1 /*
2 * Licensed to the Apache Software Foundation (ASF) under one or more
3 * contributor license agreements. See the NOTICE file distributed with
4 * this work for additional information regarding copyright ownership.
5 * The ASF licenses this file to You under the Apache License, Version 2.0
6 * (the "License"); you may not use this file except in compliance with
7 * the License. You may obtain a copy of the License at
8 *
9 * https://www.apache.org/licenses/LICENSE-2.0
10 *
11 * Unless required by applicable law or agreed to in writing, software
12 * distributed under the License is distributed on an "AS IS" BASIS,
13 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
14 * See the License for the specific language governing permissions and
15 * limitations under the License.
16 */
17
18 package org.apache.commons.lang3.text.translate;
19
20 import java.io.IOException;
21 import java.io.Writer;
22
23 import org.apache.commons.lang3.CharUtils;
24
25 /**
26 * Translates escaped Unicode values of the form \\u+\d\d\d\d back to Unicode. It supports multiple 'u' characters and will work with or without the +.
27 * <p>
28 * Only ASCII hexadecimal digits ({@code [0-9a-fA-F]}) are accepted in the four-digit value. Malformed sequences - non-ASCII-hex digits, sign characters,
29 * or an escape truncated by the end of the input - are not translated and pass through unchanged.
30 * </p>
31 *
32 * @since 3.0
33 * @deprecated As of <a href="https://commons.apache.org/proper/commons-lang/changes-report.html#a3.6">3.6</a>, use Apache Commons Text
34 * <a href="https://commons.apache.org/proper/commons-text/javadocs/api-release/org/apache/commons/text/translate/UnicodeUnescaper.html">
35 * UnicodeUnescaper</a>.
36 */
37 @Deprecated
38 public class UnicodeUnescaper extends CharSequenceTranslator {
39
40 /**
41 * Constructs a new instance.
42 */
43 public UnicodeUnescaper() {
44 // empty
45 }
46
47 /**
48 * {@inheritDoc}
49 */
50 @Override
51 public int translate(final CharSequence input, final int index, final Writer out) throws IOException {
52 if (input.charAt(index) == '\\' && index + 1 < input.length() && input.charAt(index + 1) == 'u') {
53 // consume optional additional 'u' chars
54 int i = 2;
55 while (index + i < input.length() && input.charAt(index + i) == 'u') {
56 i++;
57 }
58 if (index + i < input.length() && input.charAt(index + i) == '+') {
59 i++;
60 }
61 if (index + i + 4 <= input.length()) {
62 // Get 4 hex digits
63 final CharSequence unicode = input.subSequence(index + i, index + i + 4);
64 // Pre-validate that all four characters are ASCII hexadecimal digits, mirroring
65 // NumericEntityUnescaper's CharUtils.isHex discipline. Integer.parseInt is looser than
66 // the \\uXXXX format: it accepts a leading sign and, via Character.digit, decimal digits
67 // from any Unicode script and fullwidth Latin hex letters. Anything that is not a
68 // well-formed escape is not translated and passes through verbatim.
69 for (int j = 0; j < 4; j++) {
70 if (!CharUtils.isHex(unicode.charAt(j))) {
71 return 0;
72 }
73 }
74 out.write((char) Integer.parseInt(unicode.toString(), 16));
75 return i + 4;
76 }
77 // Truncated escape at the end of the input: not a well-formed escape, pass through verbatim.
78 return 0;
79 }
80 return 0;
81 }
82 }