View Javadoc
1   /*
2    * Licensed to the Apache Software Foundation (ASF) under one or more
3    * contributor license agreements.  See the NOTICE file distributed with
4    * this work for additional information regarding copyright ownership.
5    * The ASF licenses this file to You under the Apache License, Version 2.0
6    * (the "License"); you may not use this file except in compliance with
7    * the License.  You may obtain a copy of the License at
8    *
9    *      https://www.apache.org/licenses/LICENSE-2.0
10   *
11   * Unless required by applicable law or agreed to in writing, software
12   * distributed under the License is distributed on an "AS IS" BASIS,
13   * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
14   * See the License for the specific language governing permissions and
15   * limitations under the License.
16   */
17  
18  package org.apache.commons.lang3.text.translate;
19  
20  import java.io.IOException;
21  import java.io.Writer;
22  
23  import org.apache.commons.lang3.CharUtils;
24  
25  /**
26   * Translates escaped Unicode values of the form \\u+\d\d\d\d back to Unicode. It supports multiple 'u' characters and will work with or without the +.
27   * <p>
28   * Only ASCII hexadecimal digits ({@code [0-9a-fA-F]}) are accepted in the four-digit value. Malformed sequences - non-ASCII-hex digits, sign characters,
29   * or an escape truncated by the end of the input - are not translated and pass through unchanged.
30   * </p>
31   *
32   * @since 3.0
33   * @deprecated As of <a href="https://commons.apache.org/proper/commons-lang/changes-report.html#a3.6">3.6</a>, use Apache Commons Text
34   *             <a href="https://commons.apache.org/proper/commons-text/javadocs/api-release/org/apache/commons/text/translate/UnicodeUnescaper.html">
35   *             UnicodeUnescaper</a>.
36   */
37  @Deprecated
38  public class UnicodeUnescaper extends CharSequenceTranslator {
39  
40      /**
41       * Constructs a new instance.
42       */
43      public UnicodeUnescaper() {
44          // empty
45      }
46  
47      /**
48       * {@inheritDoc}
49       */
50      @Override
51      public int translate(final CharSequence input, final int index, final Writer out) throws IOException {
52          if (input.charAt(index) == '\\' && index + 1 < input.length() && input.charAt(index + 1) == 'u') {
53              // consume optional additional 'u' chars
54              int i = 2;
55              while (index + i < input.length() && input.charAt(index + i) == 'u') {
56                  i++;
57              }
58              if (index + i < input.length() && input.charAt(index + i) == '+') {
59                  i++;
60              }
61              if (index + i + 4 <= input.length()) {
62                  // Get 4 hex digits
63                  final CharSequence unicode = input.subSequence(index + i, index + i + 4);
64                  // Pre-validate that all four characters are ASCII hexadecimal digits, mirroring
65                  // NumericEntityUnescaper's CharUtils.isHex discipline. Integer.parseInt is looser than
66                  // the \\uXXXX format: it accepts a leading sign and, via Character.digit, decimal digits
67                  // from any Unicode script and fullwidth Latin hex letters. Anything that is not a
68                  // well-formed escape is not translated and passes through verbatim.
69                  for (int j = 0; j < 4; j++) {
70                      if (!CharUtils.isHex(unicode.charAt(j))) {
71                          return 0;
72                      }
73                  }
74                  out.write((char) Integer.parseInt(unicode.toString(), 16));
75                  return i + 4;
76              }
77              // Truncated escape at the end of the input: not a well-formed escape, pass through verbatim.
78              return 0;
79          }
80          return 0;
81      }
82  }