View Javadoc
1   /*
2    * Licensed to the Apache Software Foundation (ASF) under one or more
3    * contributor license agreements.  See the NOTICE file distributed with
4    * this work for additional information regarding copyright ownership.
5    * The ASF licenses this file to You under the Apache License, Version 2.0
6    * (the "License"); you may not use this file except in compliance with
7    * the License.  You may obtain a copy of the License at
8    *
9    *      https://www.apache.org/licenses/LICENSE-2.0
10   *
11   * Unless required by applicable law or agreed to in writing, software
12   * distributed under the License is distributed on an "AS IS" BASIS,
13   * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
14   * See the License for the specific language governing permissions and
15   * limitations under the License.
16   */
17  package org.apache.commons.lang3;
18  
19  import static org.apache.commons.lang3.LangAssertions.assertNullPointerException;
20  import static org.junit.jupiter.api.Assertions.assertEquals;
21  import static org.junit.jupiter.api.Assertions.assertFalse;
22  import static org.junit.jupiter.api.Assertions.assertNotNull;
23  import static org.junit.jupiter.api.Assertions.assertNull;
24  import static org.junit.jupiter.api.Assertions.assertThrows;
25  import static org.junit.jupiter.api.Assertions.assertTrue;
26  
27  import java.io.IOException;
28  import java.io.StringWriter;
29  import java.lang.reflect.Constructor;
30  import java.lang.reflect.Modifier;
31  import java.nio.charset.StandardCharsets;
32  import java.nio.file.Files;
33  import java.nio.file.Paths;
34  
35  import org.apache.commons.lang3.text.translate.CharSequenceTranslator;
36  import org.apache.commons.lang3.text.translate.NumericEntityEscaper;
37  import org.junit.jupiter.api.Test;
38  
39  /**
40   * Tests for {@link StringEscapeUtils}.
41   */
42  @Deprecated
43  class StringEscapeUtilsTest extends AbstractLangTest {
44      private static final String FOO = "foo";
45  
46      /** HTML and XML */
47      private static final String[][] HTML_ESCAPES = {
48          {"no escaping", "plain text", "plain text"},
49          {"no escaping", "plain text", "plain text"},
50          {"empty string", "", ""},
51          {"null", null, null},
52          {"ampersand", "bread & butter", "bread & butter"},
53          {"quotes", ""bread" & butter", "\"bread\" & butter"},
54          {"final character only", "greater than >", "greater than >"},
55          {"first character only", "&lt; less than", "< less than"},
56          {"apostrophe", "Huntington&#39;s chorea", "Huntington's chorea"},
57          {"languages", "English,Fran&ccedil;ais,\u65E5\u672C\u8A9E (nihongo)", "English,Fran\u00E7ais,\u65E5\u672C\u8A9E (nihongo)"},
58          {"8-bit ascii shouldn't number-escape", "\u0080\u009F", "\u0080\u009F"},
59      };
60  
61      private void assertEscapeJava(final String escaped, final String original) throws IOException {
62          assertEscapeJava(null, escaped, original);
63      }
64  
65      private void assertEscapeJava(String message, final String expected, final String original) throws IOException {
66          final String converted = StringEscapeUtils.escapeJava(original);
67          message = "escapeJava(String) failed" + (message == null ? "" : ": " + message);
68          assertEquals(expected, converted, message);
69  
70          final StringWriter writer = new StringWriter();
71          StringEscapeUtils.ESCAPE_JAVA.translate(original, writer);
72          assertEquals(expected, writer.toString());
73      }
74  
75      private void assertUnescapeJava(final String unescaped, final String original) throws IOException {
76          assertUnescapeJava(null, unescaped, original);
77      }
78  
79      private void assertUnescapeJava(final String message, final String unescaped, final String original) throws IOException {
80          final String expected = unescaped;
81          final String actual = StringEscapeUtils.unescapeJava(original);
82          assertEquals(expected, actual,
83                  "unescape(String) failed" + (message == null ? "" : ": " + message) + ": expected '" + StringEscapeUtils.escapeJava(expected) +
84                  // we escape this so we can see it in the error message
85                          "' actual '" + StringEscapeUtils.escapeJava(actual) + "'");
86          final StringWriter writer = new StringWriter();
87          StringEscapeUtils.UNESCAPE_JAVA.translate(original, writer);
88          assertEquals(unescaped, writer.toString());
89      }
90  
91      private void checkCsvEscapeWriter(final String expected, final String value) throws IOException {
92          final StringWriter writer = new StringWriter();
93          StringEscapeUtils.ESCAPE_CSV.translate(value, writer);
94          assertEquals(expected, writer.toString());
95      }
96  
97      private void checkCsvUnescapeWriter(final String expected, final String value) throws IOException {
98          final StringWriter writer = new StringWriter();
99          StringEscapeUtils.UNESCAPE_CSV.translate(value, writer);
100         assertEquals(expected, writer.toString());
101     }
102 
103     @Test
104     void testConstructor() {
105         assertNotNull(new StringEscapeUtils());
106         final Constructor<?>[] cons = StringEscapeUtils.class.getDeclaredConstructors();
107         assertEquals(1, cons.length);
108         assertTrue(Modifier.isPublic(cons[0].getModifiers()));
109         assertTrue(Modifier.isPublic(StringEscapeUtils.class.getModifiers()));
110         assertFalse(Modifier.isFinal(StringEscapeUtils.class.getModifiers()));
111     }
112 
113     @Test
114     void testEscapeCsvIllegalStateException() {
115         final StringWriter writer = new StringWriter();
116         assertThrows(IllegalStateException.class, () -> StringEscapeUtils.ESCAPE_CSV.translate("foo", -1, writer));
117     }
118 
119     @Test
120     void testEscapeCsvString() {
121         assertEquals("foo.bar", StringEscapeUtils.escapeCsv("foo.bar"));
122         assertEquals("\"foo,bar\"", StringEscapeUtils.escapeCsv("foo,bar"));
123         assertEquals("\"foo\nbar\"", StringEscapeUtils.escapeCsv("foo\nbar"));
124         assertEquals("\"foo\rbar\"", StringEscapeUtils.escapeCsv("foo\rbar"));
125         assertEquals("\"foo\"\"bar\"", StringEscapeUtils.escapeCsv("foo\"bar"));
126         assertEquals("foo\uD84C\uDFB4bar", StringEscapeUtils.escapeCsv("foo\uD84C\uDFB4bar"));
127         assertEquals("", StringEscapeUtils.escapeCsv(""));
128         assertNull(StringEscapeUtils.escapeCsv(null));
129     }
130 
131     @Test
132     void testEscapeCsvWriter() throws Exception {
133         checkCsvEscapeWriter("foo.bar", "foo.bar");
134         checkCsvEscapeWriter("\"foo,bar\"", "foo,bar");
135         checkCsvEscapeWriter("\"foo\nbar\"", "foo\nbar");
136         checkCsvEscapeWriter("\"foo\rbar\"", "foo\rbar");
137         checkCsvEscapeWriter("\"foo\"\"bar\"", "foo\"bar");
138         checkCsvEscapeWriter("foo\uD84C\uDFB4bar", "foo\uD84C\uDFB4bar");
139         checkCsvEscapeWriter("", null);
140         checkCsvEscapeWriter("", "");
141     }
142 
143     @Test
144     void testEscapeEcmaScript() {
145         assertNull(StringEscapeUtils.escapeEcmaScript(null));
146         assertNullPointerException(() -> StringEscapeUtils.ESCAPE_ECMASCRIPT.translate(null, null));
147         assertNullPointerException(() -> StringEscapeUtils.ESCAPE_ECMASCRIPT.translate("", null));
148         assertEquals("He didn\\'t say, \\\"stop!\\\"", StringEscapeUtils.escapeEcmaScript("He didn't say, \"stop!\""));
149         assertEquals("document.getElementById(\\\"test\\\").value = \\'<script>alert(\\'aaa\\');<\\/script>\\';",
150                 StringEscapeUtils.escapeEcmaScript("document.getElementById(\"test\").value = '<script>alert('aaa');</script>';"));
151     }
152 
153     @Test
154     void testEscapeEcmaScriptInlineScriptSequences() {
155         // HTML parser-state sequences remain unchanged, as documented for this string escaper.
156         assertEquals("<!--", StringEscapeUtils.escapeEcmaScript("<!--"));
157         assertEquals("<script", StringEscapeUtils.escapeEcmaScript("<script"));
158         assertEquals("<!--<script>", StringEscapeUtils.escapeEcmaScript("<!--<script>"));
159         assertEquals("<\\/script>", StringEscapeUtils.escapeEcmaScript("</script>"));
160     }
161 
162     @Test
163     void testEscapeEcmaScriptLineSeparators() {
164         assertEquals("\\u2028\\u2029", StringEscapeUtils.escapeEcmaScript("\u2028\u2029"));
165     }
166 
167     @Test
168     void testEscapeEcmaScriptTemplateLiterals() throws IOException {
169         final String[][] cases = {
170             {"`", "\\`"},
171             {"${", "\\${"},
172             {"${alert(document.cookie)}", "\\${alert(document.cookie)}"},
173             {"`;alert(document.cookie);//", "\\`;alert(document.cookie);\\/\\/"},
174             {"Hello `${name}`!", "Hello \\`\\${name}\\`!"},
175             {"${first}${second}", "\\${first}\\${second}"},
176             {"$${name}", "$\\${name}"},
177             {"\\`", "\\\\\\`"},
178             {"\\${name}", "\\\\\\${name}"},
179             {"$ {name} { } $", "$ {name} { } $"}
180         };
181         for (final String[] pair : cases) {
182             final String input = pair[0];
183             final String expected = pair[1];
184             assertEquals(expected, StringEscapeUtils.escapeEcmaScript(input), input);
185             final StringWriter writer = new StringWriter();
186             StringEscapeUtils.ESCAPE_ECMASCRIPT.translate(input, writer);
187             assertEquals(expected, writer.toString(), input);
188             assertEquals(input, StringEscapeUtils.unescapeEcmaScript(expected), input);
189         }
190     }
191 
192     /**
193      * Tests https://issues.apache.org/jira/browse/LANG-339
194      */
195     @Test
196     void testEscapeHiragana() {
197         // Some random Japanese Unicode characters
198         final String original = "\u304B\u304C\u3068";
199         final String escaped = StringEscapeUtils.escapeHtml4(original);
200         assertEquals(original, escaped, "Hiragana character Unicode behavior should not be being escaped by escapeHtml4");
201         final String unescaped = StringEscapeUtils.unescapeHtml4(escaped);
202         assertEquals(escaped, unescaped, "Hiragana character Unicode behavior has changed - expected no unescaping");
203     }
204 
205     @Test
206     void testEscapeHtml() throws IOException {
207         for (final String[] element : HTML_ESCAPES) {
208             final String message = element[0];
209             final String expected = element[1];
210             final String original = element[2];
211             assertEquals(expected, StringEscapeUtils.escapeHtml4(original), message);
212             final StringWriter sw = new StringWriter();
213             StringEscapeUtils.ESCAPE_HTML4.translate(original, sw);
214             final String actual = original == null ? null : sw.toString();
215             assertEquals(expected, actual, message);
216         }
217     }
218 
219     @Test
220     void testEscapeHtmlApostrophes() throws IOException {
221         final String input = "' autofocus onfocus=alert(1) x='";
222         final String expected = "&#39; autofocus onfocus=alert(1) x=&#39;";
223         assertEquals(expected, StringEscapeUtils.escapeHtml3(input));
224         assertEquals(expected, StringEscapeUtils.escapeHtml4(input));
225         for (final CharSequenceTranslator translator : new CharSequenceTranslator[] {
226                 StringEscapeUtils.ESCAPE_HTML3, StringEscapeUtils.ESCAPE_HTML4 }) {
227             final StringWriter writer = new StringWriter();
228             translator.translate(input, writer);
229             assertEquals(expected, writer.toString());
230             assertEquals("&#39;&#39;", translator.translate("''"));
231             assertEquals("&amp;#39;", translator.translate("&#39;"));
232         }
233         assertEquals(input, StringEscapeUtils.unescapeHtml3(expected));
234         assertEquals(input, StringEscapeUtils.unescapeHtml4(expected));
235     }
236 
237     /**
238      * Tests // https://issues.apache.org/jira/browse/LANG-480
239      */
240     @Test
241     void testEscapeHtmlHighUnicode() {
242         // this is the utf8 representation of the character:
243         // COUNTING ROD UNIT DIGIT THREE
244         // in Unicode
245         // code point: U+1D362
246         final byte[] data = { (byte) 0xF0, (byte) 0x9D, (byte) 0x8D, (byte) 0xA2 };
247         final String original = new String(data, StandardCharsets.UTF_8);
248         final String escaped = StringEscapeUtils.escapeHtml4(original);
249         assertEquals(original, escaped, "High Unicode should not have been escaped");
250         final String unescaped = StringEscapeUtils.unescapeHtml4(escaped);
251         assertEquals(original, unescaped, "High Unicode should have been unchanged");
252 // TODO: I think this should hold, needs further investigation
253 //        String unescapedFromEntity = StringEscapeUtils.unescapeHtml4("&#119650;");
254 //        assertEquals("High Unicode should have been unescaped", original, unescapedFromEntity);
255     }
256 
257     @Test
258     void testEscapeHtmlVersions() {
259         assertEquals("&Beta;", StringEscapeUtils.escapeHtml4("\u0392"));
260         assertEquals("\u0392", StringEscapeUtils.unescapeHtml4("&Beta;"));
261         // TODO: refine API for escaping/unescaping specific HTML versions
262     }
263 
264     @Test
265     void testEscapeJava() throws IOException {
266         assertNull(StringEscapeUtils.escapeJava(null));
267         assertNullPointerException(() -> StringEscapeUtils.ESCAPE_JAVA.translate(null, null));
268         assertNullPointerException(() -> StringEscapeUtils.ESCAPE_JAVA.translate("", null));
269         assertEscapeJava("empty string", "", "");
270         assertEscapeJava(FOO, FOO);
271         assertEscapeJava("tab", "\\t", "\t");
272         assertEscapeJava("backslash", "\\\\", "\\");
273         assertEscapeJava("single quote should not be escaped", "'", "'");
274         assertEscapeJava("\\\\\\b\\t\\r", "\\\b\t\r");
275         assertEscapeJava("\\u1234", "\u1234");
276         assertEscapeJava("\\u0234", "\u0234");
277         assertEscapeJava("\\u00EF", "\u00ef");
278         assertEscapeJava("\\u0001", "\u0001");
279         assertEscapeJava("Should use capitalized Unicode hex", "\\uABCD", "\uabcd");
280         assertEscapeJava("He didn't say, \\\"stop!\\\"", "He didn't say, \"stop!\"");
281         assertEscapeJava("non-breaking space", "This space is non-breaking:\\u00A0", "This space is non-breaking:\u00a0");
282         assertEscapeJava("\\uABCD\\u1234\\u012C", "\uABCD\u1234\u012C");
283     }
284 
285     /**
286      * Tests https://issues.apache.org/jira/browse/LANG-421
287      */
288     @Test
289     void testEscapeJavaWithSlash() {
290         final String input = "String with a slash (/) in it";
291         final String expected = input;
292         final String actual = StringEscapeUtils.escapeJava(input);
293         // In 2.4, StringEscapeUtils.escapeJava(String) escapes '/' characters, which are not a valid character to escape in a Java string.
294         assertEquals(expected, actual);
295     }
296 
297     @Test
298     void testEscapeJson() {
299         assertNull(StringEscapeUtils.escapeJson(null));
300         assertNullPointerException(() -> StringEscapeUtils.ESCAPE_JSON.translate(null, null));
301         assertNullPointerException(() -> StringEscapeUtils.ESCAPE_JSON.translate("", null));
302         assertEquals("He didn't say, \\\"stop!\\\"", StringEscapeUtils.escapeJson("He didn't say, \"stop!\""));
303         final String expected = "\\\"foo\\\" isn't \\\"bar\\\". specials: \\b\\r\\n\\f\\t\\\\\\/";
304         final String input = "\"foo\" isn't \"bar\". specials: \b\r\n\f\t\\/";
305         assertEquals(expected, StringEscapeUtils.escapeJson(input));
306     }
307 
308     @Test
309     void testEscapeJsonTemplateLiteralAndInlineScriptSequences() {
310         // JSON escaping preserves these sequences and does not provide HTML-context encoding.
311         assertEquals("`${alert(document.cookie)}`", StringEscapeUtils.escapeJson("`${alert(document.cookie)}`"));
312         assertEquals("<!--", StringEscapeUtils.escapeJson("<!--"));
313         assertEquals("<script", StringEscapeUtils.escapeJson("<script"));
314         assertEquals("<!--<script>", StringEscapeUtils.escapeJson("<!--<script>"));
315         assertEquals("<\\/script>", StringEscapeUtils.escapeJson("</script>"));
316     }
317 
318     @Test
319     void testEscapeXml() throws Exception {
320         assertEquals("&lt;abc&gt;", StringEscapeUtils.escapeXml("<abc>"));
321         assertEquals("<abc>", StringEscapeUtils.unescapeXml("&lt;abc&gt;"));
322         assertEquals("\u00A1", StringEscapeUtils.escapeXml("\u00A1"), "XML should not escape >0x7f values");
323         assertEquals("\u00A0", StringEscapeUtils.unescapeXml("&#160;"), "XML should be able to unescape >0x7f values");
324         assertEquals("\u00A0", StringEscapeUtils.unescapeXml("&#0160;"), "XML should be able to unescape >0x7f values with one leading 0");
325         assertEquals("\u00A0", StringEscapeUtils.unescapeXml("&#00160;"), "XML should be able to unescape >0x7f values with two leading 0s");
326         assertEquals("\u00A0", StringEscapeUtils.unescapeXml("&#000160;"), "XML should be able to unescape >0x7f values with three leading 0s");
327         assertEquals("ain't", StringEscapeUtils.unescapeXml("ain&apos;t"));
328         assertEquals("ain&apos;t", StringEscapeUtils.escapeXml("ain't"));
329         assertEquals("", StringEscapeUtils.escapeXml(""));
330         assertNull(StringEscapeUtils.escapeXml(null));
331         assertNull(StringEscapeUtils.unescapeXml(null));
332         StringWriter sw = new StringWriter();
333         StringEscapeUtils.ESCAPE_XML.translate("<abc>", sw);
334         assertEquals("&lt;abc&gt;", sw.toString(), "XML was escaped incorrectly");
335         sw = new StringWriter();
336         StringEscapeUtils.UNESCAPE_XML.translate("&lt;abc&gt;", sw);
337         assertEquals("<abc>", sw.toString(), "XML was unescaped incorrectly");
338     }
339 
340     @Test
341     void testEscapeXml10() {
342         assertEquals("a&lt;b&gt;c&quot;d&apos;e&amp;f", StringEscapeUtils.escapeXml10("a<b>c\"d'e&f"));
343         assertEquals("a\tb\rc\nd", StringEscapeUtils.escapeXml10("a\tb\rc\nd"), "XML 1.0 should not escape \t \n \r");
344         assertEquals("ab", StringEscapeUtils.escapeXml10("a\u0000\u0001\u0008\u000b\u000c\u000e\u001fb"),
345                 "XML 1.0 should omit most #x0-x8 | #xb | #xc | #xe-#x19");
346         assertEquals("a\ud7ff  \ue000b", StringEscapeUtils.escapeXml10("a\ud7ff\ud800 \udfff \ue000b"), "XML 1.0 should omit #xd800-#xdfff");
347         assertEquals("a\ufffdb", StringEscapeUtils.escapeXml10("a\ufffd\ufffe\uffffb"), "XML 1.0 should omit #xfffe | #xffff");
348         assertEquals("a\u007e&#127;&#132;\u0085&#134;&#159;\u00a0b", StringEscapeUtils.escapeXml10("a\u007e\u007f\u0084\u0085\u0086\u009f\u00a0b"),
349                 "XML 1.0 should escape #x7f-#x84 | #x86 - #x9f, for XML 1.1 compatibility");
350     }
351 
352     @Test
353     void testEscapeXml11() {
354         assertEquals("a&lt;b&gt;c&quot;d&apos;e&amp;f", StringEscapeUtils.escapeXml11("a<b>c\"d'e&f"));
355         assertEquals("a\tb\rc\nd", StringEscapeUtils.escapeXml11("a\tb\rc\nd"), "XML 1.1 should not escape \t \n \r");
356         assertEquals("ab", StringEscapeUtils.escapeXml11("a\u0000b"), "XML 1.1 should omit #x0");
357         assertEquals("a&#1;&#8;&#11;&#12;&#14;&#31;b", StringEscapeUtils.escapeXml11("a\u0001\u0008\u000b\u000c\u000e\u001fb"),
358                 "XML 1.1 should escape #x1-x8 | #xb | #xc | #xe-#x19");
359         assertEquals("a\u007e&#127;&#132;\u0085&#134;&#159;\u00a0b", StringEscapeUtils.escapeXml11("a\u007e\u007f\u0084\u0085\u0086\u009f\u00a0b"),
360                 "XML 1.1 should escape #x7F-#x84 | #x86-#x9F");
361         assertEquals("a\ud7ff  \ue000b", StringEscapeUtils.escapeXml11("a\ud7ff\ud800 \udfff \ue000b"), "XML 1.1 should omit #xd800-#xdfff");
362         assertEquals("a\ufffdb", StringEscapeUtils.escapeXml11("a\ufffd\ufffe\uffffb"), "XML 1.1 should omit #xfffe | #xffff");
363     }
364 
365     @Test
366     void testEscapeXmlAllCharacters() {
367         // https://www.w3.org/TR/xml/#charsets says:
368         // Char ::= #x9 | #xA | #xD | [#x20-#xD7FF] | [#xE000-#xFFFD] | [#x10000-#x10FFFF] /* any Unicode character,
369         // excluding the surrogate blocks, FFFE, and FFFF. */
370         final CharSequenceTranslator escapeXml = StringEscapeUtils.ESCAPE_XML.with(NumericEntityEscaper.below(9), NumericEntityEscaper.between(0xB, 0xC),
371                 NumericEntityEscaper.between(0xE, 0x19), NumericEntityEscaper.between(0xD800, 0xDFFF), NumericEntityEscaper.between(0xFFFE, 0xFFFF),
372                 NumericEntityEscaper.above(0x110000));
373         assertEquals("&#0;&#1;&#2;&#3;&#4;&#5;&#6;&#7;&#8;", escapeXml.translate("\u0000\u0001\u0002\u0003\u0004\u0005\u0006\u0007\u0008"));
374         assertEquals("\t", escapeXml.translate("\t")); // 0x9
375         assertEquals("\n", escapeXml.translate("\n")); // 0xA
376         assertEquals("&#11;&#12;", escapeXml.translate("\u000B\u000C"));
377         assertEquals("\r", escapeXml.translate("\r")); // 0xD
378         assertEquals("Hello World! Ain&apos;t this great?", escapeXml.translate("Hello World! Ain't this great?"));
379         assertEquals("&#14;&#15;&#24;&#25;", escapeXml.translate("\u000E\u000F\u0018\u0019"));
380     }
381 
382     /**
383      * Tests Supplementary characters.
384      * <p>
385      * From https://www.w3.org/International/questions/qa-escapes
386      * </p>
387      * <blockquote> Supplementary characters are those Unicode characters that have code points higher than the characters in the Basic Multilingual Plane
388      * (BMP). In UTF-16 a supplementary character is encoded using two 16-bit surrogate code points from the BMP. Because of this, some people think that
389      * supplementary characters need to be represented using two escapes, but this is incorrect - you must use the single, code point value for that character.
390      * For example, use &amp;&#35;x233B4&#59; rather than &amp;&#35;xD84C&#59;&amp;&#35;xDFB4&#59;. </blockquote>
391      *
392      * @see <a href="https://www.w3.org/International/questions/qa-escapes">Using character escapes in markup and CSS</a>
393      * @see <a href="https://issues.apache.org/jira/browse/LANG-728">LANG-728</a>
394      */
395     @Test
396     void testEscapeXmlSupplementaryCharacters() {
397         final CharSequenceTranslator escapeXml = StringEscapeUtils.ESCAPE_XML.with(NumericEntityEscaper.between(0x7f, Integer.MAX_VALUE));
398         assertEquals("&#144308;", escapeXml.translate("\uD84C\uDFB4"), "Supplementary character must be represented using a single escape");
399         assertEquals("a b c &#144308;", escapeXml.translate("a b c \uD84C\uDFB4"),
400                 "Supplementary characters mixed with basic characters should be encoded correctly");
401     }
402 
403     @Test
404     void testLang313() {
405         assertEquals("& &", StringEscapeUtils.unescapeHtml4("& &amp;"));
406     }
407 
408     /**
409      * Tests https://issues.apache.org/jira/browse/LANG-708
410      *
411      * @throws IOException Thrown if an I/O error occurs
412      */
413     @Test
414     void testLang708() throws IOException {
415         final byte[] inputBytes = Files.readAllBytes(Paths.get("src/test/resources/lang-708-input.txt"));
416         final String input = new String(inputBytes, StandardCharsets.UTF_8);
417         final String escaped = StringEscapeUtils.escapeEcmaScript(input);
418         // just the end:
419         assertTrue(escaped.endsWith("}]"), escaped);
420         // a little more:
421         assertTrue(escaped.endsWith("\"valueCode\\\":\\\"\\\"}]"), escaped);
422     }
423 
424     /**
425      * Tests https://issues.apache.org/jira/browse/LANG-720
426      */
427     @Test
428     void testLang720() {
429         final String input = "\ud842\udfb7" + "A";
430         final String escaped = StringEscapeUtils.escapeXml(input);
431         assertEquals(input, escaped);
432     }
433 
434     /**
435      * Tests https://issues.apache.org/jira/browse/LANG-911
436      */
437     @Test
438     void testLang911() {
439         final String bellsTest = "\ud83d\udc80\ud83d\udd14";
440         final String value = StringEscapeUtils.escapeJava(bellsTest);
441         final String valueTest = StringEscapeUtils.unescapeJava(value);
442         assertEquals(bellsTest, valueTest);
443     }
444 
445     // Tests issue LANG-150
446     // https://issues.apache.org/jira/browse/LANG-150
447     @Test
448     void testStandaloneAmphersand() {
449         assertEquals("<P&O>", StringEscapeUtils.unescapeHtml4("&lt;P&O&gt;"));
450         assertEquals("test & <", StringEscapeUtils.unescapeHtml4("test & &lt;"));
451         assertEquals("<P&O>", StringEscapeUtils.unescapeXml("&lt;P&O&gt;"));
452         assertEquals("test & <", StringEscapeUtils.unescapeXml("test & &lt;"));
453     }
454 
455     @Test
456     void testUnescapeCsvIllegalStateException() {
457         final StringWriter writer = new StringWriter();
458         assertThrows(IllegalStateException.class, () -> StringEscapeUtils.UNESCAPE_CSV.translate("foo", -1, writer));
459     }
460 
461     @Test
462     void testUnescapeCsvString() {
463         assertEquals("foo.bar", StringEscapeUtils.unescapeCsv("foo.bar"));
464         assertEquals("foo,bar", StringEscapeUtils.unescapeCsv("\"foo,bar\""));
465         assertEquals("foo\nbar", StringEscapeUtils.unescapeCsv("\"foo\nbar\""));
466         assertEquals("foo\rbar", StringEscapeUtils.unescapeCsv("\"foo\rbar\""));
467         assertEquals("foo\"bar", StringEscapeUtils.unescapeCsv("\"foo\"\"bar\""));
468         assertEquals("foo\uD84C\uDFB4bar", StringEscapeUtils.unescapeCsv("foo\uD84C\uDFB4bar"));
469         assertEquals("", StringEscapeUtils.unescapeCsv(""));
470         assertNull(StringEscapeUtils.unescapeCsv(null));
471         assertEquals("\"foo.bar\"", StringEscapeUtils.unescapeCsv("\"foo.bar\""));
472 
473         // a single quote is not an enclosing pair, so it passes through unchanged
474         assertEquals("\"", StringEscapeUtils.unescapeCsv("\""));
475     }
476 
477     @Test
478     void testUnescapeCsvWriter() throws Exception {
479         checkCsvUnescapeWriter("foo.bar", "foo.bar");
480         checkCsvUnescapeWriter("foo,bar", "\"foo,bar\"");
481         checkCsvUnescapeWriter("foo\nbar", "\"foo\nbar\"");
482         checkCsvUnescapeWriter("foo\rbar", "\"foo\rbar\"");
483         checkCsvUnescapeWriter("foo\"bar", "\"foo\"\"bar\"");
484         checkCsvUnescapeWriter("foo\uD84C\uDFB4bar", "foo\uD84C\uDFB4bar");
485         checkCsvUnescapeWriter("", null);
486         checkCsvUnescapeWriter("", "");
487         checkCsvUnescapeWriter("\"foo.bar\"", "\"foo.bar\"");
488     }
489 
490     @Test
491     void testUnescapeEcmaScript() {
492         assertNull(StringEscapeUtils.escapeEcmaScript(null));
493         assertNullPointerException(() -> StringEscapeUtils.UNESCAPE_ECMASCRIPT.translate(null, null));
494         assertNullPointerException(() -> StringEscapeUtils.UNESCAPE_ECMASCRIPT.translate("", null));
495         assertEquals("He didn't say, \"stop!\"", StringEscapeUtils.unescapeEcmaScript("He didn\\'t say, \\\"stop!\\\""));
496         assertEquals("document.getElementById(\"test\").value = '<script>alert('aaa');</script>';",
497                 StringEscapeUtils.unescapeEcmaScript("document.getElementById(\\\"test\\\").value = \\'<script>alert(\\'aaa\\');<\\/script>\\';"));
498     }
499 
500     @Test
501     void testUnescapeHexCharsHtml() {
502         // Simple easy to grok test
503         assertEquals("\u0080\u009F", StringEscapeUtils.unescapeHtml4("&#x80;&#x9F;"), "hex number unescape");
504         assertEquals("\u0080\u009F", StringEscapeUtils.unescapeHtml4("&#X80;&#X9F;"), "hex number unescape");
505         // Test all Character values:
506         for (char i = Character.MIN_VALUE; i < Character.MAX_VALUE; i++) {
507             final Character c1 = Character.valueOf(i);
508             final Character c2 = Character.valueOf((char) (i + 1));
509             final String expected = c1.toString() + c2;
510             final String escapedC1 = "&#x" + Integer.toHexString(c1.charValue()) + ";";
511             final String escapedC2 = "&#x" + Integer.toHexString(c2.charValue()) + ";";
512             assertEquals(expected, StringEscapeUtils.unescapeHtml4(escapedC1 + escapedC2), "hex number unescape index " + (int) i);
513         }
514     }
515 
516     @Test
517     void testUnescapeHtml4() throws IOException {
518         for (final String[] element : HTML_ESCAPES) {
519             final String message = element[0];
520             final String expected = element[2];
521             final String original = element[1];
522             assertEquals(expected, StringEscapeUtils.unescapeHtml4(original), message);
523             final StringWriter sw = new StringWriter();
524             StringEscapeUtils.UNESCAPE_HTML4.translate(original, sw);
525             final String actual = original == null ? null : sw.toString();
526             assertEquals(expected, actual, message);
527         }
528         // \u00E7 is a cedilla (c with wiggle under)
529         // note that the test string must be 7-bit-clean (Unicode escaped) or else it will compile incorrectly
530         // on some locales
531         assertEquals("Fran\u00E7ais", StringEscapeUtils.unescapeHtml4("Fran\u00E7ais"), "funny chars pass through OK");
532         assertEquals("Hello&;World", StringEscapeUtils.unescapeHtml4("Hello&;World"));
533         assertEquals("Hello&#;World", StringEscapeUtils.unescapeHtml4("Hello&#;World"));
534         assertEquals("Hello&# ;World", StringEscapeUtils.unescapeHtml4("Hello&# ;World"));
535         assertEquals("Hello&##;World", StringEscapeUtils.unescapeHtml4("Hello&##;World"));
536     }
537 
538     @Test
539     void testUnescapeJava() throws IOException {
540         assertNull(StringEscapeUtils.unescapeJava(null));
541         assertNullPointerException(() -> StringEscapeUtils.UNESCAPE_JAVA.translate(null, null));
542         assertNullPointerException(() -> StringEscapeUtils.UNESCAPE_JAVA.translate("", null));
543         // A malformed Unicode escape is not translated by the Unicode unescaper; the aggregate's
544         // stray-backslash rule then drops the lone backslash (same as unescapeJava("\\") == "").
545         assertEquals("u02-3", StringEscapeUtils.unescapeJava("\\u02-3"));
546         assertUnescapeJava("", "");
547         assertUnescapeJava("test", "test");
548         assertUnescapeJava("\ntest\b", "\\ntest\\b");
549         assertUnescapeJava("\u123425foo\ntest\b", "\\u123425foo\\ntest\\b");
550         assertUnescapeJava("'\foo\teste\r", "\\'\\foo\\teste\\r");
551         assertUnescapeJava("", "\\");
552         assertUnescapeJava("lowercase Unicode", "\uABCDx", "\\uabcdx");
553         assertUnescapeJava("uppercase Unicode", "\uABCDx", "\\uABCDx");
554         assertUnescapeJava("Unicode as final character", "\uABCD", "\\uabcd");
555     }
556 
557     @Test
558     void testUnescapeJson() {
559         assertNull(StringEscapeUtils.unescapeJson(null));
560         assertNullPointerException(() -> StringEscapeUtils.UNESCAPE_JSON.translate(null, null));
561         assertNullPointerException(() -> StringEscapeUtils.UNESCAPE_JSON.translate("", null));
562         assertEquals("He didn't say, \"stop!\"", StringEscapeUtils.unescapeJson("He didn't say, \\\"stop!\\\""));
563         final String expected = "\"foo\" isn't \"bar\". specials: \b\r\n\f\t\\/";
564         final String input = "\\\"foo\\\" isn't \\\"bar\\\". specials: \\b\\r\\n\\f\\t\\\\\\/";
565         assertEquals(expected, StringEscapeUtils.unescapeJson(input));
566     }
567 
568     @Test
569     void testUnescapeUnknownEntity() {
570         assertEquals("&zzzz;", StringEscapeUtils.unescapeHtml4("&zzzz;"));
571     }
572 
573     /**
574      * Reverse of the above.
575      *
576      * @see <a href="https://issues.apache.org/jira/browse/LANG-729">LANG-729</a>
577      */
578     @Test
579     void testUnescapeXmlSupplementaryCharacters() {
580         assertEquals("\uD84C\uDFB4", StringEscapeUtils.unescapeXml("&#144308;"), "Supplementary character must be represented using a single escape");
581         assertEquals("a b c \uD84C\uDFB4", StringEscapeUtils.unescapeXml("a b c &#144308;"),
582                 "Supplementary characters mixed with basic characters should be decoded correctly");
583     }
584 }