View Javadoc
1   /*
2    * Licensed to the Apache Software Foundation (ASF) under one or more
3    * contributor license agreements.  See the NOTICE file distributed with
4    * this work for additional information regarding copyright ownership.
5    * The ASF licenses this file to You under the Apache License, Version 2.0
6    * (the "License"); you may not use this file except in compliance with
7    * the License.  You may obtain a copy of the License at
8    *
9    *      https://www.apache.org/licenses/LICENSE-2.0
10   *
11   * Unless required by applicable law or agreed to in writing, software
12   * distributed under the License is distributed on an "AS IS" BASIS,
13   * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
14   * See the License for the specific language governing permissions and
15   * limitations under the License.
16   */
17  package org.apache.commons.lang3;
18  
19  import java.io.IOException;
20  import java.io.Writer;
21  
22  import org.apache.commons.lang3.text.translate.AggregateTranslator;
23  import org.apache.commons.lang3.text.translate.CharSequenceTranslator;
24  import org.apache.commons.lang3.text.translate.EntityArrays;
25  import org.apache.commons.lang3.text.translate.JavaUnicodeEscaper;
26  import org.apache.commons.lang3.text.translate.LookupTranslator;
27  import org.apache.commons.lang3.text.translate.NumericEntityEscaper;
28  import org.apache.commons.lang3.text.translate.NumericEntityUnescaper;
29  import org.apache.commons.lang3.text.translate.OctalUnescaper;
30  import org.apache.commons.lang3.text.translate.UnicodeUnescaper;
31  import org.apache.commons.lang3.text.translate.UnicodeUnpairedSurrogateRemover;
32  
33  /**
34   * Escapes and unescapes {@link String}s for
35   * Java, Java Script, HTML and XML.
36   *
37   * <p>
38   * #ThreadSafe#
39   * </p>
40   *
41   * @since 2.0
42   * @deprecated As of 3.6, use Apache Commons Text
43   * <a href="https://commons.apache.org/proper/commons-text/javadocs/api-release/org/apache/commons/text/StringEscapeUtils.html">
44   * StringEscapeUtils</a> instead.
45   */
46  @Deprecated
47  public class StringEscapeUtils {
48  
49      /* ESCAPE TRANSLATORS */
50  
51      private static final class CsvEscaper extends CharSequenceTranslator {
52  
53          private static final char CSV_DELIMITER = ',';
54          private static final char CSV_QUOTE = '"';
55          private static final String CSV_QUOTE_STR = String.valueOf(CSV_QUOTE);
56          private static final char[] CSV_SEARCH_CHARS = { CSV_DELIMITER, CSV_QUOTE, CharUtils.CR, CharUtils.LF };
57  
58          @Override
59          public int translate(final CharSequence input, final int index, final Writer out) throws IOException {
60              if (index != 0) {
61                  throw new IllegalStateException("CsvEscaper should never reach the [1] index");
62              }
63              if (StringUtils.containsNone(input.toString(), CSV_SEARCH_CHARS)) {
64                  out.write(input.toString());
65              } else {
66                  out.write(CSV_QUOTE);
67                  out.write(Strings.CS.replace(input.toString(), CSV_QUOTE_STR, CSV_QUOTE_STR + CSV_QUOTE_STR));
68                  out.write(CSV_QUOTE);
69              }
70              return Character.codePointCount(input, 0, input.length());
71          }
72      }
73  
74      private static final class CsvUnescaper extends CharSequenceTranslator {
75  
76          private static final char CSV_DELIMITER = ',';
77          private static final char CSV_QUOTE = '"';
78          private static final String CSV_QUOTE_STR = String.valueOf(CSV_QUOTE);
79          private static final char[] CSV_SEARCH_CHARS = {CSV_DELIMITER, CSV_QUOTE, CharUtils.CR, CharUtils.LF};
80  
81          @Override
82          public int translate(final CharSequence input, final int index, final Writer out) throws IOException {
83              if (index != 0) {
84                  throw new IllegalStateException("CsvUnescaper should never reach the [1] index");
85              }
86              if (input.length() < 2 || input.charAt(0) != CSV_QUOTE || input.charAt(input.length() - 1) != CSV_QUOTE) {
87                  out.write(input.toString());
88                  return Character.codePointCount(input, 0, input.length());
89              }
90              // strip quotes
91              final String quoteless = input.subSequence(1, input.length() - 1).toString();
92              if (StringUtils.containsAny(quoteless, CSV_SEARCH_CHARS)) {
93                  // deal with escaped quotes; ie) ""
94                  out.write(Strings.CS.replace(quoteless, CSV_QUOTE_STR + CSV_QUOTE_STR, CSV_QUOTE_STR));
95              } else {
96                  out.write(input.toString());
97              }
98              return Character.codePointCount(input, 0, input.length());
99          }
100     }
101 
102     /**
103      * Translator object for escaping Java.
104      *
105      * While {@link #escapeJava(String)} is the expected method of use, this
106      * object allows the Java escaping functionality to be used
107      * as the foundation for a custom translator.
108      *
109      * @since 3.0
110      */
111     public static final CharSequenceTranslator ESCAPE_JAVA =
112           new LookupTranslator(
113             new String[][] {
114               {"\"", "\\\""},
115               {"\\", "\\\\"},
116           }).with(
117             new LookupTranslator(EntityArrays.JAVA_CTRL_CHARS_ESCAPE())
118           ).with(
119             JavaUnicodeEscaper.outsideOf(32, 0x7f)
120         );
121 
122     /**
123      * Translator object for escaping EcmaScript/JavaScript.
124      *
125      * While {@link #escapeEcmaScript(String)} is the expected method of use, this
126      * object allows the EcmaScript escaping functionality to be used
127      * as the foundation for a custom translator.
128      *
129      * @since 3.0
130      */
131     public static final CharSequenceTranslator ESCAPE_ECMASCRIPT =
132         new AggregateTranslator(
133             new LookupTranslator(
134                       new String[][] {
135                             {"'", "\\'"},
136                             {"\"", "\\\""},
137                             {"`", "\\`"},
138                             {"${", "\\${"},
139                             {"\\", "\\\\"},
140                             {"/", "\\/"}
141                       }),
142             new LookupTranslator(EntityArrays.JAVA_CTRL_CHARS_ESCAPE()),
143             JavaUnicodeEscaper.outsideOf(32, 0x7f)
144         );
145 
146     /**
147      * Translator object for escaping Json.
148      *
149      * While {@link #escapeJson(String)} is the expected method of use, this
150      * object allows the Json escaping functionality to be used
151      * as the foundation for a custom translator.
152      *
153      * @since 3.2
154      */
155     public static final CharSequenceTranslator ESCAPE_JSON =
156         new AggregateTranslator(
157             new LookupTranslator(
158                       new String[][] {
159                             {"\"", "\\\""},
160                             {"\\", "\\\\"},
161                             {"/", "\\/"}
162                       }),
163             new LookupTranslator(EntityArrays.JAVA_CTRL_CHARS_ESCAPE()),
164             JavaUnicodeEscaper.outsideOf(32, 0x7f)
165         );
166 
167     /**
168      * Translator object for escaping XML.
169      *
170      * While {@link #escapeXml(String)} is the expected method of use, this
171      * object allows the XML escaping functionality to be used
172      * as the foundation for a custom translator.
173      *
174      * @since 3.0
175      * @deprecated Use {@link #ESCAPE_XML10} or {@link #ESCAPE_XML11} instead.
176      */
177     @Deprecated
178     public static final CharSequenceTranslator ESCAPE_XML =
179         new AggregateTranslator(
180             new LookupTranslator(EntityArrays.APOS_ESCAPE()),
181             new LookupTranslator(EntityArrays.BASIC_ESCAPE())
182         );
183 
184     /**
185      * Translator object for escaping XML 1.0.
186      *
187      * While {@link #escapeXml10(String)} is the expected method of use, this
188      * object allows the XML escaping functionality to be used
189      * as the foundation for a custom translator.
190      *
191      * @since 3.3
192      */
193     public static final CharSequenceTranslator ESCAPE_XML10 =
194         new AggregateTranslator(
195             new LookupTranslator(EntityArrays.APOS_ESCAPE()),
196             new LookupTranslator(EntityArrays.BASIC_ESCAPE()),
197             new LookupTranslator(
198                     new String[][] {
199                             { "\u0000", StringUtils.EMPTY },
200                             { "\u0001", StringUtils.EMPTY },
201                             { "\u0002", StringUtils.EMPTY },
202                             { "\u0003", StringUtils.EMPTY },
203                             { "\u0004", StringUtils.EMPTY },
204                             { "\u0005", StringUtils.EMPTY },
205                             { "\u0006", StringUtils.EMPTY },
206                             { "\u0007", StringUtils.EMPTY },
207                             { "\u0008", StringUtils.EMPTY },
208                             { "\u000b", StringUtils.EMPTY },
209                             { "\u000c", StringUtils.EMPTY },
210                             { "\u000e", StringUtils.EMPTY },
211                             { "\u000f", StringUtils.EMPTY },
212                             { "\u0010", StringUtils.EMPTY },
213                             { "\u0011", StringUtils.EMPTY },
214                             { "\u0012", StringUtils.EMPTY },
215                             { "\u0013", StringUtils.EMPTY },
216                             { "\u0014", StringUtils.EMPTY },
217                             { "\u0015", StringUtils.EMPTY },
218                             { "\u0016", StringUtils.EMPTY },
219                             { "\u0017", StringUtils.EMPTY },
220                             { "\u0018", StringUtils.EMPTY },
221                             { "\u0019", StringUtils.EMPTY },
222                             { "\u001a", StringUtils.EMPTY },
223                             { "\u001b", StringUtils.EMPTY },
224                             { "\u001c", StringUtils.EMPTY },
225                             { "\u001d", StringUtils.EMPTY },
226                             { "\u001e", StringUtils.EMPTY },
227                             { "\u001f", StringUtils.EMPTY },
228                             { "\ufffe", StringUtils.EMPTY },
229                             { "\uffff", StringUtils.EMPTY }
230                     }),
231             NumericEntityEscaper.between(0x7f, 0x84),
232             NumericEntityEscaper.between(0x86, 0x9f),
233             new UnicodeUnpairedSurrogateRemover()
234         );
235 
236     /**
237      * Translator object for escaping XML 1.1.
238      *
239      * While {@link #escapeXml11(String)} is the expected method of use, this
240      * object allows the XML escaping functionality to be used
241      * as the foundation for a custom translator.
242      *
243      * @since 3.3
244      */
245     public static final CharSequenceTranslator ESCAPE_XML11 =
246         new AggregateTranslator(
247             new LookupTranslator(EntityArrays.APOS_ESCAPE()),
248             new LookupTranslator(EntityArrays.BASIC_ESCAPE()),
249             new LookupTranslator(
250                     new String[][] {
251                             { "\u0000", StringUtils.EMPTY },
252                             { "\u000b", "&#11;" },
253                             { "\u000c", "&#12;" },
254                             { "\ufffe", StringUtils.EMPTY },
255                             { "\uffff", StringUtils.EMPTY }
256                     }),
257             NumericEntityEscaper.between(0x1, 0x8),
258             NumericEntityEscaper.between(0xe, 0x1f),
259             NumericEntityEscaper.between(0x7f, 0x84),
260             NumericEntityEscaper.between(0x86, 0x9f),
261             new UnicodeUnpairedSurrogateRemover()
262         );
263 
264     /**
265      * Translator object for escaping HTML version 3.0.
266      *
267      * While {@link #escapeHtml3(String)} is the expected method of use, this
268      * object allows the HTML escaping functionality to be used
269      * as the foundation for a custom translator.
270      *
271      * @since 3.0
272      */
273     public static final CharSequenceTranslator ESCAPE_HTML3 =
274         new AggregateTranslator(
275             new LookupTranslator(EntityArrays.BASIC_ESCAPE()),
276             new LookupTranslator(EntityArrays.ISO8859_1_ESCAPE())
277         );
278 
279     /**
280      * Translator object for escaping HTML version 4.0.
281      *
282      * While {@link #escapeHtml4(String)} is the expected method of use, this
283      * object allows the HTML escaping functionality to be used
284      * as the foundation for a custom translator.
285      *
286      * @since 3.0
287      */
288     public static final CharSequenceTranslator ESCAPE_HTML4 =
289         new AggregateTranslator(
290             new LookupTranslator(EntityArrays.BASIC_ESCAPE()),
291             new LookupTranslator(EntityArrays.ISO8859_1_ESCAPE()),
292             new LookupTranslator(EntityArrays.HTML40_EXTENDED_ESCAPE())
293         );
294 
295     /* UNESCAPE TRANSLATORS */
296 
297     /**
298      * Translator object for escaping individual Comma Separated Values.
299      *
300      * While {@link #escapeCsv(String)} is the expected method of use, this
301      * object allows the CSV escaping functionality to be used
302      * as the foundation for a custom translator.
303      *
304      * @since 3.0
305      */
306     public static final CharSequenceTranslator ESCAPE_CSV = new CsvEscaper();
307 
308     /**
309      * Translator object for unescaping escaped Java.
310      *
311      * While {@link #unescapeJava(String)} is the expected method of use, this
312      * object allows the Java unescaping functionality to be used
313      * as the foundation for a custom translator.
314      *
315      * @since 3.0
316      */
317     // TODO: throw "illegal character: \92" as an Exception if a \ on the end of the Java (as per the compiler)?
318     public static final CharSequenceTranslator UNESCAPE_JAVA =
319         new AggregateTranslator(
320             new OctalUnescaper(),     // .between('\1', '\377'),
321             new UnicodeUnescaper(),
322             new LookupTranslator(EntityArrays.JAVA_CTRL_CHARS_UNESCAPE()),
323             new LookupTranslator(
324                       new String[][] {
325                             {"\\\\", "\\"},
326                             {"\\\"", "\""},
327                             {"\\'", "'"},
328                             {"\\", ""}
329                       })
330         );
331 
332     /**
333      * Translator object for unescaping escaped EcmaScript.
334      *
335      * While {@link #unescapeEcmaScript(String)} is the expected method of use, this
336      * object allows the EcmaScript unescaping functionality to be used
337      * as the foundation for a custom translator.
338      *
339      * @since 3.0
340      */
341     public static final CharSequenceTranslator UNESCAPE_ECMASCRIPT = UNESCAPE_JAVA;
342 
343     /**
344      * Translator object for unescaping escaped Json.
345      *
346      * While {@link #unescapeJson(String)} is the expected method of use, this
347      * object allows the Json unescaping functionality to be used
348      * as the foundation for a custom translator.
349      *
350      * @since 3.2
351      */
352     public static final CharSequenceTranslator UNESCAPE_JSON = UNESCAPE_JAVA;
353 
354     /**
355      * Translator object for unescaping escaped HTML 3.0.
356      *
357      * While {@link #unescapeHtml3(String)} is the expected method of use, this
358      * object allows the HTML unescaping functionality to be used
359      * as the foundation for a custom translator.
360      *
361      * @since 3.0
362      */
363     public static final CharSequenceTranslator UNESCAPE_HTML3 =
364         new AggregateTranslator(
365             new LookupTranslator(EntityArrays.BASIC_UNESCAPE()),
366             new LookupTranslator(EntityArrays.ISO8859_1_UNESCAPE()),
367             new NumericEntityUnescaper()
368         );
369 
370     /**
371      * Translator object for unescaping escaped HTML 4.0.
372      *
373      * While {@link #unescapeHtml4(String)} is the expected method of use, this
374      * object allows the HTML unescaping functionality to be used
375      * as the foundation for a custom translator.
376      *
377      * @since 3.0
378      */
379     public static final CharSequenceTranslator UNESCAPE_HTML4 =
380         new AggregateTranslator(
381             new LookupTranslator(EntityArrays.BASIC_UNESCAPE()),
382             new LookupTranslator(EntityArrays.ISO8859_1_UNESCAPE()),
383             new LookupTranslator(EntityArrays.HTML40_EXTENDED_UNESCAPE()),
384             new NumericEntityUnescaper()
385         );
386 
387     /**
388      * Translator object for unescaping escaped XML.
389      *
390      * While {@link #unescapeXml(String)} is the expected method of use, this
391      * object allows the XML unescaping functionality to be used
392      * as the foundation for a custom translator.
393      *
394      * @since 3.0
395      */
396     public static final CharSequenceTranslator UNESCAPE_XML =
397         new AggregateTranslator(
398             new LookupTranslator(EntityArrays.BASIC_UNESCAPE()),
399             new LookupTranslator(EntityArrays.APOS_UNESCAPE()),
400             new NumericEntityUnescaper()
401         );
402 
403     /**
404      * Translator object for unescaping escaped Comma Separated Value entries.
405      *
406      * While {@link #unescapeCsv(String)} is the expected method of use, this
407      * object allows the CSV unescaping functionality to be used
408      * as the foundation for a custom translator.
409      *
410      * @since 3.0
411      */
412     public static final CharSequenceTranslator UNESCAPE_CSV = new CsvUnescaper();
413 
414     /* Helper functions */
415 
416     /**
417      * Returns a {@link String} value for a CSV column enclosed in double quotes, if required.
418      * <p>
419      * If the value contains a comma, newline or double quote, then the String value is returned enclosed in double quotes.
420      * </p>
421      * <p>
422      * Any double quote characters in the value are escaped with another double quote.
423      * </p>
424      * <p>
425      * If the value does not contain a comma, newline or double quote, then the String value is returned unchanged.
426      * </p>
427      * <p>
428      * See <a href="https://en.wikipedia.org/wiki/Comma-separated_values">Wikipedia</a> and <a href="https://datatracker.ietf.org/doc/html/rfc4180">RFC
429      * 4180</a>.
430      * </p>
431      *
432      * @param input The input CSV column String, may be null
433      * @return The input String, enclosed in double quotes if the value contains a comma, newline or double quote, {@code null} if null string input
434      * @since 2.4
435      */
436     public static final String escapeCsv(final String input) {
437         return ESCAPE_CSV.translate(input);
438     }
439 
440     /**
441      * Escapes the characters in a {@link String} using EcmaScript String rules.
442      * <p>
443      * Escapes any values it finds into their EcmaScript String form. Escapes the EcmaScript string delimiters (single quote, double quote and (since ES6) the
444      * backtick) as well as the template-literal interpolation sequence <code>${</code> and control-chars (tab, backslash, cr, ff, etc.).
445      * </p>
446      * <p>
447      * So a tab becomes the characters {@code '\\'} and {@code 't'}.
448      * </p>
449      * <p>
450      * The differences between Java strings and EcmaScript strings handled here are that in EcmaScript, a single quote, the backtick, the <code>${</code>
451      * sequence and forward-slash (/) are escaped.
452      * </p>
453      * <p>
454      * <strong>Scope:</strong> the output is a correctly escaped EcmaScript string literal for any of the three string delimiters, but it is <em>not</em> made
455      * safe for direct embedding inside an HTML inline {@code <script>} block: HTML parser-state sequences such as {@code <!--} and {@code <script} pass through
456      * unchanged (a literal {@code </script>} is neutralized by the forward-slash escape). HTML-context encoding must be applied separately when the result is
457      * placed in HTML.
458      * </p>
459      * <p>
460      * Note that EcmaScript is best known by the JavaScript and ActionScript dialects.
461      * </p>
462      * <p>
463      * Example:
464      * </p>
465      *
466      * <pre>
467      * input string: He didn't say, "Stop!"
468      * output string: He didn\'t say, \"Stop!\"
469      * </pre>
470      *
471      * @param input String to escape values in, may be null
472      * @return String with escaped values, {@code null} if null string input
473      * @since 3.0
474      */
475     public static final String escapeEcmaScript(final String input) {
476         return ESCAPE_ECMASCRIPT.translate(input);
477     }
478 
479     /**
480      * Escapes the characters in a {@link String} using HTML entities.
481      *
482      * <p>
483      * Supports only the HTML 3.0 entities. Apostrophes are escaped as the numeric reference {@code &#39;}.
484      * </p>
485      * <p>
486      * HTML attribute values must be quoted. This method does not escape all characters that delimit unquoted attribute values.
487      * </p>
488      *
489      * @param input  The {@link String} to escape, may be null
490      * @return A new escaped {@link String}, {@code null} if null string input
491      * @since 3.0
492      */
493     public static final String escapeHtml3(final String input) {
494         return ESCAPE_HTML3.translate(input);
495     }
496 
497     /**
498      * Escapes the characters in a {@link String} using HTML entities.
499      *
500      * <p>
501      * For example:
502      * </p>
503      * <p>
504      * {@code "bread" &amp; "butter"}
505      * </p>
506      * becomes:
507      * <p>
508      * {@code &amp;quot;bread&amp;quot; &amp;amp; &amp;quot;butter&amp;quot;}.
509      * </p>
510      * <p>
511      * Supports all known HTML 4.0 entities, including funky accents.
512      * Apostrophes are escaped as the numeric reference {@code &#39;} because {@code &apos;} is not a legal HTML 4.0 entity.
513      * </p>
514      * <p>
515      * HTML attribute values must be quoted. This method does not escape all characters that delimit unquoted attribute values.
516      * </p>
517      *
518      * @param input  The {@link String} to escape, may be null
519      * @return A new escaped {@link String}, {@code null} if null string input
520      * @see <a href="https://web.archive.org/web/20060225074150/https://hotwired.lycos.com/webmonkey/reference/special_characters/">ISO Entities</a>
521      * @see <a href="https://www.w3.org/TR/REC-html32#latin1">HTML 3.2 Character Entities for ISO Latin-1</a>
522      * @see <a href="https://www.w3.org/TR/REC-html40/sgml/entities.html">HTML 4.0 Character entity references</a>
523      * @see <a href="https://www.w3.org/TR/html401/charset.html#h-5.3">HTML 4.01 Character References</a>
524      * @see <a href="https://www.w3.org/TR/html401/charset.html#code-position">HTML 4.01 Code positions</a>
525      * @since 3.0
526      */
527     public static final String escapeHtml4(final String input) {
528         return ESCAPE_HTML4.translate(input);
529     }
530 
531     /**
532      * Escapes the characters in a {@link String} using Java String rules.
533      *
534      * <p>
535      * Deals correctly with quotes and control-chars (tab, backslash, cr, ff, etc.)
536      * </p>
537      *
538      * <p>
539      * So a tab becomes the characters {@code '\\'} and
540      * {@code 't'}.
541      * </p>
542      *
543      * <p>
544      * The only difference between Java strings and JavaScript strings
545      * is that in JavaScript, a single quote and forward-slash (/) are escaped.
546      * </p>
547      *
548      * <p>
549      * Example:
550      * </p>
551      * <pre>
552      * input string: He didn't say, "Stop!"
553      * output string: He didn't say, \"Stop!\"
554      * </pre>
555      *
556      * @param input  String to escape values in, may be null
557      * @return String with escaped values, {@code null} if null string input
558      */
559     public static final String escapeJava(final String input) {
560         return ESCAPE_JAVA.translate(input);
561     }
562 
563     /**
564      * Escapes the characters in a {@link String} using Json String rules.
565      * <p>
566      * Escapes any values it finds into their JSON String form. Deals correctly with quotes and control-chars (tab, backslash, cr, ff, etc.)
567      * </p>
568      * <p>
569      * So a tab becomes the characters {@code '\\'} and {@code 't'}.
570      * </p>
571      * <p>
572      * The only difference between Java strings and Json strings is that in Json, forward-slash (/) is escaped.
573      * </p>
574      * <p>
575      * <strong>Scope:</strong> the output is a correctly escaped JSON string, but it is <em>not</em> made safe for direct embedding inside an HTML inline
576      * {@code <script>} block: the backtick, <code>${</code>, and HTML parser-state sequences such as {@code <!--} and {@code <script} pass through unchanged
577      * (JSON offers no backslash escape for them; only a literal {@code </script>} is neutralized by the forward-slash escape). HTML-context encoding must be
578      * applied separately when the result is placed in HTML.
579      * </p>
580      * <p>
581      * See https://www.ietf.org/rfc/rfc4627.txt for further details.
582      * </p>
583      * <p>
584      * Example:
585      * </p>
586      *
587      * <pre>
588      * input string: He didn't say, "Stop!"
589      * output string: He didn't say, \"Stop!\"
590      * </pre>
591      *
592      * @param input String to escape values in, may be null
593      * @return String with escaped values, {@code null} if null string input
594      * @since 3.2
595      */
596     public static final String escapeJson(final String input) {
597         return ESCAPE_JSON.translate(input);
598     }
599 
600     /**
601      * Escapes the characters in a {@link String} using XML entities.
602      *
603      * <p>
604      * For example: {@code "bread" & "butter"} =&gt;
605      * {@code &quot;bread&quot; &amp; &quot;butter&quot;}.
606      * </p>
607      *
608      * <p>
609      * Supports only the five basic XML entities (gt, lt, quot, amp, apos).
610      * Does not support DTDs or external entities.
611      * </p>
612      *
613      * <p>
614      * Note that Unicode characters greater than 0x7f are as of 3.0, no longer
615      *    escaped. If you still wish this functionality, you can achieve it
616      *    via the following:
617      * {@code StringEscapeUtils.ESCAPE_XML.with( NumericEntityEscaper.between(0x7f, Integer.MAX_VALUE));}
618      * </p>
619      *
620      * @param input  The {@link String} to escape, may be null
621      * @return A new escaped {@link String}, {@code null} if null string input
622      * @see #unescapeXml(String)
623      * @deprecated Use {@link #escapeXml10(java.lang.String)} or {@link #escapeXml11(java.lang.String)} instead.
624      */
625     @Deprecated
626     public static final String escapeXml(final String input) {
627         return ESCAPE_XML.translate(input);
628     }
629 
630     /**
631      * Escapes the characters in a {@link String} using XML entities.
632      * <p>
633      * For example:
634      * </p>
635      *
636      * <pre>{@code
637      * "bread" & "butter"
638      * }</pre>
639      * <p>
640      * converts to:
641      * </p>
642      *
643      * <pre>
644      * {@code
645      * &quot;bread&quot; &amp; &quot;butter&quot;
646      * }
647      * </pre>
648      *
649      * <p>
650      * Note that XML 1.0 is a text-only format: it cannot represent control characters or unpaired Unicode surrogate code points, even after escaping. The
651      * method {@code escapeXml10} will remove characters that do not fit in the following ranges:
652      * </p>
653      *
654      * <p>
655      * {@code #x9 | #xA | #xD | [#x20-#xD7FF] | [#xE000-#xFFFD] | [#x10000-#x10FFFF]}
656      * </p>
657      *
658      * <p>
659      * Though not strictly necessary, {@code escapeXml10} will escape characters in the following ranges:
660      * </p>
661      *
662      * <p>
663      * {@code [#x7F-#x84] | [#x86-#x9F]}
664      * </p>
665      *
666      * <p>
667      * The returned string can be inserted into a valid XML 1.0 or XML 1.1 document. If you want to allow more non-text characters in an XML 1.1 document, use
668      * {@link #escapeXml11(String)}.
669      * </p>
670      *
671      * @param input The {@link String} to escape, may be null
672      * @return A new escaped {@link String}, {@code null} if null string input
673      * @see #unescapeXml(String)
674      * @since 3.3
675      */
676     public static String escapeXml10(final String input) {
677         return ESCAPE_XML10.translate(input);
678     }
679 
680     /**
681      * Escapes the characters in a {@link String} using XML entities.
682      *
683      * <p>
684      * For example: {@code "bread" & "butter"} =&gt;
685      * {@code &quot;bread&quot; &amp; &quot;butter&quot;}.
686      * </p>
687      *
688      * <p>
689      * XML 1.1 can represent certain control characters, but it cannot represent
690      * the null byte or unpaired Unicode surrogate code points, even after escaping.
691      * {@code escapeXml11} will remove characters that do not fit in the following
692      * ranges:
693      * </p>
694      *
695      * <p>
696      * {@code [#x1-#xD7FF] | [#xE000-#xFFFD] | [#x10000-#x10FFFF]}
697      * </p>
698      *
699      * <p>
700      * {@code escapeXml11} will escape characters in the following ranges:
701      * </p>
702      *
703      * <p>
704      * {@code [#x1-#x8] | [#xB-#xC] | [#xE-#x1F] | [#x7F-#x84] | [#x86-#x9F]}
705      * </p>
706      *
707      * <p>
708      * The returned string can be inserted into a valid XML 1.1 document. Do not
709      * use it for XML 1.0 documents.
710      * </p>
711      *
712      * @param input  The {@link String} to escape, may be null
713      * @return A new escaped {@link String}, {@code null} if null string input
714      * @see #unescapeXml(String)
715      * @since 3.3
716      */
717     public static String escapeXml11(final String input) {
718         return ESCAPE_XML11.translate(input);
719     }
720 
721     /**
722      * Returns a {@link String} value for an unescaped CSV column.
723      * <p>
724      * If the value is enclosed in double quotes, and contains a comma, newline or double quote, then quotes are removed.
725      * </p>
726      * <p>
727      * Any double quote escaped characters (a pair of double quotes) are unescaped to just one double quote.
728      * </p>
729      * <p>
730      * If the value is not enclosed in double quotes, or is and does not contain a comma, newline or double quote, then the String value is returned unchanged.
731      * </p>
732      * <p>
733      * See <a href="https://en.wikipedia.org/wiki/Comma-separated_values">Wikipedia</a> and <a href="https://datatracker.ietf.org/doc/html/rfc4180">RFC
734      * 4180</a>.
735      * </p>
736      *
737      * @param input The input CSV column String, may be null
738      * @return The input String, with enclosing double quotes removed and embedded double quotes unescaped, {@code null} if null string input
739      * @since 2.4
740      */
741     public static final String unescapeCsv(final String input) {
742         return UNESCAPE_CSV.translate(input);
743     }
744 
745     /**
746      * Unescapes any EcmaScript literals found in the {@link String}.
747      *
748      * <p>
749      * For example, it will turn a sequence of {@code '\'} and {@code 'n'}
750      * into a newline character, unless the {@code '\'} is preceded by another
751      * {@code '\'}.
752      * </p>
753      *
754      * @see #unescapeJava(String)
755      * @param input  The {@link String} to unescape, may be null
756      * @return A new unescaped {@link String}, {@code null} if null string input
757      * @since 3.0
758      */
759     public static final String unescapeEcmaScript(final String input) {
760         return UNESCAPE_ECMASCRIPT.translate(input);
761     }
762 
763     /**
764      * Unescapes a string containing entity escapes to a string
765      * containing the actual Unicode characters corresponding to the
766      * escapes. Supports only HTML 3.0 entities.
767      *
768      * @param input  The {@link String} to unescape, may be null
769      * @return A new unescaped {@link String}, {@code null} if null string input
770      * @since 3.0
771      */
772     public static final String unescapeHtml3(final String input) {
773         return UNESCAPE_HTML3.translate(input);
774     }
775 
776     /**
777      * Unescapes a string containing entity escapes to a string
778      * containing the actual Unicode characters corresponding to the
779      * escapes. Supports HTML 4.0 entities.
780      *
781      * <p>
782      * For example, the string {@code "&lt;Fran&ccedil;ais&gt;"}
783      * will become {@code "<Français>"}
784      * </p>
785      *
786      * <p>
787      * If an entity is unrecognized, it is left alone, and inserted
788      * verbatim into the result string. e.g. {@code "&gt;&zzzz;x"} will
789      * become {@code ">&zzzz;x"}.
790      * </p>
791      *
792      * @param input  The {@link String} to unescape, may be null
793      * @return A new unescaped {@link String}, {@code null} if null string input
794      * @since 3.0
795      */
796     public static final String unescapeHtml4(final String input) {
797         return UNESCAPE_HTML4.translate(input);
798     }
799 
800     /**
801      * Unescapes any Java literals found in the {@link String}.
802      * For example, it will turn a sequence of {@code '\'} and
803      * {@code 'n'} into a newline character, unless the {@code '\'}
804      * is preceded by another {@code '\'}.
805      *
806      * @param input  The {@link String} to unescape, may be null
807      * @return A new unescaped {@link String}, {@code null} if null string input
808      */
809     public static final String unescapeJava(final String input) {
810         return UNESCAPE_JAVA.translate(input);
811     }
812 
813     /**
814      * Unescapes any Json literals found in the {@link String}.
815      *
816      * <p>
817      * For example, it will turn a sequence of {@code '\'} and {@code 'n'}
818      * into a newline character, unless the {@code '\'} is preceded by another
819      * {@code '\'}.
820      * </p>
821      *
822      * @see #unescapeJava(String)
823      * @param input  The {@link String} to unescape, may be null
824      * @return A new unescaped {@link String}, {@code null} if null string input
825      * @since 3.2
826      */
827     public static final String unescapeJson(final String input) {
828         return UNESCAPE_JSON.translate(input);
829     }
830 
831     /**
832      * Unescapes a string containing XML entity escapes to a string
833      * containing the actual Unicode characters corresponding to the
834      * escapes.
835      *
836      * <p>
837      * Supports only the five basic XML entities (gt, lt, quot, amp, apos).
838      * Does not support DTDs or external entities.
839      * </p>
840      *
841      * <p>
842      * Note that numerical \\u Unicode codes are unescaped to their respective
843      *    Unicode characters. This may change in future releases.
844      *    </p>
845      *
846      * @param input  The {@link String} to unescape, may be null
847      * @return A new unescaped {@link String}, {@code null} if null string input
848      * @see #escapeXml(String)
849      * @see #escapeXml10(String)
850      * @see #escapeXml11(String)
851      */
852     public static final String unescapeXml(final String input) {
853         return UNESCAPE_XML.translate(input);
854     }
855 
856     /**
857      * {@link StringEscapeUtils} instances should NOT be constructed in
858      * standard programming.
859      *
860      * <p>
861      * Instead, the class should be used as:
862      * </p>
863      * <pre>StringEscapeUtils.escapeJava("foo");</pre>
864      *
865      * <p>
866      * This constructor is public to permit tools that require a JavaBean
867      * instance to operate.
868      * </p>
869      *
870      * @deprecated TODO Make private in 4.0.
871      */
872     @Deprecated
873     public StringEscapeUtils() {
874         // empty
875     }
876 
877 }