1 /*
2 * Licensed to the Apache Software Foundation (ASF) under one or more
3 * contributor license agreements. See the NOTICE file distributed with
4 * this work for additional information regarding copyright ownership.
5 * The ASF licenses this file to You under the Apache License, Version 2.0
6 * (the "License"); you may not use this file except in compliance with
7 * the License. You may obtain a copy of the License at
8 *
9 * https://www.apache.org/licenses/LICENSE-2.0
10 *
11 * Unless required by applicable law or agreed to in writing, software
12 * distributed under the License is distributed on an "AS IS" BASIS,
13 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
14 * See the License for the specific language governing permissions and
15 * limitations under the License.
16 */
17 package org.apache.commons.lang3;
18
19 import java.io.IOException;
20 import java.io.Writer;
21
22 import org.apache.commons.lang3.text.translate.AggregateTranslator;
23 import org.apache.commons.lang3.text.translate.CharSequenceTranslator;
24 import org.apache.commons.lang3.text.translate.EntityArrays;
25 import org.apache.commons.lang3.text.translate.JavaUnicodeEscaper;
26 import org.apache.commons.lang3.text.translate.LookupTranslator;
27 import org.apache.commons.lang3.text.translate.NumericEntityEscaper;
28 import org.apache.commons.lang3.text.translate.NumericEntityUnescaper;
29 import org.apache.commons.lang3.text.translate.OctalUnescaper;
30 import org.apache.commons.lang3.text.translate.UnicodeUnescaper;
31 import org.apache.commons.lang3.text.translate.UnicodeUnpairedSurrogateRemover;
32
33 /**
34 * Escapes and unescapes {@link String}s for
35 * Java, Java Script, HTML and XML.
36 *
37 * <p>
38 * #ThreadSafe#
39 * </p>
40 *
41 * @since 2.0
42 * @deprecated As of 3.6, use Apache Commons Text
43 * <a href="https://commons.apache.org/proper/commons-text/javadocs/api-release/org/apache/commons/text/StringEscapeUtils.html">
44 * StringEscapeUtils</a> instead.
45 */
46 @Deprecated
47 public class StringEscapeUtils {
48
49 /* ESCAPE TRANSLATORS */
50
51 private static final class CsvEscaper extends CharSequenceTranslator {
52
53 private static final char CSV_DELIMITER = ',';
54 private static final char CSV_QUOTE = '"';
55 private static final String CSV_QUOTE_STR = String.valueOf(CSV_QUOTE);
56 private static final char[] CSV_SEARCH_CHARS = { CSV_DELIMITER, CSV_QUOTE, CharUtils.CR, CharUtils.LF };
57
58 @Override
59 public int translate(final CharSequence input, final int index, final Writer out) throws IOException {
60 if (index != 0) {
61 throw new IllegalStateException("CsvEscaper should never reach the [1] index");
62 }
63 if (StringUtils.containsNone(input.toString(), CSV_SEARCH_CHARS)) {
64 out.write(input.toString());
65 } else {
66 out.write(CSV_QUOTE);
67 out.write(Strings.CS.replace(input.toString(), CSV_QUOTE_STR, CSV_QUOTE_STR + CSV_QUOTE_STR));
68 out.write(CSV_QUOTE);
69 }
70 return Character.codePointCount(input, 0, input.length());
71 }
72 }
73
74 private static final class CsvUnescaper extends CharSequenceTranslator {
75
76 private static final char CSV_DELIMITER = ',';
77 private static final char CSV_QUOTE = '"';
78 private static final String CSV_QUOTE_STR = String.valueOf(CSV_QUOTE);
79 private static final char[] CSV_SEARCH_CHARS = {CSV_DELIMITER, CSV_QUOTE, CharUtils.CR, CharUtils.LF};
80
81 @Override
82 public int translate(final CharSequence input, final int index, final Writer out) throws IOException {
83 if (index != 0) {
84 throw new IllegalStateException("CsvUnescaper should never reach the [1] index");
85 }
86 if (input.length() < 2 || input.charAt(0) != CSV_QUOTE || input.charAt(input.length() - 1) != CSV_QUOTE) {
87 out.write(input.toString());
88 return Character.codePointCount(input, 0, input.length());
89 }
90 // strip quotes
91 final String quoteless = input.subSequence(1, input.length() - 1).toString();
92 if (StringUtils.containsAny(quoteless, CSV_SEARCH_CHARS)) {
93 // deal with escaped quotes; ie) ""
94 out.write(Strings.CS.replace(quoteless, CSV_QUOTE_STR + CSV_QUOTE_STR, CSV_QUOTE_STR));
95 } else {
96 out.write(input.toString());
97 }
98 return Character.codePointCount(input, 0, input.length());
99 }
100 }
101
102 /**
103 * Translator object for escaping Java.
104 *
105 * While {@link #escapeJava(String)} is the expected method of use, this
106 * object allows the Java escaping functionality to be used
107 * as the foundation for a custom translator.
108 *
109 * @since 3.0
110 */
111 public static final CharSequenceTranslator ESCAPE_JAVA =
112 new LookupTranslator(
113 new String[][] {
114 {"\"", "\\\""},
115 {"\\", "\\\\"},
116 }).with(
117 new LookupTranslator(EntityArrays.JAVA_CTRL_CHARS_ESCAPE())
118 ).with(
119 JavaUnicodeEscaper.outsideOf(32, 0x7f)
120 );
121
122 /**
123 * Translator object for escaping EcmaScript/JavaScript.
124 *
125 * While {@link #escapeEcmaScript(String)} is the expected method of use, this
126 * object allows the EcmaScript escaping functionality to be used
127 * as the foundation for a custom translator.
128 *
129 * @since 3.0
130 */
131 public static final CharSequenceTranslator ESCAPE_ECMASCRIPT =
132 new AggregateTranslator(
133 new LookupTranslator(
134 new String[][] {
135 {"'", "\\'"},
136 {"\"", "\\\""},
137 {"`", "\\`"},
138 {"${", "\\${"},
139 {"\\", "\\\\"},
140 {"/", "\\/"}
141 }),
142 new LookupTranslator(EntityArrays.JAVA_CTRL_CHARS_ESCAPE()),
143 JavaUnicodeEscaper.outsideOf(32, 0x7f)
144 );
145
146 /**
147 * Translator object for escaping Json.
148 *
149 * While {@link #escapeJson(String)} is the expected method of use, this
150 * object allows the Json escaping functionality to be used
151 * as the foundation for a custom translator.
152 *
153 * @since 3.2
154 */
155 public static final CharSequenceTranslator ESCAPE_JSON =
156 new AggregateTranslator(
157 new LookupTranslator(
158 new String[][] {
159 {"\"", "\\\""},
160 {"\\", "\\\\"},
161 {"/", "\\/"}
162 }),
163 new LookupTranslator(EntityArrays.JAVA_CTRL_CHARS_ESCAPE()),
164 JavaUnicodeEscaper.outsideOf(32, 0x7f)
165 );
166
167 /**
168 * Translator object for escaping XML.
169 *
170 * While {@link #escapeXml(String)} is the expected method of use, this
171 * object allows the XML escaping functionality to be used
172 * as the foundation for a custom translator.
173 *
174 * @since 3.0
175 * @deprecated Use {@link #ESCAPE_XML10} or {@link #ESCAPE_XML11} instead.
176 */
177 @Deprecated
178 public static final CharSequenceTranslator ESCAPE_XML =
179 new AggregateTranslator(
180 new LookupTranslator(EntityArrays.APOS_ESCAPE()),
181 new LookupTranslator(EntityArrays.BASIC_ESCAPE())
182 );
183
184 /**
185 * Translator object for escaping XML 1.0.
186 *
187 * While {@link #escapeXml10(String)} is the expected method of use, this
188 * object allows the XML escaping functionality to be used
189 * as the foundation for a custom translator.
190 *
191 * @since 3.3
192 */
193 public static final CharSequenceTranslator ESCAPE_XML10 =
194 new AggregateTranslator(
195 new LookupTranslator(EntityArrays.APOS_ESCAPE()),
196 new LookupTranslator(EntityArrays.BASIC_ESCAPE()),
197 new LookupTranslator(
198 new String[][] {
199 { "\u0000", StringUtils.EMPTY },
200 { "\u0001", StringUtils.EMPTY },
201 { "\u0002", StringUtils.EMPTY },
202 { "\u0003", StringUtils.EMPTY },
203 { "\u0004", StringUtils.EMPTY },
204 { "\u0005", StringUtils.EMPTY },
205 { "\u0006", StringUtils.EMPTY },
206 { "\u0007", StringUtils.EMPTY },
207 { "\u0008", StringUtils.EMPTY },
208 { "\u000b", StringUtils.EMPTY },
209 { "\u000c", StringUtils.EMPTY },
210 { "\u000e", StringUtils.EMPTY },
211 { "\u000f", StringUtils.EMPTY },
212 { "\u0010", StringUtils.EMPTY },
213 { "\u0011", StringUtils.EMPTY },
214 { "\u0012", StringUtils.EMPTY },
215 { "\u0013", StringUtils.EMPTY },
216 { "\u0014", StringUtils.EMPTY },
217 { "\u0015", StringUtils.EMPTY },
218 { "\u0016", StringUtils.EMPTY },
219 { "\u0017", StringUtils.EMPTY },
220 { "\u0018", StringUtils.EMPTY },
221 { "\u0019", StringUtils.EMPTY },
222 { "\u001a", StringUtils.EMPTY },
223 { "\u001b", StringUtils.EMPTY },
224 { "\u001c", StringUtils.EMPTY },
225 { "\u001d", StringUtils.EMPTY },
226 { "\u001e", StringUtils.EMPTY },
227 { "\u001f", StringUtils.EMPTY },
228 { "\ufffe", StringUtils.EMPTY },
229 { "\uffff", StringUtils.EMPTY }
230 }),
231 NumericEntityEscaper.between(0x7f, 0x84),
232 NumericEntityEscaper.between(0x86, 0x9f),
233 new UnicodeUnpairedSurrogateRemover()
234 );
235
236 /**
237 * Translator object for escaping XML 1.1.
238 *
239 * While {@link #escapeXml11(String)} is the expected method of use, this
240 * object allows the XML escaping functionality to be used
241 * as the foundation for a custom translator.
242 *
243 * @since 3.3
244 */
245 public static final CharSequenceTranslator ESCAPE_XML11 =
246 new AggregateTranslator(
247 new LookupTranslator(EntityArrays.APOS_ESCAPE()),
248 new LookupTranslator(EntityArrays.BASIC_ESCAPE()),
249 new LookupTranslator(
250 new String[][] {
251 { "\u0000", StringUtils.EMPTY },
252 { "\u000b", "" },
253 { "\u000c", "" },
254 { "\ufffe", StringUtils.EMPTY },
255 { "\uffff", StringUtils.EMPTY }
256 }),
257 NumericEntityEscaper.between(0x1, 0x8),
258 NumericEntityEscaper.between(0xe, 0x1f),
259 NumericEntityEscaper.between(0x7f, 0x84),
260 NumericEntityEscaper.between(0x86, 0x9f),
261 new UnicodeUnpairedSurrogateRemover()
262 );
263
264 /**
265 * Translator object for escaping HTML version 3.0.
266 *
267 * While {@link #escapeHtml3(String)} is the expected method of use, this
268 * object allows the HTML escaping functionality to be used
269 * as the foundation for a custom translator.
270 *
271 * @since 3.0
272 */
273 public static final CharSequenceTranslator ESCAPE_HTML3 =
274 new AggregateTranslator(
275 new LookupTranslator(EntityArrays.BASIC_ESCAPE()),
276 new LookupTranslator(EntityArrays.ISO8859_1_ESCAPE())
277 );
278
279 /**
280 * Translator object for escaping HTML version 4.0.
281 *
282 * While {@link #escapeHtml4(String)} is the expected method of use, this
283 * object allows the HTML escaping functionality to be used
284 * as the foundation for a custom translator.
285 *
286 * @since 3.0
287 */
288 public static final CharSequenceTranslator ESCAPE_HTML4 =
289 new AggregateTranslator(
290 new LookupTranslator(EntityArrays.BASIC_ESCAPE()),
291 new LookupTranslator(EntityArrays.ISO8859_1_ESCAPE()),
292 new LookupTranslator(EntityArrays.HTML40_EXTENDED_ESCAPE())
293 );
294
295 /* UNESCAPE TRANSLATORS */
296
297 /**
298 * Translator object for escaping individual Comma Separated Values.
299 *
300 * While {@link #escapeCsv(String)} is the expected method of use, this
301 * object allows the CSV escaping functionality to be used
302 * as the foundation for a custom translator.
303 *
304 * @since 3.0
305 */
306 public static final CharSequenceTranslator ESCAPE_CSV = new CsvEscaper();
307
308 /**
309 * Translator object for unescaping escaped Java.
310 *
311 * While {@link #unescapeJava(String)} is the expected method of use, this
312 * object allows the Java unescaping functionality to be used
313 * as the foundation for a custom translator.
314 *
315 * @since 3.0
316 */
317 // TODO: throw "illegal character: \92" as an Exception if a \ on the end of the Java (as per the compiler)?
318 public static final CharSequenceTranslator UNESCAPE_JAVA =
319 new AggregateTranslator(
320 new OctalUnescaper(), // .between('\1', '\377'),
321 new UnicodeUnescaper(),
322 new LookupTranslator(EntityArrays.JAVA_CTRL_CHARS_UNESCAPE()),
323 new LookupTranslator(
324 new String[][] {
325 {"\\\\", "\\"},
326 {"\\\"", "\""},
327 {"\\'", "'"},
328 {"\\", ""}
329 })
330 );
331
332 /**
333 * Translator object for unescaping escaped EcmaScript.
334 *
335 * While {@link #unescapeEcmaScript(String)} is the expected method of use, this
336 * object allows the EcmaScript unescaping functionality to be used
337 * as the foundation for a custom translator.
338 *
339 * @since 3.0
340 */
341 public static final CharSequenceTranslator UNESCAPE_ECMASCRIPT = UNESCAPE_JAVA;
342
343 /**
344 * Translator object for unescaping escaped Json.
345 *
346 * While {@link #unescapeJson(String)} is the expected method of use, this
347 * object allows the Json unescaping functionality to be used
348 * as the foundation for a custom translator.
349 *
350 * @since 3.2
351 */
352 public static final CharSequenceTranslator UNESCAPE_JSON = UNESCAPE_JAVA;
353
354 /**
355 * Translator object for unescaping escaped HTML 3.0.
356 *
357 * While {@link #unescapeHtml3(String)} is the expected method of use, this
358 * object allows the HTML unescaping functionality to be used
359 * as the foundation for a custom translator.
360 *
361 * @since 3.0
362 */
363 public static final CharSequenceTranslator UNESCAPE_HTML3 =
364 new AggregateTranslator(
365 new LookupTranslator(EntityArrays.BASIC_UNESCAPE()),
366 new LookupTranslator(EntityArrays.ISO8859_1_UNESCAPE()),
367 new NumericEntityUnescaper()
368 );
369
370 /**
371 * Translator object for unescaping escaped HTML 4.0.
372 *
373 * While {@link #unescapeHtml4(String)} is the expected method of use, this
374 * object allows the HTML unescaping functionality to be used
375 * as the foundation for a custom translator.
376 *
377 * @since 3.0
378 */
379 public static final CharSequenceTranslator UNESCAPE_HTML4 =
380 new AggregateTranslator(
381 new LookupTranslator(EntityArrays.BASIC_UNESCAPE()),
382 new LookupTranslator(EntityArrays.ISO8859_1_UNESCAPE()),
383 new LookupTranslator(EntityArrays.HTML40_EXTENDED_UNESCAPE()),
384 new NumericEntityUnescaper()
385 );
386
387 /**
388 * Translator object for unescaping escaped XML.
389 *
390 * While {@link #unescapeXml(String)} is the expected method of use, this
391 * object allows the XML unescaping functionality to be used
392 * as the foundation for a custom translator.
393 *
394 * @since 3.0
395 */
396 public static final CharSequenceTranslator UNESCAPE_XML =
397 new AggregateTranslator(
398 new LookupTranslator(EntityArrays.BASIC_UNESCAPE()),
399 new LookupTranslator(EntityArrays.APOS_UNESCAPE()),
400 new NumericEntityUnescaper()
401 );
402
403 /**
404 * Translator object for unescaping escaped Comma Separated Value entries.
405 *
406 * While {@link #unescapeCsv(String)} is the expected method of use, this
407 * object allows the CSV unescaping functionality to be used
408 * as the foundation for a custom translator.
409 *
410 * @since 3.0
411 */
412 public static final CharSequenceTranslator UNESCAPE_CSV = new CsvUnescaper();
413
414 /* Helper functions */
415
416 /**
417 * Returns a {@link String} value for a CSV column enclosed in double quotes, if required.
418 * <p>
419 * If the value contains a comma, newline or double quote, then the String value is returned enclosed in double quotes.
420 * </p>
421 * <p>
422 * Any double quote characters in the value are escaped with another double quote.
423 * </p>
424 * <p>
425 * If the value does not contain a comma, newline or double quote, then the String value is returned unchanged.
426 * </p>
427 * <p>
428 * See <a href="https://en.wikipedia.org/wiki/Comma-separated_values">Wikipedia</a> and <a href="https://datatracker.ietf.org/doc/html/rfc4180">RFC
429 * 4180</a>.
430 * </p>
431 *
432 * @param input The input CSV column String, may be null
433 * @return The input String, enclosed in double quotes if the value contains a comma, newline or double quote, {@code null} if null string input
434 * @since 2.4
435 */
436 public static final String escapeCsv(final String input) {
437 return ESCAPE_CSV.translate(input);
438 }
439
440 /**
441 * Escapes the characters in a {@link String} using EcmaScript String rules.
442 * <p>
443 * Escapes any values it finds into their EcmaScript String form. Escapes the EcmaScript string delimiters (single quote, double quote and (since ES6) the
444 * backtick) as well as the template-literal interpolation sequence <code>${</code> and control-chars (tab, backslash, cr, ff, etc.).
445 * </p>
446 * <p>
447 * So a tab becomes the characters {@code '\\'} and {@code 't'}.
448 * </p>
449 * <p>
450 * The differences between Java strings and EcmaScript strings handled here are that in EcmaScript, a single quote, the backtick, the <code>${</code>
451 * sequence and forward-slash (/) are escaped.
452 * </p>
453 * <p>
454 * <strong>Scope:</strong> the output is a correctly escaped EcmaScript string literal for any of the three string delimiters, but it is <em>not</em> made
455 * safe for direct embedding inside an HTML inline {@code <script>} block: HTML parser-state sequences such as {@code <!--} and {@code <script} pass through
456 * unchanged (a literal {@code </script>} is neutralized by the forward-slash escape). HTML-context encoding must be applied separately when the result is
457 * placed in HTML.
458 * </p>
459 * <p>
460 * Note that EcmaScript is best known by the JavaScript and ActionScript dialects.
461 * </p>
462 * <p>
463 * Example:
464 * </p>
465 *
466 * <pre>
467 * input string: He didn't say, "Stop!"
468 * output string: He didn\'t say, \"Stop!\"
469 * </pre>
470 *
471 * @param input String to escape values in, may be null
472 * @return String with escaped values, {@code null} if null string input
473 * @since 3.0
474 */
475 public static final String escapeEcmaScript(final String input) {
476 return ESCAPE_ECMASCRIPT.translate(input);
477 }
478
479 /**
480 * Escapes the characters in a {@link String} using HTML entities.
481 *
482 * <p>
483 * Supports only the HTML 3.0 entities. Apostrophes are escaped as the numeric reference {@code '}.
484 * </p>
485 * <p>
486 * HTML attribute values must be quoted. This method does not escape all characters that delimit unquoted attribute values.
487 * </p>
488 *
489 * @param input The {@link String} to escape, may be null
490 * @return A new escaped {@link String}, {@code null} if null string input
491 * @since 3.0
492 */
493 public static final String escapeHtml3(final String input) {
494 return ESCAPE_HTML3.translate(input);
495 }
496
497 /**
498 * Escapes the characters in a {@link String} using HTML entities.
499 *
500 * <p>
501 * For example:
502 * </p>
503 * <p>
504 * {@code "bread" & "butter"}
505 * </p>
506 * becomes:
507 * <p>
508 * {@code &quot;bread&quot; &amp; &quot;butter&quot;}.
509 * </p>
510 * <p>
511 * Supports all known HTML 4.0 entities, including funky accents.
512 * Apostrophes are escaped as the numeric reference {@code '} because {@code '} is not a legal HTML 4.0 entity.
513 * </p>
514 * <p>
515 * HTML attribute values must be quoted. This method does not escape all characters that delimit unquoted attribute values.
516 * </p>
517 *
518 * @param input The {@link String} to escape, may be null
519 * @return A new escaped {@link String}, {@code null} if null string input
520 * @see <a href="https://web.archive.org/web/20060225074150/https://hotwired.lycos.com/webmonkey/reference/special_characters/">ISO Entities</a>
521 * @see <a href="https://www.w3.org/TR/REC-html32#latin1">HTML 3.2 Character Entities for ISO Latin-1</a>
522 * @see <a href="https://www.w3.org/TR/REC-html40/sgml/entities.html">HTML 4.0 Character entity references</a>
523 * @see <a href="https://www.w3.org/TR/html401/charset.html#h-5.3">HTML 4.01 Character References</a>
524 * @see <a href="https://www.w3.org/TR/html401/charset.html#code-position">HTML 4.01 Code positions</a>
525 * @since 3.0
526 */
527 public static final String escapeHtml4(final String input) {
528 return ESCAPE_HTML4.translate(input);
529 }
530
531 /**
532 * Escapes the characters in a {@link String} using Java String rules.
533 *
534 * <p>
535 * Deals correctly with quotes and control-chars (tab, backslash, cr, ff, etc.)
536 * </p>
537 *
538 * <p>
539 * So a tab becomes the characters {@code '\\'} and
540 * {@code 't'}.
541 * </p>
542 *
543 * <p>
544 * The only difference between Java strings and JavaScript strings
545 * is that in JavaScript, a single quote and forward-slash (/) are escaped.
546 * </p>
547 *
548 * <p>
549 * Example:
550 * </p>
551 * <pre>
552 * input string: He didn't say, "Stop!"
553 * output string: He didn't say, \"Stop!\"
554 * </pre>
555 *
556 * @param input String to escape values in, may be null
557 * @return String with escaped values, {@code null} if null string input
558 */
559 public static final String escapeJava(final String input) {
560 return ESCAPE_JAVA.translate(input);
561 }
562
563 /**
564 * Escapes the characters in a {@link String} using Json String rules.
565 * <p>
566 * Escapes any values it finds into their JSON String form. Deals correctly with quotes and control-chars (tab, backslash, cr, ff, etc.)
567 * </p>
568 * <p>
569 * So a tab becomes the characters {@code '\\'} and {@code 't'}.
570 * </p>
571 * <p>
572 * The only difference between Java strings and Json strings is that in Json, forward-slash (/) is escaped.
573 * </p>
574 * <p>
575 * <strong>Scope:</strong> the output is a correctly escaped JSON string, but it is <em>not</em> made safe for direct embedding inside an HTML inline
576 * {@code <script>} block: the backtick, <code>${</code>, and HTML parser-state sequences such as {@code <!--} and {@code <script} pass through unchanged
577 * (JSON offers no backslash escape for them; only a literal {@code </script>} is neutralized by the forward-slash escape). HTML-context encoding must be
578 * applied separately when the result is placed in HTML.
579 * </p>
580 * <p>
581 * See https://www.ietf.org/rfc/rfc4627.txt for further details.
582 * </p>
583 * <p>
584 * Example:
585 * </p>
586 *
587 * <pre>
588 * input string: He didn't say, "Stop!"
589 * output string: He didn't say, \"Stop!\"
590 * </pre>
591 *
592 * @param input String to escape values in, may be null
593 * @return String with escaped values, {@code null} if null string input
594 * @since 3.2
595 */
596 public static final String escapeJson(final String input) {
597 return ESCAPE_JSON.translate(input);
598 }
599
600 /**
601 * Escapes the characters in a {@link String} using XML entities.
602 *
603 * <p>
604 * For example: {@code "bread" & "butter"} =>
605 * {@code "bread" & "butter"}.
606 * </p>
607 *
608 * <p>
609 * Supports only the five basic XML entities (gt, lt, quot, amp, apos).
610 * Does not support DTDs or external entities.
611 * </p>
612 *
613 * <p>
614 * Note that Unicode characters greater than 0x7f are as of 3.0, no longer
615 * escaped. If you still wish this functionality, you can achieve it
616 * via the following:
617 * {@code StringEscapeUtils.ESCAPE_XML.with( NumericEntityEscaper.between(0x7f, Integer.MAX_VALUE));}
618 * </p>
619 *
620 * @param input The {@link String} to escape, may be null
621 * @return A new escaped {@link String}, {@code null} if null string input
622 * @see #unescapeXml(String)
623 * @deprecated Use {@link #escapeXml10(java.lang.String)} or {@link #escapeXml11(java.lang.String)} instead.
624 */
625 @Deprecated
626 public static final String escapeXml(final String input) {
627 return ESCAPE_XML.translate(input);
628 }
629
630 /**
631 * Escapes the characters in a {@link String} using XML entities.
632 * <p>
633 * For example:
634 * </p>
635 *
636 * <pre>{@code
637 * "bread" & "butter"
638 * }</pre>
639 * <p>
640 * converts to:
641 * </p>
642 *
643 * <pre>
644 * {@code
645 * "bread" & "butter"
646 * }
647 * </pre>
648 *
649 * <p>
650 * Note that XML 1.0 is a text-only format: it cannot represent control characters or unpaired Unicode surrogate code points, even after escaping. The
651 * method {@code escapeXml10} will remove characters that do not fit in the following ranges:
652 * </p>
653 *
654 * <p>
655 * {@code #x9 | #xA | #xD | [#x20-#xD7FF] | [#xE000-#xFFFD] | [#x10000-#x10FFFF]}
656 * </p>
657 *
658 * <p>
659 * Though not strictly necessary, {@code escapeXml10} will escape characters in the following ranges:
660 * </p>
661 *
662 * <p>
663 * {@code [#x7F-#x84] | [#x86-#x9F]}
664 * </p>
665 *
666 * <p>
667 * The returned string can be inserted into a valid XML 1.0 or XML 1.1 document. If you want to allow more non-text characters in an XML 1.1 document, use
668 * {@link #escapeXml11(String)}.
669 * </p>
670 *
671 * @param input The {@link String} to escape, may be null
672 * @return A new escaped {@link String}, {@code null} if null string input
673 * @see #unescapeXml(String)
674 * @since 3.3
675 */
676 public static String escapeXml10(final String input) {
677 return ESCAPE_XML10.translate(input);
678 }
679
680 /**
681 * Escapes the characters in a {@link String} using XML entities.
682 *
683 * <p>
684 * For example: {@code "bread" & "butter"} =>
685 * {@code "bread" & "butter"}.
686 * </p>
687 *
688 * <p>
689 * XML 1.1 can represent certain control characters, but it cannot represent
690 * the null byte or unpaired Unicode surrogate code points, even after escaping.
691 * {@code escapeXml11} will remove characters that do not fit in the following
692 * ranges:
693 * </p>
694 *
695 * <p>
696 * {@code [#x1-#xD7FF] | [#xE000-#xFFFD] | [#x10000-#x10FFFF]}
697 * </p>
698 *
699 * <p>
700 * {@code escapeXml11} will escape characters in the following ranges:
701 * </p>
702 *
703 * <p>
704 * {@code [#x1-#x8] | [#xB-#xC] | [#xE-#x1F] | [#x7F-#x84] | [#x86-#x9F]}
705 * </p>
706 *
707 * <p>
708 * The returned string can be inserted into a valid XML 1.1 document. Do not
709 * use it for XML 1.0 documents.
710 * </p>
711 *
712 * @param input The {@link String} to escape, may be null
713 * @return A new escaped {@link String}, {@code null} if null string input
714 * @see #unescapeXml(String)
715 * @since 3.3
716 */
717 public static String escapeXml11(final String input) {
718 return ESCAPE_XML11.translate(input);
719 }
720
721 /**
722 * Returns a {@link String} value for an unescaped CSV column.
723 * <p>
724 * If the value is enclosed in double quotes, and contains a comma, newline or double quote, then quotes are removed.
725 * </p>
726 * <p>
727 * Any double quote escaped characters (a pair of double quotes) are unescaped to just one double quote.
728 * </p>
729 * <p>
730 * If the value is not enclosed in double quotes, or is and does not contain a comma, newline or double quote, then the String value is returned unchanged.
731 * </p>
732 * <p>
733 * See <a href="https://en.wikipedia.org/wiki/Comma-separated_values">Wikipedia</a> and <a href="https://datatracker.ietf.org/doc/html/rfc4180">RFC
734 * 4180</a>.
735 * </p>
736 *
737 * @param input The input CSV column String, may be null
738 * @return The input String, with enclosing double quotes removed and embedded double quotes unescaped, {@code null} if null string input
739 * @since 2.4
740 */
741 public static final String unescapeCsv(final String input) {
742 return UNESCAPE_CSV.translate(input);
743 }
744
745 /**
746 * Unescapes any EcmaScript literals found in the {@link String}.
747 *
748 * <p>
749 * For example, it will turn a sequence of {@code '\'} and {@code 'n'}
750 * into a newline character, unless the {@code '\'} is preceded by another
751 * {@code '\'}.
752 * </p>
753 *
754 * @see #unescapeJava(String)
755 * @param input The {@link String} to unescape, may be null
756 * @return A new unescaped {@link String}, {@code null} if null string input
757 * @since 3.0
758 */
759 public static final String unescapeEcmaScript(final String input) {
760 return UNESCAPE_ECMASCRIPT.translate(input);
761 }
762
763 /**
764 * Unescapes a string containing entity escapes to a string
765 * containing the actual Unicode characters corresponding to the
766 * escapes. Supports only HTML 3.0 entities.
767 *
768 * @param input The {@link String} to unescape, may be null
769 * @return A new unescaped {@link String}, {@code null} if null string input
770 * @since 3.0
771 */
772 public static final String unescapeHtml3(final String input) {
773 return UNESCAPE_HTML3.translate(input);
774 }
775
776 /**
777 * Unescapes a string containing entity escapes to a string
778 * containing the actual Unicode characters corresponding to the
779 * escapes. Supports HTML 4.0 entities.
780 *
781 * <p>
782 * For example, the string {@code "<Français>"}
783 * will become {@code "<Français>"}
784 * </p>
785 *
786 * <p>
787 * If an entity is unrecognized, it is left alone, and inserted
788 * verbatim into the result string. e.g. {@code ">&zzzz;x"} will
789 * become {@code ">&zzzz;x"}.
790 * </p>
791 *
792 * @param input The {@link String} to unescape, may be null
793 * @return A new unescaped {@link String}, {@code null} if null string input
794 * @since 3.0
795 */
796 public static final String unescapeHtml4(final String input) {
797 return UNESCAPE_HTML4.translate(input);
798 }
799
800 /**
801 * Unescapes any Java literals found in the {@link String}.
802 * For example, it will turn a sequence of {@code '\'} and
803 * {@code 'n'} into a newline character, unless the {@code '\'}
804 * is preceded by another {@code '\'}.
805 *
806 * @param input The {@link String} to unescape, may be null
807 * @return A new unescaped {@link String}, {@code null} if null string input
808 */
809 public static final String unescapeJava(final String input) {
810 return UNESCAPE_JAVA.translate(input);
811 }
812
813 /**
814 * Unescapes any Json literals found in the {@link String}.
815 *
816 * <p>
817 * For example, it will turn a sequence of {@code '\'} and {@code 'n'}
818 * into a newline character, unless the {@code '\'} is preceded by another
819 * {@code '\'}.
820 * </p>
821 *
822 * @see #unescapeJava(String)
823 * @param input The {@link String} to unescape, may be null
824 * @return A new unescaped {@link String}, {@code null} if null string input
825 * @since 3.2
826 */
827 public static final String unescapeJson(final String input) {
828 return UNESCAPE_JSON.translate(input);
829 }
830
831 /**
832 * Unescapes a string containing XML entity escapes to a string
833 * containing the actual Unicode characters corresponding to the
834 * escapes.
835 *
836 * <p>
837 * Supports only the five basic XML entities (gt, lt, quot, amp, apos).
838 * Does not support DTDs or external entities.
839 * </p>
840 *
841 * <p>
842 * Note that numerical \\u Unicode codes are unescaped to their respective
843 * Unicode characters. This may change in future releases.
844 * </p>
845 *
846 * @param input The {@link String} to unescape, may be null
847 * @return A new unescaped {@link String}, {@code null} if null string input
848 * @see #escapeXml(String)
849 * @see #escapeXml10(String)
850 * @see #escapeXml11(String)
851 */
852 public static final String unescapeXml(final String input) {
853 return UNESCAPE_XML.translate(input);
854 }
855
856 /**
857 * {@link StringEscapeUtils} instances should NOT be constructed in
858 * standard programming.
859 *
860 * <p>
861 * Instead, the class should be used as:
862 * </p>
863 * <pre>StringEscapeUtils.escapeJava("foo");</pre>
864 *
865 * <p>
866 * This constructor is public to permit tools that require a JavaBean
867 * instance to operate.
868 * </p>
869 *
870 * @deprecated TODO Make private in 4.0.
871 */
872 @Deprecated
873 public StringEscapeUtils() {
874 // empty
875 }
876
877 }