001/*
002 * Licensed to the Apache Software Foundation (ASF) under one or more
003 * contributor license agreements.  See the NOTICE file distributed with
004 * this work for additional information regarding copyright ownership.
005 * The ASF licenses this file to You under the Apache License, Version 2.0
006 * (the "License"); you may not use this file except in compliance with
007 * the License.  You may obtain a copy of the License at
008 *
009 *      https://www.apache.org/licenses/LICENSE-2.0
010 *
011 * Unless required by applicable law or agreed to in writing, software
012 * distributed under the License is distributed on an "AS IS" BASIS,
013 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
014 * See the License for the specific language governing permissions and
015 * limitations under the License.
016 */
017package org.apache.commons.lang3;
018
019import java.io.IOException;
020import java.io.Writer;
021
022import org.apache.commons.lang3.text.translate.AggregateTranslator;
023import org.apache.commons.lang3.text.translate.CharSequenceTranslator;
024import org.apache.commons.lang3.text.translate.EntityArrays;
025import org.apache.commons.lang3.text.translate.JavaUnicodeEscaper;
026import org.apache.commons.lang3.text.translate.LookupTranslator;
027import org.apache.commons.lang3.text.translate.NumericEntityEscaper;
028import org.apache.commons.lang3.text.translate.NumericEntityUnescaper;
029import org.apache.commons.lang3.text.translate.OctalUnescaper;
030import org.apache.commons.lang3.text.translate.UnicodeUnescaper;
031import org.apache.commons.lang3.text.translate.UnicodeUnpairedSurrogateRemover;
032
033/**
034 * Escapes and unescapes {@link String}s for
035 * Java, Java Script, HTML and XML.
036 *
037 * <p>
038 * #ThreadSafe#
039 * </p>
040 *
041 * @since 2.0
042 * @deprecated As of 3.6, use Apache Commons Text
043 * <a href="https://commons.apache.org/proper/commons-text/javadocs/api-release/org/apache/commons/text/StringEscapeUtils.html">
044 * StringEscapeUtils</a> instead.
045 */
046@Deprecated
047public class StringEscapeUtils {
048
049    /* ESCAPE TRANSLATORS */
050
051    private static final class CsvEscaper extends CharSequenceTranslator {
052
053        private static final char CSV_DELIMITER = ',';
054        private static final char CSV_QUOTE = '"';
055        private static final String CSV_QUOTE_STR = String.valueOf(CSV_QUOTE);
056        private static final char[] CSV_SEARCH_CHARS = { CSV_DELIMITER, CSV_QUOTE, CharUtils.CR, CharUtils.LF };
057
058        @Override
059        public int translate(final CharSequence input, final int index, final Writer out) throws IOException {
060            if (index != 0) {
061                throw new IllegalStateException("CsvEscaper should never reach the [1] index");
062            }
063            if (StringUtils.containsNone(input.toString(), CSV_SEARCH_CHARS)) {
064                out.write(input.toString());
065            } else {
066                out.write(CSV_QUOTE);
067                out.write(Strings.CS.replace(input.toString(), CSV_QUOTE_STR, CSV_QUOTE_STR + CSV_QUOTE_STR));
068                out.write(CSV_QUOTE);
069            }
070            return Character.codePointCount(input, 0, input.length());
071        }
072    }
073
074    private static final class CsvUnescaper extends CharSequenceTranslator {
075
076        private static final char CSV_DELIMITER = ',';
077        private static final char CSV_QUOTE = '"';
078        private static final String CSV_QUOTE_STR = String.valueOf(CSV_QUOTE);
079        private static final char[] CSV_SEARCH_CHARS = {CSV_DELIMITER, CSV_QUOTE, CharUtils.CR, CharUtils.LF};
080
081        @Override
082        public int translate(final CharSequence input, final int index, final Writer out) throws IOException {
083            if (index != 0) {
084                throw new IllegalStateException("CsvUnescaper should never reach the [1] index");
085            }
086            if (input.length() < 2 || input.charAt(0) != CSV_QUOTE || input.charAt(input.length() - 1) != CSV_QUOTE) {
087                out.write(input.toString());
088                return Character.codePointCount(input, 0, input.length());
089            }
090            // strip quotes
091            final String quoteless = input.subSequence(1, input.length() - 1).toString();
092            if (StringUtils.containsAny(quoteless, CSV_SEARCH_CHARS)) {
093                // deal with escaped quotes; ie) ""
094                out.write(Strings.CS.replace(quoteless, CSV_QUOTE_STR + CSV_QUOTE_STR, CSV_QUOTE_STR));
095            } else {
096                out.write(input.toString());
097            }
098            return Character.codePointCount(input, 0, input.length());
099        }
100    }
101
102    /**
103     * Translator object for escaping Java.
104     *
105     * While {@link #escapeJava(String)} is the expected method of use, this
106     * object allows the Java escaping functionality to be used
107     * as the foundation for a custom translator.
108     *
109     * @since 3.0
110     */
111    public static final CharSequenceTranslator ESCAPE_JAVA =
112          new LookupTranslator(
113            new String[][] {
114              {"\"", "\\\""},
115              {"\\", "\\\\"},
116          }).with(
117            new LookupTranslator(EntityArrays.JAVA_CTRL_CHARS_ESCAPE())
118          ).with(
119            JavaUnicodeEscaper.outsideOf(32, 0x7f)
120        );
121
122    /**
123     * Translator object for escaping EcmaScript/JavaScript.
124     *
125     * While {@link #escapeEcmaScript(String)} is the expected method of use, this
126     * object allows the EcmaScript escaping functionality to be used
127     * as the foundation for a custom translator.
128     *
129     * @since 3.0
130     */
131    public static final CharSequenceTranslator ESCAPE_ECMASCRIPT =
132        new AggregateTranslator(
133            new LookupTranslator(
134                      new String[][] {
135                            {"'", "\\'"},
136                            {"\"", "\\\""},
137                            {"`", "\\`"},
138                            {"${", "\\${"},
139                            {"\\", "\\\\"},
140                            {"/", "\\/"}
141                      }),
142            new LookupTranslator(EntityArrays.JAVA_CTRL_CHARS_ESCAPE()),
143            JavaUnicodeEscaper.outsideOf(32, 0x7f)
144        );
145
146    /**
147     * Translator object for escaping Json.
148     *
149     * While {@link #escapeJson(String)} is the expected method of use, this
150     * object allows the Json escaping functionality to be used
151     * as the foundation for a custom translator.
152     *
153     * @since 3.2
154     */
155    public static final CharSequenceTranslator ESCAPE_JSON =
156        new AggregateTranslator(
157            new LookupTranslator(
158                      new String[][] {
159                            {"\"", "\\\""},
160                            {"\\", "\\\\"},
161                            {"/", "\\/"}
162                      }),
163            new LookupTranslator(EntityArrays.JAVA_CTRL_CHARS_ESCAPE()),
164            JavaUnicodeEscaper.outsideOf(32, 0x7f)
165        );
166
167    /**
168     * Translator object for escaping XML.
169     *
170     * While {@link #escapeXml(String)} is the expected method of use, this
171     * object allows the XML escaping functionality to be used
172     * as the foundation for a custom translator.
173     *
174     * @since 3.0
175     * @deprecated Use {@link #ESCAPE_XML10} or {@link #ESCAPE_XML11} instead.
176     */
177    @Deprecated
178    public static final CharSequenceTranslator ESCAPE_XML =
179        new AggregateTranslator(
180            new LookupTranslator(EntityArrays.APOS_ESCAPE()),
181            new LookupTranslator(EntityArrays.BASIC_ESCAPE())
182        );
183
184    /**
185     * Translator object for escaping XML 1.0.
186     *
187     * While {@link #escapeXml10(String)} is the expected method of use, this
188     * object allows the XML escaping functionality to be used
189     * as the foundation for a custom translator.
190     *
191     * @since 3.3
192     */
193    public static final CharSequenceTranslator ESCAPE_XML10 =
194        new AggregateTranslator(
195            new LookupTranslator(EntityArrays.APOS_ESCAPE()),
196            new LookupTranslator(EntityArrays.BASIC_ESCAPE()),
197            new LookupTranslator(
198                    new String[][] {
199                            { "\u0000", StringUtils.EMPTY },
200                            { "\u0001", StringUtils.EMPTY },
201                            { "\u0002", StringUtils.EMPTY },
202                            { "\u0003", StringUtils.EMPTY },
203                            { "\u0004", StringUtils.EMPTY },
204                            { "\u0005", StringUtils.EMPTY },
205                            { "\u0006", StringUtils.EMPTY },
206                            { "\u0007", StringUtils.EMPTY },
207                            { "\u0008", StringUtils.EMPTY },
208                            { "\u000b", StringUtils.EMPTY },
209                            { "\u000c", StringUtils.EMPTY },
210                            { "\u000e", StringUtils.EMPTY },
211                            { "\u000f", StringUtils.EMPTY },
212                            { "\u0010", StringUtils.EMPTY },
213                            { "\u0011", StringUtils.EMPTY },
214                            { "\u0012", StringUtils.EMPTY },
215                            { "\u0013", StringUtils.EMPTY },
216                            { "\u0014", StringUtils.EMPTY },
217                            { "\u0015", StringUtils.EMPTY },
218                            { "\u0016", StringUtils.EMPTY },
219                            { "\u0017", StringUtils.EMPTY },
220                            { "\u0018", StringUtils.EMPTY },
221                            { "\u0019", StringUtils.EMPTY },
222                            { "\u001a", StringUtils.EMPTY },
223                            { "\u001b", StringUtils.EMPTY },
224                            { "\u001c", StringUtils.EMPTY },
225                            { "\u001d", StringUtils.EMPTY },
226                            { "\u001e", StringUtils.EMPTY },
227                            { "\u001f", StringUtils.EMPTY },
228                            { "\ufffe", StringUtils.EMPTY },
229                            { "\uffff", StringUtils.EMPTY }
230                    }),
231            NumericEntityEscaper.between(0x7f, 0x84),
232            NumericEntityEscaper.between(0x86, 0x9f),
233            new UnicodeUnpairedSurrogateRemover()
234        );
235
236    /**
237     * Translator object for escaping XML 1.1.
238     *
239     * While {@link #escapeXml11(String)} is the expected method of use, this
240     * object allows the XML escaping functionality to be used
241     * as the foundation for a custom translator.
242     *
243     * @since 3.3
244     */
245    public static final CharSequenceTranslator ESCAPE_XML11 =
246        new AggregateTranslator(
247            new LookupTranslator(EntityArrays.APOS_ESCAPE()),
248            new LookupTranslator(EntityArrays.BASIC_ESCAPE()),
249            new LookupTranslator(
250                    new String[][] {
251                            { "\u0000", StringUtils.EMPTY },
252                            { "\u000b", "&#11;" },
253                            { "\u000c", "&#12;" },
254                            { "\ufffe", StringUtils.EMPTY },
255                            { "\uffff", StringUtils.EMPTY }
256                    }),
257            NumericEntityEscaper.between(0x1, 0x8),
258            NumericEntityEscaper.between(0xe, 0x1f),
259            NumericEntityEscaper.between(0x7f, 0x84),
260            NumericEntityEscaper.between(0x86, 0x9f),
261            new UnicodeUnpairedSurrogateRemover()
262        );
263
264    /**
265     * Translator object for escaping HTML version 3.0.
266     *
267     * While {@link #escapeHtml3(String)} is the expected method of use, this
268     * object allows the HTML escaping functionality to be used
269     * as the foundation for a custom translator.
270     *
271     * @since 3.0
272     */
273    public static final CharSequenceTranslator ESCAPE_HTML3 =
274        new AggregateTranslator(
275            new LookupTranslator(EntityArrays.BASIC_ESCAPE()),
276            new LookupTranslator(EntityArrays.ISO8859_1_ESCAPE())
277        );
278
279    /**
280     * Translator object for escaping HTML version 4.0.
281     *
282     * While {@link #escapeHtml4(String)} is the expected method of use, this
283     * object allows the HTML escaping functionality to be used
284     * as the foundation for a custom translator.
285     *
286     * @since 3.0
287     */
288    public static final CharSequenceTranslator ESCAPE_HTML4 =
289        new AggregateTranslator(
290            new LookupTranslator(EntityArrays.BASIC_ESCAPE()),
291            new LookupTranslator(EntityArrays.ISO8859_1_ESCAPE()),
292            new LookupTranslator(EntityArrays.HTML40_EXTENDED_ESCAPE())
293        );
294
295    /* UNESCAPE TRANSLATORS */
296
297    /**
298     * Translator object for escaping individual Comma Separated Values.
299     *
300     * While {@link #escapeCsv(String)} is the expected method of use, this
301     * object allows the CSV escaping functionality to be used
302     * as the foundation for a custom translator.
303     *
304     * @since 3.0
305     */
306    public static final CharSequenceTranslator ESCAPE_CSV = new CsvEscaper();
307
308    /**
309     * Translator object for unescaping escaped Java.
310     *
311     * While {@link #unescapeJava(String)} is the expected method of use, this
312     * object allows the Java unescaping functionality to be used
313     * as the foundation for a custom translator.
314     *
315     * @since 3.0
316     */
317    // TODO: throw "illegal character: \92" as an Exception if a \ on the end of the Java (as per the compiler)?
318    public static final CharSequenceTranslator UNESCAPE_JAVA =
319        new AggregateTranslator(
320            new OctalUnescaper(),     // .between('\1', '\377'),
321            new UnicodeUnescaper(),
322            new LookupTranslator(EntityArrays.JAVA_CTRL_CHARS_UNESCAPE()),
323            new LookupTranslator(
324                      new String[][] {
325                            {"\\\\", "\\"},
326                            {"\\\"", "\""},
327                            {"\\'", "'"},
328                            {"\\", ""}
329                      })
330        );
331
332    /**
333     * Translator object for unescaping escaped EcmaScript.
334     *
335     * While {@link #unescapeEcmaScript(String)} is the expected method of use, this
336     * object allows the EcmaScript unescaping functionality to be used
337     * as the foundation for a custom translator.
338     *
339     * @since 3.0
340     */
341    public static final CharSequenceTranslator UNESCAPE_ECMASCRIPT = UNESCAPE_JAVA;
342
343    /**
344     * Translator object for unescaping escaped Json.
345     *
346     * While {@link #unescapeJson(String)} is the expected method of use, this
347     * object allows the Json unescaping functionality to be used
348     * as the foundation for a custom translator.
349     *
350     * @since 3.2
351     */
352    public static final CharSequenceTranslator UNESCAPE_JSON = UNESCAPE_JAVA;
353
354    /**
355     * Translator object for unescaping escaped HTML 3.0.
356     *
357     * While {@link #unescapeHtml3(String)} is the expected method of use, this
358     * object allows the HTML unescaping functionality to be used
359     * as the foundation for a custom translator.
360     *
361     * @since 3.0
362     */
363    public static final CharSequenceTranslator UNESCAPE_HTML3 =
364        new AggregateTranslator(
365            new LookupTranslator(EntityArrays.BASIC_UNESCAPE()),
366            new LookupTranslator(EntityArrays.ISO8859_1_UNESCAPE()),
367            new NumericEntityUnescaper()
368        );
369
370    /**
371     * Translator object for unescaping escaped HTML 4.0.
372     *
373     * While {@link #unescapeHtml4(String)} is the expected method of use, this
374     * object allows the HTML unescaping functionality to be used
375     * as the foundation for a custom translator.
376     *
377     * @since 3.0
378     */
379    public static final CharSequenceTranslator UNESCAPE_HTML4 =
380        new AggregateTranslator(
381            new LookupTranslator(EntityArrays.BASIC_UNESCAPE()),
382            new LookupTranslator(EntityArrays.ISO8859_1_UNESCAPE()),
383            new LookupTranslator(EntityArrays.HTML40_EXTENDED_UNESCAPE()),
384            new NumericEntityUnescaper()
385        );
386
387    /**
388     * Translator object for unescaping escaped XML.
389     *
390     * While {@link #unescapeXml(String)} is the expected method of use, this
391     * object allows the XML unescaping functionality to be used
392     * as the foundation for a custom translator.
393     *
394     * @since 3.0
395     */
396    public static final CharSequenceTranslator UNESCAPE_XML =
397        new AggregateTranslator(
398            new LookupTranslator(EntityArrays.BASIC_UNESCAPE()),
399            new LookupTranslator(EntityArrays.APOS_UNESCAPE()),
400            new NumericEntityUnescaper()
401        );
402
403    /**
404     * Translator object for unescaping escaped Comma Separated Value entries.
405     *
406     * While {@link #unescapeCsv(String)} is the expected method of use, this
407     * object allows the CSV unescaping functionality to be used
408     * as the foundation for a custom translator.
409     *
410     * @since 3.0
411     */
412    public static final CharSequenceTranslator UNESCAPE_CSV = new CsvUnescaper();
413
414    /* Helper functions */
415
416    /**
417     * Returns a {@link String} value for a CSV column enclosed in double quotes, if required.
418     * <p>
419     * If the value contains a comma, newline or double quote, then the String value is returned enclosed in double quotes.
420     * </p>
421     * <p>
422     * Any double quote characters in the value are escaped with another double quote.
423     * </p>
424     * <p>
425     * If the value does not contain a comma, newline or double quote, then the String value is returned unchanged.
426     * </p>
427     * <p>
428     * See <a href="https://en.wikipedia.org/wiki/Comma-separated_values">Wikipedia</a> and <a href="https://datatracker.ietf.org/doc/html/rfc4180">RFC
429     * 4180</a>.
430     * </p>
431     *
432     * @param input The input CSV column String, may be null
433     * @return The input String, enclosed in double quotes if the value contains a comma, newline or double quote, {@code null} if null string input
434     * @since 2.4
435     */
436    public static final String escapeCsv(final String input) {
437        return ESCAPE_CSV.translate(input);
438    }
439
440    /**
441     * Escapes the characters in a {@link String} using EcmaScript String rules.
442     * <p>
443     * Escapes any values it finds into their EcmaScript String form. Escapes the EcmaScript string delimiters (single quote, double quote and (since ES6) the
444     * backtick) as well as the template-literal interpolation sequence <code>${</code> and control-chars (tab, backslash, cr, ff, etc.).
445     * </p>
446     * <p>
447     * So a tab becomes the characters {@code '\\'} and {@code 't'}.
448     * </p>
449     * <p>
450     * The differences between Java strings and EcmaScript strings handled here are that in EcmaScript, a single quote, the backtick, the <code>${</code>
451     * sequence and forward-slash (/) are escaped.
452     * </p>
453     * <p>
454     * <strong>Scope:</strong> the output is a correctly escaped EcmaScript string literal for any of the three string delimiters, but it is <em>not</em> made
455     * safe for direct embedding inside an HTML inline {@code <script>} block: HTML parser-state sequences such as {@code <!--} and {@code <script} pass through
456     * unchanged (a literal {@code </script>} is neutralized by the forward-slash escape). HTML-context encoding must be applied separately when the result is
457     * placed in HTML.
458     * </p>
459     * <p>
460     * Note that EcmaScript is best known by the JavaScript and ActionScript dialects.
461     * </p>
462     * <p>
463     * Example:
464     * </p>
465     *
466     * <pre>
467     * input string: He didn't say, "Stop!"
468     * output string: He didn\'t say, \"Stop!\"
469     * </pre>
470     *
471     * @param input String to escape values in, may be null
472     * @return String with escaped values, {@code null} if null string input
473     * @since 3.0
474     */
475    public static final String escapeEcmaScript(final String input) {
476        return ESCAPE_ECMASCRIPT.translate(input);
477    }
478
479    /**
480     * Escapes the characters in a {@link String} using HTML entities.
481     *
482     * <p>
483     * Supports only the HTML 3.0 entities. Apostrophes are escaped as the numeric reference {@code &#39;}.
484     * </p>
485     * <p>
486     * HTML attribute values must be quoted. This method does not escape all characters that delimit unquoted attribute values.
487     * </p>
488     *
489     * @param input  The {@link String} to escape, may be null
490     * @return A new escaped {@link String}, {@code null} if null string input
491     * @since 3.0
492     */
493    public static final String escapeHtml3(final String input) {
494        return ESCAPE_HTML3.translate(input);
495    }
496
497    /**
498     * Escapes the characters in a {@link String} using HTML entities.
499     *
500     * <p>
501     * For example:
502     * </p>
503     * <p>
504     * {@code "bread" &amp; "butter"}
505     * </p>
506     * becomes:
507     * <p>
508     * {@code &amp;quot;bread&amp;quot; &amp;amp; &amp;quot;butter&amp;quot;}.
509     * </p>
510     * <p>
511     * Supports all known HTML 4.0 entities, including funky accents.
512     * Apostrophes are escaped as the numeric reference {@code &#39;} because {@code &apos;} is not a legal HTML 4.0 entity.
513     * </p>
514     * <p>
515     * HTML attribute values must be quoted. This method does not escape all characters that delimit unquoted attribute values.
516     * </p>
517     *
518     * @param input  The {@link String} to escape, may be null
519     * @return A new escaped {@link String}, {@code null} if null string input
520     * @see <a href="https://web.archive.org/web/20060225074150/https://hotwired.lycos.com/webmonkey/reference/special_characters/">ISO Entities</a>
521     * @see <a href="https://www.w3.org/TR/REC-html32#latin1">HTML 3.2 Character Entities for ISO Latin-1</a>
522     * @see <a href="https://www.w3.org/TR/REC-html40/sgml/entities.html">HTML 4.0 Character entity references</a>
523     * @see <a href="https://www.w3.org/TR/html401/charset.html#h-5.3">HTML 4.01 Character References</a>
524     * @see <a href="https://www.w3.org/TR/html401/charset.html#code-position">HTML 4.01 Code positions</a>
525     * @since 3.0
526     */
527    public static final String escapeHtml4(final String input) {
528        return ESCAPE_HTML4.translate(input);
529    }
530
531    /**
532     * Escapes the characters in a {@link String} using Java String rules.
533     *
534     * <p>
535     * Deals correctly with quotes and control-chars (tab, backslash, cr, ff, etc.)
536     * </p>
537     *
538     * <p>
539     * So a tab becomes the characters {@code '\\'} and
540     * {@code 't'}.
541     * </p>
542     *
543     * <p>
544     * The only difference between Java strings and JavaScript strings
545     * is that in JavaScript, a single quote and forward-slash (/) are escaped.
546     * </p>
547     *
548     * <p>
549     * Example:
550     * </p>
551     * <pre>
552     * input string: He didn't say, "Stop!"
553     * output string: He didn't say, \"Stop!\"
554     * </pre>
555     *
556     * @param input  String to escape values in, may be null
557     * @return String with escaped values, {@code null} if null string input
558     */
559    public static final String escapeJava(final String input) {
560        return ESCAPE_JAVA.translate(input);
561    }
562
563    /**
564     * Escapes the characters in a {@link String} using Json String rules.
565     * <p>
566     * Escapes any values it finds into their JSON String form. Deals correctly with quotes and control-chars (tab, backslash, cr, ff, etc.)
567     * </p>
568     * <p>
569     * So a tab becomes the characters {@code '\\'} and {@code 't'}.
570     * </p>
571     * <p>
572     * The only difference between Java strings and Json strings is that in Json, forward-slash (/) is escaped.
573     * </p>
574     * <p>
575     * <strong>Scope:</strong> the output is a correctly escaped JSON string, but it is <em>not</em> made safe for direct embedding inside an HTML inline
576     * {@code <script>} block: the backtick, <code>${</code>, and HTML parser-state sequences such as {@code <!--} and {@code <script} pass through unchanged
577     * (JSON offers no backslash escape for them; only a literal {@code </script>} is neutralized by the forward-slash escape). HTML-context encoding must be
578     * applied separately when the result is placed in HTML.
579     * </p>
580     * <p>
581     * See https://www.ietf.org/rfc/rfc4627.txt for further details.
582     * </p>
583     * <p>
584     * Example:
585     * </p>
586     *
587     * <pre>
588     * input string: He didn't say, "Stop!"
589     * output string: He didn't say, \"Stop!\"
590     * </pre>
591     *
592     * @param input String to escape values in, may be null
593     * @return String with escaped values, {@code null} if null string input
594     * @since 3.2
595     */
596    public static final String escapeJson(final String input) {
597        return ESCAPE_JSON.translate(input);
598    }
599
600    /**
601     * Escapes the characters in a {@link String} using XML entities.
602     *
603     * <p>
604     * For example: {@code "bread" & "butter"} =&gt;
605     * {@code &quot;bread&quot; &amp; &quot;butter&quot;}.
606     * </p>
607     *
608     * <p>
609     * Supports only the five basic XML entities (gt, lt, quot, amp, apos).
610     * Does not support DTDs or external entities.
611     * </p>
612     *
613     * <p>
614     * Note that Unicode characters greater than 0x7f are as of 3.0, no longer
615     *    escaped. If you still wish this functionality, you can achieve it
616     *    via the following:
617     * {@code StringEscapeUtils.ESCAPE_XML.with( NumericEntityEscaper.between(0x7f, Integer.MAX_VALUE));}
618     * </p>
619     *
620     * @param input  The {@link String} to escape, may be null
621     * @return A new escaped {@link String}, {@code null} if null string input
622     * @see #unescapeXml(String)
623     * @deprecated Use {@link #escapeXml10(java.lang.String)} or {@link #escapeXml11(java.lang.String)} instead.
624     */
625    @Deprecated
626    public static final String escapeXml(final String input) {
627        return ESCAPE_XML.translate(input);
628    }
629
630    /**
631     * Escapes the characters in a {@link String} using XML entities.
632     * <p>
633     * For example:
634     * </p>
635     *
636     * <pre>{@code
637     * "bread" & "butter"
638     * }</pre>
639     * <p>
640     * converts to:
641     * </p>
642     *
643     * <pre>
644     * {@code
645     * &quot;bread&quot; &amp; &quot;butter&quot;
646     * }
647     * </pre>
648     *
649     * <p>
650     * Note that XML 1.0 is a text-only format: it cannot represent control characters or unpaired Unicode surrogate code points, even after escaping. The
651     * method {@code escapeXml10} will remove characters that do not fit in the following ranges:
652     * </p>
653     *
654     * <p>
655     * {@code #x9 | #xA | #xD | [#x20-#xD7FF] | [#xE000-#xFFFD] | [#x10000-#x10FFFF]}
656     * </p>
657     *
658     * <p>
659     * Though not strictly necessary, {@code escapeXml10} will escape characters in the following ranges:
660     * </p>
661     *
662     * <p>
663     * {@code [#x7F-#x84] | [#x86-#x9F]}
664     * </p>
665     *
666     * <p>
667     * The returned string can be inserted into a valid XML 1.0 or XML 1.1 document. If you want to allow more non-text characters in an XML 1.1 document, use
668     * {@link #escapeXml11(String)}.
669     * </p>
670     *
671     * @param input The {@link String} to escape, may be null
672     * @return A new escaped {@link String}, {@code null} if null string input
673     * @see #unescapeXml(String)
674     * @since 3.3
675     */
676    public static String escapeXml10(final String input) {
677        return ESCAPE_XML10.translate(input);
678    }
679
680    /**
681     * Escapes the characters in a {@link String} using XML entities.
682     *
683     * <p>
684     * For example: {@code "bread" & "butter"} =&gt;
685     * {@code &quot;bread&quot; &amp; &quot;butter&quot;}.
686     * </p>
687     *
688     * <p>
689     * XML 1.1 can represent certain control characters, but it cannot represent
690     * the null byte or unpaired Unicode surrogate code points, even after escaping.
691     * {@code escapeXml11} will remove characters that do not fit in the following
692     * ranges:
693     * </p>
694     *
695     * <p>
696     * {@code [#x1-#xD7FF] | [#xE000-#xFFFD] | [#x10000-#x10FFFF]}
697     * </p>
698     *
699     * <p>
700     * {@code escapeXml11} will escape characters in the following ranges:
701     * </p>
702     *
703     * <p>
704     * {@code [#x1-#x8] | [#xB-#xC] | [#xE-#x1F] | [#x7F-#x84] | [#x86-#x9F]}
705     * </p>
706     *
707     * <p>
708     * The returned string can be inserted into a valid XML 1.1 document. Do not
709     * use it for XML 1.0 documents.
710     * </p>
711     *
712     * @param input  The {@link String} to escape, may be null
713     * @return A new escaped {@link String}, {@code null} if null string input
714     * @see #unescapeXml(String)
715     * @since 3.3
716     */
717    public static String escapeXml11(final String input) {
718        return ESCAPE_XML11.translate(input);
719    }
720
721    /**
722     * Returns a {@link String} value for an unescaped CSV column.
723     * <p>
724     * If the value is enclosed in double quotes, and contains a comma, newline or double quote, then quotes are removed.
725     * </p>
726     * <p>
727     * Any double quote escaped characters (a pair of double quotes) are unescaped to just one double quote.
728     * </p>
729     * <p>
730     * If the value is not enclosed in double quotes, or is and does not contain a comma, newline or double quote, then the String value is returned unchanged.
731     * </p>
732     * <p>
733     * See <a href="https://en.wikipedia.org/wiki/Comma-separated_values">Wikipedia</a> and <a href="https://datatracker.ietf.org/doc/html/rfc4180">RFC
734     * 4180</a>.
735     * </p>
736     *
737     * @param input The input CSV column String, may be null
738     * @return The input String, with enclosing double quotes removed and embedded double quotes unescaped, {@code null} if null string input
739     * @since 2.4
740     */
741    public static final String unescapeCsv(final String input) {
742        return UNESCAPE_CSV.translate(input);
743    }
744
745    /**
746     * Unescapes any EcmaScript literals found in the {@link String}.
747     *
748     * <p>
749     * For example, it will turn a sequence of {@code '\'} and {@code 'n'}
750     * into a newline character, unless the {@code '\'} is preceded by another
751     * {@code '\'}.
752     * </p>
753     *
754     * @see #unescapeJava(String)
755     * @param input  The {@link String} to unescape, may be null
756     * @return A new unescaped {@link String}, {@code null} if null string input
757     * @since 3.0
758     */
759    public static final String unescapeEcmaScript(final String input) {
760        return UNESCAPE_ECMASCRIPT.translate(input);
761    }
762
763    /**
764     * Unescapes a string containing entity escapes to a string
765     * containing the actual Unicode characters corresponding to the
766     * escapes. Supports only HTML 3.0 entities.
767     *
768     * @param input  The {@link String} to unescape, may be null
769     * @return A new unescaped {@link String}, {@code null} if null string input
770     * @since 3.0
771     */
772    public static final String unescapeHtml3(final String input) {
773        return UNESCAPE_HTML3.translate(input);
774    }
775
776    /**
777     * Unescapes a string containing entity escapes to a string
778     * containing the actual Unicode characters corresponding to the
779     * escapes. Supports HTML 4.0 entities.
780     *
781     * <p>
782     * For example, the string {@code "&lt;Fran&ccedil;ais&gt;"}
783     * will become {@code "<Français>"}
784     * </p>
785     *
786     * <p>
787     * If an entity is unrecognized, it is left alone, and inserted
788     * verbatim into the result string. e.g. {@code "&gt;&zzzz;x"} will
789     * become {@code ">&zzzz;x"}.
790     * </p>
791     *
792     * @param input  The {@link String} to unescape, may be null
793     * @return A new unescaped {@link String}, {@code null} if null string input
794     * @since 3.0
795     */
796    public static final String unescapeHtml4(final String input) {
797        return UNESCAPE_HTML4.translate(input);
798    }
799
800    /**
801     * Unescapes any Java literals found in the {@link String}.
802     * For example, it will turn a sequence of {@code '\'} and
803     * {@code 'n'} into a newline character, unless the {@code '\'}
804     * is preceded by another {@code '\'}.
805     *
806     * @param input  The {@link String} to unescape, may be null
807     * @return A new unescaped {@link String}, {@code null} if null string input
808     */
809    public static final String unescapeJava(final String input) {
810        return UNESCAPE_JAVA.translate(input);
811    }
812
813    /**
814     * Unescapes any Json literals found in the {@link String}.
815     *
816     * <p>
817     * For example, it will turn a sequence of {@code '\'} and {@code 'n'}
818     * into a newline character, unless the {@code '\'} is preceded by another
819     * {@code '\'}.
820     * </p>
821     *
822     * @see #unescapeJava(String)
823     * @param input  The {@link String} to unescape, may be null
824     * @return A new unescaped {@link String}, {@code null} if null string input
825     * @since 3.2
826     */
827    public static final String unescapeJson(final String input) {
828        return UNESCAPE_JSON.translate(input);
829    }
830
831    /**
832     * Unescapes a string containing XML entity escapes to a string
833     * containing the actual Unicode characters corresponding to the
834     * escapes.
835     *
836     * <p>
837     * Supports only the five basic XML entities (gt, lt, quot, amp, apos).
838     * Does not support DTDs or external entities.
839     * </p>
840     *
841     * <p>
842     * Note that numerical \\u Unicode codes are unescaped to their respective
843     *    Unicode characters. This may change in future releases.
844     *    </p>
845     *
846     * @param input  The {@link String} to unescape, may be null
847     * @return A new unescaped {@link String}, {@code null} if null string input
848     * @see #escapeXml(String)
849     * @see #escapeXml10(String)
850     * @see #escapeXml11(String)
851     */
852    public static final String unescapeXml(final String input) {
853        return UNESCAPE_XML.translate(input);
854    }
855
856    /**
857     * {@link StringEscapeUtils} instances should NOT be constructed in
858     * standard programming.
859     *
860     * <p>
861     * Instead, the class should be used as:
862     * </p>
863     * <pre>StringEscapeUtils.escapeJava("foo");</pre>
864     *
865     * <p>
866     * This constructor is public to permit tools that require a JavaBean
867     * instance to operate.
868     * </p>
869     *
870     * @deprecated TODO Make private in 4.0.
871     */
872    @Deprecated
873    public StringEscapeUtils() {
874        // empty
875    }
876
877}