001/* 002 * Licensed to the Apache Software Foundation (ASF) under one or more 003 * contributor license agreements. See the NOTICE file distributed with 004 * this work for additional information regarding copyright ownership. 005 * The ASF licenses this file to You under the Apache License, Version 2.0 006 * (the "License"); you may not use this file except in compliance with 007 * the License. You may obtain a copy of the License at 008 * 009 * https://www.apache.org/licenses/LICENSE-2.0 010 * 011 * Unless required by applicable law or agreed to in writing, software 012 * distributed under the License is distributed on an "AS IS" BASIS, 013 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. 014 * See the License for the specific language governing permissions and 015 * limitations under the License. 016 */ 017package org.apache.commons.lang3; 018 019import java.io.IOException; 020import java.io.Writer; 021 022import org.apache.commons.lang3.text.translate.AggregateTranslator; 023import org.apache.commons.lang3.text.translate.CharSequenceTranslator; 024import org.apache.commons.lang3.text.translate.EntityArrays; 025import org.apache.commons.lang3.text.translate.JavaUnicodeEscaper; 026import org.apache.commons.lang3.text.translate.LookupTranslator; 027import org.apache.commons.lang3.text.translate.NumericEntityEscaper; 028import org.apache.commons.lang3.text.translate.NumericEntityUnescaper; 029import org.apache.commons.lang3.text.translate.OctalUnescaper; 030import org.apache.commons.lang3.text.translate.UnicodeUnescaper; 031import org.apache.commons.lang3.text.translate.UnicodeUnpairedSurrogateRemover; 032 033/** 034 * Escapes and unescapes {@link String}s for 035 * Java, Java Script, HTML and XML. 036 * 037 * <p> 038 * #ThreadSafe# 039 * </p> 040 * 041 * @since 2.0 042 * @deprecated As of 3.6, use Apache Commons Text 043 * <a href="https://commons.apache.org/proper/commons-text/javadocs/api-release/org/apache/commons/text/StringEscapeUtils.html"> 044 * StringEscapeUtils</a> instead. 045 */ 046@Deprecated 047public class StringEscapeUtils { 048 049 /* ESCAPE TRANSLATORS */ 050 051 private static final class CsvEscaper extends CharSequenceTranslator { 052 053 private static final char CSV_DELIMITER = ','; 054 private static final char CSV_QUOTE = '"'; 055 private static final String CSV_QUOTE_STR = String.valueOf(CSV_QUOTE); 056 private static final char[] CSV_SEARCH_CHARS = { CSV_DELIMITER, CSV_QUOTE, CharUtils.CR, CharUtils.LF }; 057 058 @Override 059 public int translate(final CharSequence input, final int index, final Writer out) throws IOException { 060 if (index != 0) { 061 throw new IllegalStateException("CsvEscaper should never reach the [1] index"); 062 } 063 if (StringUtils.containsNone(input.toString(), CSV_SEARCH_CHARS)) { 064 out.write(input.toString()); 065 } else { 066 out.write(CSV_QUOTE); 067 out.write(Strings.CS.replace(input.toString(), CSV_QUOTE_STR, CSV_QUOTE_STR + CSV_QUOTE_STR)); 068 out.write(CSV_QUOTE); 069 } 070 return Character.codePointCount(input, 0, input.length()); 071 } 072 } 073 074 private static final class CsvUnescaper extends CharSequenceTranslator { 075 076 private static final char CSV_DELIMITER = ','; 077 private static final char CSV_QUOTE = '"'; 078 private static final String CSV_QUOTE_STR = String.valueOf(CSV_QUOTE); 079 private static final char[] CSV_SEARCH_CHARS = {CSV_DELIMITER, CSV_QUOTE, CharUtils.CR, CharUtils.LF}; 080 081 @Override 082 public int translate(final CharSequence input, final int index, final Writer out) throws IOException { 083 if (index != 0) { 084 throw new IllegalStateException("CsvUnescaper should never reach the [1] index"); 085 } 086 if (input.length() < 2 || input.charAt(0) != CSV_QUOTE || input.charAt(input.length() - 1) != CSV_QUOTE) { 087 out.write(input.toString()); 088 return Character.codePointCount(input, 0, input.length()); 089 } 090 // strip quotes 091 final String quoteless = input.subSequence(1, input.length() - 1).toString(); 092 if (StringUtils.containsAny(quoteless, CSV_SEARCH_CHARS)) { 093 // deal with escaped quotes; ie) "" 094 out.write(Strings.CS.replace(quoteless, CSV_QUOTE_STR + CSV_QUOTE_STR, CSV_QUOTE_STR)); 095 } else { 096 out.write(input.toString()); 097 } 098 return Character.codePointCount(input, 0, input.length()); 099 } 100 } 101 102 /** 103 * Translator object for escaping Java. 104 * 105 * While {@link #escapeJava(String)} is the expected method of use, this 106 * object allows the Java escaping functionality to be used 107 * as the foundation for a custom translator. 108 * 109 * @since 3.0 110 */ 111 public static final CharSequenceTranslator ESCAPE_JAVA = 112 new LookupTranslator( 113 new String[][] { 114 {"\"", "\\\""}, 115 {"\\", "\\\\"}, 116 }).with( 117 new LookupTranslator(EntityArrays.JAVA_CTRL_CHARS_ESCAPE()) 118 ).with( 119 JavaUnicodeEscaper.outsideOf(32, 0x7f) 120 ); 121 122 /** 123 * Translator object for escaping EcmaScript/JavaScript. 124 * 125 * While {@link #escapeEcmaScript(String)} is the expected method of use, this 126 * object allows the EcmaScript escaping functionality to be used 127 * as the foundation for a custom translator. 128 * 129 * @since 3.0 130 */ 131 public static final CharSequenceTranslator ESCAPE_ECMASCRIPT = 132 new AggregateTranslator( 133 new LookupTranslator( 134 new String[][] { 135 {"'", "\\'"}, 136 {"\"", "\\\""}, 137 {"`", "\\`"}, 138 {"${", "\\${"}, 139 {"\\", "\\\\"}, 140 {"/", "\\/"} 141 }), 142 new LookupTranslator(EntityArrays.JAVA_CTRL_CHARS_ESCAPE()), 143 JavaUnicodeEscaper.outsideOf(32, 0x7f) 144 ); 145 146 /** 147 * Translator object for escaping Json. 148 * 149 * While {@link #escapeJson(String)} is the expected method of use, this 150 * object allows the Json escaping functionality to be used 151 * as the foundation for a custom translator. 152 * 153 * @since 3.2 154 */ 155 public static final CharSequenceTranslator ESCAPE_JSON = 156 new AggregateTranslator( 157 new LookupTranslator( 158 new String[][] { 159 {"\"", "\\\""}, 160 {"\\", "\\\\"}, 161 {"/", "\\/"} 162 }), 163 new LookupTranslator(EntityArrays.JAVA_CTRL_CHARS_ESCAPE()), 164 JavaUnicodeEscaper.outsideOf(32, 0x7f) 165 ); 166 167 /** 168 * Translator object for escaping XML. 169 * 170 * While {@link #escapeXml(String)} is the expected method of use, this 171 * object allows the XML escaping functionality to be used 172 * as the foundation for a custom translator. 173 * 174 * @since 3.0 175 * @deprecated Use {@link #ESCAPE_XML10} or {@link #ESCAPE_XML11} instead. 176 */ 177 @Deprecated 178 public static final CharSequenceTranslator ESCAPE_XML = 179 new AggregateTranslator( 180 new LookupTranslator(EntityArrays.APOS_ESCAPE()), 181 new LookupTranslator(EntityArrays.BASIC_ESCAPE()) 182 ); 183 184 /** 185 * Translator object for escaping XML 1.0. 186 * 187 * While {@link #escapeXml10(String)} is the expected method of use, this 188 * object allows the XML escaping functionality to be used 189 * as the foundation for a custom translator. 190 * 191 * @since 3.3 192 */ 193 public static final CharSequenceTranslator ESCAPE_XML10 = 194 new AggregateTranslator( 195 new LookupTranslator(EntityArrays.APOS_ESCAPE()), 196 new LookupTranslator(EntityArrays.BASIC_ESCAPE()), 197 new LookupTranslator( 198 new String[][] { 199 { "\u0000", StringUtils.EMPTY }, 200 { "\u0001", StringUtils.EMPTY }, 201 { "\u0002", StringUtils.EMPTY }, 202 { "\u0003", StringUtils.EMPTY }, 203 { "\u0004", StringUtils.EMPTY }, 204 { "\u0005", StringUtils.EMPTY }, 205 { "\u0006", StringUtils.EMPTY }, 206 { "\u0007", StringUtils.EMPTY }, 207 { "\u0008", StringUtils.EMPTY }, 208 { "\u000b", StringUtils.EMPTY }, 209 { "\u000c", StringUtils.EMPTY }, 210 { "\u000e", StringUtils.EMPTY }, 211 { "\u000f", StringUtils.EMPTY }, 212 { "\u0010", StringUtils.EMPTY }, 213 { "\u0011", StringUtils.EMPTY }, 214 { "\u0012", StringUtils.EMPTY }, 215 { "\u0013", StringUtils.EMPTY }, 216 { "\u0014", StringUtils.EMPTY }, 217 { "\u0015", StringUtils.EMPTY }, 218 { "\u0016", StringUtils.EMPTY }, 219 { "\u0017", StringUtils.EMPTY }, 220 { "\u0018", StringUtils.EMPTY }, 221 { "\u0019", StringUtils.EMPTY }, 222 { "\u001a", StringUtils.EMPTY }, 223 { "\u001b", StringUtils.EMPTY }, 224 { "\u001c", StringUtils.EMPTY }, 225 { "\u001d", StringUtils.EMPTY }, 226 { "\u001e", StringUtils.EMPTY }, 227 { "\u001f", StringUtils.EMPTY }, 228 { "\ufffe", StringUtils.EMPTY }, 229 { "\uffff", StringUtils.EMPTY } 230 }), 231 NumericEntityEscaper.between(0x7f, 0x84), 232 NumericEntityEscaper.between(0x86, 0x9f), 233 new UnicodeUnpairedSurrogateRemover() 234 ); 235 236 /** 237 * Translator object for escaping XML 1.1. 238 * 239 * While {@link #escapeXml11(String)} is the expected method of use, this 240 * object allows the XML escaping functionality to be used 241 * as the foundation for a custom translator. 242 * 243 * @since 3.3 244 */ 245 public static final CharSequenceTranslator ESCAPE_XML11 = 246 new AggregateTranslator( 247 new LookupTranslator(EntityArrays.APOS_ESCAPE()), 248 new LookupTranslator(EntityArrays.BASIC_ESCAPE()), 249 new LookupTranslator( 250 new String[][] { 251 { "\u0000", StringUtils.EMPTY }, 252 { "\u000b", "" }, 253 { "\u000c", "" }, 254 { "\ufffe", StringUtils.EMPTY }, 255 { "\uffff", StringUtils.EMPTY } 256 }), 257 NumericEntityEscaper.between(0x1, 0x8), 258 NumericEntityEscaper.between(0xe, 0x1f), 259 NumericEntityEscaper.between(0x7f, 0x84), 260 NumericEntityEscaper.between(0x86, 0x9f), 261 new UnicodeUnpairedSurrogateRemover() 262 ); 263 264 /** 265 * Translator object for escaping HTML version 3.0. 266 * 267 * While {@link #escapeHtml3(String)} is the expected method of use, this 268 * object allows the HTML escaping functionality to be used 269 * as the foundation for a custom translator. 270 * 271 * @since 3.0 272 */ 273 public static final CharSequenceTranslator ESCAPE_HTML3 = 274 new AggregateTranslator( 275 new LookupTranslator(EntityArrays.BASIC_ESCAPE()), 276 new LookupTranslator(EntityArrays.ISO8859_1_ESCAPE()) 277 ); 278 279 /** 280 * Translator object for escaping HTML version 4.0. 281 * 282 * While {@link #escapeHtml4(String)} is the expected method of use, this 283 * object allows the HTML escaping functionality to be used 284 * as the foundation for a custom translator. 285 * 286 * @since 3.0 287 */ 288 public static final CharSequenceTranslator ESCAPE_HTML4 = 289 new AggregateTranslator( 290 new LookupTranslator(EntityArrays.BASIC_ESCAPE()), 291 new LookupTranslator(EntityArrays.ISO8859_1_ESCAPE()), 292 new LookupTranslator(EntityArrays.HTML40_EXTENDED_ESCAPE()) 293 ); 294 295 /* UNESCAPE TRANSLATORS */ 296 297 /** 298 * Translator object for escaping individual Comma Separated Values. 299 * 300 * While {@link #escapeCsv(String)} is the expected method of use, this 301 * object allows the CSV escaping functionality to be used 302 * as the foundation for a custom translator. 303 * 304 * @since 3.0 305 */ 306 public static final CharSequenceTranslator ESCAPE_CSV = new CsvEscaper(); 307 308 /** 309 * Translator object for unescaping escaped Java. 310 * 311 * While {@link #unescapeJava(String)} is the expected method of use, this 312 * object allows the Java unescaping functionality to be used 313 * as the foundation for a custom translator. 314 * 315 * @since 3.0 316 */ 317 // TODO: throw "illegal character: \92" as an Exception if a \ on the end of the Java (as per the compiler)? 318 public static final CharSequenceTranslator UNESCAPE_JAVA = 319 new AggregateTranslator( 320 new OctalUnescaper(), // .between('\1', '\377'), 321 new UnicodeUnescaper(), 322 new LookupTranslator(EntityArrays.JAVA_CTRL_CHARS_UNESCAPE()), 323 new LookupTranslator( 324 new String[][] { 325 {"\\\\", "\\"}, 326 {"\\\"", "\""}, 327 {"\\'", "'"}, 328 {"\\", ""} 329 }) 330 ); 331 332 /** 333 * Translator object for unescaping escaped EcmaScript. 334 * 335 * While {@link #unescapeEcmaScript(String)} is the expected method of use, this 336 * object allows the EcmaScript unescaping functionality to be used 337 * as the foundation for a custom translator. 338 * 339 * @since 3.0 340 */ 341 public static final CharSequenceTranslator UNESCAPE_ECMASCRIPT = UNESCAPE_JAVA; 342 343 /** 344 * Translator object for unescaping escaped Json. 345 * 346 * While {@link #unescapeJson(String)} is the expected method of use, this 347 * object allows the Json unescaping functionality to be used 348 * as the foundation for a custom translator. 349 * 350 * @since 3.2 351 */ 352 public static final CharSequenceTranslator UNESCAPE_JSON = UNESCAPE_JAVA; 353 354 /** 355 * Translator object for unescaping escaped HTML 3.0. 356 * 357 * While {@link #unescapeHtml3(String)} is the expected method of use, this 358 * object allows the HTML unescaping functionality to be used 359 * as the foundation for a custom translator. 360 * 361 * @since 3.0 362 */ 363 public static final CharSequenceTranslator UNESCAPE_HTML3 = 364 new AggregateTranslator( 365 new LookupTranslator(EntityArrays.BASIC_UNESCAPE()), 366 new LookupTranslator(EntityArrays.ISO8859_1_UNESCAPE()), 367 new NumericEntityUnescaper() 368 ); 369 370 /** 371 * Translator object for unescaping escaped HTML 4.0. 372 * 373 * While {@link #unescapeHtml4(String)} is the expected method of use, this 374 * object allows the HTML unescaping functionality to be used 375 * as the foundation for a custom translator. 376 * 377 * @since 3.0 378 */ 379 public static final CharSequenceTranslator UNESCAPE_HTML4 = 380 new AggregateTranslator( 381 new LookupTranslator(EntityArrays.BASIC_UNESCAPE()), 382 new LookupTranslator(EntityArrays.ISO8859_1_UNESCAPE()), 383 new LookupTranslator(EntityArrays.HTML40_EXTENDED_UNESCAPE()), 384 new NumericEntityUnescaper() 385 ); 386 387 /** 388 * Translator object for unescaping escaped XML. 389 * 390 * While {@link #unescapeXml(String)} is the expected method of use, this 391 * object allows the XML unescaping functionality to be used 392 * as the foundation for a custom translator. 393 * 394 * @since 3.0 395 */ 396 public static final CharSequenceTranslator UNESCAPE_XML = 397 new AggregateTranslator( 398 new LookupTranslator(EntityArrays.BASIC_UNESCAPE()), 399 new LookupTranslator(EntityArrays.APOS_UNESCAPE()), 400 new NumericEntityUnescaper() 401 ); 402 403 /** 404 * Translator object for unescaping escaped Comma Separated Value entries. 405 * 406 * While {@link #unescapeCsv(String)} is the expected method of use, this 407 * object allows the CSV unescaping functionality to be used 408 * as the foundation for a custom translator. 409 * 410 * @since 3.0 411 */ 412 public static final CharSequenceTranslator UNESCAPE_CSV = new CsvUnescaper(); 413 414 /* Helper functions */ 415 416 /** 417 * Returns a {@link String} value for a CSV column enclosed in double quotes, if required. 418 * <p> 419 * If the value contains a comma, newline or double quote, then the String value is returned enclosed in double quotes. 420 * </p> 421 * <p> 422 * Any double quote characters in the value are escaped with another double quote. 423 * </p> 424 * <p> 425 * If the value does not contain a comma, newline or double quote, then the String value is returned unchanged. 426 * </p> 427 * <p> 428 * See <a href="https://en.wikipedia.org/wiki/Comma-separated_values">Wikipedia</a> and <a href="https://datatracker.ietf.org/doc/html/rfc4180">RFC 429 * 4180</a>. 430 * </p> 431 * 432 * @param input The input CSV column String, may be null 433 * @return The input String, enclosed in double quotes if the value contains a comma, newline or double quote, {@code null} if null string input 434 * @since 2.4 435 */ 436 public static final String escapeCsv(final String input) { 437 return ESCAPE_CSV.translate(input); 438 } 439 440 /** 441 * Escapes the characters in a {@link String} using EcmaScript String rules. 442 * <p> 443 * Escapes any values it finds into their EcmaScript String form. Escapes the EcmaScript string delimiters (single quote, double quote and (since ES6) the 444 * backtick) as well as the template-literal interpolation sequence <code>${</code> and control-chars (tab, backslash, cr, ff, etc.). 445 * </p> 446 * <p> 447 * So a tab becomes the characters {@code '\\'} and {@code 't'}. 448 * </p> 449 * <p> 450 * The differences between Java strings and EcmaScript strings handled here are that in EcmaScript, a single quote, the backtick, the <code>${</code> 451 * sequence and forward-slash (/) are escaped. 452 * </p> 453 * <p> 454 * <strong>Scope:</strong> the output is a correctly escaped EcmaScript string literal for any of the three string delimiters, but it is <em>not</em> made 455 * safe for direct embedding inside an HTML inline {@code <script>} block: HTML parser-state sequences such as {@code <!--} and {@code <script} pass through 456 * unchanged (a literal {@code </script>} is neutralized by the forward-slash escape). HTML-context encoding must be applied separately when the result is 457 * placed in HTML. 458 * </p> 459 * <p> 460 * Note that EcmaScript is best known by the JavaScript and ActionScript dialects. 461 * </p> 462 * <p> 463 * Example: 464 * </p> 465 * 466 * <pre> 467 * input string: He didn't say, "Stop!" 468 * output string: He didn\'t say, \"Stop!\" 469 * </pre> 470 * 471 * @param input String to escape values in, may be null 472 * @return String with escaped values, {@code null} if null string input 473 * @since 3.0 474 */ 475 public static final String escapeEcmaScript(final String input) { 476 return ESCAPE_ECMASCRIPT.translate(input); 477 } 478 479 /** 480 * Escapes the characters in a {@link String} using HTML entities. 481 * 482 * <p> 483 * Supports only the HTML 3.0 entities. Apostrophes are escaped as the numeric reference {@code '}. 484 * </p> 485 * <p> 486 * HTML attribute values must be quoted. This method does not escape all characters that delimit unquoted attribute values. 487 * </p> 488 * 489 * @param input The {@link String} to escape, may be null 490 * @return A new escaped {@link String}, {@code null} if null string input 491 * @since 3.0 492 */ 493 public static final String escapeHtml3(final String input) { 494 return ESCAPE_HTML3.translate(input); 495 } 496 497 /** 498 * Escapes the characters in a {@link String} using HTML entities. 499 * 500 * <p> 501 * For example: 502 * </p> 503 * <p> 504 * {@code "bread" & "butter"} 505 * </p> 506 * becomes: 507 * <p> 508 * {@code &quot;bread&quot; &amp; &quot;butter&quot;}. 509 * </p> 510 * <p> 511 * Supports all known HTML 4.0 entities, including funky accents. 512 * Apostrophes are escaped as the numeric reference {@code '} because {@code '} is not a legal HTML 4.0 entity. 513 * </p> 514 * <p> 515 * HTML attribute values must be quoted. This method does not escape all characters that delimit unquoted attribute values. 516 * </p> 517 * 518 * @param input The {@link String} to escape, may be null 519 * @return A new escaped {@link String}, {@code null} if null string input 520 * @see <a href="https://web.archive.org/web/20060225074150/https://hotwired.lycos.com/webmonkey/reference/special_characters/">ISO Entities</a> 521 * @see <a href="https://www.w3.org/TR/REC-html32#latin1">HTML 3.2 Character Entities for ISO Latin-1</a> 522 * @see <a href="https://www.w3.org/TR/REC-html40/sgml/entities.html">HTML 4.0 Character entity references</a> 523 * @see <a href="https://www.w3.org/TR/html401/charset.html#h-5.3">HTML 4.01 Character References</a> 524 * @see <a href="https://www.w3.org/TR/html401/charset.html#code-position">HTML 4.01 Code positions</a> 525 * @since 3.0 526 */ 527 public static final String escapeHtml4(final String input) { 528 return ESCAPE_HTML4.translate(input); 529 } 530 531 /** 532 * Escapes the characters in a {@link String} using Java String rules. 533 * 534 * <p> 535 * Deals correctly with quotes and control-chars (tab, backslash, cr, ff, etc.) 536 * </p> 537 * 538 * <p> 539 * So a tab becomes the characters {@code '\\'} and 540 * {@code 't'}. 541 * </p> 542 * 543 * <p> 544 * The only difference between Java strings and JavaScript strings 545 * is that in JavaScript, a single quote and forward-slash (/) are escaped. 546 * </p> 547 * 548 * <p> 549 * Example: 550 * </p> 551 * <pre> 552 * input string: He didn't say, "Stop!" 553 * output string: He didn't say, \"Stop!\" 554 * </pre> 555 * 556 * @param input String to escape values in, may be null 557 * @return String with escaped values, {@code null} if null string input 558 */ 559 public static final String escapeJava(final String input) { 560 return ESCAPE_JAVA.translate(input); 561 } 562 563 /** 564 * Escapes the characters in a {@link String} using Json String rules. 565 * <p> 566 * Escapes any values it finds into their JSON String form. Deals correctly with quotes and control-chars (tab, backslash, cr, ff, etc.) 567 * </p> 568 * <p> 569 * So a tab becomes the characters {@code '\\'} and {@code 't'}. 570 * </p> 571 * <p> 572 * The only difference between Java strings and Json strings is that in Json, forward-slash (/) is escaped. 573 * </p> 574 * <p> 575 * <strong>Scope:</strong> the output is a correctly escaped JSON string, but it is <em>not</em> made safe for direct embedding inside an HTML inline 576 * {@code <script>} block: the backtick, <code>${</code>, and HTML parser-state sequences such as {@code <!--} and {@code <script} pass through unchanged 577 * (JSON offers no backslash escape for them; only a literal {@code </script>} is neutralized by the forward-slash escape). HTML-context encoding must be 578 * applied separately when the result is placed in HTML. 579 * </p> 580 * <p> 581 * See https://www.ietf.org/rfc/rfc4627.txt for further details. 582 * </p> 583 * <p> 584 * Example: 585 * </p> 586 * 587 * <pre> 588 * input string: He didn't say, "Stop!" 589 * output string: He didn't say, \"Stop!\" 590 * </pre> 591 * 592 * @param input String to escape values in, may be null 593 * @return String with escaped values, {@code null} if null string input 594 * @since 3.2 595 */ 596 public static final String escapeJson(final String input) { 597 return ESCAPE_JSON.translate(input); 598 } 599 600 /** 601 * Escapes the characters in a {@link String} using XML entities. 602 * 603 * <p> 604 * For example: {@code "bread" & "butter"} => 605 * {@code "bread" & "butter"}. 606 * </p> 607 * 608 * <p> 609 * Supports only the five basic XML entities (gt, lt, quot, amp, apos). 610 * Does not support DTDs or external entities. 611 * </p> 612 * 613 * <p> 614 * Note that Unicode characters greater than 0x7f are as of 3.0, no longer 615 * escaped. If you still wish this functionality, you can achieve it 616 * via the following: 617 * {@code StringEscapeUtils.ESCAPE_XML.with( NumericEntityEscaper.between(0x7f, Integer.MAX_VALUE));} 618 * </p> 619 * 620 * @param input The {@link String} to escape, may be null 621 * @return A new escaped {@link String}, {@code null} if null string input 622 * @see #unescapeXml(String) 623 * @deprecated Use {@link #escapeXml10(java.lang.String)} or {@link #escapeXml11(java.lang.String)} instead. 624 */ 625 @Deprecated 626 public static final String escapeXml(final String input) { 627 return ESCAPE_XML.translate(input); 628 } 629 630 /** 631 * Escapes the characters in a {@link String} using XML entities. 632 * <p> 633 * For example: 634 * </p> 635 * 636 * <pre>{@code 637 * "bread" & "butter" 638 * }</pre> 639 * <p> 640 * converts to: 641 * </p> 642 * 643 * <pre> 644 * {@code 645 * "bread" & "butter" 646 * } 647 * </pre> 648 * 649 * <p> 650 * Note that XML 1.0 is a text-only format: it cannot represent control characters or unpaired Unicode surrogate code points, even after escaping. The 651 * method {@code escapeXml10} will remove characters that do not fit in the following ranges: 652 * </p> 653 * 654 * <p> 655 * {@code #x9 | #xA | #xD | [#x20-#xD7FF] | [#xE000-#xFFFD] | [#x10000-#x10FFFF]} 656 * </p> 657 * 658 * <p> 659 * Though not strictly necessary, {@code escapeXml10} will escape characters in the following ranges: 660 * </p> 661 * 662 * <p> 663 * {@code [#x7F-#x84] | [#x86-#x9F]} 664 * </p> 665 * 666 * <p> 667 * The returned string can be inserted into a valid XML 1.0 or XML 1.1 document. If you want to allow more non-text characters in an XML 1.1 document, use 668 * {@link #escapeXml11(String)}. 669 * </p> 670 * 671 * @param input The {@link String} to escape, may be null 672 * @return A new escaped {@link String}, {@code null} if null string input 673 * @see #unescapeXml(String) 674 * @since 3.3 675 */ 676 public static String escapeXml10(final String input) { 677 return ESCAPE_XML10.translate(input); 678 } 679 680 /** 681 * Escapes the characters in a {@link String} using XML entities. 682 * 683 * <p> 684 * For example: {@code "bread" & "butter"} => 685 * {@code "bread" & "butter"}. 686 * </p> 687 * 688 * <p> 689 * XML 1.1 can represent certain control characters, but it cannot represent 690 * the null byte or unpaired Unicode surrogate code points, even after escaping. 691 * {@code escapeXml11} will remove characters that do not fit in the following 692 * ranges: 693 * </p> 694 * 695 * <p> 696 * {@code [#x1-#xD7FF] | [#xE000-#xFFFD] | [#x10000-#x10FFFF]} 697 * </p> 698 * 699 * <p> 700 * {@code escapeXml11} will escape characters in the following ranges: 701 * </p> 702 * 703 * <p> 704 * {@code [#x1-#x8] | [#xB-#xC] | [#xE-#x1F] | [#x7F-#x84] | [#x86-#x9F]} 705 * </p> 706 * 707 * <p> 708 * The returned string can be inserted into a valid XML 1.1 document. Do not 709 * use it for XML 1.0 documents. 710 * </p> 711 * 712 * @param input The {@link String} to escape, may be null 713 * @return A new escaped {@link String}, {@code null} if null string input 714 * @see #unescapeXml(String) 715 * @since 3.3 716 */ 717 public static String escapeXml11(final String input) { 718 return ESCAPE_XML11.translate(input); 719 } 720 721 /** 722 * Returns a {@link String} value for an unescaped CSV column. 723 * <p> 724 * If the value is enclosed in double quotes, and contains a comma, newline or double quote, then quotes are removed. 725 * </p> 726 * <p> 727 * Any double quote escaped characters (a pair of double quotes) are unescaped to just one double quote. 728 * </p> 729 * <p> 730 * If the value is not enclosed in double quotes, or is and does not contain a comma, newline or double quote, then the String value is returned unchanged. 731 * </p> 732 * <p> 733 * See <a href="https://en.wikipedia.org/wiki/Comma-separated_values">Wikipedia</a> and <a href="https://datatracker.ietf.org/doc/html/rfc4180">RFC 734 * 4180</a>. 735 * </p> 736 * 737 * @param input The input CSV column String, may be null 738 * @return The input String, with enclosing double quotes removed and embedded double quotes unescaped, {@code null} if null string input 739 * @since 2.4 740 */ 741 public static final String unescapeCsv(final String input) { 742 return UNESCAPE_CSV.translate(input); 743 } 744 745 /** 746 * Unescapes any EcmaScript literals found in the {@link String}. 747 * 748 * <p> 749 * For example, it will turn a sequence of {@code '\'} and {@code 'n'} 750 * into a newline character, unless the {@code '\'} is preceded by another 751 * {@code '\'}. 752 * </p> 753 * 754 * @see #unescapeJava(String) 755 * @param input The {@link String} to unescape, may be null 756 * @return A new unescaped {@link String}, {@code null} if null string input 757 * @since 3.0 758 */ 759 public static final String unescapeEcmaScript(final String input) { 760 return UNESCAPE_ECMASCRIPT.translate(input); 761 } 762 763 /** 764 * Unescapes a string containing entity escapes to a string 765 * containing the actual Unicode characters corresponding to the 766 * escapes. Supports only HTML 3.0 entities. 767 * 768 * @param input The {@link String} to unescape, may be null 769 * @return A new unescaped {@link String}, {@code null} if null string input 770 * @since 3.0 771 */ 772 public static final String unescapeHtml3(final String input) { 773 return UNESCAPE_HTML3.translate(input); 774 } 775 776 /** 777 * Unescapes a string containing entity escapes to a string 778 * containing the actual Unicode characters corresponding to the 779 * escapes. Supports HTML 4.0 entities. 780 * 781 * <p> 782 * For example, the string {@code "<Français>"} 783 * will become {@code "<Français>"} 784 * </p> 785 * 786 * <p> 787 * If an entity is unrecognized, it is left alone, and inserted 788 * verbatim into the result string. e.g. {@code ">&zzzz;x"} will 789 * become {@code ">&zzzz;x"}. 790 * </p> 791 * 792 * @param input The {@link String} to unescape, may be null 793 * @return A new unescaped {@link String}, {@code null} if null string input 794 * @since 3.0 795 */ 796 public static final String unescapeHtml4(final String input) { 797 return UNESCAPE_HTML4.translate(input); 798 } 799 800 /** 801 * Unescapes any Java literals found in the {@link String}. 802 * For example, it will turn a sequence of {@code '\'} and 803 * {@code 'n'} into a newline character, unless the {@code '\'} 804 * is preceded by another {@code '\'}. 805 * 806 * @param input The {@link String} to unescape, may be null 807 * @return A new unescaped {@link String}, {@code null} if null string input 808 */ 809 public static final String unescapeJava(final String input) { 810 return UNESCAPE_JAVA.translate(input); 811 } 812 813 /** 814 * Unescapes any Json literals found in the {@link String}. 815 * 816 * <p> 817 * For example, it will turn a sequence of {@code '\'} and {@code 'n'} 818 * into a newline character, unless the {@code '\'} is preceded by another 819 * {@code '\'}. 820 * </p> 821 * 822 * @see #unescapeJava(String) 823 * @param input The {@link String} to unescape, may be null 824 * @return A new unescaped {@link String}, {@code null} if null string input 825 * @since 3.2 826 */ 827 public static final String unescapeJson(final String input) { 828 return UNESCAPE_JSON.translate(input); 829 } 830 831 /** 832 * Unescapes a string containing XML entity escapes to a string 833 * containing the actual Unicode characters corresponding to the 834 * escapes. 835 * 836 * <p> 837 * Supports only the five basic XML entities (gt, lt, quot, amp, apos). 838 * Does not support DTDs or external entities. 839 * </p> 840 * 841 * <p> 842 * Note that numerical \\u Unicode codes are unescaped to their respective 843 * Unicode characters. This may change in future releases. 844 * </p> 845 * 846 * @param input The {@link String} to unescape, may be null 847 * @return A new unescaped {@link String}, {@code null} if null string input 848 * @see #escapeXml(String) 849 * @see #escapeXml10(String) 850 * @see #escapeXml11(String) 851 */ 852 public static final String unescapeXml(final String input) { 853 return UNESCAPE_XML.translate(input); 854 } 855 856 /** 857 * {@link StringEscapeUtils} instances should NOT be constructed in 858 * standard programming. 859 * 860 * <p> 861 * Instead, the class should be used as: 862 * </p> 863 * <pre>StringEscapeUtils.escapeJava("foo");</pre> 864 * 865 * <p> 866 * This constructor is public to permit tools that require a JavaBean 867 * instance to operate. 868 * </p> 869 * 870 * @deprecated TODO Make private in 4.0. 871 */ 872 @Deprecated 873 public StringEscapeUtils() { 874 // empty 875 } 876 877}