diff --git a/lang/java/compiler/src/main/java/org/apache/avro/compiler/specific/SpecificCompiler.java b/lang/java/compiler/src/main/java/org/apache/avro/compiler/specific/SpecificCompiler.java index ea1e1a11b55..3d4e4263239 100644 --- a/lang/java/compiler/src/main/java/org/apache/avro/compiler/specific/SpecificCompiler.java +++ b/lang/java/compiler/src/main/java/org/apache/avro/compiler/specific/SpecificCompiler.java @@ -1109,10 +1109,20 @@ public static String javaEscape(String o) { } /** - * Utility for template use. Escapes comment end with HTML entities. + * Utility for template use. Escapes content emitted into a Javadoc comment. + * + *
+ * As well as escaping the comment terminator ({@code *}{@code /}) and HTML
+ * metacharacters, this neutralizes backslashes. This is required because the
+ * Java compiler translates Unicode escapes (of the form {@code \}{@code uXXXX})
+ * across the whole source file, including inside comments, as its first lexical
+ * step (JLS §3.3). Without this, a schema doc value such as
+ * {@code \}{@code u002a\}{@code u002f} would be decoded by the compiler to
+ * {@code *}{@code /}, prematurely closing the comment and allowing arbitrary
+ * code to be injected into the generated source.
*/
public static String escapeForJavadoc(String s) {
- return s.replace("*/", "*/").replace("<", "<").replace(">", ">");
+ return s.replace("\\", "\").replace("*/", "*/").replace("<", "<").replace(">", ">");
}
/**
diff --git a/lang/java/compiler/src/test/java/org/apache/avro/compiler/specific/TestSpecificCompiler.java b/lang/java/compiler/src/test/java/org/apache/avro/compiler/specific/TestSpecificCompiler.java
index 178f2fa2100..62d199376ba 100644
--- a/lang/java/compiler/src/test/java/org/apache/avro/compiler/specific/TestSpecificCompiler.java
+++ b/lang/java/compiler/src/test/java/org/apache/avro/compiler/specific/TestSpecificCompiler.java
@@ -1031,6 +1031,76 @@ void docsAreEscaped_avro4053() {
}
}
+ @Test
+ void unicodeEscapesInDocsAreNeutralized() {
+ // The Java compiler decodes Unicode escapes (\ uXXXX) across the whole source
+ // file, including inside comments, before comments are recognized (JLS 3.3).
+ // A doc value carrying the literal text "\ u002a\ u002f" therefore decodes to
+ // "*/" at compile time and could close the generated Javadoc comment early,
+ // enabling arbitrary code injection. Since a Unicode escape always requires a
+ // literal backslash, escapeForJavadoc neutralizes every backslash, which
+ // covers all escape variants at once.
+ String[] maliciousDocs = { //
+ "\\u002a\\u002f static { System.exit(1); } \\u002f\\u002a", // basic form
+ "\\uuuu002a\\uuuu002f System.exit(1);", // multiple 'u's are legal (JLS 3.3)
+ "\\u005cu002a\\u005cu002f System.exit(1);", // escape that would decode to a backslash
+ "\\U002A\\u002F", // uppercase hex / uppercase-U decoy
+ "prefix\\\\u002a\\\\u002f even-backslash-run", // even run of backslashes
+ "literal */ static { System.exit(1); } /* comment close", // no escape at all
+ "first line\\u002a\\u002f\nsecond line \\u002f\\u002a end" // spans multiple physical lines
+ };
+
+ for (String maliciousDoc : maliciousDocs) {
+ // Unit-level check on the escaping utility itself.
+ String escaped = SpecificCompiler.escapeForJavadoc(maliciousDoc);
+ assertFalse(escaped.contains("\\"), "Backslashes must be neutralized: " + escaped);
+ assertFalse(escaped.contains("*/"), "Comment terminator must be neutralized: " + escaped);
+
+ // End-to-end check: no raw backslash may reach the generated source outside of
+ // string literals. A Java Unicode escape always requires a literal backslash,
+ // so the absence of backslashes everywhere except string literals proves no
+ // \ uXXXX sequence can be reconstituted by the compiler to close a comment.
+ Schema schema = SchemaBuilder.record("EvilRecord").namespace("org.apache.avro.codegentest.testdata")
+ .doc(maliciousDoc).fields().name("field").doc(maliciousDoc).type().stringType().noDefault().endRecord();
+ Collection