diff --git a/src/core/src/main/kotlin/org/apache/jmeter/save/JMeterStaxDriver.kt b/src/core/src/main/kotlin/org/apache/jmeter/save/JMeterStaxDriver.kt index fef5c9383a8..a4f30a0c2c4 100644 --- a/src/core/src/main/kotlin/org/apache/jmeter/save/JMeterStaxDriver.kt +++ b/src/core/src/main/kotlin/org/apache/jmeter/save/JMeterStaxDriver.kt @@ -17,13 +17,52 @@ package org.apache.jmeter.save +import com.ctc.wstx.api.WstxInputProperties +import com.ctc.wstx.stax.WstxInputFactory +import com.ctc.wstx.stax.WstxOutputFactory import com.thoughtworks.xstream.io.xml.StaxDriver +import java.io.InputStream +import java.io.Reader +import javax.xml.stream.XMLInputFactory import javax.xml.stream.XMLOutputFactory +import javax.xml.stream.XMLStreamReader +import javax.xml.transform.Source +import javax.xml.transform.stream.StreamSource public class JMeterStaxDriver( public val xmlHeader: Boolean = true, public val indent: Boolean = true, ) : StaxDriver() { - override fun createOutputFactory(): XMLOutputFactory = - XMLOutputFactoryDelegate(super.createOutputFactory(), xmlHeader = xmlHeader, indent = indent) + override fun createOutputFactory(): XMLOutputFactory { + // A plugin StAX jar, or a javax.xml.stream.XMLOutputFactory system + // property, must not replace Woodstox. Character-reference handling + // below is Woodstox-specific. + return XMLOutputFactoryDelegate(WstxOutputFactory(), xmlHeader = xmlHeader, indent = indent) + } + + override fun createInputFactory(): XMLInputFactory { + val factory = WstxInputFactory() + factory.setProperty(XMLInputFactory.IS_SUPPORTING_EXTERNAL_ENTITIES, false) + // XML 1.0 documents may still contain the character references JMeter + // 5.6.3 wrote for C0 controls (, , ...). NUL, U+FFFE and + // U+FFFF stay rejected by Woodstox and are rewritten in createParser. + factory.setProperty(WstxInputProperties.P_ALLOW_XML11_ESCAPED_CHARS_IN_XML10, true) + return factory + } + + override fun createParser(reader: Reader): XMLStreamReader { + return XmlCharRefStreamReader(super.createParser(XmlIllegalCharRefReader(reader))) + } + + override fun createParser(input: InputStream): XMLStreamReader { + return XmlCharRefStreamReader(super.createParser(XmlIllegalCharRefInputStream(input))) + } + + override fun createParser(source: Source): XMLStreamReader { + if (source is StreamSource) { + source.reader?.let { return createParser(it) } + source.inputStream?.let { return createParser(it) } + } + return XmlCharRefStreamReader(super.createParser(source)) + } } diff --git a/src/core/src/main/kotlin/org/apache/jmeter/save/XMLOutputFactoryDelegate.kt b/src/core/src/main/kotlin/org/apache/jmeter/save/XMLOutputFactoryDelegate.kt index d2283a47cc3..cc009cebe55 100644 --- a/src/core/src/main/kotlin/org/apache/jmeter/save/XMLOutputFactoryDelegate.kt +++ b/src/core/src/main/kotlin/org/apache/jmeter/save/XMLOutputFactoryDelegate.kt @@ -19,7 +19,10 @@ package org.apache.jmeter.save import com.sun.xml.txw2.output.IndentingXMLStreamWriter import java.io.OutputStream +import java.io.OutputStreamWriter import java.io.Writer +import java.nio.charset.Charset +import java.nio.charset.StandardCharsets import javax.xml.stream.XMLEventWriter import javax.xml.stream.XMLOutputFactory import javax.xml.stream.XMLStreamWriter @@ -46,9 +49,12 @@ public class XMLOutputFactoryDelegate( // Methods that wrap with XMLStreamWriterSkipHeader private fun XMLStreamWriter.applyXmlStreamWriterConfiguration(): XMLStreamWriter { - var result = this + // Shield illegal characters before indentation, so both text and + // attributes are rewritten. The filter sits under Woodstox and emits + // the character reference. + var result: XMLStreamWriter = XmlCharRefStreamWriter(this) if (indent) { - result = IndentingXMLStreamWriter(this) + result = IndentingXMLStreamWriter(result) } if (!xmlHeader) { result = XMLStreamWriterSkipHeader(result) @@ -57,18 +63,27 @@ public class XMLOutputFactoryDelegate( } override fun createXMLStreamWriter(stream: Writer): XMLStreamWriter { - return delegate.createXMLStreamWriter(stream).applyXmlStreamWriterConfiguration() + return delegate.createXMLStreamWriter(XmlCharRefFilterWriter(stream)) + .applyXmlStreamWriterConfiguration() } override fun createXMLStreamWriter(stream: OutputStream): XMLStreamWriter { - return delegate.createXMLStreamWriter(stream).applyXmlStreamWriterConfiguration() + return createXMLStreamWriter(OutputStreamWriter(stream, StandardCharsets.UTF_8)) } override fun createXMLStreamWriter( stream: OutputStream, encoding: String? ): XMLStreamWriter { - return delegate.createXMLStreamWriter(stream, encoding).applyXmlStreamWriterConfiguration() + val charset = charsetOrUtf8(encoding) + return createXMLStreamWriter(OutputStreamWriter(stream, charset)) + } + + private fun charsetOrUtf8(encoding: String?): Charset { + if (encoding.isNullOrEmpty()) { + return StandardCharsets.UTF_8 + } + return Charset.forName(encoding) } override fun createXMLStreamWriter(result: Result): XMLStreamWriter { diff --git a/src/core/src/main/kotlin/org/apache/jmeter/save/XmlCharRef.kt b/src/core/src/main/kotlin/org/apache/jmeter/save/XmlCharRef.kt new file mode 100644 index 00000000000..cd542162c6f --- /dev/null +++ b/src/core/src/main/kotlin/org/apache/jmeter/save/XmlCharRef.kt @@ -0,0 +1,653 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to you under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.jmeter.save + +import java.io.BufferedReader +import java.io.FilterWriter +import java.io.InputStream +import java.io.Reader +import java.io.Writer +import javax.xml.stream.XMLStreamReader +import javax.xml.stream.XMLStreamWriter + +/** + * Characters XML 1.0 does not allow are written as character references + * (`�`, ``, `￾`) and accepted again on read. + * + * Woodstox rejects NUL, U+FFFE and U+FFFF even as character references. + * Those references are rewritten to a private-use pair before the parser + * sees them, then mapped back to the original character. + */ +internal const val XML_CHAR_ESCAPE: Char = '\uE000' + +private const val MARK_BASE: Int = 0xE100 +private const val MARK_FFFE: Char = '\uE1FE' +private const val MARK_FFFF: Char = '\uE1FF' +private const val MAX_CHAR_REF_LENGTH: Int = 16 + +internal fun isIllegalXml10Char(c: Char): Boolean { + if (c == '\t' || c == '\n' || c == '\r') { + return false + } + if (c.code < 0x20) { + return true + } + return c == '\uFFFE' || c == '\uFFFF' +} + +internal fun xmlCharRef(c: Char): String { + return "&#x" + Integer.toHexString(c.code) + ";" +} + +private fun markFor(c: Char): Char { + return when (c) { + '\uFFFE' -> MARK_FFFE + '\uFFFF' -> MARK_FFFF + else -> (MARK_BASE + c.code).toChar() + } +} + +private fun codePointForMark(mark: Char): Int { + return when (mark) { + MARK_FFFE -> 0xFFFE + MARK_FFFF -> 0xFFFF + else -> { + val cp = mark.code - MARK_BASE + if (cp in 0..0x1F) cp else -1 + } + } +} + +internal fun shieldIllegalXmlChars(text: String?): String? { + if (text == null || !needsShield(text)) { + return text + } + val out = StringBuilder(text.length + 8) + for (c in text) { + when { + c == XML_CHAR_ESCAPE -> out.append(XML_CHAR_ESCAPE).append(XML_CHAR_ESCAPE) + isIllegalXml10Char(c) -> out.append(XML_CHAR_ESCAPE).append(markFor(c)) + else -> out.append(c) + } + } + return out.toString() +} + +private fun needsShield(text: String): Boolean { + for (c in text) { + if (c == XML_CHAR_ESCAPE || isIllegalXml10Char(c)) { + return true + } + } + return false +} + +internal fun decodeXmlCharSentinels(text: String?): String? { + if (text == null || text.indexOf(XML_CHAR_ESCAPE) < 0) { + return text + } + val out = StringBuilder(text.length) + var i = 0 + while (i < text.length) { + val c = text[i] + if (c == XML_CHAR_ESCAPE && i + 1 < text.length) { + val mark = text[i + 1] + when { + mark == XML_CHAR_ESCAPE -> out.append(XML_CHAR_ESCAPE) + codePointForMark(mark) >= 0 -> out.append(codePointForMark(mark).toChar()) + else -> { + out.append(c) + i++ + continue + } + } + i += 2 + } else { + out.append(c) + i++ + } + } + return out.toString() +} + +/** + * Replaces illegal characters with a private-use pair so Woodstox will write + * them. [XmlCharRefFilterWriter] turns that pair into a character reference. + */ +internal class XmlCharRefStreamWriter( + private val delegate: XMLStreamWriter +) : XMLStreamWriter by delegate { + + override fun writeCharacters(text: String?) { + delegate.writeCharacters(shieldIllegalXmlChars(text)) + } + + override fun writeCharacters(text: CharArray?, start: Int, len: Int) { + if (text == null) { + delegate.writeCharacters(text, start, len) + return + } + delegate.writeCharacters(shieldIllegalXmlChars(String(text, start, len))) + } + + override fun writeAttribute(localName: String, value: String?) { + delegate.writeAttribute(localName, shieldIllegalXmlChars(value)) + } + + override fun writeAttribute(namespaceURI: String?, localName: String?, value: String?) { + delegate.writeAttribute(namespaceURI, localName, shieldIllegalXmlChars(value)) + } + + override fun writeAttribute( + prefix: String?, + namespaceURI: String?, + localName: String?, + value: String? + ) { + delegate.writeAttribute(prefix, namespaceURI, localName, shieldIllegalXmlChars(value)) + } + + override fun writeCData(data: String?) { + if (data != null && needsShield(data)) { + writeCharacters(data) + } else { + delegate.writeCData(data) + } + } +} + +/** + * Writes the private-use pair from [XmlCharRefStreamWriter] as `&#xN;`. + */ +internal class XmlCharRefFilterWriter(out: Writer) : FilterWriter(out) { + private var pendingEscape = false + + override fun write(cbuf: CharArray, off: Int, len: Int) { + var i = off + val end = off + len + while (i < end) { + if (pendingEscape) { + pendingEscape = false + writeMark(cbuf[i]) + i++ + continue + } + if (cbuf[i] == XML_CHAR_ESCAPE) { + if (i + 1 < end) { + writeMark(cbuf[i + 1]) + i += 2 + } else { + pendingEscape = true + i++ + } + continue + } + var j = i + 1 + while (j < end && cbuf[j] != XML_CHAR_ESCAPE) { + j++ + } + out.write(cbuf, i, j - i) + i = j + } + } + + override fun write(c: Int) { + val ch = c.toChar() + if (pendingEscape) { + pendingEscape = false + writeMark(ch) + return + } + if (ch == XML_CHAR_ESCAPE) { + pendingEscape = true + return + } + out.write(c) + } + + override fun flush() { + super.flush() + } + + override fun close() { + if (pendingEscape) { + pendingEscape = false + out.write(XML_CHAR_ESCAPE.code) + } + super.close() + } + + private fun writeMark(mark: Char) { + if (mark == XML_CHAR_ESCAPE) { + out.write(XML_CHAR_ESCAPE.code) + return + } + val cp = codePointForMark(mark) + if (cp < 0) { + out.write(XML_CHAR_ESCAPE.code) + out.write(mark.code) + return + } + out.write(xmlCharRef(cp.toChar())) + } +} + +/** + * Rewrites character references Woodstox always rejects (`�`, `￾`, + * `￿`) into the private-use pair [decodeXmlCharSentinels] understands. + */ +internal class XmlIllegalCharRefReader( + delegate: Reader +) : Reader() { + private val input = BufferedReader(delegate, 4096) + private val pending = StringBuilder() + private var pendingAt = 0 + private val tail = StringBuilder() + private var mode = Mode.TEXT + private var eof = false + + private enum class Mode { TEXT, COMMENT, CDATA } + + override fun read(cbuf: CharArray, off: Int, len: Int): Int { + if (len <= 0) { + return 0 + } + var written = 0 + while (written < len) { + if (pendingAt >= pending.length) { + pending.setLength(0) + pendingAt = 0 + if (!produce()) { + return if (written == 0) -1 else written + } + } + val n = minOf(len - written, pending.length - pendingAt) + pending.getChars(pendingAt, pendingAt + n, cbuf, off + written) + pendingAt += n + written += n + } + return written + } + + override fun close() { + input.close() + } + + private fun produce(): Boolean { + if (!eof) { + val chunk = CharArray(2048) + val n = input.read(chunk) + if (n < 0) { + eof = true + } else { + tail.append(chunk, 0, n) + } + } + drainTail() + if (pending.isNotEmpty()) { + return true + } + return if (eof) { + false + } else { + produce() + } + } + + private fun drainTail() { + var i = 0 + while (i < tail.length) { + if (mode == Mode.COMMENT) { + val end = tail.indexOf("-->", i) + if (end < 0) { + i = holdSuffix(i, 2) + break + } + pending.append(tail, i, end + 3) + i = end + 3 + mode = Mode.TEXT + continue + } + if (mode == Mode.CDATA) { + val end = tail.indexOf("]]>", i) + if (end < 0) { + i = holdSuffix(i, 2) + break + } + pending.append(tail, i, end + 3) + i = end + 3 + mode = Mode.TEXT + continue + } + val special = indexOfSpecial(i) + if (special < 0) { + appendPlain(i, tail.length) + i = tail.length + break + } + appendPlain(i, special) + val ch = tail[special] + if (ch == '&') { + val semi = tail.indexOf(";", special) + val refLen = if (semi < 0) -1 else semi - special + 1 + if (semi < 0 || refLen > MAX_CHAR_REF_LENGTH) { + if (!eof && semi < 0 && tail.length - special <= MAX_CHAR_REF_LENGTH) { + i = special + break + } + pending.append('&') + i = special + 1 + continue + } + rewriteRef(special, semi) + i = semi + 1 + continue + } + if (tail.startsWith("", i) + if (end < 0) { + i = holdSuffix(i, 2) + break + } + pending.append(tail, i, end + 3) + i = end + 3 + mode = Mode.TEXT + continue + } + if (mode == Mode.CDATA) { + val end = tail.indexOf("]]>", i) + if (end < 0) { + i = holdSuffix(i, 2) + break + } + pending.append(tail, i, end + 3) + i = end + 3 + mode = Mode.TEXT + continue + } + val special = indexOfAsciiSpecial(i) + if (special < 0) { + pending.append(tail, i, tail.length) + i = tail.length + break + } + pending.append(tail, i, special) + val ch = tail[special] + if (ch == '&') { + val semi = tail.indexOf(";", special) + val refLen = if (semi < 0) -1 else semi - special + 1 + if (semi < 0 || refLen > MAX_CHAR_REF_LENGTH) { + if (!eof && semi < 0 && tail.length - special <= MAX_CHAR_REF_LENGTH) { + i = special + break + } + pending.append('&') + i = special + 1 + continue + } + rewriteRef(special, semi) + i = semi + 1 + continue + } + if (tail.startsWith(" diff --git a/xdocs/usermanual/component_reference.xml b/xdocs/usermanual/component_reference.xml index 783051d8bb1..19d4b0b6971 100644 --- a/xdocs/usermanual/component_reference.xml +++ b/xdocs/usermanual/component_reference.xml @@ -2659,9 +2659,7 @@ before loading the file. When reading from CSV results files, the header (if present) is used to determine which fields are present. In order to interpret a header-less CSV file correctly, the appropriate properties must be set in jmeter.properties. -XML files written by JMeter have version 1.0 declared in header while actual file is serialized with 1.1 rules. -(This is done for historical compatibility reasons; see 59973 and 58679) -This causes strict XML parsers to fail. Consider using non-strict XML parsers to read JTL files. +JMeter writes an XML 1.0 declaration. Characters that XML 1.0 does not allow, such as NUL and other C0 controls, are stored as character references (for example &#x0; and &#x1f;) and JMeter reads them back. A recorded binary body therefore survives Save, and files written by JMeter 5.6.3 still open. Strict XML 1.0 parsers reject those references. See 59973 and 58679.