diff --git a/lib/blocks.js b/lib/blocks.js index 66113617..94dd7c66 100644 --- a/lib/blocks.js +++ b/lib/blocks.js @@ -70,6 +70,83 @@ var peek = function(ln, pos) { } }; +// True when `s` ends inside an unclosed HTML tag, comment, PI, +// declaration, or CDATA section. Used so paragraph-indent stripping +// does not eat significant whitespace inside a raw HTML attribute. +var isInsideUnclosedHtmlTag = function(s) { + var i = 0; + var n = s.length; + while (i < n) { + var lt = s.indexOf("<", i); + if (lt < 0) { + return false; + } + if (s.slice(lt, lt + 4) === "", lt + 4); + if (endComment < 0) { + return true; + } + i = endComment + 3; + continue; + } + if (s.slice(lt, lt + 2) === "", lt + 2); + if (endPi < 0) { + return true; + } + i = endPi + 2; + continue; + } + if (s.slice(lt, lt + 9) === "", lt + 9); + if (endCdata < 0) { + return true; + } + i = endCdata + 3; + continue; + } + if (s.slice(lt, lt + 2) === "", lt + 2); + if (endDecl < 0) { + return true; + } + i = endDecl + 1; + continue; + } + var namePos = lt + 1; + if (s.charAt(namePos) === "/") { + namePos += 1; + } + if (!/[A-Za-z]/.test(s.charAt(namePos))) { + i = lt + 1; + continue; + } + var j = namePos + 1; + while (j < n && /[A-Za-z0-9-]/.test(s.charAt(j))) { + j += 1; + } + var quote = null; + while (j < n) { + var c = s.charAt(j); + if (quote !== null) { + if (c === quote) { + quote = null; + } + } else if (c === '"' || c === "'") { + quote = c; + } else if (c === ">") { + break; + } + j += 1; + } + if (j >= n) { + return true; + } + i = j + 1; + } + return false; +}; + // DOC PARSER // These are methods of a Parser object, defined below. @@ -83,13 +160,25 @@ var endsWithBlankLine = function(block) { // Add a line to the block at the tip. We assume the tip // can accept lines -- that check should be done before calling this. var addLine = function() { - if (this.partiallyConsumedTab) { - this.offset += 1; // skip over tab + var offset = this.offset; + var partiallyConsumedTab = this.partiallyConsumedTab; + // Keep indent that was skipped before sending a paragraph to the + // inline parser when that indent sits inside unclosed raw HTML. + // Otherwise `` loses the attribute spaces. + if ( + this.tip.type === "paragraph" && + isInsideUnclosedHtmlTag(this.tip._string_content || "") + ) { + offset = this.lineContentOffset; + partiallyConsumedTab = false; + } + if (partiallyConsumedTab) { + offset += 1; // skip over tab // add space characters: var charsToTab = 4 - (this.column % 4); this.tip._string_content += " ".repeat(charsToTab); } - this.tip._string_content += this.currentLine.slice(this.offset) + "\n"; + this.tip._string_content += this.currentLine.slice(offset) + "\n"; }; // Add block of type tag as a child of the tip. If the tip can't @@ -812,6 +901,8 @@ var incorporateLine = function(ln) { this.allClosed = container === this.oldtip; this.lastMatchedContainer = container; + // Remaining line text, including indent skipped by advanceNextNonspace. + this.lineContentOffset = this.offset; var matchedLeaf = container.type !== "paragraph" && blocks[container.type].acceptsLines; @@ -988,6 +1079,7 @@ function Parser(options) { lineNumber: 0, offset: 0, column: 0, + lineContentOffset: 0, nextNonspace: 0, nextNonspaceColumn: 0, indent: 0, diff --git a/test/regression.txt b/test/regression.txt index 1703e3e3..11a5ed6e 100644 --- a/test/regression.txt +++ b/test/regression.txt @@ -625,6 +625,21 @@ A dotless i reference does not resolve against an ASCII I definition .

[ı]

```````````````````````````````` +Inline HTML attribute whitespace is not stripped as paragraph indent (#303). + +```````````````````````````````` example +
text + +text +. +
text +

text

+```````````````````````````````` + Entities inside autolinks (#263). ```````````````````````````````` example