diff --git a/scripts/generate-markdown.mjs b/scripts/generate-markdown.mjs index 445e85850f..35a6047466 100644 --- a/scripts/generate-markdown.mjs +++ b/scripts/generate-markdown.mjs @@ -106,6 +106,16 @@ const extractDemoCode = (demo, doc) => { return fragment; }; +// The converter fences a
only when it holds a element.
+const fenceLiteralBlocks = (article, doc) => {
+ article.querySelectorAll('pre').forEach((pre) => {
+ if (pre.querySelector('code')) return;
+ const code = doc.createElement('code');
+ code.append(...pre.childNodes);
+ pre.appendChild(code);
+ });
+};
+
const rewriteLiveDemos = (article, doc) => {
article.querySelectorAll('.live-demo').forEach((demo) => {
const replacement = extractDemoCode(demo, doc);
@@ -190,6 +200,115 @@ const rewriteCardTables = (article, doc) => {
});
};
+// The converter ends a table row at any line break inside a cell. So merged
+// cells are filled in, a table with block content in its cells is written as
+// one block per row, and any other table keeps each cell on one line, with
+// CELL_BREAK written as
.
+const CELL_BREAK = '\u2063br\u2063';
+
+const unwrap = (el) => el.replaceWith(...el.childNodes);
+
+const hasBlockContent = (cell) => cell.querySelector('pre, ul, ol, dl, table, blockquote') !== null;
+
+const fillMergedCells = (table) => {
+ const carried = [];
+ [ ...table.rows ].forEach((row) => {
+ const cells = [];
+ const takeCarried = () => {
+ while (carried[cells.length]?.rows > 0) {
+ carried[cells.length].rows--;
+ cells.push(carried[cells.length].cell.cloneNode(true));
+ }
+ };
+
+ takeCarried();
+ [ ...row.cells ].forEach((cell) => {
+ const { rowSpan, colSpan } = cell;
+ cell.removeAttribute('rowspan');
+ cell.removeAttribute('colspan');
+ for (let i = 0; i < Math.max(colSpan, 1); i++) {
+ const copy = i ? cell.cloneNode(true) : cell;
+ if (rowSpan > 1) carried[cells.length] = { cell: copy, rows: rowSpan - 1 };
+ cells.push(copy);
+ takeCarried();
+ }
+ });
+ row.replaceChildren(...cells);
+ });
+};
+
+// One paragraph of the row's short cells, each after its column label, then
+// each block cell under its label.
+const tableToRows = (table, doc) => {
+ const labels = [ ...(table.tHead?.rows[0]?.cells ?? []) ].map((th) => th.textContent.trim());
+ const label = (i) => {
+ const strong = doc.createElement('strong');
+ strong.textContent = labels[i] + ':';
+ return labels[i] ? [ strong, ' ' ] : [];
+ };
+
+ const output = [];
+ [ ...table.rows ].filter((row) => row.parentElement !== table.tHead).forEach((row) => {
+ const summary = doc.createElement('p');
+ const blocks = [];
+ [ ...row.cells ].forEach((cell, i) => {
+ if (hasBlockContent(cell)) {
+ const heading = doc.createElement('p');
+ heading.append(...label(i));
+ blocks.push(heading, ...cell.childNodes);
+ } else if (cell.textContent.trim() || cell.querySelector('img')) {
+ if (summary.hasChildNodes()) summary.append(' \u00b7 ');
+ cell.querySelectorAll('p').forEach((p, j) => {
+ if (j) p.before(' ');
+ unwrap(p);
+ });
+ summary.append(...label(i), ...cell.childNodes);
+ }
+ });
+ output.push(summary, ...blocks);
+ });
+ table.replaceWith(...output);
+};
+
+// The converter also puts a line break after every image, so images outside
+// links are written as markdown here (flattenLinkContent handles linked ones),
+// and it writes a header separator only after a row of header cells.
+const flattenCells = (table, doc) => {
+ [ ...table.rows ].flatMap((row) => [ ...row.cells ]).forEach((cell) => {
+ cell.querySelectorAll('br').forEach((br) => br.replaceWith(CELL_BREAK));
+ cell.querySelectorAll('img').forEach((img) => {
+ if (!img.closest('a')) img.replaceWith(`})`);
+ });
+ cell.querySelectorAll('p').forEach((p, i) => {
+ if (i) p.before(CELL_BREAK);
+ unwrap(p);
+ });
+ cell.querySelectorAll('div').forEach(unwrap);
+
+ const walker = doc.createTreeWalker(cell, doc.defaultView.NodeFilter.SHOW_TEXT);
+ for (let node = walker.nextNode(); node; node = walker.nextNode()) {
+ node.textContent = node.textContent.replace(/\s*\n\s*/g, ' ');
+ }
+ });
+
+ if (!table.tHead && table.rows[0]) {
+ [ ...table.rows[0].cells ].forEach((cell) => {
+ const th = doc.createElement('th');
+ th.append(...cell.childNodes);
+ cell.replaceWith(th);
+ });
+ }
+};
+
+const rewriteTables = (article, doc) => {
+ // Innermost first, so a nested table is reshaped before its parent.
+ [ ...article.querySelectorAll('table') ].reverse().forEach((table) => {
+ fillMergedCells(table);
+ const cells = [ ...table.rows ].flatMap((row) => [ ...row.cells ]);
+ cells.some(hasBlockContent) ? tableToRows(table, doc) : flattenCells(table, doc);
+ });
+};
+
// ---------------------------------------------------------------------------
// Preprocessing pipeline
// ---------------------------------------------------------------------------
@@ -198,8 +317,10 @@ const TRANSFORMS = [
stripNonContent,
rewriteAdmonitions,
rewriteLiveDemos,
+ fenceLiteralBlocks,
stripHeadingAnchors,
rewriteCardTables,
+ rewriteTables,
flattenLinkContent,
];
@@ -224,6 +345,34 @@ const D2M_OPTIONS = (dom) => ({
const fixBlankAnchors = (md) =>
md.replace(/about:blank#/g, '#');
+const FENCE = /^\s*(```|~~~)/;
+const LIST_ITEM = /^\s*([-*+]|\d+\.) /;
+
+// Without a blank line after it, a table takes the next paragraph as a row and
+// a list takes it as part of its last item.
+const separateBlocks = (md) => {
+ const lines = [];
+ let inFence = false;
+ let block = null;
+ for (const line of md.split('\n')) {
+ if (inFence) {
+ if (FENCE.test(line)) inFence = false;
+ lines.push(line);
+ continue;
+ }
+
+ const isRow = line.startsWith('|');
+ const isListLine = LIST_ITEM.test(line) || (block === 'list' && /^\s/.test(line));
+ const ends = (block === 'table' && !isRow) || (block === 'list' && !isListLine);
+ if (ends && line.trim()) lines.push('');
+
+ block = !line.trim() ? null : isRow ? 'table' : isListLine ? 'list' : null;
+ if (FENCE.test(line)) inFence = true;
+ lines.push(line);
+ }
+ return lines.join('\n');
+};
+
// Occurrences of " {
const walker = dom.window.document.createTreeWalker(article, dom.window.NodeFilter.SHOW_TEXT);
@@ -239,7 +388,8 @@ const countLiteralLinkText = (article, dom) => {
const toMarkdown = (articleEl, dom) => {
const article = preprocess(articleEl, dom.window.document);
const raw = convertHtmlToMarkdown(article.innerHTML, D2M_OPTIONS(dom));
- return { markdown: fixBlankAnchors(raw), literalLinkText: countLiteralLinkText(article, dom) };
+ const markdown = separateBlocks(fixBlankAnchors(raw).replaceAll(CELL_BREAK, '
'));
+ return { markdown, literalLinkText: countLiteralLinkText(article, dom) };
};
// ---------------------------------------------------------------------------
@@ -316,6 +466,29 @@ const countRawLinks = (markdown, literalLinkText) => {
return Math.max(0, count - literalLinkText);
};
+// Count table rows that do not match their header: a different number of cells
+// (pipes in code spans and escaped pipes do not count), a code fence, or a
+// merged-cell comment.
+const countCells = (row) =>
+ (row.replace(/(`+).*?\1/g, '').match(/(? {
+ let inFence = false;
+ let columns = 0;
+ return markdown.split('\n').filter((line, i, lines) => {
+ if (FENCE.test(line)) inFence = !inFence;
+ if (inFence || !line.startsWith('|')) {
+ columns = 0;
+ return false;
+ }
+ if (!columns && /^\|\s*:?-{3}/.test(lines[i + 1] ?? '')) {
+ columns = countCells(line);
+ return false;
+ }
+ return countCells(line) !== columns || /```|