Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
117 changes: 81 additions & 36 deletions lib/html-to-markdown.js
Original file line number Diff line number Diff line change
@@ -1,5 +1,8 @@
const {
LIST_INDENT,
HARD_BREAK,
collapseInline,
resolveHardBreaks,
escapeSentinels,
finalizeSentinels,
fenceLength,
Expand Down Expand Up @@ -33,16 +36,77 @@ const NAMED_ENTITIES = {
// Same value as StorageWalker's DEFAULT_MAX_DEPTH.
const MAX_LIST_DEPTH = 256;

// Flatten the inline content of a list item onto one line, matching the
// historical regex converter: paragraphs and any remaining tags become
// spaces and whitespace runs collapse.
// Decode the entities the converter understands. Run last for the document,
// and early on a continuation line whose first characters decide whether it
// needs escaping (see resolveHardBreaks).
function decodeEntities(input) {
let text = input;
text = text.replace(/ /g, ' ');
text = text.replace(/&lt;/g, '<');
text = text.replace(/&gt;/g, '>');
text = text.replace(/&amp;/g, '&');
text = text.replace(/&quot;/g, '"');
text = text.replace(/&apos;/g, '\'');
text = text.replace(/&ldquo;/g, '"');
text = text.replace(/&rdquo;/g, '"');
text = text.replace(/&lsquo;/g, '\'');
text = text.replace(/&rsquo;/g, '\'');
text = text.replace(/&mdash;/g, '—');
text = text.replace(/&ndash;/g, '–');
text = text.replace(/&hellip;/g, '...');
text = text.replace(/&bull;/g, '•');
text = text.replace(/&copy;/g, '©');
text = text.replace(/&reg;/g, '®');
text = text.replace(/&trade;/g, '™');
// Decoded codepoints are escaped like literal input so an entity for a
// LIST_INDENT sentinel is not turned into indentation.
text = text.replace(/&#(\d+);/g,
(_, code) => escapeSentinels(String.fromCharCode(parseInt(code, 10))));
text = text.replace(/&#x([0-9a-fA-F]+);/g,
(_, code) => escapeSentinels(String.fromCharCode(parseInt(code, 16))));

text = text.replace(/&([a-zA-Z]+);/g, (match, name) => NAMED_ENTITIES[name] || match);
return text;
}

const BR_RE = /<br\s*\/?>/gi;
const HARD_BREAK_RE = new RegExp(HARD_BREAK, 'g');
const CODE_SPAN_RE = /`[^`\n]*`/g;

// Mark each <br> with HARD_BREAK so it survives whitespace collapse. A code
// span cannot hold a line break, so a <br> inside one stays a space.
function markBreaks(text) {
let out = '';
let last = 0;
for (const m of text.matchAll(CODE_SPAN_RE)) {
out += text.slice(last, m.index).replace(BR_RE, HARD_BREAK) + m[0].replace(BR_RE, ' ');
last = m.index + m[0].length;
}
return out + text.slice(last).replace(BR_RE, HARD_BREAK);
}

// Entities are decoded only after list items are rendered, but a hard break
// must see what a line will really start or end with: a trailing backslash
// written as a reference would otherwise escape the break's own `\`, and a
// leading run like `&gt;`, `-&#32;` or `&#35;` would become a block opener on
// the next line. resolveHardBreaks therefore checks the decoded line; a line
// that needs escaping is written back with `&` and `<` re-encoded, because
// the passes that run before the final decode would treat them as the start
// of a reference or a tag. Sentinels were already escaped by decodeEntities.
const encodeLine = (line) => line.replace(/&/g, '&#38;').replace(/</g, '&#60;');
const NBSP_REF_RE = /&nbsp;|&#0*160;|&#x0*a0;/gi;

// Flatten the inline content of a list item, matching the historical regex
// converter: paragraphs and any remaining tags become spaces and whitespace
// runs collapse. A <br> is kept as a CommonMark hard break, as in the
// walker's renderInlineRun.
function flattenItemText(text) {
return text
.replace(/<p>/g, '')
.replace(/<\/p>/g, ' ')
.replace(/<[^>]+>/g, ' ')
.replace(/\s+/g, ' ')
.trim();
let flat = markBreaks(text.replace(/<p>/g, '').replace(/<\/p>/g, ' ')).replace(/<[^>]+>/g, ' ');
// A line holding only &nbsp; would otherwise survive the collapse and leave
// a stray break behind. Only needed (and only applied) next to a break, so
// text without breaks is untouched.
if (flat.includes(HARD_BREAK)) flat = flat.replace(NBSP_REF_RE, ' ');
return resolveHardBreaks(collapseInline(flat), { decode: decodeEntities, encode: encodeLine });
}

// Render an <li> as the walker does (storage-walker renderListItemBody):
Expand Down Expand Up @@ -333,8 +397,13 @@ function htmlToMarkdown(html) {
cellMatches.forEach(cellMatch => {
let cellText = cellMatch.replace(/<t[hd]>(.*?)<\/t[hd]>/s, '$1');
cellText = cellText.replace(/<p>/g, '').replace(/<\/p>/g, ' ');
cellText = cellText.replace(/<[^>]+>/g, ' ').replace(/\s+/g, ' ').trim();
cells.push(cellText || ' ');
cellText = markBreaks(cellText).replace(/<[^>]+>/g, ' ');
if (cellText.includes(HARD_BREAK)) cellText = cellText.replace(NBSP_REF_RE, ' ');
cellText = collapseInline(cellText);
// A GFM row cannot span lines, so a break stays inline HTML. Written
// as entities so the tag-stripping pass below leaves it alone; the
// entity pass turns it into a literal `<br>`.
cells.push(cellText.replace(HARD_BREAK_RE, '&lt;br&gt;') || ' ');
});
}

Expand Down Expand Up @@ -364,31 +433,7 @@ function htmlToMarkdown(html) {

markdown = markdown.replace(/<(?!\/?(details|summary)\b)[^>]+>/g, ' ');

markdown = markdown.replace(/&nbsp;/g, ' ');
markdown = markdown.replace(/&lt;/g, '<');
markdown = markdown.replace(/&gt;/g, '>');
markdown = markdown.replace(/&amp;/g, '&');
markdown = markdown.replace(/&quot;/g, '"');
markdown = markdown.replace(/&apos;/g, '\'');
markdown = markdown.replace(/&ldquo;/g, '"');
markdown = markdown.replace(/&rdquo;/g, '"');
markdown = markdown.replace(/&lsquo;/g, '\'');
markdown = markdown.replace(/&rsquo;/g, '\'');
markdown = markdown.replace(/&mdash;/g, '—');
markdown = markdown.replace(/&ndash;/g, '–');
markdown = markdown.replace(/&hellip;/g, '...');
markdown = markdown.replace(/&bull;/g, '•');
markdown = markdown.replace(/&copy;/g, '©');
markdown = markdown.replace(/&reg;/g, '®');
markdown = markdown.replace(/&trade;/g, '™');
// Decoded codepoints are escaped like literal input so an entity for a
// LIST_INDENT sentinel is not turned into indentation.
markdown = markdown.replace(/&#(\d+);/g,
(_, code) => escapeSentinels(String.fromCharCode(parseInt(code, 10))));
markdown = markdown.replace(/&#x([0-9a-fA-F]+);/g,
(_, code) => escapeSentinels(String.fromCharCode(parseInt(code, 16))));

markdown = markdown.replace(/&([a-zA-Z]+);/g, (match, name) => NAMED_ENTITIES[name] || match);
markdown = decodeEntities(markdown);

return finalizeSentinels(cleanupWithFences(markdown));
}
Expand Down
57 changes: 56 additions & 1 deletion lib/markdown-cleanup.js
Original file line number Diff line number Diff line change
Expand Up @@ -10,7 +10,7 @@
// 4. Provide the LIST_INDENT / QUOTE_MARK / HARD_BREAK sentinels:
// LIST_INDENT is the list continuation indent both converters emit,
// QUOTE_MARK the storage-walker blockquote prefix, and HARD_BREAK carries
// a storage-walker <br/> through inline whitespace collapse. Step 3
// a <br/> through inline whitespace collapse (both converters). Step 3
// cannot strip that indentation and step 2 still sees nested fences.
//
// Keeping these helpers in one place prevents the converters from drifting
Expand Down Expand Up @@ -98,6 +98,58 @@ function stripListIndent(text) {
return text.replace(LIST_INDENT_RUN_RE, '');
}

const HARD_BREAK_PAD_RE = new RegExp(` *${HARD_BREAK} *`, 'g');
const HARD_BREAK_EDGE_RE = new RegExp(`^[ ${HARD_BREAK}]+|[ ${HARD_BREAK}]+$`, 'g');

// Collapse whitespace in an inline run, keeping HARD_BREAK as the only line
// boundary and dropping breaks (and spaces) at either edge.
function collapseInline(text) {
return text
.replace(/\s+/g, ' ')
.replace(HARD_BREAK_PAD_RE, HARD_BREAK)
.replace(HARD_BREAK_EDGE_RE, '');
}

// A line that follows a hard break is still paragraph text, but CommonMark
// would let some line starts interrupt the paragraph (nested list, heading,
// blockquote, fence), turn the line above into a setext heading, or make it
// a GFM table header (delimiter row). Escape
// just those openers; inline syntax such as `**bold**` is left alone. A
// backtick run followed by more backticks is a code span, not a fence.
function escapeContinuationLine(line) {
if (/^(?:[-=]+|(?:[-*_] *){3,})$/.test(line)
|| (/^[-:| ]+$/.test(line) && line.includes('|') && line.includes('-'))
|| /^(?:[-+*](?= |$)|#{1,6}(?= |$)|>|~{3}|`{3,}[^`]*$)/.test(line)) {
return '\\' + line;
}
return line.replace(/^(\d{1,9})([.)])(?= |$)/, '$1\\$2');
}

// Resolve HARD_BREAK in a collapsed inline run to a CommonMark backslash hard
// break (trailing-space breaks would not survive cleanupOutsideFence).
// For callers whose text still holds undecoded entities, `decode` maps a line
// to what it will finally read as, and is what the block-opener and
// trailing-backslash checks look at; a line that needs escaping is replaced
// by `encode(escaped decoded line)`, any other line is kept as written.
function resolveHardBreaks(text, { decode = (line) => line, encode = (line) => line } = {}) {
if (!text.includes(HARD_BREAK)) return text;
const lines = text.split(HARD_BREAK);
return lines
.map((line, i) => {
let out = line;
if (i > 0) {
const plain = decode(line);
const escaped = escapeContinuationLine(plain);
if (escaped !== plain) out = encode(escaped);
}
// An odd run of trailing backslashes (`C:\temp\`) would escape the
// break's own `\`; double the last one so it stays literal.
const trailing = decode(out).match(/\\*$/)[0].length;
return i < lines.length - 1 && trailing % 2 === 1 ? out + '\\' : out;
})
.join('\\\n');
}

// Turn LIST_INDENT back into spaces and QUOTE_MARK into `>`, degrade any
// HARD_BREAK its owning context failed to resolve to a newline, then restore
// escaped literals.
Expand Down Expand Up @@ -292,6 +344,9 @@ module.exports = {
QUOTE_MARK,
escapeSentinels,
stripListIndent,
collapseInline,
escapeContinuationLine,
resolveHardBreaks,
finalizeSentinels,
fenceLength,
escapeFenceLikeText,
Expand Down
44 changes: 5 additions & 39 deletions lib/storage-walker.js
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,8 @@ const {
QUOTE_MARK,
escapeSentinels,
stripListIndent,
collapseInline,
resolveHardBreaks,
finalizeSentinels,
fenceLength,
escapeFenceLikeText,
Expand Down Expand Up @@ -66,23 +68,6 @@ const TRANSPARENT_LIST_WRAPPERS = new Set([
]);

const HARD_BREAK_RE = new RegExp(HARD_BREAK, 'g');
const HARD_BREAK_PAD_RE = new RegExp(` *${HARD_BREAK} *`, 'g');
const HARD_BREAK_EDGE_RE = new RegExp(`^[ ${HARD_BREAK}]+|[ ${HARD_BREAK}]+$`, 'g');

// A line that follows a hard break is still paragraph text, but CommonMark
// would let some line starts interrupt the paragraph (nested list, heading,
// blockquote, fence), turn the line above into a setext heading, or make it
// a GFM table header (delimiter row). Escape
// just those openers; inline syntax such as `**bold**` is left alone. A
// backtick run followed by more backticks is a code span, not a fence.
function escapeContinuationLine(line) {
if (/^(?:[-=]+|(?:[-*_] *){3,})$/.test(line)
|| (/^[-:| ]+$/.test(line) && line.includes('|') && line.includes('-'))
|| /^(?:[-+*](?= |$)|#{1,6}(?= |$)|>|~{3}|`{3,}[^`]*$)/.test(line)) {
return '\\' + line;
}
return line.replace(/^(\d{1,9})([.)])(?= |$)/, '$1\\$2');
}

// Decode HTML entity references, matching the original htmlToMarkdown
// bit-for-bit: nbsp / ldquo / rdquo / lsquo / rsquo / hellip → ASCII,
Expand Down Expand Up @@ -461,35 +446,16 @@ class StorageWalker {
}
}

// Collapse whitespace in an inline run, keeping HARD_BREAK as the only
// line boundary and dropping breaks (and spaces) at either edge.
collapseInline(text) {
return text
.replace(/\s+/g, ' ')
.replace(HARD_BREAK_PAD_RE, HARD_BREAK)
.replace(HARD_BREAK_EDGE_RE, '');
}

renderInlineRun(text) {
return this.renderHardBreaks(this.collapseInline(stripListIndent(text)));
return this.renderHardBreaks(collapseInline(stripListIndent(text)));
}

// Resolve HARD_BREAK to a CommonMark backslash hard break. Trailing-space
// breaks would not survive cleanupOutsideFence. Inside an enclosing
// collapsing context (table cell, outer list item) the sentinel is left
// for that context to resolve.
renderHardBreaks(text) {
if (this._hardBreakDepth > 0 || !text.includes(HARD_BREAK)) return text;
const lines = text.split(HARD_BREAK);
return lines
.map((line, i) => {
const out = i === 0 ? line : escapeContinuationLine(line);
// An odd run of trailing backslashes (`C:\temp\`) would escape the
// break's own `\`; double the last one so it stays literal.
const trailing = out.match(/\\*$/)[0].length;
return i < lines.length - 1 && trailing % 2 === 1 ? out + '\\' : out;
})
.join('\\\n');
return this._hardBreakDepth > 0 ? text : resolveHardBreaks(text);
}

isListItemBlock(node) {
Expand All @@ -509,7 +475,7 @@ class StorageWalker {
if (cells.length === 0) continue;
// GFM table rows cannot span lines, so <br/> stays inline HTML.
const cellTexts = cells.map((cell) =>
this.collapseInline(stripListIndent(this.walkWithHardBreaks(cell.children)))
collapseInline(stripListIndent(this.walkWithHardBreaks(cell.children)))
.replace(HARD_BREAK_RE, '<br>') || ' '
);
rows.push('| ' + cellTexts.join(' | ') + ' |');
Expand Down
Loading
Loading