diff --git a/lib/html-to-markdown.js b/lib/html-to-markdown.js
index ce05df9..8686ab0 100644
--- a/lib/html-to-markdown.js
+++ b/lib/html-to-markdown.js
@@ -1,5 +1,8 @@
const {
LIST_INDENT,
+ HARD_BREAK,
+ collapseInline,
+ resolveHardBreaks,
escapeSentinels,
finalizeSentinels,
fenceLength,
@@ -33,16 +36,77 @@ const NAMED_ENTITIES = {
// Same value as StorageWalker's DEFAULT_MAX_DEPTH.
const MAX_LIST_DEPTH = 256;
-// Flatten the inline content of a list item onto one line, matching the
-// historical regex converter: paragraphs and any remaining tags become
-// spaces and whitespace runs collapse.
+// Decode the entities the converter understands. Run last for the document,
+// and early on a continuation line whose first characters decide whether it
+// needs escaping (see resolveHardBreaks).
+function decodeEntities(input) {
+ let text = input;
+ text = text.replace(/ /g, ' ');
+ text = text.replace(/</g, '<');
+ text = text.replace(/>/g, '>');
+ text = text.replace(/&/g, '&');
+ text = text.replace(/"/g, '"');
+ text = text.replace(/'/g, '\'');
+ text = text.replace(/“/g, '"');
+ text = text.replace(/”/g, '"');
+ text = text.replace(/‘/g, '\'');
+ text = text.replace(/’/g, '\'');
+ text = text.replace(/—/g, '—');
+ text = text.replace(/–/g, '–');
+ text = text.replace(/…/g, '...');
+ text = text.replace(/•/g, '•');
+ text = text.replace(/©/g, '©');
+ text = text.replace(/®/g, '®');
+ text = text.replace(/™/g, '™');
+ // Decoded codepoints are escaped like literal input so an entity for a
+ // LIST_INDENT sentinel is not turned into indentation.
+ text = text.replace(/(\d+);/g,
+ (_, code) => escapeSentinels(String.fromCharCode(parseInt(code, 10))));
+ text = text.replace(/([0-9a-fA-F]+);/g,
+ (_, code) => escapeSentinels(String.fromCharCode(parseInt(code, 16))));
+
+ text = text.replace(/&([a-zA-Z]+);/g, (match, name) => NAMED_ENTITIES[name] || match);
+ return text;
+}
+
+const BR_RE = /
/gi;
+const HARD_BREAK_RE = new RegExp(HARD_BREAK, 'g');
+const CODE_SPAN_RE = /`[^`\n]*`/g;
+
+// Mark each
with HARD_BREAK so it survives whitespace collapse. A code
+// span cannot hold a line break, so a
inside one stays a space.
+function markBreaks(text) {
+ let out = '';
+ let last = 0;
+ for (const m of text.matchAll(CODE_SPAN_RE)) {
+ out += text.slice(last, m.index).replace(BR_RE, HARD_BREAK) + m[0].replace(BR_RE, ' ');
+ last = m.index + m[0].length;
+ }
+ return out + text.slice(last).replace(BR_RE, HARD_BREAK);
+}
+
+// Entities are decoded only after list items are rendered, but a hard break
+// must see what a line will really start or end with: a trailing backslash
+// written as a reference would otherwise escape the break's own `\`, and a
+// leading run like `>`, `- ` or `#` would become a block opener on
+// the next line. resolveHardBreaks therefore checks the decoded line; a line
+// that needs escaping is written back with `&` and `<` re-encoded, because
+// the passes that run before the final decode would treat them as the start
+// of a reference or a tag. Sentinels were already escaped by decodeEntities.
+const encodeLine = (line) => line.replace(/&/g, '&').replace(/ is kept as a CommonMark hard break, as in the
+// walker's renderInlineRun.
function flattenItemText(text) {
- return text
- .replace(/
/g, '') - .replace(/<\/p>/g, ' ') - .replace(/<[^>]+>/g, ' ') - .replace(/\s+/g, ' ') - .trim(); + let flat = markBreaks(text.replace(/
/g, '').replace(/<\/p>/g, ' ')).replace(/<[^>]+>/g, ' '); + // A line holding only would otherwise survive the collapse and leave + // a stray break behind. Only needed (and only applied) next to a break, so + // text without breaks is untouched. + if (flat.includes(HARD_BREAK)) flat = flat.replace(NBSP_REF_RE, ' '); + return resolveHardBreaks(collapseInline(flat), { decode: decodeEntities, encode: encodeLine }); } // Render an
/g, '').replace(/<\/p>/g, ' ');
- cellText = cellText.replace(/<[^>]+>/g, ' ').replace(/\s+/g, ' ').trim();
- cells.push(cellText || ' ');
+ cellText = markBreaks(cellText).replace(/<[^>]+>/g, ' ');
+ if (cellText.includes(HARD_BREAK)) cellText = cellText.replace(NBSP_REF_RE, ' ');
+ cellText = collapseInline(cellText);
+ // A GFM row cannot span lines, so a break stays inline HTML. Written
+ // as entities so the tag-stripping pass below leaves it alone; the
+ // entity pass turns it into a literal `
`.
+ cells.push(cellText.replace(HARD_BREAK_RE, '<br>') || ' ');
});
}
@@ -364,31 +433,7 @@ function htmlToMarkdown(html) {
markdown = markdown.replace(/<(?!\/?(details|summary)\b)[^>]+>/g, ' ');
- markdown = markdown.replace(/ /g, ' ');
- markdown = markdown.replace(/</g, '<');
- markdown = markdown.replace(/>/g, '>');
- markdown = markdown.replace(/&/g, '&');
- markdown = markdown.replace(/"/g, '"');
- markdown = markdown.replace(/'/g, '\'');
- markdown = markdown.replace(/“/g, '"');
- markdown = markdown.replace(/”/g, '"');
- markdown = markdown.replace(/‘/g, '\'');
- markdown = markdown.replace(/’/g, '\'');
- markdown = markdown.replace(/—/g, '—');
- markdown = markdown.replace(/–/g, '–');
- markdown = markdown.replace(/…/g, '...');
- markdown = markdown.replace(/•/g, '•');
- markdown = markdown.replace(/©/g, '©');
- markdown = markdown.replace(/®/g, '®');
- markdown = markdown.replace(/™/g, '™');
- // Decoded codepoints are escaped like literal input so an entity for a
- // LIST_INDENT sentinel is not turned into indentation.
- markdown = markdown.replace(/(\d+);/g,
- (_, code) => escapeSentinels(String.fromCharCode(parseInt(code, 10))));
- markdown = markdown.replace(/([0-9a-fA-F]+);/g,
- (_, code) => escapeSentinels(String.fromCharCode(parseInt(code, 16))));
-
- markdown = markdown.replace(/&([a-zA-Z]+);/g, (match, name) => NAMED_ENTITIES[name] || match);
+ markdown = decodeEntities(markdown);
return finalizeSentinels(cleanupWithFences(markdown));
}
diff --git a/lib/markdown-cleanup.js b/lib/markdown-cleanup.js
index 5e06f7a..1122d52 100644
--- a/lib/markdown-cleanup.js
+++ b/lib/markdown-cleanup.js
@@ -10,7 +10,7 @@
// 4. Provide the LIST_INDENT / QUOTE_MARK / HARD_BREAK sentinels:
// LIST_INDENT is the list continuation indent both converters emit,
// QUOTE_MARK the storage-walker blockquote prefix, and HARD_BREAK carries
-// a storage-walker
through inline whitespace collapse. Step 3
+// a
through inline whitespace collapse (both converters). Step 3
// cannot strip that indentation and step 2 still sees nested fences.
//
// Keeping these helpers in one place prevents the converters from drifting
@@ -98,6 +98,58 @@ function stripListIndent(text) {
return text.replace(LIST_INDENT_RUN_RE, '');
}
+const HARD_BREAK_PAD_RE = new RegExp(` *${HARD_BREAK} *`, 'g');
+const HARD_BREAK_EDGE_RE = new RegExp(`^[ ${HARD_BREAK}]+|[ ${HARD_BREAK}]+$`, 'g');
+
+// Collapse whitespace in an inline run, keeping HARD_BREAK as the only line
+// boundary and dropping breaks (and spaces) at either edge.
+function collapseInline(text) {
+ return text
+ .replace(/\s+/g, ' ')
+ .replace(HARD_BREAK_PAD_RE, HARD_BREAK)
+ .replace(HARD_BREAK_EDGE_RE, '');
+}
+
+// A line that follows a hard break is still paragraph text, but CommonMark
+// would let some line starts interrupt the paragraph (nested list, heading,
+// blockquote, fence), turn the line above into a setext heading, or make it
+// a GFM table header (delimiter row). Escape
+// just those openers; inline syntax such as `**bold**` is left alone. A
+// backtick run followed by more backticks is a code span, not a fence.
+function escapeContinuationLine(line) {
+ if (/^(?:[-=]+|(?:[-*_] *){3,})$/.test(line)
+ || (/^[-:| ]+$/.test(line) && line.includes('|') && line.includes('-'))
+ || /^(?:[-+*](?= |$)|#{1,6}(?= |$)|>|~{3}|`{3,}[^`]*$)/.test(line)) {
+ return '\\' + line;
+ }
+ return line.replace(/^(\d{1,9})([.)])(?= |$)/, '$1\\$2');
+}
+
+// Resolve HARD_BREAK in a collapsed inline run to a CommonMark backslash hard
+// break (trailing-space breaks would not survive cleanupOutsideFence).
+// For callers whose text still holds undecoded entities, `decode` maps a line
+// to what it will finally read as, and is what the block-opener and
+// trailing-backslash checks look at; a line that needs escaping is replaced
+// by `encode(escaped decoded line)`, any other line is kept as written.
+function resolveHardBreaks(text, { decode = (line) => line, encode = (line) => line } = {}) {
+ if (!text.includes(HARD_BREAK)) return text;
+ const lines = text.split(HARD_BREAK);
+ return lines
+ .map((line, i) => {
+ let out = line;
+ if (i > 0) {
+ const plain = decode(line);
+ const escaped = escapeContinuationLine(plain);
+ if (escaped !== plain) out = encode(escaped);
+ }
+ // An odd run of trailing backslashes (`C:\temp\`) would escape the
+ // break's own `\`; double the last one so it stays literal.
+ const trailing = decode(out).match(/\\*$/)[0].length;
+ return i < lines.length - 1 && trailing % 2 === 1 ? out + '\\' : out;
+ })
+ .join('\\\n');
+}
+
// Turn LIST_INDENT back into spaces and QUOTE_MARK into `>`, degrade any
// HARD_BREAK its owning context failed to resolve to a newline, then restore
// escaped literals.
@@ -292,6 +344,9 @@ module.exports = {
QUOTE_MARK,
escapeSentinels,
stripListIndent,
+ collapseInline,
+ escapeContinuationLine,
+ resolveHardBreaks,
finalizeSentinels,
fenceLength,
escapeFenceLikeText,
diff --git a/lib/storage-walker.js b/lib/storage-walker.js
index f4b51c1..50c9dcf 100644
--- a/lib/storage-walker.js
+++ b/lib/storage-walker.js
@@ -6,6 +6,8 @@ const {
QUOTE_MARK,
escapeSentinels,
stripListIndent,
+ collapseInline,
+ resolveHardBreaks,
finalizeSentinels,
fenceLength,
escapeFenceLikeText,
@@ -66,23 +68,6 @@ const TRANSPARENT_LIST_WRAPPERS = new Set([
]);
const HARD_BREAK_RE = new RegExp(HARD_BREAK, 'g');
-const HARD_BREAK_PAD_RE = new RegExp(` *${HARD_BREAK} *`, 'g');
-const HARD_BREAK_EDGE_RE = new RegExp(`^[ ${HARD_BREAK}]+|[ ${HARD_BREAK}]+$`, 'g');
-
-// A line that follows a hard break is still paragraph text, but CommonMark
-// would let some line starts interrupt the paragraph (nested list, heading,
-// blockquote, fence), turn the line above into a setext heading, or make it
-// a GFM table header (delimiter row). Escape
-// just those openers; inline syntax such as `**bold**` is left alone. A
-// backtick run followed by more backticks is a code span, not a fence.
-function escapeContinuationLine(line) {
- if (/^(?:[-=]+|(?:[-*_] *){3,})$/.test(line)
- || (/^[-:| ]+$/.test(line) && line.includes('|') && line.includes('-'))
- || /^(?:[-+*](?= |$)|#{1,6}(?= |$)|>|~{3}|`{3,}[^`]*$)/.test(line)) {
- return '\\' + line;
- }
- return line.replace(/^(\d{1,9})([.)])(?= |$)/, '$1\\$2');
-}
// Decode HTML entity references, matching the original htmlToMarkdown
// bit-for-bit: nbsp / ldquo / rdquo / lsquo / rsquo / hellip → ASCII,
@@ -461,17 +446,8 @@ class StorageWalker {
}
}
- // Collapse whitespace in an inline run, keeping HARD_BREAK as the only
- // line boundary and dropping breaks (and spaces) at either edge.
- collapseInline(text) {
- return text
- .replace(/\s+/g, ' ')
- .replace(HARD_BREAK_PAD_RE, HARD_BREAK)
- .replace(HARD_BREAK_EDGE_RE, '');
- }
-
renderInlineRun(text) {
- return this.renderHardBreaks(this.collapseInline(stripListIndent(text)));
+ return this.renderHardBreaks(collapseInline(stripListIndent(text)));
}
// Resolve HARD_BREAK to a CommonMark backslash hard break. Trailing-space
@@ -479,17 +455,7 @@ class StorageWalker {
// collapsing context (table cell, outer list item) the sentinel is left
// for that context to resolve.
renderHardBreaks(text) {
- if (this._hardBreakDepth > 0 || !text.includes(HARD_BREAK)) return text;
- const lines = text.split(HARD_BREAK);
- return lines
- .map((line, i) => {
- const out = i === 0 ? line : escapeContinuationLine(line);
- // An odd run of trailing backslashes (`C:\temp\`) would escape the
- // break's own `\`; double the last one so it stays literal.
- const trailing = out.match(/\\*$/)[0].length;
- return i < lines.length - 1 && trailing % 2 === 1 ? out + '\\' : out;
- })
- .join('\\\n');
+ return this._hardBreakDepth > 0 ? text : resolveHardBreaks(text);
}
isListItemBlock(node) {
@@ -509,7 +475,7 @@ class StorageWalker {
if (cells.length === 0) continue;
// GFM table rows cannot span lines, so
stays inline HTML.
const cellTexts = cells.map((cell) =>
- this.collapseInline(stripListIndent(this.walkWithHardBreaks(cell.children)))
+ collapseInline(stripListIndent(this.walkWithHardBreaks(cell.children)))
.replace(HARD_BREAK_RE, '
') || ' '
);
rows.push('| ' + cellTexts.join(' | ') + ' |');
diff --git a/tests/html-to-markdown.test.js b/tests/html-to-markdown.test.js
index eef00dd..17916c6 100644
--- a/tests/html-to-markdown.test.js
+++ b/tests/html-to-markdown.test.js
@@ -666,4 +666,203 @@ describe('htmlToMarkdown', () => {
});
});
});
+ describe('
in list items and table cells (#253)', () => {
+ const converter = new MacroConverter({ isCloud: true });
+
+ test('issue repro:
in a list item is a backslash hard break', () => {
+ expect(htmlToMarkdown('
| line one line two |
a
b
c
a
b xa |
a
b
x\n y';
+ expect(htmlToMarkdown(html)).toBe('- a\\\n \\`\\`\\`js\n\n```\n x\n y\n```');
+ });
+
+ test('a decoded < or & is left encoded until the final pass', () => {
+ expect(htmlToMarkdown('| a |
| |
|
| h1 h2 | k |
|---|---|
| a b | c d |
| a b |
a
b
x
- y| h |
|---|
| a b |
| h1 h2 | k |
|---|---|
| a b | c d |
a |
a
b
| a |