From ac72878d1c8d90c2d3a9241a73c0d58c2cf8b9b8 Mon Sep 17 00:00:00 2001 From: pchuri Date: Sun, 4 Oct 2026 21:35:38 +0900 Subject: [PATCH 1/2] =?UTF-8?q?fix(converter):=20keep=20
=20as=20a=20ha?= =?UTF-8?q?rd=20break=20in=20html=20=E2=86=92=20markdown=20lists=20and=20t?= =?UTF-8?q?ables?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit html → markdown turned
in list items and table cells into a space, while storage → markdown has kept them since #248 (a backslash hard break in list items, an inline
in cells), so the converters disagreed on the same markup. Move the walker's hard-break helpers (collapseInline, escapeContinuationLine and the trailing-backslash handling, now resolveHardBreaks) into markdown-cleanup so both converters share them. In html → markdown, mark each
with HARD_BREAK outside code spans, resolve it per list item, and write it as an inline
in table cells. Backslash and block-opener character references are decoded where they decide how a break is escaped, since entities are otherwise decoded after list items are rendered. Top-level
stays a soft break. Fixes #253 --- lib/html-to-markdown.js | 58 +++++++++++--- lib/markdown-cleanup.js | 48 ++++++++++++ lib/storage-walker.js | 44 ++--------- tests/html-to-markdown.test.js | 136 +++++++++++++++++++++++++++++++++ tests/markdown-cleanup.test.js | 46 +++++++++++ 5 files changed, 282 insertions(+), 50 deletions(-) diff --git a/lib/html-to-markdown.js b/lib/html-to-markdown.js index ce05df9..303427d 100644 --- a/lib/html-to-markdown.js +++ b/lib/html-to-markdown.js @@ -1,5 +1,8 @@ const { LIST_INDENT, + HARD_BREAK, + collapseInline, + resolveHardBreaks, escapeSentinels, finalizeSentinels, fenceLength, @@ -33,16 +36,46 @@ const NAMED_ENTITIES = { // Same value as StorageWalker's DEFAULT_MAX_DEPTH. const MAX_LIST_DEPTH = 256; -// Flatten the inline content of a list item onto one line, matching the -// historical regex converter: paragraphs and any remaining tags become -// spaces and whitespace runs collapse. +const BR_RE = //g; +const HARD_BREAK_RE = new RegExp(HARD_BREAK, 'g'); +const CODE_SPAN_RE = /`[^`\n]*`/g; + +// Mark each
with HARD_BREAK so it survives whitespace collapse. A code +// span cannot hold a line break, so a
inside one stays a space. +function markBreaks(text) { + let out = ''; + let last = 0; + for (const m of text.matchAll(CODE_SPAN_RE)) { + out += text.slice(last, m.index).replace(BR_RE, HARD_BREAK) + m[0].replace(BR_RE, ' '); + last = m.index + m[0].length; + } + return out + text.slice(last).replace(BR_RE, HARD_BREAK); +} + +// Entities are decoded only after list items are rendered, but a hard break +// must see what a line will really start or end with: a trailing `\` written +// as a character reference would otherwise escape the break's own `\`, and a +// leading `>` or `#` would become a block opener on the next line. +const BACKSLASH_REF_RE = /�*92;|�*5c;/gi; +const LEADING_REFS_RE = /^(?:>|&#\d+;|&#x[0-9a-f]+;)+/i; +const REF_RE = />|&#(\d+);|&#x([0-9a-f]+);/gi; +function decodeLeadingRefs(line) { + return line.replace(LEADING_REFS_RE, (run) => run.replace(REF_RE, (ref, dec, hex) => { + if (dec === undefined && hex === undefined) return '>'; + const code = dec !== undefined ? parseInt(dec, 10) : parseInt(hex, 16); + // `<` and `&` stay encoded: decoding them early would start a tag or + // another reference for the passes that run before the final decode. + return code === 0x3C || code === 0x26 || code > 0x10FFFF ? ref : String.fromCodePoint(code); + })); +} + +// Flatten the inline content of a list item, matching the historical regex +// converter: paragraphs and any remaining tags become spaces and whitespace +// runs collapse. A
is kept as a CommonMark hard break, as in the +// walker's renderInlineRun. function flattenItemText(text) { - return text - .replace(/

/g, '') - .replace(/<\/p>/g, ' ') - .replace(/<[^>]+>/g, ' ') - .replace(/\s+/g, ' ') - .trim(); + const flat = markBreaks(text.replace(/

/g, '').replace(/<\/p>/g, ' ')).replace(/<[^>]+>/g, ' '); + return resolveHardBreaks(collapseInline(flat.replace(BACKSLASH_REF_RE, '\\')), decodeLeadingRefs); } // Render an

  • as the walker does (storage-walker renderListItemBody): @@ -333,8 +366,11 @@ function htmlToMarkdown(html) { cellMatches.forEach(cellMatch => { let cellText = cellMatch.replace(/(.*?)<\/t[hd]>/s, '$1'); cellText = cellText.replace(/

    /g, '').replace(/<\/p>/g, ' '); - cellText = cellText.replace(/<[^>]+>/g, ' ').replace(/\s+/g, ' ').trim(); - cells.push(cellText || ' '); + cellText = collapseInline(markBreaks(cellText).replace(/<[^>]+>/g, ' ')); + // A GFM row cannot span lines, so a break stays inline HTML. Written + // as entities so the tag-stripping pass below leaves it alone; the + // entity pass turns it into a literal `
    `. + cells.push(cellText.replace(HARD_BREAK_RE, '<br>') || ' '); }); } diff --git a/lib/markdown-cleanup.js b/lib/markdown-cleanup.js index 5e06f7a..652409a 100644 --- a/lib/markdown-cleanup.js +++ b/lib/markdown-cleanup.js @@ -98,6 +98,51 @@ function stripListIndent(text) { return text.replace(LIST_INDENT_RUN_RE, ''); } +const HARD_BREAK_PAD_RE = new RegExp(` *${HARD_BREAK} *`, 'g'); +const HARD_BREAK_EDGE_RE = new RegExp(`^[ ${HARD_BREAK}]+|[ ${HARD_BREAK}]+$`, 'g'); + +// Collapse whitespace in an inline run, keeping HARD_BREAK as the only line +// boundary and dropping breaks (and spaces) at either edge. +function collapseInline(text) { + return text + .replace(/\s+/g, ' ') + .replace(HARD_BREAK_PAD_RE, HARD_BREAK) + .replace(HARD_BREAK_EDGE_RE, ''); +} + +// A line that follows a hard break is still paragraph text, but CommonMark +// would let some line starts interrupt the paragraph (nested list, heading, +// blockquote, fence), turn the line above into a setext heading, or make it +// a GFM table header (delimiter row). Escape +// just those openers; inline syntax such as `**bold**` is left alone. A +// backtick run followed by more backticks is a code span, not a fence. +function escapeContinuationLine(line) { + if (/^(?:[-=]+|(?:[-*_] *){3,})$/.test(line) + || (/^[-:| ]+$/.test(line) && line.includes('|') && line.includes('-')) + || /^(?:[-+*](?= |$)|#{1,6}(?= |$)|>|~{3}|`{3,}[^`]*$)/.test(line)) { + return '\\' + line; + } + return line.replace(/^(\d{1,9})([.)])(?= |$)/, '$1\\$2'); +} + +// Resolve HARD_BREAK in a collapsed inline run to a CommonMark backslash hard +// break (trailing-space breaks would not survive cleanupOutsideFence). +// `prepareLine` may rewrite a continuation line before it is checked for block +// openers, for callers whose text still holds undecoded entities. +function resolveHardBreaks(text, prepareLine = (line) => line) { + if (!text.includes(HARD_BREAK)) return text; + const lines = text.split(HARD_BREAK); + return lines + .map((line, i) => { + const out = i === 0 ? line : escapeContinuationLine(prepareLine(line)); + // An odd run of trailing backslashes (`C:\temp\`) would escape the + // break's own `\`; double the last one so it stays literal. + const trailing = out.match(/\\*$/)[0].length; + return i < lines.length - 1 && trailing % 2 === 1 ? out + '\\' : out; + }) + .join('\\\n'); +} + // Turn LIST_INDENT back into spaces and QUOTE_MARK into `>`, degrade any // HARD_BREAK its owning context failed to resolve to a newline, then restore // escaped literals. @@ -292,6 +337,9 @@ module.exports = { QUOTE_MARK, escapeSentinels, stripListIndent, + collapseInline, + escapeContinuationLine, + resolveHardBreaks, finalizeSentinels, fenceLength, escapeFenceLikeText, diff --git a/lib/storage-walker.js b/lib/storage-walker.js index f4b51c1..50c9dcf 100644 --- a/lib/storage-walker.js +++ b/lib/storage-walker.js @@ -6,6 +6,8 @@ const { QUOTE_MARK, escapeSentinels, stripListIndent, + collapseInline, + resolveHardBreaks, finalizeSentinels, fenceLength, escapeFenceLikeText, @@ -66,23 +68,6 @@ const TRANSPARENT_LIST_WRAPPERS = new Set([ ]); const HARD_BREAK_RE = new RegExp(HARD_BREAK, 'g'); -const HARD_BREAK_PAD_RE = new RegExp(` *${HARD_BREAK} *`, 'g'); -const HARD_BREAK_EDGE_RE = new RegExp(`^[ ${HARD_BREAK}]+|[ ${HARD_BREAK}]+$`, 'g'); - -// A line that follows a hard break is still paragraph text, but CommonMark -// would let some line starts interrupt the paragraph (nested list, heading, -// blockquote, fence), turn the line above into a setext heading, or make it -// a GFM table header (delimiter row). Escape -// just those openers; inline syntax such as `**bold**` is left alone. A -// backtick run followed by more backticks is a code span, not a fence. -function escapeContinuationLine(line) { - if (/^(?:[-=]+|(?:[-*_] *){3,})$/.test(line) - || (/^[-:| ]+$/.test(line) && line.includes('|') && line.includes('-')) - || /^(?:[-+*](?= |$)|#{1,6}(?= |$)|>|~{3}|`{3,}[^`]*$)/.test(line)) { - return '\\' + line; - } - return line.replace(/^(\d{1,9})([.)])(?= |$)/, '$1\\$2'); -} // Decode HTML entity references, matching the original htmlToMarkdown // bit-for-bit: nbsp / ldquo / rdquo / lsquo / rsquo / hellip → ASCII, @@ -461,17 +446,8 @@ class StorageWalker { } } - // Collapse whitespace in an inline run, keeping HARD_BREAK as the only - // line boundary and dropping breaks (and spaces) at either edge. - collapseInline(text) { - return text - .replace(/\s+/g, ' ') - .replace(HARD_BREAK_PAD_RE, HARD_BREAK) - .replace(HARD_BREAK_EDGE_RE, ''); - } - renderInlineRun(text) { - return this.renderHardBreaks(this.collapseInline(stripListIndent(text))); + return this.renderHardBreaks(collapseInline(stripListIndent(text))); } // Resolve HARD_BREAK to a CommonMark backslash hard break. Trailing-space @@ -479,17 +455,7 @@ class StorageWalker { // collapsing context (table cell, outer list item) the sentinel is left // for that context to resolve. renderHardBreaks(text) { - if (this._hardBreakDepth > 0 || !text.includes(HARD_BREAK)) return text; - const lines = text.split(HARD_BREAK); - return lines - .map((line, i) => { - const out = i === 0 ? line : escapeContinuationLine(line); - // An odd run of trailing backslashes (`C:\temp\`) would escape the - // break's own `\`; double the last one so it stays literal. - const trailing = out.match(/\\*$/)[0].length; - return i < lines.length - 1 && trailing % 2 === 1 ? out + '\\' : out; - }) - .join('\\\n'); + return this._hardBreakDepth > 0 ? text : resolveHardBreaks(text); } isListItemBlock(node) { @@ -509,7 +475,7 @@ class StorageWalker { if (cells.length === 0) continue; // GFM table rows cannot span lines, so
    stays inline HTML. const cellTexts = cells.map((cell) => - this.collapseInline(stripListIndent(this.walkWithHardBreaks(cell.children))) + collapseInline(stripListIndent(this.walkWithHardBreaks(cell.children))) .replace(HARD_BREAK_RE, '
    ') || ' ' ); rows.push('| ' + cellTexts.join(' | ') + ' |'); diff --git a/tests/html-to-markdown.test.js b/tests/html-to-markdown.test.js index eef00dd..0854bd8 100644 --- a/tests/html-to-markdown.test.js +++ b/tests/html-to-markdown.test.js @@ -666,4 +666,140 @@ describe('htmlToMarkdown', () => { }); }); }); + describe('
    in list items and table cells (#253)', () => { + const converter = new MacroConverter({ isCloud: true }); + + test('issue repro:
    in a list item is a backslash hard break', () => { + expect(htmlToMarkdown('

    • line1
      line2
    ')).toBe('- line1\\\n line2'); + }); + + test('issue repro:
    in a table cell stays an inline
    ', () => { + expect(htmlToMarkdown('
    line one
    line two
    ')) + .toBe('| line one
    line two |\n| --- |'); + }); + + test('continuation is indented to the content column of the marker', () => { + expect(htmlToMarkdown('
    1. a
      b
    2. c
    ')).toBe('1. a\\\n b\n2. c'); + const items = Array.from({ length: 9 }, (_, i) => `
  • i${i + 1}
  • `).join(''); + expect(htmlToMarkdown(`
      ${items}
    1. ten
      more
    `).split('\n').slice(-2).join('\n')) + .toBe('10. ten\\\n more'); + }); + + test('nested item continuation is indented to the nested content column', () => { + expect(htmlToMarkdown('')).toBe('- P\n - a\\\n b'); + }); + + test('surrounding whitespace is dropped and consecutive breaks are kept', () => { + expect(htmlToMarkdown('')).toBe('- a\\\n b\\\n \\\n c'); + }); + + test('leading and trailing breaks are dropped', () => { + expect(htmlToMarkdown('')).toBe('- a'); + }); + + test('a trailing backslash before a break is doubled so the break survives', () => { + expect(htmlToMarkdown('')).toBe('- C:\\temp\\\\\\\n next'); + expect(htmlToMarkdown('')).toBe('- a\\\\\\\n b'); + }); + + test('a backslash written as a character reference is counted too', () => { + expect(htmlToMarkdown('')).toBe('- C:\\temp\\\\\\\n next'); + expect(htmlToMarkdown('')).toBe('- C:\\temp\\\\\\\n next'); + }); + + test('paragraphs in an item are merged as before and keep their breaks', () => { + expect(htmlToMarkdown('')).toBe('- a\\\n b c'); + }); + + test('a break inside a code span stays a space', () => { + expect(htmlToMarkdown('')).toBe('- `a b` x\\\n y'); + expect(htmlToMarkdown('
    a
    b
    ')).toBe('| `a b` |\n| --- |'); + }); + + test('top-level
    is still a soft break', () => { + expect(htmlToMarkdown('

    a
    b

    ')).toBe('a\nb'); + }); + + describe('continuation lines that would start a block are escaped', () => { + test.each([ + ['- b', '\\- b'], + ['* b', '\\* b'], + ['+ b', '\\+ b'], + ['1. b', '1\\. b'], + ['2) b', '2\\) b'], + ['# b', '\\# b'], + ['> b', '\\> b'], + ['---', '\\---'], + ['-', '\\-'], + ['===', '\\==='], + ['* * *', '\\* * *'], + ['~~~', '\\~~~'], + ['--- | ---', '\\--- | ---'], + ['|:---|---:|', '\\|:---|---:|'], + // written as character references, which are decoded later + ['# b', '\\# b'], + ['- b', '\\- b'], + ['---', '\\---'], + ])('%s', (text, escaped) => { + expect(htmlToMarkdown(``)).toBe(`- a\\\n ${escaped}`); + }); + + test('a literal ``` line is escaped and cannot capture a later code block', () => { + const html = '
      x\n    y
    '; + expect(htmlToMarkdown(html)).toBe('- a\\\n \\`\\`\\`js\n\n```\n x\n y\n```'); + }); + + test('a decoded < or & is left encoded until the final pass', () => { + expect(htmlToMarkdown('')).toBe('- a\\\n c'); + expect(htmlToMarkdown('
    • a
      &amp; c
    ')).toBe('- a\\\n & c'); + }); + + test('inline syntax at line start is left alone', () => { + expect(htmlToMarkdown('
    • a
      b
      10 items
    ')) + .toBe('- a\\\n **b**\\\n 10 items'); + }); + }); + + test('table cells: header, several cells and surrounding whitespace', () => { + expect(htmlToMarkdown('
    h1
    h2
    k
    a
    b
    c
    d
    ')) + .toBe('| h1
    h2 | k |\n| --- | --- |\n| a
    b | c
    d |'); + }); + + test('a table cell break does not leak into a following list', () => { + expect(htmlToMarkdown('
    a
    b
    • c
      d
    ')) + .toBe('| a
    b |\n| --- |\n\n- c\\\n d'); + }); + + test('literal U+E003 in content survives next to a real break', () => { + expect(htmlToMarkdown('
    • a\uE003b
      c
    ')).toBe('- a\uE003b\\\n c'); + }); + + describe('matches storage → markdown for the same markup', () => { + const BLOCK_STARTS = ['- b', '* b', '+ b', '1. b', '2) b', '# b', '> b', '---', '-', '===', '* * *', '```js', '~~~', '--- | ---', '|:---|---:|']; + test.each([ + '
    • line1
      line2
    ', + '
    1. a
      b
    2. c
    ', + '
    • C:\\temp\\
      next
    ', + '
    • a\\\\
      b
    ', + '
    • a
      b

      • c
    ', + '
    • P
      • a
        b
    ', + '
    • a \n
      b

      c
    ', + '

    • a

    ', + '
    • a
      b
    ', + '', + '
    • x
      a | b
      --- | ---
    ', + '
    • ax
      - y
    ', + '
    h
    a
    b
    ', + '
    h1
    h2
    k
    a
    b
    c
    d
    ', + '
    a
    b
    ', + '

    a
    b

    ', + '
    • x
      # y
    ', + '
    • x
      ---
    ', + '
    • a\
      b
    ', + ...BLOCK_STARTS.map((text) => `
    • a
      ${text}
    `), + ])('%s', (html) => { + expect(htmlToMarkdown(html)).toBe(converter.storageToMarkdown(html)); + }); + }); + }); }); diff --git a/tests/markdown-cleanup.test.js b/tests/markdown-cleanup.test.js index 5b6556f..e41d110 100644 --- a/tests/markdown-cleanup.test.js +++ b/tests/markdown-cleanup.test.js @@ -6,6 +6,9 @@ const { finalizeSentinels, fenceLength, escapeFenceLikeText, + collapseInline, + escapeContinuationLine, + resolveHardBreaks, splitOnFences, isFenceOpenLine, cleanupOutsideFence, @@ -334,3 +337,46 @@ describe('escapeFenceLikeText', () => { expect(segments[1]).toBe('```\n code\n```'); }); }); + +describe('hard-break helpers (#253)', () => { + const B = HARD_BREAK; + + test('collapseInline keeps breaks, trims their padding and drops edge breaks', () => { + expect(collapseInline(`a \n ${B} b ${B}${B}c`)).toBe(`a${B}b${B}${B}c`); + expect(collapseInline(`${B} a ${B}`)).toBe('a'); + expect(collapseInline(' x \t y ')).toBe('x y'); + }); + + test.each([ + ['- b', '\\- b'], + ['1. b', '1\\. b'], + ['2) b', '2\\) b'], + ['# b', '\\# b'], + ['> b', '\\> b'], + ['---', '\\---'], + ['|:---|---:|', '\\|:---|---:|'], + ['plain', 'plain'], + ['**bold**', '**bold**'], + ['10 items', '10 items'], + ])('escapeContinuationLine(%j)', (line, expected) => { + expect(escapeContinuationLine(line)).toBe(expected); + }); + + test('resolveHardBreaks joins lines with a backslash break and escapes openers', () => { + expect(resolveHardBreaks(`a${B}b`)).toBe('a\\\nb'); + expect(resolveHardBreaks(`a${B}- b`)).toBe('a\\\n\\- b'); + expect(resolveHardBreaks('no breaks')).toBe('no breaks'); + }); + + test('an odd run of trailing backslashes is doubled, an even run is left', () => { + expect(resolveHardBreaks(`C:\\temp\\${B}next`)).toBe('C:\\temp\\\\\\\nnext'); + expect(resolveHardBreaks(`a\\\\${B}b`)).toBe('a\\\\\\\nb'); + }); + + test('prepareLine rewrites a continuation line before it is checked, but not the first line', () => { + const seen = []; + const prepare = (line) => { seen.push(line); return line.replace('>', '>'); }; + expect(resolveHardBreaks(`> a${B}> b`, prepare)).toBe('> a\\\n\\> b'); + expect(seen).toEqual(['> b']); + }); +}); From 9a65fe2be8c7e089eed3d860ff8245f63e9997f9 Mon Sep 17 00:00:00 2001 From: pchuri Date: Sun, 4 Oct 2026 23:21:20 +0900 Subject: [PATCH 2/2] fix(converter): decide hard-break escaping on decoded lines MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit html → markdown decodes entities only after list items are rendered, so the break handling looked at undecoded text. A continuation line that only decodes to a block opener (`- b`, `----`, `&#35; b`) was not escaped, a reference to a sentinel codepoint could become a live sentinel, and ` ` next to a break left a stray backslash. Extract the converter's entity pass into one decodeEntities function and let resolveHardBreaks take decode/encode hooks: the opener and trailing-backslash checks look at the decoded line, a line that needs escaping is written back with & and < re-encoded, and other lines are kept as written. Drop   lines next to a break before the collapse, only when a break is present. Treat
    as a break, as HTML tag names are case-insensitive. --- lib/html-to-markdown.js | 97 +++++++++++++++++++--------------- lib/markdown-cleanup.js | 19 ++++--- tests/html-to-markdown.test.js | 63 ++++++++++++++++++++++ tests/markdown-cleanup.test.js | 17 ++++-- 4 files changed, 141 insertions(+), 55 deletions(-) diff --git a/lib/html-to-markdown.js b/lib/html-to-markdown.js index 303427d..8686ab0 100644 --- a/lib/html-to-markdown.js +++ b/lib/html-to-markdown.js @@ -36,7 +36,40 @@ const NAMED_ENTITIES = { // Same value as StorageWalker's DEFAULT_MAX_DEPTH. const MAX_LIST_DEPTH = 256; -const BR_RE = //g; +// Decode the entities the converter understands. Run last for the document, +// and early on a continuation line whose first characters decide whether it +// needs escaping (see resolveHardBreaks). +function decodeEntities(input) { + let text = input; + text = text.replace(/ /g, ' '); + text = text.replace(/</g, '<'); + text = text.replace(/>/g, '>'); + text = text.replace(/&/g, '&'); + text = text.replace(/"/g, '"'); + text = text.replace(/'/g, '\''); + text = text.replace(/“/g, '"'); + text = text.replace(/”/g, '"'); + text = text.replace(/‘/g, '\''); + text = text.replace(/’/g, '\''); + text = text.replace(/—/g, '—'); + text = text.replace(/–/g, '–'); + text = text.replace(/…/g, '...'); + text = text.replace(/•/g, '•'); + text = text.replace(/©/g, '©'); + text = text.replace(/®/g, '®'); + text = text.replace(/™/g, '™'); + // Decoded codepoints are escaped like literal input so an entity for a + // LIST_INDENT sentinel is not turned into indentation. + text = text.replace(/&#(\d+);/g, + (_, code) => escapeSentinels(String.fromCharCode(parseInt(code, 10)))); + text = text.replace(/&#x([0-9a-fA-F]+);/g, + (_, code) => escapeSentinels(String.fromCharCode(parseInt(code, 16)))); + + text = text.replace(/&([a-zA-Z]+);/g, (match, name) => NAMED_ENTITIES[name] || match); + return text; +} + +const BR_RE = //gi; const HARD_BREAK_RE = new RegExp(HARD_BREAK, 'g'); const CODE_SPAN_RE = /`[^`\n]*`/g; @@ -53,29 +86,27 @@ function markBreaks(text) { } // Entities are decoded only after list items are rendered, but a hard break -// must see what a line will really start or end with: a trailing `\` written -// as a character reference would otherwise escape the break's own `\`, and a -// leading `>` or `#` would become a block opener on the next line. -const BACKSLASH_REF_RE = /�*92;|�*5c;/gi; -const LEADING_REFS_RE = /^(?:>|&#\d+;|&#x[0-9a-f]+;)+/i; -const REF_RE = />|&#(\d+);|&#x([0-9a-f]+);/gi; -function decodeLeadingRefs(line) { - return line.replace(LEADING_REFS_RE, (run) => run.replace(REF_RE, (ref, dec, hex) => { - if (dec === undefined && hex === undefined) return '>'; - const code = dec !== undefined ? parseInt(dec, 10) : parseInt(hex, 16); - // `<` and `&` stay encoded: decoding them early would start a tag or - // another reference for the passes that run before the final decode. - return code === 0x3C || code === 0x26 || code > 0x10FFFF ? ref : String.fromCodePoint(code); - })); -} +// must see what a line will really start or end with: a trailing backslash +// written as a reference would otherwise escape the break's own `\`, and a +// leading run like `>`, `- ` or `#` would become a block opener on +// the next line. resolveHardBreaks therefore checks the decoded line; a line +// that needs escaping is written back with `&` and `<` re-encoded, because +// the passes that run before the final decode would treat them as the start +// of a reference or a tag. Sentinels were already escaped by decodeEntities. +const encodeLine = (line) => line.replace(/&/g, '&').replace(/ is kept as a CommonMark hard break, as in the // walker's renderInlineRun. function flattenItemText(text) { - const flat = markBreaks(text.replace(/

    /g, '').replace(/<\/p>/g, ' ')).replace(/<[^>]+>/g, ' '); - return resolveHardBreaks(collapseInline(flat.replace(BACKSLASH_REF_RE, '\\')), decodeLeadingRefs); + let flat = markBreaks(text.replace(/

    /g, '').replace(/<\/p>/g, ' ')).replace(/<[^>]+>/g, ' '); + // A line holding only   would otherwise survive the collapse and leave + // a stray break behind. Only needed (and only applied) next to a break, so + // text without breaks is untouched. + if (flat.includes(HARD_BREAK)) flat = flat.replace(NBSP_REF_RE, ' '); + return resolveHardBreaks(collapseInline(flat), { decode: decodeEntities, encode: encodeLine }); } // Render an

  • as the walker does (storage-walker renderListItemBody): @@ -366,7 +397,9 @@ function htmlToMarkdown(html) { cellMatches.forEach(cellMatch => { let cellText = cellMatch.replace(/(.*?)<\/t[hd]>/s, '$1'); cellText = cellText.replace(/

    /g, '').replace(/<\/p>/g, ' '); - cellText = collapseInline(markBreaks(cellText).replace(/<[^>]+>/g, ' ')); + cellText = markBreaks(cellText).replace(/<[^>]+>/g, ' '); + if (cellText.includes(HARD_BREAK)) cellText = cellText.replace(NBSP_REF_RE, ' '); + cellText = collapseInline(cellText); // A GFM row cannot span lines, so a break stays inline HTML. Written // as entities so the tag-stripping pass below leaves it alone; the // entity pass turns it into a literal `
    `. @@ -400,31 +433,7 @@ function htmlToMarkdown(html) { markdown = markdown.replace(/<(?!\/?(details|summary)\b)[^>]+>/g, ' '); - markdown = markdown.replace(/ /g, ' '); - markdown = markdown.replace(/</g, '<'); - markdown = markdown.replace(/>/g, '>'); - markdown = markdown.replace(/&/g, '&'); - markdown = markdown.replace(/"/g, '"'); - markdown = markdown.replace(/'/g, '\''); - markdown = markdown.replace(/“/g, '"'); - markdown = markdown.replace(/”/g, '"'); - markdown = markdown.replace(/‘/g, '\''); - markdown = markdown.replace(/’/g, '\''); - markdown = markdown.replace(/—/g, '—'); - markdown = markdown.replace(/–/g, '–'); - markdown = markdown.replace(/…/g, '...'); - markdown = markdown.replace(/•/g, '•'); - markdown = markdown.replace(/©/g, '©'); - markdown = markdown.replace(/®/g, '®'); - markdown = markdown.replace(/™/g, '™'); - // Decoded codepoints are escaped like literal input so an entity for a - // LIST_INDENT sentinel is not turned into indentation. - markdown = markdown.replace(/&#(\d+);/g, - (_, code) => escapeSentinels(String.fromCharCode(parseInt(code, 10)))); - markdown = markdown.replace(/&#x([0-9a-fA-F]+);/g, - (_, code) => escapeSentinels(String.fromCharCode(parseInt(code, 16)))); - - markdown = markdown.replace(/&([a-zA-Z]+);/g, (match, name) => NAMED_ENTITIES[name] || match); + markdown = decodeEntities(markdown); return finalizeSentinels(cleanupWithFences(markdown)); } diff --git a/lib/markdown-cleanup.js b/lib/markdown-cleanup.js index 652409a..1122d52 100644 --- a/lib/markdown-cleanup.js +++ b/lib/markdown-cleanup.js @@ -10,7 +10,7 @@ // 4. Provide the LIST_INDENT / QUOTE_MARK / HARD_BREAK sentinels: // LIST_INDENT is the list continuation indent both converters emit, // QUOTE_MARK the storage-walker blockquote prefix, and HARD_BREAK carries -// a storage-walker
    through inline whitespace collapse. Step 3 +// a
    through inline whitespace collapse (both converters). Step 3 // cannot strip that indentation and step 2 still sees nested fences. // // Keeping these helpers in one place prevents the converters from drifting @@ -127,17 +127,24 @@ function escapeContinuationLine(line) { // Resolve HARD_BREAK in a collapsed inline run to a CommonMark backslash hard // break (trailing-space breaks would not survive cleanupOutsideFence). -// `prepareLine` may rewrite a continuation line before it is checked for block -// openers, for callers whose text still holds undecoded entities. -function resolveHardBreaks(text, prepareLine = (line) => line) { +// For callers whose text still holds undecoded entities, `decode` maps a line +// to what it will finally read as, and is what the block-opener and +// trailing-backslash checks look at; a line that needs escaping is replaced +// by `encode(escaped decoded line)`, any other line is kept as written. +function resolveHardBreaks(text, { decode = (line) => line, encode = (line) => line } = {}) { if (!text.includes(HARD_BREAK)) return text; const lines = text.split(HARD_BREAK); return lines .map((line, i) => { - const out = i === 0 ? line : escapeContinuationLine(prepareLine(line)); + let out = line; + if (i > 0) { + const plain = decode(line); + const escaped = escapeContinuationLine(plain); + if (escaped !== plain) out = encode(escaped); + } // An odd run of trailing backslashes (`C:\temp\`) would escape the // break's own `\`; double the last one so it stays literal. - const trailing = out.match(/\\*$/)[0].length; + const trailing = decode(out).match(/\\*$/)[0].length; return i < lines.length - 1 && trailing % 2 === 1 ? out + '\\' : out; }) .join('\\\n'); diff --git a/tests/html-to-markdown.test.js b/tests/html-to-markdown.test.js index 0854bd8..17916c6 100644 --- a/tests/html-to-markdown.test.js +++ b/tests/html-to-markdown.test.js @@ -760,6 +760,60 @@ describe('htmlToMarkdown', () => { }); }); + describe('character references around a break', () => { + test('  lines next to a break are dropped as whitespace', () => { + expect(htmlToMarkdown('

    • a
       
    • b
    ')).toBe('- a\n- b'); + expect(htmlToMarkdown('
    •  
      a
    ')).toBe('- a'); + expect(htmlToMarkdown('
    • a
       
       
    ')).toBe('- a'); + expect(htmlToMarkdown('
    • a
       
    ')).toBe('- a'); + expect(htmlToMarkdown('
    a
     
    ')).toBe('| a |\n| --- |'); + expect(htmlToMarkdown('
     
     
    ')).toBe('| |\n| --- |'); + }); + + test('  in an item without a break is left exactly as before', () => { + expect(htmlToMarkdown('
    • a  b
    ')).toBe('- a b'); + expect(htmlToMarkdown('
    •  a
    ')).toBe('- a'); + }); + + test.each([ + // an opener built from a mix of literal text and references + ['- b', '\\- b'], + ['- b', '\\- b'], + ['----', '\\----'], + ['# b', '\\# b'], + ['* b', '\\* b'], + ['1. b', '1\\. b'], + ['|---|', '\\|---|'], + // a reference to a reference (the entity pass decodes `&` first) + ['&#35; b', '\\# b'], + ['&#45; b', '\\- b'], + ])('a continuation line that only decodes to a block opener is escaped: %s', (text, escaped) => { + expect(htmlToMarkdown(`
    • a
      ${text}
    `)).toBe(`- a\\\n ${escaped}`); + }); + + test('a backslash produced by a double-decoded reference is counted before the break', () => { + expect(htmlToMarkdown('
    • a&#92;
      b
    ')).toBe('- a\\\\\\\n b'); + }); + + test('a reference to a sentinel codepoint stays literal text on a continuation line', () => { + const out = htmlToMarkdown('
    • a
      x
      y
      z
    '); + expect(out).toBe('- a\\\n \uE002x\\\n \uE003y\\\n \uE000z'); + }); + + test('< and & written as references survive on an escaped line', () => { + expect(htmlToMarkdown('
    • a
      - <b> &amp; c
    ')).toBe('- a\\\n \\- & c'); + }); + }); + + test('
    is a break too, since HTML tag names are case-insensitive', () => { + expect(htmlToMarkdown('
    • a
      b
    ')).toBe('- a\\\n b'); + }); + + test('a break in a cell holding a list is flattened with the list', () => { + expect(htmlToMarkdown('
    • a
      b
    ')) + .toBe('| a
    b |\n| --- |'); + }); + test('table cells: header, several cells and surrounding whitespace', () => { expect(htmlToMarkdown('
    h1
    h2
    k
    a
    b
    c
    d
    ')) .toBe('| h1
    h2 | k |\n| --- | --- |\n| a
    b | c
    d |'); @@ -796,6 +850,15 @@ describe('htmlToMarkdown', () => { '
    • x
      # y
    ', '
    • x
      ---
    ', '
    • a\
      b
    ', + '
    • a&#92;
      b
    ', + '
    • a
       
    • b
    ', + '
    •  
      a
    ', + '
    a
     
    ', + '
    • a
      - b
    ', + '
    • a
      ----
    ', + '
    • a
      &#35; b
    ', + '
    • a
      1. b
    ', + '
    • a
      |---|
    ', ...BLOCK_STARTS.map((text) => `
    • a
      ${text}
    `), ])('%s', (html) => { expect(htmlToMarkdown(html)).toBe(converter.storageToMarkdown(html)); diff --git a/tests/markdown-cleanup.test.js b/tests/markdown-cleanup.test.js index e41d110..4638189 100644 --- a/tests/markdown-cleanup.test.js +++ b/tests/markdown-cleanup.test.js @@ -373,10 +373,17 @@ describe('hard-break helpers (#253)', () => { expect(resolveHardBreaks(`a\\\\${B}b`)).toBe('a\\\\\\\nb'); }); - test('prepareLine rewrites a continuation line before it is checked, but not the first line', () => { - const seen = []; - const prepare = (line) => { seen.push(line); return line.replace('>', '>'); }; - expect(resolveHardBreaks(`> a${B}> b`, prepare)).toBe('> a\\\n\\> b'); - expect(seen).toEqual(['> b']); + test('decode and encode let a caller check lines it has not decoded yet', () => { + const decode = (line) => line.replace(/>/g, '>').replace(/\/g, '\\'); + const encode = (line) => line.replace(/>/g, '>'); + // A continuation line that decodes to a block opener is escaped and re-encoded; + // one that does not is kept exactly as written. + expect(resolveHardBreaks(`a${B}> b${B}x > y`, { decode, encode })).toBe('a\\\n\\> b\\\nx > y'); + // The trailing-backslash check looks at the decoded line. + expect(resolveHardBreaks(`C:\temp\${B}next`, { decode })).toBe('C:\temp\\\\\\nnext'); + }); + + test('the first line is only checked for trailing backslashes, never escaped as an opener', () => { + expect(resolveHardBreaks(`- a${B}b`)).toBe('- a\\\nb'); }); });