From 5d761708750a47338a83e27d2137fa153d3c7a3c Mon Sep 17 00:00:00 2001 From: ragool Date: Mon, 5 Oct 2026 13:43:39 +0200 Subject: [PATCH 1/6] add structural redlining for tables and refactor code and add typescript --- .gitignore | 3 +- .mocharc.json | 10 + README.md | 79 +- eslintrc.json | 15 +- js/core/atomicTags.d.ts | 63 + js/core/atomicTags.js | 81 + js/core/diff.d.ts | 29 + js/core/diff.js | 43 + js/core/matching.d.ts | 71 + js/core/matching.js | 299 ++++ js/core/operations.d.ts | 23 + js/core/operations.js | 88 ++ js/core/rendering.d.ts | 112 ++ js/core/rendering.js | 315 ++++ js/core/tokens.d.ts | 54 + js/core/tokens.js | 363 +++++ js/htmldiff.d.ts | 30 +- js/htmldiff.js | 1384 +---------------- js/tables/Cell.d.ts | 100 ++ js/tables/Cell.js | 202 +++ js/tables/ColumnAligner.d.ts | 43 + js/tables/ColumnAligner.js | 118 ++ js/tables/MergedCells.d.ts | 74 + js/tables/MergedCells.js | 294 ++++ js/tables/Row.d.ts | 66 + js/tables/Row.js | 84 + js/tables/RowAligner.d.ts | 50 + js/tables/RowAligner.js | 157 ++ js/tables/SequenceAligner.d.ts | 94 ++ js/tables/SequenceAligner.js | 167 ++ js/tables/Table.d.ts | 78 + js/tables/Table.js | 207 +++ js/tables/TableAligner.d.ts | 35 + js/tables/TableAligner.js | 96 ++ js/tables/TableMerger.d.ts | 77 + js/tables/TableMerger.js | 306 ++++ js/tables/TableRedlining.d.ts | 34 + js/tables/TableRedlining.js | 86 + js/tables/TableVersion.d.ts | 143 ++ js/tables/TableVersion.js | 248 +++ js/tables/constants.d.ts | 13 + js/tables/constants.js | 14 + js/tables/helpers.d.ts | 26 + js/tables/helpers.js | 45 + js/tables/html.d.ts | 90 ++ js/tables/html.js | 154 ++ js/tables/index.d.ts | 2 + js/tables/index.js | 5 + js/tables/similarity.d.ts | 54 + js/tables/similarity.js | 144 ++ package-lock.json | 337 +++- package.json | 24 +- src/core/atomicTags.ts | 86 + src/core/diff.ts | 46 + src/core/matching.ts | 360 +++++ src/core/operations.ts | 105 ++ src/core/rendering.ts | 445 ++++++ src/core/tokens.ts | 373 +++++ src/htmldiff.ts | 63 + src/tables/Cell.ts | 227 +++ src/tables/ColumnAligner.ts | 135 ++ src/tables/MergedCells.ts | 341 ++++ src/tables/Row.ts | 101 ++ src/tables/RowAligner.ts | 181 +++ src/tables/SequenceAligner.ts | 233 +++ src/tables/Table.ts | 235 +++ src/tables/TableAligner.ts | 104 ++ src/tables/TableMerger.ts | 343 ++++ src/tables/TableRedlining.ts | 98 ++ src/tables/TableVersion.ts | 293 ++++ src/tables/constants.ts | 15 + src/tables/helpers.ts | 42 + src/tables/html.ts | 188 +++ src/tables/index.ts | 2 + src/tables/similarity.ts | 144 ++ test/calculate_operations.spec.js | 278 ---- test/config.js | 2 - test/core/atomic_tags.spec.ts | 55 + test/core/calculate_operations.spec.ts | 189 +++ test/core/diff.spec.ts | 225 +++ test/core/diff_core.spec.ts | 31 + test/core/find_matching_blocks.spec.ts | 140 ++ test/core/html_to_tokens.spec.ts | 202 +++ test/core/inner_diff.spec.ts | 303 ++++ test/core/render_operations.spec.ts | 146 ++ test/core/rendering.spec.ts | 80 + test/core/tokens.spec.ts | 37 + test/diff.spec.js | 318 ---- ...{edge_cases.spec.js => edge_cases.spec.ts} | 28 +- test/find_matching_blocks.spec.js | 168 -- test/from_port_source.spec.js | 30 - test/from_port_source.spec.ts | 26 + test/html_to_tokens.spec.js | 249 --- test/inner_diff.spec.js | 333 ---- test/mocha.opts | 3 - test/module.spec.js | 11 - test/module.spec.ts | 8 + test/pain_games.spec.js | 18 - test/pain_games.spec.ts | 11 + test/render_operations.spec.js | 178 --- test/tables/Cell.spec.ts | 109 ++ test/tables/ColumnAligner.spec.ts | 126 ++ test/tables/MergedCells.spec.ts | 123 ++ test/tables/Row.spec.ts | 56 + test/tables/RowAligner.spec.ts | 179 +++ test/tables/SequenceAligner.spec.ts | 107 ++ test/tables/Table.spec.ts | 99 ++ test/tables/TableAligner.spec.ts | 99 ++ test/tables/TableMerger.spec.ts | 76 + test/tables/TableRedlining.spec.ts | 32 + test/tables/TableVersion.spec.ts | 122 ++ test/tables/helpers.spec.ts | 22 + test/tables/html.spec.ts | 102 ++ test/tables/redlining.spec.ts | 1032 ++++++++++++ test/tables/similarity.spec.ts | 69 + tsconfig.build.json | 16 + tsconfig.cli.json | 16 + tsconfig.json | 33 +- 118 files changed, 13359 insertions(+), 3017 deletions(-) create mode 100644 .mocharc.json create mode 100644 js/core/atomicTags.d.ts create mode 100644 js/core/atomicTags.js create mode 100644 js/core/diff.d.ts create mode 100644 js/core/diff.js create mode 100644 js/core/matching.d.ts create mode 100644 js/core/matching.js create mode 100644 js/core/operations.d.ts create mode 100644 js/core/operations.js create mode 100644 js/core/rendering.d.ts create mode 100644 js/core/rendering.js create mode 100644 js/core/tokens.d.ts create mode 100644 js/core/tokens.js create mode 100644 js/tables/Cell.d.ts create mode 100644 js/tables/Cell.js create mode 100644 js/tables/ColumnAligner.d.ts create mode 100644 js/tables/ColumnAligner.js create mode 100644 js/tables/MergedCells.d.ts create mode 100644 js/tables/MergedCells.js create mode 100644 js/tables/Row.d.ts create mode 100644 js/tables/Row.js create mode 100644 js/tables/RowAligner.d.ts create mode 100644 js/tables/RowAligner.js create mode 100644 js/tables/SequenceAligner.d.ts create mode 100644 js/tables/SequenceAligner.js create mode 100644 js/tables/Table.d.ts create mode 100644 js/tables/Table.js create mode 100644 js/tables/TableAligner.d.ts create mode 100644 js/tables/TableAligner.js create mode 100644 js/tables/TableMerger.d.ts create mode 100644 js/tables/TableMerger.js create mode 100644 js/tables/TableRedlining.d.ts create mode 100644 js/tables/TableRedlining.js create mode 100644 js/tables/TableVersion.d.ts create mode 100644 js/tables/TableVersion.js create mode 100644 js/tables/constants.d.ts create mode 100644 js/tables/constants.js create mode 100644 js/tables/helpers.d.ts create mode 100644 js/tables/helpers.js create mode 100644 js/tables/html.d.ts create mode 100644 js/tables/html.js create mode 100644 js/tables/index.d.ts create mode 100644 js/tables/index.js create mode 100644 js/tables/similarity.d.ts create mode 100644 js/tables/similarity.js create mode 100644 src/core/atomicTags.ts create mode 100644 src/core/diff.ts create mode 100644 src/core/matching.ts create mode 100644 src/core/operations.ts create mode 100644 src/core/rendering.ts create mode 100644 src/core/tokens.ts create mode 100644 src/htmldiff.ts create mode 100644 src/tables/Cell.ts create mode 100644 src/tables/ColumnAligner.ts create mode 100644 src/tables/MergedCells.ts create mode 100644 src/tables/Row.ts create mode 100644 src/tables/RowAligner.ts create mode 100644 src/tables/SequenceAligner.ts create mode 100644 src/tables/Table.ts create mode 100644 src/tables/TableAligner.ts create mode 100644 src/tables/TableMerger.ts create mode 100644 src/tables/TableRedlining.ts create mode 100644 src/tables/TableVersion.ts create mode 100644 src/tables/constants.ts create mode 100644 src/tables/helpers.ts create mode 100644 src/tables/html.ts create mode 100644 src/tables/index.ts create mode 100644 src/tables/similarity.ts delete mode 100644 test/calculate_operations.spec.js delete mode 100644 test/config.js create mode 100644 test/core/atomic_tags.spec.ts create mode 100644 test/core/calculate_operations.spec.ts create mode 100644 test/core/diff.spec.ts create mode 100644 test/core/diff_core.spec.ts create mode 100644 test/core/find_matching_blocks.spec.ts create mode 100644 test/core/html_to_tokens.spec.ts create mode 100644 test/core/inner_diff.spec.ts create mode 100644 test/core/render_operations.spec.ts create mode 100644 test/core/rendering.spec.ts create mode 100644 test/core/tokens.spec.ts delete mode 100644 test/diff.spec.js rename test/{edge_cases.spec.js => edge_cases.spec.ts} (53%) delete mode 100644 test/find_matching_blocks.spec.js delete mode 100644 test/from_port_source.spec.js create mode 100644 test/from_port_source.spec.ts delete mode 100644 test/html_to_tokens.spec.js delete mode 100644 test/inner_diff.spec.js delete mode 100644 test/mocha.opts delete mode 100644 test/module.spec.js create mode 100644 test/module.spec.ts delete mode 100644 test/pain_games.spec.js create mode 100644 test/pain_games.spec.ts delete mode 100644 test/render_operations.spec.js create mode 100644 test/tables/Cell.spec.ts create mode 100644 test/tables/ColumnAligner.spec.ts create mode 100644 test/tables/MergedCells.spec.ts create mode 100644 test/tables/Row.spec.ts create mode 100644 test/tables/RowAligner.spec.ts create mode 100644 test/tables/SequenceAligner.spec.ts create mode 100644 test/tables/Table.spec.ts create mode 100644 test/tables/TableAligner.spec.ts create mode 100644 test/tables/TableMerger.spec.ts create mode 100644 test/tables/TableRedlining.spec.ts create mode 100644 test/tables/TableVersion.spec.ts create mode 100644 test/tables/helpers.spec.ts create mode 100644 test/tables/html.spec.ts create mode 100644 test/tables/redlining.spec.ts create mode 100644 test/tables/similarity.spec.ts create mode 100644 tsconfig.build.json create mode 100644 tsconfig.cli.json diff --git a/.gitignore b/.gitignore index 44b42aa..9865a1a 100644 --- a/.gitignore +++ b/.gitignore @@ -1,2 +1,3 @@ node_modules -htmldiff-cli.js \ No newline at end of file +htmldiff-cli.js +.DS_Store diff --git a/.mocharc.json b/.mocharc.json new file mode 100644 index 0000000..f6f93a8 --- /dev/null +++ b/.mocharc.json @@ -0,0 +1,10 @@ +{ + "require": [ + "ts-node/register" + ], + "extension": [ + "ts" + ], + "spec": "test/**/*.spec.ts", + "ui": "bdd" +} diff --git a/README.md b/README.md index ea0bc78..13e615a 100644 --- a/README.md +++ b/README.md @@ -145,6 +145,48 @@ Limitations: - The recursion depth is capped at 10 levels as a backstop against deep nesting. Opted-in elements beyond the cap are rendered as their after version. +### Tables + +Tables are compared as structures before the flat diff runs. The two documents' top level +tables are paired (by their own `data-htmldiff-id` when they have one, otherwise by the values +they hold; a table in the other's place is the same table only when the two still share half of +what the smaller one holds, or when both sit in a document section), each pair is aligned +column by column and row by row, and a merged table takes the place of the after version's +table. A table without a partner is kept whole, so the flat diff wraps it as added or deleted: + +- an added or deleted row is a whole row with the class `table-row-added` or + `table-row-deleted`, +- an added or deleted column marks every one of its cells (and its ``, when the table has + a ``) with `table-cell-added` or `table-cell-deleted`, +- a kept cell holds the diff of its content, with the usual ``/`` tags, +- a row that keeps less than half of its content, or a column no kept row agrees with, is + deleted and added instead of diffed; so is a moved row or column, +- merged cells (`rowspan`, `colspan`) are kept. A row that a kept group (a merged cell and the + rows it spans) lost or gained goes under the group's kept rows and carries the change on its + cells (`table-cell-deleted` / `table-cell-added`), not on the row: the group's cell, sitting on + the group's first kept row, spans it like any other row of the group. A view that hides the + changed cells then hides nothing the span counts on, so the layout holds. A whole group that + one version has is added or deleted row by row. + +Inside an element with a `data-shadow-boundary` attribute a row is identified by the item +references it holds, as `` elements. The first reference +names the item the row is about: a row about another item is another row. A cell where a +reference was swapped for another (a trace, an execution) makes another row too. References +only added to a cell or only removed from it leave the row the same row, and then its other +cells decide as for any row: at least half of them kept means an edited row with cell diffs, +less means the row was replaced. An item both versions have, none of whose rows matched, +keeps its item cell once: it spans the old rows, whose cells are deleted, and the new rows, +whose cells are added. + +A producer may give the cells that name a row their own `data-htmldiff-id` (the item, the +trace, the execution). When both versions have such cells, those alone pair the rows: the same +identities are the same row, whatever its other cells say, and they get cell diffs; another +identity is another row. Tables without such cells are read as above. + +Both versions of a pair get the same `data-htmldiff-id` (`redline-table-` unless the table +had one), a table only one version has gets one of its own, so the flat diff keeps every table +whole and emits the merged table as it is. The styling of the classes is up to the consumer. + ### Example JavaScript: @@ -198,14 +240,35 @@ description please see API documentation above. ## Development -After cloning the repository run `npm i` or `npm install` to install the necessary -dependencies. A run of `npm run make` creates the JavaScript output file. -`npm run lint` checks the TypeScript sources with TSLint. `npm test` runs all the -tests from the `test` directory. `npm run testsample` diffs the HTML sample files -from the directory `sample` and logs the result to the console. - -The command line interface of htmldiff is developed in TypeScript so you have to run -`npm run make` once to create the JavaScript output file. +After cloning the repository run `npm install` to install the dependencies. + +Everything is TypeScript. The library lives in `src/` and is compiled to CommonJS in `js/` +(`js/htmldiff.js` is the entry point, `js/htmldiff.d.ts` the typings); `js/` is what gets +published. + +- `src/htmldiff.ts` is the facade: it runs the table pass, then the flat diff. +- `src/core/` is the flat diff, one module per stage: `atomicTags` (which elements are one + token), `tokens` (tokenizing and token keys), `matching` (matching blocks), `operations` + (insert, delete, replace, equal), `rendering` (ins/del markup and the recursive inner + diff) and `diff` (the pipeline). +- `src/tables/` is the structural table pass, one class per concern: `Cell`, `Row` and + `Table` are the model, `TableVersion` a table as the alignment reads it, `SequenceAligner`, + `ColumnAligner`, `RowAligner` and `TableAligner` decide what is the same, `MergedCells` + handles spans, `TableMerger` writes the merged table and `TableRedlining` is the pass + itself. `html.ts`, `similarity.ts` and `helpers.ts` are plain helper functions. + +Tests are TypeScript too, in `test/`, run by mocha through `ts-node` against `src/`. Every +module and class has a spec of its own (`test/core/`, `test/tables/`), next to the end to end +specs (`test/*.spec.ts`, `test/tables/redlining.spec.ts`). + +Scripts: + +- `npm run build` compiles `src/` to `js/`. +- `npm test` builds, type-checks sources and specs, then runs the specs. +- `npm run lint` checks sources and specs with ESLint. +- `npm run make` builds the library and the command line interface, `htmldiff-cli.ts`. +- `npm run testsample` diffs the HTML sample files from the directory `sample` and logs the + result to the console. ## Credits diff --git a/eslintrc.json b/eslintrc.json index 344bd3b..6f043f7 100644 --- a/eslintrc.json +++ b/eslintrc.json @@ -2,7 +2,7 @@ "root": true, "parser": "@typescript-eslint/parser", "parserOptions": { - "project": "./app/src/tsconfig.json" + "project": ["./tsconfig.json", "./tsconfig.cli.json"] }, "plugins": [ "@typescript-eslint", @@ -43,25 +43,16 @@ "warn", { "require": { - "ArrowFunctionExpression": true, "ClassDeclaration": true, - "ClassExpression": true, "FunctionDeclaration": true, - "FunctionExpression": true, "MethodDefinition": true }, "contexts": [ - "Property", - "ClassProperty:not([accessibility=\"private\"])", - "TSMethodSignature", "TSEnumDeclaration", "TSInterfaceDeclaration", - "TSTypeAliasDeclaration", - "-TSPropertySignature", - "ExportNamedDeclaration" + "TSTypeAliasDeclaration" ], - "checkGetters": true, - "checkSetters": true, + "publicOnly": true, "exemptEmptyConstructors": true } ], diff --git a/js/core/atomicTags.d.ts b/js/core/atomicTags.d.ts new file mode 100644 index 0000000..5bb5ed8 --- /dev/null +++ b/js/core/atomicTags.d.ts @@ -0,0 +1,63 @@ +/** + * Atomic tags are elements whose child nodes are never compared: the whole element is one + * token. Which tags are atomic changes while a diff runs (the caller's list for the outer + * diff, a reduced list inside a recursive inner diff), so the active regular expression is + * kept here and read by the tokenizer and the renderer. + */ +/** + * The default atomic tags. The tag name must be followed by a delimiter (not a \b word + * boundary): the tokenizer matches against partially read tags, and a word boundary would + * match at the end of an incomplete name, e.g. detecting '' as the atomic tag 'a' + * while reading '' + * into a nested child. The leading \s prevents matching 'x-data-htmldiff-id'. + */ +export declare const dataHtmlDiffIdRegExp: RegExp; +/** + * Opt-in marker for the recursive inner diff. When two matched atomic tokens (typically + * matched by data-htmldiff-id) have equal keys but different content, an element carrying + * this attribute gets its inner HTML diffed recursively instead of being rendered as is. + * The attribute must appear in the element's opening tag. Captures the attribute value; + * a bare attribute or any value other than "false" enables the opt-in. + */ +export declare const dataHtmlDiffInnerDiffRegExp: RegExp; +/** + * Per-element override for the atomic tags used inside a recursive inner diff. The value + * is a comma separated tag name list, like the atomicTags parameter of the diff function; + * an empty value means no tag name is atomic. + */ +export declare const dataHtmlDiffInnerDiffAtomicTagsRegExp: RegExp; +/** + * The atomic tags regular expression the running diff uses. + * @returns The active regular expression. + */ +export declare function getAtomicTagsRegExp(): RegExp; +/** + * Switches the atomic tags for the diff that is about to run. + * @param regExp The regular expression to match the start of an atomic tag. + */ +export declare function setAtomicTagsRegExp(regExp: RegExp): void; +/** + * Builds the atomic tags regular expression from a comma separated tag name list. + * @param atomicTags Comma separated list of tag names, e.g. 'head,script,style'. + * @returns The regular expression matching the start of those tags. + */ +export declare function buildAtomicTagsRegExp(atomicTags: string): RegExp; +/** + * Checks if the current word is the beginning of an atomic tag: one of the active atomic + * tags, or any element with a data-htmldiff-id of its own. + * @param word The characters of the current token read so far. + * @returns The name of the atomic tag if the word will be an atomic tag, null otherwise. + */ +export declare function isStartOfAtomicTag(word: string): string | null; diff --git a/js/core/atomicTags.js b/js/core/atomicTags.js new file mode 100644 index 0000000..5b1f1e9 --- /dev/null +++ b/js/core/atomicTags.js @@ -0,0 +1,81 @@ +"use strict"; +/** + * Atomic tags are elements whose child nodes are never compared: the whole element is one + * token. Which tags are atomic changes while a diff runs (the caller's list for the outer + * diff, a reduced list inside a recursive inner diff), so the active regular expression is + * kept here and read by the tokenizer and the renderer. + */ +Object.defineProperty(exports, "__esModule", { value: true }); +exports.isStartOfAtomicTag = exports.buildAtomicTagsRegExp = exports.setAtomicTagsRegExp = exports.getAtomicTagsRegExp = exports.dataHtmlDiffInnerDiffAtomicTagsRegExp = exports.dataHtmlDiffInnerDiffRegExp = exports.dataHtmlDiffIdRegExp = exports.noAtomicTagsRegExp = exports.defaultInnerDiffAtomicTagsRegExp = exports.defaultAtomicTagsRegExp = void 0; +/** + * The default atomic tags. The tag name must be followed by a delimiter (not a \b word + * boundary): the tokenizer matches against partially read tags, and a word boundary would + * match at the end of an incomplete name, e.g. detecting '' as the atomic tag 'a' + * while reading ']"); +/** + * Atomic tags used inside a recursive inner diff unless the element overrides them via + * data-htmldiff-inner-diff-atomic-tags: the default list without 'a'. + */ +exports.defaultInnerDiffAtomicTagsRegExp = new RegExp("^<(iframe|object|math|svg|script|video|head|style)[\\s/>]"); +/** Matches no tag at all: used when data-htmldiff-inner-diff-atomic-tags is empty. */ +exports.noAtomicTagsRegExp = /^<(?!)/; +/** + * Matches an element whose own opening tag carries data-htmldiff-id; captures tag name and + * attribute value. The skip before the attribute is quote aware, so it cannot run past '>' + * into a nested child. The leading \s prevents matching 'x-data-htmldiff-id'. + */ +exports.dataHtmlDiffIdRegExp = /^<([a-z-]+)(?:[^>"']|"[^"]*"|'[^']*')*\sdata-htmldiff-id=["']?((?:.(?!["']?\s+(?:\S+)=|\s*\/?[>"']))*.)["']?/; +/** + * Opt-in marker for the recursive inner diff. When two matched atomic tokens (typically + * matched by data-htmldiff-id) have equal keys but different content, an element carrying + * this attribute gets its inner HTML diffed recursively instead of being rendered as is. + * The attribute must appear in the element's opening tag. Captures the attribute value; + * a bare attribute or any value other than "false" enables the opt-in. + */ +exports.dataHtmlDiffInnerDiffRegExp = /^<[^>]*\sdata-htmldiff-inner-diff(?:\s*=\s*["']?([^"'\s/>]*)|(?=[\s/>]))/; +/** + * Per-element override for the atomic tags used inside a recursive inner diff. The value + * is a comma separated tag name list, like the atomicTags parameter of the diff function; + * an empty value means no tag name is atomic. + */ +exports.dataHtmlDiffInnerDiffAtomicTagsRegExp = /^<[^>]*\sdata-htmldiff-inner-diff-atomic-tags\s*=\s*["']([^"']*)["']/; +let atomicTagsRegExp = exports.defaultAtomicTagsRegExp; +/** + * The atomic tags regular expression the running diff uses. + * @returns The active regular expression. + */ +function getAtomicTagsRegExp() { + return atomicTagsRegExp; +} +exports.getAtomicTagsRegExp = getAtomicTagsRegExp; +/** + * Switches the atomic tags for the diff that is about to run. + * @param regExp The regular expression to match the start of an atomic tag. + */ +function setAtomicTagsRegExp(regExp) { + atomicTagsRegExp = regExp; +} +exports.setAtomicTagsRegExp = setAtomicTagsRegExp; +/** + * Builds the atomic tags regular expression from a comma separated tag name list. + * @param atomicTags Comma separated list of tag names, e.g. 'head,script,style'. + * @returns The regular expression matching the start of those tags. + */ +function buildAtomicTagsRegExp(atomicTags) { + // Require a delimiter after the name (see defaultAtomicTagsRegExp on why not \b). + return new RegExp("^<(" + atomicTags.replace(/\s*/g, "").replace(/,/g, "|") + ")[\\s/>]"); +} +exports.buildAtomicTagsRegExp = buildAtomicTagsRegExp; +/** + * Checks if the current word is the beginning of an atomic tag: one of the active atomic + * tags, or any element with a data-htmldiff-id of its own. + * @param word The characters of the current token read so far. + * @returns The name of the atomic tag if the word will be an atomic tag, null otherwise. + */ +function isStartOfAtomicTag(word) { + const result = atomicTagsRegExp.exec(word) || exports.dataHtmlDiffIdRegExp.exec(word); + return result ? result[1] : null; +} +exports.isStartOfAtomicTag = isStartOfAtomicTag; diff --git a/js/core/diff.d.ts b/js/core/diff.d.ts new file mode 100644 index 0000000..7518114 --- /dev/null +++ b/js/core/diff.d.ts @@ -0,0 +1,29 @@ +/** + * The flat diff: tokenize both documents, find the operations between them, render them. Runs + * with whatever atomic tags are active; the public entry point sets them first. + */ +import { Operation } from "./operations"; +import { ContentDiff } from "./rendering"; +import { Token } from "./tokens"; +/** + * Diffs two fragments of HTML with the atomic tags that are currently active. Used by the + * public diff function after resolving the atomicTags parameter, by the recursive inner diff, + * which sets the atomic tags itself, and by the table pass to diff cell content. + * @param before The HTML content before the changes. + * @param after The HTML content after the changes. + * @param className (Optional) The class attribute to include in and tags. + * @param dataPrefix (Optional) The data prefix to use for data attributes. + * @returns The combined HTML content with differences wrapped in and tags. + */ +export declare const diffCore: ContentDiff; +/** + * Renders a list of operations into HTML content, diffing the content of opted-in atomic + * elements with the flat diff. + * @param beforeTokens The before list of tokens. + * @param afterTokens The after list of tokens. + * @param operations The operations to render. + * @param dataPrefix (Optional) The prefix to use in data attributes. + * @param className (Optional) The class name to include in the wrapper tag. + * @returns The rendering of the list of operations. + */ +export declare function renderOperations(beforeTokens: Token[], afterTokens: Token[], operations: Operation[], dataPrefix?: string | null, className?: string | null): string; diff --git a/js/core/diff.js b/js/core/diff.js new file mode 100644 index 0000000..b67d2a9 --- /dev/null +++ b/js/core/diff.js @@ -0,0 +1,43 @@ +"use strict"; +Object.defineProperty(exports, "__esModule", { value: true }); +exports.renderOperations = exports.diffCore = void 0; +/** + * The flat diff: tokenize both documents, find the operations between them, render them. Runs + * with whatever atomic tags are active; the public entry point sets them first. + */ +const operations_1 = require("./operations"); +const rendering_1 = require("./rendering"); +const tokens_1 = require("./tokens"); +/** + * Diffs two fragments of HTML with the atomic tags that are currently active. Used by the + * public diff function after resolving the atomicTags parameter, by the recursive inner diff, + * which sets the atomic tags itself, and by the table pass to diff cell content. + * @param before The HTML content before the changes. + * @param after The HTML content after the changes. + * @param className (Optional) The class attribute to include in and tags. + * @param dataPrefix (Optional) The data prefix to use for data attributes. + * @returns The combined HTML content with differences wrapped in and tags. + */ +const diffCore = (before, after, className, dataPrefix) => { + if (before === after) + return before; + const beforeTokens = (0, tokens_1.htmlToTokens)(before); + const afterTokens = (0, tokens_1.htmlToTokens)(after); + const ops = (0, operations_1.calculateOperations)(beforeTokens, afterTokens); + return (0, rendering_1.renderOperations)(beforeTokens, afterTokens, ops, exports.diffCore, dataPrefix, className); +}; +exports.diffCore = diffCore; +/** + * Renders a list of operations into HTML content, diffing the content of opted-in atomic + * elements with the flat diff. + * @param beforeTokens The before list of tokens. + * @param afterTokens The after list of tokens. + * @param operations The operations to render. + * @param dataPrefix (Optional) The prefix to use in data attributes. + * @param className (Optional) The class name to include in the wrapper tag. + * @returns The rendering of the list of operations. + */ +function renderOperations(beforeTokens, afterTokens, operations, dataPrefix, className) { + return (0, rendering_1.renderOperations)(beforeTokens, afterTokens, operations, exports.diffCore, dataPrefix, className); +} +exports.renderOperations = renderOperations; diff --git a/js/core/matching.d.ts b/js/core/matching.d.ts new file mode 100644 index 0000000..76a4dae --- /dev/null +++ b/js/core/matching.d.ts @@ -0,0 +1,71 @@ +/** + * Matching: finds the blocks of consecutive tokens that appear in both the before and the + * after token lists. The longest block is found first, then the blocks before and after it, + * recursively, until no match is left. + */ +import { Token } from "./tokens"; +/** The part of both documents a match is searched in. */ +export interface Segment { + beforeTokens: Token[]; + afterTokens: Token[]; + beforeMap: TokenMap; + afterMap: TokenMap; + beforeIndex: number; + afterIndex: number; +} +/** Token key to the indices of the tokens with that key. */ +export declare type TokenMap = Record; +/** + * A Match stores the information of a matching block. A matching block is a list of + * consecutive tokens that appear in both the before and after lists of tokens. + */ +export declare class Match { + segment: Segment; + length: number; + startInBefore: number; + startInAfter: number; + endInBefore: number; + endInAfter: number; + segmentStartInBefore: number; + segmentStartInAfter: number; + segmentEndInBefore: number; + segmentEndInAfter: number; + /** + * @param startInBefore The index of the first token in the list of before tokens. + * @param startInAfter The index of the first token in the list of after tokens. + * @param length The number of consecutive matching tokens in this block. + * @param segment The segment where the match was found. + */ + constructor(startInBefore: number, startInAfter: number, length: number, segment: Segment); +} +/** + * Creates a map from token key to an array of indices of locations of the matching token in + * the list of all tokens. + * @param tokens The list of tokens to be mapped. + * @returns A mapping that can be used to search for tokens. + */ +export declare function createMap(tokens: Token[]): TokenMap; +/** + * Finds and returns the best match between the before and after arrays contained in the + * segment provided. + * @param segment The segment in which to look for a match. + * @returns The best match, or null when the segment has none. + */ +export declare function findBestMatch(segment: Segment): Match | null; +/** + * Creates segment objects from the original document that can be used to restrict the area + * that findBestMatch and its helper functions search to increase performance. + * @param beforeTokens Tokens from the before document. + * @param afterTokens Tokens from the after document. + * @param beforeIndex The index within the before document where this segment begins. + * @param afterIndex The index within the after document where this segment begins. + * @returns The segment object. + */ +export declare function createSegment(beforeTokens: Token[], afterTokens: Token[], beforeIndex: number, afterIndex: number): Segment; +/** + * Finds all the matching blocks within the given segment in the before and after lists of + * tokens. + * @param segment The segment that should be searched for matching blocks. + * @returns The list of matching blocks in this range. + */ +export declare function findMatchingBlocks(segment: Segment): Match[]; diff --git a/js/core/matching.js b/js/core/matching.js new file mode 100644 index 0000000..a2795b1 --- /dev/null +++ b/js/core/matching.js @@ -0,0 +1,299 @@ +"use strict"; +Object.defineProperty(exports, "__esModule", { value: true }); +exports.findMatchingBlocks = exports.createSegment = exports.findBestMatch = exports.createMap = exports.Match = void 0; +/** + * A Match stores the information of a matching block. A matching block is a list of + * consecutive tokens that appear in both the before and after lists of tokens. + */ +class Match { + /** + * @param startInBefore The index of the first token in the list of before tokens. + * @param startInAfter The index of the first token in the list of after tokens. + * @param length The number of consecutive matching tokens in this block. + * @param segment The segment where the match was found. + */ + constructor(startInBefore, startInAfter, length, segment) { + this.segment = segment; + this.length = length; + this.startInBefore = startInBefore + segment.beforeIndex; + this.startInAfter = startInAfter + segment.afterIndex; + this.endInBefore = this.startInBefore + this.length - 1; + this.endInAfter = this.startInAfter + this.length - 1; + this.segmentStartInBefore = startInBefore; + this.segmentStartInAfter = startInAfter; + this.segmentEndInBefore = this.segmentStartInBefore + this.length - 1; + this.segmentEndInAfter = this.segmentStartInAfter + this.length - 1; + } +} +exports.Match = Match; +/** + * Creates a map from token key to an array of indices of locations of the matching token in + * the list of all tokens. + * @param tokens The list of tokens to be mapped. + * @returns A mapping that can be used to search for tokens. + */ +function createMap(tokens) { + return tokens.reduce((map, token, index) => { + if (map[token.key]) { + map[token.key].push(index); + } + else { + map[token.key] = [index]; + } + return map; + }, Object.create(null)); +} +exports.createMap = createMap; +/** + * Compares two match objects to determine if the second match object comes before or after the + * first match object. + * @param m1 The first match object to compare. + * @param m2 The second match object to compare. + * @returns -1 if m2 should come before m1, 1 if m1 should come before m2, 0 if the two + * matches criss-cross each other. + */ +function compareMatches(m1, m2) { + if (m2.endInBefore < m1.startInBefore && m2.endInAfter < m1.startInAfter) { + return -1; + } + if (m2.startInBefore > m1.endInBefore && m2.startInAfter > m1.endInAfter) { + return 1; + } + return 0; +} +/** A binary search tree that keeps match objects in the proper order as they're found. */ +class MatchBinarySearchTree { + constructor() { + this.root = null; + } + /** + * Adds a match to the binary search tree. A match overlapping an existing node is dropped. + * @param value The match to add. + */ + add(value) { + const node = { value, left: null, right: null }; + let current = this.root; + if (!current) { + this.root = node; + return; + } + for (;;) { + // Determine if the match value should go to the left or right of the current node. + const position = compareMatches(current.value, value); + if (position === -1) { + if (current.left) { + current = current.left; + } + else { + current.left = node; + break; + } + } + else if (position === 1) { + if (current.right) { + current = current.right; + } + else { + current.right = node; + break; + } + } + else { + // If 0 was returned from compareMatches, that means the node cannot + // be inserted because it overlaps an existing node. + break; + } + } + } + /** + * Converts the binary search tree into an array using an in-order traversal. + * @returns The matches in the binary search tree, in order. + */ + toArray() { + function inOrder(node, nodes) { + if (node) { + inOrder(node.left, nodes); + nodes.push(node.value); + inOrder(node.right, nodes); + } + return nodes; + } + return inOrder(this.root, []); + } +} +/** + * Finds and returns the best match between the before and after arrays contained in the + * segment provided. + * @param segment The segment in which to look for a match. + * @returns The best match, or null when the segment has none. + */ +function findBestMatch(segment) { + const beforeTokens = segment.beforeTokens; + const afterMap = segment.afterMap; + let lastSpace = null; + let bestMatch = null; + // Iterate through the entirety of the beforeTokens to find the best match. + for (let beforeIndex = 0; beforeIndex < beforeTokens.length; beforeIndex++) { + let lookBehind = false; + // If the current best match is longer than the remaining tokens, we can bail because we + // won't find a better match. + const remainingTokens = beforeTokens.length - beforeIndex; + if (bestMatch && remainingTokens < bestMatch.length) { + break; + } + // If the current token is whitespace, make a note of it and move on. Trying to start a + // set of matches with whitespace is not efficient because it's too prevelant in most + // documents. Instead, if the next token yields a match, we'll see if the whitespace can + // be included in that match. + const beforeToken = beforeTokens[beforeIndex]; + if (beforeToken.key === " ") { + lastSpace = beforeIndex; + continue; + } + // Check to see if we just skipped a space, if so, we'll ask getFullMatch to look behind + // by one token to see if it can include the whitespace. + if (lastSpace === beforeIndex - 1) { + lookBehind = true; + } + // If the current token is not found in the afterTokens, it won't match and we can move on. + const afterTokenLocations = afterMap[beforeToken.key]; + if (!afterTokenLocations) { + continue; + } + // For each instance of the current token in afterTokens, let's see how big of a match + // we can build. + for (const afterIndex of afterTokenLocations) { + // getFullMatch will see how far the current token match will go in both + // beforeTokens and afterTokens. + const bestMatchLength = bestMatch ? bestMatch.length : 0; + const match = getFullMatch(segment, beforeIndex, afterIndex, bestMatchLength, lookBehind); + // If we got a new best match, we'll save it aside. + if (match && match.length > bestMatchLength) { + bestMatch = match; + } + } + } + return bestMatch; +} +exports.findBestMatch = findBestMatch; +/** + * Takes the start of a match, and expands it in the beforeTokens and afterTokens of the + * current segment as far as it can go. + * @param segment The segment object to search within when expanding the match. + * @param beforeStart The offset within beforeTokens to start looking. + * @param afterStart The offset within afterTokens to start looking. + * @param minLength The minimum length match that must be found. + * @param lookBehind If true, attempt to match a whitespace token just before the + * beforeStart and afterStart tokens. + * @returns The full match, or undefined when no match of the minimum length starts here. + */ +function getFullMatch(segment, beforeStart, afterStart, minLength, lookBehind) { + const beforeTokens = segment.beforeTokens; + const afterTokens = segment.afterTokens; + // If we already have a match that goes to the end of the document, no need to keep looking. + const minBeforeIndex = beforeStart + minLength; + const minAfterIndex = afterStart + minLength; + if (minBeforeIndex >= beforeTokens.length || minAfterIndex >= afterTokens.length) { + return undefined; + } + // If a minLength was provided, we can do a quick check to see if the tokens after that + // length match. If not, we won't be beating the previous best match, and we can bail out + // early. + if (minLength) { + const nextBeforeWord = beforeTokens[minBeforeIndex].key; + const nextAfterWord = afterTokens[minAfterIndex].key; + if (nextBeforeWord !== nextAfterWord) { + return undefined; + } + } + // Extend the current match as far foward as it can go, without overflowing beforeTokens or + // afterTokens. + let searching = true; + let currentLength = 1; + let beforeIndex = beforeStart + currentLength; + let afterIndex = afterStart + currentLength; + while (searching && beforeIndex < beforeTokens.length && afterIndex < afterTokens.length) { + const beforeWord = beforeTokens[beforeIndex].key; + const afterWord = afterTokens[afterIndex].key; + if (beforeWord === afterWord) { + currentLength++; + beforeIndex = beforeStart + currentLength; + afterIndex = afterStart + currentLength; + } + else { + searching = false; + } + } + // If we've been asked to look behind, it's because both beforeTokens and afterTokens may + // have a whitespace token just behind the current match that was previously ignored. If so, + // we'll expand the current match to include it. + if (lookBehind && beforeStart > 0 && afterStart > 0) { + const prevBeforeKey = beforeTokens[beforeStart - 1].key; + const prevAfterKey = afterTokens[afterStart - 1].key; + if (prevBeforeKey === " " && prevAfterKey === " ") { + beforeStart--; + afterStart--; + currentLength++; + } + } + return new Match(beforeStart, afterStart, currentLength, segment); +} +/** + * Creates segment objects from the original document that can be used to restrict the area + * that findBestMatch and its helper functions search to increase performance. + * @param beforeTokens Tokens from the before document. + * @param afterTokens Tokens from the after document. + * @param beforeIndex The index within the before document where this segment begins. + * @param afterIndex The index within the after document where this segment begins. + * @returns The segment object. + */ +function createSegment(beforeTokens, afterTokens, beforeIndex, afterIndex) { + return { + beforeTokens, + afterTokens, + beforeMap: createMap(beforeTokens), + afterMap: createMap(afterTokens), + beforeIndex, + afterIndex, + }; +} +exports.createSegment = createSegment; +/** + * Finds all the matching blocks within the given segment in the before and after lists of + * tokens. + * @param segment The segment that should be searched for matching blocks. + * @returns The list of matching blocks in this range. + */ +function findMatchingBlocks(segment) { + // Create a binary search tree to hold the matches we find in order. + const matches = new MatchBinarySearchTree(); + const segments = [segment]; + // Each time the best match is found in a segment, zero, one or two new segments may be + // created from the parts of the original segment not included in the match. We will + // continue to iterate until all segments have been processed. + while (segments.length) { + const current = segments.pop(); + const match = findBestMatch(current); + if (match && match.length) { + // If there's an unmatched area at the start of the segment, create a new segment + // from that area and throw it into the segments array to get processed. + if (match.segmentStartInBefore > 0 && match.segmentStartInAfter > 0) { + const leftBeforeTokens = current.beforeTokens.slice(0, match.segmentStartInBefore); + const leftAfterTokens = current.afterTokens.slice(0, match.segmentStartInAfter); + segments.push(createSegment(leftBeforeTokens, leftAfterTokens, current.beforeIndex, current.afterIndex)); + } + // If there's an unmatched area at the end of the segment, create a new segment from + // that area and throw it into the segments array to get processed. + const rightBeforeTokens = current.beforeTokens.slice(match.segmentEndInBefore + 1); + const rightAfterTokens = current.afterTokens.slice(match.segmentEndInAfter + 1); + const rightBeforeIndex = current.beforeIndex + match.segmentEndInBefore + 1; + const rightAfterIndex = current.afterIndex + match.segmentEndInAfter + 1; + if (rightBeforeTokens.length && rightAfterTokens.length) { + segments.push(createSegment(rightBeforeTokens, rightAfterTokens, rightBeforeIndex, rightAfterIndex)); + } + matches.add(match); + } + } + return matches.toArray(); +} +exports.findMatchingBlocks = findMatchingBlocks; diff --git a/js/core/operations.d.ts b/js/core/operations.d.ts new file mode 100644 index 0000000..4a603e6 --- /dev/null +++ b/js/core/operations.d.ts @@ -0,0 +1,23 @@ +import { Token } from "./tokens"; +/** What happened to a range of tokens. */ +export declare type OperationAction = "equal" | "insert" | "delete" | "replace"; +/** + * A range of tokens on both sides and what happened to it. The end is null on the side an + * operation does not touch: an insert has no before range, a delete no after range. + */ +export interface Operation { + action: OperationAction; + startInBefore: number; + endInBefore: number | null; + startInAfter: number; + endInAfter: number | null; +} +/** + * Gets a list of operations required to transform the before list of tokens into the + * after list of tokens. An operation describes whether a particular list of consecutive + * tokens are equal, replaced, inserted, or deleted. + * @param beforeTokens The before list of tokens. + * @param afterTokens The after list of tokens. + * @returns The list of operations to transform the before list of tokens into the after list. + */ +export declare function calculateOperations(beforeTokens: Token[], afterTokens: Token[]): Operation[]; diff --git a/js/core/operations.js b/js/core/operations.js new file mode 100644 index 0000000..46f1d84 --- /dev/null +++ b/js/core/operations.js @@ -0,0 +1,88 @@ +"use strict"; +Object.defineProperty(exports, "__esModule", { value: true }); +exports.calculateOperations = void 0; +/** + * Operations: turns the matching blocks into the list of equal, insert, delete and replace + * steps that transform the before tokens into the after tokens. + */ +const matching_1 = require("./matching"); +/** + * Gets a list of operations required to transform the before list of tokens into the + * after list of tokens. An operation describes whether a particular list of consecutive + * tokens are equal, replaced, inserted, or deleted. + * @param beforeTokens The before list of tokens. + * @param afterTokens The after list of tokens. + * @returns The list of operations to transform the before list of tokens into the after list. + */ +function calculateOperations(beforeTokens, afterTokens) { + if (!beforeTokens) + throw new Error("Missing beforeTokens"); + if (!afterTokens) + throw new Error("Missing afterTokens"); + let positionInBefore = 0; + let positionInAfter = 0; + const operations = []; + const segment = (0, matching_1.createSegment)(beforeTokens, afterTokens, 0, 0); + const matches = (0, matching_1.findMatchingBlocks)(segment); + matches.push(new matching_1.Match(beforeTokens.length, afterTokens.length, 0, segment)); + for (let index = 0; index < matches.length; index++) { + const match = matches[index]; + let actionUpToMatchPositions = "none"; + if (positionInBefore === match.startInBefore) { + if (positionInAfter !== match.startInAfter) { + actionUpToMatchPositions = "insert"; + } + } + else { + actionUpToMatchPositions = "delete"; + if (positionInAfter !== match.startInAfter) { + actionUpToMatchPositions = "replace"; + } + } + if (actionUpToMatchPositions !== "none") { + operations.push({ + action: actionUpToMatchPositions, + startInBefore: positionInBefore, + endInBefore: actionUpToMatchPositions !== "insert" ? match.startInBefore - 1 : null, + startInAfter: positionInAfter, + endInAfter: actionUpToMatchPositions !== "delete" ? match.startInAfter - 1 : null, + }); + } + if (match.length !== 0) { + operations.push({ + action: "equal", + startInBefore: match.startInBefore, + endInBefore: match.endInBefore, + startInAfter: match.startInAfter, + endInAfter: match.endInAfter, + }); + } + positionInBefore = match.endInBefore + 1; + positionInAfter = match.endInAfter + 1; + } + const postProcessed = []; + let lastOp = { action: "none" }; + function isSingleWhitespace(op) { + if (op.action !== "equal") { + return false; + } + if (op.endInBefore - op.startInBefore !== 0) { + return false; + } + return /^\s$/.test(String(beforeTokens.slice(op.startInBefore, op.endInBefore + 1))); + } + for (let i = 0; i < operations.length; i++) { + const op = operations[i]; + if (lastOp.action === "replace" && + (isSingleWhitespace(op) || op.action === "replace")) { + lastOp.endInBefore = op.endInBefore; + lastOp.endInAfter = op.endInAfter; + } + else { + postProcessed.push(op); + lastOp = op; + } + } + return postProcessed; +} +exports.calculateOperations = calculateOperations; diff --git a/js/core/rendering.d.ts b/js/core/rendering.d.ts new file mode 100644 index 0000000..994641c --- /dev/null +++ b/js/core/rendering.d.ts @@ -0,0 +1,112 @@ +import { Operation } from "./operations"; +import { Token } from "./tokens"; +/** + * Diffs two fragments of HTML with the active atomic tags. The renderer gets it injected to + * diff the content of opted-in elements recursively; see the diff module. + */ +export declare type ContentDiff = (before: string, after: string, className?: string | null, dataPrefix?: string | null) => string; +/** A run of tokens that are all wrappable or all not. */ +interface TokenSegment { + isWrappable: boolean; + tokens: string[]; +} +/** + * A TokenWrapper provides a utility for grouping segments of tokens based on whether they're + * wrappable or not. A tag is considered wrappable if it is closed within the given set of + * tokens. For example, given the following tokens: + * + * ['', 'this', ' ', 'is', ' ', 'a', ' ', '', 'test', '', '!'] + * + * The first '' is not considered wrappable since the tag is not fully contained within the + * array of tokens. The '', 'test', and '' would be a part of the same wrappable segment + * since the entire bold tag is within the set of tokens. + */ +export declare class TokenWrapper { + private readonly tokens; + private readonly notes; + /** + * @param tokens The tokens to group. + */ + constructor(tokens: string[]); + /** + * Wraps the contained tokens in tags based on output given by a map function. Each segment + * of tokens will be visited. A segment is a continuous run of either all wrappable tokens or + * unwrappable tokens. The given map function will be called with each segment of tokens and + * the resulting strings will be combined to form the wrapped HTML. + * @param mapFn Called with each segment; the result should be a string. + * @param tagFn Called with the opening tag of every tag inserted whole, to mark it. + * @returns The wrapped HTML. + */ + combine(mapFn: (segment: TokenSegment) => string, tagFn: (openingTag: string) => string): string; +} +/** + * Wraps and concatenates a list of tokens with a tag. Does not wrap tag tokens, unless they + * are wrappable (i.e. void and atomic tags). + * @param tag The tag name of the wrapper tags. + * @param content The list of tokens to wrap. + * @param opIndex The index of the operation, written to the data attribute. + * @param dataPrefix (Optional) The prefix to use in data attributes. + * @param className (Optional) The class name to include in the wrapper tag. + * @returns The wrapped HTML. + */ +export declare function wrap(tag: string, content: string[], opIndex: number, dataPrefix?: string | null, className?: string | null): string; +/** + * Checks whether a token is an atomic tag that opted into the recursive inner diff via the + * data-htmldiff-inner-diff attribute. A bare attribute or any value other than "false" counts + * as opted in. Opted-in elements nested inside other opted-in elements are diffed recursively + * as well, up to the depth cap; beyond it, opted-in tokens are rendered verbatim like any + * other atomic token. + * @param tokenString The token string to check. + * @returns True if the token should get a recursive inner diff. + */ +export declare function isInnerDiffToken(tokenString: string): boolean; +/** + * Finds the index of the '>' that ends the opening tag at the start of the given token + * string, skipping any '>' inside quoted attribute values (e.g. title="a > b"). + * @param tokenString The token string starting with an opening tag. + * @returns The index of the closing '>' of the opening tag, or -1 if there is none (e.g. an + * unterminated tag or an unbalanced attribute quote). + */ +export declare function findOpeningTagEnd(tokenString: string): number; +/** An atomic token split into its tags and content. */ +export interface SplitToken { + openingTag: string; + innerHtml: string; + closingTag: string; +} +/** + * Splits an atomic token string into its opening tag, inner HTML and closing tag. A token + * consisting of a single tag (a void or self-closing element) has an empty inner HTML and no + * closing tag. + * @param tokenString The atomic token string, e.g. '
content
'. + * @returns The parts, or null if the token cannot be split (e.g. an unterminated tag). + */ +export declare function splitAtomicTokenString(tokenString: string): SplitToken | null; +/** + * Renders the recursive inner diff of two matched atomic tokens with equal keys but different + * content. The after version's opening and closing tags are emitted with the diff of the two + * inner HTML fragments in between. Inside the recursion the default atomic tags without 'a' + * are used, so link text is diffed word by word and href-only changes do not produce any + * markup. The after version's data-htmldiff-inner-diff-atomic-tags attribute overrides that + * list. Nested opted-in elements are diffed recursively as well, up to a hardcoded depth cap. + * @param beforeString The before version of the atomic token. + * @param afterString The after version of the atomic token. + * @param diffContent Diffs the two inner HTML fragments. + * @param dataPrefix (Optional) The prefix to use in data attributes. + * @param className (Optional) The class name to include in the wrapper tag. + * @returns The rendered element with inner differences wrapped in ins/del tags. + */ +export declare function renderInnerDiff(beforeString: string, afterString: string, diffContent: ContentDiff, dataPrefix?: string | null, className?: string | null): string; +/** + * Renders a list of operations into HTML content. The result is the combined version of the + * before and after tokens with the differences wrapped in tags. + * @param beforeTokens The before list of tokens. + * @param afterTokens The after list of tokens. + * @param operations The list of operations to transform the before tokens into the after tokens. + * @param diffContent Diffs the content of opted-in atomic elements recursively. + * @param dataPrefix (Optional) The prefix to use in data attributes. + * @param className (Optional) The class name to include in the wrapper tag. + * @returns The rendering of the list of operations. + */ +export declare function renderOperations(beforeTokens: Token[], afterTokens: Token[], operations: Operation[], diffContent: ContentDiff, dataPrefix?: string | null, className?: string | null): string; +export {}; diff --git a/js/core/rendering.js b/js/core/rendering.js new file mode 100644 index 0000000..8d6365b --- /dev/null +++ b/js/core/rendering.js @@ -0,0 +1,315 @@ +"use strict"; +Object.defineProperty(exports, "__esModule", { value: true }); +exports.renderOperations = exports.renderInnerDiff = exports.splitAtomicTokenString = exports.findOpeningTagEnd = exports.isInnerDiffToken = exports.wrap = exports.TokenWrapper = void 0; +/** + * Rendering: writes the operations back out as HTML, wrapping inserted and deleted tokens in + * and tags and diffing the content of opted-in atomic elements recursively. + */ +const atomicTags_1 = require("./atomicTags"); +const tokens_1 = require("./tokens"); +// The number of currently active recursive inner diffs. Recursion is governed per element +// (each nesting level requires its own data-htmldiff-inner-diff attribute), so the depth is +// naturally bounded by the nesting of opted-in elements; the cap is only a backstop against +// pathologically deep documents. +let innerDiffDepth = 0; +const maxInnerDiffDepth = 10; +/** + * A TokenWrapper provides a utility for grouping segments of tokens based on whether they're + * wrappable or not. A tag is considered wrappable if it is closed within the given set of + * tokens. For example, given the following tokens: + * + * ['', 'this', ' ', 'is', ' ', 'a', ' ', '', 'test', '', '!'] + * + * The first '' is not considered wrappable since the tag is not fully contained within the + * array of tokens. The '', 'test', and '' would be a part of the same wrappable segment + * since the entire bold tag is within the set of tokens. + */ +class TokenWrapper { + /** + * @param tokens The tokens to group. + */ + constructor(tokens) { + this.tokens = tokens; + this.notes = tokens.reduce((data, token, index) => { + data.notes.push({ + isWrappable: (0, tokens_1.isWrappable)(token), + insertedTag: false, + }); + const tag = !(0, tokens_1.isVoidTag)(token) && (0, tokens_1.isTag)(token); + const lastEntry = data.tagStack[data.tagStack.length - 1]; + if (tag) { + if (lastEntry && "/" + lastEntry.tag === tag) { + data.notes[lastEntry.position].insertedTag = true; + data.tagStack.pop(); + } + else { + data.tagStack.push({ tag, position: index }); + } + } + return data; + }, { notes: [], tagStack: [] }).notes; + } + /** + * Wraps the contained tokens in tags based on output given by a map function. Each segment + * of tokens will be visited. A segment is a continuous run of either all wrappable tokens or + * unwrappable tokens. The given map function will be called with each segment of tokens and + * the resulting strings will be combined to form the wrapped HTML. + * @param mapFn Called with each segment; the result should be a string. + * @param tagFn Called with the opening tag of every tag inserted whole, to mark it. + * @returns The wrapped HTML. + */ + combine(mapFn, tagFn) { + const notes = this.notes; + const tokens = this.tokens.slice(); + const segments = tokens.reduce((data, token, index) => { + if (notes[index].insertedTag) { + tokens[index] = tagFn(tokens[index]); + } + if (data.status === null) { + data.status = notes[index].isWrappable; + } + const status = notes[index].isWrappable; + // Handling atomic tags wrapping independently + // each atomic tag is wrapped with their own ins/del tags + const isAtomic = !!(0, atomicTags_1.isStartOfAtomicTag)(token); + if (status !== data.status || (isAtomic && index > data.lastIndex) || data.lastWasAtomic) { + data.list.push({ + isWrappable: data.status, + tokens: tokens.slice(data.lastIndex, index), + }); + data.lastIndex = index; + data.status = status; + } + // tracking if the last token was an atomic tag + // if so then we break the segment and wrap them + data.lastWasAtomic = isAtomic; + if (index === tokens.length - 1) { + data.list.push({ + isWrappable: data.status, + tokens: tokens.slice(data.lastIndex, index + 1), + }); + } + return data; + }, { list: [], status: null, lastIndex: 0, lastWasAtomic: false }).list; + return segments.map(mapFn).join(""); + } +} +exports.TokenWrapper = TokenWrapper; +/** + * Wraps and concatenates a list of tokens with a tag. Does not wrap tag tokens, unless they + * are wrappable (i.e. void and atomic tags). + * @param tag The tag name of the wrapper tags. + * @param content The list of tokens to wrap. + * @param opIndex The index of the operation, written to the data attribute. + * @param dataPrefix (Optional) The prefix to use in data attributes. + * @param className (Optional) The class name to include in the wrapper tag. + * @returns The wrapped HTML. + */ +function wrap(tag, content, opIndex, dataPrefix, className) { + const wrapper = new TokenWrapper(content); + const prefix = dataPrefix ? dataPrefix + "-" : ""; + let attrs = ` data-${prefix}operation-index="${opIndex}"`; + if (className) { + attrs += ' class="' + className + '"'; + } + return wrapper.combine((segment) => { + if (segment.isWrappable) { + const val = segment.tokens.join(""); + if (val.trim()) { + return "<" + tag + attrs + ">" + val + ""; + } + } + else { + return segment.tokens.join(""); + } + return ""; + }, (openingTag) => { + let dataAttrs = ' data-diff-node="' + tag + '"'; + dataAttrs += ` data-${prefix}operation-index="${opIndex}"`; + return openingTag.replace(/>\s*$/, dataAttrs + "$&"); + }); +} +exports.wrap = wrap; +/** + * Checks whether a token is an atomic tag that opted into the recursive inner diff via the + * data-htmldiff-inner-diff attribute. A bare attribute or any value other than "false" counts + * as opted in. Opted-in elements nested inside other opted-in elements are diffed recursively + * as well, up to the depth cap; beyond it, opted-in tokens are rendered verbatim like any + * other atomic token. + * @param tokenString The token string to check. + * @returns True if the token should get a recursive inner diff. + */ +function isInnerDiffToken(tokenString) { + if (innerDiffDepth >= maxInnerDiffDepth || !(0, atomicTags_1.isStartOfAtomicTag)(tokenString)) { + return false; + } + const attr = atomicTags_1.dataHtmlDiffInnerDiffRegExp.exec(tokenString); + return !!attr && attr[1] !== "false"; +} +exports.isInnerDiffToken = isInnerDiffToken; +/** + * Finds the index of the '>' that ends the opening tag at the start of the given token + * string, skipping any '>' inside quoted attribute values (e.g. title="a > b"). + * @param tokenString The token string starting with an opening tag. + * @returns The index of the closing '>' of the opening tag, or -1 if there is none (e.g. an + * unterminated tag or an unbalanced attribute quote). + */ +function findOpeningTagEnd(tokenString) { + let quote = null; + for (let i = 0; i < tokenString.length; i++) { + const char = tokenString[i]; + // quote is closed + if (char === quote) { + quote = null; + continue; + } + // inside quote + if (quote) { + continue; + } + // quote start + if (char === '"' || char === "'") { + quote = char; + continue; + } + // not inside quote, check for tag end + if (char === ">") { + return i; + } + } + return -1; +} +exports.findOpeningTagEnd = findOpeningTagEnd; +/** + * Splits an atomic token string into its opening tag, inner HTML and closing tag. A token + * consisting of a single tag (a void or self-closing element) has an empty inner HTML and no + * closing tag. + * @param tokenString The atomic token string, e.g. '
content
'. + * @returns The parts, or null if the token cannot be split (e.g. an unterminated tag). + */ +function splitAtomicTokenString(tokenString) { + const openingTagEnd = findOpeningTagEnd(tokenString); + if (openingTagEnd === -1) { + return null; + } + if (openingTagEnd === tokenString.length - 1) { + // The token is a single tag (void or self-closing): the element has no content. + return { + openingTag: tokenString, + innerHtml: "", + closingTag: "", + }; + } + const closingTagStart = tokenString.lastIndexOf("<"); + if (closingTagStart <= openingTagEnd || tokenString[closingTagStart + 1] !== "/") { + return null; + } + return { + openingTag: tokenString.slice(0, openingTagEnd + 1), + innerHtml: tokenString.slice(openingTagEnd + 1, closingTagStart), + closingTag: tokenString.slice(closingTagStart), + }; +} +exports.splitAtomicTokenString = splitAtomicTokenString; +/** + * Renders the recursive inner diff of two matched atomic tokens with equal keys but different + * content. The after version's opening and closing tags are emitted with the diff of the two + * inner HTML fragments in between. Inside the recursion the default atomic tags without 'a' + * are used, so link text is diffed word by word and href-only changes do not produce any + * markup. The after version's data-htmldiff-inner-diff-atomic-tags attribute overrides that + * list. Nested opted-in elements are diffed recursively as well, up to a hardcoded depth cap. + * @param beforeString The before version of the atomic token. + * @param afterString The after version of the atomic token. + * @param diffContent Diffs the two inner HTML fragments. + * @param dataPrefix (Optional) The prefix to use in data attributes. + * @param className (Optional) The class name to include in the wrapper tag. + * @returns The rendered element with inner differences wrapped in ins/del tags. + */ +function renderInnerDiff(beforeString, afterString, diffContent, dataPrefix, className) { + const before = splitAtomicTokenString(beforeString); + const after = splitAtomicTokenString(afterString); + if (!before || !after) { + return afterString; + } + const atomicTagsOverride = atomicTags_1.dataHtmlDiffInnerDiffAtomicTagsRegExp.exec(afterString); + const outerAtomicTagsRegExp = (0, atomicTags_1.getAtomicTagsRegExp)(); + innerDiffDepth++; + let innerDiff; + // the depth and the atomic tags must be restored in case of an error, hence try/finally + try { + (0, atomicTags_1.setAtomicTagsRegExp)(atomicTags_1.defaultInnerDiffAtomicTagsRegExp); + if (atomicTagsOverride) { + (0, atomicTags_1.setAtomicTagsRegExp)(atomicTagsOverride[1] ? (0, atomicTags_1.buildAtomicTagsRegExp)(atomicTagsOverride[1]) : atomicTags_1.noAtomicTagsRegExp); + } + innerDiff = diffContent(before.innerHtml, after.innerHtml, className, dataPrefix); + } + finally { + innerDiffDepth--; + (0, atomicTags_1.setAtomicTagsRegExp)(outerAtomicTagsRegExp); + } + return after.openingTag + innerDiff + after.closingTag; +} +exports.renderInnerDiff = renderInnerDiff; +function renderEqual(op, beforeTokens, afterTokens, _opIndex, diffContent, dataPrefix, className) { + // Tokens in an equal operation pair up one to one between before and after. Equal keys do + // not guarantee equal strings (e.g. atomic tokens matched by data-htmldiff-id): elements + // that opted in via data-htmldiff-inner-diff get a recursive diff of their content, + // everything else renders the after version. + let result = ""; + for (let i = 0; op.startInAfter + i <= op.endInAfter; i++) { + const afterToken = afterTokens[op.startInAfter + i]; + const beforeToken = beforeTokens[op.startInBefore + i]; + if (beforeToken && beforeToken.string !== afterToken.string && isInnerDiffToken(afterToken.string)) { + result += renderInnerDiff(beforeToken.string, afterToken.string, diffContent, dataPrefix, className); + } + else { + result += afterToken.string; + } + } + return result; +} +function renderInsert(op, _beforeTokens, afterTokens, opIndex, _diffContent, dataPrefix, className) { + const tokens = afterTokens.slice(op.startInAfter, op.endInAfter + 1); + const val = tokens.map((token) => token.string); + const res = wrap("ins", val, opIndex, dataPrefix, className); + // handling inserted tags, see https://matrixreq.atlassian.net/browse/MATRIX-7876 + if (/^<[^./]+?>$/.exec(res)) { + return `${res.slice(0, res.length - 1)} data-inserted="true">`; + } + return res; +} +function renderDelete(op, beforeTokens, _afterTokens, opIndex, _diffContent, dataPrefix, className) { + const tokens = beforeTokens.slice(op.startInBefore, op.endInBefore + 1); + const val = tokens.map((token) => token.string); + const res = wrap("del", val, opIndex, dataPrefix, className); + // handling cases like deleted

, see https://matrixreq.atlassian.net/browse/MATRIX-7688 + if (/^<\/.+?><.+?>$/.exec(res) && !res.includes("del")) { + return `${val.slice(1, val.length - 1).join("")}`; + } + return res; +} +function renderReplace(op, beforeTokens, afterTokens, opIndex, diffContent, dataPrefix, className) { + return (renderDelete(op, beforeTokens, afterTokens, opIndex, diffContent, dataPrefix, className) + + renderInsert(op, beforeTokens, afterTokens, opIndex, diffContent, dataPrefix, className)); +} +const OPS = { + equal: renderEqual, + insert: renderInsert, + delete: renderDelete, + replace: renderReplace, +}; +/** + * Renders a list of operations into HTML content. The result is the combined version of the + * before and after tokens with the differences wrapped in tags. + * @param beforeTokens The before list of tokens. + * @param afterTokens The after list of tokens. + * @param operations The list of operations to transform the before tokens into the after tokens. + * @param diffContent Diffs the content of opted-in atomic elements recursively. + * @param dataPrefix (Optional) The prefix to use in data attributes. + * @param className (Optional) The class name to include in the wrapper tag. + * @returns The rendering of the list of operations. + */ +function renderOperations(beforeTokens, afterTokens, operations, diffContent, dataPrefix, className) { + return operations.reduce((rendering, op, index) => rendering + OPS[op.action](op, beforeTokens, afterTokens, index, diffContent, dataPrefix, className), ""); +} +exports.renderOperations = renderOperations; diff --git a/js/core/tokens.d.ts b/js/core/tokens.d.ts new file mode 100644 index 0000000..5b783bc --- /dev/null +++ b/js/core/tokens.d.ts @@ -0,0 +1,54 @@ +/** A token holds the text to render and the key it is compared by. */ +export interface Token { + string: string; + key: string; +} +/** + * Determines if the given token is a tag. + * @param token The token in question. + * @returns False if the token is not a tag, or the tag name otherwise. + */ +export declare function isTag(token: string): string | false; +/** + * Checks if a tag is a void tag, written with the XML style '/>'. + * @param token The token to check. + * @returns True if the token is a void tag, false otherwise. + */ +export declare function isVoidTag(token: string): boolean; +/** + * Checks if a tag name is an HTML void element. Void elements cannot have content and can + * skip a closing tag, so an atomic element with a void tag name ends with its opening tag, + * with or without the XML style '/>'. + * @param tag The tag name to check. + * @returns True if the tag name is a void element. + */ +export declare function isVoidTagName(tag: string): boolean; +/** + * Checks if a token can be wrapped inside a tag: text, images, void tags and atomic tags + * can, other tags cannot. + * @param token The token to check. + * @returns True if the token can be wrapped inside a tag, false otherwise. + */ +export declare function isWrappable(token: string): boolean; +/** + * Creates a token that holds a string and key representation. The key is used for diffing + * comparisons and the string is used to recompose the document after the diff is complete. + * @param currentWord The section of the document to create a token for. + * @returns A token object with a string and key property. + */ +export declare function createToken(currentWord: string): Token; +/** + * Creates a key that should be used to match tokens. This is useful, for example, if we want + * to consider two open tag tokens as equal, even if they don't have the same attributes. We + * use a key instead of overwriting the token because we may want to render the original + * string without losing the attributes. + * @param token The token to create the key for. + * @returns The identifying key that should be used to match before and after tokens. + */ +export declare function getKeyForToken(token: string): string; +/** + * Tokenizes a string of HTML. + * @param html The string to tokenize. + * @returns The list of tokens. + */ +export declare function htmlToTokens(html: string): Token[]; diff --git a/js/core/tokens.js b/js/core/tokens.js new file mode 100644 index 0000000..c48e06d --- /dev/null +++ b/js/core/tokens.js @@ -0,0 +1,363 @@ +"use strict"; +Object.defineProperty(exports, "__esModule", { value: true }); +exports.htmlToTokens = exports.getKeyForToken = exports.createToken = exports.isWrappable = exports.isVoidTagName = exports.isVoidTag = exports.isTag = void 0; +/** + * Tokenizing: splits HTML into words, whitespace, tags and atomic elements, and gives every + * token the key it is compared by. + */ +const atomicTags_1 = require("./atomicTags"); +/** + * Determines if the given token is a tag. + * @param token The token in question. + * @returns False if the token is not a tag, or the tag name otherwise. + */ +function isTag(token) { + const match = token.match(/^\s*<([^!>][^>]*)>\s*$/); + return !!match && match[1].trim().split(" ")[0]; +} +exports.isTag = isTag; +/** + * Checks if a tag is a void tag, written with the XML style '/>'. + * @param token The token to check. + * @returns True if the token is a void tag, false otherwise. + */ +function isVoidTag(token) { + return /^\s*<[^>]+\/>\s*$/.test(token); +} +exports.isVoidTag = isVoidTag; +/** + * Checks if a tag name is an HTML void element. Void elements cannot have content and can + * skip a closing tag, so an atomic element with a void tag name ends with its opening tag, + * with or without the XML style '/>'. + * @param tag The tag name to check. + * @returns True if the tag name is a void element. + */ +function isVoidTagName(tag) { + return /^(area|base|br|col|embed|hr|img|input|link|meta|param|source|track|wbr)$/.test(tag); +} +exports.isVoidTagName = isVoidTagName; +/** + * Checks if a token can be wrapped inside a tag: text, images, void tags and atomic tags + * can, other tags cannot. + * @param token The token to check. + * @returns True if the token can be wrapped inside a tag, false otherwise. + */ +function isWrappable(token) { + const isImage = /^]/.test(token); + return isImage || !isTag(token) || !!(0, atomicTags_1.isStartOfAtomicTag)(token) || isVoidTag(token); +} +exports.isWrappable = isWrappable; +/** + * Creates a token that holds a string and key representation. The key is used for diffing + * comparisons and the string is used to recompose the document after the diff is complete. + * @param currentWord The section of the document to create a token for. + * @returns A token object with a string and key property. + */ +function createToken(currentWord) { + return { + string: currentWord, + key: getKeyForToken(currentWord), + }; +} +exports.createToken = createToken; +/** + * Creates a key that should be used to match tokens. This is useful, for example, if we want + * to consider two open tag tokens as equal, even if they don't have the same attributes. We + * use a key instead of overwriting the token because we may want to render the original + * string without losing the attributes. + * @param token The token to create the key for. + * @returns The identifying key that should be used to match before and after tokens. + */ +function getKeyForToken(token) { + // If the token is an image element, grab it's src attribute to include in the key. + const img = /^$/.exec(token); + if (img) { + return ''; + } + // If the token is an a element, grab it's data attribute to include in the key. + // Only when is atomic: if it has been excluded from the atomic tags (as done in + // recursive inner diffs), the token is just the opening tag. + const a = /^'; + } + // If the token is an object element, grab it's data attribute to include in the key. + const object = /^'; + } + // If it's a video, math or svg element, the entire token should be compared except the + // data-uuid. + if (/^<(svg|math|video)[\s>]/.test(token)) { + const uuid = token.indexOf('data-uuid="'); + if (uuid !== -1) { + const start = token.slice(0, uuid); + const end = token.slice(uuid + 44); + return start + end; + } + return token; + } + // If the token is an iframe element, grab it's src attribute to include in it's key. + const iframe = /^/.exec(token); + if (iframe) { + return ''; + } + // if the token has data-htmldiff-uuid use it as a key + const uuidTag = atomicTags_1.dataHtmlDiffIdRegExp.exec(token); + if (uuidTag) { + return uuidTag[2]; + } + // If the token is any other element, just grab the tag name. + const tagName = /<([^\s>]+)[\s>]/.exec(token); + if (tagName) { + return "<" + tagName[1].toLowerCase() + ">"; + } + // Otherwise, the token is text, collapse the whitespace + // (except new lines, see https://matrixreq.atlassian.net/browse/MATRIX-7880) + // potentially, this also causing the problems with prettified HTML (with "\n" between the tags), + // so it's required to "flatten" html before passing it to the diffing function + if (token) { + return token.replace(/([^\S\r\n]+| | )/g, " "); + } + return token; +} +exports.getKeyForToken = getKeyForToken; +function isEndOfTag(char) { + return char === ">"; +} +function isStartOfTag(char) { + return char === "<"; +} +function isWhitespace(char) { + return /^\s+$/.test(char); +} +function isStartOfHtmlComment(word) { + return /^$/.test(word); +} +/** + * Inspects the last tag in the given string, its slice from the final '<'. A '>' before the + * slice's end means text follows e.g. "a > b" in a

')).eql(tokenize(["

", '', "

"])); + }); + + it("should identify tags with data-htmldiff-id attribute as single token", () => { + expect( + cut('
hello
goodbye
' + 'some stuff' + "
"), + ).eql( + tokenize([ + "
", + 'hello
goodbye
', + 'some stuff', + "
", + ]), + ); + }); + + describe("nested atomic tags wrapping", () => { + it("should keep a data-htmldiff-id wrapper with nested same-tag children as one token", () => { + const atomic = '' + 'AB' + "Name"; + expect(cut(`
${atomic}
`)).eql(tokenize(["
", atomic, "
"])); + }); + + it("should not close early on the first inner closing tag", () => { + const atomic = 'ab'; + expect(cut(atomic)).eql(tokenize([atomic])); + }); + + it('should not treat a stray ">" in script content as a tag boundary', () => { + const atomic = ""; + expect(cut(`

${atomic}

`)).eql(tokenize(["

", atomic, "

"])); + }); + + it("should ignore self-closing same-named children when counting depth", () => { + const atomic = 'xy'; + expect(cut(atomic)).eql(tokenize([atomic])); + }); + + it("should ignore self-closing same-named children written with a space ()", () => { + const atomic = 'xy'; + expect(cut(atomic)).eql(tokenize([atomic])); + }); + + it("should key a wrapper by its own data-htmldiff-id, not a nested child one", () => { + const atomic = '' + 'C'; + expect(cut(atomic)[0].key).eql("s1"); + }); + + it("should not key an unkeyed atomic tag by a nested child data-htmldiff-id", () => { + const atomic = 'C'; + expect(cut(atomic)[0].key).eql(""); + }); + + it("should not bump depth on differently-named tags that share a prefix", () => { + // a tag must not be matched by an atomic tag named "a" appearing as
. + const atomic = '
hi
'; + expect(cut(atomic)).eql(tokenize([atomic])); + }); + }); + + describe("tags sharing a prefix with atomic tag names", () => { + it("should not treat as the atomic tag a", () => { + expect(cut("x tail")).eql(tokenize(["", "x", "", " ", "tail"])); + }); + + it("should not treat
as the atomic tag a", () => { + expect(cut("
hi
")).eql(tokenize(["
", "hi", "
"])); + }); + }); + + describe("self-closing atomic tags", () => { + it("should end a self-closing data-htmldiff-id tag without swallowing trailing content", () => { + expect(cut('
x old')).eql(tokenize(['
', "x", " ", "old"])); + }); + + it("should end a self-closing name-based atomic tag without swallowing trailing content", () => { + expect(cut("tail")).eql(tokenize(["", "tail"])); + }); + }); + + describe("quoted attribute values containing tag delimiters", () => { + it('should not end a tag on ">" inside a double-quoted attribute value', () => { + expect(cut('

text

')).eql(tokenize(['

', "text", "

"])); + }); + + it('should not end a tag on ">" inside a single-quoted attribute value', () => { + expect(cut("

x

")).eql(tokenize(["

", "x", "

"])); + }); + + it('should not end a void atomic tag on ">" inside an attribute value', () => { + expect(cut('a > b tail')).eql(tokenize(['a > b', " ", "tail"])); + }); + + it('should not treat "/>" inside an attribute value as self-closing', () => { + const atomic = 'c'; + expect(cut(atomic)).eql(tokenize([atomic])); + }); + + it('should keep an atomic tag with ">" in an attribute as one token', () => { + expect(cut('
x
tail')).eql( + tokenize(['
x
', " ", "tail"]), + ); + }); + + it("should not treat apostrophes in atomic text content as quotes", () => { + expect(cut("
it's ok
tail")).eql(tokenize(["
it's ok
", " ", "tail"])); + }); + + it("should not treat apostrophes in comments inside atomic tags as quotes", () => { + expect(cut("x tail")).eql(tokenize(["x", " ", "tail"])); + }); + }); + + describe("void atomic tags", () => { + it("should end a void data-htmldiff-id tag written without a slash", () => { + expect(cut(' tail')).eql(tokenize(['', " ", "tail"])); + }); + + it("should end a void data-htmldiff-id br tag without swallowing trailing content", () => { + expect(cut('
y')).eql(tokenize(['
', "y"])); + }); + }); + }); +}); diff --git a/test/core/inner_diff.spec.ts b/test/core/inner_diff.spec.ts new file mode 100644 index 0000000..1959f45 --- /dev/null +++ b/test/core/inner_diff.spec.ts @@ -0,0 +1,303 @@ +import { expect } from "chai"; +import diff from "../../src/htmldiff"; + +describe("Recursive inner diff (data-htmldiff-inner-diff)", () => { + const cut = diff; + + function tocEntry(href: string, name: string): string { + return `
${name}
`; + } + + describe("when an opted-in element is renamed in place", () => { + it("renders inline ins/del inside the single emitted element", () => { + const res = cut(tocEntry("#123", "1. Old name"), tocEntry("#123", "1. New name")); + expect(res).to.equal( + '
' + + '1. Old' + + 'New name
', + ); + }); + + it("passes className and dataPrefix through to the inner ins/del tags", () => { + const res = cut(tocEntry("#123", "Old"), tocEntry("#123", "New"), "diff-cls", "pre"); + expect(res).to.equal( + '
' + + 'Old' + + 'New
', + ); + }); + + it("keeps diffing the rest of the document with the outer atomic tag list", () => { + // The outside the opted-in element must stay atomic after the inner diff ran. + const res = cut(tocEntry("#123", "Old name") + 'same link', tocEntry("#123", "New name") + 'same link'); + expect(res).to.equal( + '
' + + 'Old' + + 'New name
' + + 'same link' + + 'same link', + ); + }); + }); + + describe("anchors inside the recursive diff", () => { + it("do not treat as atomic, so href-only changes produce no diff markup", () => { + const res = cut(tocEntry("#18036", "1. Same name"), tocEntry("#18037", "1. Same name")); + expect(res).to.equal(tocEntry("#18037", "1. Same name")); + }); + + it("diffs the link text inline when the href changed", () => { + const res = cut(tocEntry("#18036", "1. Old name"), tocEntry("#18037", "1. New name")); + expect(res).to.equal( + '
' + + '1. Old' + + 'New name
', + ); + }); + }); + + describe("atomic tags inside the recursive diff", () => { + function el(extraAttrs: string, inner: string): string { + return `
${inner}
`; + } + + it("keeps the default atomic tags (without a) inside the recursion", () => { + // Embedded content like svg stays atomic by default: a changed svg is replaced + // as a whole, not word-diffed. + const res = cut(el("", 'old'), el("", 'new')); + expect(res).to.equal( + '
' + + '' + + 'old' + + '' + + 'new' + + "
", + ); + }); + + it("replaces the default list with data-htmldiff-inner-diff-atomic-tags", () => { + // The override lists only em, so svg is no longer atomic and gets word-diffed. + const attrs = ' data-htmldiff-inner-diff-atomic-tags="em"'; + const res = cut(el(attrs, 'old'), el(attrs, 'new')); + expect(res).to.equal( + `
` + + '' + + 'old' + + 'new' + + "
", + ); + }); + + it("restores atomic anchors (href comparison) when the override lists a", () => { + const attrs = ' data-htmldiff-inner-diff-atomic-tags="a"'; + const res = cut(el(attrs, 'Name'), el(attrs, 'Name')); + expect(res).to.equal( + `
` + + 'Name' + + 'Name' + + "
", + ); + }); + + it("treats no tag name as atomic when the override value is empty", () => { + const attrs = ' data-htmldiff-inner-diff-atomic-tags=""'; + const res = cut(el(attrs, 'old'), el(attrs, 'new')); + expect(res).to.equal( + `
` + + '' + + 'old' + + 'new' + + "
", + ); + }); + }); + + describe("opt-in attribute values", () => { + function entry(value: string, name: string): string { + return `
${name}
`; + } + + it('treats a double-quoted "false" value as opted out', () => { + const res = cut(entry('"false"', "old"), entry('"false"', "new")); + expect(res).to.equal(entry('"false"', "new")); + }); + + it("treats a single-quoted 'false' value as opted out", () => { + const res = cut(entry("'false'", "old"), entry("'false'", "new")); + expect(res).to.equal(entry("'false'", "new")); + }); + + it("treats an unquoted false value as opted out", () => { + const res = cut(entry("false", "old"), entry("false", "new")); + expect(res).to.equal(entry("false", "new")); + }); + + it("treats a bare attribute without a value as opted in", () => { + const res = cut('
old
', '
new
'); + expect(res).to.equal( + '
' + 'old' + 'new
', + ); + }); + + it("only matches exact attribute name", () => { + const res = cut( + '
old
', + '
new
', + ); + expect(res).to.equal('
new
'); + }); + + it("recognizes the attribute regardless of its position", () => { + const res = cut( + '
old
', + '
new
', + ); + expect(res).to.equal( + '
' + + 'old' + + 'new
', + ); + }); + }); + + describe("edge cases", () => { + it("ignores the attribute on non-atomic elements (no data-htmldiff-id)", () => { + // Without data-htmldiff-id the div is not atomic, so this is a plain word diff. + const res = cut('
old text
', '
new text
'); + expect(res).to.equal( + '
' + 'old' + 'new text
', + ); + }); + + it("diffs text following a self-closing opted-in element normally", () => { + const res = cut( + '
x old', + '
x new', + ); + expect(res).to.equal( + '
x ' + + 'old' + + 'new', + ); + }); + + it("marks the content as inserted when the before element was self-closing", () => { + // A self-closing element has empty inner content, so the new content is a + // pure insertion. + const res = cut('
', '
new
'); + expect(res).to.equal('
' + 'new
'); + }); + + it("marks the content as deleted when the after element became self-closing", () => { + // The after element has no content anymore, so the deleted content is + // rendered right after the self-closing tag. + const res = cut('
old
', '
'); + expect(res).to.equal('
' + 'old'); + }); + + it('is not confused by ">" inside attribute values', () => { + const res = cut( + '
old
', + '
new
', + ); + expect(res).to.equal( + '
' + + 'old' + + 'new
', + ); + }); + + it('is not confused by ">" inside single-quoted attribute values', () => { + const res = cut( + "
old
", + "
new
", + ); + expect(res).to.equal( + "
" + + 'old' + + 'new
', + ); + }); + + it("falls back to the after version when a token cannot be split", () => { + // An unterminated atomic tag swallows the rest of the input and has no closing + // tag to split on; the inner diff falls back instead of producing broken markup. + const after = '
new'; + const res = cut('
old', after); + expect(res).to.equal(after); + }); + + it("renders pure insertions when the before content is empty", () => { + const res = cut(tocEntry("#1", ""), tocEntry("#1", "New name")); + expect(res).to.equal( + '
' + + 'New name
', + ); + }); + + it("do not diffs inner html for moved opted-in elements", () => { + function entry(id: string, name: string): string { + return `
${name}
`; + } + const res = cut(entry("a", "First") + entry("b", "Second"), entry("b", "Second") + entry("a", "First")); + expect(res).to.equal( + `${entry("b", "Second")}` + entry("a", "First") + `${entry("b", "Second")}`, + ); + }); + }); + + describe("when the element did not opt in", () => { + it("should render the after version as is when keys match but content differs", () => { + const before = '
'; + const after = '
'; + expect(cut(before, after)).to.equal(after); + }); + }); + + describe("when key and content are both equal", () => { + it("should render the element unchanged", () => { + const before = `x ${tocEntry("#123", "1. Same name")} y`; + const after = `x ${tocEntry("#123", "1. Same name")} z`; + expect(cut(before, after)).to.equal(`x ${tocEntry("#123", "1. Same name")} ` + 'yz'); + }); + }); + + describe("when opted-in elements are nested", () => { + function nest(depth: number, content: string): string { + let html = content; + for (let i = depth; i >= 1; i--) { + html = `
${html}
`; + } + return html; + } + + it("diffs nested elements", () => { + const res = cut( + '
x ' + 'old
', + '
x ' + 'new
', + ); + expect(res).to.equal( + '
x ' + + '' + + 'old' + + 'new
', + ); + }); + + it("do not diffs nested element that do not opt-in", () => { + const after = '
t ' + 'new
'; + const res = cut('
t ' + 'old
', after); + expect(res).to.equal(after); + }); + + it("diffs all levels up to the depth cap of 10", () => { + const res = cut(nest(10, "old"), nest(10, "new")); + expect(res).to.equal(nest(10, 'old' + 'new')); + }); + + it("renders the after version as is beyond the depth cap", () => { + const res = cut(nest(11, "old"), nest(11, "new")); + expect(res).to.equal(nest(11, "new")); + }); + }); +}); diff --git a/test/core/render_operations.spec.ts b/test/core/render_operations.spec.ts new file mode 100644 index 0000000..01e0c92 --- /dev/null +++ b/test/core/render_operations.spec.ts @@ -0,0 +1,146 @@ +import { expect } from "chai"; +import { renderOperations } from "../../src/core/diff"; +import { calculateOperations } from "../../src/core/operations"; +import { createToken, Token } from "../../src/core/tokens"; + +describe("renderOperations", () => { + const tokenize = (tokens: string[]): Token[] => tokens.map((token) => createToken(token)); + const cut = (before: Token[], after: Token[]): string => renderOperations(before, after, calculateOperations(before, after)); + let res: string; + + it("should be a function", () => { + expect(cut).is.a("function"); + }); + + describe("equal", () => { + beforeEach(() => { + const before = tokenize(["this", " ", "is", " ", "a", " ", "test"]); + res = cut(before, before); + }); + + it("should output the text", () => { + expect(res).equal("this is a test"); + }); + }); + + describe("insert", () => { + beforeEach(() => { + res = cut(tokenize(["this", " ", "is"]), tokenize(["this", " ", "is", " ", "a", " ", "test"])); + }); + + it("should wrap in an ", () => { + expect(res).equal('this is a test'); + }); + }); + + describe("delete", () => { + beforeEach(() => { + res = cut(tokenize(["this", " ", "is", " ", "a", " ", "test", " ", "of", " ", "stuff"]), tokenize(["this", " ", "is", " ", "a", " ", "test"])); + }); + + it("should wrap in a ", () => { + expect(res).to.equal('this is a test of stuff'); + }); + }); + + describe("replace", () => { + beforeEach(() => { + res = cut(tokenize(["this", " ", "is", " ", "a", " ", "break"]), tokenize(["this", " ", "is", " ", "a", " ", "test"])); + }); + + it("should wrap in both and ", () => { + expect(res).to.equal('this is a break' + 'test'); + }); + }); + + describe("Dealing with tags", () => { + let before: Token[]; + let after: Token[]; + + beforeEach(() => { + before = tokenize(["

", "a", "

"]); + after = tokenize(["

", "a", " ", "b", "

", "

", "c", "

"]); + res = cut(before, after); + }); + + it("should identify contained inserted tags", () => { + expect(res).to.equal( + '

a b

' + '

' + 'c

', + ); + }); + + it("should identify contained deleted tags", () => { + res = cut(after, before); + + expect(res).to.equal( + '

a b

' + '

' + 'c

', + ); + }); + + it("should not identify partial tags", () => { + res = cut(tokenize(["test", "", "non-bold"]), tokenize(["test!", "", "non-bold", "", "bold"])); + + expect(res).to.equal( + 'test' + 'test!non-bold' + 'bold', + ); + }); + + describe("When there is a change at the beginning, in a

", () => { + beforeEach(() => { + res = cut(tokenize(["

", "this", " ", "is", " ", "awesome", "

"]), tokenize(["

", "I", " ", "is", " ", "awesome", "

"])); + }); + + it("should keep the change inside the

", () => { + expect(res).to.equal('

this' + 'I is awesome

'); + }); + }); + }); + + describe("empty tokens", () => { + it("should not be wrapped", () => { + res = cut(tokenize(["text"]), tokenize(["text", " "])); + + expect(res).to.equal("text"); + }); + }); + + describe("tags with attributes", () => { + it("should treat attribute changes as equal and output the after tag", () => { + res = cut( + tokenize(["

", "this", " ", "is", " ", "awesome", "

"]), + tokenize(['

', "this", " ", "is", " ", "awesome", "

"]), + ); + + expect(res).to.equal('

this is awesome

'); + }); + + it("should show changes within tags with different attributes", () => { + res = cut( + tokenize(["

", "this", " ", "is", " ", "awesome", "

"]), + tokenize(['

', "that", " ", "is", " ", "awesome", "

"]), + ); + + expect(res).to.equal( + '

' + 'this' + "that is awesome

", + ); + }); + }); + + describe("wrappable tags", () => { + it("should wrap void tags", () => { + res = cut(tokenize(["old", " ", "text"]), tokenize(["new", "
", " ", "text"])); + + expect(res).to.equal('old' + 'new
text'); + }); + + it("should wrap atomic tags independently", () => { + res = cut(tokenize(["old", '', " ", "text"]), tokenize(["new", " ", "text"])); + + expect(res).to.equal( + 'old' + + '' + + 'new text', + ); + }); + }); +}); diff --git a/test/core/rendering.spec.ts b/test/core/rendering.spec.ts new file mode 100644 index 0000000..2244747 --- /dev/null +++ b/test/core/rendering.spec.ts @@ -0,0 +1,80 @@ +import { expect } from "chai"; +import { findOpeningTagEnd, isInnerDiffToken, renderInnerDiff, splitAtomicTokenString, wrap } from "../../src/core/rendering"; + +describe("rendering", () => { + describe("findOpeningTagEnd", () => { + it("finds the end of a plain opening tag", () => { + expect(findOpeningTagEnd('
text
')).to.equal(14); + }); + + it("skips a > inside a quoted attribute", () => { + expect(findOpeningTagEnd('
x
')).to.equal(18); + }); + + it("returns -1 for an unterminated tag", () => { + expect(findOpeningTagEnd('
{ + it("splits an element into its tags and content", () => { + expect(splitAtomicTokenString('
content
')).to.deep.equal({ openingTag: '
', innerHtml: "content", closingTag: "
" }); + }); + + it("splits a self-closing element into its tag alone", () => { + expect(splitAtomicTokenString('')).to.deep.equal({ openingTag: '', innerHtml: "", closingTag: "" }); + }); + + it("returns null when no closing tag follows the content", () => { + expect(splitAtomicTokenString("
content")).to.equal(null); + }); + }); + + describe("isInnerDiffToken", () => { + it("accepts an atomic element that opted in", () => { + expect(isInnerDiffToken('
x
')).to.equal(true); + }); + + it("accepts a bare opt-in attribute", () => { + expect(isInnerDiffToken('
x
')).to.equal(true); + }); + + it("rejects an opt-out", () => { + expect(isInnerDiffToken('
x
')).to.equal(false); + }); + + it("rejects an element that is not atomic", () => { + expect(isInnerDiffToken('
x
')).to.equal(false); + }); + }); + + describe("wrap", () => { + it("wraps text in the tag with the operation index", () => { + expect(wrap("ins", ["hello", " ", "world"], 3)).to.equal('hello world'); + }); + + it("adds the prefix and the class", () => { + expect(wrap("del", ["x"], 0, "pre", "cls")).to.equal('x'); + }); + + it("marks a tag inserted whole instead of wrapping it", () => { + expect(wrap("ins", ["", "x", ""], 1)).to.equal('x'); + }); + }); + + describe("renderInnerDiff", () => { + it("diffs the content with the injected diff and keeps the after tags", () => { + const calls: [string, string][] = []; + const result = renderInnerDiff('
old
', '
new
', (before, after) => { + calls.push([before, after]); + return `[${before}|${after}]`; + }); + expect(calls).to.deep.equal([["old", "new"]]); + expect(result).to.equal('
[old|new]
'); + }); + + it("renders the after token when a side cannot be split", () => { + expect(renderInnerDiff("
old", "
new
", () => "x")).to.equal("
new
"); + }); + }); +}); diff --git a/test/core/tokens.spec.ts b/test/core/tokens.spec.ts new file mode 100644 index 0000000..370e682 --- /dev/null +++ b/test/core/tokens.spec.ts @@ -0,0 +1,37 @@ +import { expect } from "chai"; +import { createToken, isTag, isVoidTag, isVoidTagName, isWrappable } from "../../src/core/tokens"; + +describe("tokens", () => { + describe("isTag", () => { + it("names a tag token and rejects text", () => { + expect(isTag('

')).to.equal("p"); + expect(isTag("

")).to.equal("/p"); + expect(isTag("text")).to.equal(false); + }); + }); + + describe("void tags", () => { + it("recognises self-closing tags and void tag names", () => { + expect(isVoidTag("
")).to.equal(true); + expect(isVoidTag("
")).to.equal(false); + expect(isVoidTagName("img")).to.equal(true); + expect(isVoidTagName("div")).to.equal(false); + }); + }); + + describe("isWrappable", () => { + it("wraps text, images, void and atomic tags, not other tags", () => { + expect(isWrappable("word")).to.equal(true); + expect(isWrappable('')).to.equal(true); + expect(isWrappable("
")).to.equal(true); + expect(isWrappable("")).to.equal(true); + expect(isWrappable("

")).to.equal(false); + }); + }); + + describe("createToken", () => { + it("holds the string and its key", () => { + expect(createToken('

')).to.deep.equal({ string: '

', key: "

" }); + }); + }); +}); diff --git a/test/diff.spec.js b/test/diff.spec.js deleted file mode 100644 index 2a8ec3d..0000000 --- a/test/diff.spec.js +++ /dev/null @@ -1,318 +0,0 @@ -describe('Diff', function(){ - var cut, res, html_to_tokens, calculate_operations; - - beforeEach(function(){ - cut = require('../js/htmldiff'); - html_to_tokens = cut.htmlToTokens; - calculate_operations = cut.calculateOperations; - }); - - describe('When both inputs are the same', function(){ - beforeEach(function(){ - res = cut('input text', 'input text'); - }); - - it('should return the text', function(){ - expect(res).equal('input text'); - }); - }); // describe('When both inputs are the same') - - describe('When a letter is added', function(){ - beforeEach(function(){ - res = cut('input', 'input 2'); - }); - - it('should mark the new letter', function(){ - expect(res).to.equal('input 2'); - }); - }); // describe('When a letter is added') - - describe('Whitespace differences', function(){ - it('should collapse adjacent whitespace', function(){ - expect(cut('Much \t spaces', 'Much spaces')).to.equal('Much spaces'); - }); - - it('should consider non-breaking spaces as equal', function(){ - expect(cut('Hello world', 'Hello world')).to.equal('Hello world'); - }); - - it('should consider non-breaking spaces and non-adjacent regular spaces as equal', function(){ - expect(cut('Hello world', 'Hello world')).to.equal('Hello world'); - }); - }); // describe('Whitespace differences') - - describe('When a class name is specified', function(){ - it('should include the class in the wrapper tags', function(){ - expect(cut('input', 'input 2', 'diff-result')).to.equal( - 'input 2'); - }); - }); // describe('When a class name is specified') - - describe('Image Differences', function(){ - it('show two images as different if their src attributes are different', function() { - var before = html_to_tokens(''); - var after = html_to_tokens(''); - var ops = calculate_operations(before, after); - expect(ops.length).to.equal(1); - expect(ops[0]).to.eql({ - action: 'replace', - startInBefore: 0, - endInBefore: 0, - startInAfter: 0, - endInAfter: 0 - }); - }); - - it('should show two images are the same if their src attributes are the same', function() { - var before = html_to_tokens(''); - var after = html_to_tokens('hey!'); - var ops = calculate_operations(before, after); - expect(ops.length).to.equal(1); - expect(ops[0]).to.eql({ - action: 'equal', - startInBefore: 0, - endInBefore: 0, - startInAfter: 0, - endInAfter: 0 - }); - }); - }); // describe('Image Differences') - - describe('Widget Differences', function(){ - it('show two widgets as different if their data attributes are different', function() { - var before = html_to_tokens(''); - var after = html_to_tokens(''); - var ops = calculate_operations(before, after); - expect(ops.length).to.equal(1); - expect(ops[0]).to.eql({ - action: 'replace', - startInBefore: 0, - endInBefore: 0, - startInAfter: 0, - endInAfter: 0 - }); - }); - - it('should show two widgets are the same if their data attributes are the same', function() { - var before = html_to_tokens('yo!'); - var after = html_to_tokens(''); - var ops = calculate_operations(before, after); - expect(ops.length).to.equal(1); - expect(ops[0]).to.eql({ - action: 'equal', - startInBefore: 0, - endInBefore: 0, - startInAfter: 0, - endInAfter: 0 - }); - }); - }); // describe('Widget Differences') - - describe('Math Differences', function(){ - it('should show two math elements as different if their contents are different', function() { - var before = html_to_tokens('' + - 'b2'); - var after = html_to_tokens('' + - 'b5'); - var ops = calculate_operations(before, after); - expect(ops.length).to.equal(1); - expect(ops[0]).to.eql({ - action: 'replace', - startInBefore: 0, - endInBefore: 0, - startInAfter: 0, - endInAfter: 0 - }); - }); - - it('should show two math elements as the same if their contents are the same', function() { - var before = html_to_tokens('' + - 'b2'); - var after = html_to_tokens('' + - 'b2'); - var ops = calculate_operations(before, after); - expect(ops.length).to.equal(1); - expect(ops[0]).to.eql({ - action: 'equal', - startInBefore: 0, - endInBefore: 0, - startInAfter: 0, - endInAfter: 0 - }); - }); - }); // describe('Math Differences') - - describe('Video Differences', function(){ - it('show two widgets as different if their data attributes are different', function() { - var before = html_to_tokens(''); - var after = html_to_tokens(''); - var ops = calculate_operations(before, after); - expect(ops.length).to.equal(1); - expect(ops[0]).to.eql({ - action: 'replace', - startInBefore: 0, - endInBefore: 0, - startInAfter: 0, - endInAfter: 0 - }); - - }); - - it('should show two widgets are the same if their data attributes are the same', function() { - var before = html_to_tokens(''); - var after = html_to_tokens(''); - var ops = calculate_operations(before, after); - expect(ops.length).to.equal(1); - expect(ops[0]).to.eql({ - action: 'equal', - startInBefore: 0, - endInBefore: 0, - startInAfter: 0, - endInAfter: 0 - }); - }); - }); // describe('Video Differences') - - describe('iframe Differences', function(){ - it('show two widgets as different if their data attributes are different', function() { - var before = html_to_tokens(''); - var after = html_to_tokens(''); - var ops = calculate_operations(before, after); - expect(ops.length).to.equal(1); - expect(ops[0]).to.eql({ - action: 'replace', - startInBefore: 0, - endInBefore: 0, - startInAfter: 0, - endInAfter: 0 - }); - }); - - it('should show two widgets are the same if their data attributes are the same', function() { - var before = html_to_tokens(''); - var after = html_to_tokens(''); - var ops = calculate_operations(before, after); - expect(ops.length).to.equal(1); - expect(ops[0]).to.eql({ - action: 'equal', - startInBefore: 0, - endInBefore: 0, - startInAfter: 0, - endInAfter: 0 - }); - }); - }); // describe('iframe Differences') - - describe('Adjacent atomic tag combining', function(){ - it('should wrap each inserted atomic tag independently when multiple are inserted', function(){ - var result = cut('', ''); - expect(result).to.equal( - '' + - '' - ); - }); - - it('should wrap each deleted atomic tag independently when multiple are deleted', function(){ - var result = cut('', ''); - expect(result).to.equal( - '' + - '' - ); - }); - - it('should not merge atomic tag with adjacent text in same ins/del', function(){ - var result = cut('hello world', 'helloworld'); - expect(result).to.equal('helloworld'); - }); - - it('should wrap inserted content inside the tag, not in a standalone ins', function(){ - var result = cut('', 'hello'); - expect(result).to.equal( - 'hello' - ); - }); - - it('should mark non-atomic container tags with data-diff-node rather than wrapping with ins', function(){ - var result = cut('', '

  • content
  • '); - expect(result).to.equal( - '
  • ' + - '
    ' + - 'content' + - '
    ' + - '
  • ' - ); - }); - - it('should keep adjacent inserted tags as separate segments', function(){ - var result = cut('', 'helloworld'); - expect(result).to.equal( - 'hello' + - 'world' - ); - }); - }); // describe('Adjacent atomic tag combining') - - describe('Atomic tag list override (atomicTags parameter)', function(){ - // Regression tests for the parameter regexp being built with a literal backspace - // ('\b' instead of '\\b'), which made every override match nothing. - it('treats a listed tag as atomic', function(){ - var res = cut('', '', - null, null, 'iframe'); - expect(res).to.equal( - '' + - ''); - }); - - it('replaces the default list, so an unlisted default tag loses atomicity', function(){ - var res = cut('', '', - null, null, 'p'); - expect(res).to.equal( - ''); - }); - - it('does not match tags that merely start with a listed name', function(){ - // 'if' must not make ', '', - null, null, 'if'); - expect(res).to.equal( - ''); - }); - }); // describe('Atomic tag list override (atomicTags parameter)') - - describe('Tags sharing a prefix with atomic tag names', function(){ - it('should diff content although a is an atomic tag', function(){ - expect(cut('old t', 'new t')).to.equal( - 'old' + - 'new t'); - }); - }); // describe('Tags sharing a prefix with atomic tag names') - - describe('Quoted attribute values containing ">"', function(){ - it('should diff the content of a tag with ">" in an attribute value', function(){ - expect(cut('

    old

    ', '

    new

    ')).to.equal( - '

    ' + - 'old' + - 'new

    '); - }); - }); // describe('Quoted attribute values containing ">"') - - describe('Void atomic elements', function(){ - it('should diff text following a void data-htmldiff-id element', function(){ - var res = cut(' old', - ' new'); - expect(res).to.equal( - ' ' + - 'old' + - 'new'); - }); - }); // describe('Void atomic elements') - - }); // describe('Diff') \ No newline at end of file diff --git a/test/edge_cases.spec.js b/test/edge_cases.spec.ts similarity index 53% rename from test/edge_cases.spec.js rename to test/edge_cases.spec.ts index 91e5c9a..b2924b5 100644 --- a/test/edge_cases.spec.js +++ b/test/edge_cases.spec.ts @@ -1,17 +1,19 @@ -const diff = require('../js/htmldiff'); +import { expect } from "chai"; +import diff from "../src/htmldiff"; -describe('Edge cases', () => { - it('removes deleted closing-opening tag pair', () => { - const before = `

    something

    `; - const after = `

    something

    `; - const res = `

    something

    `; +describe("Edge cases", () => { + it("removes deleted closing-opening tag pair", () => { + const before = '

    something

    '; + const after = '

    something

    '; + const res = '

    something

    '; expect(diff(before, after)).to.eql(res); }); - it('preserves new lines', () => { - const before = `
    this isain text control Donec ullamcorper tortor quis augue egestas ultricies. Aenean vehicula molestie ex. Praesent id dolor at mauris efficitur ultrices. Vestibulum rutrum sit amet odio quis gravida. Praesent id augue maximus, faucibus tellus in, mattis augue. Pellentesque ullamcorper, ante et vestibulum tempus, libero nisl consequat massa, vel aliquam massa nibh in sapie
    `; - const after = `
    this isain text control Donec ullamcorper tortor quis augue egestas ultricies. Aenean vehicula molestie ex. Praesent id do at mauris efficitur ultrices. Vestibulum rutrum sit amet odio quis gravida. Praesent id augue maximus, faucibus tellus in, mattisugue. Pellentesque ullamcorper, ante et tempus, libero nisl consequat massa, vel aliquam massa nibh + it("preserves new lines", () => { + const before = + "
    this isain text control Donec ullamcorper tortor quis augue egestas ultricies. Aenean vehicula molestie ex. Praesent id dolor at mauris efficitur ultrices. Vestibulum rutrum sit amet odio quis gravida. Praesent id augue maximus, faucibus tellus in, mattis augue. Pellentesque ullamcorper, ante et vestibulum tempus, libero nisl consequat massa, vel aliquam massa nibh in sapie
    "; + const after = `
    this isain text control Donec ullamcorper tortor quis augue egestas ultricies. Aenean vehicula molestie ex. Praesent id do at mauris efficitur ultrices. Vestibulum rutrum sit amet odio quis gravida. Praesent id augue maximus, faucibus tellus in, mattisugue. Pellentesque ullamcorper, ante et tempus, libero nisl consequat massa, vel aliquam massa nibh jndfoewnf[owe 1123123242 %^&*()(*&^%$#@#$%^&*() @@ -19,7 +21,7 @@ jndfoewnf[owe 1123123242 BNXOISAN'FOSER3154 65+ 5SSDC
    `; - const res = `
    this isain text control Donec ullamcorper tortor quis augue egestas ultricies. Aenean vehicula molestie ex. Praesent id dolordo at mauris efficitur ultrices. Vestibulum rutrum sit amet odio quis gravida. Praesent id augue maximus, faucibus tellus in, mattis augue.mattisugue. Pellentesque ullamcorper, ante et vestibulum tempus, libero nisl consequat massa, vel aliquam massa nibh in sapie + const res = `
    this isain text control Donec ullamcorper tortor quis augue egestas ultricies. Aenean vehicula molestie ex. Praesent id dolordo at mauris efficitur ultrices. Vestibulum rutrum sit amet odio quis gravida. Praesent id augue maximus, faucibus tellus in, mattis augue.mattisugue. Pellentesque ullamcorper, ante et vestibulum tempus, libero nisl consequat massa, vel aliquam massa nibh in sapie jndfoewnf[owe 1123123242 %^&*()(*&^%$#@#$%^&*() @@ -32,9 +34,9 @@ BNXOISAN'FOSER3154 }); it('adds data-inserted="true" to inserted tags', () => { - const before = `
    some content
    `; - const after = `
    some content
    `; - const res = `
    some content
    `; + const before = "
    some content
    "; + const after = "
    some content
    "; + const res = '
    some content
    '; expect(diff(before, after)).to.eql(res); }); diff --git a/test/find_matching_blocks.spec.js b/test/find_matching_blocks.spec.js deleted file mode 100644 index 1e060bb..0000000 --- a/test/find_matching_blocks.spec.js +++ /dev/null @@ -1,168 +0,0 @@ -describe('findMatchingBlocks', function(){ - var diff, cut, res, createToken, tokenize, createSegment, htmlToTokens; - - beforeEach(function(){ - diff = require('../js/htmldiff'); - createSegment = diff.findMatchingBlocks.createSegment; - htmlToTokens = diff.htmlToTokens; - createToken = diff.findMatchingBlocks.createToken; - tokenize = function(tokens){ - return tokens.map(function(token){ - return createToken(token); - }); - }; - }); - - describe('createMap', function(){ - beforeEach(function(){ - cut = diff.findMatchingBlocks.createMap; - }); - - it('should be a function', function(){ - expect(cut).is.a('function'); - }); - - describe('When the items exist in the search target', function(){ - beforeEach(function(){ - res = cut(tokenize(['a', 'apple', 'has', 'a', 'worm'])); - }); - - it('should find "a" twice', function(){ - expect(res['a'].length).to.equal(2); - }); - - it('should find "a" at 0', function(){ - expect(res['a'][0]).to.equal(0); - }); - - it('should find "a" at 3', function(){ - expect(res['a'][1]).to.equal(3); - }); - - it('should find "has" at 2', function(){ - expect(res['has'][0]).to.equal(2); - }); - }); - }); - - describe('findBestMatch', function(){ - var invoke; - - beforeEach(function(){ - cut = diff.findMatchingBlocks.findBestMatch; - invoke = function(before, after){ - var segment = createSegment(before, after, 0, 0); - - res = cut(segment); - }; - }); - - describe('When there is a match', function(){ - beforeEach(function(){ - var before = tokenize(['a', 'dog', 'bites']); - var after = tokenize(['a', 'dog', 'bites', 'a', 'man']); - invoke(before, after); - }); - - it('should match the match', function(){ - expect(res).to.exist; - expect(res.startInBefore).equal(0); - expect(res.startInAfter).equal(0); - expect(res.length).equal(3); - expect(res.endInBefore).equal(2); - expect(res.endInAfter).equal(2); - }); - - describe('When the match is surrounded', function(){ - beforeEach(function(){ - before = tokenize(['dog', 'bites']); - after = tokenize(['the', 'dog', 'bites', 'a', 'man']); - invoke(before, after); - }); - - it('should match with appropriate indexing', function(){ - expect(res).to.exist; - expect(res.startInBefore).to.equal(0); - expect(res.startInAfter).to.equal(1); - expect(res.endInBefore).to.equal(1); - expect(res.endInAfter).to.equal(2); - }); - }); - }); - - describe('When there is no match', function(){ - beforeEach(function(){ - var before = tokenize(['the', 'rat', 'sqeaks']); - var after = tokenize(['a', 'dog', 'bites', 'a', 'man']); - invoke(before, after); - }); - - it('should return nothing', function(){ - expect(res).to.not.exist; - }); - }); - }); - - describe('findMatchingBlocks', function(){ - var segment; - - beforeEach(function(){ - cut = diff.findMatchingBlocks; - }); - - it('should be a function', function(){ - expect(cut).is.a('function'); - }); - - describe('When called with a single match', function(){ - beforeEach(function(){ - var before = htmlToTokens('a dog bites'); - var after = htmlToTokens('when a dog bites it hurts'); - segment = createSegment(before, after, 0, 0); - - res = cut(segment); - }); - - it('should return a match', function(){ - expect(res.length).to.equal(1); - }); - }); - - describe('When called with multiple matches', function(){ - beforeEach(function(){ - var before = htmlToTokens('the dog bit a man'); - var after = htmlToTokens('the large brown dog bit a tall man'); - segment = createSegment(before, after, 0, 0); - res = cut(segment); - }); - - it('should return 3 matches', function(){ - expect(res.length).to.equal(3); - }); - - it('should match "the"', function(){ - expect(res[0].startInBefore).eql(0); - expect(res[0].startInAfter).eql(0); - expect(res[0].endInBefore).eql(0); - expect(res[0].endInAfter).eql(0); - expect(res[0].length).eql(1); - }); - - it('should match "dog bit a"', function(){ - expect(res[1].startInBefore).eql(1); - expect(res[1].startInAfter).eql(5); - expect(res[1].endInBefore).eql(7); - expect(res[1].endInAfter).eql(11); - expect(res[1].length).eql(7); - }); - - it('should match "man"', function(){ - expect(res[2].startInBefore).eql(8); - expect(res[2].startInAfter).eql(14); - expect(res[2].endInBefore).eql(8); - expect(res[2].endInAfter).eql(14); - expect(res[2].length).eql(1); - }); - }); - }); -}); diff --git a/test/from_port_source.spec.js b/test/from_port_source.spec.js deleted file mode 100644 index 07ab90a..0000000 --- a/test/from_port_source.spec.js +++ /dev/null @@ -1,30 +0,0 @@ -describe('The specs from the ruby source project', function(){ - var cut; - - beforeEach(function(){ - cut = require('../js/htmldiff'); - }); - - it('should diff text', function(){ - var diff = cut('a word is here', 'a nother word is there'); - expect(diff).equal('a nother word is ' + - 'here' + - 'there'); - }); - - it('should insert a letter and a space', function(){ - var diff = cut('a c', 'a b c'); - expect(diff).equal('a b c'); - }); - - it('should remove a letter and a space', function(){ - var diff = cut('a b c', 'a c'); - diff.should == 'a b c'; - }); - - it('should change a letter', function(){ - var diff = cut('a b c', 'a d c'); - expect(diff).equal('a b' + - 'd c'); - }); -}); diff --git a/test/from_port_source.spec.ts b/test/from_port_source.spec.ts new file mode 100644 index 0000000..8219e36 --- /dev/null +++ b/test/from_port_source.spec.ts @@ -0,0 +1,26 @@ +import { expect } from "chai"; +import htmldiff from "../src/htmldiff"; + +describe("The specs from the ruby source project", () => { + const cut = htmldiff; + + it("should diff text", () => { + const diff = cut("a word is here", "a nother word is there"); + expect(diff).equal('a nother word is ' + 'here' + "there"); + }); + + it("should insert a letter and a space", () => { + const diff = cut("a c", "a b c"); + expect(diff).equal('a b c'); + }); + + it("should remove a letter and a space", () => { + const diff = cut("a b c", "a c"); + expect(diff).equal('a b c'); + }); + + it("should change a letter", () => { + const diff = cut("a b c", "a d c"); + expect(diff).equal('a b' + 'd c'); + }); +}); diff --git a/test/html_to_tokens.spec.js b/test/html_to_tokens.spec.js deleted file mode 100644 index 61875d3..0000000 --- a/test/html_to_tokens.spec.js +++ /dev/null @@ -1,249 +0,0 @@ -describe('htmlToTokens', function(){ - var cut, res, diff, createToken, tokenize; - - beforeEach(function(){ - diff = require('../js/htmldiff') - cut = diff.htmlToTokens; - - createToken = diff.findMatchingBlocks.createToken; - tokenize = function(tokens){ - return tokens.map(function(token){ - return createToken(token); - }); - }; - }); - - it('should be a function', function(){ - expect(cut).is.a('function'); - }); - - describe('when called with text', function(){ - beforeEach(function(){ - res = cut('this is a test'); - }); - - it('should return 4', function(){ - expect(res.length).to.equal(7); - }); - }); - - describe('when called with html', function(){ - beforeEach(function(){ - res = cut('

    this is a test

    '); - }); - - it('should return 11', function(){ - expect(res.length).to.equal(11); - }); - - it('should remove any html comments', function(){ - res = cut('

    this is

    '); - expect(res.length).to.equal(8); - }); - }); - - it('should identify contiguous whitespace as a single token', function(){ - expect(cut('a b')).to.eql(tokenize(['a', ' ', 'b'])); - }); - - it('should identify a single space as a single token', function(){ - expect(cut(' a b ')).to.eql(tokenize([' ', 'a', ' ', 'b', ' '])); - }); - - it('should identify self closing tags as tokens', function(){ - expect(cut('

    hello
    goodbye

    ')).eql( - tokenize(['

    ', 'hello', '
    ', 'goodbye', '

    '])); - }); - - describe('when encountering atomic tags', function(){ - it('should identify an image tag as a single token', function(){ - expect(cut('

    ')).eql( - tokenize(['

    ', '', '', '

    '])); - }); - - it('should identify an iframe tag as a single token', function(){ - expect(cut('

    ')).eql( - tokenize(['

    ', '', '

    '])); - }); - - it('should identify an object tag as a single token', function(){ - var cutResult = cut('

    '); - var tokenizeResult = tokenize( - ['

    ', '','

    ']); - expect(cutResult).eql(tokenizeResult); - }); - - it('should identify a math tag as a single token', function(){ - var cutResult = cut('

    ' + - 'π' + - '⁢' + - 'r2

    '); - var tokenizeResult = tokenize([ - '

    ', - '' + - 'π' + - '⁢' + - 'r2', - '

    ' - ]); - expect(cutResult).eql(tokenizeResult); - }); - - it('should identify an svg tag as a single token', function(){ - var cutResult = cut('

    ' + - '

    '); - var tokenizeResult = tokenize([ - '

    ', - '' + - '' + - '', - '

    ' - ]); - expect(cutResult).eql(tokenizeResult); - }); - - it('should identify a script tag as a single token', function(){ - expect(cut('

    ')).eql( - tokenize(['

    ', '', '

    '])); - }); - - - - it('should identify tags with data-htmldiff-id attribute as single token', () => { - expect( - cut('
    hello
    goodbye
    ' + - 'some stuff' + - '
    ') - ).eql(tokenize( - [ - '
    ', - 'hello
    goodbye
    ', - 'some stuff', - '
    ' - ] - )); - }); - - describe('nested atomic tags wrapping', function(){ - it('should keep a data-htmldiff-id wrapper with nested same-tag children as one token', function(){ - var atomic = '' + - 'AB' + - 'Name'; - expect(cut('
    ' + atomic + '
    ')).eql( - tokenize(['
    ', atomic, '
    '])); - }); - - it('should not close early on the first inner closing tag', function(){ - var atomic = 'ab'; - expect(cut(atomic)).eql(tokenize([atomic])); - }); - - it('should not treat a stray ">" in script content as a tag boundary', function(){ - var atomic = ''; - expect(cut('

    ' + atomic + '

    ')).eql( - tokenize(['

    ', atomic, '

    '])); - }); - - it('should ignore self-closing same-named children when counting depth', function(){ - var atomic = 'xy'; - expect(cut(atomic)).eql(tokenize([atomic])); - }); - - it('should ignore self-closing same-named children written with a space ()', function(){ - var atomic = 'xy'; - expect(cut(atomic)).eql(tokenize([atomic])); - }); - - it('should key a wrapper by its own data-htmldiff-id, not a nested child one', function(){ - var atomic = '' + - 'C'; - expect(cut(atomic)[0].key).eql('s1'); - }); - - it('should not key an unkeyed atomic tag by a nested child data-htmldiff-id', function(){ - var atomic = 'C'; - expect(cut(atomic)[0].key).eql(''); - }); - - it('should not bump depth on differently-named tags that share a prefix', function(){ - // a tag must not be matched by an atomic tag named "a" appearing as
    . - var atomic = '
    hi
    '; - expect(cut(atomic)).eql(tokenize([atomic])); - }); - }); - - describe('tags sharing a prefix with atomic tag names', function(){ - it('should not treat as the atomic tag a', function(){ - expect(cut('x tail')).eql( - tokenize(['', 'x', '', ' ', 'tail'])); - }); - - it('should not treat
    as the atomic tag a', function(){ - expect(cut('
    hi
    ')).eql( - tokenize(['
    ', 'hi', '
    '])); - }); - }); - - describe('self-closing atomic tags', function(){ - it('should end a self-closing data-htmldiff-id tag without swallowing trailing content', function(){ - expect(cut('
    x old')).eql( - tokenize(['
    ', 'x', ' ', 'old'])); - }); - - it('should end a self-closing name-based atomic tag without swallowing trailing content', function(){ - expect(cut('tail')).eql(tokenize(['', 'tail'])); - }); - }); - - describe('quoted attribute values containing tag delimiters', function(){ - it('should not end a tag on ">" inside a double-quoted attribute value', function(){ - expect(cut('

    text

    ')).eql( - tokenize(['

    ', 'text', '

    '])); - }); - - it('should not end a tag on ">" inside a single-quoted attribute value', function(){ - expect(cut("

    x

    ")).eql( - tokenize(["

    ", 'x', '

    '])); - }); - - it('should not end a void atomic tag on ">" inside an attribute value', function(){ - expect(cut('a > b tail')).eql( - tokenize(['a > b', ' ', 'tail'])); - }); - - it('should not treat "/>" inside an attribute value as self-closing', function(){ - var atomic = 'c'; - expect(cut(atomic)).eql(tokenize([atomic])); - }); - - it('should keep an atomic tag with ">" in an attribute as one token', function(){ - expect(cut('
    x
    tail')).eql( - tokenize(['
    x
    ', - ' ', 'tail'])); - }); - - it('should not treat apostrophes in atomic text content as quotes', function(){ - expect(cut("
    it's ok
    tail")).eql( - tokenize(["
    it's ok
    ", ' ', 'tail'])); - }); - - it('should not treat apostrophes in comments inside atomic tags as quotes', function(){ - expect(cut("x tail")).eql( - tokenize(["x", ' ', 'tail'])); - }); - }); - - describe('void atomic tags', function(){ - it('should end a void data-htmldiff-id tag written without a slash', function(){ - expect(cut(' tail')).eql( - tokenize(['', ' ', 'tail'])); - }); - - it('should end a void data-htmldiff-id br tag without swallowing trailing content', function(){ - expect(cut('
    y')).eql( - tokenize(['
    ', 'y'])); - }); - }); - }); -}); diff --git a/test/inner_diff.spec.js b/test/inner_diff.spec.js deleted file mode 100644 index 36a530c..0000000 --- a/test/inner_diff.spec.js +++ /dev/null @@ -1,333 +0,0 @@ -describe('Recursive inner diff (data-htmldiff-inner-diff)', function(){ - var cut; - - beforeEach(function(){ - cut = require('../js/htmldiff'); - }); - - function tocEntry(href, name){ - return '
    ' + - '' + name + '
    '; - } - - describe('when an opted-in element is renamed in place', function(){ - it('renders inline ins/del inside the single emitted element', function(){ - var res = cut(tocEntry('#123', '1. Old name'), tocEntry('#123', '1. New name')); - expect(res).to.equal( - '
    ' + - '1. Old' + - 'New name
    '); - }); - - it('passes className and dataPrefix through to the inner ins/del tags', function(){ - var res = cut(tocEntry('#123', 'Old'), tocEntry('#123', 'New'), 'diff-cls', 'pre'); - expect(res).to.equal( - '
    ' + - 'Old' + - 'New
    '); - }); - - it('keeps diffing the rest of the document with the outer atomic tag list', function(){ - // The outside the opted-in element must stay atomic after the inner diff ran. - var res = cut( - tocEntry('#123', 'Old name') + 'same link', - tocEntry('#123', 'New name') + 'same link'); - expect(res).to.equal( - '
    ' + - 'Old' + - 'New name
    ' + - 'same link' + - 'same link'); - }); - }); // describe('when an opted-in element is renamed in place') - - describe('anchors inside the recursive diff', function(){ - it('do not treat as atomic, so href-only changes produce no diff markup', function(){ - var res = cut(tocEntry('#18036', '1. Same name'), tocEntry('#18037', '1. Same name')); - expect(res).to.equal(tocEntry('#18037', '1. Same name')); - }); - - it('diffs the link text inline when the href changed', function(){ - var res = cut(tocEntry('#18036', '1. Old name'), tocEntry('#18037', '1. New name')); - expect(res).to.equal( - '
    ' + - '1. Old' + - 'New name
    '); - }); - }); // describe('anchors inside the recursive diff') - - describe('atomic tags inside the recursive diff', function(){ - function el(extraAttrs, inner){ - return '
    ' + inner + '
    '; - } - - it('keeps the default atomic tags (without a) inside the recursion', function(){ - // Embedded content like svg stays atomic by default: a changed svg is replaced - // as a whole, not word-diffed. - var res = cut( - el('', 'old'), - el('', 'new')); - expect(res).to.equal( - '
    ' + - '' + - 'old' + - '' + - 'new' + - '
    '); - }); - - it('replaces the default list with data-htmldiff-inner-diff-atomic-tags', function(){ - // The override lists only em, so svg is no longer atomic and gets word-diffed. - var attrs = ' data-htmldiff-inner-diff-atomic-tags="em"'; - var res = cut( - el(attrs, 'old'), - el(attrs, 'new')); - expect(res).to.equal( - '
    ' + - '' + - 'old' + - 'new' + - '
    '); - }); - - it('restores atomic anchors (href comparison) when the override lists a', function(){ - var attrs = ' data-htmldiff-inner-diff-atomic-tags="a"'; - var res = cut( - el(attrs, 'Name'), - el(attrs, 'Name')); - expect(res).to.equal( - '
    ' + - 'Name' + - 'Name' + - '
    '); - }); - - it('treats no tag name as atomic when the override value is empty', function(){ - var attrs = ' data-htmldiff-inner-diff-atomic-tags=""'; - var res = cut( - el(attrs, 'old'), - el(attrs, 'new')); - expect(res).to.equal( - '
    ' + - '' + - 'old' + - 'new' + - '
    '); - }); - }); // describe('atomic tags inside the recursive diff') - - describe('opt-in attribute values', function(){ - function entry(value, name){ - return '
    ' + - name + '
    '; - } - - it('treats a double-quoted "false" value as opted out', function(){ - var res = cut(entry('"false"', 'old'), entry('"false"', 'new')); - expect(res).to.equal(entry('"false"', 'new')); - }); - - it('treats a single-quoted \'false\' value as opted out', function(){ - var res = cut(entry("'false'", 'old'), entry("'false'", 'new')); - expect(res).to.equal(entry("'false'", 'new')); - }); - - it('treats an unquoted false value as opted out', function(){ - var res = cut(entry('false', 'old'), entry('false', 'new')); - expect(res).to.equal(entry('false', 'new')); - }); - - it('treats a bare attribute without a value as opted in', function(){ - var res = cut( - '
    old
    ', - '
    new
    '); - expect(res).to.equal( - '
    ' + - 'old' + - 'new
    '); - }); - - it('only matches exact attribute name', function(){ - var res = cut( - '
    old
    ', - '
    new
    '); - expect(res).to.equal( - '
    new
    '); - }); - - it('recognizes the attribute regardless of its position', function(){ - var res = cut( - '
    old
    ', - '
    new
    '); - expect(res).to.equal( - '
    ' + - 'old' + - 'new
    '); - }); - }); // describe('opt-in attribute values') - - describe('edge cases', function(){ - it('ignores the attribute on non-atomic elements (no data-htmldiff-id)', function(){ - // Without data-htmldiff-id the div is not atomic, so this is a plain word diff. - var res = cut( - '
    old text
    ', - '
    new text
    '); - expect(res).to.equal( - '
    ' + - 'old' + - 'new text
    '); - }); - - it('diffs text following a self-closing opted-in element normally', function(){ - var res = cut( - '
    x old', - '
    x new'); - expect(res).to.equal( - '
    x ' + - 'old' + - 'new'); - }); - - it('marks the content as inserted when the before element was self-closing', function(){ - // A self-closing element has empty inner content, so the new content is a - // pure insertion. - var res = cut( - '
    ', - '
    new
    '); - expect(res).to.equal( - '
    ' + - 'new
    '); - }); - - it('marks the content as deleted when the after element became self-closing', function(){ - // The after element has no content anymore, so the deleted content is - // rendered right after the self-closing tag. - var res = cut( - '
    old
    ', - '
    '); - expect(res).to.equal( - '
    ' + - 'old'); - }); - - it('is not confused by ">" inside attribute values', function(){ - var res = cut( - '
    ' + - 'old
    ', - '
    ' + - 'new
    '); - expect(res).to.equal( - '
    ' + - 'old' + - 'new
    '); - }); - - it('is not confused by ">" inside single-quoted attribute values', function(){ - var res = cut( - "
    old
    ", - "
    new
    "); - expect(res).to.equal( - "
    " + - 'old' + - 'new
    '); - }); - - it('falls back to the after version when a token cannot be split', function(){ - // An unterminated atomic tag swallows the rest of the input and has no closing - // tag to split on; the inner diff falls back instead of producing broken markup. - var after = '
    new'; - var res = cut( - '
    old', - after); - expect(res).to.equal(after); - }); - - it('renders pure insertions when the before content is empty', function(){ - var res = cut(tocEntry('#1', ''), tocEntry('#1', 'New name')); - expect(res).to.equal( - '
    ' + - 'New name
    '); - }); - - it('do not diffs inner html for moved opted-in elements', function(){ - function entry(id, name){ - return '
    ' + - name + '
    '; - } - var res = cut(entry('a', 'First') + entry('b', 'Second'), - entry('b', 'Second') + entry('a', 'First')); - expect(res).to.equal( - '' + entry('b', 'Second') + '' + - entry('a', 'First') + - '' + entry('b', 'Second') + ''); - }); - }); // describe('edge cases') - - describe('when the element did not opt in', function(){ - it('should render the after version as is when keys match but content differs', function(){ - var before = '
    ' + - '
    '; - var after = '
    ' + - '
    '; - expect(cut(before, after)).to.equal(after); - }); - }); // describe('when the element did not opt in') - - describe('when key and content are both equal', function(){ - it('should render the element unchanged', function(){ - var before = 'x ' + tocEntry('#123', '1. Same name') + ' y'; - var after = 'x ' + tocEntry('#123', '1. Same name') + ' z'; - expect(cut(before, after)).to.equal( - 'x ' + tocEntry('#123', '1. Same name') + ' ' + - 'yz'); - }); - }); // describe('when key and content are both equal') - - describe('when opted-in elements are nested', function(){ - function nest(depth, content){ - var html = content; - for (var i = depth; i >= 1; i--){ - html = '
    ' + - html + '
    '; - } - return html; - } - - it('diffs nested elements', function(){ - var res = cut( - '
    x ' + - 'old
    ', - '
    x ' + - 'new
    '); - expect(res).to.equal( - '
    x ' + - '' + - 'old' + - 'new
    '); - }); - - it('do not diffs nested element that do not opt-in', function(){ - var after = '
    t ' + - 'new
    '; - var res = cut( - '
    t ' + - 'old
    ', - after); - expect(res).to.equal(after); - }); - - it('diffs all levels up to the depth cap of 10', function(){ - var res = cut(nest(10, 'old'), nest(10, 'new')); - expect(res).to.equal(nest(10, - 'old' + - 'new')); - }); - - it('renders the after version as is beyond the depth cap', function(){ - var res = cut(nest(11, 'old'), nest(11, 'new')); - expect(res).to.equal(nest(11, 'new')); - }); - }); // describe('when opted-in elements are nested') - -}); // describe('Recursive inner diff (data-htmldiff-inner-diff)') diff --git a/test/mocha.opts b/test/mocha.opts deleted file mode 100644 index cdb0a40..0000000 --- a/test/mocha.opts +++ /dev/null @@ -1,3 +0,0 @@ ---require test/config.js ---ui bdd ---reporter spec diff --git a/test/module.spec.js b/test/module.spec.js deleted file mode 100644 index 80cccc4..0000000 --- a/test/module.spec.js +++ /dev/null @@ -1,11 +0,0 @@ -describe('The module', function(){ - var cut; - - beforeEach(function(){ - cut = require('../js/htmldiff'); - }); - - it('should return a function', function(){ - expect(cut).is.a('function'); - }); -}); diff --git a/test/module.spec.ts b/test/module.spec.ts new file mode 100644 index 0000000..bbaabb0 --- /dev/null +++ b/test/module.spec.ts @@ -0,0 +1,8 @@ +import { expect } from "chai"; +import diff from "../src/htmldiff"; + +describe("The module", () => { + it("should return a function", () => { + expect(diff).is.a("function"); + }); +}); diff --git a/test/pain_games.spec.js b/test/pain_games.spec.js deleted file mode 100644 index d921879..0000000 --- a/test/pain_games.spec.js +++ /dev/null @@ -1,18 +0,0 @@ -describe('Pain Games', function(){ - var cut, res; - - beforeEach(function(){ - cut = require('../js/htmldiff'); - }); - - describe('When an entire sentence is replaced', function(){ - beforeEach(function(){ - res = cut('this is what I had', 'and now we have a new one'); - }); - - it('should replace the whole chunk', function(){ - expect(res).to.equal('this is what I had' + - 'and now we have a new one'); - }); - }); -}); diff --git a/test/pain_games.spec.ts b/test/pain_games.spec.ts new file mode 100644 index 0000000..5cefc4c --- /dev/null +++ b/test/pain_games.spec.ts @@ -0,0 +1,11 @@ +import { expect } from "chai"; +import diff from "../src/htmldiff"; + +describe("Pain Games", () => { + describe("When an entire sentence is replaced", () => { + it("should replace the whole chunk", () => { + const res = diff("this is what I had", "and now we have a new one"); + expect(res).to.equal('this is what I had' + 'and now we have a new one'); + }); + }); +}); diff --git a/test/render_operations.spec.js b/test/render_operations.spec.js deleted file mode 100644 index 1e179b8..0000000 --- a/test/render_operations.spec.js +++ /dev/null @@ -1,178 +0,0 @@ -describe('renderOperations', function(){ - var cut, res, createToken, tokenize; - - beforeEach(function(){ - var diff = require('../js/htmldiff'); - createToken = diff.findMatchingBlocks.createToken; - - tokenize = function(tokens){ - return tokens.map(function(token){ - return createToken(token); - }); - }; - - cut = function(before, after){ - var ops = diff.calculateOperations(before, after); - return diff.renderOperations(before, after, ops); - }; - }); - - it('should be a function', function(){ - expect(cut).is.a('function'); - }); - - describe('equal', function(){ - beforeEach(function(){ - var before = tokenize(['this', ' ', 'is', ' ', 'a', ' ', 'test']); - res = cut(before, before); - }); - - it('should output the text', function(){ - expect(res).equal('this is a test'); - }); - }); - - describe('insert', function(){ - beforeEach(function(){ - before = tokenize(['this', ' ', 'is']); - after = tokenize(['this', ' ', 'is', ' ', 'a', ' ', 'test']); - res = cut(before, after); - }); - - it('should wrap in an ', function(){ - expect(res).equal('this is a test'); - }); - }); - - describe('delete', function(){ - beforeEach(function(){ - var before = tokenize(['this', ' ', 'is', ' ', 'a', ' ', 'test', - ' ', 'of', ' ', 'stuff']); - var after = tokenize(['this', ' ', 'is', ' ', 'a', ' ', 'test']); - res = cut(before, after); - }); - - it('should wrap in a ', function(){ - expect(res).to.equal('this is a test of stuff'); - }); - }); - - describe('replace', function(){ - beforeEach(function(){ - var before = tokenize(['this', ' ', 'is', ' ', 'a', ' ', 'break']); - var after = tokenize(['this', ' ', 'is', ' ', 'a', ' ', 'test']); - res = cut(before, after); - }); - - it('should wrap in both and ', function(){ - expect(res).to.equal('this is a break' + - 'test'); - }); - }); - - describe('Dealing with tags', function(){ - var before, after; - - beforeEach(function(){ - before = tokenize(['

    ', 'a', '

    ']); - after = tokenize(['

    ', 'a', ' ', 'b', '

    ', '

    ', 'c', '

    ']); - res = cut(before, after); - }); - - it('should identify contained inserted tags', function(){ - expect(res).to.equal('

    a b

    ' + - '

    ' + - 'c

    '); - }); - - it('should identify contained deleted tags', function(){ - res = cut(after, before); - - expect(res).to.equal('

    a b

    ' + - '

    ' + - 'c

    '); - }); - - it('should not identify partial tags', function(){ - var before = tokenize(['test', '', 'non-bold']); - var after = tokenize(['test!', '
    ', 'non-bold', '', 'bold']); - res = cut(before, after); - - expect(res).to.equal('test' + - 'test!non-bold' + - 'bold'); - }); - - describe('When there is a change at the beginning, in a

    ', function(){ - beforeEach(function(){ - var before = tokenize(['

    ', 'this', ' ', 'is', ' ', 'awesome', '

    ']); - var after = tokenize(['

    ', 'I', ' ', 'is', ' ', 'awesome', '

    ']); - res = cut(before, after); - }); - - it('should keep the change inside the

    ', function(){ - expect(res).to.equal('

    this' + - 'I is awesome

    '); - }); - }); - }); - - describe('empty tokens', function(){ - it('should not be wrapped', function(){ - var before = tokenize(['text']); - var after = tokenize(['text', ' ']); - - res = cut(before, after); - - expect(res).to.equal('text'); - }); - }); - - describe('tags with attributes', function(){ - it('should treat attribute changes as equal and output the after tag', function(){ - var before = tokenize(['

    ', 'this', ' ', 'is', ' ', 'awesome', '

    ']); - var after = tokenize(['

    ', 'this', ' ', 'is', ' ', - 'awesome', '

    ']); - - res = cut(before, after); - - expect(res).to.equal('

    this is awesome

    '); - }); - - it('should show changes within tags with different attributes', function(){ - var before = tokenize(['

    ', 'this', ' ', 'is', ' ', 'awesome', '

    ']); - var after = tokenize(['

    ', 'that', ' ', 'is', ' ', - 'awesome', '

    ']); - - res = cut(before, after); - - expect(res).to.equal('

    ' + - 'this' + - 'that is awesome

    '); - }); - }); - - describe('wrappable tags', function(){ - it('should wrap void tags', function(){ - var before = tokenize(['old', ' ', 'text']); - var after = tokenize(['new', '
    ', ' ', 'text']); - - res = cut(before, after); - - expect(res).to.equal('old' + - 'new
    text'); - }); - - it('should wrap atomic tags independently', function(){ - var before = tokenize(['old', '', ' ', 'text']); - var after = tokenize(['new', ' ', 'text']); - - res = cut(before, after); - - expect(res).to.equal( - 'old' + - '' + - 'new text'); - }); - }); -}); diff --git a/test/tables/Cell.spec.ts b/test/tables/Cell.spec.ts new file mode 100644 index 0000000..5670b9d --- /dev/null +++ b/test/tables/Cell.spec.ts @@ -0,0 +1,109 @@ +import { expect } from "chai"; +import { Cell } from "../../src/tables/Cell"; + +describe("Cell", () => { + const cell = (openTag = "", inner = ""): Cell => new Cell("td", openTag, inner, ""); + + describe("readRow", () => { + it("reads td and th cells", () => { + const cells = Cell.readRow('hd'); + expect(cells.map((c) => [c.tag, c.openTag, c.inner])).to.deep.equal([ + ["th", "", "h"], + ["td", '', "d"], + ]); + }); + }); + + describe("ownId", () => { + it("reads the cell's own identity", () => { + expect(cell('', "TC-3 text").ownId()).to.equal("TC-3"); + expect(cell("", 'TC-3').ownId()).to.equal(null); + }); + }); + + describe("signature", () => { + it("is the collapsed text", () => { + expect(cell("", " a b\n c ").signature()).to.equal("a b c"); + }); + + it("adds the identities of atomic elements and images", () => { + expect(cell("", 'REQ-1 t ').signature()).to.equal( + "REQ-1 t|REQ-1|i.png", + ); + }); + + it("ignores hidden helpers", () => { + expect(cell("", 'shown').signature()).to.equal("shown"); + }); + + it("decodes entities and ignores comments", () => { + expect(cell("", "a&b").signature()).to.equal("a&b"); + }); + + it("is empty for an empty cell", () => { + expect(cell().signature()).to.equal(""); + }); + }); + + describe("itemRefs", () => { + it("lists the smart link ids in order", () => { + expect( + cell("", 'A-1 B-2').itemRefs(), + ).to.deep.equal(["A-1", "B-2"]); + }); + + it("ignores other elements with an id", () => { + expect(cell("", 'x').itemRefs()).to.deep.equal([]); + }); + }); + + describe("span and shape", () => { + it("reads spans, defaulting to 1", () => { + expect(cell('').span("colspan")).to.equal(3); + expect(cell('').span("colspan")).to.equal(1); + expect(cell().span("rowspan")).to.equal(1); + }); + + it("describes the shape", () => { + expect(cell('').shape()).to.equal("td:2x3"); + }); + }); + + describe("placeholders, parts and classes", () => { + it("creates a placeholder with the siblings' tag", () => { + const th = Cell.placeholder([new Cell("th", "", "", "")], "x"); + expect(th.render()).to.equal(''); + expect(Cell.placeholder([]).render()).to.equal(""); + }); + + it("creates a part standing for a merged cell", () => { + const part = Cell.part("k", []); + expect(part.partOf).to.equal("k"); + expect(part.render()).to.equal(""); + }); + + it("adds and detects change classes", () => { + const c = cell('', "x"); + expect(c.hasChangeClass()).to.equal(false); + c.addClass("table-cell-deleted"); + expect(c.openTag).to.equal(''); + expect(c.hasChangeClass()).to.equal(true); + }); + + it("renders nothing for a removed cell", () => { + const c = cell("", "x"); + c.removed = true; + expect(c.render()).to.equal(""); + }); + + it("clones a cell without its removal", () => { + const c = cell("", "x"); + c.removed = true; + c.mergedKey = "k"; + const copy = c.clone(); + expect(copy.removed).to.equal(false); + expect(copy.mergedKey).to.equal("k"); + expect(copy).to.not.equal(c); + }); + }); +}); diff --git a/test/tables/ColumnAligner.spec.ts b/test/tables/ColumnAligner.spec.ts new file mode 100644 index 0000000..ee8f01e --- /dev/null +++ b/test/tables/ColumnAligner.spec.ts @@ -0,0 +1,126 @@ +import { expect } from "chai"; +import { ColumnAligner } from "../../src/tables/ColumnAligner"; +import { findElements } from "../../src/tables/html"; +import { Alignment } from "../../src/tables/SequenceAligner"; +import { Table } from "../../src/tables/Table"; +import { TableVersion } from "../../src/tables/TableVersion"; + +describe("ColumnAligner", () => { + const version = (markup: string): TableVersion => TableVersion.read(Table.read(findElements(markup, ["table"])[0], false)); + const rows = (cells: string[][]): string => + `${cells.map((row) => `${row.map((cell) => ``).join("")}`).join("")}
    ${cell}
    `; + const same: Alignment[] = [ + { kind: "same", oldIndex: 0, newIndex: 0 }, + { kind: "same", oldIndex: 1, newIndex: 1 }, + ]; + + describe("align", () => { + it("pairs columns by their values whatever the rows did", () => { + const aligner = new ColumnAligner( + version( + rows([ + ["a", "b"], + ["c", "d"], + ]), + ), + version( + rows([ + ["x", "b"], + ["c", "d"], + ["e", "f"], + ]), + ), + ); + expect(aligner.align()).to.deep.equal(same); + }); + + it("adds a column in the middle", () => { + expect(new ColumnAligner(version(rows([["a", "b"]])), version(rows([["a", "n", "b"]]))).align()).to.deep.equal([ + { kind: "same", oldIndex: 0, newIndex: 0 }, + { kind: "added", newIndex: 1 }, + { kind: "same", oldIndex: 1, newIndex: 2 }, + ]); + }); + + it("pairs columns in place when the values tell nothing", () => { + expect(new ColumnAligner(version(rows([["a", "b"]])), version(rows([["x", "y"]]))).align()).to.deep.equal(same); + }); + }); + + describe("splitReplaced", () => { + it("splits a column no kept row agrees with", () => { + const aligner = new ColumnAligner( + version( + rows([ + ["a", "b"], + ["c", "d"], + ]), + ), + version( + rows([ + ["a", "1"], + ["c", "2"], + ]), + ), + ); + expect(aligner.splitReplaced(same, same)).to.deep.equal([ + { kind: "same", oldIndex: 0, newIndex: 0 }, + { kind: "deleted", oldIndex: 1 }, + { kind: "added", newIndex: 1 }, + ]); + }); + + it("needs two kept rows to tell a replacement from an edit", () => { + const aligner = new ColumnAligner(version(rows([["a", "b"]])), version(rows([["a", "1"]]))); + expect(aligner.splitReplaced(same, [{ kind: "same", oldIndex: 0, newIndex: 0 }])).to.deep.equal(same); + }); + + it("never splits a line number column", () => { + const aligner = new ColumnAligner( + version( + rows([ + ["1", "a"], + ["2", "b"], + ]), + ), + version( + rows([ + ["1", "b"], + ["2", "a"], + ]), + ), + ); + const kept: Alignment[] = [ + { kind: "same", oldIndex: 1, newIndex: 0 }, + { kind: "same", oldIndex: 0, newIndex: 1 }, + ]; + expect(aligner.splitReplaced(same, kept)).to.deep.equal(same); + }); + }); + + describe("rebuildColgroup", () => { + it("writes one col per merged column with the change on it", () => { + const aligner = new ColumnAligner( + version('
    ab
    '), + version('
    a
    '), + ); + const merged = 'ab'; + expect( + aligner.rebuildColgroup( + [ + { kind: "same", oldIndex: 0, newIndex: 0 }, + { kind: "deleted", oldIndex: 1 }, + ], + merged, + ), + ).to.equal( + 'ab', + ); + }); + + it("leaves a table without colgroup alone", () => { + const v = version(rows([["a"]])); + expect(new ColumnAligner(v, v).rebuildColgroup([{ kind: "same", oldIndex: 0, newIndex: 0 }], "inner")).to.equal("inner"); + }); + }); +}); diff --git a/test/tables/MergedCells.spec.ts b/test/tables/MergedCells.spec.ts new file mode 100644 index 0000000..a973809 --- /dev/null +++ b/test/tables/MergedCells.spec.ts @@ -0,0 +1,123 @@ +import { expect } from "chai"; +import { findElements } from "../../src/tables/html"; +import { MergedCells } from "../../src/tables/MergedCells"; +import { Alignment } from "../../src/tables/SequenceAligner"; +import { Table } from "../../src/tables/Table"; +import { TableVersion } from "../../src/tables/TableVersion"; + +describe("MergedCells", () => { + const ref = (itemRef: string): string => `${itemRef}`; + const readTable = (markup: string, inSection = false): Table => Table.read(findElements(markup, ["table"])[0], inSection); + const layout = (table: Table): string[] => + table.rows().map((row) => + row.cells + .map((cell) => { + if (cell.removed) return "-"; + if (cell.mergedKey) return `[${cell.inner}]`; + return cell.partOf ? `(${cell.partOf})` : cell.inner; + }) + .join(" "), + ); + const twoColumns: Alignment[] = [ + { kind: "same", oldIndex: 0, newIndex: 0 }, + { kind: "same", oldIndex: 1, newIndex: 1 }, + ]; + + describe("expand", () => { + it("splits a merged cell into the cell and empty parts", () => { + const table = readTable('
    ga
    b
    '); + MergedCells.expand(table, "old"); + expect(layout(table)).to.deep.equal(["[g] (old-0-0) a", "(old-0-0) (old-0-0) b"]); + expect(table.rows()[0].cells[0].openTag).to.equal(""); + }); + }); + + describe("fold", () => { + it("spans a merged cell over its empty parts again", () => { + const table = readTable('
    ga
    b
    '); + MergedCells.expand(table, "new"); + const version = TableVersion.read(table); + new MergedCells(version, version).fold(); + expect(layout(table)).to.deep.equal(["[g] a", "- b"]); + expect(table.rows()[0].cells[0].openTag).to.equal(''); + }); + + it("does not span over a row of another kind", () => { + const table = readTable('
    ga
    b
    '); + MergedCells.expand(table, "new"); + const version = TableVersion.read(table); + table.rows()[1].added = true; + new MergedCells(version, version).fold(); + expect(layout(table)).to.deep.equal(["[g] a", "(new-0-0) b"]); + }); + + it("lets a changed cell span its own row only", () => { + const table = readTable('
    g
    '); + MergedCells.expand(table, "old"); + const version = TableVersion.read(table); + const rows = table.rows(); + rows[0].changedInGroup = true; + rows[0].changeClass = "table-cell-deleted"; + rows[0].cells[0].openTag = ''; + new MergedCells(version, version).fold(); + expect(table.rows()[0].cells[0].openTag).to.equal(''); + expect(layout(table)).to.deep.equal(["[g] -", "(old-0-0) (old-0-0)"]); + }); + + it("gives a part left alone in a changed row the row's change", () => { + const table = readTable("
    ab
    "); + const version = TableVersion.read(table); + const row = table.rows()[0]; + row.changedInGroup = true; + row.changeClass = "table-cell-added"; + row.cells[1].partOf = "gone"; + row.cells[1].inner = ""; + new MergedCells(version, version).fold(); + expect(row.cells[1].openTag).to.equal(''); + }); + }); + + describe("moveOwnersToKeptRows", () => { + it("moves the merged cell onto the first kept row of its group", () => { + const oldTable = readTable('
    ga
    b
    '); + const newTable = readTable("
    gb
    "); + MergedCells.expand(oldTable, "old"); + const oldVersion = TableVersion.read(oldTable); + const newVersion = TableVersion.read(newTable); + new MergedCells(oldVersion, newVersion).moveOwnersToKeptRows(twoColumns, [ + { kind: "deleted", oldIndex: 0 }, + { kind: "same", oldIndex: 1, newIndex: 0 }, + ]); + expect(layout(oldTable)).to.deep.equal(["(old-0-0) a", "[g] b"]); + // the plain new cell took over the old merged cell + expect(newVersion.cells[0][0].mergedKey).to.equal("old-0-0"); + expect(newVersion.mergedCells["old-0-0"]).to.equal("g"); + }); + }); + + describe("findReplacedItemGroups", () => { + it("finds an item none of whose rows matched", () => { + const oldVersion = TableVersion.read(readTable(`
    ${ref("S-1")}old
    `, true)); + const newVersion = TableVersion.read(readTable(`
    ${ref("S-1")}new
    `, true)); + const groups = new MergedCells(oldVersion, newVersion).findReplacedItemGroups( + twoColumns, + [ + { kind: "deleted", oldIndex: 0 }, + { kind: "added", newIndex: 0 }, + ], + ["S-1"], + ["S-1"], + ); + expect(Object.keys(groups)).to.deep.equal(["S-1"]); + expect(groups["S-1"].column).to.equal(0); + expect(groups["S-1"].key).to.equal("item-0"); + expect(groups["S-1"].itemCell).to.equal(newVersion.cells[0][0]); + }); + + it("skips an item one of whose rows was kept", () => { + const v = TableVersion.read(readTable(`
    ${ref("S-1")}x
    `, true)); + const groups = new MergedCells(v, v).findReplacedItemGroups([], [{ kind: "same", oldIndex: 0, newIndex: 0 }], ["S-1"], ["S-1"]); + expect(groups).to.deep.equal({}); + }); + }); +}); diff --git a/test/tables/Row.spec.ts b/test/tables/Row.spec.ts new file mode 100644 index 0000000..285e21d --- /dev/null +++ b/test/tables/Row.spec.ts @@ -0,0 +1,56 @@ +import { expect } from "chai"; +import { Cell } from "../../src/tables/Cell"; +import { Row } from "../../src/tables/Row"; + +describe("Row", () => { + const row = (text: string): Row => new Row("", [new Cell("td", "", text, "")], "tbody"); + + it("tells its kind", () => { + const r = row("a"); + expect(r.kind()).to.equal("kept"); + r.added = true; + expect(r.kind()).to.equal("added"); + r.deleted = true; + expect(r.kind()).to.equal("deleted"); + }); + + it("renders its cells and the rows hung onto it", () => { + const a = row("a"); + const b = row("b"); + const c = row("c"); + a.insertBefore(b); + a.insertAfter(c); + expect(a.render()).to.equal("bac"); + }); + + it("hangs later rows right after itself, like Element.after", () => { + const a = row("a"); + a.insertAfter(row("x")); + a.insertAfter(row("y")); + expect(a.render()).to.equal("ayx"); + }); + + it("gives a hung row its own row group", () => { + const a = row("a"); + const b = new Row("", [], null); + a.insertBefore(b); + expect(b.container).to.equal("tbody"); + }); + + it("collects itself and its hung rows in document order", () => { + const a = row("a"); + const b = row("b"); + const c = row("c"); + a.insertBefore(b); + b.insertAfter(c); + const rows: Row[] = []; + a.collect(rows); + expect(rows.map((r) => r.cells[0].inner)).to.deep.equal(["b", "c", "a"]); + }); + + it("skips removed cells when rendering", () => { + const r = row("a"); + r.cells[0].removed = true; + expect(r.render()).to.equal(""); + }); +}); diff --git a/test/tables/RowAligner.spec.ts b/test/tables/RowAligner.spec.ts new file mode 100644 index 0000000..ed27d4b --- /dev/null +++ b/test/tables/RowAligner.spec.ts @@ -0,0 +1,179 @@ +import { expect } from "chai"; +import { findElements } from "../../src/tables/html"; +import { RowAligner } from "../../src/tables/RowAligner"; +import { Alignment } from "../../src/tables/SequenceAligner"; +import { Table } from "../../src/tables/Table"; +import { TableVersion } from "../../src/tables/TableVersion"; + +describe("RowAligner", () => { + const ref = (itemRef: string): string => `${itemRef}`; + const version = (markup: string, inSection = false): TableVersion => TableVersion.read(Table.read(findElements(markup, ["table"])[0], inSection)); + const plain = (cells: string[][]): string => + `${cells.map((row) => `${row.map((cell) => ``).join("")}`).join("")}
    ${cell}
    `; + const twoColumns: Alignment[] = [ + { kind: "same", oldIndex: 0, newIndex: 0 }, + { kind: "same", oldIndex: 1, newIndex: 1 }, + ]; + + describe("haveCompatibleRefs", () => { + it("allows refs only added or only removed", () => { + expect(RowAligner.haveCompatibleRefs(["A"], ["A", "B"])).to.equal(true); + expect(RowAligner.haveCompatibleRefs(["A", "B"], ["A"])).to.equal(true); + expect(RowAligner.haveCompatibleRefs([], ["A"])).to.equal(true); + }); + + it("refuses a swapped ref", () => { + expect(RowAligner.haveCompatibleRefs(["A"], ["B"])).to.equal(false); + expect(RowAligner.haveCompatibleRefs(["A", "B"], ["A", "C"])).to.equal(false); + }); + }); + + describe("align", () => { + it("pairs rows by content outside a section", () => { + const aligner = new RowAligner( + version( + plain([ + ["a", "b"], + ["c", "d"], + ]), + ), + version( + plain([ + ["a", "x"], + ["e", "f"], + ]), + ), + twoColumns, + ); + expect(aligner.align()).to.deep.equal([ + { kind: "same", oldIndex: 0, newIndex: 0 }, + { kind: "deleted", oldIndex: 1 }, + { kind: "added", newIndex: 1 }, + ]); + }); + + it("keeps rows about different items apart", () => { + const aligner = new RowAligner(version(plain([[ref("A-1"), "b"]]), true), version(plain([[ref("A-2"), "b"]]), true), twoColumns); + expect(aligner.align()).to.deep.equal([ + { kind: "deleted", oldIndex: 0 }, + { kind: "added", newIndex: 0 }, + ]); + }); + + it("keeps a row apart when a ref in a cell was swapped", () => { + const aligner = new RowAligner( + version(plain([[ref("A-1"), `${ref("T-1")} text`]]), true), + version(plain([[ref("A-1"), `${ref("T-2")} text`]]), true), + twoColumns, + ); + expect(aligner.align()).to.deep.equal([ + { kind: "deleted", oldIndex: 0 }, + { kind: "added", newIndex: 0 }, + ]); + }); + + describe("keyed tables", () => { + const keyed = (keys: string[], content: string[]): string => + `${keys.map((key) => ``).join("")}${content + .map((cell) => ``) + .join("")}
    ${key} title${cell}
    `; + const columns = (count: number): Alignment[] => Array.from({ length: count }, (_, index) => ({ kind: "same", oldIndex: index, newIndex: index })); + + it("pairs rows with the same keys whatever their other cells say", () => { + const aligner = new RowAligner( + version(keyed(["TR-3", "TC-1", "XTC-11"], ["", "", "pending"]), true), + version(keyed(["TR-3", "TC-1", "XTC-11"], ["2026/10/05", "jdoe", "passed"]), true), + columns(6), + ); + expect(aligner.align()).to.deep.equal([{ kind: "same", oldIndex: 0, newIndex: 0 }]); + }); + + it("pairs rows with the same keys when a key cell's text changed", () => { + const aligner = new RowAligner( + version('
    SPEC-6TC-3 Brake test
    ', true), + version('
    SPEC-6TC-3 Brake tests run
    ', true), + twoColumns, + ); + expect(aligner.align()).to.deep.equal([{ kind: "same", oldIndex: 0, newIndex: 0 }]); + }); + + it("keeps rows with a different key apart", () => { + const aligner = new RowAligner( + version(keyed(["SPEC-6", "TC-3"], ["same"]), true), + version(keyed(["SPEC-6", "TC-4"], ["same"]), true), + columns(3), + ); + expect(aligner.align()).to.deep.equal([ + { kind: "deleted", oldIndex: 0 }, + { kind: "added", newIndex: 0 }, + ]); + }); + + it("pairs by content when only one version is keyed", () => { + const aligner = new RowAligner( + version(keyed(["RISK-1"], ["same", "text"]), true), + version(plain([[`${ref("RISK-1")} title`, "same", "text"]]), true), + columns(3), + ); + expect(aligner.align()).to.deep.equal([{ kind: "same", oldIndex: 0, newIndex: 0 }]); + }); + }); + + it("pairs a row whose cell gained a ref and keeps its content", () => { + const aligner = new RowAligner( + version(plain([[ref("A-1"), "same", ref("T-1")]]), true), + version(plain([[ref("A-1"), "same", `${ref("T-1")} ${ref("T-2")}`]]), true), + twoColumns.concat([{ kind: "same", oldIndex: 2, newIndex: 2 }]), + ); + expect(aligner.align()).to.deep.equal([{ kind: "same", oldIndex: 0, newIndex: 0 }]); + }); + + it("pairs an item row that was emptied or filled", () => { + const aligner = new RowAligner(version(plain([[ref("A-1"), "text"]]), true), version(plain([[ref("A-1"), ""]]), true), twoColumns); + expect(aligner.align()).to.deep.equal([{ kind: "same", oldIndex: 0, newIndex: 0 }]); + }); + + it("ignores a line number column", () => { + const aligner = new RowAligner( + version( + plain([ + ["1", "a"], + ["2", "b"], + ]), + ), + version( + plain([ + ["1", "b"], + ["2", "a"], + ]), + ), + twoColumns, + ); + expect(aligner.align()).to.deep.equal([ + { kind: "deleted", oldIndex: 0 }, + { kind: "same", oldIndex: 1, newIndex: 0 }, + { kind: "added", newIndex: 1 }, + ]); + }); + }); + + describe("alternateReplaced", () => { + it("alternates deleted and added rows of a replaced run", () => { + const v = version(plain([["a"], ["b"]])); + const aligner = new RowAligner(v, v, [{ kind: "same", oldIndex: 0, newIndex: 0 }]); + expect( + aligner.alternateReplaced([ + { kind: "deleted", oldIndex: 0 }, + { kind: "deleted", oldIndex: 1 }, + { kind: "added", newIndex: 0 }, + { kind: "added", newIndex: 1 }, + ]), + ).to.deep.equal([ + { kind: "deleted", oldIndex: 0 }, + { kind: "added", newIndex: 0 }, + { kind: "deleted", oldIndex: 1 }, + { kind: "added", newIndex: 1 }, + ]); + }); + }); +}); diff --git a/test/tables/SequenceAligner.spec.ts b/test/tables/SequenceAligner.spec.ts new file mode 100644 index 0000000..19db9c2 --- /dev/null +++ b/test/tables/SequenceAligner.spec.ts @@ -0,0 +1,107 @@ +import { expect } from "chai"; +import { PositionalPairing, Sequence, SequenceAligner } from "../../src/tables/SequenceAligner"; + +describe("SequenceAligner", () => { + const sequence = (oldValues: string[], newValues: string[], byPosition: Partial = {}): Sequence => ({ + oldCount: oldValues.length, + newCount: newValues.length, + similarity: (oldIndex, newIndex) => { + if (oldValues[oldIndex] === "" && newValues[newIndex] === "") return NaN; + return oldValues[oldIndex] === newValues[newIndex] ? 1 : 0; + }, + byPosition: { + isOldBlank: (oldIndex) => oldValues[oldIndex] === "", + isNewBlank: (newIndex) => newValues[newIndex] === "", + ...byPosition, + }, + }); + + describe("matchInOrder", () => { + it("finds the longest in-order run of matches", () => { + const pairs = SequenceAligner.matchInOrder( + { oldStart: 0, oldEnd: 3, newStart: 0, newEnd: 3 }, + (o, n) => ["a", "b", "c"][o] === ["c", "a", "b"][n], + ); + expect(pairs).to.deep.equal([ + { oldIndex: 0, newIndex: 1 }, + { oldIndex: 1, newIndex: 2 }, + ]); + }); + }); + + describe("toAlignments", () => { + it("puts deleted before added inside a gap", () => { + expect(SequenceAligner.toAlignments([{ oldIndex: 1, newIndex: 1 }], 2, 3)).to.deep.equal([ + { kind: "deleted", oldIndex: 0 }, + { kind: "added", newIndex: 0 }, + { kind: "same", oldIndex: 1, newIndex: 1 }, + { kind: "added", newIndex: 2 }, + ]); + }); + }); + + describe("pairByPosition", () => { + it("pairs everything in place when counts match and pairsInPlace allows", () => { + const pairs = SequenceAligner.pairByPosition( + { pairsInPlace: (o) => o !== 1, isOldBlank: () => false, isNewBlank: () => false }, + { oldStart: 0, oldEnd: 2, newStart: 0, newEnd: 2 }, + ); + expect(pairs).to.deep.equal([{ oldIndex: 0, newIndex: 0 }]); + }); + + it("pairs a blank entry with a blank one, else with the entry in its place", () => { + const blankFirst = (index: number): boolean => index === 0; + expect( + SequenceAligner.pairByPosition( + { isOldBlank: blankFirst, isNewBlank: (index) => index === 1 }, + { oldStart: 0, oldEnd: 2, newStart: 0, newEnd: 2 }, + ), + ).to.deep.equal([{ oldIndex: 0, newIndex: 1 }]); + expect( + SequenceAligner.pairByPosition({ isOldBlank: blankFirst, isNewBlank: () => false }, { oldStart: 0, oldEnd: 1, newStart: 0, newEnd: 2 }), + ).to.deep.equal([{ oldIndex: 0, newIndex: 0 }]); + }); + + it("never pairs an entry with an identity by position", () => { + const pairs = SequenceAligner.pairByPosition( + { + pairsInPlace: () => true, + isOldBlank: () => true, + isNewBlank: () => true, + hasOldIdentity: () => true, + hasNewIdentity: () => false, + }, + { oldStart: 0, oldEnd: 1, newStart: 0, newEnd: 1 }, + ); + expect(pairs).to.deep.equal([]); + }); + }); + + describe("align", () => { + it("anchors exact matches and turns a move into delete and add", () => { + expect(new SequenceAligner(sequence(["a", "b", "c"], ["c", "a", "b"])).align()).to.deep.equal([ + { kind: "added", newIndex: 0 }, + { kind: "same", oldIndex: 0, newIndex: 1 }, + { kind: "same", oldIndex: 1, newIndex: 2 }, + { kind: "deleted", oldIndex: 2 }, + ]); + }); + + it("pairs blank entries left over by position", () => { + expect(new SequenceAligner(sequence(["a", ""], ["a", ""])).align()).to.deep.equal([ + { kind: "same", oldIndex: 0, newIndex: 0 }, + { kind: "same", oldIndex: 1, newIndex: 1 }, + ]); + }); + + it("pairs similar entries inside the gaps", () => { + const seq = sequence(["a", "x", "c"], ["a", "y", "c"]); + seq.similarity = (o, n) => (o === n ? (o === 1 ? 0.6 : 1) : 0); + expect(new SequenceAligner(seq).align()).to.deep.equal([ + { kind: "same", oldIndex: 0, newIndex: 0 }, + { kind: "same", oldIndex: 1, newIndex: 1 }, + { kind: "same", oldIndex: 2, newIndex: 2 }, + ]); + }); + }); +}); diff --git a/test/tables/Table.spec.ts b/test/tables/Table.spec.ts new file mode 100644 index 0000000..657264a --- /dev/null +++ b/test/tables/Table.spec.ts @@ -0,0 +1,99 @@ +import { expect } from "chai"; +import { Cell } from "../../src/tables/Cell"; +import { findElements } from "../../src/tables/html"; +import { Row } from "../../src/tables/Row"; +import { Table } from "../../src/tables/Table"; + +describe("Table", () => { + const readFirst = (markup: string, inSection = false): Table => Table.read(findElements(markup, ["table"])[0], inSection); + const row = (text: string): Row => new Row("", [new Cell("td", "", text, "")], null); + + describe("read", () => { + it("keeps the markup around the rows as chunks", () => { + const table = readFirst("
    h
    a
    "); + const described = table.chunks.map((chunk) => { + if (typeof chunk === "string") return chunk; + if (chunk === table.appendChunk) return "[append]"; + return `[row ${(chunk as Row).container}]`; + }); + expect(described).to.deep.equal(["", "[row thead]", "", "[row tbody]", "[append]", ""]); + }); + + it("appends to the end of a table without a body", () => { + const table = readFirst("
    a
    "); + expect(table.chunks.length).to.equal(2); + expect(table.chunks[1]).to.equal(table.appendChunk); + expect(table.appendChunk.container).to.equal(null); + }); + + it("renders back exactly what it read", () => { + const markup = "
    c
    h
    ab
    "; + const table = readFirst(markup); + expect(table.openTag + table.renderInner() + table.closeTag).to.equal(markup); + }); + }); + + describe("findTopLevel", () => { + it("finds outer tables and whether a section holds them", () => { + const tables = Table.findTopLevel( + '
    a
    in
    ', + ); + expect(tables.length).to.equal(2); + expect(tables[0].inSection).to.equal(true); + expect(tables[1].inSection).to.equal(false); + }); + + it("ignores a closed section", () => { + const tables = Table.findTopLevel('
    x
    a
    '); + expect(tables[0].inSection).to.equal(false); + }); + }); + + describe("identity", () => { + it("keeps its own id and takes a fallback otherwise", () => { + const own = readFirst('
    a
    '); + expect(own.ownId()).to.equal("REQ-1"); + expect(own.ensureId("x")).to.equal("REQ-1"); + const none = readFirst("
    a
    "); + expect(none.ownId()).to.equal(undefined); + expect(none.ensureId("x")).to.equal("x"); + expect(none.openTag).to.equal(''); + }); + }); + + describe("rows", () => { + it("orders rows by their row group, direct rows last", () => { + const table = readFirst("
    direct
    body
    "); + expect(table.rows().map((r) => r.cells[0].inner)).to.deep.equal(["body", "direct"]); + }); + + it("renders hung rows in order and knows the next sibling", () => { + const table = readFirst("
    a
    "); + const a = table.rows()[0]; + const b = row("b"); + const c = row("c"); + a.insertBefore(b); + a.insertAfter(c); + expect(table.documentRows().map((r) => r.cells[0].inner)).to.deep.equal(["b", "a", "c"]); + expect(table.renderInner()).to.equal("bac"); + expect(table.nextSibling(a)).to.equal(c); + expect(table.nextSibling(c)).to.equal(null); + }); + + it("leaves a detached row out of its place", () => { + const table = readFirst("
    a
    b
    "); + const rows = table.rows(); + rows[0].detach(); + rows[1].insertAfter(rows[0]); + expect(table.renderInner()).to.equal("ba"); + }); + + it("appends a row to the body", () => { + const table = readFirst("
    h
    "); + const appended = row("x"); + table.appendRow(appended); + expect(appended.container).to.equal("tbody"); + expect(table.renderInner()).to.equal("hx"); + }); + }); +}); diff --git a/test/tables/TableAligner.spec.ts b/test/tables/TableAligner.spec.ts new file mode 100644 index 0000000..adcb2e3 --- /dev/null +++ b/test/tables/TableAligner.spec.ts @@ -0,0 +1,99 @@ +import { expect } from "chai"; +import { Table } from "../../src/tables/Table"; +import { TableAligner } from "../../src/tables/TableAligner"; + +describe("TableAligner", () => { + const plain = (cells: string[][], attributes = ""): string => + `${cells.map((row) => `${row.map((cell) => `${cell}`).join("")}`).join("")}`; + const section = (cells: string[][]): string => `
    ${plain(cells)}
    `; + + it("pairs tables by their own id before anything else", () => { + const oldTables = Table.findTopLevel(plain([["a"]], ' data-htmldiff-id="REQ-1"') + plain([["b"]], ' data-htmldiff-id="REQ-2"')); + const newTables = Table.findTopLevel(plain([["b", "x"]], ' data-htmldiff-id="REQ-2"')); + expect(new TableAligner(oldTables, newTables).align()).to.deep.equal([ + { kind: "deleted", oldIndex: 0 }, + { kind: "same", oldIndex: 1, newIndex: 0 }, + ]); + }); + + it("never pairs tables with different ids", () => { + const oldTables = Table.findTopLevel(plain([["a"]], ' data-htmldiff-id="REQ-1"')); + const newTables = Table.findTopLevel(plain([["a"]], ' data-htmldiff-id="REQ-2"')); + expect(new TableAligner(oldTables, newTables).align()).to.deep.equal([ + { kind: "deleted", oldIndex: 0 }, + { kind: "added", newIndex: 0 }, + ]); + }); + + it("pairs id-less tables by content", () => { + const oldTables = Table.findTopLevel(plain([["a", "b"]]) + plain([["c", "d"]])); + const newTables = Table.findTopLevel(plain([["c", "d", "e"]])); + expect(new TableAligner(oldTables, newTables).align()).to.deep.equal([ + { kind: "deleted", oldIndex: 0 }, + { kind: "same", oldIndex: 1, newIndex: 0 }, + ]); + }); + + it("keeps a rewritten table in place only inside a section", () => { + expect(new TableAligner(Table.findTopLevel(plain([["a", "b"]])), Table.findTopLevel(plain([["x", "y"]]))).align()).to.deep.equal([ + { kind: "deleted", oldIndex: 0 }, + { kind: "added", newIndex: 0 }, + ]); + expect(new TableAligner(Table.findTopLevel(section([["a", "b"]])), Table.findTopLevel(section([["x", "y"]]))).align()).to.deep.equal([ + { kind: "same", oldIndex: 0, newIndex: 0 }, + ]); + }); + + it("keeps section tables apart when their headers differ", () => { + const headed = (header: string[], cells: string[][]): string => + `
    ${header.map((h) => ``).join("")}${cells + .map((row) => `${row.map((cell) => ``).join("")}`) + .join("")}
    ${h}
    ${cell}
    `; + const oldTables = Table.findTopLevel(headed(["Test run", "Executed Test Case"], [["TR-4", "TC-4"]])); + const newTables = Table.findTopLevel(headed(["Items", "Executed Test Case"], [["SPEC-2", "TC-4"]])); + expect(new TableAligner(oldTables, newTables).align()).to.deep.equal([ + { kind: "deleted", oldIndex: 0 }, + { kind: "added", newIndex: 0 }, + ]); + const sameHeader = Table.findTopLevel(headed(["Items", "Executed Test Case"], [["SPEC-9", "TC-1"]])); + expect(new TableAligner(newTables, sameHeader).align()).to.deep.equal([{ kind: "same", oldIndex: 0, newIndex: 0 }]); + }); + + it("keeps tables apart when only repeated filler values recur", () => { + const oldTables = Table.findTopLevel( + plain([ + ["sdvvsdvsdv", "sd vsdv", "sfv", "sdvsd"], + ["sd vsdv", "", "sd vsdv", ""], + ["sdvsd v", "sdvsdvsdv", "sdv sdv", "sdvsdv"], + ["s dvsdv", "", "sd vsd v", "sdvsdv"], + ]), + ); + const newTables = Table.findTopLevel( + plain([ + ["sdvsdvsdv", "sdvsdv", "sdvsdv"], + ["sdvsdv", "sdvsdv", "sdvsdvsdv"], + ]), + ); + expect(new TableAligner(oldTables, newTables).align()).to.deep.equal([ + { kind: "deleted", oldIndex: 0 }, + { kind: "added", newIndex: 0 }, + ]); + }); + + it("keeps rich text tables apart when their merged cells changed", () => { + const merged = '
    ga
    b
    '; + const unmerged = "
    ga
    hb
    "; + expect(new TableAligner(Table.findTopLevel(unmerged), Table.findTopLevel(merged)).align()).to.deep.equal([ + { kind: "deleted", oldIndex: 0 }, + { kind: "added", newIndex: 0 }, + ]); + const inSection = (table: string): Table[] => Table.findTopLevel(`
    ${table}
    `); + expect(new TableAligner(inSection(unmerged), inSection(merged)).align()).to.deep.equal([{ kind: "same", oldIndex: 0, newIndex: 0 }]); + }); + + it("keeps a table in place that still shares half of the smaller one", () => { + const oldTables = Table.findTopLevel(plain([["a"], ["b"], ["c"], ["d"]])); + const newTables = Table.findTopLevel(plain([["a"]])); + expect(new TableAligner(oldTables, newTables).align()).to.deep.equal([{ kind: "same", oldIndex: 0, newIndex: 0 }]); + }); +}); diff --git a/test/tables/TableMerger.spec.ts b/test/tables/TableMerger.spec.ts new file mode 100644 index 0000000..59f1024 --- /dev/null +++ b/test/tables/TableMerger.spec.ts @@ -0,0 +1,76 @@ +import { expect } from "chai"; +import { findElements } from "../../src/tables/html"; +import { Table } from "../../src/tables/Table"; +import { TableMerger } from "../../src/tables/TableMerger"; + +describe("TableMerger", () => { + const diffContent = (before: string, after: string): string => `[${before}>${after}]`; + const readTable = (markup: string, inSection = false): Table => Table.read(findElements(markup, ["table"])[0], inSection); + const plain = (cells: string[][]): string => + `${cells.map((row) => `${row.map((cell) => ``).join("")}`).join("")}
    ${cell}
    `; + const merge = (oldMarkup: string, newMarkup: string): string => new TableMerger(readTable(oldMarkup), readTable(newMarkup), diffContent).merge(); + + it("diffs kept cells with the injected diff", () => { + expect(merge(plain([["a", "b"]]), plain([["a", "c"]]))).to.equal("a[b>c]"); + }); + + it("marks added and deleted rows whole", () => { + expect(merge(plain([["a"], ["b"]]), plain([["a"], ["c"]]))).to.equal( + 'abc', + ); + }); + + it("marks a deleted column on its cells and the colgroup", () => { + expect( + merge( + "
    ab
    ", + "
    a
    ", + ), + ).to.equal('ab'); + }); + + it("adds a placeholder for an added column in a deleted row", () => { + expect(merge(plain([["a"], ["c"]]), plain([["a", "b"]]))).to.equal( + 'abc', + ); + }); + + it("diffs by position when the merged layout did not change", () => { + const markup = (value: string): string => `
    g${value}
    c
    `; + expect(merge(markup("a"), markup("b"))).to.equal('g[a>b]c'); + }); + + it("keeps a merged header and diffs the body rows under it", () => { + const header = 'h'; + expect( + merge( + `${header}
    ab
    `, + `${header}
    ab
    cd
    `, + ), + ).to.equal(`${header}abcd`); + }); + + it("leaves a row without cells as it is", () => { + expect( + merge("
    a
    ", "
    a
    b
    "), + ).to.equal('ab'); + }); + + it("hangs a deleted row under the kept rows of its group with the change on its cells", () => { + expect( + merge( + '
    ga
    b
    ', + "
    ga
    ", + ), + ).to.equal('gab'); + }); + + it("never puts a deleted row into the header", () => { + expect( + merge( + "
    h
    a
    ", + "
    h
    ", + ), + ).to.equal('ha'); + }); +}); diff --git a/test/tables/TableRedlining.spec.ts b/test/tables/TableRedlining.spec.ts new file mode 100644 index 0000000..63890e0 --- /dev/null +++ b/test/tables/TableRedlining.spec.ts @@ -0,0 +1,32 @@ +import { expect } from "chai"; +import { TableRedlining } from "../../src/tables/TableRedlining"; + +describe("TableRedlining", () => { + const diffContent = (before: string, after: string): string => `[${before}>${after}]`; + const redline = (before: string, after: string): { before: string; after: string } => new TableRedlining(diffContent).redline(before, after); + const plain = (cells: string[][], attributes = ""): string => + `${cells.map((row) => `${row.map((cell) => `${cell}`).join("")}`).join("")}`; + + it("leaves documents without tables alone", () => { + expect(redline("

    a

    ", "

    b

    ")).to.deep.equal({ before: "

    a

    ", after: "

    b

    " }); + }); + + it("gives both versions of a pair the same id and writes the merged table into the after html", () => { + const result = redline(`

    t

    ${plain([["a", "b"]])}`, `

    t

    ${plain([["a", "c"]])}`); + expect(result.before).to.equal('

    t

    ab
    '); + expect(result.after).to.equal('

    t

    a[b>c]
    '); + }); + + it("keeps a table's own id and drops its inner diff marker", () => { + const result = redline( + plain([["a"]], ' data-htmldiff-id="REQ-1" data-htmldiff-inner-diff="true"'), + plain([["a"], ["b"]], ' data-htmldiff-id="REQ-1" data-htmldiff-inner-diff="true"'), + ); + expect(result.after).to.equal('
    a
    b
    '); + }); + + it("gives an unpaired table an id of its own", () => { + expect(redline("", plain([["a"]])).after).to.equal('
    a
    '); + expect(redline(plain([["a"]]), "").before).to.equal('
    a
    '); + }); +}); diff --git a/test/tables/TableVersion.spec.ts b/test/tables/TableVersion.spec.ts new file mode 100644 index 0000000..d3efde0 --- /dev/null +++ b/test/tables/TableVersion.spec.ts @@ -0,0 +1,122 @@ +import { expect } from "chai"; +import { findElements } from "../../src/tables/html"; +import { MergedCells } from "../../src/tables/MergedCells"; +import { Table } from "../../src/tables/Table"; +import { TableVersion } from "../../src/tables/TableVersion"; + +describe("TableVersion", () => { + const ref = (itemRef: string): string => `${itemRef}`; + const readTable = (markup: string, inSection = false): Table => Table.read(findElements(markup, ["table"])[0], inSection); + const version = (markup: string, inSection = false): TableVersion => TableVersion.read(readTable(markup, inSection)); + + it("reads signatures, shapes and spans", () => { + const v = version('
    a
    bc
    '); + expect(v.signatures).to.deep.equal([["a"], ["b", "c"]]); + expect(v.shapes).to.deep.equal(["td:2x1", "td:1x1|td:1x1"]); + expect(v.columnCount).to.equal(2); + expect(v.hasSpans).to.equal(true); + expect(v.signatureAt(1, 1)).to.equal("c"); + expect(v.signatureAt(5, 5)).to.equal(""); + }); + + it("compares shapes, counts header rows and slices", () => { + const a = version("
    h
    a
    "); + const b = version("
    h
    x
    "); + expect(a.hasSameShapeAs(b)).to.equal(true); + expect(a.headerRowCount()).to.equal(1); + expect(a.slice(1).signatures).to.deep.equal([["a"]]); + }); + + it("compares span layouts by the merged cells alone", () => { + const group = version('
    ga
    b
    '); + const groupAndRow = version('
    ga
    b
    xy
    '); + const grownGroup = version('
    ga
    b
    c
    '); + const plain = version("
    ga
    hb
    "); + expect(group.hasSameSpanLayoutAs(groupAndRow)).to.equal(true); + expect(group.hasSameSpanLayoutAs(grownGroup)).to.equal(false); + expect(group.hasSameSpanLayoutAs(plain)).to.equal(false); + expect(plain.hasSameSpanLayoutAs(version("
    z
    "))).to.equal(true); + }); + + it("recognises a line number column", () => { + const v = version("
    #n
    1a
    2b
    "); + expect(v.isSequenceColumn(0)).to.equal(true); + expect(v.isSequenceColumn(1)).to.equal(false); + }); + + it("counts the values of every column and of the table", () => { + const v = version("
    a
    ab
    "); + expect(v.columnValueCounts()).to.deep.equal([{ a: 2 }, { b: 1 }]); + expect(v.valueCounts()).to.deep.equal({ a: 2, b: 1 }); + }); + + describe("items", () => { + it("lists refs in order, the item first, only inside a section", () => { + const markup = `
    ${ref("SPEC-1")}${ref("TC-1")} ${ref("TC-2")}
    `; + expect(version(markup, true).itemRefsByRow()).to.deep.equal([["SPEC-1", "TC-1", "TC-2"]]); + expect(version(markup, false).itemRefsByRow()).to.deep.equal([[]]); + expect(version(markup, true).hasItemRows()).to.equal(true); + }); + + it("gives every row of a group the refs of its merged cell", () => { + const table = readTable( + `
    ${ref("SPEC-1")}${ref("TC-1")}
    ${ref("TC-2")}
    `, + true, + ); + MergedCells.expand(table, "new"); + const v = TableVersion.read(table); + expect(v.itemRefsByRow()).to.deep.equal([ + ["SPEC-1", "TC-1"], + ["SPEC-1", "TC-2"], + ]); + expect(v.cellRefs(1, 0)).to.deep.equal(["SPEC-1"]); + expect(v.itemCellIndex(1)).to.equal(0); + }); + + it("has no row keys unless a cell carries its own identity", () => { + const v = version(`
    ${ref("SPEC-1")}${ref("TC-1")}
    `, true); + expect(v.hasRowKeys).to.equal(false); + expect(v.rowKeys(0)).to.deep.equal([]); + }); + + it("names a keyed row by the identities of its key cells", () => { + const markup = + '' + + `
    RISK-1 Firetext${ref("SPEC-2")}${ref("XTC-11")}
    `; + const v = version(markup, true); + expect(v.hasRowKeys).to.equal(true); + expect(v.rowKeys(0)).to.deep.equal(["RISK-1", "XTC-11"]); + expect(v.itemRefsByRow()).to.deep.equal([["RISK-1", "XTC-11"]]); + expect(v.itemCellIndex(0)).to.equal(0); + }); + + it("gives every row of a group the key of its merged cell", () => { + const table = readTable( + '' + + '
    SPEC-1TC-1
    TC-2
    ', + true, + ); + MergedCells.expand(table, "new"); + const v = TableVersion.read(table); + expect(v.rowKeys(1)).to.deep.equal(["SPEC-1", "TC-2"]); + expect(v.itemCellIndex(1)).to.equal(0); + }); + + it("finds the item cell, -1 without one", () => { + expect(version(`
    x${ref("A-1")}
    `, true).itemCellIndex(0)).to.equal(1); + expect(version("
    xy
    ", true).itemCellIndex(0)).to.equal(-1); + }); + }); + + describe("groups", () => { + it("knows the continuation rows of a merged cell and their group", () => { + const table = readTable('
    ga
    b
    '); + MergedCells.expand(table, "old"); + const v = TableVersion.read(table); + expect(v.isContinuationRow(0)).to.equal(false); + expect(v.isContinuationRow(1)).to.equal(true); + expect(v.groupSignatures(1)).to.deep.equal(["g"]); + expect(v.ownerOf(v.cells[1][0])).to.equal(v.cells[0][0]); + }); + }); +}); diff --git a/test/tables/helpers.spec.ts b/test/tables/helpers.spec.ts new file mode 100644 index 0000000..e481160 --- /dev/null +++ b/test/tables/helpers.spec.ts @@ -0,0 +1,22 @@ +import { expect } from "chai"; +import { flatten, last, range, uniqueValues } from "../../src/tables/helpers"; + +describe("helpers", () => { + it("range lists the integers from start up to end", () => { + expect(range(2, 5)).to.deep.equal([2, 3, 4]); + expect(range(3, 3)).to.deep.equal([]); + }); + + it("flatten joins arrays", () => { + expect(flatten([[1], [2, 3], []])).to.deep.equal([1, 2, 3]); + }); + + it("last is the last entry, undefined when empty", () => { + expect(last([1, 2])).to.equal(2); + expect(last([])).to.equal(undefined); + }); + + it("uniqueValues keeps the first occurrence", () => { + expect(uniqueValues(["a", "b", "a"])).to.deep.equal(["a", "b"]); + }); +}); diff --git a/test/tables/html.spec.ts b/test/tables/html.spec.ts new file mode 100644 index 0000000..e25c547 --- /dev/null +++ b/test/tables/html.spec.ts @@ -0,0 +1,102 @@ +import { expect } from "chai"; +import { + addTagClass, + decodeEntities, + findElements, + getTagAttribute, + removeTagAttribute, + replaceRanges, + scanTags, + setTagAttribute, + Tag, +} from "../../src/tables/html"; + +describe("html", () => { + describe("scanTags", () => { + it("reports every tag with its name, kind and position", () => { + const tags: [string, boolean, boolean, boolean, number][] = []; + scanTags('

    x

    ', (tag: Tag) => { + tags.push([tag.name, tag.isClosing, tag.isSelfClosing, tag.isComment, tag.start]); + }); + expect(tags).to.deep.equal([ + ["p", false, false, false, 0], + ["br", false, true, false, 14], + ["p", true, false, false, 19], + ["", false, true, true, 23], + ]); + }); + + it("does not end a tag at a > inside an attribute value", () => { + const texts: string[] = []; + scanTags('t', (tag) => { + texts.push(tag.text); + }); + expect(texts).to.deep.equal(['', ""]); + }); + }); + + describe("findElements", () => { + it("finds elements with their tags, content and positions", () => { + expect(findElements("

    a

    b
    ", ["div"])).to.deep.equal([ + { name: "div", start: 8, openTag: "
    ", innerStart: 13, inner: "b", closeTag: "
    ", end: 20 }, + ]); + }); + + it("skips elements inside a nested table", () => { + const elements = findElements("
    in
    ", ["td"]); + expect(elements.length).to.equal(1); + expect(elements[0].inner).to.equal("
    in
    "); + }); + }); + + describe("tag attributes", () => { + it("reads quoted, unquoted and bare attributes", () => { + expect(getTagAttribute('', "colspan")).to.equal("2"); + expect(getTagAttribute('', "rowspan")).to.equal("3"); + expect(getTagAttribute('', "hidden")).to.equal(""); + expect(getTagAttribute('', "class")).to.equal(null); + }); + + it("does not confuse an attribute with a longer name", () => { + expect(getTagAttribute('', "id")).to.equal(null); + }); + + it("sets a new attribute last and changes an existing one in place", () => { + expect(setTagAttribute('', "b", 2)).to.equal(''); + expect(setTagAttribute('', "a", "x")).to.equal(''); + expect(setTagAttribute('', "class", "c")).to.equal(''); + }); + + it("removes an attribute", () => { + expect(removeTagAttribute('', "a")).to.equal(''); + expect(removeTagAttribute("", "a")).to.equal(""); + }); + + it("adds a class like classList.add", () => { + expect(addTagClass("", "x")).to.equal(''); + expect(addTagClass('', "x")).to.equal(''); + expect(addTagClass('', "x")).to.equal(''); + }); + }); + + describe("decodeEntities", () => { + it("decodes named and numeric entities", () => { + expect(decodeEntities("a & b < AB  ")).to.equal("a & b < AB  "); + }); + + it("leaves unknown entities as they are", () => { + expect(decodeEntities("&foo;")).to.equal("&foo;"); + }); + }); + + describe("replaceRanges", () => { + it("applies replacements whatever their order", () => { + expect( + replaceRanges("0123456789", [ + { start: 7, end: 9, html: "X" }, + { start: 1, end: 3, html: "YY" }, + ]), + ).to.equal("0YY3456X9"); + }); + }); +}); diff --git a/test/tables/redlining.spec.ts b/test/tables/redlining.spec.ts new file mode 100644 index 0000000..a7fa12a --- /dev/null +++ b/test/tables/redlining.spec.ts @@ -0,0 +1,1032 @@ +import { expect } from "chai"; +import diff from "../../src/htmldiff"; + +describe("Structural table redlining", () => { + + describe("rows", () => { + it("marks whole added row when another cell is edited", () => { + const res = diff('

    A B C D E

    ','

    A B C D E

    '); + expect(res).to.equal( + '

    A B ' + + 'C' + + 'C D E

    '); + }); + + it("leaves row without cells empty", () => { + const res = diff( + '' + + '' + + '
    a
    ', + '' + + '' + + '
    ax
    '); + expect(res).to.equal( + '' + + '' + + '
    a' + + 'x
    '); + }); + + it("marks whole deleted row", () => { + const res = diff('

    text

    ','

    text

    '); + expect(res).to.equal( + '

    ' + + 'text' + + 'text

    '); + }); + + it("marks added row without changing edited row elsewhere", () => { + const res = diff( + '' + + '' + + '' + + '
    1a
    2b
    ', + '' + + '' + + '' + + '' + + '
    1a
    2x
    3b
    '); + expect(res).to.equal( + '' + + '' + + '' + + '' + + '
    1a
    2x
    ' + + '2' + + '3b
    '); + }); + + it("marks row inserted in the middle", () => { + const res = diff('

    text

    ','

    text

    '); + expect(res).to.equal( + '
    ' + + '

    text

    ' + + 'text

    '); + }); + + it("marks moved row as deleted and added", () => { + const res = diff( + '' + + '' + + '' + + '
    ab
    cd
    ', + '' + + '' + + '' + + '
    ab
    xy
    '); + expect(res).to.equal( + '' + + '' + + '' + + '' + + '
    ab
    cd
    xy
    '); + }); + + it("keeps filling an empty cell as cell edit", () => { + const res = diff('

    text here

    ','

    text HERE

    '); + expect(res).to.equal( + '

    text ' + + 'here' + + 'HERE

    '); + }); + + it("ignores line number column when matching rows", () => { + const res = diff('
    • buy milk
    ','
    • buy oat milk
    '); + expect(res).to.equal( + '
    • buy ' + + 'oat milk
    '); + }); + + it("fills an empty row in place as a cell edit", () => { + const res = diff('
    ab
    ', + '
    ab
    x
    yz
    '); + expect(res).to.equal( + '' + + '' + + '' + + '
    ab
    x
    yz
    '); + }); + + it("keeps the line number column when every kept row renumbers", () => { + const res = diff('
    1a
    2b
    3c
    4d
    5e
    ', + '
    1c
    2d
    3e
    4a
    5b
    '); + expect(res).to.equal( + '' + + '' + + '' + + '' + + '' + + '' + + '' + + '
    1a
    2b
    31c
    42d
    53e
    4a
    5b
    '); + }); + + it("marks fully changed row as deleted and added", () => { + const res = diff( + '
    ' + + '' + + '' + + '
    TC-1a
    TC-2b
    ', + '
    ' + + '' + + '' + + '
    TC-1a
    TC-9b
    '); + expect(res).to.equal( + '
    ' + + '' + + '' + + '' + + '
    TC-1a
    TC-2b
    TC-9b
    '); + }); + }); + + describe("rows in a document section", () => { + it("keeps rows of different items apart", () => { + const res = diff('
    • task
    ','
    • task
    '); + expect(res).to.equal( + '
    • ' + + 'task' + + 'task
    '); + }); + + it("spans the item over its old and new rows when none of them match", () => { + const res = diff('
    SPEC-1not covered
    ', + '
    ' + + '
    SPEC-1TC-3XTC-22other
    TC-2XTC-23other
    '); + expect(res).to.equal( + '
    SPEC-1not covered
    TC-3XTC-22other
    TC-2XTC-23other
    '); + }); + + it("spans the item over its old and new row when its only row is replaced", () => { + const res = diff('
    SPEC-7Missing trace to TC
    ', + '
    SPEC-7TC-5
    '); + expect(res).to.equal( + '
    SPEC-7Missing trace to TC
    TC-5
    '); + }); + + it("diffs the links cell when a linked item is added", () => { + const res = diff( + '
    ' + + '' + + '
    REQ-1REQ-2
    ', + '
    ' + + '' + + '
    REQ-1REQ-2 REQ-3
    '); + expect(res).to.equal( + '
    REQ-1REQ-2REQ-3
    '); + }); + + it("diffs the row of an item whose value changed and whose controls gained a ref", () => { + const res = diff('
    RISK-9 Grab BarInadequate grip2SPEC-18 IFU
    ', + '
    RISK-9 Grab BarInadequate grip3SPEC-8 RF SPEC-18 IFU
    '); + expect(res).to.equal( + '
    RISK-9 Grab BarInadequate grip23SPEC-8 RF SPEC-18 IFU
    '); + }); + + it("keeps the item cell and replaces the row when a ref in a cell was swapped", () => { + const res = diff('
    RISK-9 Grab BarInadequate gripSPEC-18 IFU
    ', + '
    RISK-9 Grab BarInadequate gripSPEC-8 RF
    '); + expect(res).to.equal( + '
    RISK-9 Grab BarInadequate gripSPEC-18 IFU
    Inadequate gripSPEC-8 RF
    '); + }); + + // a producer may give the cells naming a row their own identity: then those alone pair the rows + describe("keyed rows", () => { + const section = (rows: string): string => `
    ${rows}
    `; + + it("diffs the cells of an executed test when its keys are unchanged", () => { + const key = (itemRef: string): string => `${itemRef}`; + const row = (cells: string[]): string => `${key("TR-3")}${key("TC-1")}${key("XTC-11")}${cells.map((cell) => `${cell}`).join("")}`; + const res = diff(section(row(["", "", "0s", "pending"])), section(row(["2026/10/05", "jdoe", "4s", "passed"]))); + expect(res).to.equal( + '
    ' + + `${key("TR-3")}${key("TC-1")}${key("XTC-11")}` + + '' + + '' + + '' + + '' + + "
    2026/10/05jdoe0s4spendingpassed
    ", + ); + }); + + it("diffs the controls cell of a keyed risk row when a control was swapped", () => { + const row = (control: string): string => + `RISK-1 FireFire` + + `${control}`; + const res = diff(section(row("SPEC-2")), section(row("SPEC-7"))); + expect(res).to.equal( + '
    ' + + '' + + '' + + "
    RISK-1 FireFireSPEC-2SPEC-7
    ", + ); + }); + + it("keeps a keyed trace row apart when its trace was swapped", () => { + const row = (trace: string): string => `SPEC-6${trace} text`; + const res = diff(section(row("TC-3")), section(row("TC-4"))); + expect(res).to.equal( + '
    ' + + '' + + '' + + "
    SPEC-6TC-3 text
    TC-4 text
    ", + ); + }); + }); + + it("keeps executions of different test cases apart", () => { + const res = diff('
    ' + + '
    SPEC-1TC-3TR-4other
    TC-2TR-4other
    ', + '
    ' + + '
    SPEC-1TC-3TR-4other
    TC-9TR-4other
    '); + expect(res).to.equal( + '
    SPEC-1TC-3TR-4other
    TC-2TR-4other
    TC-9TR-4other
    '); + }); + + it("empties the trace cell of an item as a cell edit", () => { + const res = diff('
    SPEC-6TC-3 Mounting
    TC-4 App
    ', + '
    SPEC-6
    '); + expect(res).to.equal( + '
    SPEC-6TC-3 Mounting
    TC-4 App
    '); + }); + + it("fills the empty trace cell of an item as a cell edit", () => { + const res = diff('
    SPEC-6
    ', + '
    SPEC-6TC-3 Mounting
    '); + expect(res).to.equal( + '
    SPEC-6TC-3 Mounting
    '); + }); + + it("splits row when its item changes and links stay", () => { + const res = diff( + '
    ' + + '' + + '
    SPEC-3REQ-1 REQ-2 REQ-4
    ', + '
    ' + + '' + + '
    SPEC-2REQ-1 REQ-2 REQ-4
    '); + expect(res).to.equal( + '
    ' + + '' + + '' + + '
    SPEC-3REQ-1 REQ-2 REQ-4
    SPEC-2REQ-1 REQ-2 REQ-4
    '); + }); + + it("replaces the row of an item whose content all changed", () => { + const res = diff( + '
    ' + + '' + + '
    TC-1ax
    ', + '
    ' + + '' + + '' + + '
    TC-1cz
    TC-2by
    '); + expect(res).to.equal( + '
    TC-1ax
    cz
    TC-2by
    '); + }); + + it("keeps same shaped groups of different items apart", () => { + const res = diff( + '
    ' + + '' + + '' + + '
    UC-41VAL-3
    VAL-4
    ', + '
    ' + + '' + + '' + + '
    PRODREQ-193COMP-1
    COMP-2
    '); + expect(res).to.equal( + '
    ' + + '' + + '' + + '' + + '' + + '
    UC-41VAL-3
    VAL-4
    PRODREQ-193COMP-1
    COMP-2
    '); + }); + + it("marks a gained trace as an added cell under its source and deletes the other source whole", () => { + const res = diff( + '
    ' + + '' + + '' + + '' + + '
    UC-30UREQ-114
    UC-39UREQ-116
    UREQ-312
    ', + '
    ' + + '' + + '' + + '
    UC-30UREQ-114
    UREQ-312
    '); + expect(res).to.equal( + '
    UC-30UREQ-114
    UREQ-312
    UC-39UREQ-116
    UREQ-312
    '); + }); + + it("marks a lost trace as a deleted cell under the kept rows of its group", () => { + const res = diff( + '
    ' + + '' + + '' + + '' + + '
    SPEC-15TC-1
    TC-2
    TC-4
    ', + '
    ' + + '' + + '' + + '
    SPEC-15TC-1
    TC-2
    '); + expect(res).to.equal( + '
    SPEC-15TC-1
    TC-2
    TC-4
    '); + }); + + it("moves a lost trace under the kept rows of its group", () => { + const res = diff( + '
    ' + + '' + + '' + + '' + + '
    SPEC-15TC-1
    TC-2
    TC-3
    ', + '
    ' + + '' + + '' + + '
    SPEC-15TC-1
    TC-3
    '); + expect(res).to.equal( + '
    SPEC-15TC-1
    TC-3
    TC-2
    '); + }); + + it("moves the merged cell onto the first kept row when the first row of its group is deleted", () => { + const res = diff( + '
    ' + + '' + + '' + + '' + + '
    SPEC-3TC-2
    TC-3
    TC-5
    ', + '
    ' + + '' + + '' + + '
    SPEC-3TC-3
    TC-5
    '); + expect(res).to.equal( + '
    SPEC-3TC-3
    TC-5
    TC-2
    '); + }); + + it("spans the item over its lost traces when the group shrinks to one row", () => { + const res = diff( + '
    ' + + '' + + '' + + '' + + '
    TR-3VER-213
    VER-214
    VER-211
    ', + '
    ' + + '' + + '
    TR-3VER-211
    '); + expect(res).to.equal( + '
    TR-3VER-211
    VER-213
    VER-214
    '); + }); + + it("moves the merged cell onto the first kept row when the first row of its group is added", () => { + const res = diff('
    SPEC-1TC-1
    TC-3
    ', + '
    SPEC-1TC-0
    TC-1
    TC-3
    '); + expect(res).to.equal( + '
    SPEC-1TC-1
    TC-3
    TC-0
    '); + }); + + it("moves a gained trace under the kept rows of its group", () => { + const res = diff('
    SPEC-1TC-1
    TC-3
    ', + '
    SPEC-1TC-1
    TC-2
    TC-3
    '); + expect(res).to.equal( + '
    SPEC-1TC-1
    TC-3
    TC-2
    '); + }); + + it("alternates replaced rows", () => { + const res = diff( + '
    ' + + '' + + '' + + '
    UC-41VAL-3
    UC-42VAL-5
    ', + '
    ' + + '' + + '' + + '
    PRODREQ-193COMP-1
    PRODREQ-194COMP-2
    '); + expect(res).to.equal( + '
    ' + + '' + + '' + + '' + + '' + + '
    UC-41VAL-3
    PRODREQ-193COMP-1
    UC-42VAL-5
    PRODREQ-194COMP-2
    '); + }); + + it("alternates replaced groups whole", () => { + const res = diff( + '
    ' + + '' + + '' + + '' + + '
    UC-41VAL-3
    VAL-4
    UC-42VAL-5
    ', + '
    ' + + '' + + '' + + '' + + '
    PRODREQ-193COMP-1
    COMP-2
    PRODREQ-194COMP-3
    '); + expect(res).to.equal( + '
    ' + + '' + + '' + + '' + + '' + + '' + + '' + + '
    UC-41VAL-3
    VAL-4
    PRODREQ-193COMP-1
    COMP-2
    UC-42VAL-5
    PRODREQ-194COMP-3
    '); + }); + + it("appends rows left over after alternating", () => { + const res = diff( + '
    ' + + '' + + '' + + '
    UC-41VAL-3
    UC-42VAL-5
    ', + '
    ' + + '' + + '
    PRODREQ-193COMP-1
    '); + expect(res).to.equal( + '
    ' + + '' + + '' + + '' + + '
    UC-41VAL-3
    PRODREQ-193COMP-1
    UC-42VAL-5
    '); + }); + }); + + describe("columns", () => { + it("marks added column on its cells", () => { + const res = diff( + '' + + '' + + '' + + '
    a
    ', + '' + + '' + + '' + + '
    aX
    Y
    '); + expect(res).to.equal( + '' + + '' + + '' + + '
    aX
    Y
    '); + }); + + it("marks column inserted in the middle", () => { + const res = diff( + '' + + '' + + '' + + '
    ab
    cd
    ', + '' + + '' + + '' + + '
    anb
    cmd
    '); + expect(res).to.equal( + '' + + '' + + '' + + '
    anb
    cmd
    '); + }); + + it("marks added column and added row together", () => { + const res = diff( + '' + + '' + + '' + + '
    a
    c
    ', + '' + + '' + + '' + + '' + + '
    ab
    cd
    ef
    '); + expect(res).to.equal( + '' + + '' + + '' + + '' + + '
    ab
    cd
    ef
    '); + }); + + it("adds placeholder for added column in deleted row", () => { + const res = diff( + '' + + '' + + '' + + '
    a
    c
    ', + '' + + '' + + '
    ab
    '); + expect(res).to.equal( + '' + + '' + + '' + + '
    ab
    c
    '); + }); + + it("marks replaced column as deleted and added", () => { + const res = diff( + '' + + '' + + '' + + '
    abc
    def
    ', + '' + + '' + + '' + + '
    a1c
    d2f
    '); + expect(res).to.equal( + '' + + '' + + '' + + '
    ab1c
    de2f
    '); + }); + + it("keeps a column paired when rows were added around its values", () => { + const res = diff('' + + '' + + '' + + '' + + '
    jkvbijk ihvjkbkjkjbkjjbkj
    n m kjkj jhkkj kj,m kjn kj kjbkjb kj
    ', + '' + + '' + + '' + + '' + + '' + + '
    asdvasdvsavsdv
    asdvsdvsdvasdvsdvsdv
    jkvbijk ihvkjbkjjbkj
    n m kjkj kj,m kjn kj kjbkjb kj
    sdvsadv
    '); + expect(res).to.equal( + '' + + '' + + '' + + '' + + '' + + '
    asdvasdvsavsdv
    asdvsdvsdvasdvsdvsdv
    jkvbijk ihvjkbkjkjbkjjbkj
    n m kjkj jhkkj kj,m kjn kj kjbkjb kj
    sdvsadv
    '); + }); + + it("marks moved column as deleted and added", () => { + const res = diff( + '' + + '' + + '' + + '
    ab
    cd
    ', + '' + + '' + + '' + + '
    ba
    dc
    '); + expect(res).to.equal( + '' + + '' + + '' + + '
    aba
    cdc
    '); + }); + + it("matches columns by header when cells changed", () => { + const res = diff( + '' + + '' + + '' + + '
    NameStatus
    aopen
    ', + '' + + '' + + '' + + '
    NameStatus
    adone
    '); + expect(res).to.equal( + '' + + '' + + '' + + '
    NameStatus
    a' + + 'open' + + 'done
    '); + }); + + it("marks the deleted column on its col", () => { + const res = diff( + '' + + '' + + '
    ab
    ', + '' + + '' + + '
    a
    '); + expect(res).to.equal( + '' + + '' + + '
    ab
    '); + }); + + it("marks deleted column and added row together", () => { + const res = diff( + '' + + '' + + '' + + '
    ab
    cd
    ', + '' + + '' + + '' + + '' + + '
    a
    c
    e
    '); + expect(res).to.equal( + '' + + '' + + '' + + '' + + '
    ab
    cd
    e
    '); + }); + }); + + describe("merged cells", () => { + it("diffs cells by position when the shape is unchanged", () => { + const res = diff( + '' + + '' + + '' + + '' + + '
    CategoryZone 1Total
    BeforeNumber of34
    Risk percentage75%100%
    ', + '' + + '' + + '' + + '' + + '
    CategoryZone 1Total
    BeforeNumber of35
    Risk percentage60%100%
    '); + expect(res).to.equal( + '' + + '' + + '' + + '' + + '
    CategoryZone 1Total
    BeforeNumber of3' + + '4' + + '5
    Risk percentage' + + '75%' + + '60%100%
    '); + }); + + it("marks added body row under a merged header", () => { + const res = diff( + '' + + '' + + '' + + '
    RisksZone
    R-1a1
    ', + '' + + '' + + '' + + '' + + '
    RisksZone
    R-1a1
    R-2b2
    '); + expect(res).to.equal( + '' + + '' + + '' + + '' + + '
    RisksZone
    R-1a1
    R-2b2
    '); + }); + + // merged cells are layout in a text: a table whose merged cells changed is another table + it("wraps a rich text table whole when a cell was split", () => { + const oldTable = '
    a
    cd
    '; + const newTable = '
    ab
    cd
    '; + const res = diff(oldTable, newTable); + expect(res).to.equal( + `${oldTable.replace("", '
    ')}` + + `${newTable.replace("
    ", '
    ')}`); + }); + + it("wraps a rich text table whole when cells were merged", () => { + const oldTable = '
    BrakePress
    SteeringTurn
    HornPress center
    '; + const newTable = '
    BrakePress
    Steering
    Horn
    Turn
    Press center
    '; + const res = diff(oldTable, newTable); + expect(res).to.equal( + `${oldTable.replace("", '
    ')}` + + `${newTable.replace("
    ", '
    ')}`); + }); + + it("wraps a rich text table whole when a group lost a row", () => { + const oldTable = '
    ga1
    a2
    a3
    '; + const newTable = '
    ga1
    a3
    '; + const res = diff(oldTable, newTable); + expect(res).to.equal( + `${oldTable.replace("", '
    ')}` + + `${newTable.replace("
    ", '
    ')}`); + }); + + it("wraps a rich text table whole when a title grew over an added column", () => { + const oldTable = '
    title
    ab
    '; + const newTable = '
    title
    axb
    '; + const res = diff(oldTable, newTable); + expect(res).to.equal( + `${oldTable.replace("", '
    ')}` + + `${newTable.replace("
    ", '
    ')}`); + }); + + it("still merges a section table when a group lost a row", () => { + const res = diff( + '
    ga1
    a2
    a3
    ', + '
    ga1
    a3
    '); + expect(res).to.equal( + '
    ' + + '
    ga1
    a3
    a2
    '); + }); + + it("keeps colspan of title row unchanged on merge", () => { + const res = diff( + '' + + '' + + '' + + '
    title
    abc
    ', + '' + + '' + + '' + + '' + + '
    title
    abc
    def
    '); + expect(res).to.equal( + '' + + '' + + '' + + '' + + '
    title
    abc
    def
    '); + }); + + it("keeps rowspan when a row is added below the group", () => { + const res = diff( + '' + + '' + + '' + + '
    ga
    b
    ', + '' + + '' + + '' + + '' + + '
    ga
    b
    xy
    '); + expect(res).to.equal( + '' + + '' + + '' + + '' + + '
    ga
    b
    xy
    '); + }); + + it("keeps rowspan of deleted group in a section", () => { + const res = diff( + '
    ' + + '' + + '' + + '' + + '
    ga
    b
    xy
    ', + '
    ' + + '' + + '
    xy
    '); + expect(res).to.equal( + '
    ' + + '' + + '' + + '' + + '
    ga
    b
    xy
    '); + }); + + it("wraps a rich text table whole when a group was deleted", () => { + const oldTable = '
    ga
    b
    xy
    '; + const newTable = '
    xy
    '; + const res = diff(oldTable, newTable); + expect(res).to.equal( + `${oldTable.replace("", '
    ')}` + + `${newTable.replace("
    ", '
    ')}`); + }); + + it("keeps merged cells when a body row is added under them", () => { + const res = diff( + '
    ' + + '' + + '' + + '
    BeforeNumber of3
    Risk percentage75%
    ', + '' + + '' + + '' + + '' + + '
    BeforeNumber of3
    Risk percentage75%
    AfterNumber of4
    '); + expect(res).to.equal( + '' + + '' + + '' + + '' + + '
    BeforeNumber of3
    Risk percentage75%
    AfterNumber of4
    '); + }); + }); + + describe("tables", () => { + it("wraps table present in one version whole", () => { + const res = diff('

    x

    ', + '

    x

    ' + + '' + + '
    a
    '); + expect(res).to.equal( + '

    x

    ' + + '' + + '' + + '
    a
    '); + }); + + it("pairs tables by content when one is removed", () => { + const res = diff( + '
    ' + + '' + + '
    a
    ' + + '' + + '
    b
    ', + '
    ' + + '' + + '
    bx
    '); + expect(res).to.equal( + '
    ' + + '' + + '' + + '
    a
    ' + + '' + + '
    bx
    '); + }); + + it("wraps tables whole when none share content", () => { + const res = diff( + '
    ' + + '' + + '
    a
    ' + + '' + + '
    b
    ', + '
    ' + + '' + + '
    x
    '); + expect(res).to.equal( + '
    ' + + '' + + '' + + '
    a
    ' + + '' + + '' + + '
    b
    ' + + '' + + '' + + '
    x
    '); + }); + + it("wraps replaced table whole", () => { + const res = diff( + '
    ' + + '' + + '' + + '
    ab
    cd
    ', + '
    ' + + '' + + '
    xyz
    '); + expect(res).to.equal( + '
    ' + + '' + + '' + + '' + + '
    ab
    cd
    ' + + '' + + '' + + '
    xyz
    '); + }); + + it("wraps a rewritten table with the same columns whole", () => { + const res = diff('
    ab
    cd
    ', + '
    xy
    zw
    '); + expect(res).to.equal( + '
    ab
    cd
    ' + + '
    xy
    zw
    '); + }); + + it("wraps replaced table whole when one value happens to match", () => { + const res = diff( + '
    ' + + '' + + '' + + '
    absame
    cde
    ', + '
    ' + + '' + + '' + + '
    same
    xy
    '); + expect(res).to.equal( + '
    ' + + '' + + '' + + '' + + '
    absame
    cde
    ' + + '' + + '' + + '' + + '
    same
    xy
    '); + }); + + it("wraps section tables whole when their headers differ", () => { + const headed = (header: string, item: string): string => + `
    ${header}Executed Test Case
    ${item}TC-4
    `; + const res = diff(headed("Test run", "TR-4"), headed("Items", "SPEC-2")); + expect(res).to.equal( + "
    " + + "
    Test runExecuted Test Case
    TR-4TC-4
    " + + "
    ItemsExecuted Test Case
    SPEC-2TC-4
    " + + "
    ", + ); + }); + + it("wraps tables whole when only repeated filler values recur", () => { + const table = (cells: string[][]): string => + `${cells.map((row) => `${row.map((cell) => ``).join("")}`).join("")}
    ${cell}
    `; + const oldTable = table([ + ["sdvvsdvsdv", "sd vsdv", "sfv", "sdvsd"], + ["sd vsdv", "", "sd vsdv", ""], + ["sdvsd v", "sdvsdvsdv", "sdv sdv", "sdvsdv"], + ["s dvsdv", "", "sd vsd v", "sdvsdv"], + ]); + const newTable = table([ + ["sdvsdvsdv", "sdvsdv", "sdvsdv"], + ["sdvsdv", "sdvsdv", "sdvsdvsdv"], + ]); + const res = diff(`
    ${oldTable}
    `, `
    ${newTable}
    `); + expect(res).to.equal( + "
    " + + `${oldTable.replace("", '
    ')}` + + `${newTable.replace("
    ", '
    ')}` + + "", + ); + }); + + it("pairs tables by their own id", () => { + const res = diff( + '
    ' + + '' + + '
    a
    ' + + '' + + '
    b
    ', + '
    ' + + '' + + '
    bx
    '); + expect(res).to.equal( + '
    ' + + '' + + '' + + '
    a
    ' + + '' + + '
    bx
    '); + }); + + it("never pairs tables with different ids", () => { + const res = diff( + '
    ' + + '' + + '
    a
    ', + '
    ' + + '' + + '
    a
    '); + expect(res).to.equal( + '
    ' + + '' + + '' + + '
    a
    ' + + '' + + '' + + '
    a
    '); + }); + + it("marks added rows inside table with same id", () => { + const res = diff( + '
    ' + + '' + + '
    a
    ', + '
    ' + + '' + + '' + + '
    a
    b
    '); + expect(res).to.equal( + '
    ' + + '' + + '' + + '
    a
    b
    '); + }); + + it("diffs nested table as cell content", () => { + const res = diff( + '' + + '' + + '
    x' + + '' + + '
    a
    ', + '' + + '' + + '
    x' + + '' + + '
    b
    '); + expect(res).to.equal( + '' + + '' + + '
    x' + + '' + + '
    ' + + 'a' + + 'b
    '); + }); + }); + + describe("cell diffs", () => { + it("passes className and dataPrefix into the cells", () => { + const res = diff('
    ab
    ', + '
    ac
    ', 'diff-cls', 'pre'); + expect(res).to.equal('
    a' + + 'b' + + 'c
    '); + }); + }); +}); diff --git a/test/tables/similarity.spec.ts b/test/tables/similarity.spec.ts new file mode 100644 index 0000000..8a0ed1c --- /dev/null +++ b/test/tables/similarity.spec.ts @@ -0,0 +1,69 @@ +import { expect } from "chai"; +import { cellSimilarity, countsSize, countValues, distinctSharedShare, retainedShare, sharedShare, valueOverlap } from "../../src/tables/similarity"; + +describe("similarity", () => { + describe("countValues", () => { + it("counts values, empty ones left out", () => { + const counts = countValues(["a", "", "b", "a"]); + expect(counts).to.deep.equal({ a: 2, b: 1 }); + expect(countsSize(counts)).to.equal(2); + }); + }); + + describe("retainedShare", () => { + it("is the share of old values still there", () => { + expect(retainedShare({ a: 1, b: 1 }, { a: 1, c: 1 })).to.equal(0.5); + expect(retainedShare({}, { a: 1 })).to.equal(0); + }); + }); + + describe("valueOverlap", () => { + it("is the shared share of all values", () => { + expect(valueOverlap({ a: 1, b: 1 }, { a: 1, c: 1, d: 1 })).to.equal(0.25); + }); + + it("is NaN when both sides are empty", () => { + expect(isNaN(valueOverlap({}, {}))).to.equal(true); + }); + }); + + describe("sharedShare", () => { + it("judges on the smaller side", () => { + expect(sharedShare({ a: 1, b: 1 }, { a: 1, b: 1, c: 1, d: 1, e: 1 })).to.equal(1); + expect(sharedShare({ a: 1, b: 1, c: 1, d: 1 }, { a: 1, x: 1 })).to.equal(0.5); + }); + + it("is 0 when one side is empty and NaN when both are", () => { + expect(sharedShare({}, { a: 1 })).to.equal(0); + expect(isNaN(sharedShare({}, {}))).to.equal(true); + }); + }); + + describe("distinctSharedShare", () => { + it("counts a shared value once, against the smaller side's cells", () => { + expect(distinctSharedShare({ a: 2, b: 1, c: 1 }, { a: 4, b: 2 })).to.equal(0.5); + expect(distinctSharedShare({ a: 1, b: 1, c: 1, d: 1 }, { a: 1, b: 1 })).to.equal(1); + }); + + it("is 0 when one side is empty and NaN when both are", () => { + expect(distinctSharedShare({}, { a: 1 })).to.equal(0); + expect(isNaN(distinctSharedShare({}, {}))).to.equal(true); + }); + }); + + describe("cellSimilarity", () => { + it("is the share of words in common", () => { + expect(cellSimilarity("a b c", "a b d")).to.equal(0.5); + expect(cellSimilarity("same", "same")).to.equal(1); + expect(cellSimilarity("x", "y")).to.equal(0); + }); + + it("splits signatures on the id separator too", () => { + expect(cellSimilarity("a b|X", "a c|X")).to.equal(0.5); + }); + + it("is 1 for two empty cells", () => { + expect(cellSimilarity("", "")).to.equal(1); + }); + }); +}); diff --git a/tsconfig.build.json b/tsconfig.build.json new file mode 100644 index 0000000..a996921 --- /dev/null +++ b/tsconfig.build.json @@ -0,0 +1,16 @@ +{ + "extends": "./tsconfig.json", + "compilerOptions": { + "rootDir": "src", + "outDir": "js", + "declaration": true, + "noEmit": false, + "noEmitOnError": true, + "types": [ + "node" + ] + }, + "include": [ + "src/**/*.ts" + ] +} diff --git a/tsconfig.cli.json b/tsconfig.cli.json new file mode 100644 index 0000000..03c2690 --- /dev/null +++ b/tsconfig.cli.json @@ -0,0 +1,16 @@ +{ + "extends": "./tsconfig.json", + "compilerOptions": { + "rootDir": ".", + "outDir": ".", + "noEmit": false, + "noEmitOnError": true, + "types": [ + "node" + ] + }, + "include": [], + "files": [ + "htmldiff-cli.ts" + ] +} diff --git a/tsconfig.json b/tsconfig.json index 13593a5..0089110 100644 --- a/tsconfig.json +++ b/tsconfig.json @@ -1,34 +1,27 @@ { "compilerOptions": { - "target": "ESNext", + "target": "ES2019", "module": "CommonJS", + "moduleResolution": "node", "esModuleInterop": true, + "allowSyntheticDefaultImports": true, "skipLibCheck": true, "forceConsistentCasingInFileNames": true, "strict": true, - "strictPropertyInitialization": false, - "declaration": true, - "allowSyntheticDefaultImports": true, - "noImplicitAny": true, - "noImplicitThis": true, "noImplicitReturns": true, "noImplicitOverride": true, "noFallthroughCasesInSwitch": true, "noUnusedLocals": true, "noUnusedParameters": true, - "suppressImplicitAnyIndexErrors": true, - "strictNullChecks": true, - "sourceMap": false, - "noEmitOnError": true, - "removeComments": true, - "noEmitHelpers": false, - "listEmittedFiles": false, - "listFiles": false, - "pretty": true, - "experimentalDecorators": false, - "emitDecoratorMetadata": false + "noEmit": true, + "types": [ + "node", + "mocha" + ], + "pretty": true }, - "files": [ - "htmldiff-cli.ts" + "include": [ + "src/**/*.ts", + "test/**/*.ts" ] -} \ No newline at end of file +} From 1bd23284ca60a733e102c874dd52fd9e334b872d Mon Sep 17 00:00:00 2001 From: ragool Date: Mon, 5 Oct 2026 13:58:57 +0200 Subject: [PATCH 2/6] do not add js files --- .gitignore | 1 + js/core/atomicTags.d.ts | 63 ------ js/core/atomicTags.js | 81 -------- js/core/diff.d.ts | 29 --- js/core/diff.js | 43 ---- js/core/matching.d.ts | 71 ------- js/core/matching.js | 299 --------------------------- js/core/operations.d.ts | 23 --- js/core/operations.js | 88 -------- js/core/rendering.d.ts | 112 ---------- js/core/rendering.js | 315 ---------------------------- js/core/tokens.d.ts | 54 ----- js/core/tokens.js | 363 --------------------------------- js/htmldiff.d.ts | 29 --- js/htmldiff.js | 59 ------ js/tables/Cell.d.ts | 100 --------- js/tables/Cell.js | 202 ------------------ js/tables/ColumnAligner.d.ts | 43 ---- js/tables/ColumnAligner.js | 118 ----------- js/tables/MergedCells.d.ts | 74 ------- js/tables/MergedCells.js | 294 -------------------------- js/tables/Row.d.ts | 66 ------ js/tables/Row.js | 84 -------- js/tables/RowAligner.d.ts | 50 ----- js/tables/RowAligner.js | 157 -------------- js/tables/SequenceAligner.d.ts | 94 --------- js/tables/SequenceAligner.js | 167 --------------- js/tables/Table.d.ts | 78 ------- js/tables/Table.js | 207 ------------------- js/tables/TableAligner.d.ts | 35 ---- js/tables/TableAligner.js | 96 --------- js/tables/TableMerger.d.ts | 77 ------- js/tables/TableMerger.js | 306 --------------------------- js/tables/TableRedlining.d.ts | 34 --- js/tables/TableRedlining.js | 86 -------- js/tables/TableVersion.d.ts | 143 ------------- js/tables/TableVersion.js | 248 ---------------------- js/tables/constants.d.ts | 13 -- js/tables/constants.js | 14 -- js/tables/helpers.d.ts | 26 --- js/tables/helpers.js | 45 ---- js/tables/html.d.ts | 90 -------- js/tables/html.js | 154 -------------- js/tables/index.d.ts | 2 - js/tables/index.js | 5 - js/tables/similarity.d.ts | 54 ----- js/tables/similarity.js | 144 ------------- package.json | 8 +- 48 files changed, 5 insertions(+), 4939 deletions(-) delete mode 100644 js/core/atomicTags.d.ts delete mode 100644 js/core/atomicTags.js delete mode 100644 js/core/diff.d.ts delete mode 100644 js/core/diff.js delete mode 100644 js/core/matching.d.ts delete mode 100644 js/core/matching.js delete mode 100644 js/core/operations.d.ts delete mode 100644 js/core/operations.js delete mode 100644 js/core/rendering.d.ts delete mode 100644 js/core/rendering.js delete mode 100644 js/core/tokens.d.ts delete mode 100644 js/core/tokens.js delete mode 100644 js/htmldiff.d.ts delete mode 100644 js/htmldiff.js delete mode 100644 js/tables/Cell.d.ts delete mode 100644 js/tables/Cell.js delete mode 100644 js/tables/ColumnAligner.d.ts delete mode 100644 js/tables/ColumnAligner.js delete mode 100644 js/tables/MergedCells.d.ts delete mode 100644 js/tables/MergedCells.js delete mode 100644 js/tables/Row.d.ts delete mode 100644 js/tables/Row.js delete mode 100644 js/tables/RowAligner.d.ts delete mode 100644 js/tables/RowAligner.js delete mode 100644 js/tables/SequenceAligner.d.ts delete mode 100644 js/tables/SequenceAligner.js delete mode 100644 js/tables/Table.d.ts delete mode 100644 js/tables/Table.js delete mode 100644 js/tables/TableAligner.d.ts delete mode 100644 js/tables/TableAligner.js delete mode 100644 js/tables/TableMerger.d.ts delete mode 100644 js/tables/TableMerger.js delete mode 100644 js/tables/TableRedlining.d.ts delete mode 100644 js/tables/TableRedlining.js delete mode 100644 js/tables/TableVersion.d.ts delete mode 100644 js/tables/TableVersion.js delete mode 100644 js/tables/constants.d.ts delete mode 100644 js/tables/constants.js delete mode 100644 js/tables/helpers.d.ts delete mode 100644 js/tables/helpers.js delete mode 100644 js/tables/html.d.ts delete mode 100644 js/tables/html.js delete mode 100644 js/tables/index.d.ts delete mode 100644 js/tables/index.js delete mode 100644 js/tables/similarity.d.ts delete mode 100644 js/tables/similarity.js diff --git a/.gitignore b/.gitignore index 9865a1a..aebae61 100644 --- a/.gitignore +++ b/.gitignore @@ -1,3 +1,4 @@ node_modules htmldiff-cli.js +js/ .DS_Store diff --git a/js/core/atomicTags.d.ts b/js/core/atomicTags.d.ts deleted file mode 100644 index 5bb5ed8..0000000 --- a/js/core/atomicTags.d.ts +++ /dev/null @@ -1,63 +0,0 @@ -/** - * Atomic tags are elements whose child nodes are never compared: the whole element is one - * token. Which tags are atomic changes while a diff runs (the caller's list for the outer - * diff, a reduced list inside a recursive inner diff), so the active regular expression is - * kept here and read by the tokenizer and the renderer. - */ -/** - * The default atomic tags. The tag name must be followed by a delimiter (not a \b word - * boundary): the tokenizer matches against partially read tags, and a word boundary would - * match at the end of an incomplete name, e.g. detecting '' as the atomic tag 'a' - * while reading '' - * into a nested child. The leading \s prevents matching 'x-data-htmldiff-id'. - */ -export declare const dataHtmlDiffIdRegExp: RegExp; -/** - * Opt-in marker for the recursive inner diff. When two matched atomic tokens (typically - * matched by data-htmldiff-id) have equal keys but different content, an element carrying - * this attribute gets its inner HTML diffed recursively instead of being rendered as is. - * The attribute must appear in the element's opening tag. Captures the attribute value; - * a bare attribute or any value other than "false" enables the opt-in. - */ -export declare const dataHtmlDiffInnerDiffRegExp: RegExp; -/** - * Per-element override for the atomic tags used inside a recursive inner diff. The value - * is a comma separated tag name list, like the atomicTags parameter of the diff function; - * an empty value means no tag name is atomic. - */ -export declare const dataHtmlDiffInnerDiffAtomicTagsRegExp: RegExp; -/** - * The atomic tags regular expression the running diff uses. - * @returns The active regular expression. - */ -export declare function getAtomicTagsRegExp(): RegExp; -/** - * Switches the atomic tags for the diff that is about to run. - * @param regExp The regular expression to match the start of an atomic tag. - */ -export declare function setAtomicTagsRegExp(regExp: RegExp): void; -/** - * Builds the atomic tags regular expression from a comma separated tag name list. - * @param atomicTags Comma separated list of tag names, e.g. 'head,script,style'. - * @returns The regular expression matching the start of those tags. - */ -export declare function buildAtomicTagsRegExp(atomicTags: string): RegExp; -/** - * Checks if the current word is the beginning of an atomic tag: one of the active atomic - * tags, or any element with a data-htmldiff-id of its own. - * @param word The characters of the current token read so far. - * @returns The name of the atomic tag if the word will be an atomic tag, null otherwise. - */ -export declare function isStartOfAtomicTag(word: string): string | null; diff --git a/js/core/atomicTags.js b/js/core/atomicTags.js deleted file mode 100644 index 5b1f1e9..0000000 --- a/js/core/atomicTags.js +++ /dev/null @@ -1,81 +0,0 @@ -"use strict"; -/** - * Atomic tags are elements whose child nodes are never compared: the whole element is one - * token. Which tags are atomic changes while a diff runs (the caller's list for the outer - * diff, a reduced list inside a recursive inner diff), so the active regular expression is - * kept here and read by the tokenizer and the renderer. - */ -Object.defineProperty(exports, "__esModule", { value: true }); -exports.isStartOfAtomicTag = exports.buildAtomicTagsRegExp = exports.setAtomicTagsRegExp = exports.getAtomicTagsRegExp = exports.dataHtmlDiffInnerDiffAtomicTagsRegExp = exports.dataHtmlDiffInnerDiffRegExp = exports.dataHtmlDiffIdRegExp = exports.noAtomicTagsRegExp = exports.defaultInnerDiffAtomicTagsRegExp = exports.defaultAtomicTagsRegExp = void 0; -/** - * The default atomic tags. The tag name must be followed by a delimiter (not a \b word - * boundary): the tokenizer matches against partially read tags, and a word boundary would - * match at the end of an incomplete name, e.g. detecting '' as the atomic tag 'a' - * while reading ']"); -/** - * Atomic tags used inside a recursive inner diff unless the element overrides them via - * data-htmldiff-inner-diff-atomic-tags: the default list without 'a'. - */ -exports.defaultInnerDiffAtomicTagsRegExp = new RegExp("^<(iframe|object|math|svg|script|video|head|style)[\\s/>]"); -/** Matches no tag at all: used when data-htmldiff-inner-diff-atomic-tags is empty. */ -exports.noAtomicTagsRegExp = /^<(?!)/; -/** - * Matches an element whose own opening tag carries data-htmldiff-id; captures tag name and - * attribute value. The skip before the attribute is quote aware, so it cannot run past '>' - * into a nested child. The leading \s prevents matching 'x-data-htmldiff-id'. - */ -exports.dataHtmlDiffIdRegExp = /^<([a-z-]+)(?:[^>"']|"[^"]*"|'[^']*')*\sdata-htmldiff-id=["']?((?:.(?!["']?\s+(?:\S+)=|\s*\/?[>"']))*.)["']?/; -/** - * Opt-in marker for the recursive inner diff. When two matched atomic tokens (typically - * matched by data-htmldiff-id) have equal keys but different content, an element carrying - * this attribute gets its inner HTML diffed recursively instead of being rendered as is. - * The attribute must appear in the element's opening tag. Captures the attribute value; - * a bare attribute or any value other than "false" enables the opt-in. - */ -exports.dataHtmlDiffInnerDiffRegExp = /^<[^>]*\sdata-htmldiff-inner-diff(?:\s*=\s*["']?([^"'\s/>]*)|(?=[\s/>]))/; -/** - * Per-element override for the atomic tags used inside a recursive inner diff. The value - * is a comma separated tag name list, like the atomicTags parameter of the diff function; - * an empty value means no tag name is atomic. - */ -exports.dataHtmlDiffInnerDiffAtomicTagsRegExp = /^<[^>]*\sdata-htmldiff-inner-diff-atomic-tags\s*=\s*["']([^"']*)["']/; -let atomicTagsRegExp = exports.defaultAtomicTagsRegExp; -/** - * The atomic tags regular expression the running diff uses. - * @returns The active regular expression. - */ -function getAtomicTagsRegExp() { - return atomicTagsRegExp; -} -exports.getAtomicTagsRegExp = getAtomicTagsRegExp; -/** - * Switches the atomic tags for the diff that is about to run. - * @param regExp The regular expression to match the start of an atomic tag. - */ -function setAtomicTagsRegExp(regExp) { - atomicTagsRegExp = regExp; -} -exports.setAtomicTagsRegExp = setAtomicTagsRegExp; -/** - * Builds the atomic tags regular expression from a comma separated tag name list. - * @param atomicTags Comma separated list of tag names, e.g. 'head,script,style'. - * @returns The regular expression matching the start of those tags. - */ -function buildAtomicTagsRegExp(atomicTags) { - // Require a delimiter after the name (see defaultAtomicTagsRegExp on why not \b). - return new RegExp("^<(" + atomicTags.replace(/\s*/g, "").replace(/,/g, "|") + ")[\\s/>]"); -} -exports.buildAtomicTagsRegExp = buildAtomicTagsRegExp; -/** - * Checks if the current word is the beginning of an atomic tag: one of the active atomic - * tags, or any element with a data-htmldiff-id of its own. - * @param word The characters of the current token read so far. - * @returns The name of the atomic tag if the word will be an atomic tag, null otherwise. - */ -function isStartOfAtomicTag(word) { - const result = atomicTagsRegExp.exec(word) || exports.dataHtmlDiffIdRegExp.exec(word); - return result ? result[1] : null; -} -exports.isStartOfAtomicTag = isStartOfAtomicTag; diff --git a/js/core/diff.d.ts b/js/core/diff.d.ts deleted file mode 100644 index 7518114..0000000 --- a/js/core/diff.d.ts +++ /dev/null @@ -1,29 +0,0 @@ -/** - * The flat diff: tokenize both documents, find the operations between them, render them. Runs - * with whatever atomic tags are active; the public entry point sets them first. - */ -import { Operation } from "./operations"; -import { ContentDiff } from "./rendering"; -import { Token } from "./tokens"; -/** - * Diffs two fragments of HTML with the atomic tags that are currently active. Used by the - * public diff function after resolving the atomicTags parameter, by the recursive inner diff, - * which sets the atomic tags itself, and by the table pass to diff cell content. - * @param before The HTML content before the changes. - * @param after The HTML content after the changes. - * @param className (Optional) The class attribute to include in and tags. - * @param dataPrefix (Optional) The data prefix to use for data attributes. - * @returns The combined HTML content with differences wrapped in and tags. - */ -export declare const diffCore: ContentDiff; -/** - * Renders a list of operations into HTML content, diffing the content of opted-in atomic - * elements with the flat diff. - * @param beforeTokens The before list of tokens. - * @param afterTokens The after list of tokens. - * @param operations The operations to render. - * @param dataPrefix (Optional) The prefix to use in data attributes. - * @param className (Optional) The class name to include in the wrapper tag. - * @returns The rendering of the list of operations. - */ -export declare function renderOperations(beforeTokens: Token[], afterTokens: Token[], operations: Operation[], dataPrefix?: string | null, className?: string | null): string; diff --git a/js/core/diff.js b/js/core/diff.js deleted file mode 100644 index b67d2a9..0000000 --- a/js/core/diff.js +++ /dev/null @@ -1,43 +0,0 @@ -"use strict"; -Object.defineProperty(exports, "__esModule", { value: true }); -exports.renderOperations = exports.diffCore = void 0; -/** - * The flat diff: tokenize both documents, find the operations between them, render them. Runs - * with whatever atomic tags are active; the public entry point sets them first. - */ -const operations_1 = require("./operations"); -const rendering_1 = require("./rendering"); -const tokens_1 = require("./tokens"); -/** - * Diffs two fragments of HTML with the atomic tags that are currently active. Used by the - * public diff function after resolving the atomicTags parameter, by the recursive inner diff, - * which sets the atomic tags itself, and by the table pass to diff cell content. - * @param before The HTML content before the changes. - * @param after The HTML content after the changes. - * @param className (Optional) The class attribute to include in and tags. - * @param dataPrefix (Optional) The data prefix to use for data attributes. - * @returns The combined HTML content with differences wrapped in and tags. - */ -const diffCore = (before, after, className, dataPrefix) => { - if (before === after) - return before; - const beforeTokens = (0, tokens_1.htmlToTokens)(before); - const afterTokens = (0, tokens_1.htmlToTokens)(after); - const ops = (0, operations_1.calculateOperations)(beforeTokens, afterTokens); - return (0, rendering_1.renderOperations)(beforeTokens, afterTokens, ops, exports.diffCore, dataPrefix, className); -}; -exports.diffCore = diffCore; -/** - * Renders a list of operations into HTML content, diffing the content of opted-in atomic - * elements with the flat diff. - * @param beforeTokens The before list of tokens. - * @param afterTokens The after list of tokens. - * @param operations The operations to render. - * @param dataPrefix (Optional) The prefix to use in data attributes. - * @param className (Optional) The class name to include in the wrapper tag. - * @returns The rendering of the list of operations. - */ -function renderOperations(beforeTokens, afterTokens, operations, dataPrefix, className) { - return (0, rendering_1.renderOperations)(beforeTokens, afterTokens, operations, exports.diffCore, dataPrefix, className); -} -exports.renderOperations = renderOperations; diff --git a/js/core/matching.d.ts b/js/core/matching.d.ts deleted file mode 100644 index 76a4dae..0000000 --- a/js/core/matching.d.ts +++ /dev/null @@ -1,71 +0,0 @@ -/** - * Matching: finds the blocks of consecutive tokens that appear in both the before and the - * after token lists. The longest block is found first, then the blocks before and after it, - * recursively, until no match is left. - */ -import { Token } from "./tokens"; -/** The part of both documents a match is searched in. */ -export interface Segment { - beforeTokens: Token[]; - afterTokens: Token[]; - beforeMap: TokenMap; - afterMap: TokenMap; - beforeIndex: number; - afterIndex: number; -} -/** Token key to the indices of the tokens with that key. */ -export declare type TokenMap = Record; -/** - * A Match stores the information of a matching block. A matching block is a list of - * consecutive tokens that appear in both the before and after lists of tokens. - */ -export declare class Match { - segment: Segment; - length: number; - startInBefore: number; - startInAfter: number; - endInBefore: number; - endInAfter: number; - segmentStartInBefore: number; - segmentStartInAfter: number; - segmentEndInBefore: number; - segmentEndInAfter: number; - /** - * @param startInBefore The index of the first token in the list of before tokens. - * @param startInAfter The index of the first token in the list of after tokens. - * @param length The number of consecutive matching tokens in this block. - * @param segment The segment where the match was found. - */ - constructor(startInBefore: number, startInAfter: number, length: number, segment: Segment); -} -/** - * Creates a map from token key to an array of indices of locations of the matching token in - * the list of all tokens. - * @param tokens The list of tokens to be mapped. - * @returns A mapping that can be used to search for tokens. - */ -export declare function createMap(tokens: Token[]): TokenMap; -/** - * Finds and returns the best match between the before and after arrays contained in the - * segment provided. - * @param segment The segment in which to look for a match. - * @returns The best match, or null when the segment has none. - */ -export declare function findBestMatch(segment: Segment): Match | null; -/** - * Creates segment objects from the original document that can be used to restrict the area - * that findBestMatch and its helper functions search to increase performance. - * @param beforeTokens Tokens from the before document. - * @param afterTokens Tokens from the after document. - * @param beforeIndex The index within the before document where this segment begins. - * @param afterIndex The index within the after document where this segment begins. - * @returns The segment object. - */ -export declare function createSegment(beforeTokens: Token[], afterTokens: Token[], beforeIndex: number, afterIndex: number): Segment; -/** - * Finds all the matching blocks within the given segment in the before and after lists of - * tokens. - * @param segment The segment that should be searched for matching blocks. - * @returns The list of matching blocks in this range. - */ -export declare function findMatchingBlocks(segment: Segment): Match[]; diff --git a/js/core/matching.js b/js/core/matching.js deleted file mode 100644 index a2795b1..0000000 --- a/js/core/matching.js +++ /dev/null @@ -1,299 +0,0 @@ -"use strict"; -Object.defineProperty(exports, "__esModule", { value: true }); -exports.findMatchingBlocks = exports.createSegment = exports.findBestMatch = exports.createMap = exports.Match = void 0; -/** - * A Match stores the information of a matching block. A matching block is a list of - * consecutive tokens that appear in both the before and after lists of tokens. - */ -class Match { - /** - * @param startInBefore The index of the first token in the list of before tokens. - * @param startInAfter The index of the first token in the list of after tokens. - * @param length The number of consecutive matching tokens in this block. - * @param segment The segment where the match was found. - */ - constructor(startInBefore, startInAfter, length, segment) { - this.segment = segment; - this.length = length; - this.startInBefore = startInBefore + segment.beforeIndex; - this.startInAfter = startInAfter + segment.afterIndex; - this.endInBefore = this.startInBefore + this.length - 1; - this.endInAfter = this.startInAfter + this.length - 1; - this.segmentStartInBefore = startInBefore; - this.segmentStartInAfter = startInAfter; - this.segmentEndInBefore = this.segmentStartInBefore + this.length - 1; - this.segmentEndInAfter = this.segmentStartInAfter + this.length - 1; - } -} -exports.Match = Match; -/** - * Creates a map from token key to an array of indices of locations of the matching token in - * the list of all tokens. - * @param tokens The list of tokens to be mapped. - * @returns A mapping that can be used to search for tokens. - */ -function createMap(tokens) { - return tokens.reduce((map, token, index) => { - if (map[token.key]) { - map[token.key].push(index); - } - else { - map[token.key] = [index]; - } - return map; - }, Object.create(null)); -} -exports.createMap = createMap; -/** - * Compares two match objects to determine if the second match object comes before or after the - * first match object. - * @param m1 The first match object to compare. - * @param m2 The second match object to compare. - * @returns -1 if m2 should come before m1, 1 if m1 should come before m2, 0 if the two - * matches criss-cross each other. - */ -function compareMatches(m1, m2) { - if (m2.endInBefore < m1.startInBefore && m2.endInAfter < m1.startInAfter) { - return -1; - } - if (m2.startInBefore > m1.endInBefore && m2.startInAfter > m1.endInAfter) { - return 1; - } - return 0; -} -/** A binary search tree that keeps match objects in the proper order as they're found. */ -class MatchBinarySearchTree { - constructor() { - this.root = null; - } - /** - * Adds a match to the binary search tree. A match overlapping an existing node is dropped. - * @param value The match to add. - */ - add(value) { - const node = { value, left: null, right: null }; - let current = this.root; - if (!current) { - this.root = node; - return; - } - for (;;) { - // Determine if the match value should go to the left or right of the current node. - const position = compareMatches(current.value, value); - if (position === -1) { - if (current.left) { - current = current.left; - } - else { - current.left = node; - break; - } - } - else if (position === 1) { - if (current.right) { - current = current.right; - } - else { - current.right = node; - break; - } - } - else { - // If 0 was returned from compareMatches, that means the node cannot - // be inserted because it overlaps an existing node. - break; - } - } - } - /** - * Converts the binary search tree into an array using an in-order traversal. - * @returns The matches in the binary search tree, in order. - */ - toArray() { - function inOrder(node, nodes) { - if (node) { - inOrder(node.left, nodes); - nodes.push(node.value); - inOrder(node.right, nodes); - } - return nodes; - } - return inOrder(this.root, []); - } -} -/** - * Finds and returns the best match between the before and after arrays contained in the - * segment provided. - * @param segment The segment in which to look for a match. - * @returns The best match, or null when the segment has none. - */ -function findBestMatch(segment) { - const beforeTokens = segment.beforeTokens; - const afterMap = segment.afterMap; - let lastSpace = null; - let bestMatch = null; - // Iterate through the entirety of the beforeTokens to find the best match. - for (let beforeIndex = 0; beforeIndex < beforeTokens.length; beforeIndex++) { - let lookBehind = false; - // If the current best match is longer than the remaining tokens, we can bail because we - // won't find a better match. - const remainingTokens = beforeTokens.length - beforeIndex; - if (bestMatch && remainingTokens < bestMatch.length) { - break; - } - // If the current token is whitespace, make a note of it and move on. Trying to start a - // set of matches with whitespace is not efficient because it's too prevelant in most - // documents. Instead, if the next token yields a match, we'll see if the whitespace can - // be included in that match. - const beforeToken = beforeTokens[beforeIndex]; - if (beforeToken.key === " ") { - lastSpace = beforeIndex; - continue; - } - // Check to see if we just skipped a space, if so, we'll ask getFullMatch to look behind - // by one token to see if it can include the whitespace. - if (lastSpace === beforeIndex - 1) { - lookBehind = true; - } - // If the current token is not found in the afterTokens, it won't match and we can move on. - const afterTokenLocations = afterMap[beforeToken.key]; - if (!afterTokenLocations) { - continue; - } - // For each instance of the current token in afterTokens, let's see how big of a match - // we can build. - for (const afterIndex of afterTokenLocations) { - // getFullMatch will see how far the current token match will go in both - // beforeTokens and afterTokens. - const bestMatchLength = bestMatch ? bestMatch.length : 0; - const match = getFullMatch(segment, beforeIndex, afterIndex, bestMatchLength, lookBehind); - // If we got a new best match, we'll save it aside. - if (match && match.length > bestMatchLength) { - bestMatch = match; - } - } - } - return bestMatch; -} -exports.findBestMatch = findBestMatch; -/** - * Takes the start of a match, and expands it in the beforeTokens and afterTokens of the - * current segment as far as it can go. - * @param segment The segment object to search within when expanding the match. - * @param beforeStart The offset within beforeTokens to start looking. - * @param afterStart The offset within afterTokens to start looking. - * @param minLength The minimum length match that must be found. - * @param lookBehind If true, attempt to match a whitespace token just before the - * beforeStart and afterStart tokens. - * @returns The full match, or undefined when no match of the minimum length starts here. - */ -function getFullMatch(segment, beforeStart, afterStart, minLength, lookBehind) { - const beforeTokens = segment.beforeTokens; - const afterTokens = segment.afterTokens; - // If we already have a match that goes to the end of the document, no need to keep looking. - const minBeforeIndex = beforeStart + minLength; - const minAfterIndex = afterStart + minLength; - if (minBeforeIndex >= beforeTokens.length || minAfterIndex >= afterTokens.length) { - return undefined; - } - // If a minLength was provided, we can do a quick check to see if the tokens after that - // length match. If not, we won't be beating the previous best match, and we can bail out - // early. - if (minLength) { - const nextBeforeWord = beforeTokens[minBeforeIndex].key; - const nextAfterWord = afterTokens[minAfterIndex].key; - if (nextBeforeWord !== nextAfterWord) { - return undefined; - } - } - // Extend the current match as far foward as it can go, without overflowing beforeTokens or - // afterTokens. - let searching = true; - let currentLength = 1; - let beforeIndex = beforeStart + currentLength; - let afterIndex = afterStart + currentLength; - while (searching && beforeIndex < beforeTokens.length && afterIndex < afterTokens.length) { - const beforeWord = beforeTokens[beforeIndex].key; - const afterWord = afterTokens[afterIndex].key; - if (beforeWord === afterWord) { - currentLength++; - beforeIndex = beforeStart + currentLength; - afterIndex = afterStart + currentLength; - } - else { - searching = false; - } - } - // If we've been asked to look behind, it's because both beforeTokens and afterTokens may - // have a whitespace token just behind the current match that was previously ignored. If so, - // we'll expand the current match to include it. - if (lookBehind && beforeStart > 0 && afterStart > 0) { - const prevBeforeKey = beforeTokens[beforeStart - 1].key; - const prevAfterKey = afterTokens[afterStart - 1].key; - if (prevBeforeKey === " " && prevAfterKey === " ") { - beforeStart--; - afterStart--; - currentLength++; - } - } - return new Match(beforeStart, afterStart, currentLength, segment); -} -/** - * Creates segment objects from the original document that can be used to restrict the area - * that findBestMatch and its helper functions search to increase performance. - * @param beforeTokens Tokens from the before document. - * @param afterTokens Tokens from the after document. - * @param beforeIndex The index within the before document where this segment begins. - * @param afterIndex The index within the after document where this segment begins. - * @returns The segment object. - */ -function createSegment(beforeTokens, afterTokens, beforeIndex, afterIndex) { - return { - beforeTokens, - afterTokens, - beforeMap: createMap(beforeTokens), - afterMap: createMap(afterTokens), - beforeIndex, - afterIndex, - }; -} -exports.createSegment = createSegment; -/** - * Finds all the matching blocks within the given segment in the before and after lists of - * tokens. - * @param segment The segment that should be searched for matching blocks. - * @returns The list of matching blocks in this range. - */ -function findMatchingBlocks(segment) { - // Create a binary search tree to hold the matches we find in order. - const matches = new MatchBinarySearchTree(); - const segments = [segment]; - // Each time the best match is found in a segment, zero, one or two new segments may be - // created from the parts of the original segment not included in the match. We will - // continue to iterate until all segments have been processed. - while (segments.length) { - const current = segments.pop(); - const match = findBestMatch(current); - if (match && match.length) { - // If there's an unmatched area at the start of the segment, create a new segment - // from that area and throw it into the segments array to get processed. - if (match.segmentStartInBefore > 0 && match.segmentStartInAfter > 0) { - const leftBeforeTokens = current.beforeTokens.slice(0, match.segmentStartInBefore); - const leftAfterTokens = current.afterTokens.slice(0, match.segmentStartInAfter); - segments.push(createSegment(leftBeforeTokens, leftAfterTokens, current.beforeIndex, current.afterIndex)); - } - // If there's an unmatched area at the end of the segment, create a new segment from - // that area and throw it into the segments array to get processed. - const rightBeforeTokens = current.beforeTokens.slice(match.segmentEndInBefore + 1); - const rightAfterTokens = current.afterTokens.slice(match.segmentEndInAfter + 1); - const rightBeforeIndex = current.beforeIndex + match.segmentEndInBefore + 1; - const rightAfterIndex = current.afterIndex + match.segmentEndInAfter + 1; - if (rightBeforeTokens.length && rightAfterTokens.length) { - segments.push(createSegment(rightBeforeTokens, rightAfterTokens, rightBeforeIndex, rightAfterIndex)); - } - matches.add(match); - } - } - return matches.toArray(); -} -exports.findMatchingBlocks = findMatchingBlocks; diff --git a/js/core/operations.d.ts b/js/core/operations.d.ts deleted file mode 100644 index 4a603e6..0000000 --- a/js/core/operations.d.ts +++ /dev/null @@ -1,23 +0,0 @@ -import { Token } from "./tokens"; -/** What happened to a range of tokens. */ -export declare type OperationAction = "equal" | "insert" | "delete" | "replace"; -/** - * A range of tokens on both sides and what happened to it. The end is null on the side an - * operation does not touch: an insert has no before range, a delete no after range. - */ -export interface Operation { - action: OperationAction; - startInBefore: number; - endInBefore: number | null; - startInAfter: number; - endInAfter: number | null; -} -/** - * Gets a list of operations required to transform the before list of tokens into the - * after list of tokens. An operation describes whether a particular list of consecutive - * tokens are equal, replaced, inserted, or deleted. - * @param beforeTokens The before list of tokens. - * @param afterTokens The after list of tokens. - * @returns The list of operations to transform the before list of tokens into the after list. - */ -export declare function calculateOperations(beforeTokens: Token[], afterTokens: Token[]): Operation[]; diff --git a/js/core/operations.js b/js/core/operations.js deleted file mode 100644 index 46f1d84..0000000 --- a/js/core/operations.js +++ /dev/null @@ -1,88 +0,0 @@ -"use strict"; -Object.defineProperty(exports, "__esModule", { value: true }); -exports.calculateOperations = void 0; -/** - * Operations: turns the matching blocks into the list of equal, insert, delete and replace - * steps that transform the before tokens into the after tokens. - */ -const matching_1 = require("./matching"); -/** - * Gets a list of operations required to transform the before list of tokens into the - * after list of tokens. An operation describes whether a particular list of consecutive - * tokens are equal, replaced, inserted, or deleted. - * @param beforeTokens The before list of tokens. - * @param afterTokens The after list of tokens. - * @returns The list of operations to transform the before list of tokens into the after list. - */ -function calculateOperations(beforeTokens, afterTokens) { - if (!beforeTokens) - throw new Error("Missing beforeTokens"); - if (!afterTokens) - throw new Error("Missing afterTokens"); - let positionInBefore = 0; - let positionInAfter = 0; - const operations = []; - const segment = (0, matching_1.createSegment)(beforeTokens, afterTokens, 0, 0); - const matches = (0, matching_1.findMatchingBlocks)(segment); - matches.push(new matching_1.Match(beforeTokens.length, afterTokens.length, 0, segment)); - for (let index = 0; index < matches.length; index++) { - const match = matches[index]; - let actionUpToMatchPositions = "none"; - if (positionInBefore === match.startInBefore) { - if (positionInAfter !== match.startInAfter) { - actionUpToMatchPositions = "insert"; - } - } - else { - actionUpToMatchPositions = "delete"; - if (positionInAfter !== match.startInAfter) { - actionUpToMatchPositions = "replace"; - } - } - if (actionUpToMatchPositions !== "none") { - operations.push({ - action: actionUpToMatchPositions, - startInBefore: positionInBefore, - endInBefore: actionUpToMatchPositions !== "insert" ? match.startInBefore - 1 : null, - startInAfter: positionInAfter, - endInAfter: actionUpToMatchPositions !== "delete" ? match.startInAfter - 1 : null, - }); - } - if (match.length !== 0) { - operations.push({ - action: "equal", - startInBefore: match.startInBefore, - endInBefore: match.endInBefore, - startInAfter: match.startInAfter, - endInAfter: match.endInAfter, - }); - } - positionInBefore = match.endInBefore + 1; - positionInAfter = match.endInAfter + 1; - } - const postProcessed = []; - let lastOp = { action: "none" }; - function isSingleWhitespace(op) { - if (op.action !== "equal") { - return false; - } - if (op.endInBefore - op.startInBefore !== 0) { - return false; - } - return /^\s$/.test(String(beforeTokens.slice(op.startInBefore, op.endInBefore + 1))); - } - for (let i = 0; i < operations.length; i++) { - const op = operations[i]; - if (lastOp.action === "replace" && - (isSingleWhitespace(op) || op.action === "replace")) { - lastOp.endInBefore = op.endInBefore; - lastOp.endInAfter = op.endInAfter; - } - else { - postProcessed.push(op); - lastOp = op; - } - } - return postProcessed; -} -exports.calculateOperations = calculateOperations; diff --git a/js/core/rendering.d.ts b/js/core/rendering.d.ts deleted file mode 100644 index 994641c..0000000 --- a/js/core/rendering.d.ts +++ /dev/null @@ -1,112 +0,0 @@ -import { Operation } from "./operations"; -import { Token } from "./tokens"; -/** - * Diffs two fragments of HTML with the active atomic tags. The renderer gets it injected to - * diff the content of opted-in elements recursively; see the diff module. - */ -export declare type ContentDiff = (before: string, after: string, className?: string | null, dataPrefix?: string | null) => string; -/** A run of tokens that are all wrappable or all not. */ -interface TokenSegment { - isWrappable: boolean; - tokens: string[]; -} -/** - * A TokenWrapper provides a utility for grouping segments of tokens based on whether they're - * wrappable or not. A tag is considered wrappable if it is closed within the given set of - * tokens. For example, given the following tokens: - * - * ['
    ', 'this', ' ', 'is', ' ', 'a', ' ', '', 'test', '', '!'] - * - * The first '' is not considered wrappable since the tag is not fully contained within the - * array of tokens. The '', 'test', and '' would be a part of the same wrappable segment - * since the entire bold tag is within the set of tokens. - */ -export declare class TokenWrapper { - private readonly tokens; - private readonly notes; - /** - * @param tokens The tokens to group. - */ - constructor(tokens: string[]); - /** - * Wraps the contained tokens in tags based on output given by a map function. Each segment - * of tokens will be visited. A segment is a continuous run of either all wrappable tokens or - * unwrappable tokens. The given map function will be called with each segment of tokens and - * the resulting strings will be combined to form the wrapped HTML. - * @param mapFn Called with each segment; the result should be a string. - * @param tagFn Called with the opening tag of every tag inserted whole, to mark it. - * @returns The wrapped HTML. - */ - combine(mapFn: (segment: TokenSegment) => string, tagFn: (openingTag: string) => string): string; -} -/** - * Wraps and concatenates a list of tokens with a tag. Does not wrap tag tokens, unless they - * are wrappable (i.e. void and atomic tags). - * @param tag The tag name of the wrapper tags. - * @param content The list of tokens to wrap. - * @param opIndex The index of the operation, written to the data attribute. - * @param dataPrefix (Optional) The prefix to use in data attributes. - * @param className (Optional) The class name to include in the wrapper tag. - * @returns The wrapped HTML. - */ -export declare function wrap(tag: string, content: string[], opIndex: number, dataPrefix?: string | null, className?: string | null): string; -/** - * Checks whether a token is an atomic tag that opted into the recursive inner diff via the - * data-htmldiff-inner-diff attribute. A bare attribute or any value other than "false" counts - * as opted in. Opted-in elements nested inside other opted-in elements are diffed recursively - * as well, up to the depth cap; beyond it, opted-in tokens are rendered verbatim like any - * other atomic token. - * @param tokenString The token string to check. - * @returns True if the token should get a recursive inner diff. - */ -export declare function isInnerDiffToken(tokenString: string): boolean; -/** - * Finds the index of the '>' that ends the opening tag at the start of the given token - * string, skipping any '>' inside quoted attribute values (e.g. title="a > b"). - * @param tokenString The token string starting with an opening tag. - * @returns The index of the closing '>' of the opening tag, or -1 if there is none (e.g. an - * unterminated tag or an unbalanced attribute quote). - */ -export declare function findOpeningTagEnd(tokenString: string): number; -/** An atomic token split into its tags and content. */ -export interface SplitToken { - openingTag: string; - innerHtml: string; - closingTag: string; -} -/** - * Splits an atomic token string into its opening tag, inner HTML and closing tag. A token - * consisting of a single tag (a void or self-closing element) has an empty inner HTML and no - * closing tag. - * @param tokenString The atomic token string, e.g. '
    content
    '. - * @returns The parts, or null if the token cannot be split (e.g. an unterminated tag). - */ -export declare function splitAtomicTokenString(tokenString: string): SplitToken | null; -/** - * Renders the recursive inner diff of two matched atomic tokens with equal keys but different - * content. The after version's opening and closing tags are emitted with the diff of the two - * inner HTML fragments in between. Inside the recursion the default atomic tags without 'a' - * are used, so link text is diffed word by word and href-only changes do not produce any - * markup. The after version's data-htmldiff-inner-diff-atomic-tags attribute overrides that - * list. Nested opted-in elements are diffed recursively as well, up to a hardcoded depth cap. - * @param beforeString The before version of the atomic token. - * @param afterString The after version of the atomic token. - * @param diffContent Diffs the two inner HTML fragments. - * @param dataPrefix (Optional) The prefix to use in data attributes. - * @param className (Optional) The class name to include in the wrapper tag. - * @returns The rendered element with inner differences wrapped in ins/del tags. - */ -export declare function renderInnerDiff(beforeString: string, afterString: string, diffContent: ContentDiff, dataPrefix?: string | null, className?: string | null): string; -/** - * Renders a list of operations into HTML content. The result is the combined version of the - * before and after tokens with the differences wrapped in tags. - * @param beforeTokens The before list of tokens. - * @param afterTokens The after list of tokens. - * @param operations The list of operations to transform the before tokens into the after tokens. - * @param diffContent Diffs the content of opted-in atomic elements recursively. - * @param dataPrefix (Optional) The prefix to use in data attributes. - * @param className (Optional) The class name to include in the wrapper tag. - * @returns The rendering of the list of operations. - */ -export declare function renderOperations(beforeTokens: Token[], afterTokens: Token[], operations: Operation[], diffContent: ContentDiff, dataPrefix?: string | null, className?: string | null): string; -export {}; diff --git a/js/core/rendering.js b/js/core/rendering.js deleted file mode 100644 index 8d6365b..0000000 --- a/js/core/rendering.js +++ /dev/null @@ -1,315 +0,0 @@ -"use strict"; -Object.defineProperty(exports, "__esModule", { value: true }); -exports.renderOperations = exports.renderInnerDiff = exports.splitAtomicTokenString = exports.findOpeningTagEnd = exports.isInnerDiffToken = exports.wrap = exports.TokenWrapper = void 0; -/** - * Rendering: writes the operations back out as HTML, wrapping inserted and deleted tokens in - * and tags and diffing the content of opted-in atomic elements recursively. - */ -const atomicTags_1 = require("./atomicTags"); -const tokens_1 = require("./tokens"); -// The number of currently active recursive inner diffs. Recursion is governed per element -// (each nesting level requires its own data-htmldiff-inner-diff attribute), so the depth is -// naturally bounded by the nesting of opted-in elements; the cap is only a backstop against -// pathologically deep documents. -let innerDiffDepth = 0; -const maxInnerDiffDepth = 10; -/** - * A TokenWrapper provides a utility for grouping segments of tokens based on whether they're - * wrappable or not. A tag is considered wrappable if it is closed within the given set of - * tokens. For example, given the following tokens: - * - * ['
    ', 'this', ' ', 'is', ' ', 'a', ' ', '', 'test', '', '!'] - * - * The first '' is not considered wrappable since the tag is not fully contained within the - * array of tokens. The '', 'test', and '' would be a part of the same wrappable segment - * since the entire bold tag is within the set of tokens. - */ -class TokenWrapper { - /** - * @param tokens The tokens to group. - */ - constructor(tokens) { - this.tokens = tokens; - this.notes = tokens.reduce((data, token, index) => { - data.notes.push({ - isWrappable: (0, tokens_1.isWrappable)(token), - insertedTag: false, - }); - const tag = !(0, tokens_1.isVoidTag)(token) && (0, tokens_1.isTag)(token); - const lastEntry = data.tagStack[data.tagStack.length - 1]; - if (tag) { - if (lastEntry && "/" + lastEntry.tag === tag) { - data.notes[lastEntry.position].insertedTag = true; - data.tagStack.pop(); - } - else { - data.tagStack.push({ tag, position: index }); - } - } - return data; - }, { notes: [], tagStack: [] }).notes; - } - /** - * Wraps the contained tokens in tags based on output given by a map function. Each segment - * of tokens will be visited. A segment is a continuous run of either all wrappable tokens or - * unwrappable tokens. The given map function will be called with each segment of tokens and - * the resulting strings will be combined to form the wrapped HTML. - * @param mapFn Called with each segment; the result should be a string. - * @param tagFn Called with the opening tag of every tag inserted whole, to mark it. - * @returns The wrapped HTML. - */ - combine(mapFn, tagFn) { - const notes = this.notes; - const tokens = this.tokens.slice(); - const segments = tokens.reduce((data, token, index) => { - if (notes[index].insertedTag) { - tokens[index] = tagFn(tokens[index]); - } - if (data.status === null) { - data.status = notes[index].isWrappable; - } - const status = notes[index].isWrappable; - // Handling atomic tags wrapping independently - // each atomic tag is wrapped with their own ins/del tags - const isAtomic = !!(0, atomicTags_1.isStartOfAtomicTag)(token); - if (status !== data.status || (isAtomic && index > data.lastIndex) || data.lastWasAtomic) { - data.list.push({ - isWrappable: data.status, - tokens: tokens.slice(data.lastIndex, index), - }); - data.lastIndex = index; - data.status = status; - } - // tracking if the last token was an atomic tag - // if so then we break the segment and wrap them - data.lastWasAtomic = isAtomic; - if (index === tokens.length - 1) { - data.list.push({ - isWrappable: data.status, - tokens: tokens.slice(data.lastIndex, index + 1), - }); - } - return data; - }, { list: [], status: null, lastIndex: 0, lastWasAtomic: false }).list; - return segments.map(mapFn).join(""); - } -} -exports.TokenWrapper = TokenWrapper; -/** - * Wraps and concatenates a list of tokens with a tag. Does not wrap tag tokens, unless they - * are wrappable (i.e. void and atomic tags). - * @param tag The tag name of the wrapper tags. - * @param content The list of tokens to wrap. - * @param opIndex The index of the operation, written to the data attribute. - * @param dataPrefix (Optional) The prefix to use in data attributes. - * @param className (Optional) The class name to include in the wrapper tag. - * @returns The wrapped HTML. - */ -function wrap(tag, content, opIndex, dataPrefix, className) { - const wrapper = new TokenWrapper(content); - const prefix = dataPrefix ? dataPrefix + "-" : ""; - let attrs = ` data-${prefix}operation-index="${opIndex}"`; - if (className) { - attrs += ' class="' + className + '"'; - } - return wrapper.combine((segment) => { - if (segment.isWrappable) { - const val = segment.tokens.join(""); - if (val.trim()) { - return "<" + tag + attrs + ">" + val + ""; - } - } - else { - return segment.tokens.join(""); - } - return ""; - }, (openingTag) => { - let dataAttrs = ' data-diff-node="' + tag + '"'; - dataAttrs += ` data-${prefix}operation-index="${opIndex}"`; - return openingTag.replace(/>\s*$/, dataAttrs + "$&"); - }); -} -exports.wrap = wrap; -/** - * Checks whether a token is an atomic tag that opted into the recursive inner diff via the - * data-htmldiff-inner-diff attribute. A bare attribute or any value other than "false" counts - * as opted in. Opted-in elements nested inside other opted-in elements are diffed recursively - * as well, up to the depth cap; beyond it, opted-in tokens are rendered verbatim like any - * other atomic token. - * @param tokenString The token string to check. - * @returns True if the token should get a recursive inner diff. - */ -function isInnerDiffToken(tokenString) { - if (innerDiffDepth >= maxInnerDiffDepth || !(0, atomicTags_1.isStartOfAtomicTag)(tokenString)) { - return false; - } - const attr = atomicTags_1.dataHtmlDiffInnerDiffRegExp.exec(tokenString); - return !!attr && attr[1] !== "false"; -} -exports.isInnerDiffToken = isInnerDiffToken; -/** - * Finds the index of the '>' that ends the opening tag at the start of the given token - * string, skipping any '>' inside quoted attribute values (e.g. title="a > b"). - * @param tokenString The token string starting with an opening tag. - * @returns The index of the closing '>' of the opening tag, or -1 if there is none (e.g. an - * unterminated tag or an unbalanced attribute quote). - */ -function findOpeningTagEnd(tokenString) { - let quote = null; - for (let i = 0; i < tokenString.length; i++) { - const char = tokenString[i]; - // quote is closed - if (char === quote) { - quote = null; - continue; - } - // inside quote - if (quote) { - continue; - } - // quote start - if (char === '"' || char === "'") { - quote = char; - continue; - } - // not inside quote, check for tag end - if (char === ">") { - return i; - } - } - return -1; -} -exports.findOpeningTagEnd = findOpeningTagEnd; -/** - * Splits an atomic token string into its opening tag, inner HTML and closing tag. A token - * consisting of a single tag (a void or self-closing element) has an empty inner HTML and no - * closing tag. - * @param tokenString The atomic token string, e.g. '
    content
    '. - * @returns The parts, or null if the token cannot be split (e.g. an unterminated tag). - */ -function splitAtomicTokenString(tokenString) { - const openingTagEnd = findOpeningTagEnd(tokenString); - if (openingTagEnd === -1) { - return null; - } - if (openingTagEnd === tokenString.length - 1) { - // The token is a single tag (void or self-closing): the element has no content. - return { - openingTag: tokenString, - innerHtml: "", - closingTag: "", - }; - } - const closingTagStart = tokenString.lastIndexOf("<"); - if (closingTagStart <= openingTagEnd || tokenString[closingTagStart + 1] !== "/") { - return null; - } - return { - openingTag: tokenString.slice(0, openingTagEnd + 1), - innerHtml: tokenString.slice(openingTagEnd + 1, closingTagStart), - closingTag: tokenString.slice(closingTagStart), - }; -} -exports.splitAtomicTokenString = splitAtomicTokenString; -/** - * Renders the recursive inner diff of two matched atomic tokens with equal keys but different - * content. The after version's opening and closing tags are emitted with the diff of the two - * inner HTML fragments in between. Inside the recursion the default atomic tags without 'a' - * are used, so link text is diffed word by word and href-only changes do not produce any - * markup. The after version's data-htmldiff-inner-diff-atomic-tags attribute overrides that - * list. Nested opted-in elements are diffed recursively as well, up to a hardcoded depth cap. - * @param beforeString The before version of the atomic token. - * @param afterString The after version of the atomic token. - * @param diffContent Diffs the two inner HTML fragments. - * @param dataPrefix (Optional) The prefix to use in data attributes. - * @param className (Optional) The class name to include in the wrapper tag. - * @returns The rendered element with inner differences wrapped in ins/del tags. - */ -function renderInnerDiff(beforeString, afterString, diffContent, dataPrefix, className) { - const before = splitAtomicTokenString(beforeString); - const after = splitAtomicTokenString(afterString); - if (!before || !after) { - return afterString; - } - const atomicTagsOverride = atomicTags_1.dataHtmlDiffInnerDiffAtomicTagsRegExp.exec(afterString); - const outerAtomicTagsRegExp = (0, atomicTags_1.getAtomicTagsRegExp)(); - innerDiffDepth++; - let innerDiff; - // the depth and the atomic tags must be restored in case of an error, hence try/finally - try { - (0, atomicTags_1.setAtomicTagsRegExp)(atomicTags_1.defaultInnerDiffAtomicTagsRegExp); - if (atomicTagsOverride) { - (0, atomicTags_1.setAtomicTagsRegExp)(atomicTagsOverride[1] ? (0, atomicTags_1.buildAtomicTagsRegExp)(atomicTagsOverride[1]) : atomicTags_1.noAtomicTagsRegExp); - } - innerDiff = diffContent(before.innerHtml, after.innerHtml, className, dataPrefix); - } - finally { - innerDiffDepth--; - (0, atomicTags_1.setAtomicTagsRegExp)(outerAtomicTagsRegExp); - } - return after.openingTag + innerDiff + after.closingTag; -} -exports.renderInnerDiff = renderInnerDiff; -function renderEqual(op, beforeTokens, afterTokens, _opIndex, diffContent, dataPrefix, className) { - // Tokens in an equal operation pair up one to one between before and after. Equal keys do - // not guarantee equal strings (e.g. atomic tokens matched by data-htmldiff-id): elements - // that opted in via data-htmldiff-inner-diff get a recursive diff of their content, - // everything else renders the after version. - let result = ""; - for (let i = 0; op.startInAfter + i <= op.endInAfter; i++) { - const afterToken = afterTokens[op.startInAfter + i]; - const beforeToken = beforeTokens[op.startInBefore + i]; - if (beforeToken && beforeToken.string !== afterToken.string && isInnerDiffToken(afterToken.string)) { - result += renderInnerDiff(beforeToken.string, afterToken.string, diffContent, dataPrefix, className); - } - else { - result += afterToken.string; - } - } - return result; -} -function renderInsert(op, _beforeTokens, afterTokens, opIndex, _diffContent, dataPrefix, className) { - const tokens = afterTokens.slice(op.startInAfter, op.endInAfter + 1); - const val = tokens.map((token) => token.string); - const res = wrap("ins", val, opIndex, dataPrefix, className); - // handling inserted tags, see https://matrixreq.atlassian.net/browse/MATRIX-7876 - if (/^<[^./]+?>$/.exec(res)) { - return `${res.slice(0, res.length - 1)} data-inserted="true">`; - } - return res; -} -function renderDelete(op, beforeTokens, _afterTokens, opIndex, _diffContent, dataPrefix, className) { - const tokens = beforeTokens.slice(op.startInBefore, op.endInBefore + 1); - const val = tokens.map((token) => token.string); - const res = wrap("del", val, opIndex, dataPrefix, className); - // handling cases like deleted

    , see https://matrixreq.atlassian.net/browse/MATRIX-7688 - if (/^<\/.+?><.+?>$/.exec(res) && !res.includes("del")) { - return `${val.slice(1, val.length - 1).join("")}`; - } - return res; -} -function renderReplace(op, beforeTokens, afterTokens, opIndex, diffContent, dataPrefix, className) { - return (renderDelete(op, beforeTokens, afterTokens, opIndex, diffContent, dataPrefix, className) + - renderInsert(op, beforeTokens, afterTokens, opIndex, diffContent, dataPrefix, className)); -} -const OPS = { - equal: renderEqual, - insert: renderInsert, - delete: renderDelete, - replace: renderReplace, -}; -/** - * Renders a list of operations into HTML content. The result is the combined version of the - * before and after tokens with the differences wrapped in tags. - * @param beforeTokens The before list of tokens. - * @param afterTokens The after list of tokens. - * @param operations The list of operations to transform the before tokens into the after tokens. - * @param diffContent Diffs the content of opted-in atomic elements recursively. - * @param dataPrefix (Optional) The prefix to use in data attributes. - * @param className (Optional) The class name to include in the wrapper tag. - * @returns The rendering of the list of operations. - */ -function renderOperations(beforeTokens, afterTokens, operations, diffContent, dataPrefix, className) { - return operations.reduce((rendering, op, index) => rendering + OPS[op.action](op, beforeTokens, afterTokens, index, diffContent, dataPrefix, className), ""); -} -exports.renderOperations = renderOperations; diff --git a/js/core/tokens.d.ts b/js/core/tokens.d.ts deleted file mode 100644 index 5b783bc..0000000 --- a/js/core/tokens.d.ts +++ /dev/null @@ -1,54 +0,0 @@ -/** A token holds the text to render and the key it is compared by. */ -export interface Token { - string: string; - key: string; -} -/** - * Determines if the given token is a tag. - * @param token The token in question. - * @returns False if the token is not a tag, or the tag name otherwise. - */ -export declare function isTag(token: string): string | false; -/** - * Checks if a tag is a void tag, written with the XML style '/>'. - * @param token The token to check. - * @returns True if the token is a void tag, false otherwise. - */ -export declare function isVoidTag(token: string): boolean; -/** - * Checks if a tag name is an HTML void element. Void elements cannot have content and can - * skip a closing tag, so an atomic element with a void tag name ends with its opening tag, - * with or without the XML style '/>'. - * @param tag The tag name to check. - * @returns True if the tag name is a void element. - */ -export declare function isVoidTagName(tag: string): boolean; -/** - * Checks if a token can be wrapped inside a tag: text, images, void tags and atomic tags - * can, other tags cannot. - * @param token The token to check. - * @returns True if the token can be wrapped inside a tag, false otherwise. - */ -export declare function isWrappable(token: string): boolean; -/** - * Creates a token that holds a string and key representation. The key is used for diffing - * comparisons and the string is used to recompose the document after the diff is complete. - * @param currentWord The section of the document to create a token for. - * @returns A token object with a string and key property. - */ -export declare function createToken(currentWord: string): Token; -/** - * Creates a key that should be used to match tokens. This is useful, for example, if we want - * to consider two open tag tokens as equal, even if they don't have the same attributes. We - * use a key instead of overwriting the token because we may want to render the original - * string without losing the attributes. - * @param token The token to create the key for. - * @returns The identifying key that should be used to match before and after tokens. - */ -export declare function getKeyForToken(token: string): string; -/** - * Tokenizes a string of HTML. - * @param html The string to tokenize. - * @returns The list of tokens. - */ -export declare function htmlToTokens(html: string): Token[]; diff --git a/js/core/tokens.js b/js/core/tokens.js deleted file mode 100644 index c48e06d..0000000 --- a/js/core/tokens.js +++ /dev/null @@ -1,363 +0,0 @@ -"use strict"; -Object.defineProperty(exports, "__esModule", { value: true }); -exports.htmlToTokens = exports.getKeyForToken = exports.createToken = exports.isWrappable = exports.isVoidTagName = exports.isVoidTag = exports.isTag = void 0; -/** - * Tokenizing: splits HTML into words, whitespace, tags and atomic elements, and gives every - * token the key it is compared by. - */ -const atomicTags_1 = require("./atomicTags"); -/** - * Determines if the given token is a tag. - * @param token The token in question. - * @returns False if the token is not a tag, or the tag name otherwise. - */ -function isTag(token) { - const match = token.match(/^\s*<([^!>][^>]*)>\s*$/); - return !!match && match[1].trim().split(" ")[0]; -} -exports.isTag = isTag; -/** - * Checks if a tag is a void tag, written with the XML style '/>'. - * @param token The token to check. - * @returns True if the token is a void tag, false otherwise. - */ -function isVoidTag(token) { - return /^\s*<[^>]+\/>\s*$/.test(token); -} -exports.isVoidTag = isVoidTag; -/** - * Checks if a tag name is an HTML void element. Void elements cannot have content and can - * skip a closing tag, so an atomic element with a void tag name ends with its opening tag, - * with or without the XML style '/>'. - * @param tag The tag name to check. - * @returns True if the tag name is a void element. - */ -function isVoidTagName(tag) { - return /^(area|base|br|col|embed|hr|img|input|link|meta|param|source|track|wbr)$/.test(tag); -} -exports.isVoidTagName = isVoidTagName; -/** - * Checks if a token can be wrapped inside a tag: text, images, void tags and atomic tags - * can, other tags cannot. - * @param token The token to check. - * @returns True if the token can be wrapped inside a tag, false otherwise. - */ -function isWrappable(token) { - const isImage = /^]/.test(token); - return isImage || !isTag(token) || !!(0, atomicTags_1.isStartOfAtomicTag)(token) || isVoidTag(token); -} -exports.isWrappable = isWrappable; -/** - * Creates a token that holds a string and key representation. The key is used for diffing - * comparisons and the string is used to recompose the document after the diff is complete. - * @param currentWord The section of the document to create a token for. - * @returns A token object with a string and key property. - */ -function createToken(currentWord) { - return { - string: currentWord, - key: getKeyForToken(currentWord), - }; -} -exports.createToken = createToken; -/** - * Creates a key that should be used to match tokens. This is useful, for example, if we want - * to consider two open tag tokens as equal, even if they don't have the same attributes. We - * use a key instead of overwriting the token because we may want to render the original - * string without losing the attributes. - * @param token The token to create the key for. - * @returns The identifying key that should be used to match before and after tokens. - */ -function getKeyForToken(token) { - // If the token is an image element, grab it's src attribute to include in the key. - const img = /^$/.exec(token); - if (img) { - return ''; - } - // If the token is an a element, grab it's data attribute to include in the key. - // Only when is atomic: if it has been excluded from the atomic tags (as done in - // recursive inner diffs), the token is just the opening tag. - const a = /^'; - } - // If the token is an object element, grab it's data attribute to include in the key. - const object = /^'; - } - // If it's a video, math or svg element, the entire token should be compared except the - // data-uuid. - if (/^<(svg|math|video)[\s>]/.test(token)) { - const uuid = token.indexOf('data-uuid="'); - if (uuid !== -1) { - const start = token.slice(0, uuid); - const end = token.slice(uuid + 44); - return start + end; - } - return token; - } - // If the token is an iframe element, grab it's src attribute to include in it's key. - const iframe = /^/.exec(token); - if (iframe) { - return ''; - } - // if the token has data-htmldiff-uuid use it as a key - const uuidTag = atomicTags_1.dataHtmlDiffIdRegExp.exec(token); - if (uuidTag) { - return uuidTag[2]; - } - // If the token is any other element, just grab the tag name. - const tagName = /<([^\s>]+)[\s>]/.exec(token); - if (tagName) { - return "<" + tagName[1].toLowerCase() + ">"; - } - // Otherwise, the token is text, collapse the whitespace - // (except new lines, see https://matrixreq.atlassian.net/browse/MATRIX-7880) - // potentially, this also causing the problems with prettified HTML (with "\n" between the tags), - // so it's required to "flatten" html before passing it to the diffing function - if (token) { - return token.replace(/([^\S\r\n]+| | )/g, " "); - } - return token; -} -exports.getKeyForToken = getKeyForToken; -function isEndOfTag(char) { - return char === ">"; -} -function isStartOfTag(char) { - return char === "<"; -} -function isWhitespace(char) { - return /^\s+$/.test(char); -} -function isStartOfHtmlComment(word) { - return /^$/.test(word); -} -/** - * Inspects the last tag in the given string, its slice from the final '<'. A '>' before the - * slice's end means text follows e.g. "a > b" in a