diff --git a/.gitignore b/.gitignore
index 44b42aa..aebae61 100644
--- a/.gitignore
+++ b/.gitignore
@@ -1,2 +1,4 @@
node_modules
-htmldiff-cli.js
\ No newline at end of file
+htmldiff-cli.js
+js/
+.DS_Store
diff --git a/.mocharc.json b/.mocharc.json
new file mode 100644
index 0000000..47a9bb2
--- /dev/null
+++ b/.mocharc.json
@@ -0,0 +1,13 @@
+{
+ "require": [
+ "ts-node/register"
+ ],
+ "extension": [
+ "ts"
+ ],
+ "spec": [
+ "src/**/*.spec.ts",
+ "test/**/*.spec.ts"
+ ],
+ "ui": "bdd"
+}
diff --git a/README.md b/README.md
index ea0bc78..cb1d1da 100644
--- a/README.md
+++ b/README.md
@@ -145,6 +145,48 @@ Limitations:
- The recursion depth is capped at 10 levels as a backstop against deep
nesting. Opted-in elements beyond the cap are rendered as their after version.
+### Tables
+
+Tables are compared as structures before the flat diff runs. A table with its own
+`data-htmldiff-id` takes part only when it also carries `data-htmldiff-inner-diff`, like every
+other element with an identity: without the inner-id it is one unit, shown as it is or replaced
+whole. The same goes for a table inside such an element. The two documents' top level
+tables are paired (by their own `data-htmldiff-id` when they have one, otherwise by the values
+they hold; a table in the other's place is the same table only when the two still share half of
+what the smaller one holds, or when both are generated tables, see below), each pair is aligned
+column by column and row by row, and a merged table takes the place of the after version's
+table. A table without a partner is kept whole, so the flat diff wraps it as added or deleted:
+
+- an added or deleted row is a whole row with the class `table-row-added` or
+ `table-row-deleted`,
+- an added or deleted column marks every one of its cells (and its `
`, when the table has
+ a `
`) with `table-cell-added` or `table-cell-deleted`,
+- a kept cell holds the diff of its content, with the usual ``/`` tags,
+- a row that keeps less than half of its content, or a column no kept row agrees with, is
+ deleted and added instead of diffed; so is a moved row or column,
+- merged cells (`rowspan`, `colspan`) are kept. A row that a kept group (a merged cell and the
+ rows it spans) lost or gained goes under the group's kept rows and carries the change on its
+ cells (`table-cell-deleted` / `table-cell-added`), not on the row: the group's cell, sitting on
+ the group's first kept row, spans it like any other row of the group. A view that hides the
+ changed cells then hides nothing the span counts on, so the layout holds. A whole group that
+ one version has is added or deleted row by row.
+- in a table without such a statement, a change of its merged cells (a cell merged, split, a
+ span grown or shrunk) makes it another table: both versions are kept whole.
+
+The diff reads nothing else from the content. A producer that knows what a row is about says
+so by giving the cells an identity with `data-htmldiff-id`.
+When both versions carry such cells, those alone pair the rows: the same
+identities are the same row, whatever its other cells say, and they get cell diffs; other
+identities are another row, deleted and added whole. The identity sits on the cell, not on
+the row, because such cells often span several rows: the rows under a spanning cell never
+contain it, yet they belong to it, and an id on the cell is inherited by every row it spans,
+where an id on each `
` would have to be composed and repeated by the producer. A cell id
+also says which cells name the row and which are content.
+
+Both versions of a pair get the same `data-htmldiff-id` (`redline-table-` unless the table
+had one), a table only one version has gets one of its own, so the flat diff keeps every table
+whole and emits the merged table as it is. The styling of the classes is up to the consumer.
+
### Example
JavaScript:
@@ -198,14 +240,40 @@ description please see API documentation above.
## Development
-After cloning the repository run `npm i` or `npm install` to install the necessary
-dependencies. A run of `npm run make` creates the JavaScript output file.
-`npm run lint` checks the TypeScript sources with TSLint. `npm test` runs all the
-tests from the `test` directory. `npm run testsample` diffs the HTML sample files
-from the directory `sample` and logs the result to the console.
-
-The command line interface of htmldiff is developed in TypeScript so you have to run
-`npm run make` once to create the JavaScript output file.
+After cloning the repository run `npm install` to install the dependencies.
+
+Everything is TypeScript. The library lives in `src/` and is compiled to CommonJS in `js/`
+(`js/htmldiff.js` is the entry point, `js/htmldiff.d.ts` the typings); `js/` is what gets
+published.
+
+- `src/htmldiff.ts` is the facade: it runs the table pass, then the flat diff.
+- `src/constants.ts` names the attributes the diff reads, `data-htmldiff-id` and
+ `data-htmldiff-inner-diff`: the contract between a producer of HTML and the diff.
+- `src/core/` is the flat diff, one module per stage: `atomicTags` (which elements are one
+ token), `tokens` (tokenizing and token keys), `matching` (matching blocks), `operations`
+ (insert, delete, replace, equal), `rendering` (ins/del markup and the recursive inner
+ diff) and `diff` (the pipeline).
+- `src/tables/` is the structural table pass, one class per concern: `Cell`, `Row` and
+ `Table` are the model, `TableVersion` a table as the alignment reads it, `SequenceAligner`,
+ `ColumnAligner`, `RowAligner` and `TableAligner` decide what is the same, `MergedCells`
+ handles spans, `TableMerger` writes the merged table and `TableRedlining` is the pass
+ itself. `html.ts`, `similarity.ts` and `helpers.ts` are plain helper functions.
+
+Tests are TypeScript too, run by mocha through `ts-node`. Every module and class has a spec
+next to it (`src/**/*.spec.ts`, left out of the build); the end to end specs that go through
+`diff()` itself live in `test/`.
+
+Scripts:
+
+- `npm run build` compiles `src/` to `js/`.
+- `npm test` builds, type-checks sources and specs, then runs the specs.
+- `npm run lint` checks sources and specs with ESLint.
+- `npm run verify` does all of that in one go and compiles the CLI; `npm publish` runs it first
+ (`prepublishOnly`), so nothing unbuilt or untested can be published. `npm install` in a clone
+ builds `js/` as well (`prepare`).
+- `npm run make` builds the library and the command line interface, `htmldiff-cli.ts`.
+- `npm run testsample` diffs the HTML sample files from the directory `sample` and logs the
+ result to the console.
## Credits
diff --git a/eslintrc.json b/eslintrc.json
index 344bd3b..6f043f7 100644
--- a/eslintrc.json
+++ b/eslintrc.json
@@ -2,7 +2,7 @@
"root": true,
"parser": "@typescript-eslint/parser",
"parserOptions": {
- "project": "./app/src/tsconfig.json"
+ "project": ["./tsconfig.json", "./tsconfig.cli.json"]
},
"plugins": [
"@typescript-eslint",
@@ -43,25 +43,16 @@
"warn",
{
"require": {
- "ArrowFunctionExpression": true,
"ClassDeclaration": true,
- "ClassExpression": true,
"FunctionDeclaration": true,
- "FunctionExpression": true,
"MethodDefinition": true
},
"contexts": [
- "Property",
- "ClassProperty:not([accessibility=\"private\"])",
- "TSMethodSignature",
"TSEnumDeclaration",
"TSInterfaceDeclaration",
- "TSTypeAliasDeclaration",
- "-TSPropertySignature",
- "ExportNamedDeclaration"
+ "TSTypeAliasDeclaration"
],
- "checkGetters": true,
- "checkSetters": true,
+ "publicOnly": true,
"exemptEmptyConstructors": true
}
],
diff --git a/js/htmldiff.d.ts b/js/htmldiff.d.ts
deleted file mode 100644
index 8c01205..0000000
--- a/js/htmldiff.d.ts
+++ /dev/null
@@ -1,17 +0,0 @@
-/**
- * Compares two pieces of HTML content and returns the combined content with differences
- * wrapped in and tags.
- * @param before The HTML content before the changes.
- * @param after The HTML content after the changes.
- * @param className (Optional) The class attribute to include in `` and `` tags.
- * @param dataPrefix (Optional) The data prefix to use for data attributes. The operation index data
- * attribute will be named `data-${dataPrefix-}operation-index`.
- * @param atomicTags (Optional) Comma separated list of tag names. The list has to be in the form
- * `tag1,tag2,...` e. g. `head,script,style`. An atomic tag is one whose child nodes should not be
- * compared - the entire tag should be treated as one token. This is useful for tags where it does
- * not make sense to insert `` and `` tags. If not used, the default list
- * `iframe,object,math,svg,script,video,head,style` will be used.
- * @return The combined HTML content with differences wrapped in `` and `` tags.
- */
-declare function diff(before: string, after: string, className?: string | null, dataPrefix?: string | null, atomicTags?: string | null): string;
-export = diff;
diff --git a/js/htmldiff.js b/js/htmldiff.js
deleted file mode 100644
index 0515fe2..0000000
--- a/js/htmldiff.js
+++ /dev/null
@@ -1,1351 +0,0 @@
-/**
- * htmldiff.js is a library that compares HTML content. It creates a diff between two
- * HTML documents by combining the two documents and wrapping the differences with
- * and tags. Here is a high-level overview of how the diff works.
- *
- * 1. Tokenize the before and after HTML with htmlToTokens.
- * 2. Generate a list of operations that convert the before list of tokens to the after
- * list of tokens with calculateOperations, which does the following:
- * a. Find all the matching blocks of tokens between the before and after lists of
- * tokens with findMatchingBlocks. This is done by finding the single longest
- * matching block with findMatch, then iteratively finding the next longest
- * matching blocks that precede and follow the longest matching block.
- * b. Determine insertions, deletions, and replacements from the matching blocks.
- * This is done in calculateOperations.
- * 3. Render the list of operations by wrapping tokens with and tags where
- * appropriate with renderOperations.
- *
- * Example usage:
- *
- * var htmldiff = require('htmldiff.js');
- *
- * htmldiff('
this is some text
', '
this is some more text
')
- * == '
this is some more text
'
- *
- * htmldiff('
this is some text
', '
this is some more text
', 'diff-class')
- * == '
this is some more text
'
- */
-(function(){
- 'use strict';
-
- function isEndOfTag(char){
- return char === '>';
- }
-
- function isStartOfTag(char){
- return char === '<';
- }
-
- function isWhitespace(char){
- return /^\s+$/.test(char);
- }
-
- /**
- * Determines if the given token is a tag.
- *
- * @param {string} token The token in question.
- *
- * @return {boolean|string} False if the token is not a tag, or the tag name otherwise.
- */
- function isTag(token){
- var match = token.match(/^\s*<([^!>][^>]*)>\s*$/);
- return !!match && match[1].trim().split(' ')[0];
- }
-
- function isntTag(token){
- return !isTag(token);
- }
-
- function isStartofHTMLComment(word){
- return /^ ");
+ expect(res.length).to.equal(8);
+ });
+ });
+
+ it("should identify contiguous whitespace as a single token", () => {
+ expect(cut("a b")).to.eql(tokenize(["a", " ", "b"]));
+ });
+
+ it("should identify a single space as a single token", () => {
+ expect(cut(" a b ")).to.eql(tokenize([" ", "a", " ", "b", " "]));
+ });
+
+ it("should identify self closing tags as tokens", () => {
+ expect(cut("
hellogoodbye
")).eql(tokenize(["
", "hello", "", "goodbye", "
"]));
+ });
+
+ describe("when encountering atomic tags", () => {
+ it("should identify an image tag as a single token", () => {
+ expect(cut('
')).eql(tokenize(["
", '', '', "
"]));
+ });
+
+ it("should identify an iframe tag as a single token", () => {
+ expect(cut('')).eql(tokenize(["
", '', "
"]));
+ });
+
+ it("should identify an object tag as a single token", () => {
+ expect(cut('')).eql(
+ tokenize(["
", '', "
"]),
+ );
+ });
+
+ it("should identify a math tag as a single token", () => {
+ const math =
+ '";
+ expect(cut(`
${math}
`)).eql(tokenize(["
", math, "
"]));
+ });
+
+ it("should identify an svg tag as a single token", () => {
+ const svg = '";
+ expect(cut(`
${svg}
`)).eql(tokenize(["
", svg, "
"]));
+ });
+
+ it("should identify a script tag as a single token", () => {
+ expect(cut('')).eql(tokenize(["
", '', "
"]));
+ });
+
+ it("should identify tags with data-htmldiff-id attribute as single token", () => {
+ expect(
+ cut('
hellogoodbye' + 'some stuff' + "
"),
+ ).eql(
+ tokenize([
+ "
",
+ 'hellogoodbye',
+ 'some stuff',
+ "
",
+ ]),
+ );
+ });
+
+ describe("nested atomic tags wrapping", () => {
+ it("should keep a data-htmldiff-id wrapper with nested same-tag children as one token", () => {
+ const atomic = '' + 'AB' + "Name";
+ expect(cut(`
${atomic}
`)).eql(tokenize(["
", atomic, "
"]));
+ });
+
+ it("should not close early on the first inner closing tag", () => {
+ const atomic = 'ab';
+ expect(cut(atomic)).eql(tokenize([atomic]));
+ });
+
+ it('should not treat a stray ">" in script content as a tag boundary', () => {
+ const atomic = "";
+ expect(cut(`
${atomic}
`)).eql(tokenize(["
", atomic, "
"]));
+ });
+
+ it("should ignore self-closing same-named children when counting depth", () => {
+ const atomic = 'xy';
+ expect(cut(atomic)).eql(tokenize([atomic]));
+ });
+
+ it("should ignore self-closing same-named children written with a space ()", () => {
+ const atomic = 'xy';
+ expect(cut(atomic)).eql(tokenize([atomic]));
+ });
+
+ it("should key a wrapper by its own data-htmldiff-id, not a nested child one", () => {
+ const atomic = '' + 'C';
+ expect(cut(atomic)[0].key).eql("s1");
+ });
+
+ it("should not key an unkeyed atomic tag by a nested child data-htmldiff-id", () => {
+ const atomic = 'C';
+ expect(cut(atomic)[0].key).eql("");
+ });
+
+ it("should not bump depth on differently-named tags that share a prefix", () => {
+ // a tag must not be matched by an atomic tag named "a" appearing as .
+ const atomic = 'hi';
+ expect(cut(atomic)).eql(tokenize([atomic]));
+ });
+ });
+
+ describe("tags sharing a prefix with atomic tag names", () => {
+ it("should not treat as the atomic tag a", () => {
+ expect(cut("x tail")).eql(tokenize(["", "x", "", " ", "tail"]));
+ });
+
+ it("should not treat as the atomic tag a", () => {
+ expect(cut("hi")).eql(tokenize(["", "hi", ""]));
+ });
+ });
+
+ describe("self-closing atomic tags", () => {
+ it("should end a self-closing data-htmldiff-id tag without swallowing trailing content", () => {
+ expect(cut('x old')).eql(tokenize(['', "x", " ", "old"]));
+ });
+
+ it("should end a self-closing name-based atomic tag without swallowing trailing content", () => {
+ expect(cut("tail")).eql(tokenize(["", "tail"]));
+ });
+ });
+
+ describe("quoted attribute values containing tag delimiters", () => {
+ it('should not end a tag on ">" inside a double-quoted attribute value', () => {
+ expect(cut('
text
')).eql(tokenize(['
', "text", "
"]));
+ });
+
+ it('should not end a tag on ">" inside a single-quoted attribute value', () => {
+ expect(cut("
x
")).eql(tokenize(["
", "x", "
"]));
+ });
+
+ it('should not end a void atomic tag on ">" inside an attribute value', () => {
+ expect(cut(' tail')).eql(tokenize(['', " ", "tail"]));
+ });
+
+ it('should not treat "/>" inside an attribute value as self-closing', () => {
+ const atomic = 'c';
+ expect(cut(atomic)).eql(tokenize([atomic]));
+ });
+
+ it('should keep an atomic tag with ">" in an attribute as one token', () => {
+ expect(cut('
x
tail')).eql(
+ tokenize(['
x
', " ", "tail"]),
+ );
+ });
+
+ it("should not treat apostrophes in atomic text content as quotes", () => {
+ expect(cut("
it's ok
tail")).eql(tokenize(["
it's ok
", " ", "tail"]));
+ });
+
+ it("should not treat apostrophes in comments inside atomic tags as quotes", () => {
+ expect(cut(" tail")).eql(tokenize(["", " ", "tail"]));
+ });
+ });
+
+ describe("void atomic tags", () => {
+ it("should end a void data-htmldiff-id tag written without a slash", () => {
+ expect(cut(' tail')).eql(tokenize(['', " ", "tail"]));
+ });
+
+ it("should end a void data-htmldiff-id br tag without swallowing trailing content", () => {
+ expect(cut(' y')).eql(tokenize([' ', "y"]));
+ });
+ });
+ });
+});
diff --git a/src/core/matching.spec.ts b/src/core/matching.spec.ts
new file mode 100644
index 0000000..7fea520
--- /dev/null
+++ b/src/core/matching.spec.ts
@@ -0,0 +1,140 @@
+import { expect } from "chai";
+import { createMap, createSegment, findBestMatch, findMatchingBlocks, Match, TokenMap } from "./matching";
+import { createToken, htmlToTokens, Token } from "./tokens";
+
+describe("findMatchingBlocks", () => {
+ const tokenize = (tokens: string[]): Token[] => tokens.map((token) => createToken(token));
+
+ describe("createMap", () => {
+ const cut = createMap;
+ let res: TokenMap;
+
+ it("should be a function", () => {
+ expect(cut).is.a("function");
+ });
+
+ describe("When the items exist in the search target", () => {
+ beforeEach(() => {
+ res = cut(tokenize(["a", "apple", "has", "a", "worm"]));
+ });
+
+ it('should find "a" twice', () => {
+ expect(res["a"].length).to.equal(2);
+ });
+
+ it('should find "a" at 0', () => {
+ expect(res["a"][0]).to.equal(0);
+ });
+
+ it('should find "a" at 3', () => {
+ expect(res["a"][1]).to.equal(3);
+ });
+
+ it('should find "has" at 2', () => {
+ expect(res["has"][0]).to.equal(2);
+ });
+ });
+ });
+
+ describe("findBestMatch", () => {
+ const cut = findBestMatch;
+ let res: Match | null;
+ const invoke = (before: Token[], after: Token[]): void => {
+ res = cut(createSegment(before, after, 0, 0));
+ };
+
+ describe("When there is a match", () => {
+ beforeEach(() => {
+ invoke(tokenize(["a", "dog", "bites"]), tokenize(["a", "dog", "bites", "a", "man"]));
+ });
+
+ it("should match the match", () => {
+ expect(res).to.exist;
+ const match = res as Match;
+ expect(match.startInBefore).equal(0);
+ expect(match.startInAfter).equal(0);
+ expect(match.length).equal(3);
+ expect(match.endInBefore).equal(2);
+ expect(match.endInAfter).equal(2);
+ });
+
+ describe("When the match is surrounded", () => {
+ beforeEach(() => {
+ invoke(tokenize(["dog", "bites"]), tokenize(["the", "dog", "bites", "a", "man"]));
+ });
+
+ it("should match with appropriate indexing", () => {
+ expect(res).to.exist;
+ const match = res as Match;
+ expect(match.startInBefore).to.equal(0);
+ expect(match.startInAfter).to.equal(1);
+ expect(match.endInBefore).to.equal(1);
+ expect(match.endInAfter).to.equal(2);
+ });
+ });
+ });
+
+ describe("When there is no match", () => {
+ beforeEach(() => {
+ invoke(tokenize(["the", "rat", "sqeaks"]), tokenize(["a", "dog", "bites", "a", "man"]));
+ });
+
+ it("should return nothing", () => {
+ expect(res).to.not.exist;
+ });
+ });
+ });
+
+ describe("findMatchingBlocks", () => {
+ const cut = findMatchingBlocks;
+ let res: Match[];
+
+ it("should be a function", () => {
+ expect(cut).is.a("function");
+ });
+
+ describe("When called with a single match", () => {
+ beforeEach(() => {
+ res = cut(createSegment(htmlToTokens("a dog bites"), htmlToTokens("when a dog bites it hurts"), 0, 0));
+ });
+
+ it("should return a match", () => {
+ expect(res.length).to.equal(1);
+ });
+ });
+
+ describe("When called with multiple matches", () => {
+ beforeEach(() => {
+ res = cut(createSegment(htmlToTokens("the dog bit a man"), htmlToTokens("the large brown dog bit a tall man"), 0, 0));
+ });
+
+ it("should return 3 matches", () => {
+ expect(res.length).to.equal(3);
+ });
+
+ it('should match "the"', () => {
+ expect(res[0].startInBefore).eql(0);
+ expect(res[0].startInAfter).eql(0);
+ expect(res[0].endInBefore).eql(0);
+ expect(res[0].endInAfter).eql(0);
+ expect(res[0].length).eql(1);
+ });
+
+ it('should match "dog bit a"', () => {
+ expect(res[1].startInBefore).eql(1);
+ expect(res[1].startInAfter).eql(5);
+ expect(res[1].endInBefore).eql(7);
+ expect(res[1].endInAfter).eql(11);
+ expect(res[1].length).eql(7);
+ });
+
+ it('should match "man"', () => {
+ expect(res[2].startInBefore).eql(8);
+ expect(res[2].startInAfter).eql(14);
+ expect(res[2].endInBefore).eql(8);
+ expect(res[2].endInAfter).eql(14);
+ expect(res[2].length).eql(1);
+ });
+ });
+ });
+});
diff --git a/src/core/matching.ts b/src/core/matching.ts
new file mode 100644
index 0000000..b61719a
--- /dev/null
+++ b/src/core/matching.ts
@@ -0,0 +1,360 @@
+/**
+ * Matching: finds the blocks of consecutive tokens that appear in both the before and the
+ * after token lists. The longest block is found first, then the blocks before and after it,
+ * recursively, until no match is left.
+ */
+import { Token } from "./tokens";
+
+/** The part of both documents a match is searched in. */
+export interface Segment {
+ beforeTokens: Token[];
+ afterTokens: Token[];
+ beforeMap: TokenMap;
+ afterMap: TokenMap;
+ beforeIndex: number;
+ afterIndex: number;
+}
+
+/** Token key to the indices of the tokens with that key. */
+export type TokenMap = Record;
+
+/**
+ * A Match stores the information of a matching block. A matching block is a list of
+ * consecutive tokens that appear in both the before and after lists of tokens.
+ */
+export class Match {
+ segment: Segment;
+ length: number;
+ startInBefore: number;
+ startInAfter: number;
+ endInBefore: number;
+ endInAfter: number;
+ segmentStartInBefore: number;
+ segmentStartInAfter: number;
+ segmentEndInBefore: number;
+ segmentEndInAfter: number;
+
+ /**
+ * @param startInBefore The index of the first token in the list of before tokens.
+ * @param startInAfter The index of the first token in the list of after tokens.
+ * @param length The number of consecutive matching tokens in this block.
+ * @param segment The segment where the match was found.
+ */
+ constructor(startInBefore: number, startInAfter: number, length: number, segment: Segment) {
+ this.segment = segment;
+ this.length = length;
+
+ this.startInBefore = startInBefore + segment.beforeIndex;
+ this.startInAfter = startInAfter + segment.afterIndex;
+ this.endInBefore = this.startInBefore + this.length - 1;
+ this.endInAfter = this.startInAfter + this.length - 1;
+
+ this.segmentStartInBefore = startInBefore;
+ this.segmentStartInAfter = startInAfter;
+ this.segmentEndInBefore = this.segmentStartInBefore + this.length - 1;
+ this.segmentEndInAfter = this.segmentStartInAfter + this.length - 1;
+ }
+}
+
+/**
+ * Creates a map from token key to an array of indices of locations of the matching token in
+ * the list of all tokens.
+ * @param tokens The list of tokens to be mapped.
+ * @returns A mapping that can be used to search for tokens.
+ */
+export function createMap(tokens: Token[]): TokenMap {
+ return tokens.reduce((map, token, index) => {
+ if (map[token.key]) {
+ map[token.key].push(index);
+ } else {
+ map[token.key] = [index];
+ }
+ return map;
+ }, Object.create(null) as TokenMap);
+}
+
+/**
+ * Compares two match objects to determine if the second match object comes before or after the
+ * first match object.
+ * @param m1 The first match object to compare.
+ * @param m2 The second match object to compare.
+ * @returns -1 if m2 should come before m1, 1 if m1 should come before m2, 0 if the two
+ * matches criss-cross each other.
+ */
+function compareMatches(m1: Match, m2: Match): -1 | 0 | 1 {
+ if (m2.endInBefore < m1.startInBefore && m2.endInAfter < m1.startInAfter) {
+ return -1;
+ }
+ if (m2.startInBefore > m1.endInBefore && m2.startInAfter > m1.endInAfter) {
+ return 1;
+ }
+ return 0;
+}
+
+interface MatchNode {
+ value: Match;
+ left: MatchNode | null;
+ right: MatchNode | null;
+}
+
+/** A binary search tree that keeps match objects in the proper order as they're found. */
+class MatchBinarySearchTree {
+ private root: MatchNode | null = null;
+
+ /**
+ * Adds a match to the binary search tree. A match overlapping an existing node is dropped.
+ * @param value The match to add.
+ */
+ add(value: Match): void {
+ const node: MatchNode = { value, left: null, right: null };
+
+ let current = this.root;
+ if (!current) {
+ this.root = node;
+ return;
+ }
+ for (;;) {
+ // Determine if the match value should go to the left or right of the current node.
+ const position = compareMatches(current.value, value);
+ if (position === -1) {
+ if (current.left) {
+ current = current.left;
+ } else {
+ current.left = node;
+ break;
+ }
+ } else if (position === 1) {
+ if (current.right) {
+ current = current.right;
+ } else {
+ current.right = node;
+ break;
+ }
+ } else {
+ // If 0 was returned from compareMatches, that means the node cannot
+ // be inserted because it overlaps an existing node.
+ break;
+ }
+ }
+ }
+
+ /**
+ * Converts the binary search tree into an array using an in-order traversal.
+ * @returns The matches in the binary search tree, in order.
+ */
+ toArray(): Match[] {
+ function inOrder(node: MatchNode | null, nodes: Match[]): Match[] {
+ if (node) {
+ inOrder(node.left, nodes);
+ nodes.push(node.value);
+ inOrder(node.right, nodes);
+ }
+ return nodes;
+ }
+
+ return inOrder(this.root, []);
+ }
+}
+
+/**
+ * Finds and returns the best match between the before and after arrays contained in the
+ * segment provided.
+ * @param segment The segment in which to look for a match.
+ * @returns The best match, or null when the segment has none.
+ */
+export function findBestMatch(segment: Segment): Match | null {
+ const beforeTokens = segment.beforeTokens;
+ const afterMap = segment.afterMap;
+ let lastSpace: number | null = null;
+ let bestMatch: Match | null = null;
+
+ // Iterate through the entirety of the beforeTokens to find the best match.
+ for (let beforeIndex = 0; beforeIndex < beforeTokens.length; beforeIndex++) {
+ let lookBehind = false;
+
+ // If the current best match is longer than the remaining tokens, we can bail because we
+ // won't find a better match.
+ const remainingTokens = beforeTokens.length - beforeIndex;
+ if (bestMatch && remainingTokens < bestMatch.length) {
+ break;
+ }
+
+ // If the current token is whitespace, make a note of it and move on. Trying to start a
+ // set of matches with whitespace is not efficient because it's too prevelant in most
+ // documents. Instead, if the next token yields a match, we'll see if the whitespace can
+ // be included in that match.
+ const beforeToken = beforeTokens[beforeIndex];
+ if (beforeToken.key === " ") {
+ lastSpace = beforeIndex;
+ continue;
+ }
+
+ // Check to see if we just skipped a space, if so, we'll ask getFullMatch to look behind
+ // by one token to see if it can include the whitespace.
+ if (lastSpace === beforeIndex - 1) {
+ lookBehind = true;
+ }
+
+ // If the current token is not found in the afterTokens, it won't match and we can move on.
+ const afterTokenLocations = afterMap[beforeToken.key];
+ if (!afterTokenLocations) {
+ continue;
+ }
+
+ // For each instance of the current token in afterTokens, let's see how big of a match
+ // we can build.
+ for (const afterIndex of afterTokenLocations) {
+ // getFullMatch will see how far the current token match will go in both
+ // beforeTokens and afterTokens.
+ const bestMatchLength = bestMatch ? bestMatch.length : 0;
+ const match = getFullMatch(segment, beforeIndex, afterIndex, bestMatchLength, lookBehind);
+
+ // If we got a new best match, we'll save it aside.
+ if (match && match.length > bestMatchLength) {
+ bestMatch = match;
+ }
+ }
+ }
+
+ return bestMatch;
+}
+
+/**
+ * Takes the start of a match, and expands it in the beforeTokens and afterTokens of the
+ * current segment as far as it can go.
+ * @param segment The segment object to search within when expanding the match.
+ * @param beforeStart The offset within beforeTokens to start looking.
+ * @param afterStart The offset within afterTokens to start looking.
+ * @param minLength The minimum length match that must be found.
+ * @param lookBehind If true, attempt to match a whitespace token just before the
+ * beforeStart and afterStart tokens.
+ * @returns The full match, or undefined when no match of the minimum length starts here.
+ */
+function getFullMatch(
+ segment: Segment,
+ beforeStart: number,
+ afterStart: number,
+ minLength: number,
+ lookBehind: boolean,
+): Match | undefined {
+ const beforeTokens = segment.beforeTokens;
+ const afterTokens = segment.afterTokens;
+
+ // If we already have a match that goes to the end of the document, no need to keep looking.
+ const minBeforeIndex = beforeStart + minLength;
+ const minAfterIndex = afterStart + minLength;
+ if (minBeforeIndex >= beforeTokens.length || minAfterIndex >= afterTokens.length) {
+ return undefined;
+ }
+
+ // If a minLength was provided, we can do a quick check to see if the tokens after that
+ // length match. If not, we won't be beating the previous best match, and we can bail out
+ // early.
+ if (minLength) {
+ const nextBeforeWord = beforeTokens[minBeforeIndex].key;
+ const nextAfterWord = afterTokens[minAfterIndex].key;
+ if (nextBeforeWord !== nextAfterWord) {
+ return undefined;
+ }
+ }
+
+ // Extend the current match as far foward as it can go, without overflowing beforeTokens or
+ // afterTokens.
+ let searching = true;
+ let currentLength = 1;
+ let beforeIndex = beforeStart + currentLength;
+ let afterIndex = afterStart + currentLength;
+
+ while (searching && beforeIndex < beforeTokens.length && afterIndex < afterTokens.length) {
+ const beforeWord = beforeTokens[beforeIndex].key;
+ const afterWord = afterTokens[afterIndex].key;
+ if (beforeWord === afterWord) {
+ currentLength++;
+ beforeIndex = beforeStart + currentLength;
+ afterIndex = afterStart + currentLength;
+ } else {
+ searching = false;
+ }
+ }
+
+ // If we've been asked to look behind, it's because both beforeTokens and afterTokens may
+ // have a whitespace token just behind the current match that was previously ignored. If so,
+ // we'll expand the current match to include it.
+ if (lookBehind && beforeStart > 0 && afterStart > 0) {
+ const prevBeforeKey = beforeTokens[beforeStart - 1].key;
+ const prevAfterKey = afterTokens[afterStart - 1].key;
+ if (prevBeforeKey === " " && prevAfterKey === " ") {
+ beforeStart--;
+ afterStart--;
+ currentLength++;
+ }
+ }
+
+ return new Match(beforeStart, afterStart, currentLength, segment);
+}
+
+/**
+ * Creates segment objects from the original document that can be used to restrict the area
+ * that findBestMatch and its helper functions search to increase performance.
+ * @param beforeTokens Tokens from the before document.
+ * @param afterTokens Tokens from the after document.
+ * @param beforeIndex The index within the before document where this segment begins.
+ * @param afterIndex The index within the after document where this segment begins.
+ * @returns The segment object.
+ */
+export function createSegment(beforeTokens: Token[], afterTokens: Token[], beforeIndex: number, afterIndex: number): Segment {
+ return {
+ beforeTokens,
+ afterTokens,
+ beforeMap: createMap(beforeTokens),
+ afterMap: createMap(afterTokens),
+ beforeIndex,
+ afterIndex,
+ };
+}
+
+/**
+ * Finds all the matching blocks within the given segment in the before and after lists of
+ * tokens.
+ * @param segment The segment that should be searched for matching blocks.
+ * @returns The list of matching blocks in this range.
+ */
+export function findMatchingBlocks(segment: Segment): Match[] {
+ // Create a binary search tree to hold the matches we find in order.
+ const matches = new MatchBinarySearchTree();
+ const segments = [segment];
+
+ // Each time the best match is found in a segment, zero, one or two new segments may be
+ // created from the parts of the original segment not included in the match. We will
+ // continue to iterate until all segments have been processed.
+ while (segments.length) {
+ const current = segments.pop() as Segment;
+ const match = findBestMatch(current);
+
+ if (match && match.length) {
+ // If there's an unmatched area at the start of the segment, create a new segment
+ // from that area and throw it into the segments array to get processed.
+ if (match.segmentStartInBefore > 0 && match.segmentStartInAfter > 0) {
+ const leftBeforeTokens = current.beforeTokens.slice(0, match.segmentStartInBefore);
+ const leftAfterTokens = current.afterTokens.slice(0, match.segmentStartInAfter);
+
+ segments.push(createSegment(leftBeforeTokens, leftAfterTokens, current.beforeIndex, current.afterIndex));
+ }
+
+ // If there's an unmatched area at the end of the segment, create a new segment from
+ // that area and throw it into the segments array to get processed.
+ const rightBeforeTokens = current.beforeTokens.slice(match.segmentEndInBefore + 1);
+ const rightAfterTokens = current.afterTokens.slice(match.segmentEndInAfter + 1);
+ const rightBeforeIndex = current.beforeIndex + match.segmentEndInBefore + 1;
+ const rightAfterIndex = current.afterIndex + match.segmentEndInAfter + 1;
+
+ if (rightBeforeTokens.length && rightAfterTokens.length) {
+ segments.push(createSegment(rightBeforeTokens, rightAfterTokens, rightBeforeIndex, rightAfterIndex));
+ }
+
+ matches.add(match);
+ }
+ }
+
+ return matches.toArray();
+}
diff --git a/src/core/operations.spec.ts b/src/core/operations.spec.ts
new file mode 100644
index 0000000..ebcb306
--- /dev/null
+++ b/src/core/operations.spec.ts
@@ -0,0 +1,280 @@
+// Calculates the differences into a list of edit operations.
+import { expect } from "chai";
+import { calculateOperations, Operation } from "./operations";
+import { htmlToTokens } from "./tokens";
+
+describe("calculateOperations", () => {
+ const cut = calculateOperations;
+ const tokenize = htmlToTokens;
+ let res: Operation[];
+
+ it("should be a function", () => {
+ expect(cut).is.a("function");
+ });
+
+ describe("Actions", () => {
+ describe("In the middle", () => {
+ describe("Replace", () => {
+ beforeEach(() => {
+ res = cut(tokenize("working on it"), tokenize("working in it"));
+ });
+
+ it("should result in 3 operations", () => {
+ expect(res.length).to.equal(3);
+ });
+
+ it('should replace "on"', () => {
+ expect(res[1]).eql({ action: "replace", startInBefore: 2, endInBefore: 2, startInAfter: 2, endInAfter: 2 });
+ });
+ });
+
+ describe("Insert", () => {
+ beforeEach(() => {
+ res = cut(tokenize("working it"), tokenize("working in it"));
+ });
+
+ it("should result in 3 operations", () => {
+ expect(res.length).to.equal(3);
+ });
+
+ it('should show an insert for "on"', () => {
+ expect(res[1]).eql({ action: "insert", startInBefore: 2, endInBefore: null, startInAfter: 2, endInAfter: 3 });
+ });
+
+ describe("More than one word", () => {
+ beforeEach(() => {
+ res = cut(tokenize("working it"), tokenize("working all up on it"));
+ });
+
+ it("should still have 3 operations", () => {
+ expect(res.length).to.equal(3);
+ });
+
+ it("should show a big insert", () => {
+ expect(res[1]).eql({ action: "insert", startInBefore: 2, endInBefore: null, startInAfter: 2, endInAfter: 7 });
+ });
+ });
+ });
+
+ describe("Delete", () => {
+ beforeEach(() => {
+ res = cut(tokenize("this is a lot of text"), tokenize("this is text"));
+ });
+
+ it("should return 3 operations", () => {
+ expect(res.length).to.equal(3);
+ });
+
+ it("should show the delete in the middle", () => {
+ expect(res[1]).eql({ action: "delete", startInBefore: 4, endInBefore: 9, startInAfter: 4, endInAfter: null });
+ });
+ });
+
+ describe("Equal", () => {
+ beforeEach(() => {
+ res = cut(tokenize("this is what it sounds like"), tokenize("this is what it sounds like"));
+ });
+
+ it("should return a single op", () => {
+ expect(res.length).to.equal(1);
+ expect(res[0]).eql({ action: "equal", startInBefore: 0, endInBefore: 10, startInAfter: 0, endInAfter: 10 });
+ });
+ });
+ });
+
+ describe("At the beginning", () => {
+ describe("Replace", () => {
+ beforeEach(() => {
+ res = cut(tokenize("I dont like veggies"), tokenize("Joe loves veggies"));
+ });
+
+ it("should return 2 operations", () => {
+ expect(res.length).to.equal(2);
+ });
+
+ it("should have a replace at the beginning", () => {
+ expect(res[0]).eql({ action: "replace", startInBefore: 0, endInBefore: 4, startInAfter: 0, endInAfter: 2 });
+ });
+ });
+
+ describe("Insert", () => {
+ beforeEach(() => {
+ res = cut(tokenize("dog"), tokenize("the shaggy dog"));
+ });
+
+ it("should return 2 operations", () => {
+ expect(res.length).to.equal(2);
+ });
+
+ it("should have an insert at the beginning", () => {
+ expect(res[0]).eql({ action: "insert", startInBefore: 0, endInBefore: null, startInAfter: 0, endInAfter: 3 });
+ });
+ });
+
+ describe("Delete", () => {
+ beforeEach(() => {
+ res = cut(tokenize("awesome dog barks"), tokenize("dog barks"));
+ });
+
+ it("should return 2 operations", () => {
+ expect(res.length).to.equal(2);
+ });
+
+ it("should have a delete at the beginning", () => {
+ expect(res[0]).eql({ action: "delete", startInBefore: 0, endInBefore: 1, startInAfter: 0, endInAfter: null });
+ });
+ });
+ });
+
+ describe("At the end", () => {
+ describe("Replace", () => {
+ beforeEach(() => {
+ res = cut(tokenize("the dog bit the cat"), tokenize("the dog bit a bird"));
+ });
+
+ it("should return 2 operations", () => {
+ expect(res.length).to.equal(2);
+ });
+
+ it("should have a replace at the end", () => {
+ expect(res[1]).eql({ action: "replace", startInBefore: 6, endInBefore: 8, startInAfter: 6, endInAfter: 8 });
+ });
+ });
+
+ describe("Insert", () => {
+ beforeEach(() => {
+ res = cut(tokenize("this is a dog"), tokenize("this is a dog that barks"));
+ });
+
+ it("should return 2 operations", () => {
+ expect(res.length).to.equal(2);
+ });
+
+ it("should have an Insert at the end", () => {
+ expect(res[1]).eql({ action: "insert", startInBefore: 7, endInBefore: null, startInAfter: 7, endInAfter: 10 });
+ });
+ });
+
+ describe("Delete", () => {
+ beforeEach(() => {
+ res = cut(tokenize("this is a dog that barks"), tokenize("this is a dog"));
+ });
+
+ it("should have 2 operations", () => {
+ expect(res.length).to.equal(2);
+ });
+
+ it("should have a delete at the end", () => {
+ expect(res[1]).eql({ action: "delete", startInBefore: 7, endInBefore: 10, startInAfter: 7, endInAfter: null });
+ });
+ });
+ });
+ });
+
+ describe("Action Combination", () => {
+ describe("dont absorb non-single-whitespace tokens", () => {
+ beforeEach(() => {
+ res = cut(tokenize("I am awesome"), tokenize("You are great"));
+ });
+
+ it("should return 3 actions", () => {
+ expect(res.length).to.equal(1);
+ });
+
+ it("should have a replace first", () => {
+ expect(res[0].action).to.equal("replace");
+ });
+ });
+ });
+
+ // elements htmldiff treats as one token: a change inside them is a replacement of the whole
+ describe("atomic elements", () => {
+ const equalOperation = { action: "equal", startInBefore: 0, endInBefore: 0, startInAfter: 0, endInAfter: 0 };
+ const replaceOperation = { action: "replace", startInBefore: 0, endInBefore: 0, startInAfter: 0, endInAfter: 0 };
+
+ describe("Image Differences", () => {
+ it("show two images as different if their src attributes are different", () => {
+ const ops = cut(tokenize(''), tokenize(''));
+ expect(ops.length).to.equal(1);
+ expect(ops[0]).to.eql(replaceOperation);
+ });
+
+ it("should show two images are the same if their src attributes are the same", () => {
+ const ops = cut(tokenize(''), tokenize(''));
+ expect(ops.length).to.equal(1);
+ expect(ops[0]).to.eql(equalOperation);
+ });
+ });
+
+ describe("Widget Differences", () => {
+ it("show two widgets as different if their data attributes are different", () => {
+ const ops = cut(tokenize(''), tokenize(''));
+ expect(ops.length).to.equal(1);
+ expect(ops[0]).to.eql(replaceOperation);
+ });
+
+ it("should show two widgets are the same if their data attributes are the same", () => {
+ const ops = cut(
+ tokenize(''),
+ tokenize(''),
+ );
+ expect(ops.length).to.equal(1);
+ expect(ops[0]).to.eql(equalOperation);
+ });
+ });
+
+ describe("Math Differences", () => {
+ it("should show two math elements as different if their contents are different", () => {
+ const ops = cut(
+ tokenize(''),
+ tokenize(''),
+ );
+ expect(ops.length).to.equal(1);
+ expect(ops[0]).to.eql(replaceOperation);
+ });
+
+ it("should show two math elements as the same if their contents are the same", () => {
+ const ops = cut(
+ tokenize(''),
+ tokenize(''),
+ );
+ expect(ops.length).to.equal(1);
+ expect(ops[0]).to.eql(equalOperation);
+ });
+ });
+
+ describe("Video Differences", () => {
+ it("show two widgets as different if their data attributes are different", () => {
+ const ops = cut(
+ tokenize(''),
+ tokenize(''),
+ );
+ expect(ops.length).to.equal(1);
+ expect(ops[0]).to.eql(replaceOperation);
+ });
+
+ it("should show two widgets are the same if their data attributes are the same", () => {
+ const ops = cut(
+ tokenize(''),
+ tokenize(''),
+ );
+ expect(ops.length).to.equal(1);
+ expect(ops[0]).to.eql(equalOperation);
+ });
+ });
+
+ describe("iframe Differences", () => {
+ it("show two widgets as different if their data attributes are different", () => {
+ const ops = cut(tokenize(''), tokenize(''));
+ expect(ops.length).to.equal(1);
+ expect(ops[0]).to.eql(replaceOperation);
+ });
+
+ it("should show two widgets are the same if their data attributes are the same", () => {
+ const ops = cut(tokenize(''), tokenize(''));
+ expect(ops.length).to.equal(1);
+ expect(ops[0]).to.eql(equalOperation);
+ });
+ });
+ });
+});
diff --git a/src/core/operations.ts b/src/core/operations.ts
new file mode 100644
index 0000000..804f58b
--- /dev/null
+++ b/src/core/operations.ts
@@ -0,0 +1,105 @@
+/**
+ * Operations: turns the matching blocks into the list of equal, insert, delete and replace
+ * steps that transform the before tokens into the after tokens.
+ */
+import { createSegment, findMatchingBlocks, Match } from "./matching";
+import { Token } from "./tokens";
+
+/** What happened to a range of tokens. */
+export type OperationAction = "equal" | "insert" | "delete" | "replace";
+
+/**
+ * A range of tokens on both sides and what happened to it. The end is null on the side an
+ * operation does not touch: an insert has no before range, a delete no after range.
+ */
+export interface Operation {
+ action: OperationAction;
+ startInBefore: number;
+ endInBefore: number | null;
+ startInAfter: number;
+ endInAfter: number | null;
+}
+
+/**
+ * Gets a list of operations required to transform the before list of tokens into the
+ * after list of tokens. An operation describes whether a particular list of consecutive
+ * tokens are equal, replaced, inserted, or deleted.
+ * @param beforeTokens The before list of tokens.
+ * @param afterTokens The after list of tokens.
+ * @returns The list of operations to transform the before list of tokens into the after list.
+ */
+export function calculateOperations(beforeTokens: Token[], afterTokens: Token[]): Operation[] {
+ if (!beforeTokens) throw new Error("Missing beforeTokens");
+ if (!afterTokens) throw new Error("Missing afterTokens");
+
+ let positionInBefore = 0;
+ let positionInAfter = 0;
+ const operations: Operation[] = [];
+ const segment = createSegment(beforeTokens, afterTokens, 0, 0);
+ const matches = findMatchingBlocks(segment);
+ matches.push(new Match(beforeTokens.length, afterTokens.length, 0, segment));
+
+ for (let index = 0; index < matches.length; index++) {
+ const match = matches[index];
+ let actionUpToMatchPositions: OperationAction | "none" = "none";
+ if (positionInBefore === match.startInBefore) {
+ if (positionInAfter !== match.startInAfter) {
+ actionUpToMatchPositions = "insert";
+ }
+ } else {
+ actionUpToMatchPositions = "delete";
+ if (positionInAfter !== match.startInAfter) {
+ actionUpToMatchPositions = "replace";
+ }
+ }
+ if (actionUpToMatchPositions !== "none") {
+ operations.push({
+ action: actionUpToMatchPositions,
+ startInBefore: positionInBefore,
+ endInBefore: actionUpToMatchPositions !== "insert" ? match.startInBefore - 1 : null,
+ startInAfter: positionInAfter,
+ endInAfter: actionUpToMatchPositions !== "delete" ? match.startInAfter - 1 : null,
+ });
+ }
+ if (match.length !== 0) {
+ operations.push({
+ action: "equal",
+ startInBefore: match.startInBefore,
+ endInBefore: match.endInBefore,
+ startInAfter: match.startInAfter,
+ endInAfter: match.endInAfter,
+ });
+ }
+ positionInBefore = match.endInBefore + 1;
+ positionInAfter = match.endInAfter + 1;
+ }
+
+ const postProcessed: Operation[] = [];
+ let lastOp: Operation | { action: "none" } = { action: "none" };
+
+ function isSingleWhitespace(op: Operation): boolean {
+ if (op.action !== "equal") {
+ return false;
+ }
+ if ((op.endInBefore as number) - op.startInBefore !== 0) {
+ return false;
+ }
+ return /^\s$/.test(String(beforeTokens.slice(op.startInBefore, (op.endInBefore as number) + 1)));
+ }
+
+ for (let i = 0; i < operations.length; i++) {
+ const op = operations[i];
+
+ if (
+ lastOp.action === "replace" &&
+ (isSingleWhitespace(op) || op.action === "replace")
+ ) {
+ lastOp.endInBefore = op.endInBefore;
+ lastOp.endInAfter = op.endInAfter;
+ } else {
+ postProcessed.push(op);
+ lastOp = op;
+ }
+ }
+ return postProcessed;
+}
diff --git a/src/core/rendering.spec.ts b/src/core/rendering.spec.ts
new file mode 100644
index 0000000..0c1a304
--- /dev/null
+++ b/src/core/rendering.spec.ts
@@ -0,0 +1,225 @@
+import { expect } from "chai";
+import { diffCore } from "./diff";
+import { calculateOperations } from "./operations";
+import { findOpeningTagEnd, isInnerDiffToken, renderInnerDiff, renderOperations, splitAtomicTokenString, wrap } from "./rendering";
+import { createToken, Token } from "./tokens";
+
+describe("rendering", () => {
+ describe("findOpeningTagEnd", () => {
+ it("finds the end of a plain opening tag", () => {
+ expect(findOpeningTagEnd('
text
')).to.equal(14);
+ });
+
+ it("skips a > inside a quoted attribute", () => {
+ expect(findOpeningTagEnd('
x
')).to.equal(18);
+ });
+
+ it("returns -1 for an unterminated tag", () => {
+ expect(findOpeningTagEnd('
{
+ it("splits an element into its tags and content", () => {
+ expect(splitAtomicTokenString('
content
')).to.deep.equal({ openingTag: '
', innerHtml: "content", closingTag: "
" });
+ });
+
+ it("splits a self-closing element into its tag alone", () => {
+ expect(splitAtomicTokenString('')).to.deep.equal({ openingTag: '', innerHtml: "", closingTag: "" });
+ });
+
+ it("returns null when no closing tag follows the content", () => {
+ expect(splitAtomicTokenString("
content")).to.equal(null);
+ });
+ });
+
+ describe("isInnerDiffToken", () => {
+ it("accepts an atomic element that opted in", () => {
+ expect(isInnerDiffToken('
x
')).to.equal(true);
+ });
+
+ it("accepts a bare opt-in attribute", () => {
+ expect(isInnerDiffToken('
"]));
+ });
+
+ it("should keep the change inside the
", () => {
+ expect(res).to.equal('
this' + 'I is awesome
');
+ });
+ });
+ });
+
+ describe("empty tokens", () => {
+ it("should not be wrapped", () => {
+ res = cut(tokenize(["text"]), tokenize(["text", " "]));
+
+ expect(res).to.equal("text");
+ });
+ });
+
+ describe("tags with attributes", () => {
+ it("should treat attribute changes as equal and output the after tag", () => {
+ res = cut(
+ tokenize(["
", "this", " ", "is", " ", "awesome", "
"]),
+ tokenize(['
', "this", " ", "is", " ", "awesome", "
"]),
+ );
+
+ expect(res).to.equal('
this is awesome
');
+ });
+
+ it("should show changes within tags with different attributes", () => {
+ res = cut(
+ tokenize(["
", "this", " ", "is", " ", "awesome", "
"]),
+ tokenize(['
', "that", " ", "is", " ", "awesome", "
"]),
+ );
+
+ expect(res).to.equal(
+ '
' + 'this' + "that is awesome
",
+ );
+ });
+ });
+
+ describe("wrappable tags", () => {
+ it("should wrap void tags", () => {
+ res = cut(tokenize(["old", " ", "text"]), tokenize(["new", " ", " ", "text"]));
+
+ expect(res).to.equal('old' + 'new text');
+ });
+
+ it("should wrap atomic tags independently", () => {
+ res = cut(tokenize(["old", '', " ", "text"]), tokenize(["new", " ", "text"]));
+
+ expect(res).to.equal(
+ 'old' +
+ '' +
+ 'new text',
+ );
+ });
+ });
+ });
+});
diff --git a/src/core/rendering.ts b/src/core/rendering.ts
new file mode 100644
index 0000000..89ec7eb
--- /dev/null
+++ b/src/core/rendering.ts
@@ -0,0 +1,445 @@
+/**
+ * Rendering: writes the operations back out as HTML, wrapping inserted and deleted tokens in
+ * and tags and diffing the content of opted-in atomic elements recursively.
+ */
+import {
+ buildAtomicTagsRegExp,
+ dataHtmlDiffInnerDiffAtomicTagsRegExp,
+ dataHtmlDiffInnerDiffRegExp,
+ defaultInnerDiffAtomicTagsRegExp,
+ getAtomicTagsRegExp,
+ isStartOfAtomicTag,
+ noAtomicTagsRegExp,
+ setAtomicTagsRegExp,
+} from "./atomicTags";
+import { Operation } from "./operations";
+import { isTag, isVoidTag, isWrappable, Token } from "./tokens";
+
+/**
+ * Diffs two fragments of HTML with the active atomic tags. The renderer gets it injected to
+ * diff the content of opted-in elements recursively; see the diff module.
+ */
+export type ContentDiff = (before: string, after: string, className?: string | null, dataPrefix?: string | null) => string;
+
+// The number of currently active recursive inner diffs. Recursion is governed per element
+// (each nesting level requires its own data-htmldiff-inner-diff attribute), so the depth is
+// naturally bounded by the nesting of opted-in elements; the cap is only a backstop against
+// pathologically deep documents.
+let innerDiffDepth = 0;
+const maxInnerDiffDepth = 10;
+
+interface TokenNote {
+ isWrappable: boolean;
+ insertedTag: boolean;
+}
+
+/** A run of tokens that are all wrappable or all not. */
+interface TokenSegment {
+ isWrappable: boolean;
+ tokens: string[];
+}
+
+/**
+ * A TokenWrapper provides a utility for grouping segments of tokens based on whether they're
+ * wrappable or not. A tag is considered wrappable if it is closed within the given set of
+ * tokens. For example, given the following tokens:
+ *
+ * ['', 'this', ' ', 'is', ' ', 'a', ' ', '', 'test', '', '!']
+ *
+ * The first '' is not considered wrappable since the tag is not fully contained within the
+ * array of tokens. The '', 'test', and '' would be a part of the same wrappable segment
+ * since the entire bold tag is within the set of tokens.
+ */
+export class TokenWrapper {
+ private readonly tokens: string[];
+ private readonly notes: TokenNote[];
+
+ /**
+ * @param tokens The tokens to group.
+ */
+ constructor(tokens: string[]) {
+ this.tokens = tokens;
+ this.notes = tokens.reduce<{ notes: TokenNote[]; tagStack: { tag: string; position: number }[] }>(
+ (data, token, index) => {
+ data.notes.push({
+ isWrappable: isWrappable(token),
+ insertedTag: false,
+ });
+
+ const tag = !isVoidTag(token) && isTag(token);
+ const lastEntry = data.tagStack[data.tagStack.length - 1];
+ if (tag) {
+ if (lastEntry && "/" + lastEntry.tag === tag) {
+ data.notes[lastEntry.position].insertedTag = true;
+ data.tagStack.pop();
+ } else {
+ data.tagStack.push({ tag, position: index });
+ }
+ }
+ return data;
+ },
+ { notes: [], tagStack: [] },
+ ).notes;
+ }
+
+ /**
+ * Wraps the contained tokens in tags based on output given by a map function. Each segment
+ * of tokens will be visited. A segment is a continuous run of either all wrappable tokens or
+ * unwrappable tokens. The given map function will be called with each segment of tokens and
+ * the resulting strings will be combined to form the wrapped HTML.
+ * @param mapFn Called with each segment; the result should be a string.
+ * @param tagFn Called with the opening tag of every tag inserted whole, to mark it.
+ * @returns The wrapped HTML.
+ */
+ combine(mapFn: (segment: TokenSegment) => string, tagFn: (openingTag: string) => string): string {
+ const notes = this.notes;
+ const tokens = this.tokens.slice();
+ const segments = tokens.reduce<{
+ list: TokenSegment[];
+ status: boolean | null;
+ lastIndex: number;
+ lastWasAtomic: boolean;
+ }>(
+ (data, token, index) => {
+ if (notes[index].insertedTag) {
+ tokens[index] = tagFn(tokens[index]);
+ }
+ if (data.status === null) {
+ data.status = notes[index].isWrappable;
+ }
+ const status = notes[index].isWrappable;
+
+ // Handling atomic tags wrapping independently
+ // each atomic tag is wrapped with their own ins/del tags
+ const isAtomic = !!isStartOfAtomicTag(token);
+ if (status !== data.status || (isAtomic && index > data.lastIndex) || data.lastWasAtomic) {
+ data.list.push({
+ isWrappable: data.status,
+ tokens: tokens.slice(data.lastIndex, index),
+ });
+ data.lastIndex = index;
+ data.status = status;
+ }
+ // tracking if the last token was an atomic tag
+ // if so then we break the segment and wrap them
+ data.lastWasAtomic = isAtomic;
+ if (index === tokens.length - 1) {
+ data.list.push({
+ isWrappable: data.status,
+ tokens: tokens.slice(data.lastIndex, index + 1),
+ });
+ }
+ return data;
+ },
+ { list: [], status: null, lastIndex: 0, lastWasAtomic: false },
+ ).list;
+
+ return segments.map(mapFn).join("");
+ }
+}
+
+/**
+ * Wraps and concatenates a list of tokens with a tag. Does not wrap tag tokens, unless they
+ * are wrappable (i.e. void and atomic tags).
+ * @param tag The tag name of the wrapper tags.
+ * @param content The list of tokens to wrap.
+ * @param opIndex The index of the operation, written to the data attribute.
+ * @param dataPrefix (Optional) The prefix to use in data attributes.
+ * @param className (Optional) The class name to include in the wrapper tag.
+ * @returns The wrapped HTML.
+ */
+export function wrap(
+ tag: string,
+ content: string[],
+ opIndex: number,
+ dataPrefix?: string | null,
+ className?: string | null,
+): string {
+ const wrapper = new TokenWrapper(content);
+ const prefix = dataPrefix ? dataPrefix + "-" : "";
+ let attrs = ` data-${prefix}operation-index="${opIndex}"`;
+ if (className) {
+ attrs += ' class="' + className + '"';
+ }
+
+ return wrapper.combine(
+ (segment) => {
+ if (segment.isWrappable) {
+ const val = segment.tokens.join("");
+ if (val.trim()) {
+ return "<" + tag + attrs + ">" + val + "" + tag + ">";
+ }
+ } else {
+ return segment.tokens.join("");
+ }
+ return "";
+ },
+ (openingTag) => {
+ let dataAttrs = ' data-diff-node="' + tag + '"';
+ dataAttrs += ` data-${prefix}operation-index="${opIndex}"`;
+
+ return openingTag.replace(/>\s*$/, dataAttrs + "$&");
+ },
+ );
+}
+
+/**
+ * Checks whether a token is an atomic tag that opted into the recursive inner diff via the
+ * data-htmldiff-inner-diff attribute. A bare attribute or any value other than "false" counts
+ * as opted in. Opted-in elements nested inside other opted-in elements are diffed recursively
+ * as well, up to the depth cap; beyond it, opted-in tokens are rendered verbatim like any
+ * other atomic token.
+ * @param tokenString The token string to check.
+ * @returns True if the token should get a recursive inner diff.
+ */
+export function isInnerDiffToken(tokenString: string): boolean {
+ if (innerDiffDepth >= maxInnerDiffDepth || !isStartOfAtomicTag(tokenString)) {
+ return false;
+ }
+ const attr = dataHtmlDiffInnerDiffRegExp.exec(tokenString);
+ return !!attr && attr[1] !== "false";
+}
+
+/**
+ * Finds the index of the '>' that ends the opening tag at the start of the given token
+ * string, skipping any '>' inside quoted attribute values (e.g. title="a > b").
+ * @param tokenString The token string starting with an opening tag.
+ * @returns The index of the closing '>' of the opening tag, or -1 if there is none (e.g. an
+ * unterminated tag or an unbalanced attribute quote).
+ */
+export function findOpeningTagEnd(tokenString: string): number {
+ let quote: string | null = null;
+ for (let i = 0; i < tokenString.length; i++) {
+ const char = tokenString[i];
+ // quote is closed
+ if (char === quote) {
+ quote = null;
+ continue;
+ }
+ // inside quote
+ if (quote) {
+ continue;
+ }
+ // quote start
+ if (char === '"' || char === "'") {
+ quote = char;
+ continue;
+ }
+ // not inside quote, check for tag end
+ if (char === ">") {
+ return i;
+ }
+ }
+ return -1;
+}
+
+/** An atomic token split into its tags and content. */
+export interface SplitToken {
+ openingTag: string;
+ innerHtml: string;
+ closingTag: string;
+}
+
+/**
+ * Splits an atomic token string into its opening tag, inner HTML and closing tag. A token
+ * consisting of a single tag (a void or self-closing element) has an empty inner HTML and no
+ * closing tag.
+ * @param tokenString The atomic token string, e.g. '
content
'.
+ * @returns The parts, or null if the token cannot be split (e.g. an unterminated tag).
+ */
+export function splitAtomicTokenString(tokenString: string): SplitToken | null {
+ const openingTagEnd = findOpeningTagEnd(tokenString);
+ if (openingTagEnd === -1) {
+ return null;
+ }
+ if (openingTagEnd === tokenString.length - 1) {
+ // The token is a single tag (void or self-closing): the element has no content.
+ return {
+ openingTag: tokenString,
+ innerHtml: "",
+ closingTag: "",
+ };
+ }
+ const closingTagStart = tokenString.lastIndexOf("<");
+ if (closingTagStart <= openingTagEnd || tokenString[closingTagStart + 1] !== "/") {
+ return null;
+ }
+ return {
+ openingTag: tokenString.slice(0, openingTagEnd + 1),
+ innerHtml: tokenString.slice(openingTagEnd + 1, closingTagStart),
+ closingTag: tokenString.slice(closingTagStart),
+ };
+}
+
+/**
+ * Renders the recursive inner diff of two matched atomic tokens with equal keys but different
+ * content. The after version's opening and closing tags are emitted with the diff of the two
+ * inner HTML fragments in between. Inside the recursion the default atomic tags without 'a'
+ * are used, so link text is diffed word by word and href-only changes do not produce any
+ * markup. The after version's data-htmldiff-inner-diff-atomic-tags attribute overrides that
+ * list. Nested opted-in elements are diffed recursively as well, up to a hardcoded depth cap.
+ * @param beforeString The before version of the atomic token.
+ * @param afterString The after version of the atomic token.
+ * @param diffContent Diffs the two inner HTML fragments.
+ * @param dataPrefix (Optional) The prefix to use in data attributes.
+ * @param className (Optional) The class name to include in the wrapper tag.
+ * @returns The rendered element with inner differences wrapped in ins/del tags.
+ */
+export function renderInnerDiff(
+ beforeString: string,
+ afterString: string,
+ diffContent: ContentDiff,
+ dataPrefix?: string | null,
+ className?: string | null,
+): string {
+ const before = splitAtomicTokenString(beforeString);
+ const after = splitAtomicTokenString(afterString);
+ if (!before || !after) {
+ return afterString;
+ }
+ const atomicTagsOverride = dataHtmlDiffInnerDiffAtomicTagsRegExp.exec(afterString);
+ const outerAtomicTagsRegExp = getAtomicTagsRegExp();
+ innerDiffDepth++;
+ let innerDiff: string;
+ // the depth and the atomic tags must be restored in case of an error, hence try/finally
+ try {
+ setAtomicTagsRegExp(defaultInnerDiffAtomicTagsRegExp);
+
+ if (atomicTagsOverride) {
+ setAtomicTagsRegExp(atomicTagsOverride[1] ? buildAtomicTagsRegExp(atomicTagsOverride[1]) : noAtomicTagsRegExp);
+ }
+
+ innerDiff = diffContent(before.innerHtml, after.innerHtml, className, dataPrefix);
+ } finally {
+ innerDiffDepth--;
+ setAtomicTagsRegExp(outerAtomicTagsRegExp);
+ }
+ return after.openingTag + innerDiff + after.closingTag;
+}
+
+type OperationRenderer = (
+ op: Operation,
+ beforeTokens: Token[],
+ afterTokens: Token[],
+ opIndex: number,
+ diffContent: ContentDiff,
+ dataPrefix?: string | null,
+ className?: string | null,
+) => string;
+
+function renderEqual(
+ op: Operation,
+ beforeTokens: Token[],
+ afterTokens: Token[],
+ _opIndex: number,
+ diffContent: ContentDiff,
+ dataPrefix?: string | null,
+ className?: string | null,
+): string {
+ // Tokens in an equal operation pair up one to one between before and after. Equal keys do
+ // not guarantee equal strings (e.g. atomic tokens matched by data-htmldiff-id): elements
+ // that opted in via data-htmldiff-inner-diff get a recursive diff of their content,
+ // everything else renders the after version.
+ let result = "";
+ for (let i = 0; op.startInAfter + i <= (op.endInAfter as number); i++) {
+ const afterToken = afterTokens[op.startInAfter + i];
+ const beforeToken = beforeTokens[op.startInBefore + i];
+ if (beforeToken && beforeToken.string !== afterToken.string && isInnerDiffToken(afterToken.string)) {
+ result += renderInnerDiff(beforeToken.string, afterToken.string, diffContent, dataPrefix, className);
+ } else {
+ result += afterToken.string;
+ }
+ }
+ return result;
+}
+
+function renderInsert(
+ op: Operation,
+ _beforeTokens: Token[],
+ afterTokens: Token[],
+ opIndex: number,
+ _diffContent: ContentDiff,
+ dataPrefix?: string | null,
+ className?: string | null,
+): string {
+ const tokens = afterTokens.slice(op.startInAfter, (op.endInAfter as number) + 1);
+ const val = tokens.map((token) => token.string);
+
+ const res = wrap("ins", val, opIndex, dataPrefix, className);
+
+ // handling inserted tags, see https://matrixreq.atlassian.net/browse/MATRIX-7876
+ if (/^<[^./]+?>$/.exec(res)) {
+ return `${res.slice(0, res.length - 1)} data-inserted="true">`;
+ }
+
+ return res;
+}
+
+function renderDelete(
+ op: Operation,
+ beforeTokens: Token[],
+ _afterTokens: Token[],
+ opIndex: number,
+ _diffContent: ContentDiff,
+ dataPrefix?: string | null,
+ className?: string | null,
+): string {
+ const tokens = beforeTokens.slice(op.startInBefore, (op.endInBefore as number) + 1);
+ const val = tokens.map((token) => token.string);
+ const res = wrap("del", val, opIndex, dataPrefix, className);
+
+ // handling cases like deleted
, see https://matrixreq.atlassian.net/browse/MATRIX-7688
+ if (/^<\/.+?><.+?>$/.exec(res) && !res.includes("del")) {
+ return `${val.slice(1, val.length - 1).join("")}`;
+ }
+
+ return res;
+}
+
+function renderReplace(
+ op: Operation,
+ beforeTokens: Token[],
+ afterTokens: Token[],
+ opIndex: number,
+ diffContent: ContentDiff,
+ dataPrefix?: string | null,
+ className?: string | null,
+): string {
+ return (
+ renderDelete(op, beforeTokens, afterTokens, opIndex, diffContent, dataPrefix, className) +
+ renderInsert(op, beforeTokens, afterTokens, opIndex, diffContent, dataPrefix, className)
+ );
+}
+
+const OPS: Record = {
+ equal: renderEqual,
+ insert: renderInsert,
+ delete: renderDelete,
+ replace: renderReplace,
+};
+
+/**
+ * Renders a list of operations into HTML content. The result is the combined version of the
+ * before and after tokens with the differences wrapped in tags.
+ * @param beforeTokens The before list of tokens.
+ * @param afterTokens The after list of tokens.
+ * @param operations The list of operations to transform the before tokens into the after tokens.
+ * @param diffContent Diffs the content of opted-in atomic elements recursively.
+ * @param dataPrefix (Optional) The prefix to use in data attributes.
+ * @param className (Optional) The class name to include in the wrapper tag.
+ * @returns The rendering of the list of operations.
+ */
+export function renderOperations(
+ beforeTokens: Token[],
+ afterTokens: Token[],
+ operations: Operation[],
+ diffContent: ContentDiff,
+ dataPrefix?: string | null,
+ className?: string | null,
+): string {
+ return operations.reduce(
+ (rendering, op, index) =>
+ rendering + OPS[op.action](op, beforeTokens, afterTokens, index, diffContent, dataPrefix, className),
+ "",
+ );
+}
diff --git a/src/core/tokens.spec.ts b/src/core/tokens.spec.ts
new file mode 100644
index 0000000..75cef84
--- /dev/null
+++ b/src/core/tokens.spec.ts
@@ -0,0 +1,37 @@
+import { expect } from "chai";
+import { createToken, isTag, isVoidTag, isVoidTagName, isWrappable } from "./tokens";
+
+describe("tokens", () => {
+ describe("isTag", () => {
+ it("names a tag token and rejects text", () => {
+ expect(isTag('
")).to.equal(false);
+ });
+ });
+
+ describe("createToken", () => {
+ it("holds the string and its key", () => {
+ expect(createToken('
')).to.deep.equal({ string: '
', key: "
" });
+ });
+ });
+});
diff --git a/src/core/tokens.ts b/src/core/tokens.ts
new file mode 100644
index 0000000..286499f
--- /dev/null
+++ b/src/core/tokens.ts
@@ -0,0 +1,373 @@
+/**
+ * Tokenizing: splits HTML into words, whitespace, tags and atomic elements, and gives every
+ * token the key it is compared by.
+ */
+import { dataHtmlDiffIdRegExp, getAtomicTagsRegExp, isStartOfAtomicTag } from "./atomicTags";
+
+/** A token holds the text to render and the key it is compared by. */
+export interface Token {
+ string: string;
+ key: string;
+}
+
+/**
+ * Determines if the given token is a tag.
+ * @param token The token in question.
+ * @returns False if the token is not a tag, or the tag name otherwise.
+ */
+export function isTag(token: string): string | false {
+ const match = token.match(/^\s*<([^!>][^>]*)>\s*$/);
+ return !!match && match[1].trim().split(" ")[0];
+}
+
+/**
+ * Checks if a tag is a void tag, written with the XML style '/>'.
+ * @param token The token to check.
+ * @returns True if the token is a void tag, false otherwise.
+ */
+export function isVoidTag(token: string): boolean {
+ return /^\s*<[^>]+\/>\s*$/.test(token);
+}
+
+/**
+ * Checks if a tag name is an HTML void element. Void elements cannot have content and can
+ * skip a closing tag, so an atomic element with a void tag name ends with its opening tag,
+ * with or without the XML style '/>'.
+ * @param tag The tag name to check.
+ * @returns True if the tag name is a void element.
+ */
+export function isVoidTagName(tag: string): boolean {
+ return /^(area|base|br|col|embed|hr|img|input|link|meta|param|source|track|wbr)$/.test(tag);
+}
+
+/**
+ * Checks if a token can be wrapped inside a tag: text, images, void tags and atomic tags
+ * can, other tags cannot.
+ * @param token The token to check.
+ * @returns True if the token can be wrapped inside a tag, false otherwise.
+ */
+export function isWrappable(token: string): boolean {
+ const isImage = /^]/.test(token);
+ return isImage || !isTag(token) || !!isStartOfAtomicTag(token) || isVoidTag(token);
+}
+
+/**
+ * Creates a token that holds a string and key representation. The key is used for diffing
+ * comparisons and the string is used to recompose the document after the diff is complete.
+ * @param currentWord The section of the document to create a token for.
+ * @returns A token object with a string and key property.
+ */
+export function createToken(currentWord: string): Token {
+ return {
+ string: currentWord,
+ key: getKeyForToken(currentWord),
+ };
+}
+
+/**
+ * Creates a key that should be used to match tokens. This is useful, for example, if we want
+ * to consider two open tag tokens as equal, even if they don't have the same attributes. We
+ * use a key instead of overwriting the token because we may want to render the original
+ * string without losing the attributes.
+ * @param token The token to create the key for.
+ * @returns The identifying key that should be used to match before and after tokens.
+ */
+export function getKeyForToken(token: string): string {
+ // If the token is an image element, grab it's src attribute to include in the key.
+ const img = /^$/.exec(token);
+ if (img) {
+ return '';
+ }
+
+ // If the token is an a element, grab it's data attribute to include in the key.
+ // Only when is atomic: if it has been excluded from the atomic tags (as done in
+ // recursive inner diffs), the token is just the opening tag.
+ const a = /^';
+ }
+
+ // If the token is an object element, grab it's data attribute to include in the key.
+ const object = /^';
+ }
+
+ // If it's a video, math or svg element, the entire token should be compared except the
+ // data-uuid.
+ if (/^<(svg|math|video)[\s>]/.test(token)) {
+ const uuid = token.indexOf('data-uuid="');
+ if (uuid !== -1) {
+ const start = token.slice(0, uuid);
+ const end = token.slice(uuid + 44);
+ return start + end;
+ }
+ return token;
+ }
+
+ // If the token is an iframe element, grab it's src attribute to include in it's key.
+ const iframe = /^/.exec(token);
+ if (iframe) {
+ return '';
+ }
+
+ // if the token has data-htmldiff-uuid use it as a key
+ const uuidTag = dataHtmlDiffIdRegExp.exec(token);
+ if (uuidTag) {
+ return uuidTag[2];
+ }
+
+ // If the token is any other element, just grab the tag name.
+ const tagName = /<([^\s>]+)[\s>]/.exec(token);
+ if (tagName) {
+ return "<" + tagName[1].toLowerCase() + ">";
+ }
+
+ // Otherwise, the token is text, collapse the whitespace
+ // (except new lines, see https://matrixreq.atlassian.net/browse/MATRIX-7880)
+ // potentially, this also causing the problems with prettified HTML (with "\n" between the tags),
+ // so it's required to "flatten" html before passing it to the diffing function
+ if (token) {
+ return token.replace(/([^\S\r\n]+| | )/g, " ");
+ }
+ return token;
+}
+
+function isEndOfTag(char: string): boolean {
+ return char === ">";
+}
+
+function isStartOfTag(char: string): boolean {
+ return char === "<";
+}
+
+function isWhitespace(char: string): boolean {
+ return /^\s+$/.test(char);
+}
+
+function isStartOfHtmlComment(word: string): boolean {
+ return /^$/.test(word);
+}
+
+/**
+ * Inspects the last tag in the given string, its slice from the final '<'. A '>' before the
+ * slice's end means text follows e.g. "a > b" in a
')).eql(
- tokenize(['
', '', '
']));
- });
-
-
-
- it('should identify tags with data-htmldiff-id attribute as single token', () => {
- expect(
- cut('
hellogoodbye' +
- 'some stuff' +
- '
')
- ).eql(tokenize(
- [
- '
',
- 'hellogoodbye',
- 'some stuff',
- '
'
- ]
- ));
- });
-
- describe('nested atomic tags wrapping', function(){
- it('should keep a data-htmldiff-id wrapper with nested same-tag children as one token', function(){
- var atomic = '' +
- 'AB' +
- 'Name';
- expect(cut('
' + atomic + '
')).eql(
- tokenize(['
', atomic, '
']));
- });
-
- it('should not close early on the first inner closing tag', function(){
- var atomic = 'ab';
- expect(cut(atomic)).eql(tokenize([atomic]));
- });
-
- it('should not treat a stray ">" in script content as a tag boundary', function(){
- var atomic = '';
- expect(cut('
' + atomic + '
')).eql(
- tokenize(['
', atomic, '
']));
- });
-
- it('should ignore self-closing same-named children when counting depth', function(){
- var atomic = 'xy';
- expect(cut(atomic)).eql(tokenize([atomic]));
- });
-
- it('should ignore self-closing same-named children written with a space ()', function(){
- var atomic = 'xy';
- expect(cut(atomic)).eql(tokenize([atomic]));
- });
-
- it('should key a wrapper by its own data-htmldiff-id, not a nested child one', function(){
- var atomic = '' +
- 'C';
- expect(cut(atomic)[0].key).eql('s1');
- });
-
- it('should not key an unkeyed atomic tag by a nested child data-htmldiff-id', function(){
- var atomic = 'C';
- expect(cut(atomic)[0].key).eql('');
- });
-
- it('should not bump depth on differently-named tags that share a prefix', function(){
- // a tag must not be matched by an atomic tag named "a" appearing as .
- var atomic = 'hi';
- expect(cut(atomic)).eql(tokenize([atomic]));
- });
- });
-
- describe('tags sharing a prefix with atomic tag names', function(){
- it('should not treat as the atomic tag a', function(){
- expect(cut('x tail')).eql(
- tokenize(['', 'x', '', ' ', 'tail']));
- });
-
- it('should not treat as the atomic tag a', function(){
- expect(cut('hi')).eql(
- tokenize(['', 'hi', '']));
- });
- });
-
- describe('self-closing atomic tags', function(){
- it('should end a self-closing data-htmldiff-id tag without swallowing trailing content', function(){
- expect(cut('x old')).eql(
- tokenize(['', 'x', ' ', 'old']));
- });
-
- it('should end a self-closing name-based atomic tag without swallowing trailing content', function(){
- expect(cut('tail')).eql(tokenize(['', 'tail']));
- });
- });
-
- describe('quoted attribute values containing tag delimiters', function(){
- it('should not end a tag on ">" inside a double-quoted attribute value', function(){
- expect(cut('
text
')).eql(
- tokenize(['
', 'text', '
']));
- });
-
- it('should not end a tag on ">" inside a single-quoted attribute value', function(){
- expect(cut("
x
")).eql(
- tokenize(["
", 'x', '
']));
- });
-
- it('should not end a void atomic tag on ">" inside an attribute value', function(){
- expect(cut(' tail')).eql(
- tokenize(['', ' ', 'tail']));
- });
-
- it('should not treat "/>" inside an attribute value as self-closing', function(){
- var atomic = 'c';
- expect(cut(atomic)).eql(tokenize([atomic]));
- });
-
- it('should keep an atomic tag with ">" in an attribute as one token', function(){
- expect(cut('
x
tail')).eql(
- tokenize(['
x
',
- ' ', 'tail']));
- });
-
- it('should not treat apostrophes in atomic text content as quotes', function(){
- expect(cut("
it's ok
tail")).eql(
- tokenize(["
it's ok
", ' ', 'tail']));
- });
-
- it('should not treat apostrophes in comments inside atomic tags as quotes', function(){
- expect(cut(" tail")).eql(
- tokenize(["", ' ', 'tail']));
- });
- });
-
- describe('void atomic tags', function(){
- it('should end a void data-htmldiff-id tag written without a slash', function(){
- expect(cut(' tail')).eql(
- tokenize(['', ' ', 'tail']));
- });
-
- it('should end a void data-htmldiff-id br tag without swallowing trailing content', function(){
- expect(cut(' y')).eql(
- tokenize([' ', 'y']));
- });
- });
- });
-});
diff --git a/test/innerDiff.spec.ts b/test/innerDiff.spec.ts
new file mode 100644
index 0000000..ba05190
--- /dev/null
+++ b/test/innerDiff.spec.ts
@@ -0,0 +1,303 @@
+import { expect } from "chai";
+import diff from "../src/htmldiff";
+
+describe("Recursive inner diff (data-htmldiff-inner-diff)", () => {
+ const cut = diff;
+
+ function tocEntry(href: string, name: string): string {
+ return `
`;
+ }
+
+ describe("when an opted-in element is renamed in place", () => {
+ it("renders inline ins/del inside the single emitted element", () => {
+ const res = cut(tocEntry("#123", "1. Old name"), tocEntry("#123", "1. New name"));
+ expect(res).to.equal(
+ '
`;
+ }
+
+ it("keeps the default atomic tags (without a) inside the recursion", () => {
+ // Embedded content like svg stays atomic by default: a changed svg is replaced
+ // as a whole, not word-diffed.
+ const res = cut(el("", ''), el("", ''));
+ expect(res).to.equal(
+ '
' +
+ '' +
+ '' +
+ '' +
+ '' +
+ "
",
+ );
+ });
+
+ it("replaces the default list with data-htmldiff-inner-diff-atomic-tags", () => {
+ // The override lists only em, so svg is no longer atomic and gets word-diffed.
+ const attrs = ' data-htmldiff-inner-diff-atomic-tags="em"';
+ const res = cut(el(attrs, ''), el(attrs, ''));
+ expect(res).to.equal(
+ `
",
+ );
+ });
+
+ it("treats no tag name as atomic when the override value is empty", () => {
+ const attrs = ' data-htmldiff-inner-diff-atomic-tags=""';
+ const res = cut(el(attrs, ''), el(attrs, ''));
+ expect(res).to.equal(
+ `
');
+ });
+
+ it("recognizes the attribute regardless of its position", () => {
+ const res = cut(
+ '
old
',
+ '
new
',
+ );
+ expect(res).to.equal(
+ '
' +
+ 'old' +
+ 'new
',
+ );
+ });
+ });
+
+ describe("edge cases", () => {
+ it("ignores the attribute on non-atomic elements (no data-htmldiff-id)", () => {
+ // Without data-htmldiff-id the div is not atomic, so this is a plain word diff.
+ const res = cut('
old text
', '
new text
');
+ expect(res).to.equal(
+ '
' + 'old' + 'new text
',
+ );
+ });
+
+ it("diffs text following a self-closing opted-in element normally", () => {
+ const res = cut(
+ 'x old',
+ 'x new',
+ );
+ expect(res).to.equal(
+ 'x ' +
+ 'old' +
+ 'new',
+ );
+ });
+
+ it("marks the content as inserted when the before element was self-closing", () => {
+ // A self-closing element has empty inner content, so the new content is a
+ // pure insertion.
+ const res = cut('', '
new
');
+ expect(res).to.equal('
' + 'new
');
+ });
+
+ it("marks the content as deleted when the after element became self-closing", () => {
+ // The after element has no content anymore, so the deleted content is
+ // rendered right after the self-closing tag.
+ const res = cut('
old
', '');
+ expect(res).to.equal('' + 'old');
+ });
+
+ it('is not confused by ">" inside attribute values', () => {
+ const res = cut(
+ '
old
',
+ '
new
',
+ );
+ expect(res).to.equal(
+ '
' +
+ 'old' +
+ 'new
',
+ );
+ });
+
+ it('is not confused by ">" inside single-quoted attribute values', () => {
+ const res = cut(
+ "
old
",
+ "
new
",
+ );
+ expect(res).to.equal(
+ "
" +
+ 'old' +
+ 'new
',
+ );
+ });
+
+ it("falls back to the after version when a token cannot be split", () => {
+ // An unterminated atomic tag swallows the rest of the input and has no closing
+ // tag to split on; the inner diff falls back instead of producing broken markup.
+ const after = '
new';
+ const res = cut('
old', after);
+ expect(res).to.equal(after);
+ });
+
+ it("renders pure insertions when the before content is empty", () => {
+ const res = cut(tocEntry("#1", ""), tocEntry("#1", "New name"));
+ expect(res).to.equal(
+ '
';
- }
-
- describe('when an opted-in element is renamed in place', function(){
- it('renders inline ins/del inside the single emitted element', function(){
- var res = cut(tocEntry('#123', '1. Old name'), tocEntry('#123', '1. New name'));
- expect(res).to.equal(
- '
');
- });
-
- it('passes className and dataPrefix through to the inner ins/del tags', function(){
- var res = cut(tocEntry('#123', 'Old'), tocEntry('#123', 'New'), 'diff-cls', 'pre');
- expect(res).to.equal(
- '
');
- });
- }); // describe('anchors inside the recursive diff')
-
- describe('atomic tags inside the recursive diff', function(){
- function el(extraAttrs, inner){
- return '
' + inner + '
';
- }
-
- it('keeps the default atomic tags (without a) inside the recursion', function(){
- // Embedded content like svg stays atomic by default: a changed svg is replaced
- // as a whole, not word-diffed.
- var res = cut(
- el('', ''),
- el('', ''));
- expect(res).to.equal(
- '
' +
- '' +
- '' +
- '' +
- '' +
- '
');
- });
-
- it('replaces the default list with data-htmldiff-inner-diff-atomic-tags', function(){
- // The override lists only em, so svg is no longer atomic and gets word-diffed.
- var attrs = ' data-htmldiff-inner-diff-atomic-tags="em"';
- var res = cut(
- el(attrs, ''),
- el(attrs, ''));
- expect(res).to.equal(
- '
' +
- '
');
- });
-
- it('restores atomic anchors (href comparison) when the override lists a', function(){
- var attrs = ' data-htmldiff-inner-diff-atomic-tags="a"';
- var res = cut(
- el(attrs, 'Name'),
- el(attrs, 'Name'));
- expect(res).to.equal(
- '
');
- });
-
- it('treats no tag name as atomic when the override value is empty', function(){
- var attrs = ' data-htmldiff-inner-diff-atomic-tags=""';
- var res = cut(
- el(attrs, ''),
- el(attrs, ''));
- expect(res).to.equal(
- '
';
- }
-
- it('treats a double-quoted "false" value as opted out', function(){
- var res = cut(entry('"false"', 'old'), entry('"false"', 'new'));
- expect(res).to.equal(entry('"false"', 'new'));
- });
-
- it('treats a single-quoted \'false\' value as opted out', function(){
- var res = cut(entry("'false'", 'old'), entry("'false'", 'new'));
- expect(res).to.equal(entry("'false'", 'new'));
- });
-
- it('treats an unquoted false value as opted out', function(){
- var res = cut(entry('false', 'old'), entry('false', 'new'));
- expect(res).to.equal(entry('false', 'new'));
- });
-
- it('treats a bare attribute without a value as opted in', function(){
- var res = cut(
- '
old
',
- '
new
');
- expect(res).to.equal(
- '
' +
- 'old' +
- 'new
');
- });
-
- it('only matches exact attribute name', function(){
- var res = cut(
- '
old
',
- '
new
');
- expect(res).to.equal(
- '
new
');
- });
-
- it('recognizes the attribute regardless of its position', function(){
- var res = cut(
- '
old
',
- '
new
');
- expect(res).to.equal(
- '
' +
- 'old' +
- 'new
');
- });
- }); // describe('opt-in attribute values')
-
- describe('edge cases', function(){
- it('ignores the attribute on non-atomic elements (no data-htmldiff-id)', function(){
- // Without data-htmldiff-id the div is not atomic, so this is a plain word diff.
- var res = cut(
- '
old text
',
- '
new text
');
- expect(res).to.equal(
- '
' +
- 'old' +
- 'new text
');
- });
-
- it('diffs text following a self-closing opted-in element normally', function(){
- var res = cut(
- 'x old',
- 'x new');
- expect(res).to.equal(
- 'x ' +
- 'old' +
- 'new');
- });
-
- it('marks the content as inserted when the before element was self-closing', function(){
- // A self-closing element has empty inner content, so the new content is a
- // pure insertion.
- var res = cut(
- '',
- '
new
');
- expect(res).to.equal(
- '
' +
- 'new
');
- });
-
- it('marks the content as deleted when the after element became self-closing', function(){
- // The after element has no content anymore, so the deleted content is
- // rendered right after the self-closing tag.
- var res = cut(
- '
old
',
- '');
- expect(res).to.equal(
- '' +
- 'old');
- });
-
- it('is not confused by ">" inside attribute values', function(){
- var res = cut(
- '
' +
- 'old
',
- '
' +
- 'new
');
- expect(res).to.equal(
- '
' +
- 'old' +
- 'new
');
- });
-
- it('is not confused by ">" inside single-quoted attribute values', function(){
- var res = cut(
- "
old
",
- "
new
");
- expect(res).to.equal(
- "
" +
- 'old' +
- 'new
');
- });
-
- it('falls back to the after version when a token cannot be split', function(){
- // An unterminated atomic tag swallows the rest of the input and has no closing
- // tag to split on; the inner diff falls back instead of producing broken markup.
- var after = '
new';
- var res = cut(
- '
old',
- after);
- expect(res).to.equal(after);
- });
-
- it('renders pure insertions when the before content is empty', function(){
- var res = cut(tocEntry('#1', ''), tocEntry('#1', 'New name'));
- expect(res).to.equal(
- '
');
- });
-
- it('do not diffs inner html for moved opted-in elements', function(){
- function entry(id, name){
- return '
' +
- name + '
';
- }
- var res = cut(entry('a', 'First') + entry('b', 'Second'),
- entry('b', 'Second') + entry('a', 'First'));
- expect(res).to.equal(
- '' + entry('b', 'Second') + '' +
- entry('a', 'First') +
- '' + entry('b', 'Second') + '');
- });
- }); // describe('edge cases')
-
- describe('when the element did not opt in', function(){
- it('should render the after version as is when keys match but content differs', function(){
- var before = '';
- var after = '';
- expect(cut(before, after)).to.equal(after);
- });
- }); // describe('when the element did not opt in')
-
- describe('when key and content are both equal', function(){
- it('should render the element unchanged', function(){
- var before = 'x ' + tocEntry('#123', '1. Same name') + ' y';
- var after = 'x ' + tocEntry('#123', '1. Same name') + ' z';
- expect(cut(before, after)).to.equal(
- 'x ' + tocEntry('#123', '1. Same name') + ' ' +
- 'yz');
- });
- }); // describe('when key and content are both equal')
-
- describe('when opted-in elements are nested', function(){
- function nest(depth, content){
- var html = content;
- for (var i = depth; i >= 1; i--){
- html = '
');
- });
-
- it('should not identify partial tags', function(){
- var before = tokenize(['test', '', 'non-bold']);
- var after = tokenize(['test!', '', 'non-bold', '', 'bold']);
- res = cut(before, after);
-
- expect(res).to.equal('test' +
- 'test!non-bold' +
- 'bold');
- });
-
- describe('When there is a change at the beginning, in a
', function(){
- beforeEach(function(){
- var before = tokenize(['
', 'this', ' ', 'is', ' ', 'awesome', '
']);
- var after = tokenize(['
', 'I', ' ', 'is', ' ', 'awesome', '
']);
- res = cut(before, after);
- });
-
- it('should keep the change inside the
', function(){
- expect(res).to.equal('
this' +
- 'I is awesome
');
- });
- });
- });
-
- describe('empty tokens', function(){
- it('should not be wrapped', function(){
- var before = tokenize(['text']);
- var after = tokenize(['text', ' ']);
-
- res = cut(before, after);
-
- expect(res).to.equal('text');
- });
- });
-
- describe('tags with attributes', function(){
- it('should treat attribute changes as equal and output the after tag', function(){
- var before = tokenize(['
', 'this', ' ', 'is', ' ', 'awesome', '
']);
- var after = tokenize(['
', 'this', ' ', 'is', ' ',
- 'awesome', '
']);
-
- res = cut(before, after);
-
- expect(res).to.equal('
this is awesome
');
- });
-
- it('should show changes within tags with different attributes', function(){
- var before = tokenize(['
', 'this', ' ', 'is', ' ', 'awesome', '
']);
- var after = tokenize(['
', 'that', ' ', 'is', ' ',
- 'awesome', '
']);
-
- res = cut(before, after);
-
- expect(res).to.equal('
' +
- 'this' +
- 'that is awesome
');
- });
- });
-
- describe('wrappable tags', function(){
- it('should wrap void tags', function(){
- var before = tokenize(['old', ' ', 'text']);
- var after = tokenize(['new', ' ', ' ', 'text']);
-
- res = cut(before, after);
-
- expect(res).to.equal('old' +
- 'new text');
- });
-
- it('should wrap atomic tags independently', function(){
- var before = tokenize(['old', '', ' ', 'text']);
- var after = tokenize(['new', ' ', 'text']);
-
- res = cut(before, after);
-
- expect(res).to.equal(
- 'old' +
- '' +
- 'new text');
- });
- });
-});
diff --git a/test/tables.spec.ts b/test/tables.spec.ts
new file mode 100644
index 0000000..d082b27
--- /dev/null
+++ b/test/tables.spec.ts
@@ -0,0 +1,992 @@
+import { expect } from "chai";
+import diff from "../src/htmldiff";
+
+describe("tables", () => {
+
+ describe("rows", () => {
+ it("marks whole added row when another cell is edited", () => {
+ const res = diff('
A B C D E
','
A B C D E
');
+ expect(res).to.equal(
+ '
A B ' +
+ 'C' +
+ 'C D E
');
+ });
+
+ it("leaves row without cells empty", () => {
+ const res = diff(
+ '
');
+ });
+
+ it("marks row inserted in the middle", () => {
+ const res = diff('
text
','
text
');
+ expect(res).to.equal(
+ '
' +
+ '
text
' +
+ 'text
');
+ });
+
+ it("marks moved row as deleted and added", () => {
+ const res = diff(
+ '
' +
+ '
a
b
' +
+ '
c
d
' +
+ '
',
+ '
' +
+ '
a
b
' +
+ '
x
y
' +
+ '
');
+ expect(res).to.equal(
+ '
' +
+ '
a
b
' +
+ '
c
d
' +
+ '
x
y
' +
+ '
');
+ });
+
+ it("keeps filling an empty cell as cell edit", () => {
+ const res = diff('
text here
','
text HERE
');
+ expect(res).to.equal(
+ '
text ' +
+ 'here' +
+ 'HERE
');
+ });
+
+ it("ignores line number column when matching rows", () => {
+ const res = diff('
buy milk
','
buy oat milk
');
+ expect(res).to.equal(
+ '
buy ' +
+ 'oat milk
');
+ });
+
+ it("fills an empty row in place as a cell edit", () => {
+ const res = diff('
a
b
',
+ '
a
b
x
y
z
');
+ expect(res).to.equal(
+ '
' +
+ '
a
b
' +
+ '
x
' +
+ '
y
z
');
+ });
+
+ it("keeps the line number column when every kept row renumbers", () => {
+ const res = diff('
1
a
2
b
3
c
4
d
5
e
',
+ '
1
c
2
d
3
e
4
a
5
b
');
+ expect(res).to.equal(
+ '
' +
+ '
1
a
' +
+ '
2
b
' +
+ '
31
c
' +
+ '
42
d
' +
+ '
53
e
' +
+ '
4
a
' +
+ '
5
b
');
+ });
+
+ it("marks fully changed row as deleted and added", () => {
+ const res = diff(
+ '
' +
+ '
TC-1
a
' +
+ '
TC-2
b
' +
+ '
',
+ '
' +
+ '
TC-1
a
' +
+ '
TC-9
b
' +
+ '
');
+ expect(res).to.equal('
TC-1
a
TC-2
b
TC-9
b
');
+ });
+ });
+
+ describe("rows of generated tables", () => {
+ it("keeps rows of different items apart", () => {
+ const res = diff('
task
','
task
');
+ expect(res).to.equal(
+ '
' +
+ 'task' +
+ 'task
');
+ });
+
+ it("replaces the rows of an item when none of them keep their keys", () => {
+ const res = diff('
SPEC-1
not covered
',
+ '
SPEC-1
TC-3
XTC-22
other
' +
+ '
TC-2
XTC-23
other
');
+ expect(res).to.equal('
SPEC-1
not covered
SPEC-1
TC-3
XTC-22
other
TC-2
XTC-23
other
');
+ });
+
+ it("replaces the missing-trace row when the item gains its first trace", () => {
+ const res = diff('
SPEC-7
Missing trace to TC
',
+ '
SPEC-7
TC-5
');
+ expect(res).to.equal('
SPEC-7
Missing trace to TC
SPEC-7
TC-5
');
+ });
+
+ it("diffs the links cell when a linked item is added", () => {
+ const res = diff(
+ '
' +
+ '
REQ-1
REQ-2
' +
+ '
',
+ '
' +
+ '
REQ-1
REQ-2REQ-3
' +
+ '
');
+ expect(res).to.equal('
REQ-1
REQ-2REQ-3
');
+ });
+
+ it("diffs the row of an item whose value changed and whose controls gained a ref", () => {
+ const res = diff('
RISK-9 Grab Bar
Inadequate grip
2
SPEC-18 IFU
',
+ '
RISK-9 Grab Bar
Inadequate grip
3
SPEC-8 RF SPEC-18 IFU
');
+ expect(res).to.equal('
RISK-9 Grab Bar
Inadequate grip
23
SPEC-8 RF SPEC-18 IFU
');
+ });
+
+
+ // a producer may give the cells naming a row their own identity: then those alone pair the rows
+ describe("keyed rows", () => {
+ const section = (rows: string): string => `
${rows}
`;
+
+ it("diffs the cells of an executed test when its keys are unchanged", () => {
+ const key = (itemRef: string): string => `
');
+ });
+ });
+
+ describe("merged cells", () => {
+ it("diffs cells by position when the shape is unchanged", () => {
+ const res = diff(
+ '
' +
+ '
Category
Zone 1
Total
' +
+ '
Before
Number of
3
4
' +
+ '
Risk percentage
75%
100%
' +
+ '
',
+ '
' +
+ '
Category
Zone 1
Total
' +
+ '
Before
Number of
3
5
' +
+ '
Risk percentage
60%
100%
' +
+ '
');
+ expect(res).to.equal(
+ '
' +
+ '
Category
Zone 1
Total
' +
+ '
Before
Number of
3
' +
+ '4' +
+ '5
' +
+ '
Risk percentage
' +
+ '75%' +
+ '60%
100%
' +
+ '
');
+ });
+
+ it("marks added body row under a merged header", () => {
+ const res = diff(
+ '
' +
+ '
Risks
Zone
' +
+ '
R-1
a
1
' +
+ '
',
+ '
' +
+ '
Risks
Zone
' +
+ '
R-1
a
1
' +
+ '
R-2
b
2
' +
+ '
');
+ expect(res).to.equal(
+ '
' +
+ '
Risks
Zone
' +
+ '
R-1
a
1
' +
+ '
R-2
b
2
' +
+ '
');
+ });
+
+ // merged cells are layout in a text: a table whose merged cells changed is another table
+ it("wraps a rich text table whole when a cell was split", () => {
+ const oldTable = '