Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -2,3 +2,4 @@ node_modules
htmldiff-cli.js
js/
.DS_Store
htmldiff-cli.d.ts
4 changes: 2 additions & 2 deletions package-lock.json

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

2 changes: 1 addition & 1 deletion package.json
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@
"name": "@matrixreq/htmldiff",
"description": "Diff and markup HTML with <ins> and <del> tags",
"companyname": "Matrix Requirements",
"version": "2.0.0",
"version": "2.0.1",

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

we never published it, so it can stay 2.0

"keywords": [
"diff",
"html",
Expand Down
41 changes: 23 additions & 18 deletions src/core/tokens.ts
Original file line number Diff line number Diff line change
Expand Up @@ -154,29 +154,27 @@ function isEndOfHtmlComment(word: string): boolean {
}

/**
* Inspects the last tag in the given string, its slice from the final '<'. A '>' before the
* slice's end means text follows e.g. "a > b" in a <script>, not a clean tag, so returns false.
* @param word The characters read so far.
* Inspects the last tag read, the text from its final '<'. A '>' before the text's end means
* text follows e.g. "a > b" in a <script>, not a clean tag, so returns false.
* @param tagText The characters from the last '<' read up to the current one.
* @param tag The tag name.
* @returns True if word ends with an opening (non-self-closing) tag for the given tag name.
* @returns True if the text is an opening (non-self-closing) tag for the given tag name.
*/
function isOpeningTagOf(word: string, tag: string): boolean {
const tagText = word.substring(word.lastIndexOf("<"));
function isOpeningTagOf(tagText: string, tag: string): boolean {
if (tagText.indexOf(">") !== tagText.length - 1) {
return false;
}
return new RegExp("^<" + tag + "(\\s|>)").test(tagText) && !/\/>$/.test(tagText);
}

/**
* Inspects the last tag in the given string, its slice from the final '<'. A '>' before the
* slice's end means text follows e.g. "a > b" in a <script>, not a clean tag, so returns false.
* @param word The characters read so far.
* Inspects the last tag read, the text from its final '<'. A '>' before the text's end means
* text follows e.g. "a > b" in a <script>, not a clean tag, so returns false.
* @param tagText The characters from the last '<' read up to the current one.
* @param tag The tag name.
* @returns True if word ends with a closing tag for the given tag name.
* @returns True if the text is a closing tag for the given tag name.
*/
function isClosingTagOf(word: string, tag: string): boolean {
const tagText = word.substring(word.lastIndexOf("<"));
function isClosingTagOf(tagText: string, tag: string): boolean {
if (tagText.indexOf(">") !== tagText.length - 1) {
return false;
}
Expand All @@ -203,6 +201,10 @@ export function htmlToTokens(html: string): Token[] {
// mode is limited to that region: quotes in the element's content (text
// apostrophes, comments, nested tags) must not affect how the token ends.
let inAtomicOpeningTag = false;
// Where the last '<' read stands in the html: the tag being read inside an atomic
// element is inspected there, instead of searching back through the whole token on
// every '>', which made a large atomic element (a merged table) cost its size per tag.
let lastTagStart = -1;
const words: Token[] = [];

/**
Expand All @@ -229,6 +231,9 @@ export function htmlToTokens(html: string): Token[] {

for (let i = 0; i < html.length; i++) {
const char = html[i];
if (isStartOfTag(char)) {
lastTagStart = i;
}
switch (mode) {
case "tag": {
// Quote handling must come before the atomic tag detection:
Expand Down Expand Up @@ -289,7 +294,10 @@ export function htmlToTokens(html: string): Token[] {
// end the atomic token only when it returns to 0.
if (isEndOfTag(char)) {
inAtomicOpeningTag = false;
if (isClosingTagOf(currentWord, currentAtomicTag)) {
// the current word holds the html from the element's '<' up to here, so
// its last tag is the html from the last '<' up to here
const tagText = html.slice(lastTagStart, i + 1);
if (isClosingTagOf(tagText, currentAtomicTag)) {
currentAtomicTagDepth--;
if (currentAtomicTagDepth <= 0) {
words.push(createToken(currentWord));
Expand All @@ -298,18 +306,15 @@ export function htmlToTokens(html: string): Token[] {
currentAtomicTagDepth = 0;
mode = "char";
}
} else if (
currentAtomicTagDepth === 0 &&
(isVoidTagName(currentAtomicTag) || /\/>$/.test(currentWord))
) {
} else if (currentAtomicTagDepth === 0 && (isVoidTagName(currentAtomicTag) || html[i - 1] === "/")) {
// At depth 0 this '>' can only end the atomic tag's own opening
// tag. When the element is void or self-closing it has no
// content: end the token so trailing content tokenizes normally.
words.push(createToken(currentWord));
currentWord = "";
currentAtomicTag = "";
mode = "char";
} else if (isOpeningTagOf(currentWord, currentAtomicTag)) {
} else if (isOpeningTagOf(tagText, currentAtomicTag)) {
currentAtomicTagDepth++;
}
}
Expand Down
5 changes: 5 additions & 0 deletions src/htmldiff.ts
Original file line number Diff line number Diff line change
Expand Up @@ -36,6 +36,11 @@ import { TableRedlining } from "./tables";
* @returns The combined HTML content with differences wrapped in <ins> and <del> tags.
*/
function diff(before: string, after: string, className?: string | null, dataPrefix?: string | null, atomicTags?: string | null): string {
// nothing changed, nothing to diff
if (before === after) {
return before;
}

// Enable user provided atomic tag list.
setAtomicTagsRegExp(atomicTags ? buildAtomicTagsRegExp(atomicTags) : defaultAtomicTagsRegExp);

Expand Down
95 changes: 77 additions & 18 deletions src/tables/RowAligner.ts
Original file line number Diff line number Diff line change
Expand Up @@ -4,9 +4,23 @@
*/
import { last, range } from "./helpers";
import { Alignment, SameAlignment, SequenceAligner } from "./SequenceAligner";
import { cellSimilarity } from "./similarity";
import { signatureWords, wordSimilarity } from "./similarity";
import { TableVersion } from "./TableVersion";

/** What a row says on the compared columns, read once: every old row is compared with every new one. */
interface RowContent {
/** Per compared column. */
signatures: string[];
/** Per compared column, the distinct words of the signature. */
words: string[][];
/** Nothing on any compared column. */
blank: boolean;
/** The producer's keys of the row, when the table is keyed. */
keys: string[];
/** Rows with the same key are exactly the same row; an integer, so the exact pass only compares numbers. */
exactKey: number;
}

/** Which row of the old table is which row of the new one. */
export class RowAligner {
private readonly oldVersion: TableVersion;
Expand Down Expand Up @@ -37,44 +51,89 @@ export class RowAligner {
* @returns The row alignments.
*/
align(): Alignment[] {
const oldVersion = this.oldVersion;
const newVersion = this.newVersion;
const comparedColumns = this.comparedColumns;
const keyed = oldVersion.hasRowKeys && newVersion.hasRowKeys;
const sameKeys = (oldKeys: string[], newKeys: string[]): boolean =>
oldKeys.length === newKeys.length && oldKeys.every((key, index) => key === newKeys[index]);
const keyed = this.oldVersion.hasRowKeys && this.newVersion.hasRowKeys;
const exactKeys = new Map<string, number>();
const oldRows = RowAligner.readRows(
this.oldVersion,
this.comparedColumns.map((column) => column.oldIndex),
keyed,
exactKeys,
);
const newRows = RowAligner.readRows(
this.newVersion,
this.comparedColumns.map((column) => column.newIndex),
keyed,
exactKeys,
);

// the share of what the cells said that is still there, averaged over the columns that
// say anything; the same keys are the same row, other keys another row
const similarity = (oldIndex: number, newIndex: number): number => {
const oldRow = oldRows[oldIndex];
const newRow = newRows[newIndex];
if (keyed) {
return sameKeys(oldVersion.rowKeys(oldIndex), newVersion.rowKeys(newIndex)) ? 1 : 0;
return oldRow.exactKey === newRow.exactKey ? 1 : 0;
}
let compared = 0;
let kept = 0;
comparedColumns.forEach((column) => {
const oldSignature = oldVersion.signatureAt(oldIndex, column.oldIndex);
const newSignature = newVersion.signatureAt(newIndex, column.newIndex);
oldRow.signatures.forEach((oldSignature, column) => {
const newSignature = newRow.signatures[column];
if (oldSignature === "" && newSignature === "") {
return;
}
compared++;
kept += cellSimilarity(oldSignature, newSignature);
kept += oldSignature === newSignature ? 1 : wordSimilarity(oldRow.words[column], newRow.words[column]);
});
return compared === 0 ? NaN : kept / compared;
};
// similarity is 1 exactly when every column that says anything keeps the same words, or
// the keys are the same: two blank rows have no similarity at all
const isExact = (oldIndex: number, newIndex: number): boolean => {
const oldRow = oldRows[oldIndex];
const newRow = newRows[newIndex];
return oldRow.exactKey === newRow.exactKey && (keyed || !(oldRow.blank && newRow.blank));
};

return new SequenceAligner({
oldCount: oldVersion.rows.length,
newCount: newVersion.rows.length,
oldCount: oldRows.length,
newCount: newRows.length,
similarity,
isExact,
byPosition: {
isOldBlank: (oldIndex) => comparedColumns.every((column) => oldVersion.signatureAt(oldIndex, column.oldIndex) === ""),
isNewBlank: (newIndex) => comparedColumns.every((column) => newVersion.signatureAt(newIndex, column.newIndex) === ""),
hasOldIdentity: (oldIndex) => keyed && oldVersion.rowKeys(oldIndex).length > 0,
hasNewIdentity: (newIndex) => keyed && newVersion.rowKeys(newIndex).length > 0,
isOldBlank: (oldIndex) => oldRows[oldIndex].blank,
isNewBlank: (newIndex) => newRows[newIndex].blank,
hasOldIdentity: (oldIndex) => keyed && oldRows[oldIndex].keys.length > 0,
hasNewIdentity: (newIndex) => keyed && newRows[newIndex].keys.length > 0,
},
}).align();
}

/**
* Reads what every row says on the compared columns.
* @param version The version.
* @param columnIndexes The compared columns of this version.
* @param keyed Whether the keys name the rows.
* @param exactKeys The keys seen so far, shared by both versions.
* @returns The rows.
*/
private static readRows(version: TableVersion, columnIndexes: number[], keyed: boolean, exactKeys: Map<string, number>): RowContent[] {
const keysByRow = keyed ? version.rowKeysByRow() : [];
return range(0, version.rows.length).map((rowIndex) => {
const signatures = columnIndexes.map((columnIndex) => version.signatureAt(rowIndex, columnIndex));
const words = signatures.map((signature) => signatureWords(signature));
const keys = keyed ? keysByRow[rowIndex] : [];
// keyed: the keys in order. else: the set of words of every column, columns apart
// (words never contain whitespace or '|')
const exactKey = keyed ? JSON.stringify(keys) : words.map((cellWords) => cellWords.slice().sort().join(" ")).join("|");
let interned = exactKeys.get(exactKey);
if (interned === undefined) {
interned = exactKeys.size;
exactKeys.set(exactKey, interned);
}
return { signatures, words, blank: signatures.every((signature) => signature === ""), keys, exactKey: interned };
});
}

/**
* Where rows were replaced, each deleted row is followed by the added row in its place, so
* the reader sees old and new together. A group (a row and the rows its merged cell spans)
Expand Down
11 changes: 11 additions & 0 deletions src/tables/SequenceAligner.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -94,6 +94,17 @@ describe("SequenceAligner", () => {
]);
});

it("anchors on isExact when the sequence gives one", () => {
const seq = sequence(["a", "b"], ["x", "y"]);
// similarity never reaches 1 on its own, isExact alone decides the exact pass
seq.similarity = () => 0;
seq.isExact = (o, n) => o === n;
expect(new SequenceAligner(seq).align()).to.deep.equal([
{ kind: "same", oldIndex: 0, newIndex: 0 },
{ kind: "same", oldIndex: 1, newIndex: 1 },
]);
});

it("pairs similar entries inside the gaps", () => {
const seq = sequence(["a", "x", "c"], ["a", "y", "c"]);
seq.similarity = (o, n) => (o === n ? (o === 1 ? 0.6 : 1) : 0);
Expand Down
23 changes: 15 additions & 8 deletions src/tables/SequenceAligner.ts
Original file line number Diff line number Diff line change
Expand Up @@ -47,6 +47,11 @@ export interface Sequence {
newCount: number;
/** 0..1, NaN when both entries are blank and carry no identity. */
similarity: (oldIndex: number, newIndex: number) => number;
/**
* (Optional) Whether two entries are exactly the same, when the sequence can tell that
* cheaper than through similarity: must agree with similarity === 1.
*/
isExact?: (oldIndex: number, newIndex: number) => boolean;
byPosition: PositionalPairing;
}

Expand All @@ -69,7 +74,7 @@ export class SequenceAligner {
*/
align(): Alignment[] {
const sequence = this.sequence;
const exact = (oldIndex: number, newIndex: number): boolean => sequence.similarity(oldIndex, newIndex) === 1;
const exact = sequence.isExact ?? ((oldIndex: number, newIndex: number): boolean => sequence.similarity(oldIndex, newIndex) === 1);
// half the cells (rows) or values (columns) in common is enough to be the same entry, edited
const similar = (oldIndex: number, newIndex: number): boolean => sequence.similarity(oldIndex, newIndex) >= 0.5;
const whole: UnmatchedRange = { oldStart: 0, oldEnd: sequence.oldCount, newStart: 0, newEnd: sequence.newCount };
Expand All @@ -89,13 +94,13 @@ export class SequenceAligner {
*/
private fillUnmatchedRanges(pairs: IndexPair[], pairUnmatched: (unmatched: UnmatchedRange) => IndexPair[]): IndexPair[] {
const sequence = this.sequence;
let result: IndexPair[] = [];
const result: IndexPair[] = [];
let oldStart = 0;
let newStart = 0;

pairs.concat([{ oldIndex: sequence.oldCount, newIndex: sequence.newCount }]).forEach((pair) => {
if (pair.oldIndex > oldStart && pair.newIndex > newStart) {
result = result.concat(pairUnmatched({ oldStart, oldEnd: pair.oldIndex, newStart, newEnd: pair.newIndex }));
pairUnmatched({ oldStart, oldEnd: pair.oldIndex, newStart, newEnd: pair.newIndex }).forEach((unmatchedPair) => result.push(unmatchedPair));
}
if (pair.oldIndex < sequence.oldCount) {
result.push(pair);
Expand Down Expand Up @@ -171,13 +176,15 @@ export class SequenceAligner {
const newStart = unmatched.newStart;
const oldLength = unmatched.oldEnd - oldStart;
const newLength = unmatched.newEnd - newStart;
const lengths = range(0, oldLength + 1).map(() => range(0, newLength + 1).map(() => 0));
// one flat table, lengths[i][j] at i * width + j, the last row and column stay 0
const width = newLength + 1;
const lengths = new Int32Array((oldLength + 1) * width);

for (let i = oldLength - 1; i >= 0; i--) {
for (let j = newLength - 1; j >= 0; j--) {
lengths[i][j] = matches(oldStart + i, newStart + j)
? lengths[i + 1][j + 1] + 1
: Math.max(lengths[i + 1][j], lengths[i][j + 1]);
lengths[i * width + j] = matches(oldStart + i, newStart + j)
? lengths[(i + 1) * width + j + 1] + 1
: Math.max(lengths[(i + 1) * width + j], lengths[i * width + j + 1]);
}
}

Expand All @@ -191,7 +198,7 @@ export class SequenceAligner {
newOffset++;
continue;
}
if (lengths[oldOffset + 1][newOffset] >= lengths[oldOffset][newOffset + 1]) {
if (lengths[(oldOffset + 1) * width + newOffset] >= lengths[oldOffset * width + newOffset + 1]) {
oldOffset++;
continue;
}
Expand Down
17 changes: 11 additions & 6 deletions src/tables/TableMerger.ts
Original file line number Diff line number Diff line change
Expand Up @@ -206,12 +206,17 @@ export class TableMerger {
mergedCells.moveOwnersToKeptRows(columns, rows);

const keptRows = rows.filter((row): row is SameAlignment => row.kind === "same");
// the last kept row of the group a row belongs to, by the merged cells of either version
// the groups of every kept row, by the merged cells of either version, read once: every
// changed row looks for its group among all kept rows
const keptGroupSignatures = keptRows.map((row) => newVersion.groupSignatures(row.newIndex).concat(oldVersion.groupSignatures(row.oldIndex)));
// the last kept row of the group a row belongs to
const lastKeptRowOfGroup = (signatures: string[]): Row | null => {
if (signatures.length === 0) {
return null;
}
let found: Row | null = null;
keptRows.forEach((row) => {
const rowSignatures = newVersion.groupSignatures(row.newIndex).concat(oldVersion.groupSignatures(row.oldIndex));
if (rowSignatures.some((signature) => signatures.indexOf(signature) !== -1)) {
keptRows.forEach((row, index) => {
if (keptGroupSignatures[index].some((signature) => signatures.indexOf(signature) !== -1)) {
found = newVersion.rows[row.newIndex];
}
});
Expand All @@ -228,12 +233,12 @@ export class TableMerger {

const placeDeletedRow = (tr: Row, oldIndex: number, following: Alignment[]): void => {
const group = oldVersion.groupSignatures(oldIndex);
const next = following.filter(
const next = following.find(
(row): row is Exclude<Alignment, { kind: "deleted" }> =>
row.kind !== "deleted" &&
(!newVersion.isContinuationRow(row.newIndex) ||
newVersion.groupSignatures(row.newIndex).some((signature) => group.indexOf(signature) !== -1)),
)[0];
);
if (next) {
newVersion.rows[next.newIndex].insertBefore(tr);
return;
Expand Down
Loading
Loading