diff --git a/.gitignore b/.gitignore index ffc653a..6dec064 100644 --- a/.gitignore +++ b/.gitignore @@ -2,5 +2,6 @@ dist .DS_Store *.tsbuildinfo +*.tgz # Ignore generated worker files src/lib/stream-list-diff/server/worker/node-worker.cjs \ No newline at end of file diff --git a/README.md b/README.md index 00a687b..6af3ca9 100644 --- a/README.md +++ b/README.md @@ -11,7 +11,7 @@ # WHAT IS IT? -**Superdiff** provides a rich and readable diff for **arrays**, **objects**, **texts** and **coordinates**. It supports **stream** and file inputs for handling large datasets efficiently, is battle-tested, has zero dependencies, and offers a **top-tier performance**. +**Superdiff** provides a rich and readable diff for **arrays**, **objects**, **code**, **text** and **coordinates**. It supports **stream** and file inputs for handling large datasets efficiently, is battle-tested, has zero dependencies, and offers **top-tier performance**. ℹ️ The documentation is also available on our [website](https://superdiff.gitbook.io/donedeal0-superdiff)! @@ -19,12 +19,13 @@ ## FEATURES -**Superdiff** exports 5 functions: +**Superdiff** exports 6 functions: - [getObjectDiff](#getobjectdiff) - recursively diff nested objects - [getListDiff](#getlistdiff) - detect additions, deletions, updates and moves in arrays - [streamListDiff](#streamlistdiff) - diff large lists incrementally via streams - [getTextDiff](#gettextdiff) - diff text by character, word or sentence +- [getCodeDiff](#getcodediff) - diff code line by line, then token by token - [getGeoDiff](#getgeodiff) - detect coordinate changes, distance and direction
@@ -62,6 +63,14 @@
+![superdiff-get-code-diff-demo](https://raw.githubusercontent.com/DoneDeal0/superdiff/main/assets/get-code-diff-demo.png) + +

+Track code changes with getCodeDiff +

+ +
+
## ⚔ COMPETITORS @@ -72,6 +81,7 @@ | List diff | ✅ | ❌ | ⚠️ | ❌ | ⚠️ | | Text diff | ✅ | ❌ | ✅ | ✅ | ❌ | | Coordinates diff | ✅ | ❌ | ❌ | ❌ | ❌ | +| Code diff | ✅ | ❌ | ❌ | ✅ | ❌ | | Streaming for huge datasets | ✅ | ❌ | ❌ | ❌ | ❌ | | Move detection | ✅ | ❌ | ❌ | ❌ | ❌ | | Output refinement | ✅ | ❌ | ❌ | ❌ | ❌ | @@ -104,13 +114,23 @@ Method: Warm up runs, then each script is executed 20 times, and we keep the med | Scenario | superdiff | diff | | ----------------------- | ------------ | ---------- | -| 10k words | **1.38 ms** | 3.86 ms | -| 100k words | **21.68 ms** | 45.93 ms | -| 10k sentences | **2.30 ms** | 5.61 ms | -| 100k sentences | **21.95 ms** | 62.03 ms | +| 10k words | **1.32 ms** | 3.86 ms | +| 100k words | **12.09 ms** | 45.64 ms | +| 10k sentences | **1.58 ms** | 4.53 ms | +| 100k sentences | **22.31 ms** | 55.60 ms | (Superdiff uses its `normal` accuracy settings to match diff's behavior) +### Code diff + +| Scenario | superdiff | diff | +| ---------- | ------------- | ---------- | +| 1k lines | **0.36 ms** | 1.14 ms | +| 10k lines | **9.55 ms** | 67.21 ms | +| 100k lines | **911.80 ms** | 6995.84 ms | + +(diff has no line + token API, so both passes are run; its line pass alone is 66 ms at 10k lines) + > 👉 Despite providing a full structural diff with a richer output, **Superdiff consistently matches or outperforms the fastest alternatives tested**. It also scales linearly, even with deeply nested data.
@@ -752,6 +772,129 @@ getTextDiff(
+### getCodeDiff + +```js +import { getCodeDiff } from "@donedeal0/superdiff"; +``` + +Compares two codes and returns a structured diff: lines first, then the tokens of each changed line. Whitespace is kept, so indentation changes are reported too. + +> ℹ️ To diff files, read them first: `getCodeDiff(await previousFile.text(), await currentFile.text())`. + +#### FORMAT + +**Input** + +```ts + previousCode: string | null | undefined, + currentCode: string | null | undefined, +``` + +**Output** + +```ts +type CodeDiff = { + type: "code"; + status: "added" | "deleted" | "equal" | "updated"; + diff: { + value: string; + previousValue?: string; + line: number | null; + previousLine: number | null; + status: "added" | "deleted" | "equal" | "updated"; + diff?: { + value: string; + previousValue?: string; + status: "added" | "deleted" | "equal" | "updated"; + }[]; + }[]; +}; +``` + +Lines start at 1. `line` is `null` if the line was deleted, `previousLine` is `null` if it was added. `diff` holds the token changes of an `updated` line. + +#### USAGE + +**Input** + +```diff +getCodeDiff( +- "const MAX = 280;\n\nfunction preview(post) {\n return post.body;\n}", ++ "const MAX_LENGTH = 320;\n\nfunction preview(post) {\n const text = post.body;\n return text.slice(0, MAX_LENGTH);\n}" +); +``` + +**Output** + +```diff +{ + type: "code", ++ status: "updated", + diff: [ + { ++ value: "const MAX_LENGTH = 320;", ++ previousValue: "const MAX = 280;", + line: 1, + previousLine: 1, ++ status: "updated", + diff: [ + { value: "const", status: "equal" }, ++ { value: " MAX_LENGTH", previousValue: " MAX", status: "updated" }, + { value: " =", status: "equal" }, ++ { value: " 320", previousValue: " 280", status: "updated" }, + { value: ";", status: "equal" }, + ], + }, + { + value: "", + previousValue: "", + line: 2, + previousLine: 2, + status: "equal", + }, + { + value: "function preview(post) {", + previousValue: "function preview(post) {", + line: 3, + previousLine: 3, + status: "equal", + }, + { ++ value: " const text = post.body;", ++ previousValue: " return post.body;", + line: 4, + previousLine: 4, ++ status: "updated", + diff: [ ++ { value: " const", previousValue: " return", status: "updated" }, ++ { value: " text", status: "added" }, ++ { value: " =", status: "added" }, + { value: " post", status: "equal" }, + { value: ".", status: "equal" }, + { value: "body", status: "equal" }, + { value: ";", status: "equal" }, + ], + }, + { ++ value: " return text.slice(0, MAX_LENGTH);", + line: 5, + previousLine: null, ++ status: "added", + }, + { + value: "}", + previousValue: "}", + line: 6, + previousLine: 5, + status: "equal", + }, + ], +} +``` + +
+ ### getGeoDiff ```js diff --git a/assets/get-code-diff-demo.png b/assets/get-code-diff-demo.png new file mode 100644 index 0000000..e1dd532 Binary files /dev/null and b/assets/get-code-diff-demo.png differ diff --git a/benchmark/code.ts b/benchmark/code.ts new file mode 100644 index 0000000..849fe8f --- /dev/null +++ b/benchmark/code.ts @@ -0,0 +1,75 @@ +import { diffLines, diffWordsWithSpace } from "diff"; +import { getCodeDiff } from "../src"; +import { bench } from "./utils"; + +function generateCode(functionCount: number): string { + const lines: string[] = [ + "import { helper } from './helper';", + "", + "const MAX_LENGTH = 280;", + "", + ]; + + for (let i = 0; i < functionCount; i++) { + lines.push(`export function process${i}(input: Item${i}): Result${i} {`); + lines.push(` const cache = new Map();`); + lines.push(` for (const entry of input.entries) {`); + lines.push(` if (entry.weight > ${i % 997}) {`); + lines.push(` cache.set(entry.id, helper(entry, MAX_LENGTH));`); + lines.push(` }`); + lines.push(` }`); + lines.push(` return { id: ${i}, cache, total: cache.size };`); + lines.push(`}`); + lines.push(""); + } + + return lines.join("\n"); +} + +function mutateCode(code: string, changeRate: number): string { + return code + .split("\n") + .map((line, i) => { + if (i % changeRate !== 0 || !line.trim()) return line; + return `${line.replace(/cache/g, "store").replace(/entry/g, "record")} // reviewed`; + }) + .join("\n"); +} + +function runCodeBench(functionCount: number, label: string, runs = 20) { + const previous = generateCode(functionCount); + const current = mutateCode(previous, 20); + console.log(`\nCode diff – ${label}`); + + // diff has no line + token API, so the equivalent of Superdiff's output is a + // line diff followed by a word diff on each rewritten line. + const jsdiff = bench("diff", runs, () => { + const parts = diffLines(previous, current); + for (let i = 0; i < parts.length; i++) { + const removed = parts[i]; + const added = parts[i + 1]; + if (removed.removed && added?.added) { + diffWordsWithSpace(removed.value, added.value); + i++; + } + } + }); + + const superdiff = bench("Superdiff", runs, () => { + getCodeDiff(previous, current); + }); + + return { superdiff, jsdiff }; +} + +export function runCodeBench1K() { + return runCodeBench(100, "1k lines"); +} + +export function runCodeBench10K() { + return runCodeBench(1_000, "10k lines"); +} + +export function runCodeBench100K() { + return runCodeBench(10_000, "100k lines", 3); +} diff --git a/benchmark/index.ts b/benchmark/index.ts index 076de05..a91fc0a 100644 --- a/benchmark/index.ts +++ b/benchmark/index.ts @@ -4,7 +4,13 @@ import { runNestedObjectBench, } from "./objects"; import { runListBench100K, runListBench10K } from "./lists"; -import { runTextBench10KWords, runTextBench10KSentences } from "./texts"; +import { + runTextBench10KWords, + runTextBench100KWords, + runTextBench10KSentences, + runTextBench100KSentences, +} from "./texts"; +import { runCodeBench1K, runCodeBench10K, runCodeBench100K } from "./code"; // Method: Warm up runs, then each script is executed 20 times, and we keep the median time. // To guarantee a fair assessment, all scenarios must be run individually, with a clean heap memory. @@ -23,6 +29,13 @@ runListBench100K(); // Text runTextBench10KWords(); +runTextBench100KWords(); runTextBench10KSentences(); +runTextBench100KSentences(); + +// Code +runCodeBench1K(); +runCodeBench10K(); +runCodeBench100K(); console.log("\n- BENCHMARK COMPLETE -"); diff --git a/benchmark/texts.ts b/benchmark/texts.ts index 6223a93..4aef898 100644 --- a/benchmark/texts.ts +++ b/benchmark/texts.ts @@ -31,26 +31,42 @@ function generateSentences(sentenceCount: number, mutate = false): string { return mutated.join(" "); } -export function runTextBench10KWords() { - const prev = generateText(10_000); - const curr = generateText(10_000, true); - console.log("\nText diff – 10k words"); +function runWordsBench(wordCount: number, label: string) { + const prev = generateText(wordCount); + const curr = generateText(wordCount, true); + console.log(`\nText diff – ${label} words`); - const diff = bench("diff", 1, () => diffWords(prev, curr)); - const superdiff = bench("Superdiff", 1, () => { + const diff = bench("diff", 20, () => diffWords(prev, curr)); + const superdiff = bench("Superdiff", 20, () => { getTextDiff(prev, curr, { separation: "word" }); }); return { superdiff, diff }; } -export function runTextBench10KSentences() { - const prev = generateSentences(10_000); - const curr = generateSentences(10_000, true); - console.log("\nText diff – 10k sentences"); +function runSentencesBench(sentenceCount: number, label: string) { + const prev = generateSentences(sentenceCount); + const curr = generateSentences(sentenceCount, true); + console.log(`\nText diff – ${label} sentences`); - const diff = bench("diff", 1, () => diffSentences(prev, curr, {})); - const superdiff = bench("Superdiff", 1, () => { + const diff = bench("diff", 20, () => diffSentences(prev, curr, {})); + const superdiff = bench("Superdiff", 20, () => { getTextDiff(prev, curr, { separation: "sentence" }); }); return { superdiff, diff }; } + +export function runTextBench10KWords() { + return runWordsBench(10_000, "10k"); +} + +export function runTextBench100KWords() { + return runWordsBench(100_000, "100k"); +} + +export function runTextBench10KSentences() { + return runSentencesBench(10_000, "10k"); +} + +export function runTextBench100KSentences() { + return runSentencesBench(100_000, "100k"); +} diff --git a/jest.config.mjs b/jest.config.mjs index c8184af..2b03a1c 100644 --- a/jest.config.mjs +++ b/jest.config.mjs @@ -12,9 +12,10 @@ const config = { dynamicImport: true, }, paths: { + "@core/*": ["./src/core/*"], + "@lib/*": ["./src/lib/*"], "@mocks/*": ["./src/mocks/*"], "@models/*": ["./src/models/*"], - "@lib/*": ["./src/lib/*"], }, target: "esnext", }, diff --git a/package-lock.json b/package-lock.json index d888ab2..bb25254 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "@donedeal0/superdiff", - "version": "4.2.2", + "version": "5.0.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "@donedeal0/superdiff", - "version": "4.2.2", + "version": "5.0.0", "license": "ISC", "devDependencies": { "@swc/core": "^1.15.47", diff --git a/package.json b/package.json index e416de6..5a373df 100644 --- a/package.json +++ b/package.json @@ -1,8 +1,8 @@ { "name": "@donedeal0/superdiff", - "version": "4.2.4", + "version": "5.0.0", "type": "module", - "description": "Superdiff provides a rich and readable diff for arrays, objects, texts and coordinates. It supports stream and file inputs for handling large datasets efficiently, is battle-tested, has zero dependencies, and offers a top-tier performance.", + "description": "Superdiff provides a rich and readable diff for arrays, objects, code, text and coordinates. It supports stream and file inputs for handling large datasets efficiently, is battle-tested, has zero dependencies, and offers top-tier performance.", "main": "dist/index.js", "module": "dist/index.js", "types": "dist/index.d.ts", @@ -54,40 +54,17 @@ ] }, "keywords": [ - "array-comparison", "array-diff", - "chunks", "code-diff", - "compare", - "comparison-tool", - "coordinates", - "data-diff", - "deep-comparison", - "deep-diff", - "deep-object-diff", "diff", - "file-diff", "geo-diff", - "geo-distance", "haversine", - "isequal", "json-diff", - "json", - "lat-long", - "lcs", "list-diff", - "object-comparison", "object-diff", - "object-difference", - "object", "stream-diff", - "streaming-diff", - "streaming", "string-diff", - "text-diff", - "textdiff", - "vincenty", - "word-diff" + "text-diff" ], "scripts": { "benchmark": "tsx benchmark/index.ts", diff --git a/src/lib/text-diff/lcs/myers.ts b/src/core/myers/index.ts similarity index 78% rename from src/lib/text-diff/lcs/myers.ts rename to src/core/myers/index.ts index cdc07f6..36725ee 100644 --- a/src/lib/text-diff/lcs/myers.ts +++ b/src/core/myers/index.ts @@ -1,9 +1,4 @@ -import { TextStatus, TextToken } from "@models/text"; - -type MyersEdit = - | { status: TextStatus.EQUAL; prev: number; curr: number } - | { status: TextStatus.ADDED; curr: number } - | { status: TextStatus.DELETED; prev: number }; +import { LCSStatus, MyersEdit, Token } from "@models/lcs"; type Trace = Int32Array[]; @@ -13,7 +8,7 @@ function readDiagonal(trace: Int32Array, k: number, d: number): number { return trace[index]; } -function backtrack(trace: Trace, a: TextToken[], b: TextToken[]): MyersEdit[] { +function backtrack(trace: Trace, a: Token[], b: Token[]): MyersEdit[] { let x = a.length; let y = b.length; const edits: MyersEdit[] = []; @@ -37,7 +32,7 @@ function backtrack(trace: Trace, a: TextToken[], b: TextToken[]): MyersEdit[] { while (x > prevX && y > prevY) { edits.push({ - status: TextStatus.EQUAL, + status: LCSStatus.EQUAL, prev: x - 1, curr: y - 1, }); @@ -49,13 +44,13 @@ function backtrack(trace: Trace, a: TextToken[], b: TextToken[]): MyersEdit[] { if (x === prevX) { edits.push({ - status: TextStatus.ADDED, + status: LCSStatus.ADDED, curr: y - 1, }); y--; } else { edits.push({ - status: TextStatus.DELETED, + status: LCSStatus.DELETED, prev: x - 1, }); x--; @@ -65,7 +60,7 @@ function backtrack(trace: Trace, a: TextToken[], b: TextToken[]): MyersEdit[] { return edits.reverse(); } -export function myersDiff(a: TextToken[], b: TextToken[]): MyersEdit[] { +export function myersDiff(a: Token[], b: Token[]): MyersEdit[] { const N = a.length; const M = b.length; const max = N + M; diff --git a/src/core/myers/myers.test.ts b/src/core/myers/myers.test.ts new file mode 100644 index 0000000..96307f5 --- /dev/null +++ b/src/core/myers/myers.test.ts @@ -0,0 +1,163 @@ +import { myersDiff } from "@core/myers"; +import { LCSStatus, Token } from "@models/lcs"; + +const tok = (values: string[]): Token[] => + values.map((value, index) => ({ value, normalizedValue: value, index })); + +describe("myersDiff", () => { + it("returns no edit for two empty token lists", () => { + expect(myersDiff(tok([]), tok([]))).toStrictEqual([]); + }); + + it("marks every token equal for identical lists", () => { + expect(myersDiff(tok(["a", "b"]), tok(["a", "b"]))).toStrictEqual([ + { status: LCSStatus.EQUAL, prev: 0, curr: 0 }, + { status: LCSStatus.EQUAL, prev: 1, curr: 1 }, + ]); + }); + + it("marks every token added when previous is empty", () => { + expect(myersDiff(tok([]), tok(["a", "b"]))).toStrictEqual([ + { status: LCSStatus.ADDED, curr: 0 }, + { status: LCSStatus.ADDED, curr: 1 }, + ]); + }); + + it("marks every token deleted when current is empty", () => { + expect(myersDiff(tok(["a", "b"]), tok([]))).toStrictEqual([ + { status: LCSStatus.DELETED, prev: 0 }, + { status: LCSStatus.DELETED, prev: 1 }, + ]); + }); + + it("reports a single insertion", () => { + expect(myersDiff(tok(["a", "c"]), tok(["a", "b", "c"]))).toStrictEqual([ + { status: LCSStatus.EQUAL, prev: 0, curr: 0 }, + { status: LCSStatus.ADDED, curr: 1 }, + { status: LCSStatus.EQUAL, prev: 1, curr: 2 }, + ]); + }); + + it("reports a single deletion", () => { + expect(myersDiff(tok(["a", "b", "c"]), tok(["a", "c"]))).toStrictEqual([ + { status: LCSStatus.EQUAL, prev: 0, curr: 0 }, + { status: LCSStatus.DELETED, prev: 1 }, + { status: LCSStatus.EQUAL, prev: 2, curr: 1 }, + ]); + }); + + it("reports a replacement", () => { + expect(myersDiff(tok(["a", "b", "c"]), tok(["a", "x", "c"]))).toStrictEqual( + [ + { status: LCSStatus.EQUAL, prev: 0, curr: 0 }, + { status: LCSStatus.DELETED, prev: 1 }, + { status: LCSStatus.ADDED, curr: 1 }, + { status: LCSStatus.EQUAL, prev: 2, curr: 2 }, + ], + ); + }); + + it("keeps a common prefix and suffix equal", () => { + expect( + myersDiff(tok(["a", "b", "c", "d"]), tok(["a", "x", "y", "d"])), + ).toStrictEqual([ + { status: LCSStatus.EQUAL, prev: 0, curr: 0 }, + { status: LCSStatus.DELETED, prev: 1 }, + { status: LCSStatus.DELETED, prev: 2 }, + { status: LCSStatus.ADDED, curr: 1 }, + { status: LCSStatus.ADDED, curr: 2 }, + { status: LCSStatus.EQUAL, prev: 3, curr: 3 }, + ]); + }); + + it("handles two lists with nothing in common", () => { + expect(myersDiff(tok(["a", "b"]), tok(["x", "y"]))).toStrictEqual([ + { status: LCSStatus.DELETED, prev: 0 }, + { status: LCSStatus.DELETED, prev: 1 }, + { status: LCSStatus.ADDED, curr: 0 }, + { status: LCSStatus.ADDED, curr: 1 }, + ]); + }); + + it("handles a repeated token", () => { + expect(myersDiff(tok(["a", "a", "b"]), tok(["a", "b"]))).toStrictEqual([ + { status: LCSStatus.EQUAL, prev: 0, curr: 0 }, + { status: LCSStatus.DELETED, prev: 1 }, + { status: LCSStatus.EQUAL, prev: 2, curr: 1 }, + ]); + }); + + it("compares on normalizedValue, not on value", () => { + const previous: Token[] = [ + { value: "Foo", normalizedValue: "foo", index: 0 }, + ]; + const current: Token[] = [ + { value: "FOO", normalizedValue: "foo", index: 0 }, + ]; + + expect(myersDiff(previous, current)).toStrictEqual([ + { status: LCSStatus.EQUAL, prev: 0, curr: 0 }, + ]); + }); + + it("produces an edit script that rebuilds both token lists", () => { + let seed = 99; + const rnd = () => + (seed = (seed * 1103515245 + 12345) & 0x7fffffff) / 0x7fffffff; + const alphabet = "abcde"; + const make = () => { + const size = Math.floor(rnd() * 12); + const values: string[] = []; + for (let i = 0; i < size; i++) { + values.push(alphabet[Math.floor(rnd() * alphabet.length)]); + } + return tok(values); + }; + + for (let i = 0; i < 500; i++) { + const previous = make(); + const current = make(); + const edits = myersDiff(previous, current); + + let rebuiltPrevious = ""; + let rebuiltCurrent = ""; + for (const edit of edits) { + if (edit.status === LCSStatus.EQUAL) { + rebuiltPrevious += previous[edit.prev].value; + rebuiltCurrent += current[edit.curr].value; + } else if (edit.status === LCSStatus.ADDED) { + rebuiltCurrent += current[edit.curr].value; + } else { + rebuiltPrevious += previous[edit.prev].value; + } + } + + expect(rebuiltPrevious).toBe(previous.map((t) => t.value).join("")); + expect(rebuiltCurrent).toBe(current.map((t) => t.value).join("")); + } + }); + + it("walks indexes forward without repeating or skipping", () => { + const previous = tok(["a", "b", "c", "d", "e"]); + const current = tok(["a", "x", "c", "y", "e"]); + const edits = myersDiff(previous, current); + + const prevIndexes: number[] = []; + const currIndexes: number[] = []; + for (const edit of edits) { + if (edit.status !== LCSStatus.ADDED) prevIndexes.push(edit.prev); + if (edit.status !== LCSStatus.DELETED) currIndexes.push(edit.curr); + } + + expect(prevIndexes).toStrictEqual([0, 1, 2, 3, 4]); + expect(currIndexes).toStrictEqual([0, 1, 2, 3, 4]); + }); + + it("stays linear on a long identical list", () => { + const values = Array.from({ length: 5000 }, (_unused, i) => `t${i}`); + const edits = myersDiff(tok(values), tok(values)); + + expect(edits).toHaveLength(5000); + expect(edits.every((edit) => edit.status === LCSStatus.EQUAL)).toBe(true); + }); +}); diff --git a/src/index.ts b/src/index.ts index 648e741..4824cd7 100644 --- a/src/index.ts +++ b/src/index.ts @@ -2,8 +2,10 @@ export { getListDiff } from "./lib/list-diff"; export { getObjectDiff } from "./lib/object-diff"; export { getTextDiff } from "./lib/text-diff"; export { getGeoDiff } from "./lib/geo-diff"; +export { getCodeDiff } from "./lib/code-diff"; export * from "./models/list"; export * from "./models/object"; export * from "./models/stream"; export * from "./models/text"; export * from "./models/geo"; +export * from "./models/code"; diff --git a/src/lib/code-diff/code-diff.test.ts b/src/lib/code-diff/code-diff.test.ts new file mode 100644 index 0000000..86cdec2 --- /dev/null +++ b/src/lib/code-diff/code-diff.test.ts @@ -0,0 +1,1123 @@ +import { getCodeDiff } from "@lib/code-diff"; +import { CodeDiff } from "@models/code"; + +function rebuild(diff: CodeDiff): { previous: string; current: string } { + const previous: string[] = []; + const current: string[] = []; + for (const line of diff.diff) { + switch (line.status) { + case "equal": + previous.push(line.previousValue ?? line.value); + current.push(line.value); + break; + case "updated": + previous.push(line.previousValue ?? ""); + current.push(line.value); + break; + case "deleted": + previous.push(line.value); + break; + case "added": + current.push(line.value); + break; + } + } + return { previous: previous.join("\n"), current: current.join("\n") }; +} + +describe("getCodeDiff - general", () => { + it("returns equal for two empty codes", () => { + expect(getCodeDiff(null, null)).toStrictEqual({ + type: "code", + status: "equal", + diff: [], + }); + }); + + it("marks every line added when there is no previous code", () => { + expect(getCodeDiff(null, "const a = 1;\nconst b = 2;")).toStrictEqual({ + type: "code", + status: "added", + diff: [ + { + value: "const a = 1;", + line: 1, + previousLine: null, + status: "added", + }, + { + value: "const b = 2;", + line: 2, + previousLine: null, + status: "added", + }, + ], + }); + }); + + it("marks every line deleted when there is no current code", () => { + expect(getCodeDiff("const a = 1;\nconst b = 2;", null)).toStrictEqual({ + type: "code", + status: "deleted", + diff: [ + { + value: "const a = 1;", + line: null, + previousLine: 1, + status: "deleted", + }, + { + value: "const b = 2;", + line: null, + previousLine: 2, + status: "deleted", + }, + ], + }); + }); + + it("returns equal for identical code", () => { + expect(getCodeDiff("const a = 1;", "const a = 1;")).toStrictEqual({ + type: "code", + status: "equal", + diff: [ + { + value: "const a = 1;", + previousValue: "const a = 1;", + line: 1, + previousLine: 1, + status: "equal", + }, + ], + }); + }); +}); + +describe("getCodeDiff - lines", () => { + it("numbers the lines of a mixed change", () => { + expect( + getCodeDiff("a();\nb();\nc();", "a();\nb2();\nc();\nd();"), + ).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: "a();", + previousValue: "a();", + line: 1, + previousLine: 1, + status: "equal", + }, + { + value: "b2();", + previousValue: "b();", + line: 2, + previousLine: 2, + status: "updated", + diff: [ + { value: "b2", previousValue: "b", status: "updated" }, + { value: "(", status: "equal" }, + { value: ")", status: "equal" }, + { value: ";", status: "equal" }, + ], + }, + { + value: "c();", + previousValue: "c();", + line: 3, + previousLine: 3, + status: "equal", + }, + { + value: "d();", + line: 4, + previousLine: null, + status: "added", + }, + ], + }); + }); + + it("reports an inserted line without touching its neighbours", () => { + expect(getCodeDiff("a();\nc();", "a();\nb();\nc();")).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: "a();", + previousValue: "a();", + line: 1, + previousLine: 1, + status: "equal", + }, + { + value: "b();", + line: 2, + previousLine: null, + status: "added", + }, + { + value: "c();", + previousValue: "c();", + line: 3, + previousLine: 2, + status: "equal", + }, + ], + }); + }); + + it("reports a deleted line", () => { + expect(getCodeDiff("a();\nb();\nc();", "a();\nc();")).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: "a();", + previousValue: "a();", + line: 1, + previousLine: 1, + status: "equal", + }, + { + value: "b();", + line: null, + previousLine: 2, + status: "deleted", + }, + { + value: "c();", + previousValue: "c();", + line: 2, + previousLine: 3, + status: "equal", + }, + ], + }); + }); + + it("reports a line ending change", () => { + expect(getCodeDiff("a\r\nb", "a\nb")).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: "a\r", + line: null, + previousLine: 1, + status: "deleted", + }, + { + value: "a", + line: 1, + previousLine: null, + status: "added", + }, + { + value: "b", + previousValue: "b", + line: 2, + previousLine: 2, + status: "equal", + }, + ], + }); + }); +}); + +describe("getCodeDiff - tokens", () => { + it("reports the changed tokens of an updated line", () => { + expect(getCodeDiff("foo(a).bar;", "foo(b).bar;")).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: "foo(b).bar;", + previousValue: "foo(a).bar;", + line: 1, + previousLine: 1, + status: "updated", + diff: [ + { value: "foo", status: "equal" }, + { value: "(", status: "equal" }, + { value: "b", previousValue: "a", status: "updated" }, + { value: ")", status: "equal" }, + { value: ".", status: "equal" }, + { value: "bar", status: "equal" }, + { value: ";", status: "equal" }, + ], + }, + ], + }); + }); +}); + +describe("getCodeDiff - deletions inside a line", () => { + it("reports a removed argument while the line stays", () => { + expect(getCodeDiff("foo(a, b);", "foo(a);")).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: "foo(a);", + previousValue: "foo(a, b);", + line: 1, + previousLine: 1, + status: "updated", + diff: [ + { value: "foo", status: "equal" }, + { value: "(", status: "equal" }, + { value: "a", status: "equal" }, + { value: ",", status: "deleted" }, + { value: " b", status: "deleted" }, + { value: ")", status: "equal" }, + { value: ";", status: "equal" }, + ], + }, + ], + }); + }); + + it("reports a removed trailing comment while the line stays", () => { + expect(getCodeDiff("const x = 1; // note", "const x = 1;")).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: "const x = 1;", + previousValue: "const x = 1; // note", + line: 1, + previousLine: 1, + status: "updated", + diff: [ + { value: "const", status: "equal" }, + { value: " x", status: "equal" }, + { value: " =", status: "equal" }, + { value: " 1", status: "equal" }, + { value: ";", status: "equal" }, + { value: " /", status: "deleted" }, + { value: "/", status: "deleted" }, + { value: " note", status: "deleted" }, + ], + }, + ], + }); + }); + + it("keeps the neighbouring lines equal", () => { + expect( + getCodeDiff("a();\nfoo(a, b);\nc();", "a();\nfoo(a);\nc();"), + ).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: "a();", + previousValue: "a();", + line: 1, + previousLine: 1, + status: "equal", + }, + { + value: "foo(a);", + previousValue: "foo(a, b);", + line: 2, + previousLine: 2, + status: "updated", + diff: [ + { value: "foo", status: "equal" }, + { value: "(", status: "equal" }, + { value: "a", status: "equal" }, + { value: ",", status: "deleted" }, + { value: " b", status: "deleted" }, + { value: ")", status: "equal" }, + { value: ";", status: "equal" }, + ], + }, + { + value: "c();", + previousValue: "c();", + line: 3, + previousLine: 3, + status: "equal", + }, + ], + }); + }); +}); + +describe("getCodeDiff - whitespace", () => { + it("reports a re-indentation", () => { + expect(getCodeDiff(" return x;", " return x;")).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: " return x;", + previousValue: " return x;", + line: 1, + previousLine: 1, + status: "updated", + diff: [ + { + value: " return", + previousValue: " return", + status: "updated", + }, + { value: " x", status: "equal" }, + { value: ";", status: "equal" }, + ], + }, + ], + }); + }); + + it("reports tabs converted to spaces", () => { + expect(getCodeDiff("\tif (a) {", " if (a) {")).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: " if (a) {", + previousValue: "\tif (a) {", + line: 1, + previousLine: 1, + status: "updated", + diff: [ + { value: " if", previousValue: "\tif", status: "updated" }, + { value: " (", status: "equal" }, + { value: "a", status: "equal" }, + { value: ")", status: "equal" }, + { value: " {", status: "equal" }, + ], + }, + ], + }); + }); +}); + +describe("getCodeDiff - languages", () => { + it("python - added default argument", () => { + expect( + getCodeDiff("def load(path, mode):", "def load(path, mode='r'):"), + ).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: "def load(path, mode='r'):", + previousValue: "def load(path, mode):", + line: 1, + previousLine: 1, + status: "updated", + diff: [ + { value: "def", status: "equal" }, + { value: " load", status: "equal" }, + { value: "(", status: "equal" }, + { value: "path", status: "equal" }, + { value: ",", status: "equal" }, + { value: " mode", status: "equal" }, + { value: "=", status: "added" }, + { value: "'", status: "added" }, + { value: "r", status: "added" }, + { value: "'", status: "added" }, + { value: ")", status: "equal" }, + { value: ":", status: "equal" }, + ], + }, + ], + }); + }); + + it("css - hex colour", () => { + expect( + getCodeDiff(".btn { color: #fff; }", ".btn { color: #000; }"), + ).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: ".btn { color: #000; }", + previousValue: ".btn { color: #fff; }", + line: 1, + previousLine: 1, + status: "updated", + diff: [ + { value: ".", status: "equal" }, + { value: "btn", status: "equal" }, + { value: " {", status: "equal" }, + { value: " color", status: "equal" }, + { value: ":", status: "equal" }, + { value: " #", status: "equal" }, + { value: "000", previousValue: "fff", status: "updated" }, + { value: ";", status: "equal" }, + { value: " }", status: "equal" }, + ], + }, + ], + }); + }); + + it("chinese - identifier kept whole", () => { + expect( + getCodeDiff( + "const \u7528\u6237\u540d = '\u5f20\u4e09';", + "const \u7528\u6237\u540d = '\u674e\u56db';", + ), + ).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: "const \u7528\u6237\u540d = '\u674e\u56db';", + previousValue: "const \u7528\u6237\u540d = '\u5f20\u4e09';", + line: 1, + previousLine: 1, + status: "updated", + diff: [ + { value: "const", status: "equal" }, + { value: " \u7528\u6237\u540d", status: "equal" }, + { value: " =", status: "equal" }, + { value: " '", status: "equal" }, + { + value: "\u674e\u56db", + previousValue: "\u5f20\u4e09", + status: "updated", + }, + { value: "'", status: "equal" }, + { value: ";", status: "equal" }, + ], + }, + ], + }); + }); +}); + +describe("getCodeDiff - line pairing", () => { + it("pairs a rewritten line with its replacement, not with its neighbour", () => { + const previous = ["let sum = 0;", "return sum;"].join("\n"); + const current = [ + "let sum = 0;", + "const tax = sum * rate;", + "return sum + tax;", + ].join("\n"); + + const diff = getCodeDiff(previous, current); + + expect( + diff.diff.map((line) => [ + line.status, + line.previousValue ?? null, + line.value, + ]), + ).toStrictEqual([ + ["equal", "let sum = 0;", "let sum = 0;"], + ["added", null, "const tax = sum * rate;"], + ["updated", "return sum;", "return sum + tax;"], + ]); + }); + + it("reports unrelated lines as a deletion and an insertion", () => { + const diff = getCodeDiff("const a = 1;", "import x from 'y';"); + + expect(diff.diff.map((line) => line.status)).toStrictEqual([ + "deleted", + "added", + ]); + }); +}); + +describe("getCodeDiff - unicode safety", () => { + const GRAPHEMES: [string, string][] = [ + ["decomposed accent", "cafe\u0301"], + ["devanagari", "\u0915\u094D\u0937\u0924\u094D\u0930\u093F\u092F"], + ["hebrew with niqqud", "\u05E9\u05B8\u05C1\u05DC\u05D5\u05B9\u05DD"], + ["zwj family", "\u{1F468}\u200D\u{1F469}\u200D\u{1F467}"], + ["flag", "\u{1F1EB}\u{1F1F7}"], + ["skin tone", "\u{1F44D}\u{1F3FD}"], + ["keycap", "1\uFE0F\u20E3"], + ]; + + it.each(GRAPHEMES)("keeps %s in a single token", (_name, code) => { + const diff = getCodeDiff(code, code + ";"); + const tokens = diff.diff[0].diff ?? []; + + expect(tokens.map((token) => token.value)).toStrictEqual([code, ";"]); + }); + + it("never splits a grapheme cluster across a random corpus", () => { + const SCRIPTS = [ + "abcXYZ_$", + "0123", + "(){}[];:.,=+-*/<>!?&|^~@#%'\"`\\", + "cafe\u0301", + "\u4E2D\u6587", + "\u65E5\u672C\u8A9E", + "\u0645\u0631\u062D\u0628\u0627", + "\u0915\u094D\u0937\u0924\u094D\u0930\u093F\u092F", + "\u{1F468}\u200D\u{1F469}\u200D\u{1F467}", + "\u{1F1EB}\u{1F1F7}", + "1\uFE0F\u20E3", + "\u{1F600}", + " ", + "\t", + " ", + ]; + const segmenter = new Intl.Segmenter("en", { granularity: "grapheme" }); + + let seed = 12345; + const rnd = () => + (seed = (seed * 1103515245 + 12345) & 0x7fffffff) / 0x7fffffff; + + for (let i = 0; i < 500; i++) { + let code = ""; + const parts = Math.floor(rnd() * 10); + for (let j = 0; j < parts; j++) { + code += SCRIPTS[Math.floor(rnd() * SCRIPTS.length)]; + } + // A shared prefix keeps the two lines similar enough to be paired, so + // there is a token level diff to inspect. + const previous = "const a = " + code; + const current = "const b = " + code; + + const tokens = getCodeDiff(previous, current).diff[0].diff ?? []; + const values = tokens.map((token) => token.value); + expect(values.join("")).toBe(current); + + const boundaries = new Set([0]); + let end = 0; + for (const grapheme of segmenter.segment(current)) { + end += grapheme.segment.length; + boundaries.add(end); + } + + let offset = 0; + for (const value of values) { + offset += value.length; + expect(boundaries.has(offset)).toBe(true); + } + } + }); +}); + +describe("getCodeDiff - large code snippet", () => { + const PREVIOUS = [ + "import { useState, useEffect } from 'react';", + "import { fetchPosts } from '../api/posts';", + "", + "const MAX_LENGTH = 280;", + "", + "export function PostList({ userId, isSocialPost }) {", + " const [posts, setPosts] = useState([]);", + " const [loading, setLoading] = useState(true);", + "", + " useEffect(() => {", + " let cancelled = false;", + " fetchPosts(userId).then((result) => {", + " if (!cancelled) {", + " setPosts(result.items);", + " setLoading(false);", + " }", + " });", + " return () => {", + " cancelled = true;", + " };", + " }, [userId]);", + "", + " if (loading) {", + " return ;", + " }", + "", + " return (", + "
    ", + " {posts.map((post) => {", + " const hitSentence = post.body;", + " const showMoreButton = isSocialPost && hitSentence?.length > MAX_LENGTH;", + " return
  • {hitSentence}
  • ;", + " })}", + "
", + " );", + "}", + ].join("\n"); + + const CURRENT = [ + "import { useState, useEffect, useMemo } from 'react';", + "import { fetchPosts } from '../api/posts';", + "", + "const SOCIAL_TEXT_MAX_LENGTH = 320;", + "", + "export function PostList({ userId, isYoutubePost }) {", + " const [posts, setPosts] = useState([]);", + " const [loading, setLoading] = useState(true);", + "", + " useEffect(() => {", + " let cancelled = false;", + " fetchPosts(userId).then((result) => {", + " if (!cancelled) {", + " setPosts(result.items);", + " setLoading(false);", + " }", + " });", + " return () => {", + " cancelled = true;", + " };", + " }, [userId]);", + "", + " if (loading) {", + " return ;", + " }", + "", + " return (", + "
    ", + " {posts.map((post) => {", + " const hitSentence = post.body;", + " const showMoreButton = isYoutubePost && hitSentence?.length > SOCIAL_TEXT_MAX_LENGTH;", + " return
  • {hitSentence}
  • ;", + " })}", + "
", + " );", + "}", + ].join("\n"); + + const diff = getCodeDiff(PREVIOUS, CURRENT); + + it("rebuilds both files exactly", () => { + expect(rebuild(diff)).toEqual({ previous: PREVIOUS, current: CURRENT }); + }); + + it("reports the file as updated", () => { + expect(diff.status).toBe("updated"); + }); + + it("leaves every untouched line equal", () => { + const changed = diff.diff.filter((line) => line.status !== "equal"); + + expect(changed).toHaveLength(5); + expect(changed.every((line) => line.status === "updated")).toBe(true); + }); + + it("numbers every line of both files", () => { + expect(diff.diff).toHaveLength(PREVIOUS.split("\n").length); + expect(diff.diff.map((line) => line.line)).toStrictEqual( + diff.diff.map((_unused, i) => i + 1), + ); + expect(diff.diff.map((line) => line.previousLine)).toStrictEqual( + diff.diff.map((_unused, i) => i + 1), + ); + }); + + it("highlights only the renamed identifiers inside the changed lines", () => { + const showMore = diff.diff.find((line) => + line.value.includes("showMoreButton"), + ); + + expect(showMore?.status).toBe("updated"); + expect( + (showMore?.diff ?? []) + .filter((token) => token.status !== "equal") + .map((token) => [token.previousValue, token.value]), + ).toStrictEqual([ + [" isSocialPost", " isYoutubePost"], + [" MAX_LENGTH", " SOCIAL_TEXT_MAX_LENGTH"], + ]); + }); + + it("reports the added import without rewriting the line", () => { + const importLine = diff.diff[0]; + const added = (importLine.diff ?? []).filter( + (token) => token.status === "added", + ); + + expect(importLine.status).toBe("updated"); + expect(added.map((token) => token.value)).toStrictEqual([",", " useMemo"]); + }); +}); + +describe("getCodeDiff - full output", () => { + it("shows every line status in one diff", () => { + expect( + getCodeDiff( + "function total(items) {\n let sum = 0;\n console.log(sum);\n return sum;\n}", + "function total(items, rate) {\n let sum = 0;\n const tax = sum * rate;\n return sum + tax;\n}", + ), + ).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: "function total(items, rate) {", + previousValue: "function total(items) {", + line: 1, + previousLine: 1, + status: "updated", + diff: [ + { value: "function", status: "equal" }, + { value: " total", status: "equal" }, + { value: "(", status: "equal" }, + { value: "items", status: "equal" }, + { value: ",", status: "added" }, + { value: " rate", status: "added" }, + { value: ")", status: "equal" }, + { value: " {", status: "equal" }, + ], + }, + { + value: " let sum = 0;", + previousValue: " let sum = 0;", + line: 2, + previousLine: 2, + status: "equal", + }, + { + value: " console.log(sum);", + line: null, + previousLine: 3, + status: "deleted", + }, + { + value: " const tax = sum * rate;", + line: 3, + previousLine: null, + status: "added", + }, + { + value: " return sum + tax;", + previousValue: " return sum;", + line: 4, + previousLine: 4, + status: "updated", + diff: [ + { value: " return", status: "equal" }, + { value: " sum", status: "equal" }, + { value: " +", status: "added" }, + { value: " tax", status: "added" }, + { value: ";", status: "equal" }, + ], + }, + { + value: "}", + previousValue: "}", + line: 5, + previousLine: 5, + status: "equal", + }, + ], + }); + }); + + it("javascript", () => { + expect(getCodeDiff("foo(a).bar = 1;", "foo(b).bar = 2;")).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: "foo(b).bar = 2;", + previousValue: "foo(a).bar = 1;", + line: 1, + previousLine: 1, + status: "updated", + diff: [ + { value: "foo", status: "equal" }, + { value: "(", status: "equal" }, + { value: "b", previousValue: "a", status: "updated" }, + { value: ")", status: "equal" }, + { value: ".", status: "equal" }, + { value: "bar", status: "equal" }, + { value: " =", status: "equal" }, + { value: " 2", previousValue: " 1", status: "updated" }, + { value: ";", status: "equal" }, + ], + }, + ], + }); + }); + + it("clojure", () => { + expect(getCodeDiff("(set! foo-bar 1)", "(set! foo-baz 2)")).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: "(set! foo-baz 2)", + previousValue: "(set! foo-bar 1)", + line: 1, + previousLine: 1, + status: "updated", + diff: [ + { value: "(", status: "equal" }, + { value: "set", status: "equal" }, + { value: "!", status: "equal" }, + { value: " foo", status: "equal" }, + { value: "-", status: "equal" }, + { value: "bar", status: "deleted" }, + { value: "baz", previousValue: " 1", status: "updated" }, + { value: " 2", status: "added" }, + { value: ")", status: "equal" }, + ], + }, + ], + }); + }); + + it("ruby", () => { + expect(getCodeDiff("@x.nil? && y!", "@x.present? || y!")).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: "@x.present? || y!", + previousValue: "@x.nil? && y!", + line: 1, + previousLine: 1, + status: "updated", + diff: [ + { value: "@", status: "equal" }, + { value: "x", status: "equal" }, + { value: ".", status: "equal" }, + { value: "present", previousValue: "nil", status: "updated" }, + { value: "?", status: "equal" }, + { value: " &", status: "deleted" }, + { value: " |", previousValue: "&", status: "updated" }, + { value: "|", status: "added" }, + { value: " y", status: "equal" }, + { value: "!", status: "equal" }, + ], + }, + ], + }); + }); + + it("css", () => { + expect(getCodeDiff(".a--b{color:#fff}", ".a--c{color:#000}")).toStrictEqual( + { + type: "code", + status: "updated", + diff: [ + { + value: ".a--c{color:#000}", + previousValue: ".a--b{color:#fff}", + line: 1, + previousLine: 1, + status: "updated", + diff: [ + { value: ".", status: "equal" }, + { value: "a", status: "equal" }, + { value: "-", status: "equal" }, + { value: "-", status: "equal" }, + { value: "c", previousValue: "b", status: "updated" }, + { value: "{", status: "equal" }, + { value: "color", status: "equal" }, + { value: ":", status: "equal" }, + { value: "#", status: "equal" }, + { value: "000", previousValue: "fff", status: "updated" }, + { value: "}", status: "equal" }, + ], + }, + ], + }, + ); + }); + + it("sql", () => { + expect( + getCodeDiff( + "SELECT * FROM t WHERE a <> 'b';", + "SELECT id FROM t WHERE a = 'c';", + ), + ).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: "SELECT id FROM t WHERE a = 'c';", + previousValue: "SELECT * FROM t WHERE a <> 'b';", + line: 1, + previousLine: 1, + status: "updated", + diff: [ + { value: "SELECT", status: "equal" }, + { value: " id", previousValue: " *", status: "updated" }, + { value: " FROM", status: "equal" }, + { value: " t", status: "equal" }, + { value: " WHERE", status: "equal" }, + { value: " a", status: "equal" }, + { value: " <", status: "deleted" }, + { value: " =", previousValue: ">", status: "updated" }, + { value: " '", status: "equal" }, + { value: "c", previousValue: "b", status: "updated" }, + { value: "'", status: "equal" }, + { value: ";", status: "equal" }, + ], + }, + ], + }); + }); + + it("php", () => { + expect( + getCodeDiff("$x = @$y['k'] ?? 'd';", "$x = @$z['k'] ?? 'e';"), + ).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: "$x = @$z['k'] ?? 'e';", + previousValue: "$x = @$y['k'] ?? 'd';", + line: 1, + previousLine: 1, + status: "updated", + diff: [ + { value: "$x", status: "equal" }, + { value: " =", status: "equal" }, + { value: " @", status: "equal" }, + { value: "$z", previousValue: "$y", status: "updated" }, + { value: "[", status: "equal" }, + { value: "'", status: "equal" }, + { value: "k", status: "equal" }, + { value: "'", status: "equal" }, + { value: "]", status: "equal" }, + { value: " ?", status: "equal" }, + { value: "?", status: "equal" }, + { value: " '", status: "equal" }, + { value: "e", previousValue: "d", status: "updated" }, + { value: "'", status: "equal" }, + { value: ";", status: "equal" }, + ], + }, + ], + }); + }); + + it("tabs to spaces", () => { + expect( + getCodeDiff( + "\tif (a) {\n\t\treturn;\n\t}", + " if (b) {\n return;\n }", + ), + ).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: " if (b) {", + previousValue: "\tif (a) {", + line: 1, + previousLine: 1, + status: "updated", + diff: [ + { value: " if", previousValue: "\tif", status: "updated" }, + { value: " (", status: "equal" }, + { value: "b", previousValue: "a", status: "updated" }, + { value: ")", status: "equal" }, + { value: " {", status: "equal" }, + ], + }, + { + value: " return;", + previousValue: "\t\treturn;", + line: 2, + previousLine: 2, + status: "updated", + diff: [ + { + value: " return", + previousValue: "\t\treturn", + status: "updated", + }, + { value: ";", status: "equal" }, + ], + }, + { + value: "\t}", + line: null, + previousLine: 3, + status: "deleted", + }, + { + value: " }", + line: 3, + previousLine: null, + status: "added", + }, + ], + }); + }); + + it("blank lines removed", () => { + expect(getCodeDiff("a\n\n\nb", "a\n\nb")).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: "a", + previousValue: "a", + line: 1, + previousLine: 1, + status: "equal", + }, + { + value: "", + previousValue: "", + line: 2, + previousLine: 2, + status: "equal", + }, + { + value: "", + line: null, + previousLine: 3, + status: "deleted", + }, + { + value: "b", + previousValue: "b", + line: 3, + previousLine: 4, + status: "equal", + }, + ], + }); + }); + + it("whitespace only", () => { + expect(getCodeDiff(" ", " ")).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: " ", + line: null, + previousLine: 1, + status: "deleted", + }, + { + value: " ", + line: 1, + previousLine: null, + status: "added", + }, + ], + }); + }); + + it("trailing whitespace removed", () => { + expect(getCodeDiff("trailing ", "trailing")).toStrictEqual({ + type: "code", + status: "updated", + diff: [ + { + value: "trailing ", + line: null, + previousLine: 1, + status: "deleted", + }, + { + value: "trailing", + line: 1, + previousLine: null, + status: "added", + }, + ], + }); + }); +}); diff --git a/src/lib/code-diff/diff.ts b/src/lib/code-diff/diff.ts new file mode 100644 index 0000000..a5c09e9 --- /dev/null +++ b/src/lib/code-diff/diff.ts @@ -0,0 +1,132 @@ +import { myersDiff } from "@core/myers"; +import { + CodeDiff, + CodeLineDiff, + CodeStatus, + CodeTokenDiff, +} from "@models/code"; +import { LCSStatus, Token } from "@models/lcs"; +import { getLCSTokenDiff } from "./lcs"; +import { tokenizeLines } from "./lines"; +import { getLinePairs } from "./pair"; +import { tokenizeCode } from "./tokenize"; +import { getDiffStatus } from "./utils"; + +function getTokenDiff(previousLine: string, currentLine: string) { + return getLCSTokenDiff(tokenizeCode(previousLine), tokenizeCode(currentLine)); +} + +function getLineDiff( + value: string, + previousValue: string | undefined, + line: number | null, + previousLine: number | null, + status: CodeStatus, + diff?: CodeTokenDiff[], +): CodeLineDiff { + const entry: CodeLineDiff = { value, line, previousLine, status }; + if (previousValue !== undefined) entry.previousValue = previousValue; + if (diff) entry.diff = diff; + return entry; +} + +export function getCodeLinesDiff( + previousCode: string, + currentCode: string, +): CodeDiff { + const previousLines = tokenizeLines(previousCode); + const currentLines = tokenizeLines(currentCode); + const edits = myersDiff(previousLines, currentLines); + + const diff: CodeLineDiff[] = []; + const statusSet = new Set(); + + let i = 0; + while (i < edits.length) { + const edit = edits[i]; + + if (edit.status === LCSStatus.EQUAL) { + diff.push( + getLineDiff( + currentLines[edit.curr].value, + previousLines[edit.prev].value, + edit.curr + 1, + edit.prev + 1, + CodeStatus.EQUAL, + ), + ); + statusSet.add(CodeStatus.EQUAL); + i++; + continue; + } + + const removed: Token[] = []; + const added: Token[] = []; + while (i < edits.length && edits[i].status !== LCSStatus.EQUAL) { + const blockEdit = edits[i]; + if (blockEdit.status === LCSStatus.DELETED) { + removed.push(previousLines[blockEdit.prev]); + } else if (blockEdit.status === LCSStatus.ADDED) { + added.push(currentLines[blockEdit.curr]); + } + i++; + } + + const pairs = getLinePairs(removed, added); + let removedIndex = 0; + let addedIndex = 0; + + const pushDeletedLine = (line: Token) => { + diff.push( + getLineDiff( + line.value, + undefined, + null, + line.index + 1, + CodeStatus.DELETED, + ), + ); + statusSet.add(CodeStatus.DELETED); + }; + const pushAddedLine = (line: Token) => { + diff.push( + getLineDiff( + line.value, + undefined, + line.index + 1, + null, + CodeStatus.ADDED, + ), + ); + statusSet.add(CodeStatus.ADDED); + }; + + for (const pair of pairs) { + while (removedIndex < pair.removed) + pushDeletedLine(removed[removedIndex++]); + while (addedIndex < pair.added) pushAddedLine(added[addedIndex++]); + + const removedLine = removed[pair.removed]; + const addedLine = added[pair.added]; + diff.push( + getLineDiff( + addedLine.value, + removedLine.value, + addedLine.index + 1, + removedLine.index + 1, + CodeStatus.UPDATED, + getTokenDiff(removedLine.value, addedLine.value), + ), + ); + statusSet.add(CodeStatus.UPDATED); + removedIndex = pair.removed + 1; + addedIndex = pair.added + 1; + } + + while (removedIndex < removed.length) + pushDeletedLine(removed[removedIndex++]); + while (addedIndex < added.length) pushAddedLine(added[addedIndex++]); + } + + return { type: "code", status: getDiffStatus(statusSet), diff }; +} diff --git a/src/lib/code-diff/index.ts b/src/lib/code-diff/index.ts new file mode 100644 index 0000000..7d3dacc --- /dev/null +++ b/src/lib/code-diff/index.ts @@ -0,0 +1,41 @@ +import { CodeDiff, CodeStatus } from "@models/code"; +import { getCodeLinesDiff } from "./diff"; +import { tokenizeLines } from "./lines"; + +function getSingleCodeDiff(code: string, status: CodeStatus): CodeDiff { + const isAdded = status === CodeStatus.ADDED; + return { + type: "code", + status, + diff: tokenizeLines(code).map((line) => ({ + value: line.value, + line: isAdded ? line.index + 1 : null, + previousLine: isAdded ? null : line.index + 1, + status, + })), + }; +} + +/** + *Compares two codes and returns a structured diff. + * Lines are diffed first, then the lines that changed are diffed token by token. + * To diff files, read them first: `getCodeDiff(await previous.text(), await current.text())`. + * @param {string | null | undefined} previousCode - The original code. + * @param {string | null | undefined} currentCode - The current code. + * @returns CodeDiff + */ +export function getCodeDiff( + previousCode: string | null | undefined, + currentCode: string | null | undefined, +): CodeDiff { + if (!previousCode && !currentCode) { + return { type: "code", status: CodeStatus.EQUAL, diff: [] }; + } + if (!previousCode) { + return getSingleCodeDiff(currentCode as string, CodeStatus.ADDED); + } + if (!currentCode) { + return getSingleCodeDiff(previousCode as string, CodeStatus.DELETED); + } + return getCodeLinesDiff(previousCode, currentCode); +} diff --git a/src/lib/code-diff/lcs/index.ts b/src/lib/code-diff/lcs/index.ts new file mode 100644 index 0000000..36e5f66 --- /dev/null +++ b/src/lib/code-diff/lcs/index.ts @@ -0,0 +1,48 @@ +import { myersDiff } from "@core/myers"; +import { CodeStatus, CodeTokenDiff } from "@models/code"; +import { LCSStatus, Token } from "@models/lcs"; + +export function getLCSTokenDiff( + previousTokens: Token[], + currentTokens: Token[], +): CodeTokenDiff[] { + const edits = myersDiff(previousTokens, currentTokens); + const diff: CodeTokenDiff[] = []; + + for (let i = 0; i < edits.length; i++) { + const edit = edits[i]; + + if (edit.status === LCSStatus.EQUAL) { + diff.push({ + value: currentTokens[edit.curr].value, + status: CodeStatus.EQUAL, + }); + continue; + } + + if (edit.status === LCSStatus.DELETED) { + const next = edits[i + 1]; + if (next && next.status === LCSStatus.ADDED) { + diff.push({ + value: currentTokens[next.curr].value, + previousValue: previousTokens[edit.prev].value, + status: CodeStatus.UPDATED, + }); + i++; + continue; + } + diff.push({ + value: previousTokens[edit.prev].value, + status: CodeStatus.DELETED, + }); + continue; + } + + diff.push({ + value: currentTokens[edit.curr].value, + status: CodeStatus.ADDED, + }); + } + + return diff; +} diff --git a/src/lib/code-diff/lines.ts b/src/lib/code-diff/lines.ts new file mode 100644 index 0000000..4da2c25 --- /dev/null +++ b/src/lib/code-diff/lines.ts @@ -0,0 +1,8 @@ +import { Token } from "@models/lcs"; + +export const tokenizeLines = (code: string): Token[] => + code.split("\n").map((value, index) => ({ + value, + normalizedValue: value, + index, + })); diff --git a/src/lib/code-diff/pair.ts b/src/lib/code-diff/pair.ts new file mode 100644 index 0000000..ce9028a --- /dev/null +++ b/src/lib/code-diff/pair.ts @@ -0,0 +1,66 @@ +import { Token } from "@models/lcs"; +import { tokenizeCode } from "./tokenize"; + +const MAX_PAIRING_BLOCK = 40; +const MIN_SIMILARITY_SCORE = 0.34; + +export type LinePair = { removed: number; added: number }; + +function getSimilarityScore(previousLine: string, currentLine: string): number { + const previousTokens = tokenizeCode(previousLine); + const currentTokens = tokenizeCode(currentLine); + if (previousTokens.length === 0 && currentTokens.length === 0) return 1; + if (previousTokens.length === 0 || currentTokens.length === 0) return 0; + + const pool = new Map(); + for (const token of previousTokens) { + const key = token.normalizedValue; + pool.set(key, (pool.get(key) || 0) + 1); + } + + let shared = 0; + for (const token of currentTokens) { + const remaining = pool.get(token.normalizedValue); + if (remaining) { + shared++; + pool.set(token.normalizedValue, remaining - 1); + } + } + + return (2 * shared) / (previousTokens.length + currentTokens.length); +} + +export function getLinePairs(removed: Token[], added: Token[]): LinePair[] { + if (removed.length === 0 || added.length === 0) return []; + + if (removed.length * added.length > MAX_PAIRING_BLOCK * MAX_PAIRING_BLOCK) { + const size = Math.min(removed.length, added.length); + return Array.from({ length: size }, (_unused, i) => ({ + removed: i, + added: i, + })); + } + + const scored: { score: number; removed: number; added: number }[] = []; + for (let r = 0; r < removed.length; r++) { + for (let a = 0; a < added.length; a++) { + const score = getSimilarityScore(removed[r].value, added[a].value); + if (score >= MIN_SIMILARITY_SCORE) + scored.push({ score, removed: r, added: a }); + } + } + scored.sort((a, b) => b.score - a.score || a.removed - b.removed); + + const pairs: LinePair[] = []; + for (const candidate of scored) { + const crosses = pairs.some( + (pair) => + (candidate.removed - pair.removed) * (candidate.added - pair.added) <= + 0, + ); + if (crosses) continue; + pairs.push({ removed: candidate.removed, added: candidate.added }); + } + + return pairs.sort((a, b) => a.removed - b.removed); +} diff --git a/src/lib/code-diff/tokenize.ts b/src/lib/code-diff/tokenize.ts new file mode 100644 index 0000000..dd501e9 --- /dev/null +++ b/src/lib/code-diff/tokenize.ts @@ -0,0 +1,46 @@ +import { Token } from "@models/lcs"; + +const CODE_EMOJI = + String.raw`\p{Regional_Indicator}{2}` + + String.raw`|\p{Extended_Pictographic}(?:\p{Emoji_Modifier}|\uFE0F|\u200D\p{Extended_Pictographic})*`; +const CODE_IDENTIFIER = String.raw`[\p{L}\p{N}_$][\p{L}\p{N}\p{M}_$]*`; +const CODE_SYMBOL = String.raw`[^\s\p{L}\p{N}_$\p{M}]\p{M}*`; +const CODE_ORPHAN_MARKS = String.raw`\p{M}+`; +const CODE_TOKEN = new RegExp( + String.raw`\s*(?:${CODE_EMOJI}|${CODE_IDENTIFIER}|${CODE_SYMBOL}|${CODE_ORPHAN_MARKS})`, + "gu", +); + +export const tokenizeCode = (code: string | null | undefined): Token[] => { + const result: Token[] = []; + if (!code) return result; + + const tokens = code.match(CODE_TOKEN) || []; + let matchedLength = 0; + for (let i = 0; i < tokens.length; i++) { + const value = tokens[i]; + matchedLength += value.length; + result.push({ + value, + normalizedValue: value, + index: i, + }); + } + + const trailingWhitespace = code.slice(matchedLength); + if (!trailingWhitespace) return result; + + if (result.length === 0) { + result.push({ + value: trailingWhitespace, + normalizedValue: trailingWhitespace, + index: 0, + }); + return result; + } + + const lastToken = result[result.length - 1]; + lastToken.value += trailingWhitespace; + lastToken.normalizedValue += trailingWhitespace; + return result; +}; diff --git a/src/lib/code-diff/utils/index.ts b/src/lib/code-diff/utils/index.ts new file mode 100644 index 0000000..c20bac2 --- /dev/null +++ b/src/lib/code-diff/utils/index.ts @@ -0,0 +1,18 @@ +import { CodeDiff, CodeStatus } from "@models/code"; + +export function getDiffStatus(statusMap: Set): CodeDiff["status"] { + if (statusMap.has(CodeStatus.UPDATED)) return CodeStatus.UPDATED; + + const isUniqueStatus = (status: CodeStatus) => { + for (const value of statusMap) { + if (value !== status) return false; + } + return true; + }; + + if (statusMap.size === 0) return CodeStatus.EQUAL; + if (isUniqueStatus(CodeStatus.ADDED)) return CodeStatus.ADDED; + if (isUniqueStatus(CodeStatus.DELETED)) return CodeStatus.DELETED; + if (isUniqueStatus(CodeStatus.EQUAL)) return CodeStatus.EQUAL; + return CodeStatus.UPDATED; +} diff --git a/src/lib/text-diff/index.ts b/src/lib/text-diff/index.ts index b7adcbb..dd02c8e 100644 --- a/src/lib/text-diff/index.ts +++ b/src/lib/text-diff/index.ts @@ -5,9 +5,9 @@ import { TextStatus, } from "@models/text"; import { getPositionalTextDiff } from "./positional"; -import { getLCSTextDiff } from "./lcs"; import { tokenizeNormalText } from "./tokenize/normal"; import { tokenizeStrictText } from "./tokenize/strict"; +import { getLCSTextDiff } from "./lcs"; /** *Compares two texts and returns a structured diff at a character, word, or sentence level. diff --git a/src/lib/text-diff/lcs/index.ts b/src/lib/text-diff/lcs/index.ts index 5f3ed11..8aa6775 100644 --- a/src/lib/text-diff/lcs/index.ts +++ b/src/lib/text-diff/lcs/index.ts @@ -1,19 +1,20 @@ -import { TextDiff, TextStatus, TextToken, TextTokenDiff } from "@models/text"; -import { myersDiff } from "./myers"; +import { myersDiff } from "@core/myers"; +import { TextDiff, TextStatus } from "@models/text"; import { getDiffStatus } from "../utils/status"; +import { LCSStatus, Token, TokenDiff } from "@models/lcs"; export function getLCSTextDiff( - previousTokens: TextToken[], - currentTokens: TextToken[], + previousTokens: Token[], + currentTokens: Token[], ): TextDiff { const edits = myersDiff(previousTokens, currentTokens); - const diff: TextTokenDiff[] = []; + const diff: TokenDiff[] = []; const statusSet = new Set(); for (let i = 0; i < edits.length; i++) { const edit = edits[i]; - if (edit.status === TextStatus.EQUAL) { + if (edit.status === LCSStatus.EQUAL) { diff.push({ value: currentTokens[edit.curr].value, index: edit.curr, @@ -23,7 +24,7 @@ export function getLCSTextDiff( statusSet.add(TextStatus.EQUAL); } - if (edit.status === TextStatus.ADDED) { + if (edit.status === LCSStatus.ADDED) { diff.push({ value: currentTokens[edit.curr].value, index: edit.curr, @@ -33,7 +34,7 @@ export function getLCSTextDiff( statusSet.add(TextStatus.ADDED); } - if (edit.status === TextStatus.DELETED) { + if (edit.status === LCSStatus.DELETED) { diff.push({ value: previousTokens[edit.prev].value, index: null, diff --git a/src/lib/text-diff/positional/index.ts b/src/lib/text-diff/positional/index.ts index fe9a4b1..832372f 100644 --- a/src/lib/text-diff/positional/index.ts +++ b/src/lib/text-diff/positional/index.ts @@ -1,14 +1,15 @@ -import { TextDiff, TextToken, TextTokenDiff, TextStatus } from "@models/text"; +import { TextDiff, TextStatus } from "@models/text"; import { getDiffStatus } from "../utils/status"; +import { Token, TokenDiff } from "@models/lcs"; export function getPositionalTextDiff( - previousTokens: TextToken[], - currentTokens: TextToken[], + previousTokens: Token[], + currentTokens: Token[], ): TextDiff { - const previousTokensMap = new Map(); - const addedTokensMap = new Map(); + const previousTokensMap = new Map(); + const addedTokensMap = new Map(); const statusSet = new Set(); - const diff: TextTokenDiff[] = []; + const diff: TokenDiff[] = []; for (let i = 0; i < previousTokens.length; i++) { const token = previousTokens[i]; diff --git a/src/lib/text-diff/tokenize/normal.ts b/src/lib/text-diff/tokenize/normal.ts index 707a84e..55787d0 100644 --- a/src/lib/text-diff/tokenize/normal.ts +++ b/src/lib/text-diff/tokenize/normal.ts @@ -1,9 +1,9 @@ +import { Token } from "@models/lcs"; import { DEFAULT_TEXT_DIFF_OPTIONS, PUNCTUATION_REGEX, TextDiffOptions, TextSeparation, - TextToken, } from "@models/text"; function normalizeToken(token: string, options: TextDiffOptions): string { @@ -33,8 +33,8 @@ function tokenizePreservingWhitespace( text: string, options: TextDiffOptions, separation: TextSeparation, -): TextToken[] { - const result: TextToken[] = []; +): Token[] { + const result: Token[] = []; const tokens = text.match(TOKEN_WITH_LEADING_WHITESPACE[separation]) || []; let matchedLength = 0; @@ -69,9 +69,9 @@ function tokenizePreservingWhitespace( export const tokenizeNormalText = ( text: string | null | undefined, options: TextDiffOptions = DEFAULT_TEXT_DIFF_OPTIONS, -): TextToken[] => { +): Token[] => { const separation = options.separation || DEFAULT_TEXT_DIFF_OPTIONS.separation; - const result: TextToken[] = []; + const result: Token[] = []; if (!text) return result; if (options.preserveWhitespace) { diff --git a/src/lib/text-diff/tokenize/strict.ts b/src/lib/text-diff/tokenize/strict.ts index f4f0904..8367284 100644 --- a/src/lib/text-diff/tokenize/strict.ts +++ b/src/lib/text-diff/tokenize/strict.ts @@ -1,9 +1,9 @@ +import { Token } from "@models/lcs"; import { DEFAULT_TEXT_DIFF_OPTIONS, EMOJI_SPLIT_REGEX, PUNCTUATION_REGEX, TextDiffOptions, - TextToken, } from "@models/text"; const segmenterCache = new Map(); @@ -35,8 +35,8 @@ function normalizeToken(token: string, options: TextDiffOptions): string { export const tokenizeStrictText = ( text: string | null | undefined, options: TextDiffOptions = DEFAULT_TEXT_DIFF_OPTIONS, -): TextToken[] => { - const result: TextToken[] = []; +): Token[] => { + const result: Token[] = []; if (!text || !text.trim()) return result; const separation = options.separation || DEFAULT_TEXT_DIFF_OPTIONS.separation; diff --git a/src/models/code/index.ts b/src/models/code/index.ts new file mode 100644 index 0000000..9cd8b51 --- /dev/null +++ b/src/models/code/index.ts @@ -0,0 +1,27 @@ +export enum CodeStatus { + ADDED = "added", + DELETED = "deleted", + EQUAL = "equal", + UPDATED = "updated", +} + +export type CodeTokenDiff = { + value: string; + previousValue?: string; + status: `${CodeStatus}`; +}; + +export type CodeLineDiff = { + value: string; + previousValue?: string; + line: number | null; + previousLine: number | null; + status: `${CodeStatus}`; + diff?: CodeTokenDiff[]; +}; + +export type CodeDiff = { + type: "code"; + status: `${CodeStatus}`; + diff: CodeLineDiff[]; +}; diff --git a/src/models/lcs/index.ts b/src/models/lcs/index.ts new file mode 100644 index 0000000..718b044 --- /dev/null +++ b/src/models/lcs/index.ts @@ -0,0 +1,33 @@ +import { CodeStatus } from "@models/code"; +import { TextStatus } from "@models/text"; + +export type Token = { + value: string; + normalizedValue: string; + index: number; +}; + +export type TokenDiff = T extends CodeStatus + ? { + value: string; + previousValue?: string; + status: T; + } + : { + value: string; + index: number | null; + previousValue?: string; + previousIndex: number | null; + status: T; + }; + +export enum LCSStatus { + ADDED = "added", + DELETED = "deleted", + EQUAL = "equal", +} + +export type MyersEdit = + | { status: LCSStatus.EQUAL; prev: number; curr: number } + | { status: LCSStatus.ADDED; curr: number } + | { status: LCSStatus.DELETED; prev: number }; diff --git a/src/models/text/index.ts b/src/models/text/index.ts index b935ea6..963c366 100644 --- a/src/models/text/index.ts +++ b/src/models/text/index.ts @@ -13,20 +13,6 @@ export const DEFAULT_TEXT_DIFF_OPTIONS: TextDiffOptions = { locale: undefined, }; -export type TextToken = { - value: string; - normalizedValue: string; - index: number; -}; - -export type TextTokenDiff = { - value: string; - index: number | null; - previousValue?: string; - previousIndex: number | null; - status: TextStatus; -}; - export enum TextStatus { ADDED = "added", EQUAL = "equal", diff --git a/tsconfig.json b/tsconfig.json index 9588839..12a543a 100644 --- a/tsconfig.json +++ b/tsconfig.json @@ -16,6 +16,7 @@ "allowSyntheticDefaultImports": true, "skipLibCheck": true , "paths": { + "@core/*": ["./src/core/*"], "@lib/*": ["./src/lib/*"], "@mocks/*": ["./src/mocks/*"], "@models/*": ["./src/models/*"]