diff --git a/CHANGELOG.md b/CHANGELOG.md index e7559656f7..60ef5dfd0b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -172,6 +172,10 @@ and adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). #### MCP / indexing - On Windows, the shared MCP daemon now waits longer for another program — an antivirus scan, an indexer, or another session reading its lock file — to let go of that file, so it starts instead of leaving the session to fall back to a slower in-process server. (#1773) +- An MPEG transport stream video that happens to be named `.ts` (golden fixtures under `testdata/`, e2e clips) is now recognised from its first bytes and skipped as non-source instead of being fed to the TypeScript parser — a 900 KB clip used to cost about 28 seconds of CPU per file for no symbols, and a folder of them minutes. Real TypeScript is never affected. (#1910) +- A file over the size limit is no longer read before it is skipped: committed video and blob fixtures used to be decoded in full — a 400 MB fixture cost 3.4 GB of memory — only to be discarded, and the same file was read again by every resolution pass. Its size stamp now stands in for its content, during indexing and when checking for changes. (#1910) +- A file over the size limit is no longer read before it is skipped. Large committed video and blob fixtures used to be loaded into memory in full, once per indexing pass, only to be discarded; indexing, change detection and the viewer now go by the file's size instead. (#1910) +- An MPEG transport stream video that happens to be named `.ts` (golden fixtures under `testdata/`, e2e clips) is now recognised from its first bytes and skipped instead of being fed to the TypeScript parser, which spent a long time on each clip for no symbols. Real TypeScript files are still indexed. (#1910) - File watching no longer drops the full re-scan a removed directory asks for when that sync fails, so the deleted files leave the index instead of lingering. (#1964) - Daemon startup and cleanup now preserve live legacy PID-only locks while still reclaiming dead or identity-disproved records, preventing two writers from serving the same project. - Incremental sync now keeps edge rebinding crash-safe: replacing a resolved edge with its recovery reference commits atomically, so an interruption cannot permanently remove the relationship. diff --git a/__tests__/bounded-source.test.ts b/__tests__/bounded-source.test.ts new file mode 100644 index 0000000000..e531b02a2a --- /dev/null +++ b/__tests__/bounded-source.test.ts @@ -0,0 +1,96 @@ +/** + * Source reads are bounded by the read itself, not only by a stat taken + * before it (#1910). A file can grow between the stat and the read; the + * reader must still never hold more than the limit plus one byte. + */ +import { describe, it, expect, afterEach, vi } from 'vitest'; +import * as fs from 'fs'; +import * as fsp from 'fs/promises'; +import * as os from 'os'; +import * as path from 'path'; +import { readBoundedSource, readBoundedSourceSync, MAX_SOURCE_FILE_SIZE_BYTES as LIMIT } from '../src/file-limits'; + +// Pass-through wrappers, so a test can inject a file growing mid-read into the +// exact calls file-limits.ts makes. +vi.mock('fs', async (importOriginal) => { + const actual = await importOriginal(); + return { ...actual, readSync: vi.fn(actual.readSync) }; +}); +vi.mock('fs/promises', async (importOriginal) => { + const actual = await importOriginal(); + return { ...actual, open: vi.fn(actual.open) }; +}); + +const dirs: string[] = []; +function sourceFile(bytes: Buffer): string { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'cg-bounded-')); + dirs.push(dir); + const file = path.join(dir, 'source.ts'); + fs.writeFileSync(file, bytes); + return file; +} + +afterEach(() => { + for (const dir of dirs.splice(0)) fs.rmSync(dir, { recursive: true, force: true }); +}); + +describe('bounded source reads (#1910)', () => { + for (const size of [0, 100, LIMIT, LIMIT + 1]) { + it(`returns the bytes up to the limit and null past it (${size} bytes)`, async () => { + const file = sourceFile(Buffer.alloc(size, 0x61)); + for (const result of [readBoundedSourceSync(file), await readBoundedSource(file)]) { + expect(result.stats.size).toBe(size); + if (size > LIMIT) expect(result.bytes).toBeNull(); + else expect(result.bytes?.length).toBe(size); + } + }); + } + + it('stops at the limit when the file grows during a synchronous read', () => { + const file = sourceFile(Buffer.from('hello')); + const readSync = vi.mocked(fs.readSync); + const original = readSync.getMockImplementation()!; + let requested = 0; + let grown = false; + readSync.mockImplementation(((fd: number, buf: Buffer, off: number, len: number, pos: number) => { + requested += len; + if (!grown) { + grown = true; + fs.writeFileSync(file, Buffer.alloc(LIMIT + 10, 0x61)); + } + return original(fd, buf, off, len, pos); + }) as typeof fs.readSync); + let result; + try { result = readBoundedSourceSync(file); } finally { readSync.mockImplementation(original); } + expect(grown).toBe(true); + expect(requested).toBeLessThanOrEqual(LIMIT + 1); + expect(result.bytes).toBeNull(); + }); + + it('rechecks the open descriptor when the file grows between stat and open', async () => { + const file = sourceFile(Buffer.from('hello')); + const open = vi.mocked(fsp.open); + const original = open.getMockImplementation()!; + let grown = false; + open.mockImplementationOnce((async (...args: Parameters) => { + grown = true; + fs.writeFileSync(file, Buffer.alloc(LIMIT + 1)); + return original(...args); + }) as typeof fsp.open); + expect((await readBoundedSource(file)).bytes).toBeNull(); + expect(grown).toBe(true); + }); + + it('returns multibyte source byte-for-byte', async () => { + const text = 'export const greeting = "こんにちは 🌿";'; + const file = sourceFile(Buffer.from(text)); + expect((await readBoundedSource(file)).bytes?.toString('utf8')).toBe(text); + expect(readBoundedSourceSync(file).bytes?.toString('utf8')).toBe(text); + }); + + it('refuses a path that is not a regular file', () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'cg-bounded-')); + dirs.push(dir); + expect(() => readBoundedSourceSync(dir)).toThrow(/not a regular file/); + }); +}); diff --git a/__tests__/git-index-currency.test.ts b/__tests__/git-index-currency.test.ts index fbe79fb722..a5ec39195f 100644 --- a/__tests__/git-index-currency.test.ts +++ b/__tests__/git-index-currency.test.ts @@ -132,12 +132,18 @@ describe('git index currency across commits and restores (#1829)', () => { it('keeps a committed path pending when sync cannot read it', async () => { write('new.ts', 'newSymbol'); commit(); - const real = fs.readFileSync; + // Sync reads a source file through a bounded reader that opens a + // descriptor (#1910), so the failure is injected at openSync as well as + // readFileSync: it must reach whichever one the read goes through. + const realRead = fs.readFileSync; + const realOpen = fs.openSync; let injected = 0; - vi.spyOn(fs, 'readFileSync').mockImplementation(((file: any, ...args: any[]) => { + const failNewTs = (real: (...args: any[]) => unknown) => (file: any, ...args: any[]) => { if (String(file) === path.join(root, 'new.ts')) { injected++; throw new Error('Injected transient read error'); } - return (real as any)(file, ...args); - }) as typeof fs.readFileSync); + return real(file, ...args); + }; + vi.spyOn(fs, 'readFileSync').mockImplementation(failNewTs(realRead as any) as typeof fs.readFileSync); + vi.spyOn(fs, 'openSync').mockImplementation(failNewTs(realOpen as any) as typeof fs.openSync); await cg.sync(); expect(injected).toBeGreaterThan(0); expect(symbols('newSymbol')).not.toContain('newSymbol'); diff --git a/__tests__/mpeg-ts-not-typescript.test.ts b/__tests__/mpeg-ts-not-typescript.test.ts new file mode 100644 index 0000000000..19fd75718c --- /dev/null +++ b/__tests__/mpeg-ts-not-typescript.test.ts @@ -0,0 +1,230 @@ +/** + * An MPEG transport stream named `.ts` is not TypeScript (#1910). + * + * Golden video fixtures (`testdata/*.ts`) share TypeScript's extension; fed to + * the tree-sitter TypeScript parser a 900 KB clip costs ~28 s of CPU for zero + * symbols. The fix recognises the stream from the head of the file (0x47 sync + * byte at each 188-byte packet boundary, plus a NUL byte no UTF-8 source has) + * and drops it at discovery — not indexed, not parsed, not counted, not + * reported as an unsupported language. + */ +import { describe, it, expect, afterEach } from 'vitest'; +import * as fs from 'fs'; +import * as path from 'path'; +import * as os from 'os'; +import { execFileSync } from 'child_process'; +import { CodeGraph } from '../src'; +import { scanDirectoryAsync, type ScanSkipStats } from '../src/extraction'; +import { detectLanguage, isMpegTransportStream, MPEG_TS_SNIFF_BYTES } from '../src/extraction/grammars'; + +const PACKET = 188; + +/** A synthetic transport stream: `packets` × 188 bytes, 0x47 then pseudo-random payload. */ +function makeMpegTs(packets: number, seed = 1): Buffer { + const buf = Buffer.alloc(packets * PACKET); + let x = seed >>> 0; + for (let i = 0; i < buf.length; i++) { + x = (x * 1664525 + 1013904223) >>> 0; + buf[i] = i % PACKET === 0 ? 0x47 : x >>> 24; + } + // Every real stream opens with PSI tables whose pointer field is 0x00. + buf[4] = 0; + return buf; +} + +/** + * Real TypeScript engineered to put the letter `G` (0x47) at the start of each + * of its first `packets` 188-byte strides — the sync-byte pattern alone. With + * `nul`, one raw NUL sits inside a comment (the review's counterexample on + * #1915). It must stay TypeScript either way. + */ +function makeGammaSource(packets = 4, nul = false): string { + let text = ''; + for (let i = 0; i < packets; i++) { + const line = `Gamma${i}();` + (nul && i === 1 ? ' // \u0000' : ''); + const pad = PACKET - line.length - 1; + text += line + ' '.repeat(pad) + '\n'; + } + for (let i = 0; i < packets; i++) text += `export function Gamma${i}() { return ${i}; }\n`; + text += 'export function realFn() { return 1; }\n'; + for (let off = 0; off < packets * PACKET; off += PACKET) { + if (text.charCodeAt(off) !== 0x47) throw new Error(`fixture: expected G at ${off}`); + } + return text; +} + +/** A stream that opens the usual way: PAT and PMT packets stuffed with 0xFF, then payload. */ +function makePsiLedMpegTs(packets: number): Buffer { + const buf = makeMpegTs(packets, 7); + for (const start of [0, PACKET]) { + buf.fill(0xff, start + 4, start + PACKET); + buf[start + 1] = 0x40; + buf[start + 2] = start === 0 ? 0x00 : 0x10; + buf[start + 3] = 0x10; + } + return buf; +} + +const tempDirs: string[] = []; +function createProject(): string { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'codegraph-mpegts-')); + tempDirs.push(dir); + return dir; +} + +afterEach(() => { + for (const d of tempDirs.splice(0)) fs.rmSync(d, { recursive: true, force: true }); +}); + +describe('isMpegTransportStream', () => { + it('recognises a transport stream from its head', () => { + const ts = makeMpegTs(40); + expect(isMpegTransportStream(ts.subarray(0, MPEG_TS_SNIFF_BYTES))).toBe(true); + expect(isMpegTransportStream(ts)).toBe(true); + }); + + it('needs sixteen aligned sync bytes — a head too short, or one packet off, is not video', () => { + const ts = makeMpegTs(40); + expect(isMpegTransportStream(ts.subarray(0, 15 * PACKET))).toBe(false); + for (const packet of [2, 15]) { + const broken = Buffer.from(ts); + broken[packet * PACKET] = 0x48; + expect(isMpegTransportStream(broken)).toBe(false); + } + expect(isMpegTransportStream(Buffer.alloc(0))).toBe(false); + }); + + it('recognises a stream that opens with 0xFF-stuffed PAT and PMT packets', () => { + expect(isMpegTransportStream(makePsiLedMpegTs(40).subarray(0, MPEG_TS_SNIFF_BYTES))).toBe(true); + }); + + it('does not take source text with G at every 188th byte for video', () => { + const bytes = Buffer.from(makeGammaSource(), 'utf-8'); + expect(bytes[0]).toBe(0x47); + expect(bytes[3 * PACKET]).toBe(0x47); + expect(isMpegTransportStream(bytes)).toBe(false); + expect(detectLanguage('gamma.ts', makeGammaSource())).toBe('typescript'); + }); + + it('does not take source with G at every stride and a NUL in a comment for video (#1915 review)', () => { + for (const packets of [4, 16, 20]) { + const bytes = Buffer.from(makeGammaSource(packets, true), 'utf-8'); + expect(bytes.includes(0)).toBe(true); + expect(isMpegTransportStream(bytes.subarray(0, MPEG_TS_SNIFF_BYTES))).toBe(false); + } + }); +}); + +describe('MPEG-TS video named .ts is skipped, real TypeScript is indexed (#1910)', () => { + it('drops the clip at discovery: no file record, no nodes, no unsupported-language report', async () => { + const dir = createProject(); + fs.mkdirSync(path.join(dir, 'testdata')); + fs.writeFileSync(path.join(dir, 'testdata', 'clip.ts'), makeMpegTs(40)); + fs.writeFileSync(path.join(dir, 'app.ts'), 'export function greet(n: string) { return `hi ${n}`; }\n'); + fs.writeFileSync(path.join(dir, 'gamma.ts'), makeGammaSource()); + fs.writeFileSync(path.join(dir, 'gamma-nul.ts'), makeGammaSource(16, true)); + + const stats: ScanSkipStats = { unsupportedByExtension: new Map() }; + const scanned = await scanDirectoryAsync(dir, undefined, stats); + expect(scanned.sort()).toEqual(['app.ts', 'gamma-nul.ts', 'gamma.ts']); + expect(stats.unsupportedByExtension.size).toBe(0); + + const cg = await CodeGraph.init(dir, { index: true }); + try { + const files = cg.getFiles().map((f) => f.path).sort(); + expect(files).toEqual(['app.ts', 'gamma-nul.ts', 'gamma.ts']); + expect(cg.searchNodes('greet').some((r) => r.node.name === 'greet')).toBe(true); + expect(cg.searchNodes('Gamma2').some((r) => r.node.name === 'Gamma2')).toBe(true); + // The review's counterexample: G at every stride and a NUL in a comment. + expect(cg.searchNodes('realFn').filter((r) => r.node.filePath === 'gamma-nul.ts')).toHaveLength(1); + + // A named re-sync (the watcher / `sync` path hands files in by name) must + // not let the clip back in either. + await cg.sync({ paths: ['testdata/clip.ts', 'app.ts'] }); + expect(cg.getFiles().map((f) => f.path).sort()).toEqual(['app.ts', 'gamma-nul.ts', 'gamma.ts']); + } finally { + await cg.close(); + } + }); + + it('reports nothing skipped for a project of only source and video', async () => { + const dir = createProject(); + fs.writeFileSync(path.join(dir, 'clip.ts'), makeMpegTs(40)); + fs.writeFileSync(path.join(dir, 'app.ts'), 'export const a = 1;\n'); + const cg = await CodeGraph.init(dir); + try { + const result = await cg.indexAll(); + expect(result.filesSkippedUnsupported).toBeUndefined(); + expect(result.topUnsupportedExtensions).toBeUndefined(); + expect(result.errors.filter((e) => e.filePath === 'clip.ts')).toEqual([]); + expect(cg.getFiles().map((f) => f.path)).toEqual(['app.ts']); + } finally { + await cg.close(); + } + }); + + it('indexes past a 900 KB clip in well under two seconds', async () => { + const dir = createProject(); + fs.mkdirSync(path.join(dir, 'testdata')); + fs.writeFileSync(path.join(dir, 'testdata', 'golden.ts'), makeMpegTs(Math.ceil((900 * 1024) / PACKET))); + fs.writeFileSync(path.join(dir, 'app.ts'), 'export function greet(n: string) { return `hi ${n}`; }\n'); + const cg = await CodeGraph.init(dir); + try { + const t0 = Date.now(); + const result = await cg.indexAll(); + const elapsed = Date.now() - t0; + expect(result.filesIndexed).toBe(1); + expect(cg.getFiles().map((f) => f.path)).toEqual(['app.ts']); + expect(elapsed).toBeLessThan(2000); + } finally { + await cg.close(); + } + }); +}); + +describe('a video .ts never stays pending (#1910)', () => { + const dirs: string[] = []; + afterEach(() => { for (const d of dirs.splice(0)) fs.rmSync(d, { recursive: true, force: true }); }); + const gitProject = (): string => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'codegraph-mpegts-git-')); + dirs.push(dir); + const git = (...a: string[]) => execFileSync('git', a, { cwd: dir, stdio: 'pipe' }); + git('init', '-q'); git('config', 'user.email', 't@t'); git('config', 'user.name', 't'); + fs.writeFileSync(path.join(dir, 'app.ts'), 'export const a = 1;\n'); + git('add', '.'); git('commit', '-qm', 'init'); + return dir; + }; + + it('an untracked clip is not reported as added, before or after sync', async () => { + const dir = gitProject(); + const cg = await CodeGraph.init(dir, { index: true }); + try { + fs.writeFileSync(path.join(dir, 'clip.ts'), makeMpegTs(40)); + expect(cg.getChangedFiles()).toEqual({ added: [], modified: [], removed: [] }); + await cg.sync(); + expect(cg.getChangedFiles()).toEqual({ added: [], modified: [], removed: [] }); + expect(cg.getFiles().map((f) => f.path)).toEqual(['app.ts']); + } finally { + cg.close(); + } + }); + + it('a tracked TypeScript file that becomes a clip is removed, on the git path and on a scoped sync', async () => { + for (const scoped of [false, true]) { + const dir = gitProject(); + fs.writeFileSync(path.join(dir, 'clip.ts'), 'export const clip = 1;\n'); + const cg = await CodeGraph.init(dir, { index: true }); + try { + expect(cg.getNodesInFile('clip.ts').some((n) => n.name === 'clip')).toBe(true); + fs.writeFileSync(path.join(dir, 'clip.ts'), makeMpegTs(40)); + if (!scoped) expect(cg.getChangedFiles()).toEqual({ added: [], modified: [], removed: ['clip.ts'] }); + await (scoped ? cg.sync({ paths: ['clip.ts'] }) : cg.sync()); + expect(cg.getFiles().map((f) => f.path)).toEqual(['app.ts']); + expect(cg.getNodesInFile('clip.ts')).toEqual([]); + expect(cg.getChangedFiles()).toEqual({ added: [], modified: [], removed: [] }); + } finally { + cg.close(); + } + } + }); +}); diff --git a/__tests__/oversize-file-not-read.test.ts b/__tests__/oversize-file-not-read.test.ts new file mode 100644 index 0000000000..f32267f73f --- /dev/null +++ b/__tests__/oversize-file-not-read.test.ts @@ -0,0 +1,117 @@ +import { afterEach, describe, expect, it } from 'vitest'; +import * as fs from 'fs'; +import * as os from 'os'; +import * as path from 'path'; +import { CodeGraph } from '../src'; +import { oversizeStamp, hashContent } from '../src/extraction'; +import { hasDriftedOnDisk, readFileShape } from '../src/ui-server/api/source'; +import { ToolHandler } from '../src/mcp/tools'; +import { validateAnswerFiles } from '../src/mcp/answer-freshness'; + +/** + * A file over the size limit is stored as skipped without ever being read + * (#1910): committed video/blob fixtures used to be decoded in full — and + * hashed — only to be discarded, costing their size in RSS per file. + */ +describe('oversize files are stat-gated, never read (#1910)', () => { + let dir: string; + afterEach(() => { if (dir) fs.rmSync(dir, { recursive: true, force: true }); }); + + const big = (bytes: number, fill = 0x41) => Buffer.alloc(bytes, fill); + + it('indexes the neighbours, records the oversize file as skipped with a size-stamp hash, and does not decode it', async () => { + dir = fs.mkdtempSync(path.join(os.tmpdir(), 'codegraph-oversize-')); + fs.writeFileSync(path.join(dir, 'app.ts'), 'export function alpha() { return beta(); }\nexport function beta() { return 1; }\n'); + // 1 MB + 1: over the limit; invalid UTF-8 (0xFF) so any decode would be visible as replacement chars. + fs.writeFileSync(path.join(dir, 'blob.ts'), big(1024 * 1024 + 1, 0xff)); + const cg = await CodeGraph.init(dir, { index: true }); + try { + expect(cg.getNodesByKind('function').map(n => n.name).sort()).toEqual(['alpha', 'beta']); + const rec = cg.getFiles().find(f => f.path === 'blob.ts'); + expect(rec).toBeDefined(); + expect(rec!.contentHash).toBe(hashContent(oversizeStamp(1024 * 1024 + 1))); + // Nothing from the blob reached the graph. + expect(cg.getNodesInFile('blob.ts').filter(n => n.kind !== 'file')).toEqual([]); + // Change detection agrees with what was stored: nothing pending. + expect(cg.getChangedFiles()).toEqual({ added: [], modified: [], removed: [] }); + // A same-size rewrite is not a change (nothing about it is indexed)... + fs.writeFileSync(path.join(dir, 'blob.ts'), big(1024 * 1024 + 1, 0xfe)); + expect(cg.getChangedFiles().modified).toEqual([]); + // ...crossing the limit is: the file becomes ordinary source. + fs.writeFileSync(path.join(dir, 'blob.ts'), 'export const gamma = 3;\n'); + expect(cg.getChangedFiles().modified).toEqual(['blob.ts']); + await cg.sync(); + expect(cg.getNodesInFile('blob.ts').some(n => n.name === 'gamma')).toBe(true); + // ...and growing back over it is a change too, stored as the stamp again. + fs.writeFileSync(path.join(dir, 'blob.ts'), big(2 * 1024 * 1024, 0xff)); + expect(cg.getChangedFiles().modified).toEqual(['blob.ts']); + await cg.sync(); + expect(cg.getFiles().find(f => f.path === 'blob.ts')!.contentHash).toBe(hashContent(oversizeStamp(2 * 1024 * 1024))); + expect(cg.getNodesInFile('blob.ts').some(n => n.name === 'gamma')).toBe(false); + } finally { + cg.close(); + } + }); + + it('a 400 MB sparse fixture indexes in well under a second and without growing the heap by its size', async () => { + dir = fs.mkdtempSync(path.join(os.tmpdir(), 'codegraph-oversize-big-')); + fs.writeFileSync(path.join(dir, 'ok.ts'), 'export const one = 1;\n'); + // Sparse: occupies no disk, but stat() reports 400 MB — a read would decode all of it. + const fd = fs.openSync(path.join(dir, 'huge.ts'), 'w'); + fs.ftruncateSync(fd, 400 * 1024 * 1024); + fs.closeSync(fd); + // Warm the engine on a sibling project first, so the grammar and worker + // start-up cost is not mistaken for the file being read. + const warm = fs.mkdtempSync(path.join(os.tmpdir(), 'codegraph-oversize-warm-')); + fs.writeFileSync(path.join(warm, 'w.ts'), 'export const w = 1;\n'); + (await CodeGraph.init(warm, { index: true })).close(); + fs.rmSync(warm, { recursive: true, force: true }); + const before = process.memoryUsage().rss; + const t0 = Date.now(); + const cg = await CodeGraph.init(dir, { index: true }); + try { + expect(Date.now() - t0).toBeLessThan(5000); + // Reading 400 MB would show as at least that much RSS; the stamp shows as none. + expect(process.memoryUsage().rss - before).toBeLessThan(100 * 1024 * 1024); + expect(cg.getFiles().map(f => f.path).sort()).toEqual(['huge.ts', 'ok.ts']); + expect(cg.getChangedFiles()).toEqual({ added: [], modified: [], removed: [] }); + } finally { + cg.close(); + } + }); +}); + +describe('an unchanged file over the size limit is not reported as drifted (#1910, #1915 review)', () => { + let dir: string; + afterEach(() => { if (dir) fs.rmSync(dir, { recursive: true, force: true }); }); + + it('reads as current in the viewer and in MCP until its size changes', async () => { + dir = fs.mkdtempSync(path.join(os.tmpdir(), 'codegraph-oversize-drift-')); + fs.writeFileSync(path.join(dir, 'app.ts'), 'export function alpha() { return 1; }\n'); + // 1.4 MB of ordinary text: over the index limit, under the viewer's 8 MB read cap. + const line = 'const x = 1;\n'; + fs.writeFileSync(path.join(dir, 'big.js'), line.repeat(Math.ceil((1.4 * 1024 * 1024) / line.length))); + const cg = await CodeGraph.init(dir, { index: true }); + try { + const record = cg.getFiles().find((f) => f.path === 'big.js')!; + expect(record).toBeDefined(); + // Touch it: the stat fast path no longer answers, so the hash has to. + const later = new Date(Date.now() + 5000); + fs.utimesSync(path.join(dir, 'big.js'), later, later); + + expect(readFileShape(dir, 'big.js', record).drift).toBe(false); + expect(hasDriftedOnDisk(dir, 'big.js', record)).toBe(false); + expect((new ToolHandler(cg) as any).isFileStaleOnDisk(cg, 'big.js')).toBe(false); + const answer = [{ path: 'big.js', contentHash: record.contentHash }]; + expect((await validateAnswerFiles(dir, answer)).stale).toEqual([]); + + // A different size is a different stamp, and that is drift. + fs.appendFileSync(path.join(dir, 'big.js'), line); + expect(readFileShape(dir, 'big.js', record).drift).toBe(true); + expect(hasDriftedOnDisk(dir, 'big.js', record)).toBe(true); + expect((await validateAnswerFiles(dir, answer)).stale).toEqual(['big.js']); + } finally { + cg.close(); + } + }); +}); diff --git a/src/extraction/grammars.ts b/src/extraction/grammars.ts index c7710f2007..84596d81c5 100644 --- a/src/extraction/grammars.ts +++ b/src/extraction/grammars.ts @@ -172,6 +172,67 @@ export const EXTENSION_MAP: Record = { '.tofu': 'terraform', }; +/** MPEG transport stream: fixed 188-byte packets, each opening with 0x47. */ +const MPEG_TS_PACKET_SIZE = 188; +const MPEG_TS_SYNC_BYTE = 0x47; +/** + * Consecutive packets whose sync byte must line up before a file counts as + * video — 3 KB of head. A stream shorter than that is cheap to parse anyway; + * the cost #1910 is about comes from clips hundreds of KB long. + */ +const MPEG_TS_MIN_PACKETS = 16; +/** + * Share of the head that must be control bytes (below 0x20, other than the + * whitespace ones) for it to count as binary. Compressed audio and video put + * about one byte in eight there; source text puts none. + */ +const MPEG_TS_MIN_CONTROL_SHARE = 1 / 64; +/** + * How many bytes of a file's head `isMpegTransportStream` needs — enough to + * see `MPEG_TS_MIN_PACKETS` sync bytes plus the packets between them. + */ +export const MPEG_TS_SNIFF_BYTES = MPEG_TS_PACKET_SIZE * MPEG_TS_MIN_PACKETS; + +/** + * Whether these leading bytes are an MPEG transport stream — the OTHER thing a + * `.ts` file can be. Golden video fixtures (`testdata/*.ts`, e2e clips) share + * TypeScript's extension, and tree-sitter takes ~28 s to chew through a 900 KB + * clip for zero symbols (#1910), so the decision has to be made from the head + * of the file, before any parse. + * + * Two conditions, both required: + * 1. the sync byte 0x47 sits at every 188-byte packet boundary of the first + * `MPEG_TS_MIN_PACKETS` packets — every packet of a transport stream + * opens with it, and nothing else pads to 188; + * 2. the head is binary: at least `MPEG_TS_MIN_CONTROL_SHARE` of it is + * control bytes, as any compressed payload is. + * 0x47 is the letter `G`, so (1) alone could match source whose lines happen + * to put a `G` at every 188-byte stride. Checking for a single NUL was not + * enough to close that: one NUL in a comment is still TypeScript. (2) asks for + * dozens of control bytes, which no source file carries. + * + * `head` is the first `MPEG_TS_SNIFF_BYTES` (or fewer) bytes of the file. + */ +export function isMpegTransportStream(head: Uint8Array): boolean { + const lastSync = MPEG_TS_PACKET_SIZE * (MPEG_TS_MIN_PACKETS - 1); + if (head.length <= lastSync) return false; + for (let off = 0; off <= lastSync; off += MPEG_TS_PACKET_SIZE) { + if (head[off] !== MPEG_TS_SYNC_BYTE) return false; + } + let control = 0; + for (let i = 0; i < head.length; i++) { + const b = head[i]!; + // Tab, newline, vertical tab, form feed and carriage return are text. + if (b < 0x20 && (b < 0x09 || b > 0x0d)) control++; + } + return control >= head.length * MPEG_TS_MIN_CONTROL_SHARE; +} + +/** Whether `filePath` carries the one extension MPEG-TS shares with a language. */ +export function hasMpegTsExtension(filePath: string): boolean { + return filePath.length > 3 && filePath.slice(-3).toLowerCase() === '.ts'; +} + /** * Whether a file is one CodeGraph can parse, based purely on its extension. * This is the single source of truth for "should we index this file" — derived diff --git a/src/extraction/index.ts b/src/extraction/index.ts index 399f333acf..1906c83adf 100644 --- a/src/extraction/index.ts +++ b/src/extraction/index.ts @@ -26,7 +26,7 @@ import { ParseWorkerPool, resolveParsePoolSize, resolveParseTimeoutMs } from './ import { StoreWriter, StoreBundle, finalizeStoreBundle } from './store-writer'; import { materializeKernelResult } from './kernel'; import { detectGeneratedFile } from './generated-detection'; -import { detectLanguage, isSourceFile, isLanguageSupported, isFileLevelOnlyLanguage, initGrammars, loadGrammarsForLanguages, readGrammarWasmBytes } from './grammars'; +import { detectLanguage, isSourceFile, isLanguageSupported, isFileLevelOnlyLanguage, initGrammars, loadGrammarsForLanguages, readGrammarWasmBytes, isMpegTransportStream, hasMpegTsExtension, MPEG_TS_SNIFF_BYTES } from './grammars'; import { loadExtensionOverrides, loadIncludeIgnoredPatterns, loadExcludePatterns, loadIncludePatterns, PROJECT_CONFIG_FILENAME } from '../project-config'; import { isCodeGraphDataDir } from '../directory'; import { logDebug, logWarn } from '../errors'; @@ -35,7 +35,8 @@ import ignore, { Ignore } from 'ignore'; import { detectFrameworks } from '../resolution/frameworks'; import type { ResolutionContext } from '../resolution/types'; import { createYielder, type MaybeYield } from '../resolution/cooperative-yield'; -import { MAX_SOURCE_FILE_SIZE_BYTES } from '../file-limits'; +import { MAX_SOURCE_FILE_SIZE_BYTES, oversizeStamp, readBoundedSource, readBoundedSourceSync } from '../file-limits'; +export { oversizeStamp }; /** * Number of files to read in parallel during indexing. @@ -158,6 +159,16 @@ export function hashContent(content: string): string { return crypto.createHash('sha256').update(content).digest('hex'); } + +/** + * What change detection hashes for a file: its text when it is under the size + * limit, the size stamp when it is over — an oversize file is never decoded. + */ +function readSourceOrStamp(fullPath: string): string { + const { stats, bytes } = readBoundedSourceSync(fullPath); + return bytes === null ? oversizeStamp(stats.size) : bytes.toString('utf8'); +} + /** * Directory names that are dependency, build, cache, or tooling output across the * languages/frameworks CodeGraph supports — curated from the canonical @@ -566,6 +577,7 @@ function collectIncludedFiles( if (!include.ignores(rel)) return; if (exclude && exclude.ignores(rel)) return; if (!isSourceFile(rel, overrides)) return; + if (isMpegTsVideoFile(rootDir, rel)) return; out.add(rel); } }; @@ -1504,6 +1516,7 @@ export function scanDirectory( let count = 0; for (const filePath of gitFiles) { if (isSourceFile(filePath, overrides)) { + if (isMpegTsVideoFile(rootDir, filePath)) continue; files.push(filePath); count++; onProgress?.(count, filePath); @@ -1520,6 +1533,31 @@ export function scanDirectory( * Async variant of scanDirectory that yields to the event loop periodically, * allowing worker threads to receive and render progress messages. */ +/** + * Whether a `.ts` file on disk is an MPEG transport stream rather than + * TypeScript (#1910). Reads only the file's head — `MPEG_TS_SNIFF_BYTES`, under + * 1 KB — so the check costs one small read per `.ts` file at discovery, never a + * whole-file read; any file the extension does not make ambiguous is not + * touched at all. A file that cannot be read is left to the indexing path, + * which reports the read error itself. + */ +function isMpegTsVideoFile(rootDir: string, relativePath: string): boolean { + if (!hasMpegTsExtension(relativePath)) return false; + let fd: number | null = null; + try { + fd = fs.openSync(path.join(rootDir, relativePath), 'r'); + const head = Buffer.allocUnsafe(MPEG_TS_SNIFF_BYTES); + const n = fs.readSync(fd, head, 0, head.length, 0); + if (!isMpegTransportStream(head.subarray(0, n))) return false; + } catch { + return false; + } finally { + if (fd !== null) fs.closeSync(fd); + } + logDebug('Skipping MPEG transport stream named .ts — not TypeScript', { filePath: relativePath }); + return true; +} + /** * What a scan saw but could not index, tallied by extension. * @@ -1554,6 +1592,7 @@ export async function scanDirectoryAsync( let count = 0; for (const filePath of gitFiles) { if (isSourceFile(filePath, overrides)) { + if (isMpegTsVideoFile(rootDir, filePath)) continue; files.push(filePath); count++; onProgress?.(count, filePath); @@ -1662,6 +1701,7 @@ function scanDirectoryWalk( } else if (stat.isFile()) { if (!isIgnored(fullPath, false, active)) { if (isSourceFile(relativePath, overrides)) { + if (isMpegTsVideoFile(rootDir, relativePath)) continue; files.push(relativePath); count++; onProgress?.(count, relativePath); @@ -1683,6 +1723,7 @@ function scanDirectoryWalk( } else if (entry.isFile()) { if (!isIgnored(fullPath, false, active)) { if (isSourceFile(relativePath, overrides)) { + if (isMpegTsVideoFile(rootDir, relativePath)) continue; files.push(relativePath); count++; onProgress?.(count, relativePath); @@ -1846,7 +1887,9 @@ export class ExtractionOrchestrator { const full = validatePathWithinRoot(rootDir, relativePath); if (!full) return null; try { - return fs.readFileSync(full, 'utf-8'); + // Framework detectors scan source by name; a file over the size + // limit was never indexed and must not be decoded here either (#1910). + return readBoundedSourceSync(full).bytes?.toString('utf8') ?? null; } catch { return null; } @@ -2263,8 +2306,21 @@ export class ExtractionOrchestrator { logWarn('Path traversal blocked in batch reader', { filePath: fp }); return { filePath: fp, content: null as string | null, stats: null as fs.Stats | null, error: new Error('Path traversal blocked') }; } - const content = await fsp.readFile(fullPath, 'utf-8'); - const stats = await fsp.stat(fullPath); + // Stat first: a file over the size limit is stored as skipped + // without ever being read or decoded (#1910), so ten oversize + // fixtures in one I/O batch no longer cost their size in RSS. + const { stats, bytes } = await readBoundedSource(fullPath); + if (bytes === null) { + return { filePath: fp, content: oversizeStamp(stats.size), stats, error: null as Error | null }; + } + // Read bytes, not text: a `.ts` that is really an MPEG transport + // stream (#1910) is recognised from its head here, at no extra I/O, + // and never decoded or parsed. + if (hasMpegTsExtension(fp) && isMpegTransportStream(bytes.subarray(0, MPEG_TS_SNIFF_BYTES))) { + logDebug('Skipping MPEG transport stream named .ts — not TypeScript', { filePath: fp }); + return { filePath: fp, content: null as string | null, stats: null as fs.Stats | null, error: null as Error | null, skipped: true }; + } + const content = bytes.toString('utf-8'); return { filePath: fp, content, stats, error: null as Error | null }; } catch (err) { return { filePath: fp, content: null as string | null, stats: null as fs.Stats | null, error: err as Error }; @@ -2274,9 +2330,16 @@ export class ExtractionOrchestrator { // Dispatch each readable file into the bounded parse window; the window // stores results on the main thread as they arrive. - for (const { filePath, content, stats, error } of fileContents) { + for (const { filePath, content, stats, error, skipped } of fileContents) { if (signal?.aborted) { aborted = true; break; } + if (skipped) { + // Not a source file after all — counted as done, stored as nothing. + processed++; + onProgress?.({ phase: 'parsing', current: processed, total }); + continue; + } + if (error || content === null || stats === null) { processed++; filesErrored++; @@ -2404,7 +2467,10 @@ export class ExtractionOrchestrator { try { const fullPath = validatePathWithinRoot(this.rootDir, filePath); if (!fullPath) continue; - content = await fsp.readFile(fullPath, 'utf-8'); + // Bounded like the first read: the file may have grown since (#1910). + const bytes = (await readBoundedSource(fullPath)).bytes; + if (bytes === null) continue; + content = bytes.toString('utf8'); } catch { continue; } @@ -2456,7 +2522,9 @@ export class ExtractionOrchestrator { try { const fullPath = validatePathWithinRoot(this.rootDir, filePath); if (!fullPath) continue; - fullContent = await fsp.readFile(fullPath, 'utf-8'); + const bytes = (await readBoundedSource(fullPath)).bytes; + if (bytes === null) continue; + fullContent = bytes.toString('utf8'); } catch { continue; } @@ -2586,8 +2654,10 @@ export class ExtractionOrchestrator { let content: string; let stats: fs.Stats; try { - stats = await fsp.stat(fullPath); - content = await fsp.readFile(fullPath, 'utf-8'); + // An oversize file is stored as skipped; its bytes are never needed (#1910). + const read = await readBoundedSource(fullPath); + stats = read.stats; + content = read.bytes === null ? oversizeStamp(stats.size) : read.bytes.toString('utf8'); } catch (error) { return { nodes: [], @@ -2630,6 +2700,12 @@ export class ExtractionOrchestrator { }; } + // An MPEG transport stream named `.ts` is not TypeScript (#1910): one + // sub-KB head read decides it, before the language lookup and the parse. + if (isMpegTsVideoFile(this.rootDir, relativePath)) { + return { nodes: [], edges: [], unresolvedReferences: [], errors: [], durationMs: 0 }; + } + const language = detectLanguage(relativePath, content, loadExtensionOverrides(this.rootDir)); // Check file size @@ -3101,7 +3177,10 @@ export class ExtractionOrchestrator { (p) => isSourceFile(p, overrides) && !scope.ignores(p) && - fs.existsSync(path.join(this.rootDir, p)) + fs.existsSync(path.join(this.rootDir, p)) && + // Same rule as the scan (#1910): a reported video clip is not a + // source file, so a tracked one is removed and a new one ignored. + !isMpegTsVideoFile(this.rootDir, p) ); trackedFiles = []; for (const p of unique) { @@ -3199,9 +3278,10 @@ export class ExtractionOrchestrator { } // New, or size/mtime changed — read + hash to confirm a real content change. + // (An oversize file hashes as its size stamp, unread — #1910.) let content: string; try { - content = fs.readFileSync(fullPath, 'utf-8'); + content = readSourceOrStamp(fullPath); } catch (error) { logDebug('Skipping unreadable file during sync', { filePath, error: String(error) }); failedFilePaths.push(filePath); @@ -3359,12 +3439,16 @@ export class ExtractionOrchestrator { for (const filePath of candidates) { const tracked = this.queries.getFileByPath(filePath); const fullPath = path.join(this.rootDir, filePath); - if (!isSourceFile(filePath, overrides) || scope.ignores(filePath) || !fs.existsSync(fullPath)) { + // A `.ts` that is an MPEG transport stream is not source (#1910): the + // scan never lists it, so git must not report it as pending either — + // an untracked clip would otherwise stay "added" after every sync. + if (!isSourceFile(filePath, overrides) || scope.ignores(filePath) || !fs.existsSync(fullPath) + || isMpegTsVideoFile(this.rootDir, filePath)) { if (tracked) removed.push(filePath); continue; } let content: string; - try { content = fs.readFileSync(fullPath, 'utf-8'); } + try { content = readSourceOrStamp(fullPath); } catch (error) { logDebug('Skipping unreadable file while detecting changes', { filePath, error: String(error) }); continue; @@ -3402,7 +3486,7 @@ export class ExtractionOrchestrator { const fullPath = path.join(this.rootDir, filePath); let content: string; try { - content = fs.readFileSync(fullPath, 'utf-8'); + content = readSourceOrStamp(fullPath); } catch (error) { logDebug('Skipping unreadable file while detecting changes', { filePath, error: String(error) }); continue; diff --git a/src/file-limits.ts b/src/file-limits.ts index 4bb4d6da1a..276d7f2021 100644 --- a/src/file-limits.ts +++ b/src/file-limits.ts @@ -1,6 +1,110 @@ +import * as fs from 'fs'; +import * as fsp from 'fs/promises'; + /** * Largest source file CodeGraph will parse or read during resolution. Generated * bundles, minified sources, and dependency archives above this limit provide no * useful symbols; 1 MB covers essentially all hand-written source. */ export const MAX_SOURCE_FILE_SIZE_BYTES = 1024 * 1024; + +/** + * What stands in for the content of a file over MAX_SOURCE_FILE_SIZE_BYTES. Such a file is + * never parsed, so its bytes are never needed — reading them only to hash and + * discard cost multi-GB RSS spikes on committed video/blob fixtures and could + * fail outright with `Invalid string length` (#1910). The stamp is a function + * of size alone: change detection compares it to the stored hash, so a + * same-size rewrite of an oversize file is not a change (nothing about it is + * indexed), while crossing the limit in either direction is. + */ +export function oversizeStamp(size: number): string { + return `codegraph:oversize:${size}`; +} + +/** + * The text whose hash the index stores for a file of `size` bytes: its content + * within the limit, the size stamp over it. Anything that checks a file on disk + * against its stored `contentHash` has to hash this, or every unchanged file + * over the limit reads as drifted. `content` is only read within the limit. + */ +export function indexedHashInput(size: number, content: () => string): string { + return size > MAX_SOURCE_FILE_SIZE_BYTES ? oversizeStamp(size) : content(); +} + +/** A source file's stats, and its bytes when they are within the limit (null = oversize). */ +export interface BoundedSource { + stats: fs.Stats; + bytes: Buffer | null; +} + +const READ_CHUNK_BYTES = 64 * 1024; + +function assertRegularFile(stats: fs.Stats): void { + if (!stats.isFile()) throw new Error('Source path is not a regular file'); +} + +/** + * Read a source file without ever holding more than the limit plus one byte. + * A stat before the read is not enough: the file can grow between the stat + * and the read (a log, a download, a build output being written), so the + * descriptor is re-checked after opening and the read itself stops one byte + * past the limit. Returns `bytes: null` for an oversize file. + */ +export async function readBoundedSource(file: string): Promise { + const initial = await fsp.stat(file); + assertRegularFile(initial); + if (initial.size > MAX_SOURCE_FILE_SIZE_BYTES) return { stats: initial, bytes: null }; + const handle = await fsp.open(file, 'r'); + try { + let stats = await handle.stat(); + assertRegularFile(stats); + if (stats.size > MAX_SOURCE_FILE_SIZE_BYTES) return { stats, bytes: null }; + const chunks: Buffer[] = []; + let size = 0; + while (size <= MAX_SOURCE_FILE_SIZE_BYTES) { + const chunk = Buffer.allocUnsafe(Math.min(READ_CHUNK_BYTES, MAX_SOURCE_FILE_SIZE_BYTES + 1 - size)); + const { bytesRead } = await handle.read(chunk, 0, chunk.length, size); + if (!bytesRead) break; + chunks.push(chunk.subarray(0, bytesRead)); + size += bytesRead; + } + stats = await handle.stat(); + if (size > MAX_SOURCE_FILE_SIZE_BYTES || stats.size > MAX_SOURCE_FILE_SIZE_BYTES) { + stats.size = Math.max(size, stats.size); + return { stats, bytes: null }; + } + return { stats, bytes: Buffer.concat(chunks, size) }; + } finally { + await handle.close(); + } +} + +/** Synchronous {@link readBoundedSource}. */ +export function readBoundedSourceSync(file: string): BoundedSource { + const initial = fs.statSync(file); + assertRegularFile(initial); + if (initial.size > MAX_SOURCE_FILE_SIZE_BYTES) return { stats: initial, bytes: null }; + const fd = fs.openSync(file, 'r'); + try { + let stats = fs.fstatSync(fd); + assertRegularFile(stats); + if (stats.size > MAX_SOURCE_FILE_SIZE_BYTES) return { stats, bytes: null }; + const chunks: Buffer[] = []; + let size = 0; + while (size <= MAX_SOURCE_FILE_SIZE_BYTES) { + const chunk = Buffer.allocUnsafe(Math.min(READ_CHUNK_BYTES, MAX_SOURCE_FILE_SIZE_BYTES + 1 - size)); + const bytesRead = fs.readSync(fd, chunk, 0, chunk.length, size); + if (!bytesRead) break; + chunks.push(chunk.subarray(0, bytesRead)); + size += bytesRead; + } + stats = fs.fstatSync(fd); + if (size > MAX_SOURCE_FILE_SIZE_BYTES || stats.size > MAX_SOURCE_FILE_SIZE_BYTES) { + stats.size = Math.max(size, stats.size); + return { stats, bytes: null }; + } + return { stats, bytes: Buffer.concat(chunks, size) }; + } finally { + fs.closeSync(fd); + } +} diff --git a/src/mcp/answer-freshness.ts b/src/mcp/answer-freshness.ts index 297eab4728..5f5e224f59 100644 --- a/src/mcp/answer-freshness.ts +++ b/src/mcp/answer-freshness.ts @@ -1,5 +1,7 @@ import { createHash } from 'crypto'; import { createReadStream } from 'fs'; +import { stat } from 'fs/promises'; +import { MAX_SOURCE_FILE_SIZE_BYTES, oversizeStamp } from '../file-limits'; import { validatePathWithinRoot } from '../utils'; export interface AnswerFile { @@ -35,6 +37,14 @@ export async function validateAnswerFiles(root: string, files: AnswerFile[]): Pr const hash = createHash('sha256'); const absolute = validatePathWithinRoot(root, file.path); if (!absolute) { stale.push(file.path); continue; } + // A file over the index's size limit is stored as its size stamp + // (#1910): when the stamp still matches, the file is current without + // being read. Anything else falls through to the bounded read below. + const { size } = await stat(absolute); + if (size > MAX_SOURCE_FILE_SIZE_BYTES && + createHash('sha256').update(oversizeStamp(size)).digest('hex') === file.contentHash) { + continue; + } const stream = createReadStream(absolute, { encoding: 'utf8', highWaterMark: 64 * 1024, signal, }); diff --git a/src/mcp/tools.ts b/src/mcp/tools.ts index 5c99378c85..5deb9bd86c 100644 --- a/src/mcp/tools.ts +++ b/src/mcp/tools.ts @@ -119,6 +119,7 @@ function wslSharedIndexGuidance(err: WslSharedIndexError): string { */ export { PathRefusalError } from '../errors'; import { PathRefusalError } from '../errors'; +import { indexedHashInput } from '../file-limits'; import { resolve as resolvePath, relative as relativePath } from 'path'; /** Maximum output length to prevent context bloat (characters) */ @@ -2500,7 +2501,9 @@ export class ToolHandler { // Same freshness test as the sync fast path (extraction/index.ts): // equal size + equal floored mtime ⇒ unchanged, no read needed. if (st.size !== rec.size || Math.floor(st.mtimeMs) !== Math.floor(rec.modifiedAt)) { - const data = content ?? readFileSync(absPath, 'utf-8'); + // A file over the index's size limit is stored as its size stamp + // (#1910), so it is compared as one, without reading it. + const data = indexedHashInput(st.size, () => content ?? readFileSync(absPath, 'utf-8')); // Must stay byte-identical to extraction's `hashContent` (sha256 over // the utf-8 string) — the identical-rewrite test in // mcp-stale-slice.test.ts pins the parity. Inlined (not imported) diff --git a/src/ui-server/api/source.ts b/src/ui-server/api/source.ts index d2f32069bd..a81ce031e2 100644 --- a/src/ui-server/api/source.ts +++ b/src/ui-server/api/source.ts @@ -39,6 +39,7 @@ import * as path from 'path'; import type { FileRecord } from '../../types'; import type { CodeGraph } from '../../index'; import { resolveProjectFile } from '../security'; +import { indexedHashInput } from '../../file-limits'; import { highlightLines, type HighlightResult } from '../highlight'; import { ApiError, badRequest, intParam, notFound, textParam } from './respond'; @@ -209,9 +210,10 @@ export function hasDriftedOnDisk( if (stats.size === record.size && Math.floor(stats.mtimeMs) === Math.floor(record.modifiedAt)) { return false; } - if (stats.size > MAX_SOURCE_BYTES) return true; - const content = fs.readFileSync(absolute, 'utf-8'); - return createHash('sha256').update(content).digest('hex') !== record.contentHash; + // A file over the index's size limit is stored as its size stamp (#1910), + // so it is compared as one — without reading it. + const hashed = indexedHashInput(stats.size, () => fs.readFileSync(absolute, 'utf-8')); + return createHash('sha256').update(hashed).digest('hex') !== record.contentHash; } catch { return false; } @@ -250,7 +252,8 @@ export function readFileShape( return { drift: false, totalLines: null, reason: 'The file is too large to read here.' }; } const content = fs.readFileSync(absolute, 'utf-8'); - const drift = createHash('sha256').update(content).digest('hex') !== record.contentHash; + const hashed = indexedHashInput(stats.size, () => content); + const drift = createHash('sha256').update(hashed).digest('hex') !== record.contentHash; return { drift, totalLines: splitLines(content).length, @@ -394,7 +397,8 @@ export async function buildSource( // Byte-identical to extraction's `hashContent` (sha256 over the utf-8 // string). A touch or a checkout that rewrote the same bytes must not count // as drift, which is exactly what hashing content rather than mtime buys. - const hash = createHash('sha256').update(content).digest('hex'); + // A file over the index's size limit is stored as its size stamp (#1910). + const hash = createHash('sha256').update(indexedHashInput(stats.size, () => content)).digest('hex'); const drift = hash !== record.contentHash; if (drift && onDrift === 'omit') { return {