From 260d095f6ff2690c712ce67ff1e52194c07c116e Mon Sep 17 00:00:00 2001 From: danusha2345 Date: Tue, 22 Sep 2026 18:01:22 +0300 Subject: [PATCH 1/9] fix(extraction): an MPEG-TS video named .ts is not TypeScript (#1910) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `.ts` mapped to TypeScript by extension alone, so MPEG transport-stream fixtures (`testdata/*.ts`) were parsed by tree-sitter: ~28 s of CPU for a 900 KB clip, minutes for a directory of them, for zero symbols. Recognise the stream from the head of the file — the 0x47 sync byte at offsets 0, 188, 376 and 564 plus a NUL byte, which every stream carries in its first packets and UTF-8 source never does — and drop it at discovery: not indexed, not parsed, not counted, not tallied as an unsupported language. The check reads under 1 KB and only for `.ts` files; the batch reader sniffs the bytes it already read, and the single-file path (sync, watcher) does the same head read, so a clip handed in by name is skipped too. The skip happens before parse dispatch, so no kernel mirror is needed. Co-Authored-By: Claude Fable 5.1 --- CHANGELOG.md | 1 + __tests__/mpeg-ts-not-typescript.test.ts | 156 +++++++++++++++++++++++ src/extraction/grammars.ts | 47 +++++++ src/extraction/index.ts | 58 ++++++++- 4 files changed, 259 insertions(+), 3 deletions(-) create mode 100644 __tests__/mpeg-ts-not-typescript.test.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index e7559656f7..d9880d7bcc 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -172,6 +172,7 @@ and adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). #### MCP / indexing - On Windows, the shared MCP daemon now waits longer for another program — an antivirus scan, an indexer, or another session reading its lock file — to let go of that file, so it starts instead of leaving the session to fall back to a slower in-process server. (#1773) +- An MPEG transport stream video that happens to be named `.ts` (golden fixtures under `testdata/`, e2e clips) is now recognised from its first bytes and skipped as non-source instead of being fed to the TypeScript parser — a 900 KB clip used to cost about 28 seconds of CPU per file for no symbols, and a folder of them minutes. Real TypeScript is never affected. (#1910) - File watching no longer drops the full re-scan a removed directory asks for when that sync fails, so the deleted files leave the index instead of lingering. (#1964) - Daemon startup and cleanup now preserve live legacy PID-only locks while still reclaiming dead or identity-disproved records, preventing two writers from serving the same project. - Incremental sync now keeps edge rebinding crash-safe: replacing a resolved edge with its recovery reference commits atomically, so an interruption cannot permanently remove the relationship. diff --git a/__tests__/mpeg-ts-not-typescript.test.ts b/__tests__/mpeg-ts-not-typescript.test.ts new file mode 100644 index 0000000000..2a6e12870b --- /dev/null +++ b/__tests__/mpeg-ts-not-typescript.test.ts @@ -0,0 +1,156 @@ +/** + * An MPEG transport stream named `.ts` is not TypeScript (#1910). + * + * Golden video fixtures (`testdata/*.ts`) share TypeScript's extension; fed to + * the tree-sitter TypeScript parser a 900 KB clip costs ~28 s of CPU for zero + * symbols. The fix recognises the stream from the head of the file (0x47 sync + * byte at each 188-byte packet boundary, plus a NUL byte no UTF-8 source has) + * and drops it at discovery — not indexed, not parsed, not counted, not + * reported as an unsupported language. + */ +import { describe, it, expect, afterEach } from 'vitest'; +import * as fs from 'fs'; +import * as path from 'path'; +import * as os from 'os'; +import { CodeGraph } from '../src'; +import { scanDirectoryAsync, type ScanSkipStats } from '../src/extraction'; +import { detectLanguage, isMpegTransportStream, MPEG_TS_SNIFF_BYTES } from '../src/extraction/grammars'; + +const PACKET = 188; + +/** A synthetic transport stream: `packets` × 188 bytes, 0x47 then pseudo-random payload. */ +function makeMpegTs(packets: number, seed = 1): Buffer { + const buf = Buffer.alloc(packets * PACKET); + let x = seed >>> 0; + for (let i = 0; i < buf.length; i++) { + x = (x * 1664525 + 1013904223) >>> 0; + buf[i] = i % PACKET === 0 ? 0x47 : x >>> 24; + } + // Every real stream opens with PSI tables whose pointer field is 0x00. + buf[4] = 0; + return buf; +} + +/** + * Real TypeScript engineered to put the letter `G` (0x47) at offsets 0, 188, + * 376 and 564 — the sync-byte pattern alone. It must stay TypeScript. + */ +function makeGammaSource(): string { + const lines: string[] = []; + let text = ''; + for (let i = 0; i < 4; i++) { + const line = `Gamma${i}();`; + const pad = PACKET - line.length - 1; + text += line + ' '.repeat(pad) + '\n'; + lines.push(line); + } + text += 'export function Gamma0() { return 0; }\n'; + text += 'export function Gamma1() { return 1; }\n'; + text += 'export function Gamma2() { return 2; }\n'; + text += 'export function Gamma3() { return 3; }\n'; + for (const off of [0, PACKET, 2 * PACKET, 3 * PACKET]) { + if (text.charCodeAt(off) !== 0x47) throw new Error(`fixture: expected G at ${off}`); + } + void lines; + return text; +} + +const tempDirs: string[] = []; +function createProject(): string { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'codegraph-mpegts-')); + tempDirs.push(dir); + return dir; +} + +afterEach(() => { + for (const d of tempDirs.splice(0)) fs.rmSync(d, { recursive: true, force: true }); +}); + +describe('isMpegTransportStream', () => { + it('recognises a transport stream from its head', () => { + const ts = makeMpegTs(40); + expect(isMpegTransportStream(ts.subarray(0, MPEG_TS_SNIFF_BYTES))).toBe(true); + expect(isMpegTransportStream(ts)).toBe(true); + }); + + it('needs four aligned sync bytes — a head too short, or one packet off, is not video', () => { + const ts = makeMpegTs(40); + expect(isMpegTransportStream(ts.subarray(0, 3 * PACKET))).toBe(false); + const broken = Buffer.from(ts); + broken[2 * PACKET] = 0x48; + expect(isMpegTransportStream(broken)).toBe(false); + expect(isMpegTransportStream(Buffer.alloc(0))).toBe(false); + }); + + it('does not take source text with G at every 188th byte for video', () => { + const bytes = Buffer.from(makeGammaSource(), 'utf-8'); + expect(bytes[0]).toBe(0x47); + expect(bytes[3 * PACKET]).toBe(0x47); + expect(isMpegTransportStream(bytes)).toBe(false); + expect(detectLanguage('gamma.ts', makeGammaSource())).toBe('typescript'); + }); +}); + +describe('MPEG-TS video named .ts is skipped, real TypeScript is indexed (#1910)', () => { + it('drops the clip at discovery: no file record, no nodes, no unsupported-language report', async () => { + const dir = createProject(); + fs.mkdirSync(path.join(dir, 'testdata')); + fs.writeFileSync(path.join(dir, 'testdata', 'clip.ts'), makeMpegTs(40)); + fs.writeFileSync(path.join(dir, 'app.ts'), 'export function greet(n: string) { return `hi ${n}`; }\n'); + fs.writeFileSync(path.join(dir, 'gamma.ts'), makeGammaSource()); + + const stats: ScanSkipStats = { unsupportedByExtension: new Map() }; + const scanned = await scanDirectoryAsync(dir, undefined, stats); + expect(scanned.sort()).toEqual(['app.ts', 'gamma.ts']); + expect(stats.unsupportedByExtension.size).toBe(0); + + const cg = await CodeGraph.init(dir, { index: true }); + try { + const files = cg.getFiles().map((f) => f.path).sort(); + expect(files).toEqual(['app.ts', 'gamma.ts']); + expect(cg.searchNodes('greet').some((r) => r.node.name === 'greet')).toBe(true); + expect(cg.searchNodes('Gamma2').some((r) => r.node.name === 'Gamma2')).toBe(true); + + // A named re-sync (the watcher / `sync` path hands files in by name) must + // not let the clip back in either. + await cg.sync({ paths: ['testdata/clip.ts', 'app.ts'] }); + expect(cg.getFiles().map((f) => f.path).sort()).toEqual(['app.ts', 'gamma.ts']); + } finally { + await cg.close(); + } + }); + + it('reports nothing skipped for a project of only source and video', async () => { + const dir = createProject(); + fs.writeFileSync(path.join(dir, 'clip.ts'), makeMpegTs(40)); + fs.writeFileSync(path.join(dir, 'app.ts'), 'export const a = 1;\n'); + const cg = await CodeGraph.init(dir); + try { + const result = await cg.indexAll(); + expect(result.filesSkippedUnsupported).toBeUndefined(); + expect(result.topUnsupportedExtensions).toBeUndefined(); + expect(result.errors.filter((e) => e.filePath === 'clip.ts')).toEqual([]); + expect(cg.getFiles().map((f) => f.path)).toEqual(['app.ts']); + } finally { + await cg.close(); + } + }); + + it('indexes past a 900 KB clip in well under two seconds', async () => { + const dir = createProject(); + fs.mkdirSync(path.join(dir, 'testdata')); + fs.writeFileSync(path.join(dir, 'testdata', 'golden.ts'), makeMpegTs(Math.ceil((900 * 1024) / PACKET))); + fs.writeFileSync(path.join(dir, 'app.ts'), 'export function greet(n: string) { return `hi ${n}`; }\n'); + const cg = await CodeGraph.init(dir); + try { + const t0 = Date.now(); + const result = await cg.indexAll(); + const elapsed = Date.now() - t0; + expect(result.filesIndexed).toBe(1); + expect(cg.getFiles().map((f) => f.path)).toEqual(['app.ts']); + expect(elapsed).toBeLessThan(2000); + } finally { + await cg.close(); + } + }); +}); diff --git a/src/extraction/grammars.ts b/src/extraction/grammars.ts index c7710f2007..d75180e4e9 100644 --- a/src/extraction/grammars.ts +++ b/src/extraction/grammars.ts @@ -172,6 +172,53 @@ export const EXTENSION_MAP: Record = { '.tofu': 'terraform', }; +/** MPEG transport stream: fixed 188-byte packets, each opening with 0x47. */ +const MPEG_TS_PACKET_SIZE = 188; +const MPEG_TS_SYNC_BYTE = 0x47; +/** Consecutive packets whose sync byte must line up before a file counts as video. */ +const MPEG_TS_MIN_PACKETS = 4; +/** + * How many bytes of a file's head `isMpegTransportStream` needs — enough to + * see `MPEG_TS_MIN_PACKETS` sync bytes plus the packets between them. + */ +export const MPEG_TS_SNIFF_BYTES = MPEG_TS_PACKET_SIZE * MPEG_TS_MIN_PACKETS; + +/** + * Whether these leading bytes are an MPEG transport stream — the OTHER thing a + * `.ts` file can be. Golden video fixtures (`testdata/*.ts`, e2e clips) share + * TypeScript's extension, and tree-sitter takes ~28 s to chew through a 900 KB + * clip for zero symbols (#1910), so the decision has to be made from the head + * of the file, before any parse. + * + * Two conditions, both required: + * 1. the sync byte 0x47 sits at offsets 0, 188, 376 and 564 — every packet + * of a transport stream opens with it, and nothing else pads to 188; + * 2. a NUL byte appears somewhere in the head — every stream carries one + * within its first packets (the PSI pointer field, table reserved bits, + * the `00 00 01` PES start codes), and UTF-8 source text never does. + * 0x47 is the letter `G`, so (1) alone could in principle match a source file + * whose lines happen to put a `G` at four 188-byte strides; (2) closes that + * door, because a text file with a NUL in it is not TypeScript either. + * + * `head` is the first `MPEG_TS_SNIFF_BYTES` (or fewer) bytes of the file. + */ +export function isMpegTransportStream(head: Uint8Array): boolean { + const lastSync = MPEG_TS_PACKET_SIZE * (MPEG_TS_MIN_PACKETS - 1); + if (head.length <= lastSync) return false; + for (let off = 0; off <= lastSync; off += MPEG_TS_PACKET_SIZE) { + if (head[off] !== MPEG_TS_SYNC_BYTE) return false; + } + for (let i = 0; i < head.length; i++) { + if (head[i] === 0) return true; + } + return false; +} + +/** Whether `filePath` carries the one extension MPEG-TS shares with a language. */ +export function hasMpegTsExtension(filePath: string): boolean { + return filePath.length > 3 && filePath.slice(-3).toLowerCase() === '.ts'; +} + /** * Whether a file is one CodeGraph can parse, based purely on its extension. * This is the single source of truth for "should we index this file" — derived diff --git a/src/extraction/index.ts b/src/extraction/index.ts index 399f333acf..c456f971fe 100644 --- a/src/extraction/index.ts +++ b/src/extraction/index.ts @@ -26,7 +26,7 @@ import { ParseWorkerPool, resolveParsePoolSize, resolveParseTimeoutMs } from './ import { StoreWriter, StoreBundle, finalizeStoreBundle } from './store-writer'; import { materializeKernelResult } from './kernel'; import { detectGeneratedFile } from './generated-detection'; -import { detectLanguage, isSourceFile, isLanguageSupported, isFileLevelOnlyLanguage, initGrammars, loadGrammarsForLanguages, readGrammarWasmBytes } from './grammars'; +import { detectLanguage, isSourceFile, isLanguageSupported, isFileLevelOnlyLanguage, initGrammars, loadGrammarsForLanguages, readGrammarWasmBytes, isMpegTransportStream, hasMpegTsExtension, MPEG_TS_SNIFF_BYTES } from './grammars'; import { loadExtensionOverrides, loadIncludeIgnoredPatterns, loadExcludePatterns, loadIncludePatterns, PROJECT_CONFIG_FILENAME } from '../project-config'; import { isCodeGraphDataDir } from '../directory'; import { logDebug, logWarn } from '../errors'; @@ -566,6 +566,7 @@ function collectIncludedFiles( if (!include.ignores(rel)) return; if (exclude && exclude.ignores(rel)) return; if (!isSourceFile(rel, overrides)) return; + if (isMpegTsVideoFile(rootDir, rel)) return; out.add(rel); } }; @@ -1504,6 +1505,7 @@ export function scanDirectory( let count = 0; for (const filePath of gitFiles) { if (isSourceFile(filePath, overrides)) { + if (isMpegTsVideoFile(rootDir, filePath)) continue; files.push(filePath); count++; onProgress?.(count, filePath); @@ -1520,6 +1522,31 @@ export function scanDirectory( * Async variant of scanDirectory that yields to the event loop periodically, * allowing worker threads to receive and render progress messages. */ +/** + * Whether a `.ts` file on disk is an MPEG transport stream rather than + * TypeScript (#1910). Reads only the file's head — `MPEG_TS_SNIFF_BYTES`, under + * 1 KB — so the check costs one small read per `.ts` file at discovery, never a + * whole-file read; any file the extension does not make ambiguous is not + * touched at all. A file that cannot be read is left to the indexing path, + * which reports the read error itself. + */ +function isMpegTsVideoFile(rootDir: string, relativePath: string): boolean { + if (!hasMpegTsExtension(relativePath)) return false; + let fd: number | null = null; + try { + fd = fs.openSync(path.join(rootDir, relativePath), 'r'); + const head = Buffer.allocUnsafe(MPEG_TS_SNIFF_BYTES); + const n = fs.readSync(fd, head, 0, head.length, 0); + if (!isMpegTransportStream(head.subarray(0, n))) return false; + } catch { + return false; + } finally { + if (fd !== null) fs.closeSync(fd); + } + logDebug('Skipping MPEG transport stream named .ts — not TypeScript', { filePath: relativePath }); + return true; +} + /** * What a scan saw but could not index, tallied by extension. * @@ -1554,6 +1581,7 @@ export async function scanDirectoryAsync( let count = 0; for (const filePath of gitFiles) { if (isSourceFile(filePath, overrides)) { + if (isMpegTsVideoFile(rootDir, filePath)) continue; files.push(filePath); count++; onProgress?.(count, filePath); @@ -1662,6 +1690,7 @@ function scanDirectoryWalk( } else if (stat.isFile()) { if (!isIgnored(fullPath, false, active)) { if (isSourceFile(relativePath, overrides)) { + if (isMpegTsVideoFile(rootDir, relativePath)) continue; files.push(relativePath); count++; onProgress?.(count, relativePath); @@ -1683,6 +1712,7 @@ function scanDirectoryWalk( } else if (entry.isFile()) { if (!isIgnored(fullPath, false, active)) { if (isSourceFile(relativePath, overrides)) { + if (isMpegTsVideoFile(rootDir, relativePath)) continue; files.push(relativePath); count++; onProgress?.(count, relativePath); @@ -2263,7 +2293,16 @@ export class ExtractionOrchestrator { logWarn('Path traversal blocked in batch reader', { filePath: fp }); return { filePath: fp, content: null as string | null, stats: null as fs.Stats | null, error: new Error('Path traversal blocked') }; } - const content = await fsp.readFile(fullPath, 'utf-8'); + // Read bytes, not text: a `.ts` that is really an MPEG transport + // stream (#1910) is recognised from its head here, at no extra I/O, + // and never decoded or parsed. The scan already drops these; this + // guards the paths that hand files in by name (sync, watcher). + const bytes = await fsp.readFile(fullPath); + if (hasMpegTsExtension(fp) && isMpegTransportStream(bytes.subarray(0, MPEG_TS_SNIFF_BYTES))) { + logDebug('Skipping MPEG transport stream named .ts — not TypeScript', { filePath: fp }); + return { filePath: fp, content: null as string | null, stats: null as fs.Stats | null, error: null as Error | null, skipped: true }; + } + const content = bytes.toString('utf-8'); const stats = await fsp.stat(fullPath); return { filePath: fp, content, stats, error: null as Error | null }; } catch (err) { @@ -2274,9 +2313,16 @@ export class ExtractionOrchestrator { // Dispatch each readable file into the bounded parse window; the window // stores results on the main thread as they arrive. - for (const { filePath, content, stats, error } of fileContents) { + for (const { filePath, content, stats, error, skipped } of fileContents) { if (signal?.aborted) { aborted = true; break; } + if (skipped) { + // Not a source file after all — counted as done, stored as nothing. + processed++; + onProgress?.({ phase: 'parsing', current: processed, total }); + continue; + } + if (error || content === null || stats === null) { processed++; filesErrored++; @@ -2630,6 +2676,12 @@ export class ExtractionOrchestrator { }; } + // An MPEG transport stream named `.ts` is not TypeScript (#1910): one + // sub-KB head read decides it, before the language lookup and the parse. + if (isMpegTsVideoFile(this.rootDir, relativePath)) { + return { nodes: [], edges: [], unresolvedReferences: [], errors: [], durationMs: 0 }; + } + const language = detectLanguage(relativePath, content, loadExtensionOverrides(this.rootDir)); // Check file size From b4e28d16e786c356af8beac5d1de12d8f362a0e4 Mon Sep 17 00:00:00 2001 From: danusha2345 Date: Tue, 22 Sep 2026 20:23:56 +0300 Subject: [PATCH 2/9] fix(extraction): a video .ts is never pending on the git path or a scoped sync Review found two places that decide "is this a source file" without the MPEG-TS check the scan applies: the git fast path's candidate loop in getChangedFiles and the scoped-sync path filter (watcher events). An untracked clip in a git repo was reported as `added`, skipped by sync without a record, and reported as `added` again on every status; a tracked .ts that became a clip stayed `modified` with its stale nodes. Both now treat the clip as non-source: removed when tracked, ignored otherwise. Regression tests fail without this change and pass with it, on both arms. Co-Authored-By: Claude Opus 5.5 --- __tests__/mpeg-ts-not-typescript.test.ts | 48 ++++++++++++++++++++++++ src/extraction/index.ts | 11 +++++- 2 files changed, 57 insertions(+), 2 deletions(-) diff --git a/__tests__/mpeg-ts-not-typescript.test.ts b/__tests__/mpeg-ts-not-typescript.test.ts index 2a6e12870b..3d45924bdd 100644 --- a/__tests__/mpeg-ts-not-typescript.test.ts +++ b/__tests__/mpeg-ts-not-typescript.test.ts @@ -12,6 +12,7 @@ import { describe, it, expect, afterEach } from 'vitest'; import * as fs from 'fs'; import * as path from 'path'; import * as os from 'os'; +import { execFileSync } from 'child_process'; import { CodeGraph } from '../src'; import { scanDirectoryAsync, type ScanSkipStats } from '../src/extraction'; import { detectLanguage, isMpegTransportStream, MPEG_TS_SNIFF_BYTES } from '../src/extraction/grammars'; @@ -154,3 +155,50 @@ describe('MPEG-TS video named .ts is skipped, real TypeScript is indexed (#1910) } }); }); + +describe('a video .ts never stays pending (#1910)', () => { + const dirs: string[] = []; + afterEach(() => { for (const d of dirs.splice(0)) fs.rmSync(d, { recursive: true, force: true }); }); + const gitProject = (): string => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'codegraph-mpegts-git-')); + dirs.push(dir); + const git = (...a: string[]) => execFileSync('git', a, { cwd: dir, stdio: 'pipe' }); + git('init', '-q'); git('config', 'user.email', 't@t'); git('config', 'user.name', 't'); + fs.writeFileSync(path.join(dir, 'app.ts'), 'export const a = 1;\n'); + git('add', '.'); git('commit', '-qm', 'init'); + return dir; + }; + + it('an untracked clip is not reported as added, before or after sync', async () => { + const dir = gitProject(); + const cg = await CodeGraph.init(dir, { index: true }); + try { + fs.writeFileSync(path.join(dir, 'clip.ts'), makeMpegTs(40)); + expect(cg.getChangedFiles()).toEqual({ added: [], modified: [], removed: [] }); + await cg.sync(); + expect(cg.getChangedFiles()).toEqual({ added: [], modified: [], removed: [] }); + expect(cg.getFiles().map((f) => f.path)).toEqual(['app.ts']); + } finally { + cg.close(); + } + }); + + it('a tracked TypeScript file that becomes a clip is removed, on the git path and on a scoped sync', async () => { + for (const scoped of [false, true]) { + const dir = gitProject(); + fs.writeFileSync(path.join(dir, 'clip.ts'), 'export const clip = 1;\n'); + const cg = await CodeGraph.init(dir, { index: true }); + try { + expect(cg.getNodesInFile('clip.ts').some((n) => n.name === 'clip')).toBe(true); + fs.writeFileSync(path.join(dir, 'clip.ts'), makeMpegTs(40)); + if (!scoped) expect(cg.getChangedFiles()).toEqual({ added: [], modified: [], removed: ['clip.ts'] }); + await (scoped ? cg.sync({ paths: ['clip.ts'] }) : cg.sync()); + expect(cg.getFiles().map((f) => f.path)).toEqual(['app.ts']); + expect(cg.getNodesInFile('clip.ts')).toEqual([]); + expect(cg.getChangedFiles()).toEqual({ added: [], modified: [], removed: [] }); + } finally { + cg.close(); + } + } + }); +}); diff --git a/src/extraction/index.ts b/src/extraction/index.ts index c456f971fe..251245bfc6 100644 --- a/src/extraction/index.ts +++ b/src/extraction/index.ts @@ -3153,7 +3153,10 @@ export class ExtractionOrchestrator { (p) => isSourceFile(p, overrides) && !scope.ignores(p) && - fs.existsSync(path.join(this.rootDir, p)) + fs.existsSync(path.join(this.rootDir, p)) && + // Same rule as the scan (#1910): a reported video clip is not a + // source file, so a tracked one is removed and a new one ignored. + !isMpegTsVideoFile(this.rootDir, p) ); trackedFiles = []; for (const p of unique) { @@ -3411,7 +3414,11 @@ export class ExtractionOrchestrator { for (const filePath of candidates) { const tracked = this.queries.getFileByPath(filePath); const fullPath = path.join(this.rootDir, filePath); - if (!isSourceFile(filePath, overrides) || scope.ignores(filePath) || !fs.existsSync(fullPath)) { + // A `.ts` that is an MPEG transport stream is not source (#1910): the + // scan never lists it, so git must not report it as pending either — + // an untracked clip would otherwise stay "added" after every sync. + if (!isSourceFile(filePath, overrides) || scope.ignores(filePath) || !fs.existsSync(fullPath) + || isMpegTsVideoFile(this.rootDir, filePath)) { if (tracked) removed.push(filePath); continue; } From 538fda8cc613a641e9e5d270fd8cdf3ee96df43e Mon Sep 17 00:00:00 2001 From: danusha2345 Date: Sun, 27 Sep 2026 10:57:40 +0300 Subject: [PATCH 3/9] fix(extraction): require a binary head before calling a .ts file video (#1910) Review on #1915: the sniff took four aligned 0x47 sync bytes plus any NUL as proof of a transport stream, so TypeScript with a `G` at four 188-byte strides and one raw NUL in a comment was dropped unindexed. The sniff now needs the sync byte on sixteen consecutive packets (3 KB of head) and at least 1/64 of the head to be control bytes. Compressed payload carries 8-20% of them (checked on ffmpeg-made H.264/AAC, MPEG-2 and MP2 streams); source text carries none. A shorter clip is cheap to parse anyway. Also drops the benchmark figures from the changelog entry. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 1 + __tests__/mpeg-ts-not-typescript.test.ts | 68 ++++++++++++++++-------- src/extraction/grammars.ts | 38 ++++++++----- 3 files changed, 74 insertions(+), 33 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index d9880d7bcc..d634363afe 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -173,6 +173,7 @@ and adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). - On Windows, the shared MCP daemon now waits longer for another program — an antivirus scan, an indexer, or another session reading its lock file — to let go of that file, so it starts instead of leaving the session to fall back to a slower in-process server. (#1773) - An MPEG transport stream video that happens to be named `.ts` (golden fixtures under `testdata/`, e2e clips) is now recognised from its first bytes and skipped as non-source instead of being fed to the TypeScript parser — a 900 KB clip used to cost about 28 seconds of CPU per file for no symbols, and a folder of them minutes. Real TypeScript is never affected. (#1910) +- An MPEG transport stream video that happens to be named `.ts` (golden fixtures under `testdata/`, e2e clips) is now recognised from its first bytes and skipped instead of being fed to the TypeScript parser, which spent a long time on each clip for no symbols. Real TypeScript files are still indexed. (#1910) - File watching no longer drops the full re-scan a removed directory asks for when that sync fails, so the deleted files leave the index instead of lingering. (#1964) - Daemon startup and cleanup now preserve live legacy PID-only locks while still reclaiming dead or identity-disproved records, preventing two writers from serving the same project. - Incremental sync now keeps edge rebinding crash-safe: replacing a resolved edge with its recovery reference commits atomically, so an interruption cannot permanently remove the relationship. diff --git a/__tests__/mpeg-ts-not-typescript.test.ts b/__tests__/mpeg-ts-not-typescript.test.ts index 3d45924bdd..19fd75718c 100644 --- a/__tests__/mpeg-ts-not-typescript.test.ts +++ b/__tests__/mpeg-ts-not-typescript.test.ts @@ -33,29 +33,38 @@ function makeMpegTs(packets: number, seed = 1): Buffer { } /** - * Real TypeScript engineered to put the letter `G` (0x47) at offsets 0, 188, - * 376 and 564 — the sync-byte pattern alone. It must stay TypeScript. + * Real TypeScript engineered to put the letter `G` (0x47) at the start of each + * of its first `packets` 188-byte strides — the sync-byte pattern alone. With + * `nul`, one raw NUL sits inside a comment (the review's counterexample on + * #1915). It must stay TypeScript either way. */ -function makeGammaSource(): string { - const lines: string[] = []; +function makeGammaSource(packets = 4, nul = false): string { let text = ''; - for (let i = 0; i < 4; i++) { - const line = `Gamma${i}();`; + for (let i = 0; i < packets; i++) { + const line = `Gamma${i}();` + (nul && i === 1 ? ' // \u0000' : ''); const pad = PACKET - line.length - 1; text += line + ' '.repeat(pad) + '\n'; - lines.push(line); } - text += 'export function Gamma0() { return 0; }\n'; - text += 'export function Gamma1() { return 1; }\n'; - text += 'export function Gamma2() { return 2; }\n'; - text += 'export function Gamma3() { return 3; }\n'; - for (const off of [0, PACKET, 2 * PACKET, 3 * PACKET]) { + for (let i = 0; i < packets; i++) text += `export function Gamma${i}() { return ${i}; }\n`; + text += 'export function realFn() { return 1; }\n'; + for (let off = 0; off < packets * PACKET; off += PACKET) { if (text.charCodeAt(off) !== 0x47) throw new Error(`fixture: expected G at ${off}`); } - void lines; return text; } +/** A stream that opens the usual way: PAT and PMT packets stuffed with 0xFF, then payload. */ +function makePsiLedMpegTs(packets: number): Buffer { + const buf = makeMpegTs(packets, 7); + for (const start of [0, PACKET]) { + buf.fill(0xff, start + 4, start + PACKET); + buf[start + 1] = 0x40; + buf[start + 2] = start === 0 ? 0x00 : 0x10; + buf[start + 3] = 0x10; + } + return buf; +} + const tempDirs: string[] = []; function createProject(): string { const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'codegraph-mpegts-')); @@ -74,15 +83,21 @@ describe('isMpegTransportStream', () => { expect(isMpegTransportStream(ts)).toBe(true); }); - it('needs four aligned sync bytes — a head too short, or one packet off, is not video', () => { + it('needs sixteen aligned sync bytes — a head too short, or one packet off, is not video', () => { const ts = makeMpegTs(40); - expect(isMpegTransportStream(ts.subarray(0, 3 * PACKET))).toBe(false); - const broken = Buffer.from(ts); - broken[2 * PACKET] = 0x48; - expect(isMpegTransportStream(broken)).toBe(false); + expect(isMpegTransportStream(ts.subarray(0, 15 * PACKET))).toBe(false); + for (const packet of [2, 15]) { + const broken = Buffer.from(ts); + broken[packet * PACKET] = 0x48; + expect(isMpegTransportStream(broken)).toBe(false); + } expect(isMpegTransportStream(Buffer.alloc(0))).toBe(false); }); + it('recognises a stream that opens with 0xFF-stuffed PAT and PMT packets', () => { + expect(isMpegTransportStream(makePsiLedMpegTs(40).subarray(0, MPEG_TS_SNIFF_BYTES))).toBe(true); + }); + it('does not take source text with G at every 188th byte for video', () => { const bytes = Buffer.from(makeGammaSource(), 'utf-8'); expect(bytes[0]).toBe(0x47); @@ -90,6 +105,14 @@ describe('isMpegTransportStream', () => { expect(isMpegTransportStream(bytes)).toBe(false); expect(detectLanguage('gamma.ts', makeGammaSource())).toBe('typescript'); }); + + it('does not take source with G at every stride and a NUL in a comment for video (#1915 review)', () => { + for (const packets of [4, 16, 20]) { + const bytes = Buffer.from(makeGammaSource(packets, true), 'utf-8'); + expect(bytes.includes(0)).toBe(true); + expect(isMpegTransportStream(bytes.subarray(0, MPEG_TS_SNIFF_BYTES))).toBe(false); + } + }); }); describe('MPEG-TS video named .ts is skipped, real TypeScript is indexed (#1910)', () => { @@ -99,23 +122,26 @@ describe('MPEG-TS video named .ts is skipped, real TypeScript is indexed (#1910) fs.writeFileSync(path.join(dir, 'testdata', 'clip.ts'), makeMpegTs(40)); fs.writeFileSync(path.join(dir, 'app.ts'), 'export function greet(n: string) { return `hi ${n}`; }\n'); fs.writeFileSync(path.join(dir, 'gamma.ts'), makeGammaSource()); + fs.writeFileSync(path.join(dir, 'gamma-nul.ts'), makeGammaSource(16, true)); const stats: ScanSkipStats = { unsupportedByExtension: new Map() }; const scanned = await scanDirectoryAsync(dir, undefined, stats); - expect(scanned.sort()).toEqual(['app.ts', 'gamma.ts']); + expect(scanned.sort()).toEqual(['app.ts', 'gamma-nul.ts', 'gamma.ts']); expect(stats.unsupportedByExtension.size).toBe(0); const cg = await CodeGraph.init(dir, { index: true }); try { const files = cg.getFiles().map((f) => f.path).sort(); - expect(files).toEqual(['app.ts', 'gamma.ts']); + expect(files).toEqual(['app.ts', 'gamma-nul.ts', 'gamma.ts']); expect(cg.searchNodes('greet').some((r) => r.node.name === 'greet')).toBe(true); expect(cg.searchNodes('Gamma2').some((r) => r.node.name === 'Gamma2')).toBe(true); + // The review's counterexample: G at every stride and a NUL in a comment. + expect(cg.searchNodes('realFn').filter((r) => r.node.filePath === 'gamma-nul.ts')).toHaveLength(1); // A named re-sync (the watcher / `sync` path hands files in by name) must // not let the clip back in either. await cg.sync({ paths: ['testdata/clip.ts', 'app.ts'] }); - expect(cg.getFiles().map((f) => f.path).sort()).toEqual(['app.ts', 'gamma.ts']); + expect(cg.getFiles().map((f) => f.path).sort()).toEqual(['app.ts', 'gamma-nul.ts', 'gamma.ts']); } finally { await cg.close(); } diff --git a/src/extraction/grammars.ts b/src/extraction/grammars.ts index d75180e4e9..84596d81c5 100644 --- a/src/extraction/grammars.ts +++ b/src/extraction/grammars.ts @@ -175,8 +175,18 @@ export const EXTENSION_MAP: Record = { /** MPEG transport stream: fixed 188-byte packets, each opening with 0x47. */ const MPEG_TS_PACKET_SIZE = 188; const MPEG_TS_SYNC_BYTE = 0x47; -/** Consecutive packets whose sync byte must line up before a file counts as video. */ -const MPEG_TS_MIN_PACKETS = 4; +/** + * Consecutive packets whose sync byte must line up before a file counts as + * video — 3 KB of head. A stream shorter than that is cheap to parse anyway; + * the cost #1910 is about comes from clips hundreds of KB long. + */ +const MPEG_TS_MIN_PACKETS = 16; +/** + * Share of the head that must be control bytes (below 0x20, other than the + * whitespace ones) for it to count as binary. Compressed audio and video put + * about one byte in eight there; source text puts none. + */ +const MPEG_TS_MIN_CONTROL_SHARE = 1 / 64; /** * How many bytes of a file's head `isMpegTransportStream` needs — enough to * see `MPEG_TS_MIN_PACKETS` sync bytes plus the packets between them. @@ -191,14 +201,15 @@ export const MPEG_TS_SNIFF_BYTES = MPEG_TS_PACKET_SIZE * MPEG_TS_MIN_PACKETS; * of the file, before any parse. * * Two conditions, both required: - * 1. the sync byte 0x47 sits at offsets 0, 188, 376 and 564 — every packet - * of a transport stream opens with it, and nothing else pads to 188; - * 2. a NUL byte appears somewhere in the head — every stream carries one - * within its first packets (the PSI pointer field, table reserved bits, - * the `00 00 01` PES start codes), and UTF-8 source text never does. - * 0x47 is the letter `G`, so (1) alone could in principle match a source file - * whose lines happen to put a `G` at four 188-byte strides; (2) closes that - * door, because a text file with a NUL in it is not TypeScript either. + * 1. the sync byte 0x47 sits at every 188-byte packet boundary of the first + * `MPEG_TS_MIN_PACKETS` packets — every packet of a transport stream + * opens with it, and nothing else pads to 188; + * 2. the head is binary: at least `MPEG_TS_MIN_CONTROL_SHARE` of it is + * control bytes, as any compressed payload is. + * 0x47 is the letter `G`, so (1) alone could match source whose lines happen + * to put a `G` at every 188-byte stride. Checking for a single NUL was not + * enough to close that: one NUL in a comment is still TypeScript. (2) asks for + * dozens of control bytes, which no source file carries. * * `head` is the first `MPEG_TS_SNIFF_BYTES` (or fewer) bytes of the file. */ @@ -208,10 +219,13 @@ export function isMpegTransportStream(head: Uint8Array): boolean { for (let off = 0; off <= lastSync; off += MPEG_TS_PACKET_SIZE) { if (head[off] !== MPEG_TS_SYNC_BYTE) return false; } + let control = 0; for (let i = 0; i < head.length; i++) { - if (head[i] === 0) return true; + const b = head[i]!; + // Tab, newline, vertical tab, form feed and carriage return are text. + if (b < 0x20 && (b < 0x09 || b > 0x0d)) control++; } - return false; + return control >= head.length * MPEG_TS_MIN_CONTROL_SHARE; } /** Whether `filePath` carries the one extension MPEG-TS shares with a language. */ From d0b4917f5644d10181e8631b5a0d75c018121a38 Mon Sep 17 00:00:00 2001 From: danusha2345 Date: Tue, 22 Sep 2026 18:12:36 +0300 Subject: [PATCH 4/9] fix(extraction): a file over the size limit is never read (#1910) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The size gate ran after `readFile`, so a committed video or blob fixture was decoded in full — 3.4 GB of RSS for a 400 MB file, `Invalid string length` past ~512 MB — only to be stored as skipped. It was then read again by every pass that scans files by name: the framework detectors' `readFile` and the resolver's cached reader, three full passes on the reporter's fixtures. Stat first, everywhere a source file is read by path: the batch reader and `extractFile` store an oversize file with a size stamp in place of its content, change detection hashes the same stamp (so a same-size rewrite of a file nothing is indexed from is not a change, while crossing the limit in either direction is), and both by-name readers return null above the limit. The limit moves to `src/file-limits.ts` so the three sites share one number. strace on a sparse 400 MB `.ts`: 153 603 reads before, 0 after; indexing it next to a real file costs no measurable RSS. Stacked on #1914, which shares the batch-reader hunk. Co-Authored-By: Claude Fable 5.1 --- CHANGELOG.md | 1 + __tests__/oversize-file-not-read.test.ts | 79 ++++++++++++++++++++++++ src/extraction/index.ts | 45 ++++++++++++-- 3 files changed, 120 insertions(+), 5 deletions(-) create mode 100644 __tests__/oversize-file-not-read.test.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index d634363afe..6547bde1ca 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -173,6 +173,7 @@ and adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). - On Windows, the shared MCP daemon now waits longer for another program — an antivirus scan, an indexer, or another session reading its lock file — to let go of that file, so it starts instead of leaving the session to fall back to a slower in-process server. (#1773) - An MPEG transport stream video that happens to be named `.ts` (golden fixtures under `testdata/`, e2e clips) is now recognised from its first bytes and skipped as non-source instead of being fed to the TypeScript parser — a 900 KB clip used to cost about 28 seconds of CPU per file for no symbols, and a folder of them minutes. Real TypeScript is never affected. (#1910) +- A file over the size limit is no longer read before it is skipped: committed video and blob fixtures used to be decoded in full — a 400 MB fixture cost 3.4 GB of memory — only to be discarded, and the same file was read again by every resolution pass. Its size stamp now stands in for its content, during indexing and when checking for changes. (#1910) - An MPEG transport stream video that happens to be named `.ts` (golden fixtures under `testdata/`, e2e clips) is now recognised from its first bytes and skipped instead of being fed to the TypeScript parser, which spent a long time on each clip for no symbols. Real TypeScript files are still indexed. (#1910) - File watching no longer drops the full re-scan a removed directory asks for when that sync fails, so the deleted files leave the index instead of lingering. (#1964) - Daemon startup and cleanup now preserve live legacy PID-only locks while still reclaiming dead or identity-disproved records, preventing two writers from serving the same project. diff --git a/__tests__/oversize-file-not-read.test.ts b/__tests__/oversize-file-not-read.test.ts new file mode 100644 index 0000000000..246e2b4458 --- /dev/null +++ b/__tests__/oversize-file-not-read.test.ts @@ -0,0 +1,79 @@ +import { afterEach, describe, expect, it } from 'vitest'; +import * as fs from 'fs'; +import * as os from 'os'; +import * as path from 'path'; +import { CodeGraph } from '../src'; +import { oversizeStamp, hashContent } from '../src/extraction'; + +/** + * A file over the size limit is stored as skipped without ever being read + * (#1910): committed video/blob fixtures used to be decoded in full — and + * hashed — only to be discarded, costing their size in RSS per file. + */ +describe('oversize files are stat-gated, never read (#1910)', () => { + let dir: string; + afterEach(() => { if (dir) fs.rmSync(dir, { recursive: true, force: true }); }); + + const big = (bytes: number, fill = 0x41) => Buffer.alloc(bytes, fill); + + it('indexes the neighbours, records the oversize file as skipped with a size-stamp hash, and does not decode it', async () => { + dir = fs.mkdtempSync(path.join(os.tmpdir(), 'codegraph-oversize-')); + fs.writeFileSync(path.join(dir, 'app.ts'), 'export function alpha() { return beta(); }\nexport function beta() { return 1; }\n'); + // 1 MB + 1: over the limit; invalid UTF-8 (0xFF) so any decode would be visible as replacement chars. + fs.writeFileSync(path.join(dir, 'blob.ts'), big(1024 * 1024 + 1, 0xff)); + const cg = await CodeGraph.init(dir, { index: true }); + try { + expect(cg.getNodesByKind('function').map(n => n.name).sort()).toEqual(['alpha', 'beta']); + const rec = cg.getFiles().find(f => f.path === 'blob.ts'); + expect(rec).toBeDefined(); + expect(rec!.contentHash).toBe(hashContent(oversizeStamp(1024 * 1024 + 1))); + // Nothing from the blob reached the graph. + expect(cg.getNodesInFile('blob.ts').filter(n => n.kind !== 'file')).toEqual([]); + // Change detection agrees with what was stored: nothing pending. + expect(cg.getChangedFiles()).toEqual({ added: [], modified: [], removed: [] }); + // A same-size rewrite is not a change (nothing about it is indexed)... + fs.writeFileSync(path.join(dir, 'blob.ts'), big(1024 * 1024 + 1, 0xfe)); + expect(cg.getChangedFiles().modified).toEqual([]); + // ...crossing the limit is: the file becomes ordinary source. + fs.writeFileSync(path.join(dir, 'blob.ts'), 'export const gamma = 3;\n'); + expect(cg.getChangedFiles().modified).toEqual(['blob.ts']); + await cg.sync(); + expect(cg.getNodesInFile('blob.ts').some(n => n.name === 'gamma')).toBe(true); + // ...and growing back over it is a change too, stored as the stamp again. + fs.writeFileSync(path.join(dir, 'blob.ts'), big(2 * 1024 * 1024, 0xff)); + expect(cg.getChangedFiles().modified).toEqual(['blob.ts']); + await cg.sync(); + expect(cg.getFiles().find(f => f.path === 'blob.ts')!.contentHash).toBe(hashContent(oversizeStamp(2 * 1024 * 1024))); + expect(cg.getNodesInFile('blob.ts').some(n => n.name === 'gamma')).toBe(false); + } finally { + cg.close(); + } + }); + + it('a 400 MB sparse fixture indexes in well under a second and without growing the heap by its size', async () => { + dir = fs.mkdtempSync(path.join(os.tmpdir(), 'codegraph-oversize-big-')); + fs.writeFileSync(path.join(dir, 'ok.ts'), 'export const one = 1;\n'); + // Sparse: occupies no disk, but stat() reports 400 MB — a read would decode all of it. + const fd = fs.openSync(path.join(dir, 'huge.ts'), 'w'); + fs.ftruncateSync(fd, 400 * 1024 * 1024); + fs.closeSync(fd); + // Warm the engine on a sibling project first, so the grammar and worker + // start-up cost is not mistaken for the file being read. + const warm = fs.mkdtempSync(path.join(os.tmpdir(), 'codegraph-oversize-warm-')); + fs.writeFileSync(path.join(warm, 'w.ts'), 'export const w = 1;\n'); + (await CodeGraph.init(warm, { index: true })).close(); + fs.rmSync(warm, { recursive: true, force: true }); + const before = process.memoryUsage().rss; + const t0 = Date.now(); + const cg = await CodeGraph.init(dir, { index: true }); + try { + expect(Date.now() - t0).toBeLessThan(5000); + // Reading 400 MB would show as at least that much RSS; the stamp shows as none. + expect(process.memoryUsage().rss - before).toBeLessThan(100 * 1024 * 1024); + expect(cg.getFiles().map(f => f.path).sort()).toEqual(['huge.ts', 'ok.ts']); + expect(cg.getChangedFiles()).toEqual({ added: [], modified: [], removed: [] }); + } finally { + cg.close(); + } + }); +}); diff --git a/src/extraction/index.ts b/src/extraction/index.ts index 251245bfc6..b491199345 100644 --- a/src/extraction/index.ts +++ b/src/extraction/index.ts @@ -158,6 +158,30 @@ export function hashContent(content: string): string { return crypto.createHash('sha256').update(content).digest('hex'); } +/** + * What stands in for the content of a file over MAX_SOURCE_FILE_SIZE_BYTES. Such a file is + * never parsed, so its bytes are never needed — reading them only to hash and + * discard cost multi-GB RSS spikes on committed video/blob fixtures and could + * fail outright with `Invalid string length` (#1910). The stamp is a function + * of size alone: change detection compares it to the stored hash, so a + * same-size rewrite of an oversize file is not a change (nothing about it is + * indexed), while crossing the limit in either direction is. + */ +export function oversizeStamp(size: number): string { + return `codegraph:oversize:${size}`; +} + +/** + * Read a file for hashing/indexing: the whole text when it is under the size + * limit, the size stamp when it is over — the caller never decodes an oversize + * file. `stats` is what the caller already has; without it the file is stat'ed. + */ +function readSourceOrStamp(fullPath: string, stats?: fs.Stats): { content: string; stats: fs.Stats } { + const st = stats ?? fs.statSync(fullPath); + if (st.size > MAX_SOURCE_FILE_SIZE_BYTES) return { content: oversizeStamp(st.size), stats: st }; + return { content: fs.readFileSync(fullPath, 'utf-8'), stats: st }; +} + /** * Directory names that are dependency, build, cache, or tooling output across the * languages/frameworks CodeGraph supports — curated from the canonical @@ -1876,6 +1900,9 @@ export class ExtractionOrchestrator { const full = validatePathWithinRoot(rootDir, relativePath); if (!full) return null; try { + // Framework detectors scan source by name; a file over the size + // limit was never indexed and must not be decoded here either (#1910). + if (fs.statSync(full).size > MAX_SOURCE_FILE_SIZE_BYTES) return null; return fs.readFileSync(full, 'utf-8'); } catch { return null; @@ -2297,13 +2324,19 @@ export class ExtractionOrchestrator { // stream (#1910) is recognised from its head here, at no extra I/O, // and never decoded or parsed. The scan already drops these; this // guards the paths that hand files in by name (sync, watcher). + // Stat first: a file over the size limit is stored as skipped + // without ever being read or decoded (#1910), so ten oversize + // fixtures in one I/O batch no longer cost their size in RSS. + const stats = await fsp.stat(fullPath); + if (stats.size > MAX_SOURCE_FILE_SIZE_BYTES) { + return { filePath: fp, content: oversizeStamp(stats.size), stats, error: null as Error | null }; + } const bytes = await fsp.readFile(fullPath); if (hasMpegTsExtension(fp) && isMpegTransportStream(bytes.subarray(0, MPEG_TS_SNIFF_BYTES))) { logDebug('Skipping MPEG transport stream named .ts — not TypeScript', { filePath: fp }); return { filePath: fp, content: null as string | null, stats: null as fs.Stats | null, error: null as Error | null, skipped: true }; } const content = bytes.toString('utf-8'); - const stats = await fsp.stat(fullPath); return { filePath: fp, content, stats, error: null as Error | null }; } catch (err) { return { filePath: fp, content: null as string | null, stats: null as fs.Stats | null, error: err as Error }; @@ -2633,7 +2666,8 @@ export class ExtractionOrchestrator { let stats: fs.Stats; try { stats = await fsp.stat(fullPath); - content = await fsp.readFile(fullPath, 'utf-8'); + // An oversize file is stored as skipped; its bytes are never needed (#1910). + content = stats.size > MAX_SOURCE_FILE_SIZE_BYTES ? oversizeStamp(stats.size) : await fsp.readFile(fullPath, 'utf-8'); } catch (error) { return { nodes: [], @@ -3254,9 +3288,10 @@ export class ExtractionOrchestrator { } // New, or size/mtime changed — read + hash to confirm a real content change. + // (An oversize file hashes as its size stamp, unread — #1910.) let content: string; try { - content = fs.readFileSync(fullPath, 'utf-8'); + content = readSourceOrStamp(fullPath).content; } catch (error) { logDebug('Skipping unreadable file during sync', { filePath, error: String(error) }); failedFilePaths.push(filePath); @@ -3423,7 +3458,7 @@ export class ExtractionOrchestrator { continue; } let content: string; - try { content = fs.readFileSync(fullPath, 'utf-8'); } + try { content = readSourceOrStamp(fullPath).content; } catch (error) { logDebug('Skipping unreadable file while detecting changes', { filePath, error: String(error) }); continue; @@ -3461,7 +3496,7 @@ export class ExtractionOrchestrator { const fullPath = path.join(this.rootDir, filePath); let content: string; try { - content = fs.readFileSync(fullPath, 'utf-8'); + content = readSourceOrStamp(fullPath).content; } catch (error) { logDebug('Skipping unreadable file while detecting changes', { filePath, error: String(error) }); continue; From 53aebeb898152ed91bb47e461c67ca877af98801 Mon Sep 17 00:00:00 2001 From: danusha2345 Date: Tue, 22 Sep 2026 20:26:02 +0300 Subject: [PATCH 5/9] refactor(extraction): one-purpose size-stamp reader, comments in operation order `readSourceOrStamp` took a `stats` argument no caller passed and returned stats no caller read; it now returns the text or stamp change detection hashes. The batch reader's two comments are back in the order its code runs: the size gate, then the MPEG-TS head check. No behaviour change. Co-Authored-By: Claude Opus 5.5 --- src/extraction/index.ts | 25 +++++++++++-------------- 1 file changed, 11 insertions(+), 14 deletions(-) diff --git a/src/extraction/index.ts b/src/extraction/index.ts index b491199345..c1291676ef 100644 --- a/src/extraction/index.ts +++ b/src/extraction/index.ts @@ -172,14 +172,12 @@ export function oversizeStamp(size: number): string { } /** - * Read a file for hashing/indexing: the whole text when it is under the size - * limit, the size stamp when it is over — the caller never decodes an oversize - * file. `stats` is what the caller already has; without it the file is stat'ed. + * What change detection hashes for a file: its text when it is under the size + * limit, the size stamp when it is over — an oversize file is never decoded. */ -function readSourceOrStamp(fullPath: string, stats?: fs.Stats): { content: string; stats: fs.Stats } { - const st = stats ?? fs.statSync(fullPath); - if (st.size > MAX_SOURCE_FILE_SIZE_BYTES) return { content: oversizeStamp(st.size), stats: st }; - return { content: fs.readFileSync(fullPath, 'utf-8'), stats: st }; +function readSourceOrStamp(fullPath: string): string { + const size = fs.statSync(fullPath).size; + return size > MAX_SOURCE_FILE_SIZE_BYTES ? oversizeStamp(size) : fs.readFileSync(fullPath, 'utf-8'); } /** @@ -2320,10 +2318,6 @@ export class ExtractionOrchestrator { logWarn('Path traversal blocked in batch reader', { filePath: fp }); return { filePath: fp, content: null as string | null, stats: null as fs.Stats | null, error: new Error('Path traversal blocked') }; } - // Read bytes, not text: a `.ts` that is really an MPEG transport - // stream (#1910) is recognised from its head here, at no extra I/O, - // and never decoded or parsed. The scan already drops these; this - // guards the paths that hand files in by name (sync, watcher). // Stat first: a file over the size limit is stored as skipped // without ever being read or decoded (#1910), so ten oversize // fixtures in one I/O batch no longer cost their size in RSS. @@ -2331,6 +2325,9 @@ export class ExtractionOrchestrator { if (stats.size > MAX_SOURCE_FILE_SIZE_BYTES) { return { filePath: fp, content: oversizeStamp(stats.size), stats, error: null as Error | null }; } + // Read bytes, not text: a `.ts` that is really an MPEG transport + // stream (#1910) is recognised from its head here, at no extra I/O, + // and never decoded or parsed. const bytes = await fsp.readFile(fullPath); if (hasMpegTsExtension(fp) && isMpegTransportStream(bytes.subarray(0, MPEG_TS_SNIFF_BYTES))) { logDebug('Skipping MPEG transport stream named .ts — not TypeScript', { filePath: fp }); @@ -3291,7 +3288,7 @@ export class ExtractionOrchestrator { // (An oversize file hashes as its size stamp, unread — #1910.) let content: string; try { - content = readSourceOrStamp(fullPath).content; + content = readSourceOrStamp(fullPath); } catch (error) { logDebug('Skipping unreadable file during sync', { filePath, error: String(error) }); failedFilePaths.push(filePath); @@ -3458,7 +3455,7 @@ export class ExtractionOrchestrator { continue; } let content: string; - try { content = readSourceOrStamp(fullPath).content; } + try { content = readSourceOrStamp(fullPath); } catch (error) { logDebug('Skipping unreadable file while detecting changes', { filePath, error: String(error) }); continue; @@ -3496,7 +3493,7 @@ export class ExtractionOrchestrator { const fullPath = path.join(this.rootDir, filePath); let content: string; try { - content = readSourceOrStamp(fullPath).content; + content = readSourceOrStamp(fullPath); } catch (error) { logDebug('Skipping unreadable file while detecting changes', { filePath, error: String(error) }); continue; From 157f3d2265ca1e00d6e55bf9e37e7d4745124ad1 Mon Sep 17 00:00:00 2001 From: danusha2345 Date: Thu, 24 Sep 2026 11:05:42 +0300 Subject: [PATCH 6/9] fix(extraction): bound the read itself, not only the stat before it (#1910) A file can grow between the stat that gates the size limit and the read that follows (a log, a download, a build output being written). Every source read now goes through readBoundedSource/readBoundedSourceSync: the open descriptor is re-checked, and the read stops one byte past the limit, so an oversize file is still stamped rather than decoded. The extraction batch reader, single-file extraction, framework detectors and the resolver's file cache all use it. Ported from the maintainer's hardening of this fix in #1919. Co-Authored-By: Claude Opus 5.5 --- __tests__/bounded-source.test.ts | 96 ++++++++++++++++++++++++++++++++ src/extraction/index.ts | 19 +++---- src/file-limits.ts | 81 +++++++++++++++++++++++++++ 3 files changed, 186 insertions(+), 10 deletions(-) create mode 100644 __tests__/bounded-source.test.ts diff --git a/__tests__/bounded-source.test.ts b/__tests__/bounded-source.test.ts new file mode 100644 index 0000000000..e531b02a2a --- /dev/null +++ b/__tests__/bounded-source.test.ts @@ -0,0 +1,96 @@ +/** + * Source reads are bounded by the read itself, not only by a stat taken + * before it (#1910). A file can grow between the stat and the read; the + * reader must still never hold more than the limit plus one byte. + */ +import { describe, it, expect, afterEach, vi } from 'vitest'; +import * as fs from 'fs'; +import * as fsp from 'fs/promises'; +import * as os from 'os'; +import * as path from 'path'; +import { readBoundedSource, readBoundedSourceSync, MAX_SOURCE_FILE_SIZE_BYTES as LIMIT } from '../src/file-limits'; + +// Pass-through wrappers, so a test can inject a file growing mid-read into the +// exact calls file-limits.ts makes. +vi.mock('fs', async (importOriginal) => { + const actual = await importOriginal(); + return { ...actual, readSync: vi.fn(actual.readSync) }; +}); +vi.mock('fs/promises', async (importOriginal) => { + const actual = await importOriginal(); + return { ...actual, open: vi.fn(actual.open) }; +}); + +const dirs: string[] = []; +function sourceFile(bytes: Buffer): string { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'cg-bounded-')); + dirs.push(dir); + const file = path.join(dir, 'source.ts'); + fs.writeFileSync(file, bytes); + return file; +} + +afterEach(() => { + for (const dir of dirs.splice(0)) fs.rmSync(dir, { recursive: true, force: true }); +}); + +describe('bounded source reads (#1910)', () => { + for (const size of [0, 100, LIMIT, LIMIT + 1]) { + it(`returns the bytes up to the limit and null past it (${size} bytes)`, async () => { + const file = sourceFile(Buffer.alloc(size, 0x61)); + for (const result of [readBoundedSourceSync(file), await readBoundedSource(file)]) { + expect(result.stats.size).toBe(size); + if (size > LIMIT) expect(result.bytes).toBeNull(); + else expect(result.bytes?.length).toBe(size); + } + }); + } + + it('stops at the limit when the file grows during a synchronous read', () => { + const file = sourceFile(Buffer.from('hello')); + const readSync = vi.mocked(fs.readSync); + const original = readSync.getMockImplementation()!; + let requested = 0; + let grown = false; + readSync.mockImplementation(((fd: number, buf: Buffer, off: number, len: number, pos: number) => { + requested += len; + if (!grown) { + grown = true; + fs.writeFileSync(file, Buffer.alloc(LIMIT + 10, 0x61)); + } + return original(fd, buf, off, len, pos); + }) as typeof fs.readSync); + let result; + try { result = readBoundedSourceSync(file); } finally { readSync.mockImplementation(original); } + expect(grown).toBe(true); + expect(requested).toBeLessThanOrEqual(LIMIT + 1); + expect(result.bytes).toBeNull(); + }); + + it('rechecks the open descriptor when the file grows between stat and open', async () => { + const file = sourceFile(Buffer.from('hello')); + const open = vi.mocked(fsp.open); + const original = open.getMockImplementation()!; + let grown = false; + open.mockImplementationOnce((async (...args: Parameters) => { + grown = true; + fs.writeFileSync(file, Buffer.alloc(LIMIT + 1)); + return original(...args); + }) as typeof fsp.open); + expect((await readBoundedSource(file)).bytes).toBeNull(); + expect(grown).toBe(true); + }); + + it('returns multibyte source byte-for-byte', async () => { + const text = 'export const greeting = "こんにちは 🌿";'; + const file = sourceFile(Buffer.from(text)); + expect((await readBoundedSource(file)).bytes?.toString('utf8')).toBe(text); + expect(readBoundedSourceSync(file).bytes?.toString('utf8')).toBe(text); + }); + + it('refuses a path that is not a regular file', () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'cg-bounded-')); + dirs.push(dir); + expect(() => readBoundedSourceSync(dir)).toThrow(/not a regular file/); + }); +}); diff --git a/src/extraction/index.ts b/src/extraction/index.ts index c1291676ef..d17948cce2 100644 --- a/src/extraction/index.ts +++ b/src/extraction/index.ts @@ -35,7 +35,7 @@ import ignore, { Ignore } from 'ignore'; import { detectFrameworks } from '../resolution/frameworks'; import type { ResolutionContext } from '../resolution/types'; import { createYielder, type MaybeYield } from '../resolution/cooperative-yield'; -import { MAX_SOURCE_FILE_SIZE_BYTES } from '../file-limits'; +import { MAX_SOURCE_FILE_SIZE_BYTES, readBoundedSource, readBoundedSourceSync } from '../file-limits'; /** * Number of files to read in parallel during indexing. @@ -176,8 +176,8 @@ export function oversizeStamp(size: number): string { * limit, the size stamp when it is over — an oversize file is never decoded. */ function readSourceOrStamp(fullPath: string): string { - const size = fs.statSync(fullPath).size; - return size > MAX_SOURCE_FILE_SIZE_BYTES ? oversizeStamp(size) : fs.readFileSync(fullPath, 'utf-8'); + const { stats, bytes } = readBoundedSourceSync(fullPath); + return bytes === null ? oversizeStamp(stats.size) : bytes.toString('utf8'); } /** @@ -1900,8 +1900,7 @@ export class ExtractionOrchestrator { try { // Framework detectors scan source by name; a file over the size // limit was never indexed and must not be decoded here either (#1910). - if (fs.statSync(full).size > MAX_SOURCE_FILE_SIZE_BYTES) return null; - return fs.readFileSync(full, 'utf-8'); + return readBoundedSourceSync(full).bytes?.toString('utf8') ?? null; } catch { return null; } @@ -2321,14 +2320,13 @@ export class ExtractionOrchestrator { // Stat first: a file over the size limit is stored as skipped // without ever being read or decoded (#1910), so ten oversize // fixtures in one I/O batch no longer cost their size in RSS. - const stats = await fsp.stat(fullPath); - if (stats.size > MAX_SOURCE_FILE_SIZE_BYTES) { + const { stats, bytes } = await readBoundedSource(fullPath); + if (bytes === null) { return { filePath: fp, content: oversizeStamp(stats.size), stats, error: null as Error | null }; } // Read bytes, not text: a `.ts` that is really an MPEG transport // stream (#1910) is recognised from its head here, at no extra I/O, // and never decoded or parsed. - const bytes = await fsp.readFile(fullPath); if (hasMpegTsExtension(fp) && isMpegTransportStream(bytes.subarray(0, MPEG_TS_SNIFF_BYTES))) { logDebug('Skipping MPEG transport stream named .ts — not TypeScript', { filePath: fp }); return { filePath: fp, content: null as string | null, stats: null as fs.Stats | null, error: null as Error | null, skipped: true }; @@ -2662,9 +2660,10 @@ export class ExtractionOrchestrator { let content: string; let stats: fs.Stats; try { - stats = await fsp.stat(fullPath); // An oversize file is stored as skipped; its bytes are never needed (#1910). - content = stats.size > MAX_SOURCE_FILE_SIZE_BYTES ? oversizeStamp(stats.size) : await fsp.readFile(fullPath, 'utf-8'); + const read = await readBoundedSource(fullPath); + stats = read.stats; + content = read.bytes === null ? oversizeStamp(stats.size) : read.bytes.toString('utf8'); } catch (error) { return { nodes: [], diff --git a/src/file-limits.ts b/src/file-limits.ts index 4bb4d6da1a..ed3136f936 100644 --- a/src/file-limits.ts +++ b/src/file-limits.ts @@ -1,6 +1,87 @@ +import * as fs from 'fs'; +import * as fsp from 'fs/promises'; + /** * Largest source file CodeGraph will parse or read during resolution. Generated * bundles, minified sources, and dependency archives above this limit provide no * useful symbols; 1 MB covers essentially all hand-written source. */ export const MAX_SOURCE_FILE_SIZE_BYTES = 1024 * 1024; + +/** A source file's stats, and its bytes when they are within the limit (null = oversize). */ +export interface BoundedSource { + stats: fs.Stats; + bytes: Buffer | null; +} + +const READ_CHUNK_BYTES = 64 * 1024; + +function assertRegularFile(stats: fs.Stats): void { + if (!stats.isFile()) throw new Error('Source path is not a regular file'); +} + +/** + * Read a source file without ever holding more than the limit plus one byte. + * A stat before the read is not enough: the file can grow between the stat + * and the read (a log, a download, a build output being written), so the + * descriptor is re-checked after opening and the read itself stops one byte + * past the limit. Returns `bytes: null` for an oversize file. + */ +export async function readBoundedSource(file: string): Promise { + const initial = await fsp.stat(file); + assertRegularFile(initial); + if (initial.size > MAX_SOURCE_FILE_SIZE_BYTES) return { stats: initial, bytes: null }; + const handle = await fsp.open(file, 'r'); + try { + let stats = await handle.stat(); + assertRegularFile(stats); + if (stats.size > MAX_SOURCE_FILE_SIZE_BYTES) return { stats, bytes: null }; + const chunks: Buffer[] = []; + let size = 0; + while (size <= MAX_SOURCE_FILE_SIZE_BYTES) { + const chunk = Buffer.allocUnsafe(Math.min(READ_CHUNK_BYTES, MAX_SOURCE_FILE_SIZE_BYTES + 1 - size)); + const { bytesRead } = await handle.read(chunk, 0, chunk.length, size); + if (!bytesRead) break; + chunks.push(chunk.subarray(0, bytesRead)); + size += bytesRead; + } + stats = await handle.stat(); + if (size > MAX_SOURCE_FILE_SIZE_BYTES || stats.size > MAX_SOURCE_FILE_SIZE_BYTES) { + stats.size = Math.max(size, stats.size); + return { stats, bytes: null }; + } + return { stats, bytes: Buffer.concat(chunks, size) }; + } finally { + await handle.close(); + } +} + +/** Synchronous {@link readBoundedSource}. */ +export function readBoundedSourceSync(file: string): BoundedSource { + const initial = fs.statSync(file); + assertRegularFile(initial); + if (initial.size > MAX_SOURCE_FILE_SIZE_BYTES) return { stats: initial, bytes: null }; + const fd = fs.openSync(file, 'r'); + try { + let stats = fs.fstatSync(fd); + assertRegularFile(stats); + if (stats.size > MAX_SOURCE_FILE_SIZE_BYTES) return { stats, bytes: null }; + const chunks: Buffer[] = []; + let size = 0; + while (size <= MAX_SOURCE_FILE_SIZE_BYTES) { + const chunk = Buffer.allocUnsafe(Math.min(READ_CHUNK_BYTES, MAX_SOURCE_FILE_SIZE_BYTES + 1 - size)); + const bytesRead = fs.readSync(fd, chunk, 0, chunk.length, size); + if (!bytesRead) break; + chunks.push(chunk.subarray(0, bytesRead)); + size += bytesRead; + } + stats = fs.fstatSync(fd); + if (size > MAX_SOURCE_FILE_SIZE_BYTES || stats.size > MAX_SOURCE_FILE_SIZE_BYTES) { + stats.size = Math.max(size, stats.size); + return { stats, bytes: null }; + } + return { stats, bytes: Buffer.concat(chunks, size) }; + } finally { + fs.closeSync(fd); + } +} From ef8a03fd8b9c397f6829ead45c2c565320be1784 Mon Sep 17 00:00:00 2001 From: danusha2345 Date: Thu, 24 Sep 2026 11:09:00 +0300 Subject: [PATCH 7/9] fix(extraction): bound the parse-retry reads too (#1910) The retry passes for files whose worker crashed or timed out re-read the source with an unbounded fsp.readFile. The file can have grown since the first, bounded read; an oversize file is now skipped by the retry instead of being decoded. Co-Authored-By: Claude Opus 5.5 --- src/extraction/index.ts | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/src/extraction/index.ts b/src/extraction/index.ts index d17948cce2..0c2d4de16a 100644 --- a/src/extraction/index.ts +++ b/src/extraction/index.ts @@ -2478,7 +2478,10 @@ export class ExtractionOrchestrator { try { const fullPath = validatePathWithinRoot(this.rootDir, filePath); if (!fullPath) continue; - content = await fsp.readFile(fullPath, 'utf-8'); + // Bounded like the first read: the file may have grown since (#1910). + const bytes = (await readBoundedSource(fullPath)).bytes; + if (bytes === null) continue; + content = bytes.toString('utf8'); } catch { continue; } @@ -2530,7 +2533,9 @@ export class ExtractionOrchestrator { try { const fullPath = validatePathWithinRoot(this.rootDir, filePath); if (!fullPath) continue; - fullContent = await fsp.readFile(fullPath, 'utf-8'); + const bytes = (await readBoundedSource(fullPath)).bytes; + if (bytes === null) continue; + fullContent = bytes.toString('utf8'); } catch { continue; } From 4c74148de2d93bf9927e3f6157bbdc7634f2880d Mon Sep 17 00:00:00 2001 From: danusha2345 Date: Sun, 27 Sep 2026 11:01:16 +0300 Subject: [PATCH 8/9] fix(extraction): check an oversize file against its size stamp everywhere (#1910) Review on #1915: - The viewer (readFileShape, hasDriftedOnDisk, the source endpoint) and MCP's drift gate hashed a file's real bytes, but the index stores the size stamp for a file over the limit, so every unchanged 1-8 MB file read as "changed on disk". oversizeStamp moves to file-limits.ts with an indexedHashInput helper, and all four checks hash what the index stored. An oversize file is now compared without being read. - git-index-currency's "keeps a committed path pending when sync cannot read it" injected its failure only into readFileSync, which the bounded reader no longer calls, so it failed. The failure now reaches openSync too. - The changelog entry drops its benchmark figures. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 1 + __tests__/git-index-currency.test.ts | 14 +++++++--- __tests__/oversize-file-not-read.test.ts | 34 ++++++++++++++++++++++++ src/extraction/index.ts | 15 ++--------- src/file-limits.ts | 23 ++++++++++++++++ src/mcp/tools.ts | 5 +++- src/ui-server/api/source.ts | 14 ++++++---- 7 files changed, 83 insertions(+), 23 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 6547bde1ca..60ef5dfd0b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -174,6 +174,7 @@ and adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). - On Windows, the shared MCP daemon now waits longer for another program — an antivirus scan, an indexer, or another session reading its lock file — to let go of that file, so it starts instead of leaving the session to fall back to a slower in-process server. (#1773) - An MPEG transport stream video that happens to be named `.ts` (golden fixtures under `testdata/`, e2e clips) is now recognised from its first bytes and skipped as non-source instead of being fed to the TypeScript parser — a 900 KB clip used to cost about 28 seconds of CPU per file for no symbols, and a folder of them minutes. Real TypeScript is never affected. (#1910) - A file over the size limit is no longer read before it is skipped: committed video and blob fixtures used to be decoded in full — a 400 MB fixture cost 3.4 GB of memory — only to be discarded, and the same file was read again by every resolution pass. Its size stamp now stands in for its content, during indexing and when checking for changes. (#1910) +- A file over the size limit is no longer read before it is skipped. Large committed video and blob fixtures used to be loaded into memory in full, once per indexing pass, only to be discarded; indexing, change detection and the viewer now go by the file's size instead. (#1910) - An MPEG transport stream video that happens to be named `.ts` (golden fixtures under `testdata/`, e2e clips) is now recognised from its first bytes and skipped instead of being fed to the TypeScript parser, which spent a long time on each clip for no symbols. Real TypeScript files are still indexed. (#1910) - File watching no longer drops the full re-scan a removed directory asks for when that sync fails, so the deleted files leave the index instead of lingering. (#1964) - Daemon startup and cleanup now preserve live legacy PID-only locks while still reclaiming dead or identity-disproved records, preventing two writers from serving the same project. diff --git a/__tests__/git-index-currency.test.ts b/__tests__/git-index-currency.test.ts index fbe79fb722..a5ec39195f 100644 --- a/__tests__/git-index-currency.test.ts +++ b/__tests__/git-index-currency.test.ts @@ -132,12 +132,18 @@ describe('git index currency across commits and restores (#1829)', () => { it('keeps a committed path pending when sync cannot read it', async () => { write('new.ts', 'newSymbol'); commit(); - const real = fs.readFileSync; + // Sync reads a source file through a bounded reader that opens a + // descriptor (#1910), so the failure is injected at openSync as well as + // readFileSync: it must reach whichever one the read goes through. + const realRead = fs.readFileSync; + const realOpen = fs.openSync; let injected = 0; - vi.spyOn(fs, 'readFileSync').mockImplementation(((file: any, ...args: any[]) => { + const failNewTs = (real: (...args: any[]) => unknown) => (file: any, ...args: any[]) => { if (String(file) === path.join(root, 'new.ts')) { injected++; throw new Error('Injected transient read error'); } - return (real as any)(file, ...args); - }) as typeof fs.readFileSync); + return real(file, ...args); + }; + vi.spyOn(fs, 'readFileSync').mockImplementation(failNewTs(realRead as any) as typeof fs.readFileSync); + vi.spyOn(fs, 'openSync').mockImplementation(failNewTs(realOpen as any) as typeof fs.openSync); await cg.sync(); expect(injected).toBeGreaterThan(0); expect(symbols('newSymbol')).not.toContain('newSymbol'); diff --git a/__tests__/oversize-file-not-read.test.ts b/__tests__/oversize-file-not-read.test.ts index 246e2b4458..24c1069c63 100644 --- a/__tests__/oversize-file-not-read.test.ts +++ b/__tests__/oversize-file-not-read.test.ts @@ -4,6 +4,8 @@ import * as os from 'os'; import * as path from 'path'; import { CodeGraph } from '../src'; import { oversizeStamp, hashContent } from '../src/extraction'; +import { hasDriftedOnDisk, readFileShape } from '../src/ui-server/api/source'; +import { ToolHandler } from '../src/mcp/tools'; /** * A file over the size limit is stored as skipped without ever being read @@ -77,3 +79,35 @@ describe('oversize files are stat-gated, never read (#1910)', () => { } }); }); + +describe('an unchanged file over the size limit is not reported as drifted (#1910, #1915 review)', () => { + let dir: string; + afterEach(() => { if (dir) fs.rmSync(dir, { recursive: true, force: true }); }); + + it('reads as current in the viewer and in MCP until its size changes', async () => { + dir = fs.mkdtempSync(path.join(os.tmpdir(), 'codegraph-oversize-drift-')); + fs.writeFileSync(path.join(dir, 'app.ts'), 'export function alpha() { return 1; }\n'); + // 1.4 MB of ordinary text: over the index limit, under the viewer's 8 MB read cap. + const line = 'const x = 1;\n'; + fs.writeFileSync(path.join(dir, 'big.js'), line.repeat(Math.ceil((1.4 * 1024 * 1024) / line.length))); + const cg = await CodeGraph.init(dir, { index: true }); + try { + const record = cg.getFiles().find((f) => f.path === 'big.js')!; + expect(record).toBeDefined(); + // Touch it: the stat fast path no longer answers, so the hash has to. + const later = new Date(Date.now() + 5000); + fs.utimesSync(path.join(dir, 'big.js'), later, later); + + expect(readFileShape(dir, 'big.js', record).drift).toBe(false); + expect(hasDriftedOnDisk(dir, 'big.js', record)).toBe(false); + expect((new ToolHandler(cg) as any).isFileStaleOnDisk(cg, 'big.js')).toBe(false); + + // A different size is a different stamp, and that is drift. + fs.appendFileSync(path.join(dir, 'big.js'), line); + expect(readFileShape(dir, 'big.js', record).drift).toBe(true); + expect(hasDriftedOnDisk(dir, 'big.js', record)).toBe(true); + } finally { + cg.close(); + } + }); +}); diff --git a/src/extraction/index.ts b/src/extraction/index.ts index 0c2d4de16a..1906c83adf 100644 --- a/src/extraction/index.ts +++ b/src/extraction/index.ts @@ -35,7 +35,8 @@ import ignore, { Ignore } from 'ignore'; import { detectFrameworks } from '../resolution/frameworks'; import type { ResolutionContext } from '../resolution/types'; import { createYielder, type MaybeYield } from '../resolution/cooperative-yield'; -import { MAX_SOURCE_FILE_SIZE_BYTES, readBoundedSource, readBoundedSourceSync } from '../file-limits'; +import { MAX_SOURCE_FILE_SIZE_BYTES, oversizeStamp, readBoundedSource, readBoundedSourceSync } from '../file-limits'; +export { oversizeStamp }; /** * Number of files to read in parallel during indexing. @@ -158,18 +159,6 @@ export function hashContent(content: string): string { return crypto.createHash('sha256').update(content).digest('hex'); } -/** - * What stands in for the content of a file over MAX_SOURCE_FILE_SIZE_BYTES. Such a file is - * never parsed, so its bytes are never needed — reading them only to hash and - * discard cost multi-GB RSS spikes on committed video/blob fixtures and could - * fail outright with `Invalid string length` (#1910). The stamp is a function - * of size alone: change detection compares it to the stored hash, so a - * same-size rewrite of an oversize file is not a change (nothing about it is - * indexed), while crossing the limit in either direction is. - */ -export function oversizeStamp(size: number): string { - return `codegraph:oversize:${size}`; -} /** * What change detection hashes for a file: its text when it is under the size diff --git a/src/file-limits.ts b/src/file-limits.ts index ed3136f936..276d7f2021 100644 --- a/src/file-limits.ts +++ b/src/file-limits.ts @@ -8,6 +8,29 @@ import * as fsp from 'fs/promises'; */ export const MAX_SOURCE_FILE_SIZE_BYTES = 1024 * 1024; +/** + * What stands in for the content of a file over MAX_SOURCE_FILE_SIZE_BYTES. Such a file is + * never parsed, so its bytes are never needed — reading them only to hash and + * discard cost multi-GB RSS spikes on committed video/blob fixtures and could + * fail outright with `Invalid string length` (#1910). The stamp is a function + * of size alone: change detection compares it to the stored hash, so a + * same-size rewrite of an oversize file is not a change (nothing about it is + * indexed), while crossing the limit in either direction is. + */ +export function oversizeStamp(size: number): string { + return `codegraph:oversize:${size}`; +} + +/** + * The text whose hash the index stores for a file of `size` bytes: its content + * within the limit, the size stamp over it. Anything that checks a file on disk + * against its stored `contentHash` has to hash this, or every unchanged file + * over the limit reads as drifted. `content` is only read within the limit. + */ +export function indexedHashInput(size: number, content: () => string): string { + return size > MAX_SOURCE_FILE_SIZE_BYTES ? oversizeStamp(size) : content(); +} + /** A source file's stats, and its bytes when they are within the limit (null = oversize). */ export interface BoundedSource { stats: fs.Stats; diff --git a/src/mcp/tools.ts b/src/mcp/tools.ts index 5c99378c85..5deb9bd86c 100644 --- a/src/mcp/tools.ts +++ b/src/mcp/tools.ts @@ -119,6 +119,7 @@ function wslSharedIndexGuidance(err: WslSharedIndexError): string { */ export { PathRefusalError } from '../errors'; import { PathRefusalError } from '../errors'; +import { indexedHashInput } from '../file-limits'; import { resolve as resolvePath, relative as relativePath } from 'path'; /** Maximum output length to prevent context bloat (characters) */ @@ -2500,7 +2501,9 @@ export class ToolHandler { // Same freshness test as the sync fast path (extraction/index.ts): // equal size + equal floored mtime ⇒ unchanged, no read needed. if (st.size !== rec.size || Math.floor(st.mtimeMs) !== Math.floor(rec.modifiedAt)) { - const data = content ?? readFileSync(absPath, 'utf-8'); + // A file over the index's size limit is stored as its size stamp + // (#1910), so it is compared as one, without reading it. + const data = indexedHashInput(st.size, () => content ?? readFileSync(absPath, 'utf-8')); // Must stay byte-identical to extraction's `hashContent` (sha256 over // the utf-8 string) — the identical-rewrite test in // mcp-stale-slice.test.ts pins the parity. Inlined (not imported) diff --git a/src/ui-server/api/source.ts b/src/ui-server/api/source.ts index d2f32069bd..a81ce031e2 100644 --- a/src/ui-server/api/source.ts +++ b/src/ui-server/api/source.ts @@ -39,6 +39,7 @@ import * as path from 'path'; import type { FileRecord } from '../../types'; import type { CodeGraph } from '../../index'; import { resolveProjectFile } from '../security'; +import { indexedHashInput } from '../../file-limits'; import { highlightLines, type HighlightResult } from '../highlight'; import { ApiError, badRequest, intParam, notFound, textParam } from './respond'; @@ -209,9 +210,10 @@ export function hasDriftedOnDisk( if (stats.size === record.size && Math.floor(stats.mtimeMs) === Math.floor(record.modifiedAt)) { return false; } - if (stats.size > MAX_SOURCE_BYTES) return true; - const content = fs.readFileSync(absolute, 'utf-8'); - return createHash('sha256').update(content).digest('hex') !== record.contentHash; + // A file over the index's size limit is stored as its size stamp (#1910), + // so it is compared as one — without reading it. + const hashed = indexedHashInput(stats.size, () => fs.readFileSync(absolute, 'utf-8')); + return createHash('sha256').update(hashed).digest('hex') !== record.contentHash; } catch { return false; } @@ -250,7 +252,8 @@ export function readFileShape( return { drift: false, totalLines: null, reason: 'The file is too large to read here.' }; } const content = fs.readFileSync(absolute, 'utf-8'); - const drift = createHash('sha256').update(content).digest('hex') !== record.contentHash; + const hashed = indexedHashInput(stats.size, () => content); + const drift = createHash('sha256').update(hashed).digest('hex') !== record.contentHash; return { drift, totalLines: splitLines(content).length, @@ -394,7 +397,8 @@ export async function buildSource( // Byte-identical to extraction's `hashContent` (sha256 over the utf-8 // string). A touch or a checkout that rewrote the same bytes must not count // as drift, which is exactly what hashing content rather than mtime buys. - const hash = createHash('sha256').update(content).digest('hex'); + // A file over the index's size limit is stored as its size stamp (#1910). + const hash = createHash('sha256').update(indexedHashInput(stats.size, () => content)).digest('hex'); const drift = hash !== record.contentHash; if (drift && onDrift === 'omit') { return { From 5915534af3f20c332b03e5cc9b783d096d9ebd7e Mon Sep 17 00:00:00 2001 From: danusha2345 Date: Sun, 27 Sep 2026 17:51:11 +0300 Subject: [PATCH 9/9] fix(mcp): check an oversize answer file against its size stamp (#1910) #2043's answer-freshness check hashes each contributing file whole and compares it with the stored contentHash. For a file over the index's size limit the stored hash is the size stamp, so an unchanged oversize file would read as stale. Compare it by its stamp, without reading it, like the viewer and the MCP drift gate do. Co-Authored-By: Claude Opus 5.5 --- __tests__/oversize-file-not-read.test.ts | 4 ++++ src/mcp/answer-freshness.ts | 10 ++++++++++ 2 files changed, 14 insertions(+) diff --git a/__tests__/oversize-file-not-read.test.ts b/__tests__/oversize-file-not-read.test.ts index 24c1069c63..f32267f73f 100644 --- a/__tests__/oversize-file-not-read.test.ts +++ b/__tests__/oversize-file-not-read.test.ts @@ -6,6 +6,7 @@ import { CodeGraph } from '../src'; import { oversizeStamp, hashContent } from '../src/extraction'; import { hasDriftedOnDisk, readFileShape } from '../src/ui-server/api/source'; import { ToolHandler } from '../src/mcp/tools'; +import { validateAnswerFiles } from '../src/mcp/answer-freshness'; /** * A file over the size limit is stored as skipped without ever being read @@ -101,11 +102,14 @@ describe('an unchanged file over the size limit is not reported as drifted (#191 expect(readFileShape(dir, 'big.js', record).drift).toBe(false); expect(hasDriftedOnDisk(dir, 'big.js', record)).toBe(false); expect((new ToolHandler(cg) as any).isFileStaleOnDisk(cg, 'big.js')).toBe(false); + const answer = [{ path: 'big.js', contentHash: record.contentHash }]; + expect((await validateAnswerFiles(dir, answer)).stale).toEqual([]); // A different size is a different stamp, and that is drift. fs.appendFileSync(path.join(dir, 'big.js'), line); expect(readFileShape(dir, 'big.js', record).drift).toBe(true); expect(hasDriftedOnDisk(dir, 'big.js', record)).toBe(true); + expect((await validateAnswerFiles(dir, answer)).stale).toEqual(['big.js']); } finally { cg.close(); } diff --git a/src/mcp/answer-freshness.ts b/src/mcp/answer-freshness.ts index 297eab4728..5f5e224f59 100644 --- a/src/mcp/answer-freshness.ts +++ b/src/mcp/answer-freshness.ts @@ -1,5 +1,7 @@ import { createHash } from 'crypto'; import { createReadStream } from 'fs'; +import { stat } from 'fs/promises'; +import { MAX_SOURCE_FILE_SIZE_BYTES, oversizeStamp } from '../file-limits'; import { validatePathWithinRoot } from '../utils'; export interface AnswerFile { @@ -35,6 +37,14 @@ export async function validateAnswerFiles(root: string, files: AnswerFile[]): Pr const hash = createHash('sha256'); const absolute = validatePathWithinRoot(root, file.path); if (!absolute) { stale.push(file.path); continue; } + // A file over the index's size limit is stored as its size stamp + // (#1910): when the stamp still matches, the file is current without + // being read. Anything else falls through to the bounded read below. + const { size } = await stat(absolute); + if (size > MAX_SOURCE_FILE_SIZE_BYTES && + createHash('sha256').update(oversizeStamp(size)).digest('hex') === file.contentHash) { + continue; + } const stream = createReadStream(absolute, { encoding: 'utf8', highWaterMark: 64 * 1024, signal, });