From 02335e575b8af251f7dd44c259f4803f3a299a78 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Fri, 9 Oct 2026 22:13:30 +0000 Subject: [PATCH 1/2] fix: skip binary and non-UTF-8 files opened as source text Walks for vg build and vg scan no longer decode a blob just because its name looks like source. Those files are skipped, one stderr notice names them and shows a vg --exclude example, and the bytes stay out of the map and the scan artifact. Signed-off-by: Cursor Agent Co-authored-by: vibgrate-team --- DOCS.md | 33 ++++- src/core-open/run-core-scan.ts | 2 + src/core-open/utils/fs.ts | 39 +++++- src/core-open/utils/source-text.ts | 182 ++++++++++++++++++++++++++ src/engine/build.ts | 20 ++- src/engine/discover.ts | 22 ++++ src/engine/docs-ingest.ts | 18 ++- src/engine/literal-scan.ts | 16 ++- src/engine/manifests.ts | 13 +- src/engine/parse-worker.ts | 20 ++- src/engine/pool.ts | 21 ++- src/engine/toolchain/index.ts | 7 +- test/non-utf8-source.test.ts | 198 +++++++++++++++++++++++++++++ 13 files changed, 569 insertions(+), 22 deletions(-) create mode 100644 src/core-open/utils/source-text.ts create mode 100644 test/non-utf8-source.test.ts diff --git a/DOCS.md b/DOCS.md index b6b753ce..e17783c3 100644 --- a/DOCS.md +++ b/DOCS.md @@ -113,6 +113,7 @@ For a quick overview, see the [README](./README.md). This document covers everyt - [Thresholds](#thresholds) - [Scanner Toggles](#scanner-toggles) - [Symlinks](#symlinks) + - [Binary and non-UTF-8 files](#binary-and-non-utf-8-files) - [Extended Scanners](#extended-scanners) - [Platform Matrix](#platform-matrix) - [Dependency Risk](#dependency-risk) @@ -2136,7 +2137,7 @@ Switches (flags that take no value, such as `--vulns`, `--offline` or `--no-grap By default, the scan writes `.vibgrate/scan_result.json`. Use `--no-local-artifacts` or `--max-privacy` to suppress local JSON artifact files. -`vg scan` does not follow symlinks while it indexes the tree. A skipped link is named once on stderr. See [Symlinks](#symlinks). +`vg scan` does not follow symlinks while it indexes the tree. A skipped link is named once on stderr. See [Symlinks](#symlinks). A binary or non-UTF-8 file that would otherwise be read as source is skipped. The scan prints one notice on stderr; when the scan also builds the code map, that build prints its own. See [Binary and non-UTF-8 files](#binary-and-non-utf-8-files). For offline drift scoring, pass `--package-manifest ` with a downloaded manifest bundle such as `https://github.com/vibgrate/manifests/latest-packages.zip`. The manifest shape, the fail-closed errors, and what offline mode skips are in [Offline scan with a package-version manifest](#offline-scan-with-a-package-version-manifest). @@ -2728,7 +2729,7 @@ Maps source code into a graph artifact, enabling all downstream queries (`vg sho | `--attestation ` | `.vibgrate/attestation.intoto.jsonl` | Where `--attest` writes, and where `--verify` reads | | `--pub ` | — | Public key PEM that pins the signer for `--verify` | -`vg build` does not follow symlinks while it discovers files. A skipped link is named once on stderr. See [Symlinks](#symlinks). `.gitignore` and `--exclude` still apply. +`vg build` does not follow symlinks while it discovers files. A skipped link is named once on stderr. See [Symlinks](#symlinks). `.gitignore` and `--exclude` still apply. A binary or non-UTF-8 file that would otherwise be read as source is skipped and named once on stderr. See [Binary and non-UTF-8 files](#binary-and-non-utf-8-files). **Local by default — no git churn.** The first time vg writes into `.vibgrate/` it also creates `.vibgrate/.gitignore`, keeping the graph artifacts (`graph.json`, `graph.html`, `GRAPH_REPORT.md`, `facts.jsonl`, `mcp-navigation.json`) and the cache out of git — so builds, auto-refreshes, and MCP use never leave your branch dirty. Run `vg share` when you want the map committed for your team (it rewrites that ignore file). vg never touches an existing `.vibgrate/.gitignore`, so edit it (or leave it empty) to manage the ignores yourself. @@ -5132,6 +5133,34 @@ text only; it does not hide this notice. notice: skipped 3 symlinks (alias.ts, nested/cycle, via). vg does not follow symlinks. Point the root at the link target, or pass --exclude or a narrower root. ``` +### Binary and non-UTF-8 files + +`vg build` and `vg scan` read source as UTF-8 text. A file under the walk can still be a binary, a media blob, or another encoding, including one whose extension looks like source (`payload.js`, `README.md`). vg does not decode those bytes. It leaves the file out of the map and out of scan text, prints one notice for that walk on stderr, and continues. `vg scan` builds a code map after the scan unless you pass `--no-graph`, so the map build can print a second notice for the same paths. The command does not crash, and neither notice includes file bytes. + +Known media extensions (images, fonts, audio, video, archives) are already left out of the scan walk. This notice is for a file vg would otherwise have opened as text. Output files do not change because of the notice itself, and the exit code stays the same when the rest of the command succeeds. + +The notice is the count, then the first few root-relative paths in sorted order (at most five; further files are a `+N more` count). Paths are relative to the root you passed. Absolute paths are not printed. The line goes to stderr, so `--format json` and `vg build --json` keep a JSON document on stdout. `--quiet` hides promotional text only; it does not hide this notice. + +```text +notice: skipped 2 files that are not UTF-8 text (README.md, payload.js). vg does not read binary or non-UTF-8 files as source. Ignore them with --exclude or a .gitignore rule. Example: vg build --exclude 'README.md' +``` + +To leave a file out without a notice, list it in `.gitignore` or pass `--exclude`. Both commands accept the flag. A config `exclude` list is merged in as well. + +```bash +vg build --exclude 'payload.js' +vg build --exclude '*.bin' --exclude 'vendor-blobs/**' +vg scan --exclude 'legacy/blob.js' +``` + +```gitignore +# binary checked in under a source-like name +payload.js +*.bin +``` + +A source file that is valid UTF-8, including non-ASCII text, is still read. Save a file that uses another encoding as UTF-8 if it should be part of the map. + --- ## Extended Scanners diff --git a/src/core-open/run-core-scan.ts b/src/core-open/run-core-scan.ts index a6d6ebc5..43cc1750 100644 --- a/src/core-open/run-core-scan.ts +++ b/src/core-open/run-core-scan.ts @@ -39,6 +39,7 @@ import { loadConfig, appendExcludePatterns } from './config.js'; import { pathExists, readJsonFile, writeJsonFile, writeTextFile, ensureDir, FileCache, quickTreeCount } from './utils/fs.js'; import { portableValue } from './utils/portable-path.js'; import { assertSafeWalkRoot } from './utils/root-safety.js'; +import { emitSkippedNonUtf8Notice } from './utils/source-text.js'; import { detectVcs } from './utils/vcs.js'; import { isCiEnvironment, hasVibgrateWorkflow } from './utils/ci-env.js'; import { resolveRepositoryName } from './utils/repository-name.js'; @@ -805,6 +806,7 @@ export async function runCoreScan( } } + emitSkippedNonUtf8Notice(fileCache.skippedNonUtf8); fileCache.clear(); if (allProjects.length === 0) { diff --git a/src/core-open/utils/fs.ts b/src/core-open/utils/fs.ts index 5e19c4e3..233e5f94 100644 --- a/src/core-open/utils/fs.ts +++ b/src/core-open/utils/fs.ts @@ -17,7 +17,8 @@ import { UnsafeRootError, type WalkBudgetState, } from './root-safety.js'; -import { emitSkippedSymlinkNotice, rememberSkippedSymlink } from './skipped-symlinks.js'; +import { emitSkippedSymlinkNotice, rememberSkippedSymlink, walkRelativePath } from './skipped-symlinks.js'; +import { isUtf8SourceText } from './source-text.js'; const execFileAsync = promisify(execFile); @@ -399,6 +400,13 @@ export class FileCache { return this._skippedLargeFiles; } + /** Binary or non-UTF-8 files a text read refused. Paths are root-relative. */ + private _skippedNonUtf8: string[] = []; + + get skippedNonUtf8(): readonly string[] { + return this._skippedNonUtf8; + } + // ── Directory walking ── /** @@ -770,7 +778,13 @@ export class FileCache { } } - const content = await fs.readFile(abs, 'utf8'); + const buf = await fs.readFile(abs); + if (!isUtf8SourceText(buf)) { + this.noteNonUtf8(abs); + // Cache the refusal so later scanners do not read the blob again. + return ''; + } + const content = buf.toString('utf8'); if (content.length > TEXT_CACHE_MAX_BYTES) { // Too large for cache — evict so we don't hold it this.textCache.delete(abs); @@ -824,6 +838,15 @@ export class FileCache { this.jsonCache.clear(); this.existsCache.clear(); this.sizeCache.clear(); + this._skippedNonUtf8 = []; + } + + /** Record a text read that refused binary or non-UTF-8 bytes. Path only. */ + private noteNonUtf8(abs: string): void { + const rel = this._rootDir ? walkRelativePath(this._rootDir, abs) : ''; + const label = rel || path.basename(abs); + if (!label || this._skippedNonUtf8.includes(label)) return; + this._skippedNonUtf8.push(label); } /** Number of file content entries currently held */ @@ -1135,12 +1158,18 @@ export function stripBom(text: string): string { } export async function readJsonFile(filePath: string): Promise { - const txt = await fs.readFile(filePath, 'utf8'); - return JSON.parse(stripBom(txt)) as T; + const buf = await fs.readFile(filePath); + if (!isUtf8SourceText(buf)) { + const base = path.basename(filePath); + throw new SyntaxError(`${base || 'file'} is not UTF-8 text`); + } + return JSON.parse(stripBom(buf.toString('utf8'))) as T; } export async function readTextFile(filePath: string): Promise { - return fs.readFile(filePath, 'utf8'); + const buf = await fs.readFile(filePath); + if (!isUtf8SourceText(buf)) return ''; + return buf.toString('utf8'); } export async function pathExists(p: string): Promise { diff --git a/src/core-open/utils/source-text.ts b/src/core-open/utils/source-text.ts new file mode 100644 index 00000000..7aedcf38 --- /dev/null +++ b/src/core-open/utils/source-text.ts @@ -0,0 +1,182 @@ +import * as fs from 'node:fs'; + +/** + * Decide whether bytes are safe to open as source text, and format the one + * notice a walk prints when they are not. + * + * A NUL, a UTF-16 BOM, or any byte sequence that is not UTF-8 means the file + * is binary or another encoding. Callers skip it. The notice names paths + * only — never file bytes — and tells the operator how to ignore the file. + */ + +/** How many leading bytes a large file is judged on before a full read. */ +export const NON_UTF8_PREFIX_BYTES = 8192; + +/** + * Marker on a parse row that was refused because the file is not UTF-8 text. + * The build drops the row. It is not a graph warning and not file contents. + */ +export const NON_UTF8_SKIP_MARK = 'vg-skip-non-utf8'; + +/** How many root-relative paths one notice lists. Further files are a count. */ +export const SKIPPED_NON_UTF8_NOTICE_CAP = 5; + +const HARD_FULL_READ_CAP = 32 * 1024 * 1024; + +function comparePath(a: string, b: string): number { + return a < b ? -1 : a > b ? 1 : 0; +} + +/** + * True when `bytes` must not be decoded as source text. + * `complete` means `bytes` is the whole file. A prefix may end on a cut + * multibyte character; that tail is not, by itself, a reason to skip. + */ +export function isNonUtf8Source(bytes: Uint8Array, complete: boolean): boolean { + if (bytes.length === 0) return false; + if (hasUtf16Bom(bytes)) return true; + for (let i = 0; i < bytes.length; i++) { + if (bytes[i] === 0) return true; + } + const sample = complete ? bytes : trimIncompleteUtf8Tail(bytes); + try { + new TextDecoder('utf-8', { fatal: true }).decode(sample); + return false; + } catch { + return true; + } +} + +/** True when the whole buffer is UTF-8 text with no NUL and no UTF-16 BOM. */ +export function isUtf8SourceText(bytes: Uint8Array): boolean { + return !isNonUtf8Source(bytes, true); +} + +/** + * Read `abs` and return its text, or `null` when the bytes are not UTF-8 + * source text. Throws when the file cannot be read. + */ +export function readUtf8SourceSync(abs: string): string | null { + const buf = fs.readFileSync(abs); + if (!isUtf8SourceText(buf)) return null; + return buf.toString('utf8'); +} + +/** + * True when a walked file should be left out of source-text handling. + * Files up to `fullReadCap` are checked in full. Larger files are judged + * on a prefix so a blob is never slurped just to be refused. `0` means + * prefix-only. Unreadable files return false; the caller already has a + * path for those. + */ +export function shouldSkipNonUtf8File(abs: string, fullReadCap: number): boolean { + let size: number; + try { + size = fs.statSync(abs).size; + } catch { + return false; + } + if (size === 0) return false; + const cap = fullReadCap > 0 ? Math.min(fullReadCap, HARD_FULL_READ_CAP) : 0; + try { + if (cap > 0 && size <= cap) { + return !isUtf8SourceText(fs.readFileSync(abs)); + } + const n = Math.min(size, NON_UTF8_PREFIX_BYTES); + const buf = Buffer.alloc(n); + const fd = fs.openSync(abs, 'r'); + try { + const got = fs.readSync(fd, buf, 0, n, 0); + return isNonUtf8Source(buf.subarray(0, got), got === size); + } finally { + fs.closeSync(fd); + } + } catch { + return false; + } +} + +/** + * One stderr line when a walk skips binary or non-UTF-8 files. `null` when + * `relPaths` is empty. Paths are de-duplicated, sorted, and capped. Absolute + * paths and control characters are omitted. File bytes are never included. + */ +export function formatSkippedNonUtf8Notice(relPaths: readonly string[]): string | null { + const seen = new Set(); + const unique: string[] = []; + let hidden = 0; + for (const raw of relPaths) { + const key = raw.split('\\').join('/'); + if (!key || key === '.' || seen.has(key)) continue; + seen.add(key); + const rel = displayRel(key); + if (!rel) hidden++; + else unique.push(rel); + } + unique.sort(comparePath); + const total = unique.length + hidden; + if (total === 0) return null; + const shown = unique.slice(0, SKIPPED_NON_UTF8_NOTICE_CAP); + const extra = unique.length - shown.length; + const list = extra > 0 ? `${shown.join(', ')}, +${extra} more` : shown.join(', '); + const noun = total === 1 ? 'file that is not UTF-8 text' : 'files that are not UTF-8 text'; + const names = list.length > 0 ? ` (${list})` : ''; + const example = shown[0] ? shellQuote(shown[0]) : "'*.bin'"; + return ( + `notice: skipped ${total} ${noun}${names}. ` + + 'vg does not read binary or non-UTF-8 files as source. ' + + `Ignore them with --exclude or a .gitignore rule. Example: vg build --exclude ${example}` + ); +} + +/** Write {@link formatSkippedNonUtf8Notice} to stderr. No-op when there is nothing to say. */ +export function emitSkippedNonUtf8Notice(relPaths: readonly string[]): void { + const line = formatSkippedNonUtf8Notice(relPaths); + if (!line) return; + process.stderr.write(`${line}\n`); +} + +function hasUtf16Bom(bytes: Uint8Array): boolean { + if (bytes.length < 2) return false; + const b0 = bytes[0]!; + const b1 = bytes[1]!; + return (b0 === 0xff && b1 === 0xfe) || (b0 === 0xfe && b1 === 0xff); +} + +/** Drop a trailing partial UTF-8 sequence so a prefix is not refused for being cut. */ +function trimIncompleteUtf8Tail(bytes: Uint8Array): Uint8Array { + const n = bytes.length; + if (n === 0) return bytes; + let i = n - 1; + let cont = 0; + while (i >= 0 && cont < 3 && (bytes[i]! & 0xc0) === 0x80) { + cont++; + i--; + } + if (i < 0) return bytes; + const lead = bytes[i]!; + let need = 0; + if ((lead & 0x80) === 0) need = 1; + else if ((lead & 0xe0) === 0xc0) need = 2; + else if ((lead & 0xf0) === 0xe0) need = 3; + else if ((lead & 0xf8) === 0xf0) need = 4; + else return bytes; + const have = n - i; + if (have < need) return bytes.subarray(0, i); + return bytes; +} + +/** A path safe to print. `null` when it must not appear in the notice. */ +function displayRel(rel: string): string | null { + if (!rel || rel === '.' || rel.startsWith('../') || rel === '..' || rel.startsWith('/')) return null; + if (/^[A-Za-z]:/.test(rel)) return null; + for (let i = 0; i < rel.length; i++) { + const c = rel.charCodeAt(i); + if (c < 0x20 || c === 0x7f) return null; + } + return rel; +} + +function shellQuote(value: string): string { + return `'${value.replace(/'/g, `'\\''`)}'`; +} diff --git a/src/engine/build.ts b/src/engine/build.ts index 9252c40f..f3eb783f 100644 --- a/src/engine/build.ts +++ b/src/engine/build.ts @@ -50,6 +50,7 @@ import type { FileParse } from './types.js'; import type { ResolveResult } from './resolve.js'; import { fileRolesFromParses } from './ast-roles.js'; import type { AstRoleHit } from '../core-open/scanners/architecture/ast-roles.js'; +import { emitSkippedNonUtf8Notice, NON_UTF8_SKIP_MARK } from '../core-open/utils/source-text.js'; import { stampWarning, WARNING_CODES, type CodedWarning } from '../core-open/warnings.js'; import { assembleEngineWarnings } from './warning-codes.js'; @@ -156,6 +157,15 @@ export interface BuildResult { } export async function buildGraph(options: BuildOptions): Promise { + const skippedNonUtf8: string[] = []; + try { + return await buildGraphUnchecked(options, skippedNonUtf8); + } finally { + emitSkippedNonUtf8Notice(skippedNonUtf8); + } +} + +async function buildGraphUnchecked(options: BuildOptions, skippedNonUtf8: string[]): Promise { const timer = new StageTimer(); timer.start('total'); const root = path.resolve(options.root); @@ -170,6 +180,8 @@ export async function buildGraph(options: BuildOptions): Promise { exclude, paths: options.paths, maxEntries: limits.maxFiles, + maxSourceBytes: limits.maxFileBytes === 0 ? 32 * 1024 * 1024 : limits.maxFileBytes, + skippedNonUtf8, }); timer.end('discover'); @@ -295,12 +307,16 @@ export async function buildGraph(options: BuildOptions): Promise { timer.end('hash'); timer.start('parse'); - const parsedNew = await parseFiles(toParse, { + const parsedNew = (await parseFiles(toParse, { jobs: options.jobs, inline: options.inline, onProgress: options.onParseProgress, grammarsDir: options.grammarsDir, memoryBudgetMb: limits.memoryBudgetMb, + })).filter((parsed) => { + if (!parsed.warnings?.includes(NON_UTF8_SKIP_MARK)) return true; + skippedNonUtf8.push(parsed.rel); + return false; }); timer.end('parse'); checkMemoryBudget('parse', limits.memoryBudgetMb); @@ -339,6 +355,7 @@ export async function buildGraph(options: BuildOptions): Promise { const manifests = extractManifests(root, { exclude, paths: options.paths, + skippedNonUtf8, }); if (manifests.files > 0) { const byId = new Map(resolved.nodes.map((n) => [n.id, n])); @@ -518,6 +535,7 @@ export async function buildGraph(options: BuildOptions): Promise { exclude, paths: options.paths, maxEntries: limits.maxFiles, + skippedNonUtf8, }); for (const d of docs) { try { diff --git a/src/engine/discover.ts b/src/engine/discover.ts index c5113b9a..2f8fa11d 100644 --- a/src/engine/discover.ts +++ b/src/engine/discover.ts @@ -6,7 +6,9 @@ import { requireDataConfig } from '../core-open/config.js'; import { dropBlankPatterns, gitignoreWithoutBlankLines } from '../core-open/utils/glob.js'; import { assertLockfileFile, lockfileKind } from '../core-open/utils/lockfile-parse.js'; import { assertSafeWalkRoot, createWalkBudget, noteWalkEntry } from '../core-open/utils/root-safety.js'; +import { emitSkippedNonUtf8Notice, shouldSkipNonUtf8File } from '../core-open/utils/source-text.js'; import { emitSkippedSymlinkNotice } from '../core-open/utils/skipped-symlinks.js'; +import { DEFAULT_MAX_FILE_BYTES } from './limits.js'; /** * Deterministic file discovery. @@ -167,6 +169,17 @@ export interface DiscoverOptions { * Default: `VG_MAX_FILES`, else 100000. */ maxEntries?: number; + /** + * Largest file fully checked for UTF-8 before it is treated as source. + * Larger files are judged on a prefix. Default: the build's per-file cap. + * `0` checks a prefix only. + */ + maxSourceBytes?: number; + /** + * When set, binary and non-UTF-8 source files are recorded here and this + * call does not print the notice. The caller prints once. + */ + skippedNonUtf8?: string[]; } export interface DiscoveredFile { @@ -253,6 +266,8 @@ export function discover(options: DiscoverOptions): DiscoveredFile[] { // link, so a directory link to its parent cannot re-enter this walk. The // notice lists the ones this walk skipped; ignored links stay quiet. const skippedSymlinks: string[] = []; + const skippedNonUtf8 = options.skippedNonUtf8 ?? []; + const fullReadCap = options.maxSourceBytes ?? DEFAULT_MAX_FILE_BYTES; const considerFile = (abs: string): void => { const rel = toPosix(path.relative(root, abs)); @@ -266,6 +281,12 @@ export function discover(options: DiscoverOptions): DiscoveredFile[] { } const lang = langForExtension(path.extname(abs)); if (!lang || !allowLang(lang)) return; + // A source extension does not make the bytes text. Skip before parse so + // a blob is never decoded, hashed into a symbol, or copied into a warning. + if (rel && shouldSkipNonUtf8File(abs, fullReadCap)) { + skippedNonUtf8.push(rel); + return; + } found.set(rel, { rel, abs, lang }); }; @@ -308,6 +329,7 @@ export function discover(options: DiscoverOptions): DiscoveredFile[] { } emitSkippedSymlinkNotice(skippedSymlinks); + if (!options.skippedNonUtf8) emitSkippedNonUtf8Notice(skippedNonUtf8); return [...found.values()].sort((a, b) => (a.rel < b.rel ? -1 : a.rel > b.rel ? 1 : 0)); } diff --git a/src/engine/docs-ingest.ts b/src/engine/docs-ingest.ts index 64da6fbb..e43d5965 100644 --- a/src/engine/docs-ingest.ts +++ b/src/engine/docs-ingest.ts @@ -21,6 +21,7 @@ import * as fs from 'node:fs'; import * as path from 'node:path'; import { redactSecrets } from '../core-open/utils/redact.js'; +import { emitSkippedNonUtf8Notice, isUtf8SourceText, readUtf8SourceSync } from '../core-open/utils/source-text.js'; import { nodeId } from './ids.js'; import { isSkippedDirName, loadRootIgnore, SKIP_FILES } from './discover.js'; import { assertSafeWalkRoot, createWalkBudget, noteWalkEntry, UnsafeRootError } from '../core-open/utils/root-safety.js'; @@ -57,6 +58,11 @@ export interface DiscoverDocsOptions { maxFiles?: number; /** Walk-entry ceiling. `0` disables. Default: `VG_MAX_FILES`, else 100000. */ maxEntries?: number; + /** + * When set, binary and non-UTF-8 context files are recorded here and this + * call does not print the notice. The caller prints once. + */ + skippedNonUtf8?: string[]; } export interface DiscoveredDoc { @@ -376,6 +382,7 @@ export function discoverDocs(options: DiscoverDocsOptions): DiscoveredDoc[] { const found = new Map(); const budget = createWalkBudget(root, options.maxEntries); + const skippedNonUtf8 = options.skippedNonUtf8 ?? []; const consider = (abs: string): void => { if (found.size >= maxFiles) return; @@ -390,6 +397,12 @@ export function discoverDocs(options: DiscoverDocsOptions): DiscoveredDoc[] { const st = fs.statSync(abs); if (!st.isFile() || st.size > DOC_FILE_MAX_BYTES) return; if (st.size === 0) return; + // Context files are stored as text. Refuse binary / non-UTF-8 before + // any body is copied into a document node. + if (!isUtf8SourceText(fs.readFileSync(abs))) { + skippedNonUtf8.push(rel); + return; + } } catch { return; } @@ -439,6 +452,7 @@ export function discoverDocs(options: DiscoverDocsOptions): DiscoveredDoc[] { } } + if (!options.skippedNonUtf8) emitSkippedNonUtf8Notice(skippedNonUtf8); return [...found.values()].sort((a, b) => a.rel.localeCompare(b.rel)); } @@ -503,7 +517,9 @@ export function documentNodesFromDocs(docs: DiscoveredDoc[]): GraphNode[] { for (const d of docs) { let raw = ''; try { - raw = fs.readFileSync(d.abs, 'utf8'); + const text = readUtf8SourceSync(d.abs); + if (text === null) continue; + raw = text; } catch { continue; } diff --git a/src/engine/literal-scan.ts b/src/engine/literal-scan.ts index 1660e17e..ebf13250 100644 --- a/src/engine/literal-scan.ts +++ b/src/engine/literal-scan.ts @@ -2,6 +2,7 @@ import * as fs from 'node:fs'; import * as path from 'node:path'; import * as os from 'node:os'; import { Worker } from 'node:worker_threads'; +import { isUtf8SourceText } from '../core-open/utils/source-text.js'; /** * The uniform literal-search core for `search_symbols` (VG-LITERAL-INDEX-DESIGN.md @@ -24,7 +25,6 @@ import { Worker } from 'node:worker_threads'; const MAX_FILE_BYTES = 1_000_000; const PREVIEW_CHARS = 120; -const NUL = '\u0000'; // a NUL byte marks a binary file — skipped, same as ripgrep does /** * Fan out to workers only above this many candidate BYTES. Benchmarking showed * that gating on file *count* is wrong: a few thousand small files (≈8 MB) scan @@ -149,8 +149,10 @@ function readCandidate(abs: string, knownSize?: number): string | null { try { const size = knownSize ?? fs.statSync(abs).size; if (size > MAX_FILE_BYTES) return null; - const text = fs.readFileSync(abs, 'utf8'); - return text.includes(NUL) ? null : text; + const buf = fs.readFileSync(abs); + // Binary and non-UTF-8 never become a preview line. + if (!isUtf8SourceText(buf)) return null; + return buf.toString('utf8'); } catch { return null; } @@ -218,15 +220,17 @@ const WORKER_SOURCE = [ `const MAX = ${MAX_FILE_BYTES};`, `const PREVIEW = ${PREVIEW_CHARS};`, `const CAP = ${ROW_CAP};`, - 'const NUL = String.fromCharCode(0);', 'const NL = String.fromCharCode(10);', 'const { root, files, needleLower } = workerData;', 'const hits = []; let total = 0, truncated = false;', 'for (const rel of files) {', ' const abs = path.join(root, rel);', + ' let buf;', + ' try { if (fs.statSync(abs).size > MAX) continue; buf = fs.readFileSync(abs); } catch { continue; }', + ' if (buf.includes(0)) continue;', + ' if (buf.length >= 2 && ((buf[0] === 255 && buf[1] === 254) || (buf[0] === 254 && buf[1] === 255))) continue;', ' let text;', - ' try { if (fs.statSync(abs).size > MAX) continue; text = fs.readFileSync(abs, "utf8"); } catch { continue; }', - ' if (text.includes(NUL)) continue;', + ' try { new TextDecoder("utf-8", { fatal: true }).decode(buf); text = buf.toString("utf8"); } catch { continue; }', ' const low = text.toLowerCase();', ' if (!low.includes(needleLower)) continue;', ' const lowLines = low.split(NL); const rawLines = text.split(NL);', diff --git a/src/engine/manifests.ts b/src/engine/manifests.ts index 23ee62c4..8b56a603 100644 --- a/src/engine/manifests.ts +++ b/src/engine/manifests.ts @@ -23,6 +23,7 @@ import * as path from 'node:path'; import type { Ignore } from 'ignore'; import { XMLParser } from 'fast-xml-parser'; import { nodeId, edgeId } from './ids.js'; +import { isUtf8SourceText } from '../core-open/utils/source-text.js'; import { isSkippedDirName, loadRootIgnore } from './discover.js'; import { parseToml } from '../core-open/utils/toml.js'; import type { GraphEdge, GraphNode } from '../schema.js'; @@ -62,7 +63,7 @@ const emptyNode = ( */ export function extractManifests( root: string, - opts: { exclude?: string[]; paths?: string[] } = {}, + opts: { exclude?: string[]; paths?: string[]; skippedNonUtf8?: string[] } = {}, ): ManifestExtract { const absRoot = path.resolve(root); const ig = loadRootIgnore(absRoot, opts.exclude ?? []); @@ -82,6 +83,16 @@ export function extractManifests( for (const rel of [...found.keys()].sort()) { const abs = found.get(rel)!; const base = path.posix.basename(rel); + let manifestBytes: Buffer; + try { + manifestBytes = fs.readFileSync(abs); + } catch { + continue; + } + if (!isUtf8SourceText(manifestBytes)) { + opts.skippedNonUtf8?.push(rel); + continue; + } try { if (base === 'package.json') { deps += ingestPackageJson(rel, abs, nodes, edges); diff --git a/src/engine/parse-worker.ts b/src/engine/parse-worker.ts index 47904988..77e6c997 100644 --- a/src/engine/parse-worker.ts +++ b/src/engine/parse-worker.ts @@ -1,7 +1,7 @@ -import * as fs from 'node:fs'; import { parseSource } from './parse.js'; import { setGrammarsOverride, resetParser } from './grammars.js'; import type { FileParse } from './types.js'; +import { NON_UTF8_SKIP_MARK, readUtf8SourceSync } from '../core-open/utils/source-text.js'; import { stampWarning, WARNING_CODES } from '../core-open/warnings.js'; /** @@ -30,7 +30,23 @@ export default async function run(payload: ParsePayload): Promise { const out: FileParse[] = []; for (const task of payload.tasks) { try { - const source = fs.readFileSync(task.abs, 'utf8'); + const source = readUtf8SourceSync(task.abs); + if (source === null) { + out.push({ + rel: task.rel, + lang: task.lang, + hash: '', + bytes: 0, + defs: [], + calls: [], + imports: [], + heritage: [], + typeRefs: [], + guards: [], + warnings: [NON_UTF8_SKIP_MARK], + }); + continue; + } out.push(await parseSource(task.rel, task.lang, source)); } catch (err) { // A wasm-level parse crash can leave the language's reused parser diff --git a/src/engine/pool.ts b/src/engine/pool.ts index 9837770c..865ed54e 100644 --- a/src/engine/pool.ts +++ b/src/engine/pool.ts @@ -7,6 +7,7 @@ import { setGrammarsOverride, resetParser } from './grammars.js'; import { checkMemoryBudget, envJobs, envWorkerHeapMb, ResourceLimitError } from './limits.js'; import type { DiscoveredFile } from './discover.js'; import type { FileParse } from './types.js'; +import { NON_UTF8_SKIP_MARK, readUtf8SourceSync } from '../core-open/utils/source-text.js'; import { stampWarning, WARNING_CODES } from '../core-open/warnings.js'; import type { ParseTask } from './parse-worker.js'; @@ -76,8 +77,8 @@ async function parseInline(files: DiscoveredFile[], options: ParseOptions): Prom onProgress?.(0, files.length); for (const file of files) { try { - const source = fs.readFileSync(file.abs, 'utf8'); - out.push(await parseSource(file.rel, file.lang.id, source)); + const source = readUtf8SourceSync(file.abs); + out.push(source === null ? nonUtf8Parse(file) : await parseSource(file.rel, file.lang.id, source)); } catch (err) { // A wasm-level parse crash can leave the language's reused parser // mid-state; drop it so the failure stays contained to this file. @@ -170,6 +171,22 @@ function sortByRel(parses: FileParse[]): FileParse[] { return parses.sort((a, b) => (a.rel < b.rel ? -1 : a.rel > b.rel ? 1 : 0)); } +function nonUtf8Parse(file: DiscoveredFile): FileParse { + return { + rel: file.rel, + lang: file.lang.id, + hash: '', + bytes: 0, + defs: [], + calls: [], + imports: [], + heritage: [], + typeRefs: [], + guards: [], + warnings: [NON_UTF8_SKIP_MARK], + }; +} + function emptyParse(file: DiscoveredFile, warning: string): FileParse { return { rel: file.rel, diff --git a/src/engine/toolchain/index.ts b/src/engine/toolchain/index.ts index 8f3b1aa0..d8cab316 100644 --- a/src/engine/toolchain/index.ts +++ b/src/engine/toolchain/index.ts @@ -1,6 +1,8 @@ import * as fs from 'node:fs'; import { edgeId, nodeId } from '../ids.js'; import type { GraphEdge, GraphNode } from '../../schema.js'; +import { readUtf8SourceSync } from '../../core-open/utils/source-text.js'; +import { stampWarning, WARNING_CODES } from '../../core-open/warnings.js'; import type { DiscoveredDoc } from '../docs-ingest.js'; import { composeExtractor } from './compose.js'; import { dockerfileExtractor } from './dockerfile.js'; @@ -10,7 +12,6 @@ import { terraformExtractor } from './terraform.js'; import { githubActionsExtractor, gitlabCiExtractor } from './workflows.js'; import { linkToolchain } from './link.js'; import { safeDoc, TOOLCHAIN_FILE_MAX_BYTES, toPosix } from './util.js'; -import { stampWarning, WARNING_CODES } from '../../core-open/warnings.js'; import type { ToolchainExtraction, ToolchainExtractor, @@ -139,7 +140,9 @@ async function readExtractions(docs: DiscoveredDoc[]): Promise<{ files: FileExtr try { const stat = fs.statSync(doc.abs); if (stat.size > TOOLCHAIN_FILE_MAX_BYTES) continue; - source = fs.readFileSync(doc.abs, 'utf8'); + const text = readUtf8SourceSync(doc.abs); + if (text === null) continue; + source = text; head = source.slice(0, HEAD_BYTES); } catch { continue; // unreadable — docs-ingest already tolerates this diff --git a/test/non-utf8-source.test.ts b/test/non-utf8-source.test.ts new file mode 100644 index 00000000..3fd9c9a8 --- /dev/null +++ b/test/non-utf8-source.test.ts @@ -0,0 +1,198 @@ +import { describe, it, expect, afterEach, vi } from 'vitest'; +import * as fs from 'node:fs'; +import * as os from 'node:os'; +import * as path from 'node:path'; +import { discover } from '../src/engine/discover.js'; +import { buildGraph } from '../src/engine/build.js'; +import { FileCache } from '../src/core-open/utils/fs.js'; +import { runCoreScan } from '../src/core-open/index.js'; +import { + formatSkippedNonUtf8Notice, + isNonUtf8Source, + isUtf8SourceText, +} from '../src/core-open/utils/source-text.js'; +import { advancedScanHook } from '../src/reporting/advanced-analysis.js'; + +const SECRET = 'SECRET_TOKEN_do_not_leak'; +const PIN = '2020-01-01T00:00:00.000Z'; + +/** Small blob: PNG-like header, a NUL, invalid UTF-8, and an ASCII secret. */ +function binaryBlob(): Buffer { + return Buffer.concat([ + Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a, 0x00]), + Buffer.from(SECRET), + Buffer.from([0xff, 0xfe, 0x80]), + ]); +} + +function writeFixture(opts: { gitignore?: string } = {}): string { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'vg-non-utf8-')); + fs.mkdirSync(path.join(root, 'src')); + fs.writeFileSync(path.join(root, 'src', 'keep.ts'), 'export const keep = "café";\n'); + fs.writeFileSync(path.join(root, 'payload.js'), binaryBlob()); + fs.writeFileSync(path.join(root, 'README.md'), binaryBlob()); + fs.writeFileSync(path.join(root, 'logo.png'), Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a])); + fs.writeFileSync( + path.join(root, 'package.json'), + JSON.stringify({ name: 'binary-fixture', version: '1.0.0' }), + ); + if (opts.gitignore !== undefined) fs.writeFileSync(path.join(root, '.gitignore'), opts.gitignore); + return root; +} + +function captureStderr(run: () => Promise | void): Promise { + const chunks: string[] = []; + const spy = vi.spyOn(process.stderr, 'write').mockImplementation((chunk: string | Uint8Array) => { + chunks.push(typeof chunk === 'string' ? chunk : Buffer.from(chunk).toString('utf8')); + return true; + }); + return Promise.resolve(run()).finally(() => spy.mockRestore()).then(() => chunks.join('')); +} + +const DISCOVER_NOTICE = + "notice: skipped 1 file that is not UTF-8 text (payload.js). vg does not read binary or non-UTF-8 files as source. Ignore them with --exclude or a .gitignore rule. Example: vg build --exclude 'payload.js'\n"; + +const BUILD_NOTICE = + "notice: skipped 2 files that are not UTF-8 text (README.md, payload.js). vg does not read binary or non-UTF-8 files as source. Ignore them with --exclude or a .gitignore rule. Example: vg build --exclude 'README.md'\n"; + +describe('binary and non-UTF-8 source text', () => { + const dirs: string[] = []; + afterEach(() => { + while (dirs.length) fs.rmSync(dirs.pop()!, { recursive: true, force: true }); + }); + + it('classifies UTF-8 text and refuses NUL, UTF-16, and invalid bytes', () => { + expect(isUtf8SourceText(Buffer.from(''))).toBe(true); + expect(isUtf8SourceText(Buffer.from('export const keep = "café";\n'))).toBe(true); + expect(isUtf8SourceText(Buffer.from([0xef, 0xbb, 0xbf, 0x41]))).toBe(true); + expect(isUtf8SourceText(binaryBlob())).toBe(false); + expect(isUtf8SourceText(Buffer.from([0x00]))).toBe(false); + expect(isUtf8SourceText(Buffer.from([0xff, 0xfe, 0x41, 0x00]))).toBe(false); + expect(isNonUtf8Source(Buffer.from([0x41, 0xc3]), false)).toBe(false); + expect(isNonUtf8Source(Buffer.from([0x41, 0xc3]), true)).toBe(true); + expect(isNonUtf8Source(Buffer.from([0xff]), false)).toBe(true); + }); + + it('formats one sorted notice and never prints absolute paths or control bytes', () => { + const notice = formatSkippedNonUtf8Notice([ + 'payload.js', + 'README.md', + 'payload.js', + '/tmp/secret', + 'a\u0000b', + ]); + expect(notice).toBe( + "notice: skipped 4 files that are not UTF-8 text (README.md, payload.js). vg does not read binary or non-UTF-8 files as source. Ignore them with --exclude or a .gitignore rule. Example: vg build --exclude 'README.md'", + ); + expect(notice).not.toContain(SECRET); + expect(notice).not.toContain('/tmp'); + expect(notice).not.toContain('\u0000'); + expect(formatSkippedNonUtf8Notice([])).toBeNull(); + }); + + it('discover skips a binary source file, keeps UTF-8, and prints one stable notice', async () => { + const root = writeFixture(); + dirs.push(root); + let first: string[] = []; + let second: string[] = []; + const stderr1 = await captureStderr(() => { + first = discover({ root }).map((f) => f.rel); + }); + const stderr2 = await captureStderr(() => { + second = discover({ root }).map((f) => f.rel); + }); + expect(first).toEqual(['src/keep.ts']); + expect(second).toEqual(first); + expect(stderr1).toBe(DISCOVER_NOTICE); + expect(stderr2).toBe(stderr1); + expect(stderr1).not.toContain(SECRET); + expect(stderr1).not.toContain(root); + expect(stderr1).not.toContain('logo.png'); + }); + + it('a gitignore or --exclude rule omits the file from the notice', async () => { + const ignored = writeFixture({ gitignore: 'payload.js\n' }); + dirs.push(ignored); + const fromGitignore = await captureStderr(() => { + expect(discover({ root: ignored }).map((f) => f.rel)).toEqual(['src/keep.ts']); + }); + expect(fromGitignore).toBe(''); + + const excluded = writeFixture(); + dirs.push(excluded); + const fromFlag = await captureStderr(() => { + expect(discover({ root: excluded, exclude: ['payload.js'] }).map((f) => f.rel)).toEqual(['src/keep.ts']); + }); + expect(fromFlag).toBe(''); + }); + + it('vg build keeps a deterministic map and does not copy the blob into it', async () => { + const root = writeFixture(); + dirs.push(root); + const run = () => + buildGraph({ + root, + generatedAt: PIN, + inline: true, + noCache: true, + noIndex: true, + noScip: true, + }); + let first!: Awaited>; + let second!: Awaited>; + const stderr1 = await captureStderr(async () => { + first = await run(); + }); + const stderr2 = await captureStderr(async () => { + second = await run(); + }); + + expect(stderr1).toBe(BUILD_NOTICE); + expect(stderr2).toBe(stderr1); + expect(stderr1).not.toContain(SECRET); + expect(stderr1).not.toContain(root); + + const graphText = JSON.stringify(first.graph); + expect(graphText).toBe(JSON.stringify(second.graph)); + expect(graphText).toContain('keep'); + expect(first.graph.nodes.some((n) => n.file === 'src/keep.ts')).toBe(true); + expect(graphText).not.toContain('payload.js'); + expect(graphText).not.toContain('README.md'); + expect(graphText).not.toContain(SECRET); + expect(graphText).not.toContain('vg-skip-non-utf8'); + expect(first.warnings.join('\n')).not.toContain(SECRET); + expect(first.warnings.join('\n')).not.toContain('vg-skip-non-utf8'); + }, 60_000); + + it('vg scan reads past the blob without crashing or copying it', async () => { + const root = writeFixture(); + dirs.push(root); + const logs: string[] = []; + const logSpy = vi.spyOn(console, 'log').mockImplementation((...args: unknown[]) => { + logs.push(args.map((a) => String(a)).join(' ')); + }); + let artifact!: Awaited>; + const stderr = await captureStderr(async () => { + artifact = await runCoreScan( + root, + { format: 'json', concurrency: 1, offline: true, noLocalArtifacts: true, quiet: true }, + advancedScanHook, + ); + }); + logSpy.mockRestore(); + + expect(artifact.projects.length).toBeGreaterThan(0); + const dumped = `${JSON.stringify(artifact)}\n${logs.join('\n')}`; + expect(dumped).not.toContain(SECRET); + expect(stderr).toContain('notice: skipped 1 file that is not UTF-8 text (payload.js)'); + expect(stderr).toContain("vg build --exclude 'payload.js'"); + expect(stderr).not.toContain(SECRET); + expect(stderr).not.toContain(root); + expect(stderr).not.toContain('logo.png'); + + const cache = new FileCache(); + const text = await cache.readTextFile(path.join(root, 'payload.js')); + expect(text).toBe(''); + expect(cache.skippedNonUtf8).toEqual(['payload.js']); + }, 60_000); +}); From 07cb6486071248c1dcb1a36394f897e68e78d81c Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Sat, 10 Oct 2026 22:23:32 +0000 Subject: [PATCH 2/2] fix: skip binary and non-UTF-8 files instead of crashing (#294) A file with a NUL byte, or bytes that are not UTF-8, is left out of the map and out of text parsers. One warning names the count, the first few paths, and how to ignore the files. A lockfile that is binary or not UTF-8 still stops the command. Neither message includes the file's bytes. Signed-off-by: Cursor Agent Co-authored-by: vibgrate-team --- CHANGELOG.md | 2 + DOCS.md | 29 +-- src/core-open/run-core-scan.ts | 8 +- src/core-open/utils/fs.ts | 90 +++++--- src/core-open/utils/lockfile-parse.ts | 19 +- src/core-open/utils/source-text.ts | 182 ---------------- src/core-open/utils/text-bytes.ts | 109 ++++++++++ src/core-open/warnings.ts | 5 + src/engine/build.ts | 49 +++-- src/engine/cache.ts | 4 +- src/engine/discover.ts | 30 +-- src/engine/docs-ingest.ts | 26 +-- src/engine/drift.ts | 32 +-- src/engine/hash-files.ts | 11 +- src/engine/literal-scan.ts | 16 +- src/engine/lockfile.ts | 6 +- src/engine/manifests.ts | 19 +- src/engine/module-resolver.ts | 23 ++- src/engine/parse-worker.ts | 11 +- src/engine/pool.ts | 28 +-- src/engine/toolchain/index.ts | 15 +- test/fixtures/binary-non-utf8/blob.ts | Bin 0 -> 44 bytes test/fixtures/binary-non-utf8/notes.md | 2 + test/non-text-files.test.ts | 276 +++++++++++++++++++++++++ test/non-utf8-source.test.ts | 198 ------------------ test/warning-codes.test.ts | 1 + 26 files changed, 622 insertions(+), 569 deletions(-) delete mode 100644 src/core-open/utils/source-text.ts create mode 100644 src/core-open/utils/text-bytes.ts create mode 100644 test/fixtures/binary-non-utf8/blob.ts create mode 100644 test/fixtures/binary-non-utf8/notes.md create mode 100644 test/non-text-files.test.ts delete mode 100644 test/non-utf8-source.test.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index 887e092f..f579c1d9 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -42,6 +42,8 @@ backward compatible. ### Fixed +- **`vg build` and `vg scan` skip binary and non-UTF-8 files instead of crashing.** A file with a NUL byte, or bytes that are not UTF-8, is left out of the map and out of text parsers. One warning names the count, the first few paths, and how to ignore the files with a `.gitignore` entry or `--exclude`. The warning does not include file contents. A lockfile that is binary or not UTF-8 still stops the command, and that message does not include the file's bytes either. + - **SBOM component order no longer follows scan or filesystem order.** `vg sbom export` and CycloneDX/SPDX graph export sort a component by its Package URL when one is written, otherwise by package name, then by version. diff --git a/DOCS.md b/DOCS.md index e17783c3..a6b0fcf6 100644 --- a/DOCS.md +++ b/DOCS.md @@ -113,7 +113,6 @@ For a quick overview, see the [README](./README.md). This document covers everyt - [Thresholds](#thresholds) - [Scanner Toggles](#scanner-toggles) - [Symlinks](#symlinks) - - [Binary and non-UTF-8 files](#binary-and-non-utf-8-files) - [Extended Scanners](#extended-scanners) - [Platform Matrix](#platform-matrix) - [Dependency Risk](#dependency-risk) @@ -2137,7 +2136,7 @@ Switches (flags that take no value, such as `--vulns`, `--offline` or `--no-grap By default, the scan writes `.vibgrate/scan_result.json`. Use `--no-local-artifacts` or `--max-privacy` to suppress local JSON artifact files. -`vg scan` does not follow symlinks while it indexes the tree. A skipped link is named once on stderr. See [Symlinks](#symlinks). A binary or non-UTF-8 file that would otherwise be read as source is skipped. The scan prints one notice on stderr; when the scan also builds the code map, that build prints its own. See [Binary and non-UTF-8 files](#binary-and-non-utf-8-files). +`vg scan` does not follow symlinks while it indexes the tree. A skipped link is named once on stderr. See [Symlinks](#symlinks). For offline drift scoring, pass `--package-manifest ` with a downloaded manifest bundle such as `https://github.com/vibgrate/manifests/latest-packages.zip`. The manifest shape, the fail-closed errors, and what offline mode skips are in [Offline scan with a package-version manifest](#offline-scan-with-a-package-version-manifest). @@ -2729,7 +2728,7 @@ Maps source code into a graph artifact, enabling all downstream queries (`vg sho | `--attestation ` | `.vibgrate/attestation.intoto.jsonl` | Where `--attest` writes, and where `--verify` reads | | `--pub ` | — | Public key PEM that pins the signer for `--verify` | -`vg build` does not follow symlinks while it discovers files. A skipped link is named once on stderr. See [Symlinks](#symlinks). `.gitignore` and `--exclude` still apply. A binary or non-UTF-8 file that would otherwise be read as source is skipped and named once on stderr. See [Binary and non-UTF-8 files](#binary-and-non-utf-8-files). +`vg build` does not follow symlinks while it discovers files. A skipped link is named once on stderr. See [Symlinks](#symlinks). `.gitignore` and `--exclude` still apply. **Local by default — no git churn.** The first time vg writes into `.vibgrate/` it also creates `.vibgrate/.gitignore`, keeping the graph artifacts (`graph.json`, `graph.html`, `GRAPH_REPORT.md`, `facts.jsonl`, `mcp-navigation.json`) and the cache out of git — so builds, auto-refreshes, and MCP use never leave your branch dirty. Run `vg share` when you want the map committed for your team (it rewrites that ignore file). vg never touches an existing `.vibgrate/.gitignore`, so edit it (or leave it empty) to manage the ignores yourself. @@ -5135,32 +5134,16 @@ notice: skipped 3 symlinks (alias.ts, nested/cycle, via). vg does not follow sym ### Binary and non-UTF-8 files -`vg build` and `vg scan` read source as UTF-8 text. A file under the walk can still be a binary, a media blob, or another encoding, including one whose extension looks like source (`payload.js`, `README.md`). vg does not decode those bytes. It leaves the file out of the map and out of scan text, prints one notice for that walk on stderr, and continues. `vg scan` builds a code map after the scan unless you pass `--no-graph`, so the map build can print a second notice for the same paths. The command does not crash, and neither notice includes file bytes. +`vg build` and `vg scan` read project files as text. A file that contains a NUL byte, or bytes that are not valid UTF-8, is not text. That includes a binary blob saved with a source or manifest extension, and UTF-16 saved as `.md` or `.json`. -Known media extensions (images, fonts, audio, video, archives) are already left out of the scan walk. This notice is for a file vg would otherwise have opened as text. Output files do not change because of the notice itself, and the exit code stays the same when the rest of the command succeeds. +Those files are left out. The build does not parse them and does not put their bytes in the map. The scan does not feed them to text parsers. One warning names the count and the first few root-relative paths, in sorted order (at most five; further files are a `+N more` count). It tells you to leave the files out with a `.gitignore` entry, or to pass `--exclude`. The warning does not include the file's bytes. -The notice is the count, then the first few root-relative paths in sorted order (at most five; further files are a `+N more` count). Paths are relative to the root you passed. Absolute paths are not printed. The line goes to stderr, so `--format json` and `vg build --json` keep a JSON document on stdout. `--quiet` hides promotional text only; it does not hide this notice. +A lockfile that is binary or not UTF-8 still stops the command. That message names the file and tells you to leave it out the same way, or to replace it with a text lockfile. It does not include the file's bytes. ```text -notice: skipped 2 files that are not UTF-8 text (README.md, payload.js). vg does not read binary or non-UTF-8 files as source. Ignore them with --exclude or a .gitignore rule. Example: vg build --exclude 'README.md' +warning [VG_WARN_NON_TEXT_FILE]: skipped 2 files that are binary or not UTF-8 (blob.ts, notes.md). Leave them out with a .gitignore entry, or pass --exclude. ``` -To leave a file out without a notice, list it in `.gitignore` or pass `--exclude`. Both commands accept the flag. A config `exclude` list is merged in as well. - -```bash -vg build --exclude 'payload.js' -vg build --exclude '*.bin' --exclude 'vendor-blobs/**' -vg scan --exclude 'legacy/blob.js' -``` - -```gitignore -# binary checked in under a source-like name -payload.js -*.bin -``` - -A source file that is valid UTF-8, including non-ASCII text, is still read. Save a file that uses another encoding as UTF-8 if it should be part of the map. - --- ## Extended Scanners diff --git a/src/core-open/run-core-scan.ts b/src/core-open/run-core-scan.ts index 43cc1750..810286e7 100644 --- a/src/core-open/run-core-scan.ts +++ b/src/core-open/run-core-scan.ts @@ -37,9 +37,9 @@ import { formatSarif } from './formatters/sarif.js'; import { formatMarkdown } from './formatters/markdown.js'; import { loadConfig, appendExcludePatterns } from './config.js'; import { pathExists, readJsonFile, writeJsonFile, writeTextFile, ensureDir, FileCache, quickTreeCount } from './utils/fs.js'; +import { formatNonTextWarning } from './utils/text-bytes.js'; import { portableValue } from './utils/portable-path.js'; import { assertSafeWalkRoot } from './utils/root-safety.js'; -import { emitSkippedNonUtf8Notice } from './utils/source-text.js'; import { detectVcs } from './utils/vcs.js'; import { isCiEnvironment, hasVibgrateWorkflow } from './utils/ci-env.js'; import { resolveRepositoryName } from './utils/repository-name.js'; @@ -806,7 +806,11 @@ export async function runCoreScan( } } - emitSkippedNonUtf8Notice(fileCache.skippedNonUtf8); + const nonTextMessage = formatNonTextWarning(fileCache.skippedNonTextFiles); + if (nonTextMessage) { + degradations.push(codedWarning(WARNING_CODES.NON_TEXT_FILE, nonTextMessage)); + } + fileCache.clear(); if (allProjects.length === 0) { diff --git a/src/core-open/utils/fs.ts b/src/core-open/utils/fs.ts index 233e5f94..7d87154f 100644 --- a/src/core-open/utils/fs.ts +++ b/src/core-open/utils/fs.ts @@ -17,8 +17,10 @@ import { UnsafeRootError, type WalkBudgetState, } from './root-safety.js'; -import { emitSkippedSymlinkNotice, rememberSkippedSymlink, walkRelativePath } from './skipped-symlinks.js'; -import { isUtf8SourceText } from './source-text.js'; +import { emitSkippedSymlinkNotice, rememberSkippedSymlink } from './skipped-symlinks.js'; +import { inspectUtf8, NonTextFileError } from './text-bytes.js'; +import { lockfileKind, LockfileParseError } from './lockfile-parse.js'; +import { warningPathLabel } from '../warnings.js'; const execFileAsync = promisify(execFile); @@ -41,10 +43,20 @@ interface GitignoreLevel { * appended. Returns `levels` unchanged (no allocation) when the directory has * no `.gitignore`, which is the common case. */ -async function extendGitignoreLevels(dir: string, levels: GitignoreLevel[]): Promise { +async function extendGitignoreLevels( + dir: string, + levels: GitignoreLevel[], + onNonText?: (absPath: string) => void, +): Promise { + const abs = path.join(dir, '.gitignore'); try { - const txt = await fs.readFile(path.join(dir, '.gitignore'), 'utf8'); - const rules = gitignoreWithoutBlankLines(txt); + const buf = await fs.readFile(abs); + const inspected = inspectUtf8(buf); + if (!inspected.ok) { + onNonText?.(abs); + return levels; + } + const rules = gitignoreWithoutBlankLines(inspected.text); if (!rules) return levels; return [...levels, { dir, ig: ignore().add(rules) }]; } catch { @@ -321,6 +333,10 @@ export class FileCache { private _stuckPaths: string[] = []; /** Files skipped because they exceed maxFileSizeToScan */ private _skippedLargeFiles: string[] = []; + /** Absolute paths already recorded as binary or non-UTF-8. */ + private _nonTextAbs = new Set(); + /** Root-relative labels for those files, de-duplicated. */ + private _skippedNonTextFiles: string[] = []; /** Maximum file size (bytes) we will read. 0 = unlimited. */ private _maxFileSize = 0; /** Per-project / per-directory scan timeout in ms. */ @@ -400,11 +416,23 @@ export class FileCache { return this._skippedLargeFiles; } - /** Binary or non-UTF-8 files a text read refused. Paths are root-relative. */ - private _skippedNonUtf8: string[] = []; + /** Files skipped because they are binary or not UTF-8. Labels only, never bytes. */ + get skippedNonTextFiles(): readonly string[] { + return this._skippedNonTextFiles; + } - get skippedNonUtf8(): readonly string[] { - return this._skippedNonUtf8; + private noteNonText(absPath: string): void { + const abs = path.resolve(absPath); + if (this._nonTextAbs.has(abs)) return; + this._nonTextAbs.add(abs); + const root = this._rootDir; + const rel = root ? path.relative(root, abs) : path.basename(abs); + const posix = rel.split(path.sep).join('/'); + const label = warningPathLabel(posix); + // `warningPathLabel` uses "path" when the input is empty or escapes the + // root. A file whose name is actually `path` still has that basename. + if (!label || (label === 'path' && !/(^|\/)path$/.test(posix))) return; + this._skippedNonTextFiles.push(label); } // ── Directory walking ── @@ -503,6 +531,9 @@ export class FileCache { // link to a parent cannot re-enter this walk. One notice is printed after // the walk finishes. The prelude count stays quiet so a scan says it once. const skippedSymlinks: string[] = []; + const noteNonText = (abs: string): void => { + this.noteNonText(abs); + }; async function walk(dir: string, gitignoreLevels: GitignoreLevel[]) { if (budgetError) return; @@ -517,7 +548,7 @@ export class FileCache { // Extend the .gitignore chain with this directory's own file (if any) // BEFORE reading entries, so its rules apply to its own children — // matching git's precedence (deepest applicable .gitignore wins). - const levels = await extendGitignoreLevels(dir, gitignoreLevels); + const levels = await extendGitignoreLevels(dir, gitignoreLevels, noteNonText); // Acquire the semaphore ONLY for the readdir I/O, then release // immediately so parent dirs don't hold slots while awaiting children. @@ -779,12 +810,16 @@ export class FileCache { } const buf = await fs.readFile(abs); - if (!isUtf8SourceText(buf)) { - this.noteNonUtf8(abs); - // Cache the refusal so later scanners do not read the blob again. + const inspected = inspectUtf8(buf); + if (!inspected.ok) { + this.noteNonText(abs); + this.textCache.delete(abs); + // A binary lockfile fails closed. Other files are skipped as empty + // text so a parser cannot quote their bytes. + if (lockfileKind(path.basename(abs))) throw new LockfileParseError(abs, 'binary'); return ''; } - const content = buf.toString('utf8'); + const content = inspected.text; if (content.length > TEXT_CACHE_MAX_BYTES) { // Too large for cache — evict so we don't hold it this.textCache.delete(abs); @@ -810,6 +845,7 @@ export class FileCache { const promise = this.readTextFile(abs).then((txt) => { // Evict raw text — we now have the parsed object this.textCache.delete(abs); + if (this._nonTextAbs.has(abs)) throw new NonTextFileError(abs); return JSON.parse(stripBom(txt)) as T; }); this.jsonCache.set(abs, promise); @@ -838,15 +874,6 @@ export class FileCache { this.jsonCache.clear(); this.existsCache.clear(); this.sizeCache.clear(); - this._skippedNonUtf8 = []; - } - - /** Record a text read that refused binary or non-UTF-8 bytes. Path only. */ - private noteNonUtf8(abs: string): void { - const rel = this._rootDir ? walkRelativePath(this._rootDir, abs) : ''; - const label = rel || path.basename(abs); - if (!label || this._skippedNonUtf8.includes(label)) return; - this._skippedNonUtf8.push(label); } /** Number of file content entries currently held */ @@ -1159,17 +1186,22 @@ export function stripBom(text: string): string { export async function readJsonFile(filePath: string): Promise { const buf = await fs.readFile(filePath); - if (!isUtf8SourceText(buf)) { - const base = path.basename(filePath); - throw new SyntaxError(`${base || 'file'} is not UTF-8 text`); + const inspected = inspectUtf8(buf); + if (!inspected.ok) { + if (lockfileKind(path.basename(filePath))) throw new LockfileParseError(filePath, 'binary'); + throw new NonTextFileError(filePath); } - return JSON.parse(stripBom(buf.toString('utf8'))) as T; + return JSON.parse(stripBom(inspected.text)) as T; } export async function readTextFile(filePath: string): Promise { const buf = await fs.readFile(filePath); - if (!isUtf8SourceText(buf)) return ''; - return buf.toString('utf8'); + const inspected = inspectUtf8(buf); + if (!inspected.ok) { + if (lockfileKind(path.basename(filePath))) throw new LockfileParseError(filePath, 'binary'); + return ''; + } + return inspected.text; } export async function pathExists(p: string): Promise { diff --git a/src/core-open/utils/lockfile-parse.ts b/src/core-open/utils/lockfile-parse.ts index 35a91846..c01106c3 100644 --- a/src/core-open/utils/lockfile-parse.ts +++ b/src/core-open/utils/lockfile-parse.ts @@ -17,6 +17,7 @@ import * as fs from 'node:fs'; import * as path from 'node:path'; import { parse as parseToml } from 'smol-toml'; import { parseAllDocuments } from 'yaml'; +import { inspectUtf8 } from './text-bytes.js'; export type LockfileKind = | 'JSON' @@ -60,11 +61,18 @@ export class LockfileParseError extends Error { readonly kind: string; constructor(filePath: string, kind: string) { - const detail = kind === 'unreadable' ? 'could not be read' : `is truncated or invalid ${kind}`; + const detail = + kind === 'unreadable' + ? 'could not be read' + : kind === 'binary' + ? 'is binary or not UTF-8' + : `is truncated or invalid ${kind}`; const action = kind === 'unreadable' ? 'Check that the file is accessible, then re-run the command.' - : 'Restore or regenerate the file with your package manager, then re-run the command.'; + : kind === 'binary' + ? 'Leave it out with a .gitignore entry or pass --exclude, or replace it with a text lockfile, then re-run the command.' + : 'Restore or regenerate the file with your package manager, then re-run the command.'; super(`${filePath}: lockfile ${detail}. ${action}`); this.name = 'LockfileParseError'; this.filePath = filePath; @@ -97,6 +105,8 @@ export function parseLockfileJson(filePath: string, text: string): unknown { * concern (absence is not an error); this only runs on bytes that were read. */ export function assertLockfileText(filePath: string, text: string, kind: LockfileKind): void { + // A NUL survived a lossy decode. Reject before a parser can quote the bytes. + if (text.includes('\u0000')) throw new LockfileParseError(filePath, 'binary'); switch (kind) { case 'JSON': parseLockfileJson(filePath, text); @@ -136,8 +146,11 @@ export function assertLockfileFile(filePath: string): void { if (!kind) return; let text: string; try { - text = fs.readFileSync(filePath, 'utf8'); + const inspected = inspectUtf8(fs.readFileSync(filePath)); + if (!inspected.ok) throw new LockfileParseError(filePath, 'binary'); + text = inspected.text; } catch (err) { + if (err instanceof LockfileParseError) throw err; const code = (err as NodeJS.ErrnoException).code; if (code === 'ENOENT') return; throw new LockfileParseError(filePath, 'unreadable'); diff --git a/src/core-open/utils/source-text.ts b/src/core-open/utils/source-text.ts deleted file mode 100644 index 7aedcf38..00000000 --- a/src/core-open/utils/source-text.ts +++ /dev/null @@ -1,182 +0,0 @@ -import * as fs from 'node:fs'; - -/** - * Decide whether bytes are safe to open as source text, and format the one - * notice a walk prints when they are not. - * - * A NUL, a UTF-16 BOM, or any byte sequence that is not UTF-8 means the file - * is binary or another encoding. Callers skip it. The notice names paths - * only — never file bytes — and tells the operator how to ignore the file. - */ - -/** How many leading bytes a large file is judged on before a full read. */ -export const NON_UTF8_PREFIX_BYTES = 8192; - -/** - * Marker on a parse row that was refused because the file is not UTF-8 text. - * The build drops the row. It is not a graph warning and not file contents. - */ -export const NON_UTF8_SKIP_MARK = 'vg-skip-non-utf8'; - -/** How many root-relative paths one notice lists. Further files are a count. */ -export const SKIPPED_NON_UTF8_NOTICE_CAP = 5; - -const HARD_FULL_READ_CAP = 32 * 1024 * 1024; - -function comparePath(a: string, b: string): number { - return a < b ? -1 : a > b ? 1 : 0; -} - -/** - * True when `bytes` must not be decoded as source text. - * `complete` means `bytes` is the whole file. A prefix may end on a cut - * multibyte character; that tail is not, by itself, a reason to skip. - */ -export function isNonUtf8Source(bytes: Uint8Array, complete: boolean): boolean { - if (bytes.length === 0) return false; - if (hasUtf16Bom(bytes)) return true; - for (let i = 0; i < bytes.length; i++) { - if (bytes[i] === 0) return true; - } - const sample = complete ? bytes : trimIncompleteUtf8Tail(bytes); - try { - new TextDecoder('utf-8', { fatal: true }).decode(sample); - return false; - } catch { - return true; - } -} - -/** True when the whole buffer is UTF-8 text with no NUL and no UTF-16 BOM. */ -export function isUtf8SourceText(bytes: Uint8Array): boolean { - return !isNonUtf8Source(bytes, true); -} - -/** - * Read `abs` and return its text, or `null` when the bytes are not UTF-8 - * source text. Throws when the file cannot be read. - */ -export function readUtf8SourceSync(abs: string): string | null { - const buf = fs.readFileSync(abs); - if (!isUtf8SourceText(buf)) return null; - return buf.toString('utf8'); -} - -/** - * True when a walked file should be left out of source-text handling. - * Files up to `fullReadCap` are checked in full. Larger files are judged - * on a prefix so a blob is never slurped just to be refused. `0` means - * prefix-only. Unreadable files return false; the caller already has a - * path for those. - */ -export function shouldSkipNonUtf8File(abs: string, fullReadCap: number): boolean { - let size: number; - try { - size = fs.statSync(abs).size; - } catch { - return false; - } - if (size === 0) return false; - const cap = fullReadCap > 0 ? Math.min(fullReadCap, HARD_FULL_READ_CAP) : 0; - try { - if (cap > 0 && size <= cap) { - return !isUtf8SourceText(fs.readFileSync(abs)); - } - const n = Math.min(size, NON_UTF8_PREFIX_BYTES); - const buf = Buffer.alloc(n); - const fd = fs.openSync(abs, 'r'); - try { - const got = fs.readSync(fd, buf, 0, n, 0); - return isNonUtf8Source(buf.subarray(0, got), got === size); - } finally { - fs.closeSync(fd); - } - } catch { - return false; - } -} - -/** - * One stderr line when a walk skips binary or non-UTF-8 files. `null` when - * `relPaths` is empty. Paths are de-duplicated, sorted, and capped. Absolute - * paths and control characters are omitted. File bytes are never included. - */ -export function formatSkippedNonUtf8Notice(relPaths: readonly string[]): string | null { - const seen = new Set(); - const unique: string[] = []; - let hidden = 0; - for (const raw of relPaths) { - const key = raw.split('\\').join('/'); - if (!key || key === '.' || seen.has(key)) continue; - seen.add(key); - const rel = displayRel(key); - if (!rel) hidden++; - else unique.push(rel); - } - unique.sort(comparePath); - const total = unique.length + hidden; - if (total === 0) return null; - const shown = unique.slice(0, SKIPPED_NON_UTF8_NOTICE_CAP); - const extra = unique.length - shown.length; - const list = extra > 0 ? `${shown.join(', ')}, +${extra} more` : shown.join(', '); - const noun = total === 1 ? 'file that is not UTF-8 text' : 'files that are not UTF-8 text'; - const names = list.length > 0 ? ` (${list})` : ''; - const example = shown[0] ? shellQuote(shown[0]) : "'*.bin'"; - return ( - `notice: skipped ${total} ${noun}${names}. ` + - 'vg does not read binary or non-UTF-8 files as source. ' + - `Ignore them with --exclude or a .gitignore rule. Example: vg build --exclude ${example}` - ); -} - -/** Write {@link formatSkippedNonUtf8Notice} to stderr. No-op when there is nothing to say. */ -export function emitSkippedNonUtf8Notice(relPaths: readonly string[]): void { - const line = formatSkippedNonUtf8Notice(relPaths); - if (!line) return; - process.stderr.write(`${line}\n`); -} - -function hasUtf16Bom(bytes: Uint8Array): boolean { - if (bytes.length < 2) return false; - const b0 = bytes[0]!; - const b1 = bytes[1]!; - return (b0 === 0xff && b1 === 0xfe) || (b0 === 0xfe && b1 === 0xff); -} - -/** Drop a trailing partial UTF-8 sequence so a prefix is not refused for being cut. */ -function trimIncompleteUtf8Tail(bytes: Uint8Array): Uint8Array { - const n = bytes.length; - if (n === 0) return bytes; - let i = n - 1; - let cont = 0; - while (i >= 0 && cont < 3 && (bytes[i]! & 0xc0) === 0x80) { - cont++; - i--; - } - if (i < 0) return bytes; - const lead = bytes[i]!; - let need = 0; - if ((lead & 0x80) === 0) need = 1; - else if ((lead & 0xe0) === 0xc0) need = 2; - else if ((lead & 0xf0) === 0xe0) need = 3; - else if ((lead & 0xf8) === 0xf0) need = 4; - else return bytes; - const have = n - i; - if (have < need) return bytes.subarray(0, i); - return bytes; -} - -/** A path safe to print. `null` when it must not appear in the notice. */ -function displayRel(rel: string): string | null { - if (!rel || rel === '.' || rel.startsWith('../') || rel === '..' || rel.startsWith('/')) return null; - if (/^[A-Za-z]:/.test(rel)) return null; - for (let i = 0; i < rel.length; i++) { - const c = rel.charCodeAt(i); - if (c < 0x20 || c === 0x7f) return null; - } - return rel; -} - -function shellQuote(value: string): string { - return `'${value.replace(/'/g, `'\\''`)}'`; -} diff --git a/src/core-open/utils/text-bytes.ts b/src/core-open/utils/text-bytes.ts new file mode 100644 index 00000000..864e5607 --- /dev/null +++ b/src/core-open/utils/text-bytes.ts @@ -0,0 +1,109 @@ +import { splitStampedWarning, stampWarning, warningPathLabel, WARNING_CODES } from '../warnings.js'; + +/** + * Decide whether bytes can be treated as source text. + * + * A NUL byte is binary (the same rule ripgrep uses). Any other byte sequence + * that is not valid UTF-8 is non-UTF-8, including UTF-16. Callers skip those + * files. The decoded string is returned only for valid UTF-8, so a later + * parser error cannot echo the original bytes. + */ + +export type NonTextReason = 'binary' | 'non-utf8'; + +export type Utf8Inspect = + | { ok: true; text: string } + | { ok: false; reason: NonTextReason }; + +// `ignoreBOM` keeps a leading U+FEFF in the string. Callers that already +// strip a BOM (JSON, lockfiles) keep doing that; a decoder that ate it +// would hide the mark from them. +const decoder = () => new TextDecoder('utf-8', { fatal: true, ignoreBOM: true }); + +function containsNul(bytes: Uint8Array): boolean { + const buf = Buffer.isBuffer(bytes) ? bytes : Buffer.from(bytes.buffer, bytes.byteOffset, bytes.byteLength); + return buf.includes(0); +} + +export function inspectUtf8(bytes: Uint8Array): Utf8Inspect { + if (containsNul(bytes)) return { ok: false, reason: 'binary' }; + try { + return { ok: true, text: decoder().decode(bytes) }; + } catch { + return { ok: false, reason: 'non-utf8' }; + } +} + +/** + * A file that was opened as text is binary or not UTF-8. + * The message names the file and how to leave it out. It never includes bytes. + */ +export class NonTextFileError extends Error { + readonly code = 'ENONTEXT' as const; + readonly filePath: string; + + constructor(filePath: string) { + const posix = String(filePath).split('\\').join('/'); + const label = warningPathLabel(posix); + const name = label && !(label === 'path' && !/(^|\/)path$/.test(posix)) ? label : 'file'; + super(`${name} is binary or not UTF-8. Leave it out with a .gitignore entry, or pass --exclude.`); + this.name = 'NonTextFileError'; + this.filePath = filePath; + } +} + +/** How many paths one warning lists. Further files are a count. */ +export const NON_TEXT_NOTICE_CAP = 5; + +function comparePath(a: string, b: string): number { + return a < b ? -1 : a > b ? 1 : 0; +} + +/** + * One warning for every binary or non-UTF-8 file skipped in a run. + * `null` when `relPaths` is empty. Paths are de-duplicated, sorted, and + * capped. The text is paths and the ignore hint only — never file bytes. + */ +export function formatNonTextWarning(relPaths: readonly string[]): string | null { + const seen = new Set(); + const unique: string[] = []; + for (const raw of relPaths) { + const posix = String(raw).split('\\').join('/'); + const label = warningPathLabel(posix); + if (!label) continue; + // `warningPathLabel` uses "path" when the input is empty or escapes the + // root. A file whose name is actually `path` still has that basename. + if (label === 'path' && !/(^|\/)path$/.test(posix)) continue; + if (seen.has(label)) continue; + seen.add(label); + unique.push(label); + } + unique.sort(comparePath); + if (unique.length === 0) return null; + const shown = unique.slice(0, NON_TEXT_NOTICE_CAP); + const extra = unique.length - shown.length; + const list = extra > 0 ? `${shown.join(', ')}, +${extra} more` : shown.join(', '); + const noun = unique.length === 1 ? 'file that is' : 'files that are'; + const leave = unique.length === 1 ? 'Leave it out' : 'Leave them out'; + return ( + `skipped ${unique.length} ${noun} binary or not UTF-8 (${list}). ` + + `${leave} with a .gitignore entry, or pass --exclude.` + ); +} + +/** + * Replace per-file `VG_WARN_NON_TEXT_FILE` stamps (the message is a path) + * with one aggregated warning. Other warnings are unchanged. + */ +export function collapseNonTextWarnings(stored: readonly string[]): string[] { + const rels: string[] = []; + const rest: string[] = []; + for (const line of stored) { + const split = splitStampedWarning(line); + if (split?.code === WARNING_CODES.NON_TEXT_FILE) rels.push(split.message); + else rest.push(line); + } + const message = formatNonTextWarning(rels); + if (message) rest.push(stampWarning(WARNING_CODES.NON_TEXT_FILE, message)); + return rest; +} diff --git a/src/core-open/warnings.ts b/src/core-open/warnings.ts index f2b950bc..a1b4f874 100644 --- a/src/core-open/warnings.ts +++ b/src/core-open/warnings.ts @@ -18,6 +18,11 @@ export const WARNING_CODES = { /** A source file failed to parse. The build continues without its symbols. */ PARSE_FAILED: 'VG_WARN_PARSE_FAILED', + /** + * A file under the walk is binary or not UTF-8. It is left out of the map + * and out of text parsers. The message is a count and paths, never bytes. + */ + NON_TEXT_FILE: 'VG_WARN_NON_TEXT_FILE', /** A file exceeded the per-file size cap and was left out of the map. */ BUILD_FILE_OVERSIZE: 'VG_WARN_BUILD_FILE_OVERSIZE', /** The TypeScript resolver was skipped because the corpus exceeded its file cap. */ diff --git a/src/engine/build.ts b/src/engine/build.ts index f3eb783f..9c040884 100644 --- a/src/engine/build.ts +++ b/src/engine/build.ts @@ -50,8 +50,8 @@ import type { FileParse } from './types.js'; import type { ResolveResult } from './resolve.js'; import { fileRolesFromParses } from './ast-roles.js'; import type { AstRoleHit } from '../core-open/scanners/architecture/ast-roles.js'; -import { emitSkippedNonUtf8Notice, NON_UTF8_SKIP_MARK } from '../core-open/utils/source-text.js'; import { stampWarning, WARNING_CODES, type CodedWarning } from '../core-open/warnings.js'; +import { collapseNonTextWarnings, inspectUtf8 } from '../core-open/utils/text-bytes.js'; import { assembleEngineWarnings } from './warning-codes.js'; export interface BuildOptions { @@ -157,15 +157,6 @@ export interface BuildResult { } export async function buildGraph(options: BuildOptions): Promise { - const skippedNonUtf8: string[] = []; - try { - return await buildGraphUnchecked(options, skippedNonUtf8); - } finally { - emitSkippedNonUtf8Notice(skippedNonUtf8); - } -} - -async function buildGraphUnchecked(options: BuildOptions, skippedNonUtf8: string[]): Promise { const timer = new StageTimer(); timer.start('total'); const root = path.resolve(options.root); @@ -180,8 +171,6 @@ async function buildGraphUnchecked(options: BuildOptions, skippedNonUtf8: string exclude, paths: options.paths, maxEntries: limits.maxFiles, - maxSourceBytes: limits.maxFileBytes === 0 ? 32 * 1024 * 1024 : limits.maxFileBytes, - skippedNonUtf8, }); timer.end('discover'); @@ -221,6 +210,17 @@ async function buildGraphUnchecked(options: BuildOptions, skippedNonUtf8: string const toParse: DiscoveredFile[] = []; const reused: FileParse[] = []; const buildWarnings: string[] = []; + // A binary `.gitignore` contributes no rules. Say so once, with the path + // only — the bytes are not ignore syntax and must not be echoed. + try { + const gitignorePath = path.join(root, '.gitignore'); + if (fs.existsSync(gitignorePath)) { + const inspected = inspectUtf8(fs.readFileSync(gitignorePath)); + if (!inspected.ok) buildWarnings.push(stampWarning(WARNING_CODES.NON_TEXT_FILE, '.gitignore')); + } + } catch { + // Unreadable .gitignore: discover already continued without its rules. + } /** Files skipped for size — in fileStats under a sentinel hash, never in the manifest. */ const oversizeRels = new Set(); let statHits = 0; @@ -288,13 +288,21 @@ async function buildGraphUnchecked(options: BuildOptions, skippedNonUtf8: string for (const p of pendingHash) { const r = byRel.get(p.file.rel); if (!r || !r.ok) continue; - hashes.set(p.file.rel, r.hash); fileStats.push({ rel: p.file.rel, size: p.size, mtimeMs: p.mtimeMs, hash: r.hash, }); + // Binary and non-UTF-8 files stay in the freshness snapshot (so a + // probe does not report them as newly added) but are not parsed and + // are not handed to the TypeScript program. `hashes` is the parsed + // corpus; leaving them out is what keeps tsc from opening the bytes. + if (r.nonText) { + buildWarnings.push(stampWarning(WARNING_CODES.NON_TEXT_FILE, p.file.rel)); + continue; + } + hashes.set(p.file.rel, r.hash); const cached = cache.get(p.file.rel, r.hash, p.file.lang.id); if (cached) { reused.push(cached); @@ -307,16 +315,12 @@ async function buildGraphUnchecked(options: BuildOptions, skippedNonUtf8: string timer.end('hash'); timer.start('parse'); - const parsedNew = (await parseFiles(toParse, { + const parsedNew = await parseFiles(toParse, { jobs: options.jobs, inline: options.inline, onProgress: options.onParseProgress, grammarsDir: options.grammarsDir, memoryBudgetMb: limits.memoryBudgetMb, - })).filter((parsed) => { - if (!parsed.warnings?.includes(NON_UTF8_SKIP_MARK)) return true; - skippedNonUtf8.push(parsed.rel); - return false; }); timer.end('parse'); checkMemoryBudget('parse', limits.memoryBudgetMb); @@ -355,8 +359,8 @@ async function buildGraphUnchecked(options: BuildOptions, skippedNonUtf8: string const manifests = extractManifests(root, { exclude, paths: options.paths, - skippedNonUtf8, }); + for (const rel of manifests.skippedNonText) warnings.push(stampWarning(WARNING_CODES.NON_TEXT_FILE, rel)); if (manifests.files > 0) { const byId = new Map(resolved.nodes.map((n) => [n.id, n])); for (const n of manifests.nodes) { @@ -535,7 +539,6 @@ async function buildGraphUnchecked(options: BuildOptions, skippedNonUtf8: string exclude, paths: options.paths, maxEntries: limits.maxFiles, - skippedNonUtf8, }); for (const d of docs) { try { @@ -547,7 +550,9 @@ async function buildGraphUnchecked(options: BuildOptions, skippedNonUtf8: string /* skip unreadable docs */ } } - const docNodes = documentNodesFromDocs(docs); + const skippedDocs: string[] = []; + const docNodes = documentNodesFromDocs(docs, skippedDocs); + for (const rel of skippedDocs) warnings.push(stampWarning(WARNING_CODES.NON_TEXT_FILE, rel)); if (docNodes.length) nodes = [...nodes, ...docNodes]; // Structural extraction over the same discovered set (engine/toolchain/). @@ -694,7 +699,7 @@ async function buildGraphUnchecked(options: BuildOptions, skippedNonUtf8: string timer.end('total'); const stages = timer.snapshot(); - const assembled = assembleEngineWarnings(warnings); + const assembled = assembleEngineWarnings(collapseNonTextWarnings(warnings)); return { graph, diff --git a/src/engine/cache.ts b/src/engine/cache.ts index 9233a482..fafea3ca 100644 --- a/src/engine/cache.ts +++ b/src/engine/cache.ts @@ -29,7 +29,9 @@ import type { FileParse } from './types.js'; // /6: RawCall carries `awaited`; /5 parses lack it. // /7: Prisma model-delegate writes (`prisma.post.update`) now yield `persist` // duties, so /6 parses of such files differ. -const CACHE_VERSION = 'vg-parse-cache/7'; +// /8: binary and non-UTF-8 files are no longer parsed. A cached parse of +// those bytes must not be reused via the stat fast path. +const CACHE_VERSION = 'vg-parse-cache/8'; interface CacheEntry { hash: string; diff --git a/src/engine/discover.ts b/src/engine/discover.ts index 2f8fa11d..0f73e331 100644 --- a/src/engine/discover.ts +++ b/src/engine/discover.ts @@ -5,10 +5,9 @@ import { langForExtension, langById, type LanguageDef } from './languages.js'; import { requireDataConfig } from '../core-open/config.js'; import { dropBlankPatterns, gitignoreWithoutBlankLines } from '../core-open/utils/glob.js'; import { assertLockfileFile, lockfileKind } from '../core-open/utils/lockfile-parse.js'; +import { inspectUtf8 } from '../core-open/utils/text-bytes.js'; import { assertSafeWalkRoot, createWalkBudget, noteWalkEntry } from '../core-open/utils/root-safety.js'; -import { emitSkippedNonUtf8Notice, shouldSkipNonUtf8File } from '../core-open/utils/source-text.js'; import { emitSkippedSymlinkNotice } from '../core-open/utils/skipped-symlinks.js'; -import { DEFAULT_MAX_FILE_BYTES } from './limits.js'; /** * Deterministic file discovery. @@ -169,17 +168,6 @@ export interface DiscoverOptions { * Default: `VG_MAX_FILES`, else 100000. */ maxEntries?: number; - /** - * Largest file fully checked for UTF-8 before it is treated as source. - * Larger files are judged on a prefix. Default: the build's per-file cap. - * `0` checks a prefix only. - */ - maxSourceBytes?: number; - /** - * When set, binary and non-UTF-8 source files are recorded here and this - * call does not print the notice. The caller prints once. - */ - skippedNonUtf8?: string[]; } export interface DiscoveredFile { @@ -228,8 +216,11 @@ export function loadRootIgnore(root: string, exclude: string[]): Ignore { const gitignorePath = path.join(root, '.gitignore'); try { if (fs.existsSync(gitignorePath)) { - const rules = gitignoreWithoutBlankLines(fs.readFileSync(gitignorePath, 'utf8')); - if (rules) ig.add(rules); + const inspected = inspectUtf8(fs.readFileSync(gitignorePath)); + if (inspected.ok) { + const rules = gitignoreWithoutBlankLines(inspected.text); + if (rules) ig.add(rules); + } } } catch { // Unreadable .gitignore: walk the tree rather than fail the build. @@ -266,8 +257,6 @@ export function discover(options: DiscoverOptions): DiscoveredFile[] { // link, so a directory link to its parent cannot re-enter this walk. The // notice lists the ones this walk skipped; ignored links stay quiet. const skippedSymlinks: string[] = []; - const skippedNonUtf8 = options.skippedNonUtf8 ?? []; - const fullReadCap = options.maxSourceBytes ?? DEFAULT_MAX_FILE_BYTES; const considerFile = (abs: string): void => { const rel = toPosix(path.relative(root, abs)); @@ -281,12 +270,6 @@ export function discover(options: DiscoverOptions): DiscoveredFile[] { } const lang = langForExtension(path.extname(abs)); if (!lang || !allowLang(lang)) return; - // A source extension does not make the bytes text. Skip before parse so - // a blob is never decoded, hashed into a symbol, or copied into a warning. - if (rel && shouldSkipNonUtf8File(abs, fullReadCap)) { - skippedNonUtf8.push(rel); - return; - } found.set(rel, { rel, abs, lang }); }; @@ -329,7 +312,6 @@ export function discover(options: DiscoverOptions): DiscoveredFile[] { } emitSkippedSymlinkNotice(skippedSymlinks); - if (!options.skippedNonUtf8) emitSkippedNonUtf8Notice(skippedNonUtf8); return [...found.values()].sort((a, b) => (a.rel < b.rel ? -1 : a.rel > b.rel ? 1 : 0)); } diff --git a/src/engine/docs-ingest.ts b/src/engine/docs-ingest.ts index e43d5965..90ac209d 100644 --- a/src/engine/docs-ingest.ts +++ b/src/engine/docs-ingest.ts @@ -21,7 +21,7 @@ import * as fs from 'node:fs'; import * as path from 'node:path'; import { redactSecrets } from '../core-open/utils/redact.js'; -import { emitSkippedNonUtf8Notice, isUtf8SourceText, readUtf8SourceSync } from '../core-open/utils/source-text.js'; +import { inspectUtf8 } from '../core-open/utils/text-bytes.js'; import { nodeId } from './ids.js'; import { isSkippedDirName, loadRootIgnore, SKIP_FILES } from './discover.js'; import { assertSafeWalkRoot, createWalkBudget, noteWalkEntry, UnsafeRootError } from '../core-open/utils/root-safety.js'; @@ -58,11 +58,6 @@ export interface DiscoverDocsOptions { maxFiles?: number; /** Walk-entry ceiling. `0` disables. Default: `VG_MAX_FILES`, else 100000. */ maxEntries?: number; - /** - * When set, binary and non-UTF-8 context files are recorded here and this - * call does not print the notice. The caller prints once. - */ - skippedNonUtf8?: string[]; } export interface DiscoveredDoc { @@ -382,7 +377,6 @@ export function discoverDocs(options: DiscoverDocsOptions): DiscoveredDoc[] { const found = new Map(); const budget = createWalkBudget(root, options.maxEntries); - const skippedNonUtf8 = options.skippedNonUtf8 ?? []; const consider = (abs: string): void => { if (found.size >= maxFiles) return; @@ -397,12 +391,6 @@ export function discoverDocs(options: DiscoverDocsOptions): DiscoveredDoc[] { const st = fs.statSync(abs); if (!st.isFile() || st.size > DOC_FILE_MAX_BYTES) return; if (st.size === 0) return; - // Context files are stored as text. Refuse binary / non-UTF-8 before - // any body is copied into a document node. - if (!isUtf8SourceText(fs.readFileSync(abs))) { - skippedNonUtf8.push(rel); - return; - } } catch { return; } @@ -452,7 +440,6 @@ export function discoverDocs(options: DiscoverDocsOptions): DiscoveredDoc[] { } } - if (!options.skippedNonUtf8) emitSkippedNonUtf8Notice(skippedNonUtf8); return [...found.values()].sort((a, b) => a.rel.localeCompare(b.rel)); } @@ -512,14 +499,17 @@ function summarizePackageJson(raw: string): string { /** * Build `document` graph nodes from discovered project-context files. */ -export function documentNodesFromDocs(docs: DiscoveredDoc[]): GraphNode[] { +export function documentNodesFromDocs(docs: DiscoveredDoc[], skippedNonText?: string[]): GraphNode[] { const nodes: GraphNode[] = []; for (const d of docs) { let raw = ''; try { - const text = readUtf8SourceSync(d.abs); - if (text === null) continue; - raw = text; + const inspected = inspectUtf8(fs.readFileSync(d.abs)); + if (!inspected.ok) { + skippedNonText?.push(d.rel.replace(/\\/g, '/')); + continue; + } + raw = inspected.text; } catch { continue; } diff --git a/src/engine/drift.ts b/src/engine/drift.ts index d762992f..b97aab29 100644 --- a/src/engine/drift.ts +++ b/src/engine/drift.ts @@ -1,6 +1,14 @@ import * as fs from 'node:fs'; import * as path from 'node:path'; import { assertSafeWalkRoot } from '../core-open/utils/root-safety.js'; +import { inspectUtf8 } from '../core-open/utils/text-bytes.js'; + +/** Read a manifest as UTF-8. Throws when the file is missing, binary, or not UTF-8. */ +function readProjectText(file: string): string { + const inspected = inspectUtf8(fs.readFileSync(file)); + if (!inspected.ok) throw Object.assign(new Error('non-text'), { code: 'ENONTEXT' }); + return inspected.text; +} /** * Dependency currency (VG-LOCAL-MODELS §9 / VG-DEVELOPMENT-PLAN Phase 2.4). @@ -122,7 +130,7 @@ function npmDeps(files: string[], root: string): DepRecord[] { for (const file of files) { let pkg: { dependencies?: Record; devDependencies?: Record }; try { - pkg = JSON.parse(fs.readFileSync(file, 'utf8')); + pkg = JSON.parse(readProjectText(file)); } catch { continue; } @@ -146,7 +154,7 @@ function installedNpmVersion(dir: string, name: string, root: string): string | for (let i = 0; i < 12; i++) { const p = path.join(cur, 'node_modules', name, 'package.json'); try { - return (JSON.parse(fs.readFileSync(p, 'utf8')) as { version: string }).version; + return (JSON.parse(readProjectText(p)) as { version: string }).version; } catch { /* keep climbing */ } @@ -212,7 +220,7 @@ function installedPypiVersion(root: string, name: string): string | undefined { function installedPhpVersion(root: string, name: string): string | undefined { let data: unknown; try { - data = JSON.parse(fs.readFileSync(path.join(root, 'vendor', 'composer', 'installed.json'), 'utf8')); + data = JSON.parse(readProjectText(path.join(root, 'vendor', 'composer', 'installed.json'))); } catch { return undefined; } @@ -235,7 +243,7 @@ function pypiDeps(files: string[]): DepRecord[] { for (const file of files) { let text: string; try { - text = fs.readFileSync(file, 'utf8'); + text = readProjectText(file); } catch { continue; } @@ -334,7 +342,7 @@ function goDeps(files: string[]): DepRecord[] { for (const mod of files) { let text: string; try { - text = fs.readFileSync(mod, 'utf8'); + text = readProjectText(mod); } catch { continue; } @@ -400,7 +408,7 @@ function cargoDeps(files: string[]): DepRecord[] { for (const file of files) { let text: string; try { - text = fs.readFileSync(file, 'utf8'); + text = readProjectText(file); } catch { continue; } @@ -434,7 +442,7 @@ function rubyDeps(files: string[]): DepRecord[] { for (const file of files) { let text: string; try { - text = fs.readFileSync(file, 'utf8'); + text = readProjectText(file); } catch { continue; } @@ -451,7 +459,7 @@ function phpDeps(files: string[]): DepRecord[] { for (const file of files) { let pkg: { require?: Record; 'require-dev'?: Record }; try { - pkg = JSON.parse(fs.readFileSync(file, 'utf8')); + pkg = JSON.parse(readProjectText(file)); } catch { continue; } @@ -470,7 +478,7 @@ function dotnetDeps(files: string[]): DepRecord[] { for (const file of files) { let text: string; try { - text = fs.readFileSync(file, 'utf8'); + text = readProjectText(file); } catch { continue; } @@ -496,7 +504,7 @@ function swiftDeps(files: string[]): DepRecord[] { for (const file of files) { let text: string; try { - text = fs.readFileSync(file, 'utf8'); + text = readProjectText(file); } catch { continue; } @@ -520,7 +528,7 @@ function dartDeps(files: string[]): DepRecord[] { for (const file of files) { let text: string; try { - text = fs.readFileSync(file, 'utf8'); + text = readProjectText(file); } catch { continue; } @@ -552,7 +560,7 @@ function javaDeps(files: string[]): DepRecord[] { for (const file of files) { let text: string; try { - text = fs.readFileSync(file, 'utf8'); + text = readProjectText(file); } catch { continue; } diff --git a/src/engine/hash-files.ts b/src/engine/hash-files.ts index 5e58ace3..5ebf8170 100644 --- a/src/engine/hash-files.ts +++ b/src/engine/hash-files.ts @@ -5,6 +5,7 @@ import * as fs from 'node:fs/promises'; import * as os from 'node:os'; +import { inspectUtf8, type NonTextReason } from '../core-open/utils/text-bytes.js'; import { hashBytes } from './hash.js'; export interface HashJob { @@ -17,6 +18,8 @@ export interface HashResult { hash: string; /** False if the file could not be read. */ ok: boolean; + /** Set when the bytes are binary or not valid UTF-8. Absent for text. */ + nonText?: NonTextReason; } /** @@ -43,7 +46,13 @@ export async function hashFilesParallel( const job = jobs[i]; try { const buf = await fs.readFile(job.abs); - out[i] = { rel: job.rel, hash: hashBytes(buf), ok: true }; + const inspected = inspectUtf8(buf); + out[i] = { + rel: job.rel, + hash: hashBytes(buf), + ok: true, + ...(inspected.ok ? {} : { nonText: inspected.reason }), + }; } catch { out[i] = { rel: job.rel, hash: '', ok: false }; } diff --git a/src/engine/literal-scan.ts b/src/engine/literal-scan.ts index ebf13250..1660e17e 100644 --- a/src/engine/literal-scan.ts +++ b/src/engine/literal-scan.ts @@ -2,7 +2,6 @@ import * as fs from 'node:fs'; import * as path from 'node:path'; import * as os from 'node:os'; import { Worker } from 'node:worker_threads'; -import { isUtf8SourceText } from '../core-open/utils/source-text.js'; /** * The uniform literal-search core for `search_symbols` (VG-LITERAL-INDEX-DESIGN.md @@ -25,6 +24,7 @@ import { isUtf8SourceText } from '../core-open/utils/source-text.js'; const MAX_FILE_BYTES = 1_000_000; const PREVIEW_CHARS = 120; +const NUL = '\u0000'; // a NUL byte marks a binary file — skipped, same as ripgrep does /** * Fan out to workers only above this many candidate BYTES. Benchmarking showed * that gating on file *count* is wrong: a few thousand small files (≈8 MB) scan @@ -149,10 +149,8 @@ function readCandidate(abs: string, knownSize?: number): string | null { try { const size = knownSize ?? fs.statSync(abs).size; if (size > MAX_FILE_BYTES) return null; - const buf = fs.readFileSync(abs); - // Binary and non-UTF-8 never become a preview line. - if (!isUtf8SourceText(buf)) return null; - return buf.toString('utf8'); + const text = fs.readFileSync(abs, 'utf8'); + return text.includes(NUL) ? null : text; } catch { return null; } @@ -220,17 +218,15 @@ const WORKER_SOURCE = [ `const MAX = ${MAX_FILE_BYTES};`, `const PREVIEW = ${PREVIEW_CHARS};`, `const CAP = ${ROW_CAP};`, + 'const NUL = String.fromCharCode(0);', 'const NL = String.fromCharCode(10);', 'const { root, files, needleLower } = workerData;', 'const hits = []; let total = 0, truncated = false;', 'for (const rel of files) {', ' const abs = path.join(root, rel);', - ' let buf;', - ' try { if (fs.statSync(abs).size > MAX) continue; buf = fs.readFileSync(abs); } catch { continue; }', - ' if (buf.includes(0)) continue;', - ' if (buf.length >= 2 && ((buf[0] === 255 && buf[1] === 254) || (buf[0] === 254 && buf[1] === 255))) continue;', ' let text;', - ' try { new TextDecoder("utf-8", { fatal: true }).decode(buf); text = buf.toString("utf8"); } catch { continue; }', + ' try { if (fs.statSync(abs).size > MAX) continue; text = fs.readFileSync(abs, "utf8"); } catch { continue; }', + ' if (text.includes(NUL)) continue;', ' const low = text.toLowerCase();', ' if (!low.includes(needleLower)) continue;', ' const lowLines = low.split(NL); const rawLines = text.split(NL);', diff --git a/src/engine/lockfile.ts b/src/engine/lockfile.ts index bcc88606..66ee9b7f 100644 --- a/src/engine/lockfile.ts +++ b/src/engine/lockfile.ts @@ -6,6 +6,7 @@ import { LockfileParseError, parseLockfileJson, } from '../core-open/utils/lockfile-parse.js'; +import { inspectUtf8 } from '../core-open/utils/text-bytes.js'; import { emitUnknownOptionalLockfileWarnings, lockfileWarningPath, @@ -31,8 +32,11 @@ import type { DepRecord } from './drift.js'; */ function readOptionalText(abs: string): string | undefined { try { - return fs.readFileSync(abs, 'utf8'); + const inspected = inspectUtf8(fs.readFileSync(abs)); + if (!inspected.ok) throw new LockfileParseError(abs, 'binary'); + return inspected.text; } catch (err) { + if (err instanceof LockfileParseError) throw err; if ((err as NodeJS.ErrnoException).code === 'ENOENT') return undefined; throw new LockfileParseError(abs, 'unreadable'); } diff --git a/src/engine/manifests.ts b/src/engine/manifests.ts index 8b56a603..373bf8c3 100644 --- a/src/engine/manifests.ts +++ b/src/engine/manifests.ts @@ -23,8 +23,8 @@ import * as path from 'node:path'; import type { Ignore } from 'ignore'; import { XMLParser } from 'fast-xml-parser'; import { nodeId, edgeId } from './ids.js'; -import { isUtf8SourceText } from '../core-open/utils/source-text.js'; import { isSkippedDirName, loadRootIgnore } from './discover.js'; +import { inspectUtf8 } from '../core-open/utils/text-bytes.js'; import { parseToml } from '../core-open/utils/toml.js'; import type { GraphEdge, GraphNode } from '../schema.js'; @@ -34,6 +34,8 @@ export interface ManifestExtract { /** Number of manifest files processed. */ files: number; deps: number; + /** Repo-relative paths left unread because the bytes are binary or not UTF-8. */ + skippedNonText: string[]; } const xml = new XMLParser({ ignoreAttributes: false, attributeNamePrefix: '@_' }); @@ -63,7 +65,7 @@ const emptyNode = ( */ export function extractManifests( root: string, - opts: { exclude?: string[]; paths?: string[]; skippedNonUtf8?: string[] } = {}, + opts: { exclude?: string[]; paths?: string[] } = {}, ): ManifestExtract { const absRoot = path.resolve(root); const ig = loadRootIgnore(absRoot, opts.exclude ?? []); @@ -78,21 +80,21 @@ export function extractManifests( const nodes = new Map(); const edges = new Map(); + const skippedNonText: string[] = []; let deps = 0; for (const rel of [...found.keys()].sort()) { const abs = found.get(rel)!; const base = path.posix.basename(rel); - let manifestBytes: Buffer; try { - manifestBytes = fs.readFileSync(abs); + const inspected = inspectUtf8(fs.readFileSync(abs)); + if (!inspected.ok) { + skippedNonText.push(rel); + continue; + } } catch { continue; } - if (!isUtf8SourceText(manifestBytes)) { - opts.skippedNonUtf8?.push(rel); - continue; - } try { if (base === 'package.json') { deps += ingestPackageJson(rel, abs, nodes, edges); @@ -120,6 +122,7 @@ export function extractManifests( ), files: found.size, deps, + skippedNonText, }; } diff --git a/src/engine/module-resolver.ts b/src/engine/module-resolver.ts index e00bf8e5..4427f361 100644 --- a/src/engine/module-resolver.ts +++ b/src/engine/module-resolver.ts @@ -1,5 +1,6 @@ import * as fs from 'node:fs'; import * as path from 'node:path'; +import { inspectUtf8 } from '../core-open/utils/text-bytes.js'; /** * Module resolution for import edges. Resolves an import specifier (relative, @@ -350,7 +351,9 @@ function readTsconfigChain(root: string, file: string, seen: Set): TsPat seen.add(file); let cfg: { extends?: string; compilerOptions?: { baseUrl?: string; paths?: Record } }; try { - cfg = parseJsonc(fs.readFileSync(file, 'utf8')); + const inspected = inspectUtf8(fs.readFileSync(file)); + if (!inspected.ok) return null; + cfg = parseJsonc(inspected.text); } catch { return null; } @@ -383,16 +386,26 @@ function loadWorkspacePackages(root: string): Map { const rootPkg = path.join(root, 'package.json'); if (fs.existsSync(rootPkg)) { try { - const pkg = JSON.parse(fs.readFileSync(rootPkg, 'utf8')) as { workspaces?: string[] | { packages?: string[] } }; - const ws = Array.isArray(pkg.workspaces) ? pkg.workspaces : pkg.workspaces?.packages; - if (ws) globs.push(...ws); + const inspected = inspectUtf8(fs.readFileSync(rootPkg)); + if (inspected.ok) { + const pkg = JSON.parse(inspected.text) as { workspaces?: string[] | { packages?: string[] } }; + const ws = Array.isArray(pkg.workspaces) ? pkg.workspaces : pkg.workspaces?.packages; + if (ws) globs.push(...ws); + } } catch { /* ignore */ } } const pnpmWs = path.join(root, 'pnpm-workspace.yaml'); if (fs.existsSync(pnpmWs)) { - for (const line of fs.readFileSync(pnpmWs, 'utf8').split('\n')) { + let wsText = ''; + try { + const inspected = inspectUtf8(fs.readFileSync(pnpmWs)); + if (inspected.ok) wsText = inspected.text; + } catch { + wsText = ''; + } + for (const line of wsText.split('\n')) { const m = /^\s*-\s*['"]?([^'"\n]+)['"]?\s*$/.exec(line); if (m) globs.push(m[1].trim()); } diff --git a/src/engine/parse-worker.ts b/src/engine/parse-worker.ts index 77e6c997..f3604658 100644 --- a/src/engine/parse-worker.ts +++ b/src/engine/parse-worker.ts @@ -1,7 +1,8 @@ +import * as fs from 'node:fs'; import { parseSource } from './parse.js'; import { setGrammarsOverride, resetParser } from './grammars.js'; import type { FileParse } from './types.js'; -import { NON_UTF8_SKIP_MARK, readUtf8SourceSync } from '../core-open/utils/source-text.js'; +import { inspectUtf8 } from '../core-open/utils/text-bytes.js'; import { stampWarning, WARNING_CODES } from '../core-open/warnings.js'; /** @@ -30,8 +31,8 @@ export default async function run(payload: ParsePayload): Promise { const out: FileParse[] = []; for (const task of payload.tasks) { try { - const source = readUtf8SourceSync(task.abs); - if (source === null) { + const inspected = inspectUtf8(fs.readFileSync(task.abs)); + if (!inspected.ok) { out.push({ rel: task.rel, lang: task.lang, @@ -43,11 +44,11 @@ export default async function run(payload: ParsePayload): Promise { heritage: [], typeRefs: [], guards: [], - warnings: [NON_UTF8_SKIP_MARK], + warnings: [stampWarning(WARNING_CODES.NON_TEXT_FILE, task.rel)], }); continue; } - out.push(await parseSource(task.rel, task.lang, source)); + out.push(await parseSource(task.rel, task.lang, inspected.text)); } catch (err) { // A wasm-level parse crash can leave the language's reused parser // mid-state; drop it so the failure stays contained to this file. diff --git a/src/engine/pool.ts b/src/engine/pool.ts index 865ed54e..0b3c1952 100644 --- a/src/engine/pool.ts +++ b/src/engine/pool.ts @@ -7,7 +7,7 @@ import { setGrammarsOverride, resetParser } from './grammars.js'; import { checkMemoryBudget, envJobs, envWorkerHeapMb, ResourceLimitError } from './limits.js'; import type { DiscoveredFile } from './discover.js'; import type { FileParse } from './types.js'; -import { NON_UTF8_SKIP_MARK, readUtf8SourceSync } from '../core-open/utils/source-text.js'; +import { inspectUtf8 } from '../core-open/utils/text-bytes.js'; import { stampWarning, WARNING_CODES } from '../core-open/warnings.js'; import type { ParseTask } from './parse-worker.js'; @@ -77,8 +77,14 @@ async function parseInline(files: DiscoveredFile[], options: ParseOptions): Prom onProgress?.(0, files.length); for (const file of files) { try { - const source = readUtf8SourceSync(file.abs); - out.push(source === null ? nonUtf8Parse(file) : await parseSource(file.rel, file.lang.id, source)); + const inspected = inspectUtf8(fs.readFileSync(file.abs)); + if (!inspected.ok) { + out.push(emptyParse(file, stampWarning(WARNING_CODES.NON_TEXT_FILE, file.rel))); + onProgress?.(out.length, files.length); + if (out.length % MEM_CHECK_EVERY === 0) checkMemoryBudget('parse', memoryBudgetMb); + continue; + } + out.push(await parseSource(file.rel, file.lang.id, inspected.text)); } catch (err) { // A wasm-level parse crash can leave the language's reused parser // mid-state; drop it so the failure stays contained to this file. @@ -171,22 +177,6 @@ function sortByRel(parses: FileParse[]): FileParse[] { return parses.sort((a, b) => (a.rel < b.rel ? -1 : a.rel > b.rel ? 1 : 0)); } -function nonUtf8Parse(file: DiscoveredFile): FileParse { - return { - rel: file.rel, - lang: file.lang.id, - hash: '', - bytes: 0, - defs: [], - calls: [], - imports: [], - heritage: [], - typeRefs: [], - guards: [], - warnings: [NON_UTF8_SKIP_MARK], - }; -} - function emptyParse(file: DiscoveredFile, warning: string): FileParse { return { rel: file.rel, diff --git a/src/engine/toolchain/index.ts b/src/engine/toolchain/index.ts index d8cab316..13ea0377 100644 --- a/src/engine/toolchain/index.ts +++ b/src/engine/toolchain/index.ts @@ -1,8 +1,6 @@ import * as fs from 'node:fs'; import { edgeId, nodeId } from '../ids.js'; import type { GraphEdge, GraphNode } from '../../schema.js'; -import { readUtf8SourceSync } from '../../core-open/utils/source-text.js'; -import { stampWarning, WARNING_CODES } from '../../core-open/warnings.js'; import type { DiscoveredDoc } from '../docs-ingest.js'; import { composeExtractor } from './compose.js'; import { dockerfileExtractor } from './dockerfile.js'; @@ -12,6 +10,8 @@ import { terraformExtractor } from './terraform.js'; import { githubActionsExtractor, gitlabCiExtractor } from './workflows.js'; import { linkToolchain } from './link.js'; import { safeDoc, TOOLCHAIN_FILE_MAX_BYTES, toPosix } from './util.js'; +import { inspectUtf8 } from '../../core-open/utils/text-bytes.js'; +import { stampWarning, WARNING_CODES } from '../../core-open/warnings.js'; import type { ToolchainExtraction, ToolchainExtractor, @@ -137,12 +137,16 @@ async function readExtractions(docs: DiscoveredDoc[]): Promise<{ files: FileExtr for (const doc of ordered) { let source: string; let head: string; + const rel = toPosix(doc.rel); try { const stat = fs.statSync(doc.abs); if (stat.size > TOOLCHAIN_FILE_MAX_BYTES) continue; - const text = readUtf8SourceSync(doc.abs); - if (text === null) continue; - source = text; + const inspected = inspectUtf8(fs.readFileSync(doc.abs)); + if (!inspected.ok) { + warnings.push(stampWarning(WARNING_CODES.NON_TEXT_FILE, rel)); + continue; + } + source = inspected.text; head = source.slice(0, HEAD_BYTES); } catch { continue; // unreadable — docs-ingest already tolerates this @@ -151,7 +155,6 @@ async function readExtractions(docs: DiscoveredDoc[]): Promise<{ files: FileExtr const extractor = extractorFor(doc, head); if (!extractor) continue; - const rel = toPosix(doc.rel); try { files.push({ rel, format: extractor.format, extraction: await extractor.extract(rel, source) }); } catch (err) { diff --git a/test/fixtures/binary-non-utf8/blob.ts b/test/fixtures/binary-non-utf8/blob.ts new file mode 100644 index 0000000000000000000000000000000000000000..6b72319899fa74426063cfa160cbf3747ec67bf4 GIT binary patch literal 44 zcmeAS@N?(olHy`u`2UZgAw8oY-pSL?F(}f>$KNT~)j7yD#K6$V*u>P#+` { + while (dirs.length) cleanup(dirs.pop()!); +}); + +function tempDir(): string { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'vg-non-text-')); + dirs.push(dir); + return dir; +} + +/** Project with a binary `.ts`, a non-UTF-8 `.md`, and the same blob as compose. */ +function nonTextProject(): string { + const root = tempDir(); + fs.mkdirSync(path.join(root, 'src')); + fs.writeFileSync(path.join(root, 'src/ok.ts'), 'export function kept(){ return 1 }\n'); + fs.writeFileSync( + path.join(root, 'package.json'), + JSON.stringify({ name: 'non-text-fixture', version: '1.0.0' }) + '\n', + ); + fs.copyFileSync(path.join(fixtureDir, 'blob.ts'), path.join(root, 'blob.ts')); + fs.copyFileSync(path.join(fixtureDir, 'notes.md'), path.join(root, 'notes.md')); + fs.copyFileSync(path.join(fixtureDir, 'blob.ts'), path.join(root, 'docker-compose.yml')); + return root; +} + +function nonTextWarning(result: { codedWarnings: { code: string; message: string }[] }): string { + const hits = result.codedWarnings.filter((warning) => warning.code === 'VG_WARN_NON_TEXT_FILE'); + expect(hits).toHaveLength(1); + return hits[0]!.message; +} + +function assertNoSecrets(text: string): void { + expect(text).not.toContain(BLOB_SECRET); + expect(text).not.toContain(NOTES_SECRET); + expect(text).not.toContain('ghp_'); + expect(text).not.toContain('sk-NONUTF8'); +} + +describe('inspectUtf8', () => { + it('accepts UTF-8, including empty text and a BOM', () => { + expect(inspectUtf8(Buffer.from(''))).toEqual({ ok: true, text: '' }); + expect(inspectUtf8(Buffer.from('kept'))).toEqual({ ok: true, text: 'kept' }); + const bom = Buffer.concat([Buffer.from([0xef, 0xbb, 0xbf]), Buffer.from('a')]); + expect(inspectUtf8(bom)).toEqual({ ok: true, text: '\uFEFFa' }); + }); + + it('rejects a NUL as binary and invalid UTF-8 that has no NUL', () => { + const binary = inspectUtf8(Buffer.from([0x68, 0x00, 0x69])); + expect(binary).toEqual({ ok: false, reason: 'binary' }); + const nonUtf8 = inspectUtf8(Buffer.from([0xff, 0xfe, 0x80])); + expect(nonUtf8).toEqual({ ok: false, reason: 'non-utf8' }); + }); + + it('names skipped paths in one warning and never the bytes', () => { + expect(formatNonTextWarning([])).toBeNull(); + const paths = ['notes.md', 'blob.ts', 'notes.md', 'c.ts', 'a.ts', 'b.ts', 'd.ts', 'e.ts']; + const message = formatNonTextWarning(paths); + expect(NON_TEXT_NOTICE_CAP).toBe(5); + expect(message).toBe( + 'skipped 7 files that are binary or not UTF-8 (a.ts, b.ts, blob.ts, c.ts, d.ts, +2 more). ' + + 'Leave them out with a .gitignore entry, or pass --exclude.', + ); + expect(formatNonTextWarning(['blob.ts'])).toBe( + 'skipped 1 file that is binary or not UTF-8 (blob.ts). Leave it out with a .gitignore entry, or pass --exclude.', + ); + assertNoSecrets(message ?? ''); + }); +}); + +describe('binary and non-UTF-8 files', () => { + it('vg build skips them, keeps text, and repeats the same warning', async () => { + const root = nonTextProject(); + const build = () => + buildGraph({ + root, + generatedAt: PIN, + inline: true, + noCache: true, + noIndex: true, + }); + const first = await build(); + const second = await build(); + const message = nonTextWarning(first); + expect(message).toBe( + 'skipped 3 files that are binary or not UTF-8 (blob.ts, docker-compose.yml, notes.md). ' + + 'Leave them out with a .gitignore entry, or pass --exclude.', + ); + expect(nonTextWarning(second)).toBe(message); + expect(first.graph.nodes.some((node) => node.qualifiedName === 'kept')).toBe(true); + const rendered = serializeGraph(first.graph) + '\n' + first.warnings.join('\n'); + expect(serializeGraph(second.graph)).toBe(serializeGraph(first.graph)); + assertNoSecrets(rendered); + expect(rendered).not.toContain('\u0000'); + }); + + it('a cached build still skips them and matches the cold graph', async () => { + const root = nonTextProject(); + const cold = await buildGraph({ root, generatedAt: PIN, inline: true, noCache: true, noIndex: true }); + const warm = await buildGraph({ root, generatedAt: PIN, inline: true, noCache: false, noIndex: true }); + const again = await buildGraph({ root, generatedAt: PIN, inline: true, noCache: false, noIndex: true }); + expect(nonTextWarning(warm)).toBe(nonTextWarning(cold)); + expect(nonTextWarning(again)).toBe(nonTextWarning(cold)); + expect(serializeGraph(warm.graph)).toBe(serializeGraph(cold.graph)); + expect(serializeGraph(again.graph)).toBe(serializeGraph(cold.graph)); + }); + + it('gitignore and --exclude drop the named file from the warning', async () => { + const ignored = nonTextProject(); + fs.writeFileSync(path.join(ignored, '.gitignore'), 'blob.ts\n'); + const byIgnore = await buildGraph({ + root: ignored, + generatedAt: PIN, + inline: true, + noCache: true, + noIndex: true, + }); + expect(nonTextWarning(byIgnore)).toBe( + 'skipped 2 files that are binary or not UTF-8 (docker-compose.yml, notes.md). ' + + 'Leave them out with a .gitignore entry, or pass --exclude.', + ); + + const excluded = nonTextProject(); + const byFlag = await buildGraph({ + root: excluded, + generatedAt: PIN, + inline: true, + noCache: true, + noIndex: true, + exclude: ['blob.ts'], + }); + expect(nonTextWarning(byFlag)).toBe(nonTextWarning(byIgnore)); + }); + + it('a text project has no non-text warning', async () => { + const root = tempDir(); + fs.writeFileSync(path.join(root, 'ok.ts'), 'export function kept(){ return 1 }\n'); + const result = await buildGraph({ root, generatedAt: PIN, inline: true, noCache: true, noIndex: true }); + expect(result.codedWarnings.some((warning) => warning.code === 'VG_WARN_NON_TEXT_FILE')).toBe(false); + expect(result.graph.nodes.some((node) => node.qualifiedName === 'kept')).toBe(true); + }); + + it('a binary lockfile fails closed and does not echo its bytes', () => { + const root = tempDir(); + fs.copyFileSync(path.join(fixtureDir, 'blob.ts'), path.join(root, 'package-lock.json')); + fs.writeFileSync(path.join(root, 'app.ts'), 'export const n = 1;\n'); + let caught: unknown; + try { + discover({ root }); + } catch (err) { + caught = err; + } + expect(caught).toBeInstanceOf(LockfileParseError); + const message = caught instanceof Error ? caught.message : String(caught); + expect(message).toContain('package-lock.json'); + expect(message).toContain('binary or not UTF-8'); + expect(message).toContain('.gitignore'); + expect(message).toContain('--exclude'); + assertNoSecrets(message); + }); + + it('readers skip non-text and fail a binary lockfile without its bytes', async () => { + const root = nonTextProject(); + const cache = new FileCache(); + await cache.walkDir(root); + expect(await cache.readTextFile(path.join(root, 'docker-compose.yml'))).toBe(''); + expect(await readTextFile(path.join(root, 'notes.md'))).toBe(''); + expect(cache.skippedNonTextFiles).toContain('docker-compose.yml'); + await expect(cache.readJsonFile(path.join(root, 'blob.ts'))).rejects.toBeInstanceOf(NonTextFileError); + await expect(readJsonFile(path.join(root, 'blob.ts'))).rejects.toBeInstanceOf(NonTextFileError); + const lock = path.join(root, 'package-lock.json'); + fs.copyFileSync(path.join(fixtureDir, 'blob.ts'), lock); + await expect(readTextFile(lock)).rejects.toBeInstanceOf(LockfileParseError); + try { + await cache.readTextFile(lock); + } catch (err) { + expect(err).toBeInstanceOf(LockfileParseError); + assertNoSecrets(err instanceof Error ? err.message : String(err)); + } + }); + + it('vg scan records one warning and does not put the bytes in the result', async () => { + const root = nonTextProject(); + const logs: string[] = []; + const logSpy = vi.spyOn(console, 'log').mockImplementation((...args: unknown[]) => { + logs.push(args.map((part) => String(part)).join(' ')); + }); + const errLogs: string[] = []; + const errSpy = vi.spyOn(console, 'error').mockImplementation((...args: unknown[]) => { + errLogs.push(args.map((part) => String(part)).join(' ')); + }); + try { + const artifact = await runCoreScan(root, { + format: 'json', + concurrency: 1, + offline: true, + noLocalArtifacts: true, + quiet: true, + }); + const hits = (artifact.degradations ?? []).filter((warning) => warning.code === 'VG_WARN_NON_TEXT_FILE'); + expect(hits).toHaveLength(1); + expect(hits[0]!.message).toContain('docker-compose.yml'); + expect(hits[0]!.message).toContain('--exclude'); + const rendered = JSON.stringify(artifact) + '\n' + logs.join('\n') + '\n' + errLogs.join('\n'); + assertNoSecrets(rendered); + } finally { + logSpy.mockRestore(); + errSpy.mockRestore(); + } + }, 60_000); +}); + +describe('binary files at the CLI', () => { + function run(dir: string, command: 'scan' | 'build'): { status: number | null; stdout: string; stderr: string } { + const args = + command === 'scan' + ? [cli, 'scan', dir, '--offline', '--no-graph', '--no-daemon', '--quiet'] + : [cli, 'build', '-C', dir, '--no-warm', '--no-publish', '--offline', '--no-daemon', '--no-index']; + const res = spawnSync(process.execPath, ['--import', 'tsx', ...args], { + cwd: pkgRoot, + encoding: 'utf8', + timeout: 45_000, + env: { ...process.env, NO_COLOR: '1', VIBGRATE_NO_KERNEL: '1', VIBGRATE_DSN: '' }, + }); + return { status: res.status, stdout: res.stdout ?? '', stderr: res.stderr ?? '' }; + } + + it('vg build exits 0, warns once, and writes no file bytes', () => { + const dir = nonTextProject(); + const res = run(dir, 'build'); + expect(res.status).toBe(0); + const output = res.stdout + res.stderr; + expect(output).toContain('VG_WARN_NON_TEXT_FILE'); + expect(output).toContain('blob.ts'); + expect(output).toContain('notes.md'); + expect(output.match(/VG_WARN_NON_TEXT_FILE/g)).toEqual(['VG_WARN_NON_TEXT_FILE']); + assertNoSecrets(output); + }, 60_000); + + it('vg scan exits 0 and does not print file bytes', () => { + const dir = nonTextProject(); + const res = run(dir, 'scan'); + expect(res.status).toBe(0); + const output = res.stdout + res.stderr; + expect(output).toContain('VG_WARN_NON_TEXT_FILE'); + expect(output).toContain('docker-compose.yml'); + assertNoSecrets(output); + }, 60_000); +}); diff --git a/test/non-utf8-source.test.ts b/test/non-utf8-source.test.ts deleted file mode 100644 index 3fd9c9a8..00000000 --- a/test/non-utf8-source.test.ts +++ /dev/null @@ -1,198 +0,0 @@ -import { describe, it, expect, afterEach, vi } from 'vitest'; -import * as fs from 'node:fs'; -import * as os from 'node:os'; -import * as path from 'node:path'; -import { discover } from '../src/engine/discover.js'; -import { buildGraph } from '../src/engine/build.js'; -import { FileCache } from '../src/core-open/utils/fs.js'; -import { runCoreScan } from '../src/core-open/index.js'; -import { - formatSkippedNonUtf8Notice, - isNonUtf8Source, - isUtf8SourceText, -} from '../src/core-open/utils/source-text.js'; -import { advancedScanHook } from '../src/reporting/advanced-analysis.js'; - -const SECRET = 'SECRET_TOKEN_do_not_leak'; -const PIN = '2020-01-01T00:00:00.000Z'; - -/** Small blob: PNG-like header, a NUL, invalid UTF-8, and an ASCII secret. */ -function binaryBlob(): Buffer { - return Buffer.concat([ - Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a, 0x00]), - Buffer.from(SECRET), - Buffer.from([0xff, 0xfe, 0x80]), - ]); -} - -function writeFixture(opts: { gitignore?: string } = {}): string { - const root = fs.mkdtempSync(path.join(os.tmpdir(), 'vg-non-utf8-')); - fs.mkdirSync(path.join(root, 'src')); - fs.writeFileSync(path.join(root, 'src', 'keep.ts'), 'export const keep = "café";\n'); - fs.writeFileSync(path.join(root, 'payload.js'), binaryBlob()); - fs.writeFileSync(path.join(root, 'README.md'), binaryBlob()); - fs.writeFileSync(path.join(root, 'logo.png'), Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a])); - fs.writeFileSync( - path.join(root, 'package.json'), - JSON.stringify({ name: 'binary-fixture', version: '1.0.0' }), - ); - if (opts.gitignore !== undefined) fs.writeFileSync(path.join(root, '.gitignore'), opts.gitignore); - return root; -} - -function captureStderr(run: () => Promise | void): Promise { - const chunks: string[] = []; - const spy = vi.spyOn(process.stderr, 'write').mockImplementation((chunk: string | Uint8Array) => { - chunks.push(typeof chunk === 'string' ? chunk : Buffer.from(chunk).toString('utf8')); - return true; - }); - return Promise.resolve(run()).finally(() => spy.mockRestore()).then(() => chunks.join('')); -} - -const DISCOVER_NOTICE = - "notice: skipped 1 file that is not UTF-8 text (payload.js). vg does not read binary or non-UTF-8 files as source. Ignore them with --exclude or a .gitignore rule. Example: vg build --exclude 'payload.js'\n"; - -const BUILD_NOTICE = - "notice: skipped 2 files that are not UTF-8 text (README.md, payload.js). vg does not read binary or non-UTF-8 files as source. Ignore them with --exclude or a .gitignore rule. Example: vg build --exclude 'README.md'\n"; - -describe('binary and non-UTF-8 source text', () => { - const dirs: string[] = []; - afterEach(() => { - while (dirs.length) fs.rmSync(dirs.pop()!, { recursive: true, force: true }); - }); - - it('classifies UTF-8 text and refuses NUL, UTF-16, and invalid bytes', () => { - expect(isUtf8SourceText(Buffer.from(''))).toBe(true); - expect(isUtf8SourceText(Buffer.from('export const keep = "café";\n'))).toBe(true); - expect(isUtf8SourceText(Buffer.from([0xef, 0xbb, 0xbf, 0x41]))).toBe(true); - expect(isUtf8SourceText(binaryBlob())).toBe(false); - expect(isUtf8SourceText(Buffer.from([0x00]))).toBe(false); - expect(isUtf8SourceText(Buffer.from([0xff, 0xfe, 0x41, 0x00]))).toBe(false); - expect(isNonUtf8Source(Buffer.from([0x41, 0xc3]), false)).toBe(false); - expect(isNonUtf8Source(Buffer.from([0x41, 0xc3]), true)).toBe(true); - expect(isNonUtf8Source(Buffer.from([0xff]), false)).toBe(true); - }); - - it('formats one sorted notice and never prints absolute paths or control bytes', () => { - const notice = formatSkippedNonUtf8Notice([ - 'payload.js', - 'README.md', - 'payload.js', - '/tmp/secret', - 'a\u0000b', - ]); - expect(notice).toBe( - "notice: skipped 4 files that are not UTF-8 text (README.md, payload.js). vg does not read binary or non-UTF-8 files as source. Ignore them with --exclude or a .gitignore rule. Example: vg build --exclude 'README.md'", - ); - expect(notice).not.toContain(SECRET); - expect(notice).not.toContain('/tmp'); - expect(notice).not.toContain('\u0000'); - expect(formatSkippedNonUtf8Notice([])).toBeNull(); - }); - - it('discover skips a binary source file, keeps UTF-8, and prints one stable notice', async () => { - const root = writeFixture(); - dirs.push(root); - let first: string[] = []; - let second: string[] = []; - const stderr1 = await captureStderr(() => { - first = discover({ root }).map((f) => f.rel); - }); - const stderr2 = await captureStderr(() => { - second = discover({ root }).map((f) => f.rel); - }); - expect(first).toEqual(['src/keep.ts']); - expect(second).toEqual(first); - expect(stderr1).toBe(DISCOVER_NOTICE); - expect(stderr2).toBe(stderr1); - expect(stderr1).not.toContain(SECRET); - expect(stderr1).not.toContain(root); - expect(stderr1).not.toContain('logo.png'); - }); - - it('a gitignore or --exclude rule omits the file from the notice', async () => { - const ignored = writeFixture({ gitignore: 'payload.js\n' }); - dirs.push(ignored); - const fromGitignore = await captureStderr(() => { - expect(discover({ root: ignored }).map((f) => f.rel)).toEqual(['src/keep.ts']); - }); - expect(fromGitignore).toBe(''); - - const excluded = writeFixture(); - dirs.push(excluded); - const fromFlag = await captureStderr(() => { - expect(discover({ root: excluded, exclude: ['payload.js'] }).map((f) => f.rel)).toEqual(['src/keep.ts']); - }); - expect(fromFlag).toBe(''); - }); - - it('vg build keeps a deterministic map and does not copy the blob into it', async () => { - const root = writeFixture(); - dirs.push(root); - const run = () => - buildGraph({ - root, - generatedAt: PIN, - inline: true, - noCache: true, - noIndex: true, - noScip: true, - }); - let first!: Awaited>; - let second!: Awaited>; - const stderr1 = await captureStderr(async () => { - first = await run(); - }); - const stderr2 = await captureStderr(async () => { - second = await run(); - }); - - expect(stderr1).toBe(BUILD_NOTICE); - expect(stderr2).toBe(stderr1); - expect(stderr1).not.toContain(SECRET); - expect(stderr1).not.toContain(root); - - const graphText = JSON.stringify(first.graph); - expect(graphText).toBe(JSON.stringify(second.graph)); - expect(graphText).toContain('keep'); - expect(first.graph.nodes.some((n) => n.file === 'src/keep.ts')).toBe(true); - expect(graphText).not.toContain('payload.js'); - expect(graphText).not.toContain('README.md'); - expect(graphText).not.toContain(SECRET); - expect(graphText).not.toContain('vg-skip-non-utf8'); - expect(first.warnings.join('\n')).not.toContain(SECRET); - expect(first.warnings.join('\n')).not.toContain('vg-skip-non-utf8'); - }, 60_000); - - it('vg scan reads past the blob without crashing or copying it', async () => { - const root = writeFixture(); - dirs.push(root); - const logs: string[] = []; - const logSpy = vi.spyOn(console, 'log').mockImplementation((...args: unknown[]) => { - logs.push(args.map((a) => String(a)).join(' ')); - }); - let artifact!: Awaited>; - const stderr = await captureStderr(async () => { - artifact = await runCoreScan( - root, - { format: 'json', concurrency: 1, offline: true, noLocalArtifacts: true, quiet: true }, - advancedScanHook, - ); - }); - logSpy.mockRestore(); - - expect(artifact.projects.length).toBeGreaterThan(0); - const dumped = `${JSON.stringify(artifact)}\n${logs.join('\n')}`; - expect(dumped).not.toContain(SECRET); - expect(stderr).toContain('notice: skipped 1 file that is not UTF-8 text (payload.js)'); - expect(stderr).toContain("vg build --exclude 'payload.js'"); - expect(stderr).not.toContain(SECRET); - expect(stderr).not.toContain(root); - expect(stderr).not.toContain('logo.png'); - - const cache = new FileCache(); - const text = await cache.readTextFile(path.join(root, 'payload.js')); - expect(text).toBe(''); - expect(cache.skippedNonUtf8).toEqual(['payload.js']); - }, 60_000); -}); diff --git a/test/warning-codes.test.ts b/test/warning-codes.test.ts index db06a4cb..a02e1f4c 100644 --- a/test/warning-codes.test.ts +++ b/test/warning-codes.test.ts @@ -20,6 +20,7 @@ const PUBLISHED_CODES = [ 'VG_WARN_HCL_PARTIAL', 'VG_WARN_LICENSE_UNPARSEABLE', 'VG_WARN_LICENSE_UNREPRESENTABLE', + 'VG_WARN_NON_TEXT_FILE', 'VG_WARN_PARSE_FAILED', 'VG_WARN_PURL_UNAVAILABLE', 'VG_WARN_SBOM_LOSSY_EDGES',