diff --git a/DOCS.md b/DOCS.md index b6b753c..e17783c 100644 --- a/DOCS.md +++ b/DOCS.md @@ -113,6 +113,7 @@ For a quick overview, see the [README](./README.md). This document covers everyt - [Thresholds](#thresholds) - [Scanner Toggles](#scanner-toggles) - [Symlinks](#symlinks) + - [Binary and non-UTF-8 files](#binary-and-non-utf-8-files) - [Extended Scanners](#extended-scanners) - [Platform Matrix](#platform-matrix) - [Dependency Risk](#dependency-risk) @@ -2136,7 +2137,7 @@ Switches (flags that take no value, such as `--vulns`, `--offline` or `--no-grap By default, the scan writes `.vibgrate/scan_result.json`. Use `--no-local-artifacts` or `--max-privacy` to suppress local JSON artifact files. -`vg scan` does not follow symlinks while it indexes the tree. A skipped link is named once on stderr. See [Symlinks](#symlinks). +`vg scan` does not follow symlinks while it indexes the tree. A skipped link is named once on stderr. See [Symlinks](#symlinks). A binary or non-UTF-8 file that would otherwise be read as source is skipped. The scan prints one notice on stderr; when the scan also builds the code map, that build prints its own. See [Binary and non-UTF-8 files](#binary-and-non-utf-8-files). For offline drift scoring, pass `--package-manifest ` with a downloaded manifest bundle such as `https://github.com/vibgrate/manifests/latest-packages.zip`. The manifest shape, the fail-closed errors, and what offline mode skips are in [Offline scan with a package-version manifest](#offline-scan-with-a-package-version-manifest). @@ -2728,7 +2729,7 @@ Maps source code into a graph artifact, enabling all downstream queries (`vg sho | `--attestation ` | `.vibgrate/attestation.intoto.jsonl` | Where `--attest` writes, and where `--verify` reads | | `--pub ` | — | Public key PEM that pins the signer for `--verify` | -`vg build` does not follow symlinks while it discovers files. A skipped link is named once on stderr. See [Symlinks](#symlinks). `.gitignore` and `--exclude` still apply. +`vg build` does not follow symlinks while it discovers files. A skipped link is named once on stderr. See [Symlinks](#symlinks). `.gitignore` and `--exclude` still apply. A binary or non-UTF-8 file that would otherwise be read as source is skipped and named once on stderr. See [Binary and non-UTF-8 files](#binary-and-non-utf-8-files). **Local by default — no git churn.** The first time vg writes into `.vibgrate/` it also creates `.vibgrate/.gitignore`, keeping the graph artifacts (`graph.json`, `graph.html`, `GRAPH_REPORT.md`, `facts.jsonl`, `mcp-navigation.json`) and the cache out of git — so builds, auto-refreshes, and MCP use never leave your branch dirty. Run `vg share` when you want the map committed for your team (it rewrites that ignore file). vg never touches an existing `.vibgrate/.gitignore`, so edit it (or leave it empty) to manage the ignores yourself. @@ -5132,6 +5133,34 @@ text only; it does not hide this notice. notice: skipped 3 symlinks (alias.ts, nested/cycle, via). vg does not follow symlinks. Point the root at the link target, or pass --exclude or a narrower root. ``` +### Binary and non-UTF-8 files + +`vg build` and `vg scan` read source as UTF-8 text. A file under the walk can still be a binary, a media blob, or another encoding, including one whose extension looks like source (`payload.js`, `README.md`). vg does not decode those bytes. It leaves the file out of the map and out of scan text, prints one notice for that walk on stderr, and continues. `vg scan` builds a code map after the scan unless you pass `--no-graph`, so the map build can print a second notice for the same paths. The command does not crash, and neither notice includes file bytes. + +Known media extensions (images, fonts, audio, video, archives) are already left out of the scan walk. This notice is for a file vg would otherwise have opened as text. Output files do not change because of the notice itself, and the exit code stays the same when the rest of the command succeeds. + +The notice is the count, then the first few root-relative paths in sorted order (at most five; further files are a `+N more` count). Paths are relative to the root you passed. Absolute paths are not printed. The line goes to stderr, so `--format json` and `vg build --json` keep a JSON document on stdout. `--quiet` hides promotional text only; it does not hide this notice. + +```text +notice: skipped 2 files that are not UTF-8 text (README.md, payload.js). vg does not read binary or non-UTF-8 files as source. Ignore them with --exclude or a .gitignore rule. Example: vg build --exclude 'README.md' +``` + +To leave a file out without a notice, list it in `.gitignore` or pass `--exclude`. Both commands accept the flag. A config `exclude` list is merged in as well. + +```bash +vg build --exclude 'payload.js' +vg build --exclude '*.bin' --exclude 'vendor-blobs/**' +vg scan --exclude 'legacy/blob.js' +``` + +```gitignore +# binary checked in under a source-like name +payload.js +*.bin +``` + +A source file that is valid UTF-8, including non-ASCII text, is still read. Save a file that uses another encoding as UTF-8 if it should be part of the map. + --- ## Extended Scanners diff --git a/src/core-open/run-core-scan.ts b/src/core-open/run-core-scan.ts index a6d6ebc..43cc175 100644 --- a/src/core-open/run-core-scan.ts +++ b/src/core-open/run-core-scan.ts @@ -39,6 +39,7 @@ import { loadConfig, appendExcludePatterns } from './config.js'; import { pathExists, readJsonFile, writeJsonFile, writeTextFile, ensureDir, FileCache, quickTreeCount } from './utils/fs.js'; import { portableValue } from './utils/portable-path.js'; import { assertSafeWalkRoot } from './utils/root-safety.js'; +import { emitSkippedNonUtf8Notice } from './utils/source-text.js'; import { detectVcs } from './utils/vcs.js'; import { isCiEnvironment, hasVibgrateWorkflow } from './utils/ci-env.js'; import { resolveRepositoryName } from './utils/repository-name.js'; @@ -805,6 +806,7 @@ export async function runCoreScan( } } + emitSkippedNonUtf8Notice(fileCache.skippedNonUtf8); fileCache.clear(); if (allProjects.length === 0) { diff --git a/src/core-open/utils/fs.ts b/src/core-open/utils/fs.ts index 5e19c4e..233e5f9 100644 --- a/src/core-open/utils/fs.ts +++ b/src/core-open/utils/fs.ts @@ -17,7 +17,8 @@ import { UnsafeRootError, type WalkBudgetState, } from './root-safety.js'; -import { emitSkippedSymlinkNotice, rememberSkippedSymlink } from './skipped-symlinks.js'; +import { emitSkippedSymlinkNotice, rememberSkippedSymlink, walkRelativePath } from './skipped-symlinks.js'; +import { isUtf8SourceText } from './source-text.js'; const execFileAsync = promisify(execFile); @@ -399,6 +400,13 @@ export class FileCache { return this._skippedLargeFiles; } + /** Binary or non-UTF-8 files a text read refused. Paths are root-relative. */ + private _skippedNonUtf8: string[] = []; + + get skippedNonUtf8(): readonly string[] { + return this._skippedNonUtf8; + } + // ── Directory walking ── /** @@ -770,7 +778,13 @@ export class FileCache { } } - const content = await fs.readFile(abs, 'utf8'); + const buf = await fs.readFile(abs); + if (!isUtf8SourceText(buf)) { + this.noteNonUtf8(abs); + // Cache the refusal so later scanners do not read the blob again. + return ''; + } + const content = buf.toString('utf8'); if (content.length > TEXT_CACHE_MAX_BYTES) { // Too large for cache — evict so we don't hold it this.textCache.delete(abs); @@ -824,6 +838,15 @@ export class FileCache { this.jsonCache.clear(); this.existsCache.clear(); this.sizeCache.clear(); + this._skippedNonUtf8 = []; + } + + /** Record a text read that refused binary or non-UTF-8 bytes. Path only. */ + private noteNonUtf8(abs: string): void { + const rel = this._rootDir ? walkRelativePath(this._rootDir, abs) : ''; + const label = rel || path.basename(abs); + if (!label || this._skippedNonUtf8.includes(label)) return; + this._skippedNonUtf8.push(label); } /** Number of file content entries currently held */ @@ -1135,12 +1158,18 @@ export function stripBom(text: string): string { } export async function readJsonFile(filePath: string): Promise { - const txt = await fs.readFile(filePath, 'utf8'); - return JSON.parse(stripBom(txt)) as T; + const buf = await fs.readFile(filePath); + if (!isUtf8SourceText(buf)) { + const base = path.basename(filePath); + throw new SyntaxError(`${base || 'file'} is not UTF-8 text`); + } + return JSON.parse(stripBom(buf.toString('utf8'))) as T; } export async function readTextFile(filePath: string): Promise { - return fs.readFile(filePath, 'utf8'); + const buf = await fs.readFile(filePath); + if (!isUtf8SourceText(buf)) return ''; + return buf.toString('utf8'); } export async function pathExists(p: string): Promise { diff --git a/src/core-open/utils/source-text.ts b/src/core-open/utils/source-text.ts new file mode 100644 index 0000000..7aedcf3 --- /dev/null +++ b/src/core-open/utils/source-text.ts @@ -0,0 +1,182 @@ +import * as fs from 'node:fs'; + +/** + * Decide whether bytes are safe to open as source text, and format the one + * notice a walk prints when they are not. + * + * A NUL, a UTF-16 BOM, or any byte sequence that is not UTF-8 means the file + * is binary or another encoding. Callers skip it. The notice names paths + * only — never file bytes — and tells the operator how to ignore the file. + */ + +/** How many leading bytes a large file is judged on before a full read. */ +export const NON_UTF8_PREFIX_BYTES = 8192; + +/** + * Marker on a parse row that was refused because the file is not UTF-8 text. + * The build drops the row. It is not a graph warning and not file contents. + */ +export const NON_UTF8_SKIP_MARK = 'vg-skip-non-utf8'; + +/** How many root-relative paths one notice lists. Further files are a count. */ +export const SKIPPED_NON_UTF8_NOTICE_CAP = 5; + +const HARD_FULL_READ_CAP = 32 * 1024 * 1024; + +function comparePath(a: string, b: string): number { + return a < b ? -1 : a > b ? 1 : 0; +} + +/** + * True when `bytes` must not be decoded as source text. + * `complete` means `bytes` is the whole file. A prefix may end on a cut + * multibyte character; that tail is not, by itself, a reason to skip. + */ +export function isNonUtf8Source(bytes: Uint8Array, complete: boolean): boolean { + if (bytes.length === 0) return false; + if (hasUtf16Bom(bytes)) return true; + for (let i = 0; i < bytes.length; i++) { + if (bytes[i] === 0) return true; + } + const sample = complete ? bytes : trimIncompleteUtf8Tail(bytes); + try { + new TextDecoder('utf-8', { fatal: true }).decode(sample); + return false; + } catch { + return true; + } +} + +/** True when the whole buffer is UTF-8 text with no NUL and no UTF-16 BOM. */ +export function isUtf8SourceText(bytes: Uint8Array): boolean { + return !isNonUtf8Source(bytes, true); +} + +/** + * Read `abs` and return its text, or `null` when the bytes are not UTF-8 + * source text. Throws when the file cannot be read. + */ +export function readUtf8SourceSync(abs: string): string | null { + const buf = fs.readFileSync(abs); + if (!isUtf8SourceText(buf)) return null; + return buf.toString('utf8'); +} + +/** + * True when a walked file should be left out of source-text handling. + * Files up to `fullReadCap` are checked in full. Larger files are judged + * on a prefix so a blob is never slurped just to be refused. `0` means + * prefix-only. Unreadable files return false; the caller already has a + * path for those. + */ +export function shouldSkipNonUtf8File(abs: string, fullReadCap: number): boolean { + let size: number; + try { + size = fs.statSync(abs).size; + } catch { + return false; + } + if (size === 0) return false; + const cap = fullReadCap > 0 ? Math.min(fullReadCap, HARD_FULL_READ_CAP) : 0; + try { + if (cap > 0 && size <= cap) { + return !isUtf8SourceText(fs.readFileSync(abs)); + } + const n = Math.min(size, NON_UTF8_PREFIX_BYTES); + const buf = Buffer.alloc(n); + const fd = fs.openSync(abs, 'r'); + try { + const got = fs.readSync(fd, buf, 0, n, 0); + return isNonUtf8Source(buf.subarray(0, got), got === size); + } finally { + fs.closeSync(fd); + } + } catch { + return false; + } +} + +/** + * One stderr line when a walk skips binary or non-UTF-8 files. `null` when + * `relPaths` is empty. Paths are de-duplicated, sorted, and capped. Absolute + * paths and control characters are omitted. File bytes are never included. + */ +export function formatSkippedNonUtf8Notice(relPaths: readonly string[]): string | null { + const seen = new Set(); + const unique: string[] = []; + let hidden = 0; + for (const raw of relPaths) { + const key = raw.split('\\').join('/'); + if (!key || key === '.' || seen.has(key)) continue; + seen.add(key); + const rel = displayRel(key); + if (!rel) hidden++; + else unique.push(rel); + } + unique.sort(comparePath); + const total = unique.length + hidden; + if (total === 0) return null; + const shown = unique.slice(0, SKIPPED_NON_UTF8_NOTICE_CAP); + const extra = unique.length - shown.length; + const list = extra > 0 ? `${shown.join(', ')}, +${extra} more` : shown.join(', '); + const noun = total === 1 ? 'file that is not UTF-8 text' : 'files that are not UTF-8 text'; + const names = list.length > 0 ? ` (${list})` : ''; + const example = shown[0] ? shellQuote(shown[0]) : "'*.bin'"; + return ( + `notice: skipped ${total} ${noun}${names}. ` + + 'vg does not read binary or non-UTF-8 files as source. ' + + `Ignore them with --exclude or a .gitignore rule. Example: vg build --exclude ${example}` + ); +} + +/** Write {@link formatSkippedNonUtf8Notice} to stderr. No-op when there is nothing to say. */ +export function emitSkippedNonUtf8Notice(relPaths: readonly string[]): void { + const line = formatSkippedNonUtf8Notice(relPaths); + if (!line) return; + process.stderr.write(`${line}\n`); +} + +function hasUtf16Bom(bytes: Uint8Array): boolean { + if (bytes.length < 2) return false; + const b0 = bytes[0]!; + const b1 = bytes[1]!; + return (b0 === 0xff && b1 === 0xfe) || (b0 === 0xfe && b1 === 0xff); +} + +/** Drop a trailing partial UTF-8 sequence so a prefix is not refused for being cut. */ +function trimIncompleteUtf8Tail(bytes: Uint8Array): Uint8Array { + const n = bytes.length; + if (n === 0) return bytes; + let i = n - 1; + let cont = 0; + while (i >= 0 && cont < 3 && (bytes[i]! & 0xc0) === 0x80) { + cont++; + i--; + } + if (i < 0) return bytes; + const lead = bytes[i]!; + let need = 0; + if ((lead & 0x80) === 0) need = 1; + else if ((lead & 0xe0) === 0xc0) need = 2; + else if ((lead & 0xf0) === 0xe0) need = 3; + else if ((lead & 0xf8) === 0xf0) need = 4; + else return bytes; + const have = n - i; + if (have < need) return bytes.subarray(0, i); + return bytes; +} + +/** A path safe to print. `null` when it must not appear in the notice. */ +function displayRel(rel: string): string | null { + if (!rel || rel === '.' || rel.startsWith('../') || rel === '..' || rel.startsWith('/')) return null; + if (/^[A-Za-z]:/.test(rel)) return null; + for (let i = 0; i < rel.length; i++) { + const c = rel.charCodeAt(i); + if (c < 0x20 || c === 0x7f) return null; + } + return rel; +} + +function shellQuote(value: string): string { + return `'${value.replace(/'/g, `'\\''`)}'`; +} diff --git a/src/engine/build.ts b/src/engine/build.ts index 9252c40..f3eb783 100644 --- a/src/engine/build.ts +++ b/src/engine/build.ts @@ -50,6 +50,7 @@ import type { FileParse } from './types.js'; import type { ResolveResult } from './resolve.js'; import { fileRolesFromParses } from './ast-roles.js'; import type { AstRoleHit } from '../core-open/scanners/architecture/ast-roles.js'; +import { emitSkippedNonUtf8Notice, NON_UTF8_SKIP_MARK } from '../core-open/utils/source-text.js'; import { stampWarning, WARNING_CODES, type CodedWarning } from '../core-open/warnings.js'; import { assembleEngineWarnings } from './warning-codes.js'; @@ -156,6 +157,15 @@ export interface BuildResult { } export async function buildGraph(options: BuildOptions): Promise { + const skippedNonUtf8: string[] = []; + try { + return await buildGraphUnchecked(options, skippedNonUtf8); + } finally { + emitSkippedNonUtf8Notice(skippedNonUtf8); + } +} + +async function buildGraphUnchecked(options: BuildOptions, skippedNonUtf8: string[]): Promise { const timer = new StageTimer(); timer.start('total'); const root = path.resolve(options.root); @@ -170,6 +180,8 @@ export async function buildGraph(options: BuildOptions): Promise { exclude, paths: options.paths, maxEntries: limits.maxFiles, + maxSourceBytes: limits.maxFileBytes === 0 ? 32 * 1024 * 1024 : limits.maxFileBytes, + skippedNonUtf8, }); timer.end('discover'); @@ -295,12 +307,16 @@ export async function buildGraph(options: BuildOptions): Promise { timer.end('hash'); timer.start('parse'); - const parsedNew = await parseFiles(toParse, { + const parsedNew = (await parseFiles(toParse, { jobs: options.jobs, inline: options.inline, onProgress: options.onParseProgress, grammarsDir: options.grammarsDir, memoryBudgetMb: limits.memoryBudgetMb, + })).filter((parsed) => { + if (!parsed.warnings?.includes(NON_UTF8_SKIP_MARK)) return true; + skippedNonUtf8.push(parsed.rel); + return false; }); timer.end('parse'); checkMemoryBudget('parse', limits.memoryBudgetMb); @@ -339,6 +355,7 @@ export async function buildGraph(options: BuildOptions): Promise { const manifests = extractManifests(root, { exclude, paths: options.paths, + skippedNonUtf8, }); if (manifests.files > 0) { const byId = new Map(resolved.nodes.map((n) => [n.id, n])); @@ -518,6 +535,7 @@ export async function buildGraph(options: BuildOptions): Promise { exclude, paths: options.paths, maxEntries: limits.maxFiles, + skippedNonUtf8, }); for (const d of docs) { try { diff --git a/src/engine/discover.ts b/src/engine/discover.ts index c5113b9..2f8fa11 100644 --- a/src/engine/discover.ts +++ b/src/engine/discover.ts @@ -6,7 +6,9 @@ import { requireDataConfig } from '../core-open/config.js'; import { dropBlankPatterns, gitignoreWithoutBlankLines } from '../core-open/utils/glob.js'; import { assertLockfileFile, lockfileKind } from '../core-open/utils/lockfile-parse.js'; import { assertSafeWalkRoot, createWalkBudget, noteWalkEntry } from '../core-open/utils/root-safety.js'; +import { emitSkippedNonUtf8Notice, shouldSkipNonUtf8File } from '../core-open/utils/source-text.js'; import { emitSkippedSymlinkNotice } from '../core-open/utils/skipped-symlinks.js'; +import { DEFAULT_MAX_FILE_BYTES } from './limits.js'; /** * Deterministic file discovery. @@ -167,6 +169,17 @@ export interface DiscoverOptions { * Default: `VG_MAX_FILES`, else 100000. */ maxEntries?: number; + /** + * Largest file fully checked for UTF-8 before it is treated as source. + * Larger files are judged on a prefix. Default: the build's per-file cap. + * `0` checks a prefix only. + */ + maxSourceBytes?: number; + /** + * When set, binary and non-UTF-8 source files are recorded here and this + * call does not print the notice. The caller prints once. + */ + skippedNonUtf8?: string[]; } export interface DiscoveredFile { @@ -253,6 +266,8 @@ export function discover(options: DiscoverOptions): DiscoveredFile[] { // link, so a directory link to its parent cannot re-enter this walk. The // notice lists the ones this walk skipped; ignored links stay quiet. const skippedSymlinks: string[] = []; + const skippedNonUtf8 = options.skippedNonUtf8 ?? []; + const fullReadCap = options.maxSourceBytes ?? DEFAULT_MAX_FILE_BYTES; const considerFile = (abs: string): void => { const rel = toPosix(path.relative(root, abs)); @@ -266,6 +281,12 @@ export function discover(options: DiscoverOptions): DiscoveredFile[] { } const lang = langForExtension(path.extname(abs)); if (!lang || !allowLang(lang)) return; + // A source extension does not make the bytes text. Skip before parse so + // a blob is never decoded, hashed into a symbol, or copied into a warning. + if (rel && shouldSkipNonUtf8File(abs, fullReadCap)) { + skippedNonUtf8.push(rel); + return; + } found.set(rel, { rel, abs, lang }); }; @@ -308,6 +329,7 @@ export function discover(options: DiscoverOptions): DiscoveredFile[] { } emitSkippedSymlinkNotice(skippedSymlinks); + if (!options.skippedNonUtf8) emitSkippedNonUtf8Notice(skippedNonUtf8); return [...found.values()].sort((a, b) => (a.rel < b.rel ? -1 : a.rel > b.rel ? 1 : 0)); } diff --git a/src/engine/docs-ingest.ts b/src/engine/docs-ingest.ts index 64da6fb..e43d596 100644 --- a/src/engine/docs-ingest.ts +++ b/src/engine/docs-ingest.ts @@ -21,6 +21,7 @@ import * as fs from 'node:fs'; import * as path from 'node:path'; import { redactSecrets } from '../core-open/utils/redact.js'; +import { emitSkippedNonUtf8Notice, isUtf8SourceText, readUtf8SourceSync } from '../core-open/utils/source-text.js'; import { nodeId } from './ids.js'; import { isSkippedDirName, loadRootIgnore, SKIP_FILES } from './discover.js'; import { assertSafeWalkRoot, createWalkBudget, noteWalkEntry, UnsafeRootError } from '../core-open/utils/root-safety.js'; @@ -57,6 +58,11 @@ export interface DiscoverDocsOptions { maxFiles?: number; /** Walk-entry ceiling. `0` disables. Default: `VG_MAX_FILES`, else 100000. */ maxEntries?: number; + /** + * When set, binary and non-UTF-8 context files are recorded here and this + * call does not print the notice. The caller prints once. + */ + skippedNonUtf8?: string[]; } export interface DiscoveredDoc { @@ -376,6 +382,7 @@ export function discoverDocs(options: DiscoverDocsOptions): DiscoveredDoc[] { const found = new Map(); const budget = createWalkBudget(root, options.maxEntries); + const skippedNonUtf8 = options.skippedNonUtf8 ?? []; const consider = (abs: string): void => { if (found.size >= maxFiles) return; @@ -390,6 +397,12 @@ export function discoverDocs(options: DiscoverDocsOptions): DiscoveredDoc[] { const st = fs.statSync(abs); if (!st.isFile() || st.size > DOC_FILE_MAX_BYTES) return; if (st.size === 0) return; + // Context files are stored as text. Refuse binary / non-UTF-8 before + // any body is copied into a document node. + if (!isUtf8SourceText(fs.readFileSync(abs))) { + skippedNonUtf8.push(rel); + return; + } } catch { return; } @@ -439,6 +452,7 @@ export function discoverDocs(options: DiscoverDocsOptions): DiscoveredDoc[] { } } + if (!options.skippedNonUtf8) emitSkippedNonUtf8Notice(skippedNonUtf8); return [...found.values()].sort((a, b) => a.rel.localeCompare(b.rel)); } @@ -503,7 +517,9 @@ export function documentNodesFromDocs(docs: DiscoveredDoc[]): GraphNode[] { for (const d of docs) { let raw = ''; try { - raw = fs.readFileSync(d.abs, 'utf8'); + const text = readUtf8SourceSync(d.abs); + if (text === null) continue; + raw = text; } catch { continue; } diff --git a/src/engine/literal-scan.ts b/src/engine/literal-scan.ts index 1660e17..ebf1325 100644 --- a/src/engine/literal-scan.ts +++ b/src/engine/literal-scan.ts @@ -2,6 +2,7 @@ import * as fs from 'node:fs'; import * as path from 'node:path'; import * as os from 'node:os'; import { Worker } from 'node:worker_threads'; +import { isUtf8SourceText } from '../core-open/utils/source-text.js'; /** * The uniform literal-search core for `search_symbols` (VG-LITERAL-INDEX-DESIGN.md @@ -24,7 +25,6 @@ import { Worker } from 'node:worker_threads'; const MAX_FILE_BYTES = 1_000_000; const PREVIEW_CHARS = 120; -const NUL = '\u0000'; // a NUL byte marks a binary file — skipped, same as ripgrep does /** * Fan out to workers only above this many candidate BYTES. Benchmarking showed * that gating on file *count* is wrong: a few thousand small files (≈8 MB) scan @@ -149,8 +149,10 @@ function readCandidate(abs: string, knownSize?: number): string | null { try { const size = knownSize ?? fs.statSync(abs).size; if (size > MAX_FILE_BYTES) return null; - const text = fs.readFileSync(abs, 'utf8'); - return text.includes(NUL) ? null : text; + const buf = fs.readFileSync(abs); + // Binary and non-UTF-8 never become a preview line. + if (!isUtf8SourceText(buf)) return null; + return buf.toString('utf8'); } catch { return null; } @@ -218,15 +220,17 @@ const WORKER_SOURCE = [ `const MAX = ${MAX_FILE_BYTES};`, `const PREVIEW = ${PREVIEW_CHARS};`, `const CAP = ${ROW_CAP};`, - 'const NUL = String.fromCharCode(0);', 'const NL = String.fromCharCode(10);', 'const { root, files, needleLower } = workerData;', 'const hits = []; let total = 0, truncated = false;', 'for (const rel of files) {', ' const abs = path.join(root, rel);', + ' let buf;', + ' try { if (fs.statSync(abs).size > MAX) continue; buf = fs.readFileSync(abs); } catch { continue; }', + ' if (buf.includes(0)) continue;', + ' if (buf.length >= 2 && ((buf[0] === 255 && buf[1] === 254) || (buf[0] === 254 && buf[1] === 255))) continue;', ' let text;', - ' try { if (fs.statSync(abs).size > MAX) continue; text = fs.readFileSync(abs, "utf8"); } catch { continue; }', - ' if (text.includes(NUL)) continue;', + ' try { new TextDecoder("utf-8", { fatal: true }).decode(buf); text = buf.toString("utf8"); } catch { continue; }', ' const low = text.toLowerCase();', ' if (!low.includes(needleLower)) continue;', ' const lowLines = low.split(NL); const rawLines = text.split(NL);', diff --git a/src/engine/manifests.ts b/src/engine/manifests.ts index 23ee62c..8b56a60 100644 --- a/src/engine/manifests.ts +++ b/src/engine/manifests.ts @@ -23,6 +23,7 @@ import * as path from 'node:path'; import type { Ignore } from 'ignore'; import { XMLParser } from 'fast-xml-parser'; import { nodeId, edgeId } from './ids.js'; +import { isUtf8SourceText } from '../core-open/utils/source-text.js'; import { isSkippedDirName, loadRootIgnore } from './discover.js'; import { parseToml } from '../core-open/utils/toml.js'; import type { GraphEdge, GraphNode } from '../schema.js'; @@ -62,7 +63,7 @@ const emptyNode = ( */ export function extractManifests( root: string, - opts: { exclude?: string[]; paths?: string[] } = {}, + opts: { exclude?: string[]; paths?: string[]; skippedNonUtf8?: string[] } = {}, ): ManifestExtract { const absRoot = path.resolve(root); const ig = loadRootIgnore(absRoot, opts.exclude ?? []); @@ -82,6 +83,16 @@ export function extractManifests( for (const rel of [...found.keys()].sort()) { const abs = found.get(rel)!; const base = path.posix.basename(rel); + let manifestBytes: Buffer; + try { + manifestBytes = fs.readFileSync(abs); + } catch { + continue; + } + if (!isUtf8SourceText(manifestBytes)) { + opts.skippedNonUtf8?.push(rel); + continue; + } try { if (base === 'package.json') { deps += ingestPackageJson(rel, abs, nodes, edges); diff --git a/src/engine/parse-worker.ts b/src/engine/parse-worker.ts index 4790498..77e6c99 100644 --- a/src/engine/parse-worker.ts +++ b/src/engine/parse-worker.ts @@ -1,7 +1,7 @@ -import * as fs from 'node:fs'; import { parseSource } from './parse.js'; import { setGrammarsOverride, resetParser } from './grammars.js'; import type { FileParse } from './types.js'; +import { NON_UTF8_SKIP_MARK, readUtf8SourceSync } from '../core-open/utils/source-text.js'; import { stampWarning, WARNING_CODES } from '../core-open/warnings.js'; /** @@ -30,7 +30,23 @@ export default async function run(payload: ParsePayload): Promise { const out: FileParse[] = []; for (const task of payload.tasks) { try { - const source = fs.readFileSync(task.abs, 'utf8'); + const source = readUtf8SourceSync(task.abs); + if (source === null) { + out.push({ + rel: task.rel, + lang: task.lang, + hash: '', + bytes: 0, + defs: [], + calls: [], + imports: [], + heritage: [], + typeRefs: [], + guards: [], + warnings: [NON_UTF8_SKIP_MARK], + }); + continue; + } out.push(await parseSource(task.rel, task.lang, source)); } catch (err) { // A wasm-level parse crash can leave the language's reused parser diff --git a/src/engine/pool.ts b/src/engine/pool.ts index 9837770..865ed54 100644 --- a/src/engine/pool.ts +++ b/src/engine/pool.ts @@ -7,6 +7,7 @@ import { setGrammarsOverride, resetParser } from './grammars.js'; import { checkMemoryBudget, envJobs, envWorkerHeapMb, ResourceLimitError } from './limits.js'; import type { DiscoveredFile } from './discover.js'; import type { FileParse } from './types.js'; +import { NON_UTF8_SKIP_MARK, readUtf8SourceSync } from '../core-open/utils/source-text.js'; import { stampWarning, WARNING_CODES } from '../core-open/warnings.js'; import type { ParseTask } from './parse-worker.js'; @@ -76,8 +77,8 @@ async function parseInline(files: DiscoveredFile[], options: ParseOptions): Prom onProgress?.(0, files.length); for (const file of files) { try { - const source = fs.readFileSync(file.abs, 'utf8'); - out.push(await parseSource(file.rel, file.lang.id, source)); + const source = readUtf8SourceSync(file.abs); + out.push(source === null ? nonUtf8Parse(file) : await parseSource(file.rel, file.lang.id, source)); } catch (err) { // A wasm-level parse crash can leave the language's reused parser // mid-state; drop it so the failure stays contained to this file. @@ -170,6 +171,22 @@ function sortByRel(parses: FileParse[]): FileParse[] { return parses.sort((a, b) => (a.rel < b.rel ? -1 : a.rel > b.rel ? 1 : 0)); } +function nonUtf8Parse(file: DiscoveredFile): FileParse { + return { + rel: file.rel, + lang: file.lang.id, + hash: '', + bytes: 0, + defs: [], + calls: [], + imports: [], + heritage: [], + typeRefs: [], + guards: [], + warnings: [NON_UTF8_SKIP_MARK], + }; +} + function emptyParse(file: DiscoveredFile, warning: string): FileParse { return { rel: file.rel, diff --git a/src/engine/toolchain/index.ts b/src/engine/toolchain/index.ts index 8f3b1aa..d8cab31 100644 --- a/src/engine/toolchain/index.ts +++ b/src/engine/toolchain/index.ts @@ -1,6 +1,8 @@ import * as fs from 'node:fs'; import { edgeId, nodeId } from '../ids.js'; import type { GraphEdge, GraphNode } from '../../schema.js'; +import { readUtf8SourceSync } from '../../core-open/utils/source-text.js'; +import { stampWarning, WARNING_CODES } from '../../core-open/warnings.js'; import type { DiscoveredDoc } from '../docs-ingest.js'; import { composeExtractor } from './compose.js'; import { dockerfileExtractor } from './dockerfile.js'; @@ -10,7 +12,6 @@ import { terraformExtractor } from './terraform.js'; import { githubActionsExtractor, gitlabCiExtractor } from './workflows.js'; import { linkToolchain } from './link.js'; import { safeDoc, TOOLCHAIN_FILE_MAX_BYTES, toPosix } from './util.js'; -import { stampWarning, WARNING_CODES } from '../../core-open/warnings.js'; import type { ToolchainExtraction, ToolchainExtractor, @@ -139,7 +140,9 @@ async function readExtractions(docs: DiscoveredDoc[]): Promise<{ files: FileExtr try { const stat = fs.statSync(doc.abs); if (stat.size > TOOLCHAIN_FILE_MAX_BYTES) continue; - source = fs.readFileSync(doc.abs, 'utf8'); + const text = readUtf8SourceSync(doc.abs); + if (text === null) continue; + source = text; head = source.slice(0, HEAD_BYTES); } catch { continue; // unreadable — docs-ingest already tolerates this diff --git a/test/non-utf8-source.test.ts b/test/non-utf8-source.test.ts new file mode 100644 index 0000000..3fd9c9a --- /dev/null +++ b/test/non-utf8-source.test.ts @@ -0,0 +1,198 @@ +import { describe, it, expect, afterEach, vi } from 'vitest'; +import * as fs from 'node:fs'; +import * as os from 'node:os'; +import * as path from 'node:path'; +import { discover } from '../src/engine/discover.js'; +import { buildGraph } from '../src/engine/build.js'; +import { FileCache } from '../src/core-open/utils/fs.js'; +import { runCoreScan } from '../src/core-open/index.js'; +import { + formatSkippedNonUtf8Notice, + isNonUtf8Source, + isUtf8SourceText, +} from '../src/core-open/utils/source-text.js'; +import { advancedScanHook } from '../src/reporting/advanced-analysis.js'; + +const SECRET = 'SECRET_TOKEN_do_not_leak'; +const PIN = '2020-01-01T00:00:00.000Z'; + +/** Small blob: PNG-like header, a NUL, invalid UTF-8, and an ASCII secret. */ +function binaryBlob(): Buffer { + return Buffer.concat([ + Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a, 0x00]), + Buffer.from(SECRET), + Buffer.from([0xff, 0xfe, 0x80]), + ]); +} + +function writeFixture(opts: { gitignore?: string } = {}): string { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'vg-non-utf8-')); + fs.mkdirSync(path.join(root, 'src')); + fs.writeFileSync(path.join(root, 'src', 'keep.ts'), 'export const keep = "café";\n'); + fs.writeFileSync(path.join(root, 'payload.js'), binaryBlob()); + fs.writeFileSync(path.join(root, 'README.md'), binaryBlob()); + fs.writeFileSync(path.join(root, 'logo.png'), Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a])); + fs.writeFileSync( + path.join(root, 'package.json'), + JSON.stringify({ name: 'binary-fixture', version: '1.0.0' }), + ); + if (opts.gitignore !== undefined) fs.writeFileSync(path.join(root, '.gitignore'), opts.gitignore); + return root; +} + +function captureStderr(run: () => Promise | void): Promise { + const chunks: string[] = []; + const spy = vi.spyOn(process.stderr, 'write').mockImplementation((chunk: string | Uint8Array) => { + chunks.push(typeof chunk === 'string' ? chunk : Buffer.from(chunk).toString('utf8')); + return true; + }); + return Promise.resolve(run()).finally(() => spy.mockRestore()).then(() => chunks.join('')); +} + +const DISCOVER_NOTICE = + "notice: skipped 1 file that is not UTF-8 text (payload.js). vg does not read binary or non-UTF-8 files as source. Ignore them with --exclude or a .gitignore rule. Example: vg build --exclude 'payload.js'\n"; + +const BUILD_NOTICE = + "notice: skipped 2 files that are not UTF-8 text (README.md, payload.js). vg does not read binary or non-UTF-8 files as source. Ignore them with --exclude or a .gitignore rule. Example: vg build --exclude 'README.md'\n"; + +describe('binary and non-UTF-8 source text', () => { + const dirs: string[] = []; + afterEach(() => { + while (dirs.length) fs.rmSync(dirs.pop()!, { recursive: true, force: true }); + }); + + it('classifies UTF-8 text and refuses NUL, UTF-16, and invalid bytes', () => { + expect(isUtf8SourceText(Buffer.from(''))).toBe(true); + expect(isUtf8SourceText(Buffer.from('export const keep = "café";\n'))).toBe(true); + expect(isUtf8SourceText(Buffer.from([0xef, 0xbb, 0xbf, 0x41]))).toBe(true); + expect(isUtf8SourceText(binaryBlob())).toBe(false); + expect(isUtf8SourceText(Buffer.from([0x00]))).toBe(false); + expect(isUtf8SourceText(Buffer.from([0xff, 0xfe, 0x41, 0x00]))).toBe(false); + expect(isNonUtf8Source(Buffer.from([0x41, 0xc3]), false)).toBe(false); + expect(isNonUtf8Source(Buffer.from([0x41, 0xc3]), true)).toBe(true); + expect(isNonUtf8Source(Buffer.from([0xff]), false)).toBe(true); + }); + + it('formats one sorted notice and never prints absolute paths or control bytes', () => { + const notice = formatSkippedNonUtf8Notice([ + 'payload.js', + 'README.md', + 'payload.js', + '/tmp/secret', + 'a\u0000b', + ]); + expect(notice).toBe( + "notice: skipped 4 files that are not UTF-8 text (README.md, payload.js). vg does not read binary or non-UTF-8 files as source. Ignore them with --exclude or a .gitignore rule. Example: vg build --exclude 'README.md'", + ); + expect(notice).not.toContain(SECRET); + expect(notice).not.toContain('/tmp'); + expect(notice).not.toContain('\u0000'); + expect(formatSkippedNonUtf8Notice([])).toBeNull(); + }); + + it('discover skips a binary source file, keeps UTF-8, and prints one stable notice', async () => { + const root = writeFixture(); + dirs.push(root); + let first: string[] = []; + let second: string[] = []; + const stderr1 = await captureStderr(() => { + first = discover({ root }).map((f) => f.rel); + }); + const stderr2 = await captureStderr(() => { + second = discover({ root }).map((f) => f.rel); + }); + expect(first).toEqual(['src/keep.ts']); + expect(second).toEqual(first); + expect(stderr1).toBe(DISCOVER_NOTICE); + expect(stderr2).toBe(stderr1); + expect(stderr1).not.toContain(SECRET); + expect(stderr1).not.toContain(root); + expect(stderr1).not.toContain('logo.png'); + }); + + it('a gitignore or --exclude rule omits the file from the notice', async () => { + const ignored = writeFixture({ gitignore: 'payload.js\n' }); + dirs.push(ignored); + const fromGitignore = await captureStderr(() => { + expect(discover({ root: ignored }).map((f) => f.rel)).toEqual(['src/keep.ts']); + }); + expect(fromGitignore).toBe(''); + + const excluded = writeFixture(); + dirs.push(excluded); + const fromFlag = await captureStderr(() => { + expect(discover({ root: excluded, exclude: ['payload.js'] }).map((f) => f.rel)).toEqual(['src/keep.ts']); + }); + expect(fromFlag).toBe(''); + }); + + it('vg build keeps a deterministic map and does not copy the blob into it', async () => { + const root = writeFixture(); + dirs.push(root); + const run = () => + buildGraph({ + root, + generatedAt: PIN, + inline: true, + noCache: true, + noIndex: true, + noScip: true, + }); + let first!: Awaited>; + let second!: Awaited>; + const stderr1 = await captureStderr(async () => { + first = await run(); + }); + const stderr2 = await captureStderr(async () => { + second = await run(); + }); + + expect(stderr1).toBe(BUILD_NOTICE); + expect(stderr2).toBe(stderr1); + expect(stderr1).not.toContain(SECRET); + expect(stderr1).not.toContain(root); + + const graphText = JSON.stringify(first.graph); + expect(graphText).toBe(JSON.stringify(second.graph)); + expect(graphText).toContain('keep'); + expect(first.graph.nodes.some((n) => n.file === 'src/keep.ts')).toBe(true); + expect(graphText).not.toContain('payload.js'); + expect(graphText).not.toContain('README.md'); + expect(graphText).not.toContain(SECRET); + expect(graphText).not.toContain('vg-skip-non-utf8'); + expect(first.warnings.join('\n')).not.toContain(SECRET); + expect(first.warnings.join('\n')).not.toContain('vg-skip-non-utf8'); + }, 60_000); + + it('vg scan reads past the blob without crashing or copying it', async () => { + const root = writeFixture(); + dirs.push(root); + const logs: string[] = []; + const logSpy = vi.spyOn(console, 'log').mockImplementation((...args: unknown[]) => { + logs.push(args.map((a) => String(a)).join(' ')); + }); + let artifact!: Awaited>; + const stderr = await captureStderr(async () => { + artifact = await runCoreScan( + root, + { format: 'json', concurrency: 1, offline: true, noLocalArtifacts: true, quiet: true }, + advancedScanHook, + ); + }); + logSpy.mockRestore(); + + expect(artifact.projects.length).toBeGreaterThan(0); + const dumped = `${JSON.stringify(artifact)}\n${logs.join('\n')}`; + expect(dumped).not.toContain(SECRET); + expect(stderr).toContain('notice: skipped 1 file that is not UTF-8 text (payload.js)'); + expect(stderr).toContain("vg build --exclude 'payload.js'"); + expect(stderr).not.toContain(SECRET); + expect(stderr).not.toContain(root); + expect(stderr).not.toContain('logo.png'); + + const cache = new FileCache(); + const text = await cache.readTextFile(path.join(root, 'payload.js')); + expect(text).toBe(''); + expect(cache.skippedNonUtf8).toEqual(['payload.js']); + }, 60_000); +});