diff --git a/bin/axiomcode b/bin/axiomcode index 6b902cf00..946f2c38a 100755 --- a/bin/axiomcode +++ b/bin/axiomcode @@ -324,16 +324,61 @@ case "$cmd" in # a caller that checks it is no worse off. Only the languages that produced a graph are # listed at the end, so the output stays a list of files that exist. solved=(); failed=() - for l in "${langs[@]}"; do - echo "▶ solving $l → $out/$l/graph.sqlite" - if "$0" engine --language "$l" --client-ir "$ir/$l" --out "$out/$l" --intermediate "$int" --library "$libroots" \ - --meta "source_version=$version" --meta "source_dir=$(cd "$src" && pwd)" ${rest[@]+"${rest[@]}"}; then - solved+=("$l"); [ -z "$progress" ] || echo "$l ok $out/$l/graph.sqlite" >> "$progress" + # Each language gets its OWN intermediate: run-souffle writes fixed names there + # (souffle-program.cpp, .souffle-gen.log), so two concurrent solves into one + # directory would overwrite each other's program mid-compile. + solve_one(){ local sl="$1" + "$0" engine --language "$sl" --client-ir "$ir/$sl" --out "$out/$sl" --intermediate "$int/$sl" --library "$libroots" \ + --meta "source_version=$version" --meta "source_dir=$(cd "$src" && pwd)" ${rest[@]+"${rest[@]}"} + } + record(){ local rl="$1" rrc="$2" + if [ "$rrc" -eq 0 ]; then + solved+=("$rl"); [ -z "$progress" ] || echo "$rl ok $out/$rl/graph.sqlite" >> "$progress" else - failed+=("$l"); echo "❌ $l failed; continuing with the remaining languages" >&2 - [ -z "$progress" ] || echo "$l failed" >> "$progress" + failed+=("$rl"); echo "❌ $rl failed; continuing with the remaining languages" >&2 + [ -z "$progress" ] || echo "$rl failed" >> "$progress" fi - done + } + # THE FIRST (LARGEST) LANGUAGE SOLVES ALONE, streaming, exactly as before: its graph is the one a + # --progress caller publishes first (#1555), and nothing may compete with it for cores or + # interleave its log. Only the languages AFTER it run concurrently — they used to wait in line + # behind each other for no reason: their solves share nothing (own IR, own output, own + # intermediate). Memory is the tradeoff (each is its own Soufflé process), so the width is + # modest by default and AXIOMCODE_SOLVE_JOBS raises or lowers it; 1 restores the strict line. + l="${langs[0]}" + echo "▶ solving $l → $out/$l/graph.sqlite" + rc=0; solve_one "$l" || rc=$?; record "$l" "$rc" + if [ ${#langs[@]} -gt 1 ]; then + others=("${langs[@]:1}") + JOBS="${AXIOMCODE_SOLVE_JOBS:-2}" + case "$JOBS" in (*[!0-9]*|'') JOBS=2;; esac; [ "$JOBS" -ge 1 ] || JOBS=1 + if [ "$JOBS" -eq 1 ]; then + for l in "${others[@]}"; do + echo "▶ solving $l → $out/$l/graph.sqlite" + rc=0; solve_one "$l" || rc=$?; record "$l" "$rc" + done + else + # Waves of $JOBS (bash 3.2 has no `wait -n`): start a wave, wait for all of it, replay each + # log whole so the build log never interleaves, then record in the wave's order. + i=0 + while [ $i -lt ${#others[@]} ]; do + wave=("${others[@]:$i:$JOBS}"); i=$((i + JOBS)) + wavepids=() + for l in "${wave[@]}"; do + echo "▶ solving $l → $out/$l/graph.sqlite (concurrent)" + { solve_one "$l" > "$int/solve-$l.log" 2>&1; echo $? > "$int/solve-$l.rc"; } & + wavepids+=($!) + done + for wp in "${wavepids[@]}"; do wait "$wp" || true; done + for l in "${wave[@]}"; do + cat "$int/solve-$l.log" 2>/dev/null + rc="$(cat "$int/solve-$l.rc" 2>/dev/null || echo 1)" + rm -f "$int/solve-$l.log" "$int/solve-$l.rc" + record "$l" "$rc" + done + done + fi + fi for l in ${solved[@]+"${solved[@]}"}; do echo "$out/$l/graph.sqlite"; done if [ ${#failed[@]} -gt 0 ]; then echo "❌ ${#failed[@]} of ${#langs[@]} languages failed: ${failed[*]}" >&2 diff --git a/parser/package.json b/parser/package.json index eac08e048..175cb0199 100644 --- a/parser/package.json +++ b/parser/package.json @@ -1,7 +1,7 @@ { "name": "@axiomcode/parser", - "version": "0.1.2", - "description": "AxiomCode Parser — compiles source code and build configuration into a relational intermediate representation.", + "version": "0.1.3", + "description": "AxiomCode Parser \u2014 compiles source code and build configuration into a relational intermediate representation.", "main": "dist/extract.js", "types": "dist/extract.d.ts", "scripts": { diff --git a/parser/src/parsers/csharp/csharp-parser.ts b/parser/src/parsers/csharp/csharp-parser.ts index f37143aff..2e1a43069 100644 --- a/parser/src/parsers/csharp/csharp-parser.ts +++ b/parser/src/parsers/csharp/csharp-parser.ts @@ -6,6 +6,7 @@ import Parser from 'tree-sitter'; import CSharp from 'tree-sitter-c-sharp'; import { FILE_EXTENSIONS } from '@/constants/consts'; +import { materializeTree } from '@/parsers/mirror-tree'; import { CSHARP_CALLBACK_PARSE_THRESHOLD, CSHARP_PARSE_CHUNK_SIZE, @@ -81,6 +82,20 @@ export function stripUtf8Bom(sourceCode: string): string { return sourceCode.startsWith(UTF8_BOM) ? sourceCode.slice(UTF8_BOM.length) : sourceCode; } +/** tree-sitter-c-sharp `extras`: the only node types that parse as extra. */ +const CS_EXTRA_TYPES: ReadonlySet = new Set([ + 'comment', + 'preproc_region', + 'preproc_endregion', + 'preproc_line', + 'preproc_pragma', + 'preproc_nullable', + 'preproc_error', + 'preproc_warning', + 'preproc_define', + 'preproc_undef', +]); + export class CSharpParser implements LanguageParser { readonly language = ProjectLanguage.CSHARP; readonly fileExtension = FILE_EXTENSIONS.CSHARP; @@ -142,7 +157,16 @@ export class CSharpParser implements LanguageParser { return parseWithRetry(source); } - getRootNode(tree: Parser.Tree): Parser.SyntaxNode { + getRootNode(tree: Parser.Tree, sourceCode?: string): Parser.SyntaxNode { + // With `sourceCode`, the root is a one-pass plain-JS mirror of the tree + // (see `../mirror-tree`): every stage after it reads JS properties + // instead of re-crossing the tree-sitter FFI per property access. + // Tree-sitter still parses every file. + if (sourceCode !== undefined) { + // parse() strips a leading BOM, so the mirror slices text from the SAME + // string the tree's byte offsets are relative to, whatever was passed. + return materializeTree(tree, stripUtf8Bom(sourceCode), CS_EXTRA_TYPES) as unknown as Parser.SyntaxNode; + } return tree.rootNode; } diff --git a/parser/src/parsers/csharp/extractors/cs-fact-extractor.ts b/parser/src/parsers/csharp/extractors/cs-fact-extractor.ts index 31fc4fe42..aea82ba54 100644 --- a/parser/src/parsers/csharp/extractors/cs-fact-extractor.ts +++ b/parser/src/parsers/csharp/extractors/cs-fact-extractor.ts @@ -216,11 +216,11 @@ export class CsFactExtractor { // is. Length-preserving to the character; the receiver the blanking // removes is returned in a side table. See cs-extension-block.ts. const flattened = flattenExtensionBlocks(rewritten, (source) => - this.parser.getRootNode(this.parser.parse(source)) + this.parser.getRootNode(this.parser.parse(source), source) ); const parseText = flattened.text.endsWith('\n') ? flattened.text : `${flattened.text}\n`; const tree = this.parser.parse(parseText); - const root = this.parser.getRootNode(tree); + const root = this.parser.getRootNode(tree, parseText); // The symbol set this emission is compiled under: what the caller supplied, // plus the implicit framework symbols the SDK injects and no `.csproj` diff --git a/parser/src/parsers/java/extractors/import-extractor.ts b/parser/src/parsers/java/extractors/import-extractor.ts index d6d98f026..6323cb422 100644 --- a/parser/src/parsers/java/extractors/import-extractor.ts +++ b/parser/src/parsers/java/extractors/import-extractor.ts @@ -54,7 +54,7 @@ export class ImportExtractor implements BaseExtractor { } const tree = this.javaParser.parse(fileContent); - const rootNode = this.javaParser.getRootNode(tree); + const rootNode = this.javaParser.getRootNode(tree, fileContent); this.extractImportsFromRoot(rootNode, filePath, serviceVersionHash, imports); } catch (error) { diff --git a/parser/src/parsers/java/extractors/type-registry-extractor.ts b/parser/src/parsers/java/extractors/type-registry-extractor.ts index 993630bf1..680ae6549 100644 --- a/parser/src/parsers/java/extractors/type-registry-extractor.ts +++ b/parser/src/parsers/java/extractors/type-registry-extractor.ts @@ -203,7 +203,7 @@ export class TypeRegistryExtractor implements BaseExtractor { } const tree = this.javaParser.parse(fileContent); - const rootNode = this.javaParser.getRootNode(tree); + const rootNode = this.javaParser.getRootNode(tree, fileContent); const basePath = this.extractBasePath(filePath); const fileName = path.basename(filePath); diff --git a/parser/src/parsers/java/java-parser.ts b/parser/src/parsers/java/java-parser.ts index 1ec23cc56..e444c02fa 100644 --- a/parser/src/parsers/java/java-parser.ts +++ b/parser/src/parsers/java/java-parser.ts @@ -3,9 +3,13 @@ import Java from 'tree-sitter-java'; import { FILE_EXTENSIONS } from '@/constants/consts'; import { LanguageParser } from '@/parsers/language-parser'; +import { materializeTree } from '@/parsers/mirror-tree'; import { ProjectLanguage } from '@/types/ProjectInfo'; import { withRetry } from '@/utils/retry-decorator'; +/** tree-sitter-java `extras`: the only node types that parse as extra. */ +const JAVA_EXTRA_TYPES: ReadonlySet = new Set(['line_comment', 'block_comment']); + /** * Java-specific tree-sitter parser implementation */ @@ -68,11 +72,21 @@ export class JavaParser implements LanguageParser { } /** - * Gets the root node of a parsed tree + * Gets the root node of a parsed tree. + * + * With `sourceCode`, the root is a one-pass plain-JS mirror of the tree + * (see `../mirror-tree`): every stage after it reads JS properties instead + * of re-crossing the tree-sitter FFI per property access. Tree-sitter still + * parses every file. Without it, the real tree-sitter root is returned. + * * @param tree Parsed syntax tree + * @param sourceCode The exact string the tree was parsed from * @returns Root syntax node */ - getRootNode(tree: Parser.Tree): Parser.SyntaxNode { + getRootNode(tree: Parser.Tree, sourceCode?: string): Parser.SyntaxNode { + if (sourceCode !== undefined) { + return materializeTree(tree, sourceCode, JAVA_EXTRA_TYPES) as unknown as Parser.SyntaxNode; + } return tree.rootNode; } diff --git a/parser/src/parsers/mirror-tree.ts b/parser/src/parsers/mirror-tree.ts new file mode 100644 index 000000000..ad88b17f3 --- /dev/null +++ b/parser/src/parsers/mirror-tree.ts @@ -0,0 +1,307 @@ +/** + * A plain-JS mirror of a tree-sitter tree, built in ONE cursor pass — the + * shared machinery behind each language's `materializeTree`. + * + * Tree-sitter still parses every file; what this removes is the reading cost. + * Every property access on a tree-sitter SyntaxNode crosses the JS↔C++ + * boundary and re-marshals the node handle, and the extraction stages each + * walk the same tree, so one node's properties are fetched once per stage. + * The mirror pays the boundary once per node, during the cursor walk, and + * every later read is a JS property. + * + * The surface is the union of what the tree-sitter-reading stages use + * (verified by grep per language): type, text, children/namedChildren, + * child(i)/namedChild(i), childForFieldName, fieldNameForChild, counts, + * spans, parent, named siblings, id, isNamed/isMissing/isExtra/hasError. + * Anything outside it throws at the call site rather than answering wrongly. + * + * `text` is sliced lazily from the one source string, so the mirror holds no + * copies. `isExtra` is derived from the node type against the language's own + * `extras` set, with one exception read from the real node (see constructor). + * `hasError` is computed bottom-up with tree-sitter's meaning — an ERROR or + * missing node anywhere in the subtree — and the ROOT's flag is copied from + * tree-sitter itself, because an error can live in a HIDDEN node that no + * visible child betrays. + */ +import type Parser from 'tree-sitter'; + +/** Never reset: `HashByNodeId` maps must not collide across files. */ +let nextId = 1; + +export class MirrorNode { + readonly id: number; + readonly type: string; + readonly isNamed: boolean; + readonly isMissing: boolean; + readonly startIndex: number; + readonly endIndex: number; + readonly startPosition: Parser.Point; + readonly endPosition: Parser.Point; + parent: MirrorNode | null = null; + readonly children: MirrorNode[] = []; + namedChildren: MirrorNode[] = []; + /** First child per field name — the pick childForFieldName makes. */ + private fields: Map | null = null; + /** Field name per child index, for fieldNameForChild; null when none has one. */ + private childFields: (string | null)[] | null = null; + /** Index within parent.children, for the named-sibling getters. */ + private childIndex = -1; + private errorInSubtree = false; + /** Set at build time only for ERROR nodes — see the constructor. */ + private extraOverride: boolean | null = null; + private readonly source: string; + private readonly extraTypes: ReadonlySet; + + constructor(cursor: Parser.TreeCursor, source: string, extraTypes: ReadonlySet) { + this.id = nextId++; + this.type = cursor.nodeType; + // An ERROR node can be either: tree-sitter marks an ERROR it absorbed + // during recovery as EXTRA (siblings' named counts then skip it), while a + // plain ERROR is not. The type cannot tell them apart, so this is the one + // place the real node is consulted — ERROR nodes exist only in files that + // failed to parse, so the boundary crossing stays off the healthy path. + if (this.type === 'ERROR') { + this.extraOverride = cursor.currentNode.isExtra; + } + this.isNamed = cursor.nodeIsNamed; + this.isMissing = cursor.nodeIsMissing; + this.startIndex = cursor.startIndex; + this.endIndex = cursor.endIndex; + this.startPosition = cursor.startPosition; + this.endPosition = cursor.endPosition; + this.source = source; + this.extraTypes = extraTypes; + } + + get text(): string { + return this.source.slice(this.startIndex, this.endIndex); + } + + get isExtra(): boolean { + if (this.extraOverride !== null) { + return this.extraOverride; + } + return this.extraTypes.has(this.type); + } + + get hasError(): boolean { + return this.errorInSubtree; + } + + get childCount(): number { + return this.children.length; + } + + get namedChildCount(): number { + return this.namedChildren.length; + } + + child(index: number): MirrorNode | null { + return this.children[index] ?? null; + } + + namedChild(index: number): MirrorNode | null { + return this.namedChildren[index] ?? null; + } + + // The named-sibling getters answer from ANY node, anonymous ones included + // (tree-sitter scans the parent's children positionally), so they scan from + // this node's position rather than indexing namedChildren. They are read a + // handful of times per file (comment attachment), never in a hot loop. + get previousNamedSibling(): MirrorNode | null { + if (this.parent === null) { + return null; + } + for (let i = this.childIndex - 1; i >= 0; i--) { + const sibling = this.parent.children[i]; + if (sibling !== undefined && sibling.isNamed) { + return sibling; + } + } + return null; + } + + get nextNamedSibling(): MirrorNode | null { + if (this.parent === null) { + return null; + } + for (let i = this.childIndex + 1; i < this.parent.children.length; i++) { + const sibling = this.parent.children[i]; + if (sibling !== undefined && sibling.isNamed) { + return sibling; + } + } + return null; + } + + /** + * First IMMEDIATE child carrying the field, which is what every extractor + * asks for. Tree-sitter's own lookup can additionally pierce one visible + * level where a grammar attaches a field inside a hidden rule (Python's + * `match_statement` reaches its case clauses' `alternative` through the + * match block), a quirk no stage uses: the consumers iterate those children + * by type instead. + */ + childForFieldName(fieldName: string): MirrorNode | null { + return this.fields?.get(fieldName) ?? null; + } + + fieldNameForChild(index: number): string | null { + return this.childFields?.[index] ?? null; + } + + /** @internal set when any direct child is an extra — see _repairFieldsFrom. */ + _needsFieldRepair = false; + + /** @internal build-time wiring, called only by materializeTree. */ + _addChild(child: MirrorNode, fieldName: string | null): void { + child.parent = this; + child.childIndex = this.children.length; + this.children.push(child); + if (child.isNamed) { + this.namedChildren.push(child); + } + if (child.isExtra) { + this._needsFieldRepair = true; + } + if (fieldName !== null && fieldName !== '' && fieldName !== undefined) { + if (this.fields === null) { + this.fields = new Map(); + } + if (!this.fields.has(fieldName)) { + this.fields.set(fieldName, child); + } + if (this.childFields === null) { + this.childFields = []; + } + this.childFields[this.children.length - 1] = fieldName; + } + } + + /** + * @internal Re-reads this node's field layout from the real node. + * + * Around an EXTRA child (a comment inside the construct) the cursor's field + * reporting diverges from the node API in two ways: on a chunk-parsed file + * (over tree-sitter's string-length ceiling) the field of the sibling after + * the extra can come back empty, and the node API itself labels the extra + * with the preceding field. A node with an extra child therefore copies the + * layout wholesale — these are only the comment-bearing nodes, so the + * boundary crossings stay rare. + */ + _repairFieldsFrom(real: Parser.SyntaxNode): void { + this.fields = null; + this.childFields = null; + for (let i = 0; i < this.children.length; i++) { + const fieldName = real.fieldNameForChild(i) ?? null; + if (fieldName === null || fieldName === '') { + continue; + } + const child = this.children[i]; + if (child === undefined) { + continue; + } + // The two node APIs disagree around extras, and the mirror keeps both + // behaviours: fieldNameForChild labels an extra sitting on a field + // position, while childForFieldName SKIPS extras and answers the first + // non-extra carrier. + if (!child.isExtra) { + if (this.fields === null) { + this.fields = new Map(); + } + if (!this.fields.has(fieldName)) { + this.fields.set(fieldName, child); + } + } + if (this.childFields === null) { + this.childFields = []; + } + this.childFields[i] = fieldName; + } + } + + /** @internal */ + _markError(): void { + this.errorInSubtree = true; + } +} + +/** One depth-first cursor pass over the freshly parsed tree. */ +export function materializeTree( + tree: Parser.Tree, + source: string, + extraTypes: ReadonlySet +): MirrorNode { + const cursor = tree.walk(); + const root = new MirrorNode(cursor, source, extraTypes); + // hasError at the ROOT is read from tree-sitter itself (one boundary call + // per file): an error can live in a HIDDEN node — a file whose syntax error + // is swallowed shows no visible ERROR/missing child anywhere, yet + // ts_node_has_error is true, and module-level "partial grammar" columns + // depend on exactly that. When the root does carry an error, every node's + // flag is read from the real node instead of propagated bottom-up: the + // parse-gap stages walk for the DEEPEST hasError node, and a node can + // report it on itself with no visible ERROR child (a bare preproc_pragma + // does). The per-node boundary crossings are confined to the files that + // failed to parse; a healthy file pays one. + const exactErrors = tree.rootNode.hasError; + if (exactErrors) { + root._markError(); + } + const stack: MirrorNode[] = [root]; + let current = root; + + // gotoFirstChild / gotoNextSibling / gotoParent, no recursion: a deeply + // nested file must not overflow the JS stack when the C parser handled it. + let descending = true; + for (;;) { + if (descending && cursor.gotoFirstChild()) { + const child = new MirrorNode(cursor, source, extraTypes); + if (exactErrors && cursor.currentNode.hasError) { + child._markError(); + } + current._addChild(child, cursor.currentFieldName); + stack.push(child); + current = child; + continue; + } + // finishing `current`: the cursor sits on it, so repair its fields here + // if an extra child made the cursor's reporting untrustworthy… + if (current._needsFieldRepair) { + current._repairFieldsFrom(cursor.currentNode); + current._needsFieldRepair = false; + } + // …and fold its error state into the parent + if (current.type === 'ERROR' || current.isMissing || current.hasError) { + const parent = stack[stack.length - 2]; + if (parent !== undefined) { + parent._markError(); + } + current._markError(); + } + if (cursor.gotoNextSibling()) { + stack.pop(); + const parent = stack[stack.length - 1]; + if (parent === undefined) { + // the root has no siblings; the cursor cannot get here + return root; + } + const sibling = new MirrorNode(cursor, source, extraTypes); + if (exactErrors && cursor.currentNode.hasError) { + sibling._markError(); + } + parent._addChild(sibling, cursor.currentFieldName); + stack.push(sibling); + current = sibling; + descending = true; + continue; + } + stack.pop(); + const above = stack[stack.length - 1]; + if (!cursor.gotoParent() || above === undefined) { + return root; + } + current = above; + descending = false; + } +} diff --git a/parser/src/parsers/python/extractors/python-resolution-linker.ts b/parser/src/parsers/python/extractors/python-resolution-linker.ts index 3361359ab..798ac85b1 100644 --- a/parser/src/parsers/python/extractors/python-resolution-linker.ts +++ b/parser/src/parsers/python/extractors/python-resolution-linker.ts @@ -349,6 +349,14 @@ export class PythonResolutionLinker { // ---- step 2: bases, now that imports are resolved // Per-module views of what each module's imports brought into scope. + // Modules keyed by hash once, FIRST occurrence kept — the lookup below ran as a + // linear scan per import record, which is quadratic over the project. + const moduleByHash = new Map(); + for (const module of modules) { + if (!moduleByHash.has(module.moduleHash)) { + moduleByHash.set(module.moduleHash, module); + } + } const importedTypeByName = new Map>(); const importedModuleByName = new Map>(); for (const module of modules) { @@ -365,7 +373,7 @@ export class PythonResolutionLinker { // The bound name refers to a module. Find which one by matching the // resolved module hash, so `from . import protocols` and // `import pkg.protocols` are handled by the same lookup. - const target = modules.find(m => m.moduleHash === record.getResolvedModuleLinkHash()); + const target = moduleByHash.get(record.getResolvedModuleLinkHash()); if (target) { mods.set(record.getSimpleName(), target); } @@ -383,8 +391,14 @@ export class PythonResolutionLinker { const aliasByModule = new Map>(); for (const module of modules) { const scoped = new Map(); + // Bindings keyed by hash, FIRST occurrence kept, matching the linear + // `.find` this replaces; built in the pass that already walks them. + const bindingByHash = new Map(); for (const binding of module.bindings) { scoped.set(`${binding.getPyScopeLinkHash()}::${binding.getName()}`, binding); + if (!bindingByHash.has(binding.getHash())) { + bindingByHash.set(binding.getHash(), binding); + } } const parents = new Map(); for (const scope of module.scopes) { @@ -411,7 +425,7 @@ export class PythonResolutionLinker { }); const byName = new Map(); for (const [bindingHash, entity] of aliases) { - const binding = module.bindings.find(b => b.getHash() === bindingHash); + const binding = bindingByHash.get(bindingHash); if (binding) { byName.set(binding.getName(), entity); } diff --git a/parser/src/parsers/python/extractors/python-scope-extractor.ts b/parser/src/parsers/python/extractors/python-scope-extractor.ts index cd379ac58..8d8abc5f3 100644 --- a/parser/src/parsers/python/extractors/python-scope-extractor.ts +++ b/parser/src/parsers/python/extractors/python-scope-extractor.ts @@ -30,6 +30,7 @@ import { SymbolFlags, SymbolScope, } from '@/parsers/python/extractors/python-symbol-table'; +import { materializePyTree } from '@/parsers/python/py-mirror-tree'; import { Python2Finding, SymbolBlock } from '@/parsers/python/types'; import { PythonSourcePositions } from '@/utils/python'; @@ -144,7 +145,10 @@ export class PythonScopeExtractor { } const tree = this.parser.parse(input.sourceCode); - const rootNode = this.parser.getRootNode(tree); + // One cursor pass mirrors the tree into plain JS; every stage after this + // line reads JS properties instead of re-crossing the tree-sitter FFI. + // Tree-sitter itself still parses every file — see py-mirror-tree.ts. + const rootNode = materializePyTree(tree, input.sourceCode) as unknown as Parser.SyntaxNode; const detection = this.detector.detect(rootNode, input.sourceCode); if (detection.dialect !== PythonDialect.PY3) { diff --git a/parser/src/parsers/python/py-mirror-tree.ts b/parser/src/parsers/python/py-mirror-tree.ts new file mode 100644 index 000000000..de9401848 --- /dev/null +++ b/parser/src/parsers/python/py-mirror-tree.ts @@ -0,0 +1,17 @@ +/** + * Python's one-pass plain-JS mirror of the tree-sitter tree. The machinery + * and the reasoning live in `../mirror-tree`; this module only pins the + * language's `extras`: tree-sitter-python's are exactly `comment` and + * `line_continuation` (whitespace produces no node). + */ +import type Parser from 'tree-sitter'; + +import { MirrorNode, materializeTree } from '@/parsers/mirror-tree'; + +const PY_EXTRA_TYPES: ReadonlySet = new Set(['comment', 'line_continuation']); + +export type PyMirrorNode = MirrorNode; + +export function materializePyTree(tree: Parser.Tree, source: string): MirrorNode { + return materializeTree(tree, source, PY_EXTRA_TYPES); +} diff --git a/parser/src/workflows/typescript/ts-relation-writer.ts b/parser/src/workflows/typescript/ts-relation-writer.ts index da00771ea..f431759d3 100644 --- a/parser/src/workflows/typescript/ts-relation-writer.ts +++ b/parser/src/workflows/typescript/ts-relation-writer.ts @@ -43,7 +43,9 @@ export class TsRelationWriter { private readonly outputPath: string; private buffer: string[] = []; private header = ''; + private width = 0; private rows = 0; + private bytesWritten = 0; private closed = false; constructor(outputDir: string, filename: string, uniqueSuffix: string) { @@ -71,10 +73,18 @@ export class TsRelationWriter { if (this.handle === undefined) { this.handle = await fsp.open(this.temporaryPath, 'w'); this.header = rows[0]!.getCsvHeader(); + this.width = countTabs(this.header) + 1; this.buffer.push(this.header + '\n'); } for (const row of rows) { - this.buffer.push(row.toCsv() + '\n'); + const line = row.toCsv(); + // The row is checked HERE, on the string that is about to be written, + // instead of decoding the finished file a second time: same width rule, + // same line-break alphabet, no re-read. What this no longer re-checks — + // that the bytes reached the disk whole — publish() covers by comparing + // the byte count it wrote against what the file system reports. + verifyRow(line, this.width, this.outputPath, this.rows + 2); + this.buffer.push(line + '\n'); this.rows += 1; } if (this.buffer.length >= TS_CSV_CHUNK_SIZE) { @@ -90,7 +100,8 @@ export class TsRelationWriter { // streaming cost the same as the whole-file writer did. const text = this.buffer.join(''); this.buffer = []; - await this.handle.write(text, null, 'utf-8'); + const { bytesWritten } = await this.handle.write(text, null, 'utf-8'); + this.bytesWritten += bytesWritten; } /** Flushes, verifies, and renames into place. */ @@ -107,9 +118,17 @@ export class TsRelationWriter { } await this.flush(); await this.handle.sync(); + // Every row was verified as it was appended (verifyRow); what remains to + // prove is that the bytes all arrived. The file's size must equal the sum + // of what write() reported — a mismatch is a torn write, the exact defect + // the old whole-file read-back existed to catch. + const onDisk = (await this.handle.stat()).size; await this.handle.close(); this.handle = undefined; - verifyRelationFileStreaming(this.temporaryPath, this.outputPath, this.header); + if (onDisk !== this.bytesWritten) { + throw new Error(`${path.basename(this.outputPath)}: wrote ${this.bytesWritten} byte(s) but the ` + + `file holds ${onDisk} — the write is torn`); + } await fsp.rename(this.temporaryPath, this.outputPath); } @@ -139,6 +158,43 @@ export class TsRelationWriter { */ const CONSUMER_LINE_BREAKS = /[\u000A\u000B\u000C\u000D\u001C\u001D\u001E\u0085\u2028\u2029]/; +function countTabs(line: string): number { + let tabs = 0; + for (let i = 0; i < line.length; i++) { + if (line.charCodeAt(i) === 0x09) { + tabs += 1; + } + } + return tabs; +} + +/** + * One row holds exactly the header's field count and no code point a consumer + * would break a line on \u2014 the same rules {@link verifyRelationFileStreaming} + * applies, checked on the in-memory string in one allocation-free pass. + */ +function verifyRow(line: string, width: number, outputPath: string, lineNumber: number): void { + if (line === '') { + // the streamed read-back skipped blank lines rather than calling them torn + return; + } + let tabs = 0; + for (let i = 0; i < line.length; i++) { + const c = line.charCodeAt(i); + if (c === 0x09) { + tabs += 1; + } else if ((c >= 0x0a && c <= 0x0d) || (c >= 0x1c && c <= 0x1e) || c === 0x85 + || c === 0x2028 || c === 0x2029) { + throw new Error(`${path.basename(outputPath)}: line ${lineNumber} carries a line-break code ` + + `point inside a value \u2014 the row would read torn: ${JSON.stringify(line.slice(0, 60))}`); + } + } + if (tabs + 1 !== width) { + throw new Error(`${path.basename(outputPath)}: line ${lineNumber} has ${tabs + 1} field(s) where ` + + `the header has ${width} \u2014 the row is torn: ${JSON.stringify(line.slice(0, 60))}`); + } +} + /** * Every row has exactly the header's field count, checked without holding the * file in memory. diff --git a/parser/src/workflows/typescript/typescript-project-analyzer.ts b/parser/src/workflows/typescript/typescript-project-analyzer.ts index 447c6fcdf..26287df5b 100644 --- a/parser/src/workflows/typescript/typescript-project-analyzer.ts +++ b/parser/src/workflows/typescript/typescript-project-analyzer.ts @@ -331,8 +331,13 @@ export class TypeScriptProjectAnalyzer { for (const file of files) { let sourceText: string; + // the closure walk read this file already; take its text and release it + const prefetched = rootProgram?.texts.get(file); + if (prefetched !== undefined) { + rootProgram!.texts.delete(file); + } try { - sourceText = await fsp.readFile(file, 'utf-8'); + sourceText = prefetched ?? await fsp.readFile(file, 'utf-8'); } catch (error) { this.recordSkip(file, pathAnchor, options, serviceVersionLinkHash, SkippedFileReason.READ_ERROR, String(error)); @@ -609,7 +614,13 @@ export class TypeScriptProjectAnalyzer { function filesOfRootProgram( rootDir: string, configResolver: TsConfigResolver -): { readonly files: string[]; readonly others: string[]; readonly orphans: string[] } | undefined { +): { + readonly files: string[]; + readonly others: string[]; + readonly orphans: string[]; + /** What the closure walk already read, so the extraction pass reads nothing twice. */ + readonly texts: Map; +} | undefined { const configPath = path.join(rootDir, 'tsconfig.json'); if (!fs.existsSync(configPath)) { return undefined; @@ -660,6 +671,11 @@ function filesOfRootProgram( // by a nested tsconfig stays in that program, which is what keeps a nested // project's separate global scope separate. const rootOptions = configResolver.resolve(claimed[0] ?? configPath).options; + // One resolution cache for the whole walk: ts.resolveModuleName with a bare + // ts.sys re-probes the same node_modules directories for every specifier, + // and the probing (statSync/readdirSync) was most of this pass's cost. + const resolutionCache = ts.createModuleResolutionCache(rootDir, (f) => f, rootOptions); + const texts = new Map(); const included = new Set(claimed.map((f) => path.normalize(f))); const available = new Map(unclaimed.map((f) => [path.normalize(f), f])); const queue = [...claimed]; @@ -671,13 +687,14 @@ function filesOfRootProgram( } catch { continue; } + texts.set(current, text); // No parent pointers and no type nodes needed: this pass only reads // specifiers, so the cheapest possible parse is the right one. const script = scriptTextOf(current, text); const sf = ts.createSourceFile(current, script.text, ts.ScriptTarget.Latest, false, script.scriptKind ?? (current.endsWith('.tsx') ? ts.ScriptKind.TSX : ts.ScriptKind.TS)); for (const specifier of importSpecifiersOf(sf)) { - const resolved = ts.resolveModuleName(specifier, current, rootOptions, ts.sys) + const resolved = ts.resolveModuleName(specifier, current, rootOptions, ts.sys, resolutionCache) .resolvedModule?.resolvedFileName ?? resolveVueSpecifier(specifier, current); if (resolved === undefined) { continue; @@ -702,7 +719,7 @@ function filesOfRootProgram( orphans.push(f); } } - return { files, others, orphans }; + return { files, others, orphans, texts }; } /** diff --git a/plugins/axiomcode/skills/axiomcode/scripts/ax_text.py b/plugins/axiomcode/skills/axiomcode/scripts/ax_text.py index ada6242b0..3ef55ac3f 100644 --- a/plugins/axiomcode/skills/axiomcode/scripts/ax_text.py +++ b/plugins/axiomcode/skills/axiomcode/scripts/ax_text.py @@ -644,6 +644,13 @@ def block(repo, asked, scope=None, why='unresolved', rows=ROWS): if rest > 0: lines.append(f" … +{rest} more line(s): git grep -n{'w' if w else ''} -F -e '{n}'" if not pat else f" … +{rest} more line(s): git grep -nP -e '{pat.pattern}'") + # a name no graph declares that ARRIVES THROUGH AN IMPORT is the signature of an unstaged dependency, the + # commonest setup gap there is: without the hint this answer is indistinguishable from an engine gap, and the + # fix (stage the dependency's source) is one flag away. Only for undeclared names: a string or a resolved + # target wants no staging advice. + if why == 'undeclared' and any(re.match(r'\s*(import\b|using\b|from\s)', t.strip()) and n in t for _f, _l, t in hits): + lines.append(f" this name arrives through an import, so it is likely declared in a dependency: calls through " + f"it resolve once that dependency's source is staged — `axiomcode index --library `") if not lines: return '' if approxed: lines.append("next: [approx] rows are text placed in the declaration that holds it, not resolved edges: read the " "evidence line, then `impact ` for what a change reaches") diff --git a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-build b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-build index 60068b7bc..abd5d098b 100755 --- a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-build +++ b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-build @@ -495,12 +495,16 @@ w = max([len('tier')] + [len(t) for t, _ in rows]) print(f"{'tier':<{w}} edges") for t, n in rows: print(f"{t:<{w}} {n}") PY - python3 "$H/axiomcode-impact" --warm "$REPO" || true # export the graph's facts now (~9 s on 1,373 files): the first - # query pays it otherwise, and for the plugin that first query is inside hooks/changes.py's timeout=14, several at once. + # export the graph's facts for impact IN THE BACKGROUND, after the graph is queryable: the first query pays it otherwise + # (and for the plugin that first query is inside hooks/changes.py's timeout=14), but on a large repository the export + # takes minutes and `index` must not wait on it. fd 9 is the build lock: the child closes it, so the lock is released + # when this build ends, and a query or a second build never waits on the warm-up. Its writes are atomic renames. + ( exec 9>&-; nohup python3 "$H/axiomcode-impact" --warm "$REPO" >/dev/null 2>&1 & ) || true + echo "graph ready; precomputing impact facts in the background" # the query rules: their compile started at the top of the build, detached (a caller's timeout cannot kill it half-way, # which cached nothing and repeated on every edit); this only says where it is. A query never waits for it. python3 "$H/dl_program.py" --no-wait || true - [ -n "$BASE_SAVED" ] && AXIOMCODE_GRAPH="$REPO/.axiomcode/base" python3 "$H/axiomcode-impact" --warm "$REPO" >/dev/null 2>&1 || true + [ -n "$BASE_SAVED" ] && ( exec 9>&-; AXIOMCODE_GRAPH="$REPO/.axiomcode/base" nohup python3 "$H/axiomcode-impact" --warm "$REPO" >/dev/null 2>&1 & ) || true [ -z "$OTHER_LANGS" ] || [ ! -s "$PENDING" ] || echo "$LANG_ARG graph ready at $OUT/graph.sqlite; still building: $(cut -d' ' -f1 "$PENDING" | paste -sd, - | sed 's/,/, /g')" } # EVERY OTHER LANGUAGE'S GRAPH, indexed where the engine wrote it and only then moved into place, as the main one is. diff --git a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-impact b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-impact index e8186697c..609a53b75 100755 --- a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-impact +++ b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-impact @@ -1275,8 +1275,17 @@ class Impact: # a cache written by an older program, or one an interrupted run left half written, breaks every query with # "Cannot open fact file X.facts": accept it only when it holds every relation the program reads need = set(re.findall(r'\.decl (\w+)\([^)]*\)\s+\.input', open(os.path.join(os.path.dirname(os.path.abspath(__file__)), 'dl', 'impact.dl')).read())) - PER_QUERY - if os.path.exists(stamp) and open(stamp).read() == want and all(os.path.exists(os.path.join(D, n + '.facts')) for n in need): return D - os.makedirs(D, exist_ok=True); W = lambda n, rows: g.write(n, rows, D) + def fresh(): + try: return open(stamp).read() == want and all(os.path.exists(os.path.join(D, n + '.facts')) for n in need) + except OSError: return False + if fresh(): return D + os.makedirs(D, exist_ok=True) + if not P.export_lock(D, fresh): return D # another process just exported these same facts + try: return self._export(D, stamp, want) + finally: P.export_unlock(D) + + def _export(self, D, stamp, want): + g = self.g; W = lambda n, rows: g.write(n, rows, D) # owner display -> type id. A display name is NOT unique: two files of one test suite commonly # declare a class of the same name, and keyed on the name alone the second one's members were # filed under the first one's type. Everything scope/2 reaches then crossed between them -- @@ -1508,7 +1517,8 @@ class Impact: L = self.code(r['file']); text = L[r['line'] - 1] if r['line'] <= len(L) else '' for qn in re.findall(rf'([A-Za-z_$][\w$]*)\s*\.\s*{re.escape(r["name"])}\b', text): quals.append((r['file'], r['line'], r['name'], qn)) W('ref', refs); W('qualifier', sorted(set(quals))) - W('registration', ax_registration.registrations(g.q, g.site_file)) + regs = ax_registration.registrations(g.q, g.site_file) # computed once: reg_key_fact below reads it too + W('registration', regs) trs = []; trf = set() # a reference on a module-level type alias's own lines belongs to the alias, not the module (#784) import graph_sql @@ -1590,7 +1600,7 @@ class Impact: regk = [] # `rd` and not `decl`: `decl` is the dec_literal list above for rd, _f, _l, kind, key, _why in (ax_registration.decoration_keys(g.q, g.site_file) + ax_registration.value_route_registrations(g.q, g.site_file) - + ax_registration.registrations(g.q, g.site_file) + tregs): + + regs + tregs): if rd in g.sym and key: regk.append((rd, kind, key)) W('reg_key_fact', sorted(set(regk))) # the callables in test files: a test that publishes a handler table's key drives the handler, and is not diff --git a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-index b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-index index 10a7bfecb..19c50a448 100755 --- a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-index +++ b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-index @@ -699,6 +699,12 @@ CREATE INDEX IF NOT EXISTS symbols_name ON symbols(name); CREATE INDEX IF NOT EX CREATE INDEX IF NOT EXISTS symbols_file ON symbols(file, line); CREATE INDEX IF NOT EXISTS symbols_mid ON symbols(method_id); CREATE INDEX IF NOT EXISTS symbols_tid ON symbols(type_id); CREATE INDEX IF NOT EXISTS refs_name ON refs(name); CREATE INDEX IF NOT EXISTS type_refs_name ON type_refs(name); CREATE INDEX IF NOT EXISTS decorations_owner ON decorations(owner_id); CREATE INDEX IF NOT EXISTS literals_value ON literals(value); CREATE INDEX IF NOT EXISTS comments_file ON comments(file); CREATE INDEX IF NOT EXISTS ce_callee ON call_edges(callee_method_id); CREATE INDEX IF NOT EXISTS ce_caller ON call_edges(caller_id); +-- lookups by id and by file + line: without them each per-site or per-test query scans the whole table +CREATE INDEX IF NOT EXISTS symbols_id ON symbols(id); CREATE INDEX IF NOT EXISTS refs_file_line ON refs(file, line); +CREATE INDEX IF NOT EXISTS literals_file_line ON literals(file, line); +-- the fields a caller reads: via_base_rows asks per single-target call site, and without the index each ask +-- scanned the whole table (65,797 rows × 5,162 sites on an 8,619-file repository = 24 s of a 171 s warm → 1.4 s) +CREATE INDEX IF NOT EXISTS fa_caller ON field_access(caller_id); -- who calls a method: one row per (site, target), with the engine's tier. caller_id is symbols.id: a -- Java call written in a field initializer has the TYPE as its caller, and that row is kept, not dropped CREATE VIEW callers AS diff --git a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path index 992105643..55b029565 100755 --- a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path +++ b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path @@ -128,6 +128,43 @@ def replace_file(tmp, dst): shutil.copyfile(tmp, dst); os.remove(tmp) +def export_lock(d, fresh): + """ONE exporter per facts directory. A query that found the stamp stale while another process (the build's + background warm-up, a concurrent query, several hooks at once) was already writing these facts ran the whole + export AGAIN beside it: on an 8,619-file repository the first query after `index` cost 193 s against 8.8 s + once warm, every concurrent query the same again, and the copies thrashed each other. Returns True with the + lock held — the caller exports, then export_unlock in a finally — or False once another process finished the + same export (`fresh` says so) and the caller just reads it. A lock whose writer is gone is taken over.""" + lock = os.path.join(d, '.exporting') + while True: + try: + fd = os.open(lock, os.O_CREAT | os.O_EXCL | os.O_WRONLY) + os.write(fd, str(os.getpid()).encode()); os.close(fd); return True + except FileExistsError: + pass + gone = False + try: + os.kill(int(open(lock).read().strip() or '0'), 0) + except (ProcessLookupError, ValueError): + gone = True + except PermissionError: + pass # alive, another user's process: wait for it + except OSError: # Windows cannot probe another session's pid: fall back to the lock's age + try: gone = time.time() - os.path.getmtime(lock) > 900 + except OSError: continue # the lock vanished under us: contend for it again + if gone: + try: os.remove(lock) + except OSError: pass + continue + time.sleep(0.5) + if fresh(): return False + + +def export_unlock(d): + try: os.remove(os.path.join(d, '.exporting')) + except OSError: pass + + class G: def __init__(self, repo): self.repo = os.path.realpath(repo or '.') @@ -932,10 +969,16 @@ class G: unfoll = [(r['lib'], r['n']) for r in rows if r['from_site']][:cap] return named, unfoll def site_file(self, raw): - """a call site's file as the repo sees it: Java bundles store absolute paths; the index's paths table maps them""" - if self.has('paths'): - r = self.q("SELECT rel FROM paths WHERE raw = ?", raw) - if r: return r[0]['rel'] + """a call site's file as the repo sees it: Java bundles store absolute paths; the index's paths table maps them. + The map is read ONCE PER GRAPH, not once per row: this ran a query per call (one `has`, one lookup), and the + facts export calls it for every call edge — 319k edges on an 8,619-file repository made 638k round-trips + through sqlite, which was most of the export's time (reg_key_fact 36 s, calls 34 s, registration 31 s, + via_base 25 s of a 171 s warm).""" + m = getattr(self, '_site_files', None) + if m is None: + m = self._site_files = {r['raw']: r['rel'] for r in self.q("SELECT raw, rel FROM paths")} if self.has('paths') else {} + got = m.get(raw) + if got is not None: return got return os.path.relpath(raw, self.repo).replace(os.sep, '/') if os.path.isabs(raw) and raw.startswith(self.repo) else raw # ── facts: exported once per graph, reused while graph.sqlite is unchanged ─────────────────────────────── @@ -948,8 +991,16 @@ class G: EXPORT_VERSION = '4' def export(self): stamp = os.path.join(self.facts, 'stamp'); want = f"{self.db_mtime}:{self.EXPORT_VERSION}" - if os.path.exists(stamp) and open(stamp).read() == want: return + def fresh(): + try: return open(stamp).read() == want + except OSError: return False + if fresh(): return os.makedirs(self.facts, exist_ok=True) + if not export_lock(self.facts, fresh): return # another process just exported these same facts + try: self._export(stamp, want) + finally: export_unlock(self.facts) + + def _export(self, stamp, want): # `generated` is a client declaration too: a Lombok accessor, a record member, a C# auto-property. The engine # synthesises it and RESOLVES the call sites that name it, so filtering the edge out lost every call into a # generated member — which, on a project whose models are all @Data, is most of what touches the models. @@ -997,6 +1048,9 @@ class G: for r in rows: f.write('\t'.join(str(x) for x in r) + '\n') replace_file(tmp, p) def edges(self): + # exported here, not only by the verbs that call export(): `context` read edge.facts straight after `index` + # returned, while the build's warm-up was still writing it in the background, and died on a missing file + self.export() return [tuple(l.rstrip('\n').split('\t')) for l in open(os.path.join(self.facts, 'edge.facts'))] + list(self.EXTRA) def add_outside_call_edges(self): """THE HOPS NO CALL SITE EXPRESSES, as hops of this query (#1469). A gRPC client reaches the handler that serves diff --git a/tests/export_singleflight.py b/tests/export_singleflight.py new file mode 100644 index 000000000..61ad051e8 --- /dev/null +++ b/tests/export_singleflight.py @@ -0,0 +1,83 @@ +#!/usr/bin/env python3 +"""tests/export_singleflight.py — one facts export per graph, however many queries arrive at once. + +The facts a query solves over (out/dl, out/dl/impact) are exported once per graph.sqlite and reused. The export +had no lock: a query that found the stamp stale while another process was already writing the same facts — the +build's background warm-up, a second query, several hooks at once — ran the WHOLE export again beside it. On an +8,619-file repository the first query after `index` cost 193 s against 8.8 s warm, and every concurrent query +paid the same again. Now the first comer takes out/dl/impact/.exporting and the rest wait for its stamp; a lock +whose writer died is taken over. + +Checks, on one indexed fixture: + 1. wait: a lock held by a LIVE process makes a stale --warm wait, and when the holder restores the fresh + stamp and releases, the waiter returns WITHOUT re-exporting (the stamp file is untouched). + 2. takeover: a lock left by a DEAD process does not block — the next --warm removes it and exports. + 3. unlock: a finished export leaves no .exporting behind. + +Indexes one case, so it needs the engine (AXIOMCODE_ENGINE, as tests/run.py). + + python3 tests/export_singleflight.py +""" +import os, shutil, subprocess, sys, tempfile, threading, time + +ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +AX = os.path.join(ROOT, 'plugins', 'axiomcode', 'skills', 'axiomcode', 'scripts', 'axiomcode') +IMPACT = os.path.join(ROOT, 'plugins', 'axiomcode', 'skills', 'axiomcode', 'scripts', 'axiomcode-impact') +CASE = os.path.join(ROOT, 'tests', 'cases', 'java', 'containment-not-from-a-shared-span', 'src') + + +def run(*a, timeout=600): + p = subprocess.run(['bash', AX] + list(a), capture_output=True, text=True, timeout=timeout) + return p.returncode, p.stdout + p.stderr + + +def main(): + fails = [] + def check(ok, why, detail=''): + print(('ok ' if ok else 'FAIL ') + why + ('' if ok else '\n ' + detail.strip().replace('\n', '\n '))) + if not ok: fails.append(why) + + work = tempfile.mkdtemp(prefix='axiomcode-singleflight-') + try: + repo = os.path.join(work, 'repo'); shutil.copytree(CASE, repo) + rc, out = run('index', repo, '--lang', 'java') + if rc: print(f"FAIL index: {out[-300:]}"); return 1 + rc, out = run('impact', '--warm', repo) + check(rc == 0 and 'impact facts ready' in out, 'a first --warm exports the facts', out[-300:]) + D = os.path.join(repo, '.axiomcode', 'out', 'dl', 'impact') + stamp = os.path.join(D, 'stamp'); lock = os.path.join(D, '.exporting') + check(not os.path.exists(lock), 'a finished export leaves no .exporting behind', 'lock file still there') + + # 1. WAIT, DON'T RE-EXPORT: hold the lock as a live process, hide the stamp so the waiter sees stale facts, + # then put the fresh stamp back and release. The waiter must return 0 having written nothing: the stamp's + # mtime is the proof, set well in the past so any rewrite moves it. + held = open(lock, 'w'); held.write(str(os.getpid())); held.flush() + past = time.time() - 3600; os.utime(stamp, (past, past)); before = os.path.getmtime(stamp) + saved = stamp + '.aside'; os.rename(stamp, saved) + + def release(): + time.sleep(2); os.rename(saved, stamp); held.close(); os.remove(lock) + t = threading.Thread(target=release); t.start() + t0 = time.time(); rc, out = run('impact', '--warm', repo, timeout=120); waited = time.time() - t0 + t.join() + check(rc == 0, 'a --warm against a held lock returns 0 once the holder finishes', out[-300:]) + check(waited >= 2, f'it waited for the holder ({waited:.1f}s)', 'returned before the lock was released') + check(os.path.getmtime(stamp) == before, 'and re-exported nothing: the stamp is the one the holder left', + 'stamp mtime moved — the waiter exported over the holder') + + # 2. TAKEOVER: a dead writer's lock does not block. Plant a lock naming a pid that is gone, stale the + # stamp for real, and the next --warm must remove the lock and export. + os.remove(stamp) + open(lock, 'w').write('999999999') + t0 = time.time(); rc, out = run('impact', '--warm', repo, timeout=120) + check(rc == 0 and os.path.exists(stamp), 'a dead writer\'s lock is taken over and the export runs', out[-300:]) + check(not os.path.exists(lock), 'and the taken-over lock is released', 'lock file still there') + finally: + shutil.rmtree(work, ignore_errors=True) + + print(('FAIL: ' + '; '.join(fails)) if fails else 'all export_singleflight checks passed') + return 1 if fails else 0 + + +if __name__ == '__main__': + sys.exit(main()) diff --git a/tests/run.py b/tests/run.py index 863a7c06f..772ba182c 100755 --- a/tests/run.py +++ b/tests/run.py @@ -97,7 +97,9 @@ def print(*a, flush=False, **k): builtins.print(*a, file=buf, **k) if bad: fail += 1; print(f"FAIL {l}/{name}: {ch['why']}", flush=True) for b in bad: print(f" missing/unwanted: {b}") - print(' ' + '\n '.join(text.strip().split('\n')[:14])) + lines = text.strip().split('\n') + # a traceback's last line is the error itself: keep the head (what it answered) and the tail (why it stopped) + print(' ' + '\n '.join(lines[:14] + (['…'] + lines[-12:] if len(lines) > 26 else lines[14:]))) elif verbose: print(f"ok {l}/{name}: {ch['why']}") if not keep: shutil.rmtree(os.path.join(path, '.axiomcode'), ignore_errors=True) return buf.getvalue(), tot, fail, pend