From 2f13a969794e42dfc066992fb26bc2ea08fcc1b1 Mon Sep 17 00:00:00 2001 From: swapnil Date: Thu, 1 Oct 2026 13:46:17 -0700 Subject: [PATCH 01/10] index: look up symbols by id and refs and literals by file and line through an index Co-authored-by: axiomcode-bot[bot] <334110751+axiomcode-bot[bot]@users.noreply.github.com> --- plugins/axiomcode/skills/axiomcode/scripts/axiomcode-index | 3 +++ 1 file changed, 3 insertions(+) diff --git a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-index b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-index index 10a7bfec..3b6fa191 100755 --- a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-index +++ b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-index @@ -699,6 +699,9 @@ CREATE INDEX IF NOT EXISTS symbols_name ON symbols(name); CREATE INDEX IF NOT EX CREATE INDEX IF NOT EXISTS symbols_file ON symbols(file, line); CREATE INDEX IF NOT EXISTS symbols_mid ON symbols(method_id); CREATE INDEX IF NOT EXISTS symbols_tid ON symbols(type_id); CREATE INDEX IF NOT EXISTS refs_name ON refs(name); CREATE INDEX IF NOT EXISTS type_refs_name ON type_refs(name); CREATE INDEX IF NOT EXISTS decorations_owner ON decorations(owner_id); CREATE INDEX IF NOT EXISTS literals_value ON literals(value); CREATE INDEX IF NOT EXISTS comments_file ON comments(file); CREATE INDEX IF NOT EXISTS ce_callee ON call_edges(callee_method_id); CREATE INDEX IF NOT EXISTS ce_caller ON call_edges(caller_id); +-- lookups by id and by file + line: without them each per-site or per-test query scans the whole table +CREATE INDEX IF NOT EXISTS symbols_id ON symbols(id); CREATE INDEX IF NOT EXISTS refs_file_line ON refs(file, line); +CREATE INDEX IF NOT EXISTS literals_file_line ON literals(file, line); -- who calls a method: one row per (site, target), with the engine's tier. caller_id is symbols.id: a -- Java call written in a field initializer has the TYPE as its caller, and that row is kept, not dropped CREATE VIEW callers AS From 3b2a584d89557eb01e086c64528aed06cb919b17 Mon Sep 17 00:00:00 2001 From: swapnil Date: Thu, 1 Oct 2026 15:14:46 -0700 Subject: [PATCH 02/10] build: export impact's facts in the background once the graph is queryable Co-authored-by: axiomcode-bot[bot] <334110751+axiomcode-bot[bot]@users.noreply.github.com> --- .../axiomcode/skills/axiomcode/scripts/axiomcode-build | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-build b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-build index 60068b7b..abd5d098 100755 --- a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-build +++ b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-build @@ -495,12 +495,16 @@ w = max([len('tier')] + [len(t) for t, _ in rows]) print(f"{'tier':<{w}} edges") for t, n in rows: print(f"{t:<{w}} {n}") PY - python3 "$H/axiomcode-impact" --warm "$REPO" || true # export the graph's facts now (~9 s on 1,373 files): the first - # query pays it otherwise, and for the plugin that first query is inside hooks/changes.py's timeout=14, several at once. + # export the graph's facts for impact IN THE BACKGROUND, after the graph is queryable: the first query pays it otherwise + # (and for the plugin that first query is inside hooks/changes.py's timeout=14), but on a large repository the export + # takes minutes and `index` must not wait on it. fd 9 is the build lock: the child closes it, so the lock is released + # when this build ends, and a query or a second build never waits on the warm-up. Its writes are atomic renames. + ( exec 9>&-; nohup python3 "$H/axiomcode-impact" --warm "$REPO" >/dev/null 2>&1 & ) || true + echo "graph ready; precomputing impact facts in the background" # the query rules: their compile started at the top of the build, detached (a caller's timeout cannot kill it half-way, # which cached nothing and repeated on every edit); this only says where it is. A query never waits for it. python3 "$H/dl_program.py" --no-wait || true - [ -n "$BASE_SAVED" ] && AXIOMCODE_GRAPH="$REPO/.axiomcode/base" python3 "$H/axiomcode-impact" --warm "$REPO" >/dev/null 2>&1 || true + [ -n "$BASE_SAVED" ] && ( exec 9>&-; AXIOMCODE_GRAPH="$REPO/.axiomcode/base" nohup python3 "$H/axiomcode-impact" --warm "$REPO" >/dev/null 2>&1 & ) || true [ -z "$OTHER_LANGS" ] || [ ! -s "$PENDING" ] || echo "$LANG_ARG graph ready at $OUT/graph.sqlite; still building: $(cut -d' ' -f1 "$PENDING" | paste -sd, - | sed 's/,/, /g')" } # EVERY OTHER LANGUAGE'S GRAPH, indexed where the engine wrote it and only then moved into place, as the main one is. From f57469283a42c0163482710dbb6a0b592b347530 Mon Sep 17 00:00:00 2001 From: swapnil <78632212+swapnilpaliwal-sd@users.noreply.github.com> Date: Fri, 2 Oct 2026 13:04:22 -0700 Subject: [PATCH 03/10] tests: a failing case prints the tail of its output, where a traceback names the error Co-authored-by: axiomcode-bot[bot] <334110751+axiomcode-bot[bot]@users.noreply.github.com> --- tests/run.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/tests/run.py b/tests/run.py index 863a7c06..772ba182 100755 --- a/tests/run.py +++ b/tests/run.py @@ -97,7 +97,9 @@ def print(*a, flush=False, **k): builtins.print(*a, file=buf, **k) if bad: fail += 1; print(f"FAIL {l}/{name}: {ch['why']}", flush=True) for b in bad: print(f" missing/unwanted: {b}") - print(' ' + '\n '.join(text.strip().split('\n')[:14])) + lines = text.strip().split('\n') + # a traceback's last line is the error itself: keep the head (what it answered) and the tail (why it stopped) + print(' ' + '\n '.join(lines[:14] + (['…'] + lines[-12:] if len(lines) > 26 else lines[14:]))) elif verbose: print(f"ok {l}/{name}: {ch['why']}") if not keep: shutil.rmtree(os.path.join(path, '.axiomcode'), ignore_errors=True) return buf.getvalue(), tot, fail, pend From d65fd561d3e92943c47a325762b7f1e3dd41c749 Mon Sep 17 00:00:00 2001 From: swapnil <78632212+swapnilpaliwal-sd@users.noreply.github.com> Date: Fri, 2 Oct 2026 13:08:32 -0700 Subject: [PATCH 04/10] context: export the graph's facts before reading its edges index now warms impact's facts in the background, so a query can arrive before edge.facts exists. context never exported them and died on a missing file. Co-authored-by: axiomcode-bot[bot] <334110751+axiomcode-bot[bot]@users.noreply.github.com> --- plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path | 3 +++ 1 file changed, 3 insertions(+) diff --git a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path index 99210564..e719552e 100755 --- a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path +++ b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path @@ -997,6 +997,9 @@ class G: for r in rows: f.write('\t'.join(str(x) for x in r) + '\n') replace_file(tmp, p) def edges(self): + # exported here, not only by the verbs that call export(): `context` read edge.facts straight after `index` + # returned, while the build's warm-up was still writing it in the background, and died on a missing file + self.export() return [tuple(l.rstrip('\n').split('\t')) for l in open(os.path.join(self.facts, 'edge.facts'))] + list(self.EXTRA) def add_outside_call_edges(self): """THE HOPS NO CALL SITE EXPRESSES, as hops of this query (#1469). A gRPC client reaches the handler that serves From 82a0ab22fa181b98f75d6a4d41357d3f95870087 Mon Sep 17 00:00:00 2001 From: swapnil <78632212+swapnilpaliwal-sd@users.noreply.github.com> Date: Fri, 2 Oct 2026 16:57:03 -0700 Subject: [PATCH 05/10] impact/path: one facts export per graph, however many queries arrive at once The export had no lock: a query that found the stamp stale while another process was already writing the same facts (the background warm-up after index, a second query, several hooks at once) ran the whole export again beside it. Measured on an 8,619-file Java repository: the first query after index cost 193 s against 8.8 s warm, and every concurrent query paid the same again while thrashing the writer. The first comer now takes /.exporting and the others wait for its stamp, then answer from the facts it wrote; a lock whose writer is gone is taken over. tests/export_singleflight.py pins the wait, the takeover and the unlock. Co-authored-by: axiomcode-bot[bot] <334110751+axiomcode-bot[bot]@users.noreply.github.com> --- .../skills/axiomcode/scripts/axiomcode-impact | 13 ++- .../skills/axiomcode/scripts/axiomcode-path | 47 ++++++++++- tests/export_singleflight.py | 83 +++++++++++++++++++ 3 files changed, 140 insertions(+), 3 deletions(-) create mode 100644 tests/export_singleflight.py diff --git a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-impact b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-impact index e8186697..191cb9f4 100755 --- a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-impact +++ b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-impact @@ -1275,8 +1275,17 @@ class Impact: # a cache written by an older program, or one an interrupted run left half written, breaks every query with # "Cannot open fact file X.facts": accept it only when it holds every relation the program reads need = set(re.findall(r'\.decl (\w+)\([^)]*\)\s+\.input', open(os.path.join(os.path.dirname(os.path.abspath(__file__)), 'dl', 'impact.dl')).read())) - PER_QUERY - if os.path.exists(stamp) and open(stamp).read() == want and all(os.path.exists(os.path.join(D, n + '.facts')) for n in need): return D - os.makedirs(D, exist_ok=True); W = lambda n, rows: g.write(n, rows, D) + def fresh(): + try: return open(stamp).read() == want and all(os.path.exists(os.path.join(D, n + '.facts')) for n in need) + except OSError: return False + if fresh(): return D + os.makedirs(D, exist_ok=True) + if not P.export_lock(D, fresh): return D # another process just exported these same facts + try: return self._export(D, stamp, want) + finally: P.export_unlock(D) + + def _export(self, D, stamp, want): + g = self.g; W = lambda n, rows: g.write(n, rows, D) # owner display -> type id. A display name is NOT unique: two files of one test suite commonly # declare a class of the same name, and keyed on the name alone the second one's members were # filed under the first one's type. Everything scope/2 reaches then crossed between them -- diff --git a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path index e719552e..997e29fe 100755 --- a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path +++ b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path @@ -128,6 +128,43 @@ def replace_file(tmp, dst): shutil.copyfile(tmp, dst); os.remove(tmp) +def export_lock(d, fresh): + """ONE exporter per facts directory. A query that found the stamp stale while another process (the build's + background warm-up, a concurrent query, several hooks at once) was already writing these facts ran the whole + export AGAIN beside it: on an 8,619-file repository the first query after `index` cost 193 s against 8.8 s + once warm, every concurrent query the same again, and the copies thrashed each other. Returns True with the + lock held — the caller exports, then export_unlock in a finally — or False once another process finished the + same export (`fresh` says so) and the caller just reads it. A lock whose writer is gone is taken over.""" + lock = os.path.join(d, '.exporting') + while True: + try: + fd = os.open(lock, os.O_CREAT | os.O_EXCL | os.O_WRONLY) + os.write(fd, str(os.getpid()).encode()); os.close(fd); return True + except FileExistsError: + pass + gone = False + try: + os.kill(int(open(lock).read().strip() or '0'), 0) + except (ProcessLookupError, ValueError): + gone = True + except PermissionError: + pass # alive, another user's process: wait for it + except OSError: # Windows cannot probe another session's pid: fall back to the lock's age + try: gone = time.time() - os.path.getmtime(lock) > 900 + except OSError: continue # the lock vanished under us: contend for it again + if gone: + try: os.remove(lock) + except OSError: pass + continue + time.sleep(0.5) + if fresh(): return False + + +def export_unlock(d): + try: os.remove(os.path.join(d, '.exporting')) + except OSError: pass + + class G: def __init__(self, repo): self.repo = os.path.realpath(repo or '.') @@ -948,8 +985,16 @@ class G: EXPORT_VERSION = '4' def export(self): stamp = os.path.join(self.facts, 'stamp'); want = f"{self.db_mtime}:{self.EXPORT_VERSION}" - if os.path.exists(stamp) and open(stamp).read() == want: return + def fresh(): + try: return open(stamp).read() == want + except OSError: return False + if fresh(): return os.makedirs(self.facts, exist_ok=True) + if not export_lock(self.facts, fresh): return # another process just exported these same facts + try: self._export(stamp, want) + finally: export_unlock(self.facts) + + def _export(self, stamp, want): # `generated` is a client declaration too: a Lombok accessor, a record member, a C# auto-property. The engine # synthesises it and RESOLVES the call sites that name it, so filtering the edge out lost every call into a # generated member — which, on a project whose models are all @Data, is most of what touches the models. diff --git a/tests/export_singleflight.py b/tests/export_singleflight.py new file mode 100644 index 00000000..61ad051e --- /dev/null +++ b/tests/export_singleflight.py @@ -0,0 +1,83 @@ +#!/usr/bin/env python3 +"""tests/export_singleflight.py — one facts export per graph, however many queries arrive at once. + +The facts a query solves over (out/dl, out/dl/impact) are exported once per graph.sqlite and reused. The export +had no lock: a query that found the stamp stale while another process was already writing the same facts — the +build's background warm-up, a second query, several hooks at once — ran the WHOLE export again beside it. On an +8,619-file repository the first query after `index` cost 193 s against 8.8 s warm, and every concurrent query +paid the same again. Now the first comer takes out/dl/impact/.exporting and the rest wait for its stamp; a lock +whose writer died is taken over. + +Checks, on one indexed fixture: + 1. wait: a lock held by a LIVE process makes a stale --warm wait, and when the holder restores the fresh + stamp and releases, the waiter returns WITHOUT re-exporting (the stamp file is untouched). + 2. takeover: a lock left by a DEAD process does not block — the next --warm removes it and exports. + 3. unlock: a finished export leaves no .exporting behind. + +Indexes one case, so it needs the engine (AXIOMCODE_ENGINE, as tests/run.py). + + python3 tests/export_singleflight.py +""" +import os, shutil, subprocess, sys, tempfile, threading, time + +ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +AX = os.path.join(ROOT, 'plugins', 'axiomcode', 'skills', 'axiomcode', 'scripts', 'axiomcode') +IMPACT = os.path.join(ROOT, 'plugins', 'axiomcode', 'skills', 'axiomcode', 'scripts', 'axiomcode-impact') +CASE = os.path.join(ROOT, 'tests', 'cases', 'java', 'containment-not-from-a-shared-span', 'src') + + +def run(*a, timeout=600): + p = subprocess.run(['bash', AX] + list(a), capture_output=True, text=True, timeout=timeout) + return p.returncode, p.stdout + p.stderr + + +def main(): + fails = [] + def check(ok, why, detail=''): + print(('ok ' if ok else 'FAIL ') + why + ('' if ok else '\n ' + detail.strip().replace('\n', '\n '))) + if not ok: fails.append(why) + + work = tempfile.mkdtemp(prefix='axiomcode-singleflight-') + try: + repo = os.path.join(work, 'repo'); shutil.copytree(CASE, repo) + rc, out = run('index', repo, '--lang', 'java') + if rc: print(f"FAIL index: {out[-300:]}"); return 1 + rc, out = run('impact', '--warm', repo) + check(rc == 0 and 'impact facts ready' in out, 'a first --warm exports the facts', out[-300:]) + D = os.path.join(repo, '.axiomcode', 'out', 'dl', 'impact') + stamp = os.path.join(D, 'stamp'); lock = os.path.join(D, '.exporting') + check(not os.path.exists(lock), 'a finished export leaves no .exporting behind', 'lock file still there') + + # 1. WAIT, DON'T RE-EXPORT: hold the lock as a live process, hide the stamp so the waiter sees stale facts, + # then put the fresh stamp back and release. The waiter must return 0 having written nothing: the stamp's + # mtime is the proof, set well in the past so any rewrite moves it. + held = open(lock, 'w'); held.write(str(os.getpid())); held.flush() + past = time.time() - 3600; os.utime(stamp, (past, past)); before = os.path.getmtime(stamp) + saved = stamp + '.aside'; os.rename(stamp, saved) + + def release(): + time.sleep(2); os.rename(saved, stamp); held.close(); os.remove(lock) + t = threading.Thread(target=release); t.start() + t0 = time.time(); rc, out = run('impact', '--warm', repo, timeout=120); waited = time.time() - t0 + t.join() + check(rc == 0, 'a --warm against a held lock returns 0 once the holder finishes', out[-300:]) + check(waited >= 2, f'it waited for the holder ({waited:.1f}s)', 'returned before the lock was released') + check(os.path.getmtime(stamp) == before, 'and re-exported nothing: the stamp is the one the holder left', + 'stamp mtime moved — the waiter exported over the holder') + + # 2. TAKEOVER: a dead writer's lock does not block. Plant a lock naming a pid that is gone, stale the + # stamp for real, and the next --warm must remove the lock and export. + os.remove(stamp) + open(lock, 'w').write('999999999') + t0 = time.time(); rc, out = run('impact', '--warm', repo, timeout=120) + check(rc == 0 and os.path.exists(stamp), 'a dead writer\'s lock is taken over and the export runs', out[-300:]) + check(not os.path.exists(lock), 'and the taken-over lock is released', 'lock file still there') + finally: + shutil.rmtree(work, ignore_errors=True) + + print(('FAIL: ' + '; '.join(fails)) if fails else 'all export_singleflight checks passed') + return 1 if fails else 0 + + +if __name__ == '__main__': + sys.exit(main()) From 1205b15a157fd6e4bb75e43b43d3db3e7365e730 Mon Sep 17 00:00:00 2001 From: swapnil <78632212+swapnilpaliwal-sd@users.noreply.github.com> Date: Fri, 2 Oct 2026 17:02:33 -0700 Subject: [PATCH 06/10] impact: an undeclared name that arrives through an import hints at --library staging MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A name no graph declares, reached only by text, whose lines include an import of it is the signature of an unstaged dependency — until now the answer was indistinguishable from an engine gap, and the fix is one flag away. One hint line under the [text] rows, only for undeclared names. Co-authored-by: axiomcode-bot[bot] <334110751+axiomcode-bot[bot]@users.noreply.github.com> --- plugins/axiomcode/skills/axiomcode/scripts/ax_text.py | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/plugins/axiomcode/skills/axiomcode/scripts/ax_text.py b/plugins/axiomcode/skills/axiomcode/scripts/ax_text.py index ada6242b..3ef55ac3 100644 --- a/plugins/axiomcode/skills/axiomcode/scripts/ax_text.py +++ b/plugins/axiomcode/skills/axiomcode/scripts/ax_text.py @@ -644,6 +644,13 @@ def block(repo, asked, scope=None, why='unresolved', rows=ROWS): if rest > 0: lines.append(f" … +{rest} more line(s): git grep -n{'w' if w else ''} -F -e '{n}'" if not pat else f" … +{rest} more line(s): git grep -nP -e '{pat.pattern}'") + # a name no graph declares that ARRIVES THROUGH AN IMPORT is the signature of an unstaged dependency, the + # commonest setup gap there is: without the hint this answer is indistinguishable from an engine gap, and the + # fix (stage the dependency's source) is one flag away. Only for undeclared names: a string or a resolved + # target wants no staging advice. + if why == 'undeclared' and any(re.match(r'\s*(import\b|using\b|from\s)', t.strip()) and n in t for _f, _l, t in hits): + lines.append(f" this name arrives through an import, so it is likely declared in a dependency: calls through " + f"it resolve once that dependency's source is staged — `axiomcode index --library `") if not lines: return '' if approxed: lines.append("next: [approx] rows are text placed in the declaration that holds it, not resolved edges: read the " "evidence line, then `impact ` for what a change reaches") From 6dde0de526b3ce290b15f5e83c2ffe643fe9db97 Mon Sep 17 00:00:00 2001 From: swapnil <78632212+swapnilpaliwal-sd@users.noreply.github.com> Date: Fri, 2 Oct 2026 22:50:06 -0700 Subject: [PATCH 07/10] =?UTF-8?q?impact/path:=20the=20facts=20export=20sto?= =?UTF-8?q?ps=20paying=20per-row=20SQL=20=E2=80=94=20171=20s=20to=2055=20s?= =?UTF-8?q?=20on=208,619=20files?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three wastes, found by timing each exported relation and profiling the poles standalone: - site_file() ran one has() and one paths lookup PER CALL, and the export calls it for every call edge: 638k round-trips over 319k edges. The paths map is now read once per graph (171 s -> 89 s). - via_base_rows probes field_access per single-target call site, and the table had no caller_id index: a 65,797-row scan 5,162 times. fa_caller lands with the other build-time indexes (89 s -> 63 s). - registrations() was computed twice in one export, once for the registration relation and again inside reg_key_fact (63 s -> 55 s). Every fact file is content-identical before and after (two differ in row order only, written from unsorted sets before this change too). Tests: java 329/329, python 287/287, facts_cache, latency, freshness, export_singleflight, front_door. Co-authored-by: axiomcode-bot[bot] <334110751+axiomcode-bot[bot]@users.noreply.github.com> --- .../skills/axiomcode/scripts/axiomcode-impact | 5 +++-- .../skills/axiomcode/scripts/axiomcode-index | 3 +++ .../skills/axiomcode/scripts/axiomcode-path | 14 ++++++++++---- 3 files changed, 16 insertions(+), 6 deletions(-) diff --git a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-impact b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-impact index 191cb9f4..609a53b7 100755 --- a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-impact +++ b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-impact @@ -1517,7 +1517,8 @@ class Impact: L = self.code(r['file']); text = L[r['line'] - 1] if r['line'] <= len(L) else '' for qn in re.findall(rf'([A-Za-z_$][\w$]*)\s*\.\s*{re.escape(r["name"])}\b', text): quals.append((r['file'], r['line'], r['name'], qn)) W('ref', refs); W('qualifier', sorted(set(quals))) - W('registration', ax_registration.registrations(g.q, g.site_file)) + regs = ax_registration.registrations(g.q, g.site_file) # computed once: reg_key_fact below reads it too + W('registration', regs) trs = []; trf = set() # a reference on a module-level type alias's own lines belongs to the alias, not the module (#784) import graph_sql @@ -1599,7 +1600,7 @@ class Impact: regk = [] # `rd` and not `decl`: `decl` is the dec_literal list above for rd, _f, _l, kind, key, _why in (ax_registration.decoration_keys(g.q, g.site_file) + ax_registration.value_route_registrations(g.q, g.site_file) - + ax_registration.registrations(g.q, g.site_file) + tregs): + + regs + tregs): if rd in g.sym and key: regk.append((rd, kind, key)) W('reg_key_fact', sorted(set(regk))) # the callables in test files: a test that publishes a handler table's key drives the handler, and is not diff --git a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-index b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-index index 3b6fa191..19c50a44 100755 --- a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-index +++ b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-index @@ -702,6 +702,9 @@ CREATE INDEX IF NOT EXISTS ce_callee ON call_edges(callee_method_id); CREATE IND -- lookups by id and by file + line: without them each per-site or per-test query scans the whole table CREATE INDEX IF NOT EXISTS symbols_id ON symbols(id); CREATE INDEX IF NOT EXISTS refs_file_line ON refs(file, line); CREATE INDEX IF NOT EXISTS literals_file_line ON literals(file, line); +-- the fields a caller reads: via_base_rows asks per single-target call site, and without the index each ask +-- scanned the whole table (65,797 rows × 5,162 sites on an 8,619-file repository = 24 s of a 171 s warm → 1.4 s) +CREATE INDEX IF NOT EXISTS fa_caller ON field_access(caller_id); -- who calls a method: one row per (site, target), with the engine's tier. caller_id is symbols.id: a -- Java call written in a field initializer has the TYPE as its caller, and that row is kept, not dropped CREATE VIEW callers AS diff --git a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path index 997e29fe..55b02956 100755 --- a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path +++ b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path @@ -969,10 +969,16 @@ class G: unfoll = [(r['lib'], r['n']) for r in rows if r['from_site']][:cap] return named, unfoll def site_file(self, raw): - """a call site's file as the repo sees it: Java bundles store absolute paths; the index's paths table maps them""" - if self.has('paths'): - r = self.q("SELECT rel FROM paths WHERE raw = ?", raw) - if r: return r[0]['rel'] + """a call site's file as the repo sees it: Java bundles store absolute paths; the index's paths table maps them. + The map is read ONCE PER GRAPH, not once per row: this ran a query per call (one `has`, one lookup), and the + facts export calls it for every call edge — 319k edges on an 8,619-file repository made 638k round-trips + through sqlite, which was most of the export's time (reg_key_fact 36 s, calls 34 s, registration 31 s, + via_base 25 s of a 171 s warm).""" + m = getattr(self, '_site_files', None) + if m is None: + m = self._site_files = {r['raw']: r['rel'] for r in self.q("SELECT raw, rel FROM paths")} if self.has('paths') else {} + got = m.get(raw) + if got is not None: return got return os.path.relpath(raw, self.repo).replace(os.sep, '/') if os.path.isabs(raw) and raw.startswith(self.repo) else raw # ── facts: exported once per graph, reused while graph.sqlite is unchanged ─────────────────────────────── From 966ed8554045806f60b8289452604108a9250651 Mon Sep 17 00:00:00 2001 From: swapnil <78632212+swapnilpaliwal-sd@users.noreply.github.com> Date: Sat, 3 Oct 2026 16:44:17 -0700 Subject: [PATCH 08/10] perf(python): read the parsed tree through a one-pass plain-JS mirror The Python stages each re-walked the tree-sitter tree through the JS<->C++ boundary, so a node's properties were marshalled once per stage; the parse stage of a large repository spent over half its time in those getters while the parse itself was negligible. materializePyTree now mirrors the tree in one cursor pass and every stage reads plain JS properties. Tree-sitter still parses every file. Two lookups in linkProject ran as linear scans inside loops (modules by hash per import record, bindings by hash per alias) and are now prebuilt maps with the same first-match semantics. Python analysis on two large corpus subjects drops 3.2x and 2.6x with byte-identical IR; the mirror is also A/B-asserted against the real tree node by node, which surfaced the one subtlety: an ERROR node absorbed during recovery is extra, so ERROR nodes read that flag from the real node. Parser version to 0.1.3 for the behaviour-neutral rebuild. Co-authored-by: axiomcode-bot[bot] <334110751+axiomcode-bot[bot]@users.noreply.github.com> --- parser/package.json | 4 +- .../extractors/python-resolution-linker.ts | 18 +- .../extractors/python-scope-extractor.ts | 6 +- parser/src/parsers/python/py-mirror-tree.ts | 205 ++++++++++++++++++ 4 files changed, 228 insertions(+), 5 deletions(-) create mode 100644 parser/src/parsers/python/py-mirror-tree.ts diff --git a/parser/package.json b/parser/package.json index eac08e04..175cb019 100644 --- a/parser/package.json +++ b/parser/package.json @@ -1,7 +1,7 @@ { "name": "@axiomcode/parser", - "version": "0.1.2", - "description": "AxiomCode Parser — compiles source code and build configuration into a relational intermediate representation.", + "version": "0.1.3", + "description": "AxiomCode Parser \u2014 compiles source code and build configuration into a relational intermediate representation.", "main": "dist/extract.js", "types": "dist/extract.d.ts", "scripts": { diff --git a/parser/src/parsers/python/extractors/python-resolution-linker.ts b/parser/src/parsers/python/extractors/python-resolution-linker.ts index 3361359a..798ac85b 100644 --- a/parser/src/parsers/python/extractors/python-resolution-linker.ts +++ b/parser/src/parsers/python/extractors/python-resolution-linker.ts @@ -349,6 +349,14 @@ export class PythonResolutionLinker { // ---- step 2: bases, now that imports are resolved // Per-module views of what each module's imports brought into scope. + // Modules keyed by hash once, FIRST occurrence kept — the lookup below ran as a + // linear scan per import record, which is quadratic over the project. + const moduleByHash = new Map(); + for (const module of modules) { + if (!moduleByHash.has(module.moduleHash)) { + moduleByHash.set(module.moduleHash, module); + } + } const importedTypeByName = new Map>(); const importedModuleByName = new Map>(); for (const module of modules) { @@ -365,7 +373,7 @@ export class PythonResolutionLinker { // The bound name refers to a module. Find which one by matching the // resolved module hash, so `from . import protocols` and // `import pkg.protocols` are handled by the same lookup. - const target = modules.find(m => m.moduleHash === record.getResolvedModuleLinkHash()); + const target = moduleByHash.get(record.getResolvedModuleLinkHash()); if (target) { mods.set(record.getSimpleName(), target); } @@ -383,8 +391,14 @@ export class PythonResolutionLinker { const aliasByModule = new Map>(); for (const module of modules) { const scoped = new Map(); + // Bindings keyed by hash, FIRST occurrence kept, matching the linear + // `.find` this replaces; built in the pass that already walks them. + const bindingByHash = new Map(); for (const binding of module.bindings) { scoped.set(`${binding.getPyScopeLinkHash()}::${binding.getName()}`, binding); + if (!bindingByHash.has(binding.getHash())) { + bindingByHash.set(binding.getHash(), binding); + } } const parents = new Map(); for (const scope of module.scopes) { @@ -411,7 +425,7 @@ export class PythonResolutionLinker { }); const byName = new Map(); for (const [bindingHash, entity] of aliases) { - const binding = module.bindings.find(b => b.getHash() === bindingHash); + const binding = bindingByHash.get(bindingHash); if (binding) { byName.set(binding.getName(), entity); } diff --git a/parser/src/parsers/python/extractors/python-scope-extractor.ts b/parser/src/parsers/python/extractors/python-scope-extractor.ts index cd379ac5..8d8abc5f 100644 --- a/parser/src/parsers/python/extractors/python-scope-extractor.ts +++ b/parser/src/parsers/python/extractors/python-scope-extractor.ts @@ -30,6 +30,7 @@ import { SymbolFlags, SymbolScope, } from '@/parsers/python/extractors/python-symbol-table'; +import { materializePyTree } from '@/parsers/python/py-mirror-tree'; import { Python2Finding, SymbolBlock } from '@/parsers/python/types'; import { PythonSourcePositions } from '@/utils/python'; @@ -144,7 +145,10 @@ export class PythonScopeExtractor { } const tree = this.parser.parse(input.sourceCode); - const rootNode = this.parser.getRootNode(tree); + // One cursor pass mirrors the tree into plain JS; every stage after this + // line reads JS properties instead of re-crossing the tree-sitter FFI. + // Tree-sitter itself still parses every file — see py-mirror-tree.ts. + const rootNode = materializePyTree(tree, input.sourceCode) as unknown as Parser.SyntaxNode; const detection = this.detector.detect(rootNode, input.sourceCode); if (detection.dialect !== PythonDialect.PY3) { diff --git a/parser/src/parsers/python/py-mirror-tree.ts b/parser/src/parsers/python/py-mirror-tree.ts new file mode 100644 index 00000000..1525e6f9 --- /dev/null +++ b/parser/src/parsers/python/py-mirror-tree.ts @@ -0,0 +1,205 @@ +/** + * A plain-JS mirror of a tree-sitter tree, built in ONE cursor pass. + * + * Tree-sitter still parses every file; what this removes is the reading cost. + * Every property access on a tree-sitter SyntaxNode crosses the JS↔C++ + * boundary and re-marshals the node handle, and the Python extraction stages + * each walk the same tree, so one node's properties are fetched once per + * stage. The mirror pays the boundary once per node, during the cursor walk, + * and every later read is a JS property. + * + * The surface is exactly what the Python stages use (verified by grep over + * extractors, detector and soft-keywords): type, text, children/namedChildren, + * child(i)/namedChild(i), childForFieldName, counts, spans, parent, id, + * isNamed/isMissing/isExtra/hasError. Anything outside it throws at the call + * site rather than answering wrongly. + * + * `text` is sliced lazily from the one source string, so the mirror holds no + * copies. `isExtra` is derived from the node type: tree-sitter-python's extras + * are exactly `comment` and `line_continuation` (whitespace produces no node). + * `hasError` is computed bottom-up with tree-sitter's own meaning: an ERROR or + * missing node anywhere in the subtree. + */ +import type Parser from 'tree-sitter'; + +/** tree-sitter-python `extras`: the only node types that parse as extra. */ +const PY_EXTRA_TYPES = new Set(['comment', 'line_continuation']); + +/** Never reset: `HashByNodeId` maps must not collide across files. */ +let nextId = 1; + +export class PyMirrorNode { + readonly id: number; + readonly type: string; + readonly isNamed: boolean; + readonly isMissing: boolean; + readonly startIndex: number; + readonly endIndex: number; + readonly startPosition: Parser.Point; + readonly endPosition: Parser.Point; + parent: PyMirrorNode | null = null; + readonly children: PyMirrorNode[] = []; + namedChildren: PyMirrorNode[] = []; + /** First child per field name — the pick childForFieldName makes. */ + private fields: Map | null = null; + private errorInSubtree = false; + private readonly source: string; + + constructor(cursor: Parser.TreeCursor, source: string) { + this.id = nextId++; + this.type = cursor.nodeType; + // An ERROR node can be either: tree-sitter marks an ERROR it absorbed + // during recovery as EXTRA (siblings' named counts then skip it), while a + // plain ERROR is not. The type cannot tell them apart, so this is the one + // place the real node is consulted — ERROR nodes exist only in files that + // failed to parse, so the boundary crossing stays off the healthy path. + if (this.type === 'ERROR') { + this._extraOverride = cursor.currentNode.isExtra; + } + this.isNamed = cursor.nodeIsNamed; + this.isMissing = cursor.nodeIsMissing; + this.startIndex = cursor.startIndex; + this.endIndex = cursor.endIndex; + this.startPosition = cursor.startPosition; + this.endPosition = cursor.endPosition; + this.source = source; + } + + get text(): string { + return this.source.slice(this.startIndex, this.endIndex); + } + + /** Set at build time only for ERROR nodes — see materializePyTree. */ + _extraOverride: boolean | null = null; + + get isExtra(): boolean { + if (this._extraOverride !== null) { + return this._extraOverride; + } + return PY_EXTRA_TYPES.has(this.type); + } + + get hasError(): boolean { + return this.errorInSubtree; + } + + get childCount(): number { + return this.children.length; + } + + get namedChildCount(): number { + return this.namedChildren.length; + } + + child(index: number): PyMirrorNode | null { + return this.children[index] ?? null; + } + + namedChild(index: number): PyMirrorNode | null { + return this.namedChildren[index] ?? null; + } + + /** + * First IMMEDIATE child carrying the field, which is what every extractor + * asks for. Tree-sitter's own lookup additionally pierces one visible level + * on `match_statement` (its `alternative` case clauses sit inside the match + * `block`), a quirk nothing in the Python stages uses: the match consumers + * iterate the block's namedChildren by type instead (block-extractor, + * expression-extractor), and `alternative` is read only on if/for/while, + * where it is an immediate child. + */ + childForFieldName(fieldName: string): PyMirrorNode | null { + return this.fields?.get(fieldName) ?? null; + } + + /** @internal build-time wiring, called only by materializePyTree. */ + _addChild(child: PyMirrorNode, fieldName: string | null): void { + child.parent = this; + this.children.push(child); + if (child.isNamed) { + this.namedChildren.push(child); + } + if (fieldName !== null && fieldName !== '') { + if (this.fields === null) { + this.fields = new Map(); + } + if (!this.fields.has(fieldName)) { + this.fields.set(fieldName, child); + } + } + } + + /** @internal */ + _markError(): void { + this.errorInSubtree = true; + } +} + +/** + * One depth-first cursor pass over the freshly parsed tree. + * + * The result is handed to the stages as a `Parser.SyntaxNode`: the stages are + * typed against tree-sitter's interface and use only the mirrored subset, so + * the cast is confined to the one call site that builds the mirror. + */ +export function materializePyTree(tree: Parser.Tree, source: string): PyMirrorNode { + const cursor = tree.walk(); + const root = new PyMirrorNode(cursor, source); + // hasError at the ROOT is read from tree-sitter itself (one boundary call per + // file): an error can live in a HIDDEN node — a file whose syntax error is + // swallowed shows no visible ERROR/missing child anywhere, yet + // ts_node_has_error is true, and the module row's grammar column + // (TS_PYTHON3_PARTIAL) depends on exactly that. The bottom-up propagation + // below still covers every VISIBLE error for the deeper nodes. + if (tree.rootNode.hasError) { + root._markError(); + } + const stack: PyMirrorNode[] = [root]; + let current = root; + + // gotoFirstChild / gotoNextSibling / gotoParent, no recursion: a deeply + // nested file must not overflow the JS stack when the C parser handled it. + let descending = true; + for (;;) { + if (descending && cursor.gotoFirstChild()) { + const child = new PyMirrorNode(cursor, source); + current._addChild(child, cursor.currentFieldName); + stack.push(child); + current = child; + continue; + } + // finishing `current`: fold its error state into the parent + if ( + current.type === 'ERROR' || + current.isMissing || + current.hasError + ) { + const parent = stack[stack.length - 2]; + if (parent !== undefined) { + parent._markError(); + } + current._markError(); + } + if (cursor.gotoNextSibling()) { + stack.pop(); + const parent = stack[stack.length - 1]; + if (parent === undefined) { + // the root has no siblings; the cursor cannot get here + return root; + } + const sibling = new PyMirrorNode(cursor, source); + parent._addChild(sibling, cursor.currentFieldName); + stack.push(sibling); + current = sibling; + descending = true; + continue; + } + stack.pop(); + const above = stack[stack.length - 1]; + if (!cursor.gotoParent() || above === undefined) { + return root; + } + current = above; + descending = false; + } +} From f4242c3a1767457857d22d7efa0fba6c85863672 Mon Sep 17 00:00:00 2001 From: swapnil <78632212+swapnilpaliwal-sd@users.noreply.github.com> Date: Sat, 3 Oct 2026 17:17:59 -0700 Subject: [PATCH 09/10] perf(java,csharp): read parsed trees through the shared one-pass mirror MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The mirror moves to parsers/mirror-tree.ts, parameterized by each grammar's extras set; the Python module now only pins its own. Java and C# route their getRootNode through it (the real root remains available without a source string). JavaScript and TypeScript are untouched: their extractors run on the compiler AST, not tree-sitter. The port surfaced four tree-sitter subtleties the mirror now reproduces: named-sibling getters answer from anonymous nodes too; fieldNameForChild labels an extra on a field position while childForFieldName skips extras (field layout is re-read from the real node wherever an extra child makes the cursor's reporting untrustworthy); a file whose root carries an error reads every node's hasError from the real node, because a bare directive can report an error on itself with no visible ERROR child; and the C# mirror slices text from the BOM-stripped string parse() actually parsed. One deliberate behaviour change: Java's nested-annotation arguments were linked through a Map keyed by node OBJECT, and tree-sitter hands out a fresh wrapper per access, so the join hit only when a wrapper happened to be reused — 159 of 267 nested-annotation arguments linked on a large corpus subject. The mirror's stable identities link 260 of 267; the rest of the IR is byte-identical there, as it is on both Python subjects and the two C# subjects. Parse-stage wall time on large subjects: Java 66s -> 13s, C# 73s -> 18s. Co-authored-by: axiomcode-bot[bot] <334110751+axiomcode-bot[bot]@users.noreply.github.com> --- parser/src/parsers/csharp/csharp-parser.ts | 26 +- .../csharp/extractors/cs-fact-extractor.ts | 4 +- .../java/extractors/import-extractor.ts | 2 +- .../extractors/type-registry-extractor.ts | 2 +- parser/src/parsers/java/java-parser.ts | 18 +- parser/src/parsers/mirror-tree.ts | 307 ++++++++++++++++++ parser/src/parsers/python/py-mirror-tree.ts | 206 +----------- 7 files changed, 361 insertions(+), 204 deletions(-) create mode 100644 parser/src/parsers/mirror-tree.ts diff --git a/parser/src/parsers/csharp/csharp-parser.ts b/parser/src/parsers/csharp/csharp-parser.ts index f37143af..2e1a4306 100644 --- a/parser/src/parsers/csharp/csharp-parser.ts +++ b/parser/src/parsers/csharp/csharp-parser.ts @@ -6,6 +6,7 @@ import Parser from 'tree-sitter'; import CSharp from 'tree-sitter-c-sharp'; import { FILE_EXTENSIONS } from '@/constants/consts'; +import { materializeTree } from '@/parsers/mirror-tree'; import { CSHARP_CALLBACK_PARSE_THRESHOLD, CSHARP_PARSE_CHUNK_SIZE, @@ -81,6 +82,20 @@ export function stripUtf8Bom(sourceCode: string): string { return sourceCode.startsWith(UTF8_BOM) ? sourceCode.slice(UTF8_BOM.length) : sourceCode; } +/** tree-sitter-c-sharp `extras`: the only node types that parse as extra. */ +const CS_EXTRA_TYPES: ReadonlySet = new Set([ + 'comment', + 'preproc_region', + 'preproc_endregion', + 'preproc_line', + 'preproc_pragma', + 'preproc_nullable', + 'preproc_error', + 'preproc_warning', + 'preproc_define', + 'preproc_undef', +]); + export class CSharpParser implements LanguageParser { readonly language = ProjectLanguage.CSHARP; readonly fileExtension = FILE_EXTENSIONS.CSHARP; @@ -142,7 +157,16 @@ export class CSharpParser implements LanguageParser { return parseWithRetry(source); } - getRootNode(tree: Parser.Tree): Parser.SyntaxNode { + getRootNode(tree: Parser.Tree, sourceCode?: string): Parser.SyntaxNode { + // With `sourceCode`, the root is a one-pass plain-JS mirror of the tree + // (see `../mirror-tree`): every stage after it reads JS properties + // instead of re-crossing the tree-sitter FFI per property access. + // Tree-sitter still parses every file. + if (sourceCode !== undefined) { + // parse() strips a leading BOM, so the mirror slices text from the SAME + // string the tree's byte offsets are relative to, whatever was passed. + return materializeTree(tree, stripUtf8Bom(sourceCode), CS_EXTRA_TYPES) as unknown as Parser.SyntaxNode; + } return tree.rootNode; } diff --git a/parser/src/parsers/csharp/extractors/cs-fact-extractor.ts b/parser/src/parsers/csharp/extractors/cs-fact-extractor.ts index 31fc4fe4..aea82ba5 100644 --- a/parser/src/parsers/csharp/extractors/cs-fact-extractor.ts +++ b/parser/src/parsers/csharp/extractors/cs-fact-extractor.ts @@ -216,11 +216,11 @@ export class CsFactExtractor { // is. Length-preserving to the character; the receiver the blanking // removes is returned in a side table. See cs-extension-block.ts. const flattened = flattenExtensionBlocks(rewritten, (source) => - this.parser.getRootNode(this.parser.parse(source)) + this.parser.getRootNode(this.parser.parse(source), source) ); const parseText = flattened.text.endsWith('\n') ? flattened.text : `${flattened.text}\n`; const tree = this.parser.parse(parseText); - const root = this.parser.getRootNode(tree); + const root = this.parser.getRootNode(tree, parseText); // The symbol set this emission is compiled under: what the caller supplied, // plus the implicit framework symbols the SDK injects and no `.csproj` diff --git a/parser/src/parsers/java/extractors/import-extractor.ts b/parser/src/parsers/java/extractors/import-extractor.ts index d6d98f02..6323cb42 100644 --- a/parser/src/parsers/java/extractors/import-extractor.ts +++ b/parser/src/parsers/java/extractors/import-extractor.ts @@ -54,7 +54,7 @@ export class ImportExtractor implements BaseExtractor { } const tree = this.javaParser.parse(fileContent); - const rootNode = this.javaParser.getRootNode(tree); + const rootNode = this.javaParser.getRootNode(tree, fileContent); this.extractImportsFromRoot(rootNode, filePath, serviceVersionHash, imports); } catch (error) { diff --git a/parser/src/parsers/java/extractors/type-registry-extractor.ts b/parser/src/parsers/java/extractors/type-registry-extractor.ts index 993630bf..680ae654 100644 --- a/parser/src/parsers/java/extractors/type-registry-extractor.ts +++ b/parser/src/parsers/java/extractors/type-registry-extractor.ts @@ -203,7 +203,7 @@ export class TypeRegistryExtractor implements BaseExtractor { } const tree = this.javaParser.parse(fileContent); - const rootNode = this.javaParser.getRootNode(tree); + const rootNode = this.javaParser.getRootNode(tree, fileContent); const basePath = this.extractBasePath(filePath); const fileName = path.basename(filePath); diff --git a/parser/src/parsers/java/java-parser.ts b/parser/src/parsers/java/java-parser.ts index 1ec23cc5..e444c02f 100644 --- a/parser/src/parsers/java/java-parser.ts +++ b/parser/src/parsers/java/java-parser.ts @@ -3,9 +3,13 @@ import Java from 'tree-sitter-java'; import { FILE_EXTENSIONS } from '@/constants/consts'; import { LanguageParser } from '@/parsers/language-parser'; +import { materializeTree } from '@/parsers/mirror-tree'; import { ProjectLanguage } from '@/types/ProjectInfo'; import { withRetry } from '@/utils/retry-decorator'; +/** tree-sitter-java `extras`: the only node types that parse as extra. */ +const JAVA_EXTRA_TYPES: ReadonlySet = new Set(['line_comment', 'block_comment']); + /** * Java-specific tree-sitter parser implementation */ @@ -68,11 +72,21 @@ export class JavaParser implements LanguageParser { } /** - * Gets the root node of a parsed tree + * Gets the root node of a parsed tree. + * + * With `sourceCode`, the root is a one-pass plain-JS mirror of the tree + * (see `../mirror-tree`): every stage after it reads JS properties instead + * of re-crossing the tree-sitter FFI per property access. Tree-sitter still + * parses every file. Without it, the real tree-sitter root is returned. + * * @param tree Parsed syntax tree + * @param sourceCode The exact string the tree was parsed from * @returns Root syntax node */ - getRootNode(tree: Parser.Tree): Parser.SyntaxNode { + getRootNode(tree: Parser.Tree, sourceCode?: string): Parser.SyntaxNode { + if (sourceCode !== undefined) { + return materializeTree(tree, sourceCode, JAVA_EXTRA_TYPES) as unknown as Parser.SyntaxNode; + } return tree.rootNode; } diff --git a/parser/src/parsers/mirror-tree.ts b/parser/src/parsers/mirror-tree.ts new file mode 100644 index 00000000..ad88b17f --- /dev/null +++ b/parser/src/parsers/mirror-tree.ts @@ -0,0 +1,307 @@ +/** + * A plain-JS mirror of a tree-sitter tree, built in ONE cursor pass — the + * shared machinery behind each language's `materializeTree`. + * + * Tree-sitter still parses every file; what this removes is the reading cost. + * Every property access on a tree-sitter SyntaxNode crosses the JS↔C++ + * boundary and re-marshals the node handle, and the extraction stages each + * walk the same tree, so one node's properties are fetched once per stage. + * The mirror pays the boundary once per node, during the cursor walk, and + * every later read is a JS property. + * + * The surface is the union of what the tree-sitter-reading stages use + * (verified by grep per language): type, text, children/namedChildren, + * child(i)/namedChild(i), childForFieldName, fieldNameForChild, counts, + * spans, parent, named siblings, id, isNamed/isMissing/isExtra/hasError. + * Anything outside it throws at the call site rather than answering wrongly. + * + * `text` is sliced lazily from the one source string, so the mirror holds no + * copies. `isExtra` is derived from the node type against the language's own + * `extras` set, with one exception read from the real node (see constructor). + * `hasError` is computed bottom-up with tree-sitter's meaning — an ERROR or + * missing node anywhere in the subtree — and the ROOT's flag is copied from + * tree-sitter itself, because an error can live in a HIDDEN node that no + * visible child betrays. + */ +import type Parser from 'tree-sitter'; + +/** Never reset: `HashByNodeId` maps must not collide across files. */ +let nextId = 1; + +export class MirrorNode { + readonly id: number; + readonly type: string; + readonly isNamed: boolean; + readonly isMissing: boolean; + readonly startIndex: number; + readonly endIndex: number; + readonly startPosition: Parser.Point; + readonly endPosition: Parser.Point; + parent: MirrorNode | null = null; + readonly children: MirrorNode[] = []; + namedChildren: MirrorNode[] = []; + /** First child per field name — the pick childForFieldName makes. */ + private fields: Map | null = null; + /** Field name per child index, for fieldNameForChild; null when none has one. */ + private childFields: (string | null)[] | null = null; + /** Index within parent.children, for the named-sibling getters. */ + private childIndex = -1; + private errorInSubtree = false; + /** Set at build time only for ERROR nodes — see the constructor. */ + private extraOverride: boolean | null = null; + private readonly source: string; + private readonly extraTypes: ReadonlySet; + + constructor(cursor: Parser.TreeCursor, source: string, extraTypes: ReadonlySet) { + this.id = nextId++; + this.type = cursor.nodeType; + // An ERROR node can be either: tree-sitter marks an ERROR it absorbed + // during recovery as EXTRA (siblings' named counts then skip it), while a + // plain ERROR is not. The type cannot tell them apart, so this is the one + // place the real node is consulted — ERROR nodes exist only in files that + // failed to parse, so the boundary crossing stays off the healthy path. + if (this.type === 'ERROR') { + this.extraOverride = cursor.currentNode.isExtra; + } + this.isNamed = cursor.nodeIsNamed; + this.isMissing = cursor.nodeIsMissing; + this.startIndex = cursor.startIndex; + this.endIndex = cursor.endIndex; + this.startPosition = cursor.startPosition; + this.endPosition = cursor.endPosition; + this.source = source; + this.extraTypes = extraTypes; + } + + get text(): string { + return this.source.slice(this.startIndex, this.endIndex); + } + + get isExtra(): boolean { + if (this.extraOverride !== null) { + return this.extraOverride; + } + return this.extraTypes.has(this.type); + } + + get hasError(): boolean { + return this.errorInSubtree; + } + + get childCount(): number { + return this.children.length; + } + + get namedChildCount(): number { + return this.namedChildren.length; + } + + child(index: number): MirrorNode | null { + return this.children[index] ?? null; + } + + namedChild(index: number): MirrorNode | null { + return this.namedChildren[index] ?? null; + } + + // The named-sibling getters answer from ANY node, anonymous ones included + // (tree-sitter scans the parent's children positionally), so they scan from + // this node's position rather than indexing namedChildren. They are read a + // handful of times per file (comment attachment), never in a hot loop. + get previousNamedSibling(): MirrorNode | null { + if (this.parent === null) { + return null; + } + for (let i = this.childIndex - 1; i >= 0; i--) { + const sibling = this.parent.children[i]; + if (sibling !== undefined && sibling.isNamed) { + return sibling; + } + } + return null; + } + + get nextNamedSibling(): MirrorNode | null { + if (this.parent === null) { + return null; + } + for (let i = this.childIndex + 1; i < this.parent.children.length; i++) { + const sibling = this.parent.children[i]; + if (sibling !== undefined && sibling.isNamed) { + return sibling; + } + } + return null; + } + + /** + * First IMMEDIATE child carrying the field, which is what every extractor + * asks for. Tree-sitter's own lookup can additionally pierce one visible + * level where a grammar attaches a field inside a hidden rule (Python's + * `match_statement` reaches its case clauses' `alternative` through the + * match block), a quirk no stage uses: the consumers iterate those children + * by type instead. + */ + childForFieldName(fieldName: string): MirrorNode | null { + return this.fields?.get(fieldName) ?? null; + } + + fieldNameForChild(index: number): string | null { + return this.childFields?.[index] ?? null; + } + + /** @internal set when any direct child is an extra — see _repairFieldsFrom. */ + _needsFieldRepair = false; + + /** @internal build-time wiring, called only by materializeTree. */ + _addChild(child: MirrorNode, fieldName: string | null): void { + child.parent = this; + child.childIndex = this.children.length; + this.children.push(child); + if (child.isNamed) { + this.namedChildren.push(child); + } + if (child.isExtra) { + this._needsFieldRepair = true; + } + if (fieldName !== null && fieldName !== '' && fieldName !== undefined) { + if (this.fields === null) { + this.fields = new Map(); + } + if (!this.fields.has(fieldName)) { + this.fields.set(fieldName, child); + } + if (this.childFields === null) { + this.childFields = []; + } + this.childFields[this.children.length - 1] = fieldName; + } + } + + /** + * @internal Re-reads this node's field layout from the real node. + * + * Around an EXTRA child (a comment inside the construct) the cursor's field + * reporting diverges from the node API in two ways: on a chunk-parsed file + * (over tree-sitter's string-length ceiling) the field of the sibling after + * the extra can come back empty, and the node API itself labels the extra + * with the preceding field. A node with an extra child therefore copies the + * layout wholesale — these are only the comment-bearing nodes, so the + * boundary crossings stay rare. + */ + _repairFieldsFrom(real: Parser.SyntaxNode): void { + this.fields = null; + this.childFields = null; + for (let i = 0; i < this.children.length; i++) { + const fieldName = real.fieldNameForChild(i) ?? null; + if (fieldName === null || fieldName === '') { + continue; + } + const child = this.children[i]; + if (child === undefined) { + continue; + } + // The two node APIs disagree around extras, and the mirror keeps both + // behaviours: fieldNameForChild labels an extra sitting on a field + // position, while childForFieldName SKIPS extras and answers the first + // non-extra carrier. + if (!child.isExtra) { + if (this.fields === null) { + this.fields = new Map(); + } + if (!this.fields.has(fieldName)) { + this.fields.set(fieldName, child); + } + } + if (this.childFields === null) { + this.childFields = []; + } + this.childFields[i] = fieldName; + } + } + + /** @internal */ + _markError(): void { + this.errorInSubtree = true; + } +} + +/** One depth-first cursor pass over the freshly parsed tree. */ +export function materializeTree( + tree: Parser.Tree, + source: string, + extraTypes: ReadonlySet +): MirrorNode { + const cursor = tree.walk(); + const root = new MirrorNode(cursor, source, extraTypes); + // hasError at the ROOT is read from tree-sitter itself (one boundary call + // per file): an error can live in a HIDDEN node — a file whose syntax error + // is swallowed shows no visible ERROR/missing child anywhere, yet + // ts_node_has_error is true, and module-level "partial grammar" columns + // depend on exactly that. When the root does carry an error, every node's + // flag is read from the real node instead of propagated bottom-up: the + // parse-gap stages walk for the DEEPEST hasError node, and a node can + // report it on itself with no visible ERROR child (a bare preproc_pragma + // does). The per-node boundary crossings are confined to the files that + // failed to parse; a healthy file pays one. + const exactErrors = tree.rootNode.hasError; + if (exactErrors) { + root._markError(); + } + const stack: MirrorNode[] = [root]; + let current = root; + + // gotoFirstChild / gotoNextSibling / gotoParent, no recursion: a deeply + // nested file must not overflow the JS stack when the C parser handled it. + let descending = true; + for (;;) { + if (descending && cursor.gotoFirstChild()) { + const child = new MirrorNode(cursor, source, extraTypes); + if (exactErrors && cursor.currentNode.hasError) { + child._markError(); + } + current._addChild(child, cursor.currentFieldName); + stack.push(child); + current = child; + continue; + } + // finishing `current`: the cursor sits on it, so repair its fields here + // if an extra child made the cursor's reporting untrustworthy… + if (current._needsFieldRepair) { + current._repairFieldsFrom(cursor.currentNode); + current._needsFieldRepair = false; + } + // …and fold its error state into the parent + if (current.type === 'ERROR' || current.isMissing || current.hasError) { + const parent = stack[stack.length - 2]; + if (parent !== undefined) { + parent._markError(); + } + current._markError(); + } + if (cursor.gotoNextSibling()) { + stack.pop(); + const parent = stack[stack.length - 1]; + if (parent === undefined) { + // the root has no siblings; the cursor cannot get here + return root; + } + const sibling = new MirrorNode(cursor, source, extraTypes); + if (exactErrors && cursor.currentNode.hasError) { + sibling._markError(); + } + parent._addChild(sibling, cursor.currentFieldName); + stack.push(sibling); + current = sibling; + descending = true; + continue; + } + stack.pop(); + const above = stack[stack.length - 1]; + if (!cursor.gotoParent() || above === undefined) { + return root; + } + current = above; + descending = false; + } +} diff --git a/parser/src/parsers/python/py-mirror-tree.ts b/parser/src/parsers/python/py-mirror-tree.ts index 1525e6f9..de940184 100644 --- a/parser/src/parsers/python/py-mirror-tree.ts +++ b/parser/src/parsers/python/py-mirror-tree.ts @@ -1,205 +1,17 @@ /** - * A plain-JS mirror of a tree-sitter tree, built in ONE cursor pass. - * - * Tree-sitter still parses every file; what this removes is the reading cost. - * Every property access on a tree-sitter SyntaxNode crosses the JS↔C++ - * boundary and re-marshals the node handle, and the Python extraction stages - * each walk the same tree, so one node's properties are fetched once per - * stage. The mirror pays the boundary once per node, during the cursor walk, - * and every later read is a JS property. - * - * The surface is exactly what the Python stages use (verified by grep over - * extractors, detector and soft-keywords): type, text, children/namedChildren, - * child(i)/namedChild(i), childForFieldName, counts, spans, parent, id, - * isNamed/isMissing/isExtra/hasError. Anything outside it throws at the call - * site rather than answering wrongly. - * - * `text` is sliced lazily from the one source string, so the mirror holds no - * copies. `isExtra` is derived from the node type: tree-sitter-python's extras - * are exactly `comment` and `line_continuation` (whitespace produces no node). - * `hasError` is computed bottom-up with tree-sitter's own meaning: an ERROR or - * missing node anywhere in the subtree. + * Python's one-pass plain-JS mirror of the tree-sitter tree. The machinery + * and the reasoning live in `../mirror-tree`; this module only pins the + * language's `extras`: tree-sitter-python's are exactly `comment` and + * `line_continuation` (whitespace produces no node). */ import type Parser from 'tree-sitter'; -/** tree-sitter-python `extras`: the only node types that parse as extra. */ -const PY_EXTRA_TYPES = new Set(['comment', 'line_continuation']); +import { MirrorNode, materializeTree } from '@/parsers/mirror-tree'; -/** Never reset: `HashByNodeId` maps must not collide across files. */ -let nextId = 1; +const PY_EXTRA_TYPES: ReadonlySet = new Set(['comment', 'line_continuation']); -export class PyMirrorNode { - readonly id: number; - readonly type: string; - readonly isNamed: boolean; - readonly isMissing: boolean; - readonly startIndex: number; - readonly endIndex: number; - readonly startPosition: Parser.Point; - readonly endPosition: Parser.Point; - parent: PyMirrorNode | null = null; - readonly children: PyMirrorNode[] = []; - namedChildren: PyMirrorNode[] = []; - /** First child per field name — the pick childForFieldName makes. */ - private fields: Map | null = null; - private errorInSubtree = false; - private readonly source: string; +export type PyMirrorNode = MirrorNode; - constructor(cursor: Parser.TreeCursor, source: string) { - this.id = nextId++; - this.type = cursor.nodeType; - // An ERROR node can be either: tree-sitter marks an ERROR it absorbed - // during recovery as EXTRA (siblings' named counts then skip it), while a - // plain ERROR is not. The type cannot tell them apart, so this is the one - // place the real node is consulted — ERROR nodes exist only in files that - // failed to parse, so the boundary crossing stays off the healthy path. - if (this.type === 'ERROR') { - this._extraOverride = cursor.currentNode.isExtra; - } - this.isNamed = cursor.nodeIsNamed; - this.isMissing = cursor.nodeIsMissing; - this.startIndex = cursor.startIndex; - this.endIndex = cursor.endIndex; - this.startPosition = cursor.startPosition; - this.endPosition = cursor.endPosition; - this.source = source; - } - - get text(): string { - return this.source.slice(this.startIndex, this.endIndex); - } - - /** Set at build time only for ERROR nodes — see materializePyTree. */ - _extraOverride: boolean | null = null; - - get isExtra(): boolean { - if (this._extraOverride !== null) { - return this._extraOverride; - } - return PY_EXTRA_TYPES.has(this.type); - } - - get hasError(): boolean { - return this.errorInSubtree; - } - - get childCount(): number { - return this.children.length; - } - - get namedChildCount(): number { - return this.namedChildren.length; - } - - child(index: number): PyMirrorNode | null { - return this.children[index] ?? null; - } - - namedChild(index: number): PyMirrorNode | null { - return this.namedChildren[index] ?? null; - } - - /** - * First IMMEDIATE child carrying the field, which is what every extractor - * asks for. Tree-sitter's own lookup additionally pierces one visible level - * on `match_statement` (its `alternative` case clauses sit inside the match - * `block`), a quirk nothing in the Python stages uses: the match consumers - * iterate the block's namedChildren by type instead (block-extractor, - * expression-extractor), and `alternative` is read only on if/for/while, - * where it is an immediate child. - */ - childForFieldName(fieldName: string): PyMirrorNode | null { - return this.fields?.get(fieldName) ?? null; - } - - /** @internal build-time wiring, called only by materializePyTree. */ - _addChild(child: PyMirrorNode, fieldName: string | null): void { - child.parent = this; - this.children.push(child); - if (child.isNamed) { - this.namedChildren.push(child); - } - if (fieldName !== null && fieldName !== '') { - if (this.fields === null) { - this.fields = new Map(); - } - if (!this.fields.has(fieldName)) { - this.fields.set(fieldName, child); - } - } - } - - /** @internal */ - _markError(): void { - this.errorInSubtree = true; - } -} - -/** - * One depth-first cursor pass over the freshly parsed tree. - * - * The result is handed to the stages as a `Parser.SyntaxNode`: the stages are - * typed against tree-sitter's interface and use only the mirrored subset, so - * the cast is confined to the one call site that builds the mirror. - */ -export function materializePyTree(tree: Parser.Tree, source: string): PyMirrorNode { - const cursor = tree.walk(); - const root = new PyMirrorNode(cursor, source); - // hasError at the ROOT is read from tree-sitter itself (one boundary call per - // file): an error can live in a HIDDEN node — a file whose syntax error is - // swallowed shows no visible ERROR/missing child anywhere, yet - // ts_node_has_error is true, and the module row's grammar column - // (TS_PYTHON3_PARTIAL) depends on exactly that. The bottom-up propagation - // below still covers every VISIBLE error for the deeper nodes. - if (tree.rootNode.hasError) { - root._markError(); - } - const stack: PyMirrorNode[] = [root]; - let current = root; - - // gotoFirstChild / gotoNextSibling / gotoParent, no recursion: a deeply - // nested file must not overflow the JS stack when the C parser handled it. - let descending = true; - for (;;) { - if (descending && cursor.gotoFirstChild()) { - const child = new PyMirrorNode(cursor, source); - current._addChild(child, cursor.currentFieldName); - stack.push(child); - current = child; - continue; - } - // finishing `current`: fold its error state into the parent - if ( - current.type === 'ERROR' || - current.isMissing || - current.hasError - ) { - const parent = stack[stack.length - 2]; - if (parent !== undefined) { - parent._markError(); - } - current._markError(); - } - if (cursor.gotoNextSibling()) { - stack.pop(); - const parent = stack[stack.length - 1]; - if (parent === undefined) { - // the root has no siblings; the cursor cannot get here - return root; - } - const sibling = new PyMirrorNode(cursor, source); - parent._addChild(sibling, cursor.currentFieldName); - stack.push(sibling); - current = sibling; - descending = true; - continue; - } - stack.pop(); - const above = stack[stack.length - 1]; - if (!cursor.gotoParent() || above === undefined) { - return root; - } - current = above; - descending = false; - } +export function materializePyTree(tree: Parser.Tree, source: string): MirrorNode { + return materializeTree(tree, source, PY_EXTRA_TYPES); } From 76716000543b5fb94ca996a44fcd38266a223602 Mon Sep 17 00:00:00 2001 From: swapnil <78632212+swapnilpaliwal-sd@users.noreply.github.com> Date: Sat, 3 Oct 2026 19:48:00 -0700 Subject: [PATCH 10/10] perf(typescript,build): verify CSV rows at append; reuse closure-pass reads; solve non-main languages concurrently MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three independent costs, one commit per stage touched: ts-relation-writer verified each finished relation by re-reading and re-decoding the whole file. The same rules — the header's field count and the consumer line-break alphabet — now run on each row's string as it is appended, in one allocation-free pass, and publish() proves the bytes arrived by comparing the byte count write() reported against the file's size, which is the defect the read-back existed to catch. The streaming verifier stays exported for the gate that exercises it. The root-program closure walk read and cheap-parsed every file to follow imports, then the extraction pass read every file again; the walk now hands its text over (consumed and released per file), and module resolution gets a ts.createModuleResolutionCache instead of re-probing node_modules per specifier. The all verb solved languages one after another although their solves share nothing. The largest language still solves alone first — its graph is the one a --progress caller publishes first — and the rest run in waves (AXIOMCODE_SOLVE_JOBS wide, default 2, 1 restores the strict line), each into its own intermediate, because run-souffle writes fixed names there. On a 1.1M-LOC TypeScript package, warm: stage 57.2s -> 50.9s with byte-identical IR; the removed double read was 12% of a cold parse. Serial and concurrent solves produce row-identical relations across all tables on a three-language tree. End to end on this repository (five languages), index wall time drops 37%. Co-authored-by: axiomcode-bot[bot] <334110751+axiomcode-bot[bot]@users.noreply.github.com> --- bin/axiomcode | 61 +++++++++++++++--- .../typescript/ts-relation-writer.ts | 62 ++++++++++++++++++- .../typescript/typescript-project-analyzer.ts | 25 ++++++-- 3 files changed, 133 insertions(+), 15 deletions(-) diff --git a/bin/axiomcode b/bin/axiomcode index 6b902cf0..946f2c38 100755 --- a/bin/axiomcode +++ b/bin/axiomcode @@ -324,16 +324,61 @@ case "$cmd" in # a caller that checks it is no worse off. Only the languages that produced a graph are # listed at the end, so the output stays a list of files that exist. solved=(); failed=() - for l in "${langs[@]}"; do - echo "▶ solving $l → $out/$l/graph.sqlite" - if "$0" engine --language "$l" --client-ir "$ir/$l" --out "$out/$l" --intermediate "$int" --library "$libroots" \ - --meta "source_version=$version" --meta "source_dir=$(cd "$src" && pwd)" ${rest[@]+"${rest[@]}"}; then - solved+=("$l"); [ -z "$progress" ] || echo "$l ok $out/$l/graph.sqlite" >> "$progress" + # Each language gets its OWN intermediate: run-souffle writes fixed names there + # (souffle-program.cpp, .souffle-gen.log), so two concurrent solves into one + # directory would overwrite each other's program mid-compile. + solve_one(){ local sl="$1" + "$0" engine --language "$sl" --client-ir "$ir/$sl" --out "$out/$sl" --intermediate "$int/$sl" --library "$libroots" \ + --meta "source_version=$version" --meta "source_dir=$(cd "$src" && pwd)" ${rest[@]+"${rest[@]}"} + } + record(){ local rl="$1" rrc="$2" + if [ "$rrc" -eq 0 ]; then + solved+=("$rl"); [ -z "$progress" ] || echo "$rl ok $out/$rl/graph.sqlite" >> "$progress" else - failed+=("$l"); echo "❌ $l failed; continuing with the remaining languages" >&2 - [ -z "$progress" ] || echo "$l failed" >> "$progress" + failed+=("$rl"); echo "❌ $rl failed; continuing with the remaining languages" >&2 + [ -z "$progress" ] || echo "$rl failed" >> "$progress" fi - done + } + # THE FIRST (LARGEST) LANGUAGE SOLVES ALONE, streaming, exactly as before: its graph is the one a + # --progress caller publishes first (#1555), and nothing may compete with it for cores or + # interleave its log. Only the languages AFTER it run concurrently — they used to wait in line + # behind each other for no reason: their solves share nothing (own IR, own output, own + # intermediate). Memory is the tradeoff (each is its own Soufflé process), so the width is + # modest by default and AXIOMCODE_SOLVE_JOBS raises or lowers it; 1 restores the strict line. + l="${langs[0]}" + echo "▶ solving $l → $out/$l/graph.sqlite" + rc=0; solve_one "$l" || rc=$?; record "$l" "$rc" + if [ ${#langs[@]} -gt 1 ]; then + others=("${langs[@]:1}") + JOBS="${AXIOMCODE_SOLVE_JOBS:-2}" + case "$JOBS" in (*[!0-9]*|'') JOBS=2;; esac; [ "$JOBS" -ge 1 ] || JOBS=1 + if [ "$JOBS" -eq 1 ]; then + for l in "${others[@]}"; do + echo "▶ solving $l → $out/$l/graph.sqlite" + rc=0; solve_one "$l" || rc=$?; record "$l" "$rc" + done + else + # Waves of $JOBS (bash 3.2 has no `wait -n`): start a wave, wait for all of it, replay each + # log whole so the build log never interleaves, then record in the wave's order. + i=0 + while [ $i -lt ${#others[@]} ]; do + wave=("${others[@]:$i:$JOBS}"); i=$((i + JOBS)) + wavepids=() + for l in "${wave[@]}"; do + echo "▶ solving $l → $out/$l/graph.sqlite (concurrent)" + { solve_one "$l" > "$int/solve-$l.log" 2>&1; echo $? > "$int/solve-$l.rc"; } & + wavepids+=($!) + done + for wp in "${wavepids[@]}"; do wait "$wp" || true; done + for l in "${wave[@]}"; do + cat "$int/solve-$l.log" 2>/dev/null + rc="$(cat "$int/solve-$l.rc" 2>/dev/null || echo 1)" + rm -f "$int/solve-$l.log" "$int/solve-$l.rc" + record "$l" "$rc" + done + done + fi + fi for l in ${solved[@]+"${solved[@]}"}; do echo "$out/$l/graph.sqlite"; done if [ ${#failed[@]} -gt 0 ]; then echo "❌ ${#failed[@]} of ${#langs[@]} languages failed: ${failed[*]}" >&2 diff --git a/parser/src/workflows/typescript/ts-relation-writer.ts b/parser/src/workflows/typescript/ts-relation-writer.ts index da00771e..f431759d 100644 --- a/parser/src/workflows/typescript/ts-relation-writer.ts +++ b/parser/src/workflows/typescript/ts-relation-writer.ts @@ -43,7 +43,9 @@ export class TsRelationWriter { private readonly outputPath: string; private buffer: string[] = []; private header = ''; + private width = 0; private rows = 0; + private bytesWritten = 0; private closed = false; constructor(outputDir: string, filename: string, uniqueSuffix: string) { @@ -71,10 +73,18 @@ export class TsRelationWriter { if (this.handle === undefined) { this.handle = await fsp.open(this.temporaryPath, 'w'); this.header = rows[0]!.getCsvHeader(); + this.width = countTabs(this.header) + 1; this.buffer.push(this.header + '\n'); } for (const row of rows) { - this.buffer.push(row.toCsv() + '\n'); + const line = row.toCsv(); + // The row is checked HERE, on the string that is about to be written, + // instead of decoding the finished file a second time: same width rule, + // same line-break alphabet, no re-read. What this no longer re-checks — + // that the bytes reached the disk whole — publish() covers by comparing + // the byte count it wrote against what the file system reports. + verifyRow(line, this.width, this.outputPath, this.rows + 2); + this.buffer.push(line + '\n'); this.rows += 1; } if (this.buffer.length >= TS_CSV_CHUNK_SIZE) { @@ -90,7 +100,8 @@ export class TsRelationWriter { // streaming cost the same as the whole-file writer did. const text = this.buffer.join(''); this.buffer = []; - await this.handle.write(text, null, 'utf-8'); + const { bytesWritten } = await this.handle.write(text, null, 'utf-8'); + this.bytesWritten += bytesWritten; } /** Flushes, verifies, and renames into place. */ @@ -107,9 +118,17 @@ export class TsRelationWriter { } await this.flush(); await this.handle.sync(); + // Every row was verified as it was appended (verifyRow); what remains to + // prove is that the bytes all arrived. The file's size must equal the sum + // of what write() reported — a mismatch is a torn write, the exact defect + // the old whole-file read-back existed to catch. + const onDisk = (await this.handle.stat()).size; await this.handle.close(); this.handle = undefined; - verifyRelationFileStreaming(this.temporaryPath, this.outputPath, this.header); + if (onDisk !== this.bytesWritten) { + throw new Error(`${path.basename(this.outputPath)}: wrote ${this.bytesWritten} byte(s) but the ` + + `file holds ${onDisk} — the write is torn`); + } await fsp.rename(this.temporaryPath, this.outputPath); } @@ -139,6 +158,43 @@ export class TsRelationWriter { */ const CONSUMER_LINE_BREAKS = /[\u000A\u000B\u000C\u000D\u001C\u001D\u001E\u0085\u2028\u2029]/; +function countTabs(line: string): number { + let tabs = 0; + for (let i = 0; i < line.length; i++) { + if (line.charCodeAt(i) === 0x09) { + tabs += 1; + } + } + return tabs; +} + +/** + * One row holds exactly the header's field count and no code point a consumer + * would break a line on \u2014 the same rules {@link verifyRelationFileStreaming} + * applies, checked on the in-memory string in one allocation-free pass. + */ +function verifyRow(line: string, width: number, outputPath: string, lineNumber: number): void { + if (line === '') { + // the streamed read-back skipped blank lines rather than calling them torn + return; + } + let tabs = 0; + for (let i = 0; i < line.length; i++) { + const c = line.charCodeAt(i); + if (c === 0x09) { + tabs += 1; + } else if ((c >= 0x0a && c <= 0x0d) || (c >= 0x1c && c <= 0x1e) || c === 0x85 + || c === 0x2028 || c === 0x2029) { + throw new Error(`${path.basename(outputPath)}: line ${lineNumber} carries a line-break code ` + + `point inside a value \u2014 the row would read torn: ${JSON.stringify(line.slice(0, 60))}`); + } + } + if (tabs + 1 !== width) { + throw new Error(`${path.basename(outputPath)}: line ${lineNumber} has ${tabs + 1} field(s) where ` + + `the header has ${width} \u2014 the row is torn: ${JSON.stringify(line.slice(0, 60))}`); + } +} + /** * Every row has exactly the header's field count, checked without holding the * file in memory. diff --git a/parser/src/workflows/typescript/typescript-project-analyzer.ts b/parser/src/workflows/typescript/typescript-project-analyzer.ts index 447c6fcd..26287df5 100644 --- a/parser/src/workflows/typescript/typescript-project-analyzer.ts +++ b/parser/src/workflows/typescript/typescript-project-analyzer.ts @@ -331,8 +331,13 @@ export class TypeScriptProjectAnalyzer { for (const file of files) { let sourceText: string; + // the closure walk read this file already; take its text and release it + const prefetched = rootProgram?.texts.get(file); + if (prefetched !== undefined) { + rootProgram!.texts.delete(file); + } try { - sourceText = await fsp.readFile(file, 'utf-8'); + sourceText = prefetched ?? await fsp.readFile(file, 'utf-8'); } catch (error) { this.recordSkip(file, pathAnchor, options, serviceVersionLinkHash, SkippedFileReason.READ_ERROR, String(error)); @@ -609,7 +614,13 @@ export class TypeScriptProjectAnalyzer { function filesOfRootProgram( rootDir: string, configResolver: TsConfigResolver -): { readonly files: string[]; readonly others: string[]; readonly orphans: string[] } | undefined { +): { + readonly files: string[]; + readonly others: string[]; + readonly orphans: string[]; + /** What the closure walk already read, so the extraction pass reads nothing twice. */ + readonly texts: Map; +} | undefined { const configPath = path.join(rootDir, 'tsconfig.json'); if (!fs.existsSync(configPath)) { return undefined; @@ -660,6 +671,11 @@ function filesOfRootProgram( // by a nested tsconfig stays in that program, which is what keeps a nested // project's separate global scope separate. const rootOptions = configResolver.resolve(claimed[0] ?? configPath).options; + // One resolution cache for the whole walk: ts.resolveModuleName with a bare + // ts.sys re-probes the same node_modules directories for every specifier, + // and the probing (statSync/readdirSync) was most of this pass's cost. + const resolutionCache = ts.createModuleResolutionCache(rootDir, (f) => f, rootOptions); + const texts = new Map(); const included = new Set(claimed.map((f) => path.normalize(f))); const available = new Map(unclaimed.map((f) => [path.normalize(f), f])); const queue = [...claimed]; @@ -671,13 +687,14 @@ function filesOfRootProgram( } catch { continue; } + texts.set(current, text); // No parent pointers and no type nodes needed: this pass only reads // specifiers, so the cheapest possible parse is the right one. const script = scriptTextOf(current, text); const sf = ts.createSourceFile(current, script.text, ts.ScriptTarget.Latest, false, script.scriptKind ?? (current.endsWith('.tsx') ? ts.ScriptKind.TSX : ts.ScriptKind.TS)); for (const specifier of importSpecifiersOf(sf)) { - const resolved = ts.resolveModuleName(specifier, current, rootOptions, ts.sys) + const resolved = ts.resolveModuleName(specifier, current, rootOptions, ts.sys, resolutionCache) .resolvedModule?.resolvedFileName ?? resolveVueSpecifier(specifier, current); if (resolved === undefined) { continue; @@ -702,7 +719,7 @@ function filesOfRootProgram( orphans.push(f); } } - return { files, others, orphans }; + return { files, others, orphans, texts }; } /**