diff --git a/bin/axiomcode b/bin/axiomcode index 7cc39dd11..313b110aa 100755 --- a/bin/axiomcode +++ b/bin/axiomcode @@ -1,7 +1,7 @@ #!/usr/bin/env bash # ───────────────────────────────────────────────────────────────────────────── # axiomcode — ask a repository's call graph. `axiomcode --help` prints the dispatcher's help -# (plugins/axiomcode/skills/axiomcode/scripts/axiomcode): index, impact, path and tests. +# (plugins/axiomcode/skills/axiomcode/scripts/axiomcode): index, context, impact, path and tests. # ───────────────────────────────────────────────────────────────────────────── # INTERNAL COMMANDS, not advertised: the build, the engine suites and the MCP server, which # the dispatcher, the test suites and the agent manifests call. diff --git a/plugins/axiomcode/AGENTS.md b/plugins/axiomcode/AGENTS.md index 1f6b1fcf1..f03f5c602 100644 --- a/plugins/axiomcode/AGENTS.md +++ b/plugins/axiomcode/AGENTS.md @@ -8,9 +8,11 @@ MCP tools, for what no text search answers: impact() with no name: the same for your uncommitted edits path(start, end) how A reaches B, every hop of the call chain tests() the tests your uncommitted edits reach, and the command that runs them + context(task) how something works, as a narrative: the call flow step by step; + context(task, source=True) carries each step's code Without the tools, the same from the shell: `axiomcode impact `, -`axiomcode path `, `axiomcode tests`. +`axiomcode path `, `axiomcode tests`, `axiomcode context "" --source`. Every answer is a numbered list of places, each with the code of the function it sits in and the line that matters marked `→`: answer from that code, and open a file only where a body was cut. A `resolved` place has diff --git a/plugins/axiomcode/mcp/server.py b/plugins/axiomcode/mcp/server.py index b7e72a62c..cfd5138e1 100755 --- a/plugins/axiomcode/mcp/server.py +++ b/plugins/axiomcode/mcp/server.py @@ -219,8 +219,17 @@ def plain(text): return '\n'.join(out) -# THE SMALL SURFACE. Search is grep's job; the graph answers what grep cannot. Three questions, each answered as numbered places with the code of the function each sits in, so -# a place is understood without opening its file. No options: the repository is the one the session works in. +# THE SMALL SURFACE. Search is grep's job; the graph answers what grep cannot. Three questions, each answered as +# numbered places with the code of the function each sits in, so a place is understood without opening its file, +# plus context, the one narrative verb: a task in words answered as the verb's own flow. No options beyond +# context's source: the repository is the one the session works in. +@srv.tool() +def context(task: str, source: bool = False) -> str: + """How something works, from a task in words: the files and callables the task touches, and for a "how does X + work" question the call FLOW — every step in the order the calls are written, with ⚠ where the graph lost a + call. source=True asks for the flow with each step's code, so it is read without opening files.""" + return plain(run(['context', task] + (['--source'] if source else []) + [os.getcwd()])) + @srv.tool() def impact(name: str = '') -> str: """What a change reaches. With a name (as written in the code: Owner.method, function, Type, or file.py:123): who diff --git a/plugins/axiomcode/rules/axiomcode.mdc b/plugins/axiomcode/rules/axiomcode.mdc index 05b30d8a9..129677f36 100644 --- a/plugins/axiomcode/rules/axiomcode.mdc +++ b/plugins/axiomcode/rules/axiomcode.mdc @@ -13,9 +13,11 @@ MCP tools, for what no text search answers: impact() with no name: the same for your uncommitted edits path(start, end) how A reaches B, every hop of the call chain tests() the tests your uncommitted edits reach, and the command that runs them + context(task) how something works, as a narrative: the call flow step by step; + context(task, source=True) carries each step's code Without the tools, the same from the shell: `axiomcode impact `, -`axiomcode path `, `axiomcode tests`. +`axiomcode path `, `axiomcode tests`, `axiomcode context "" --source`. Every answer is a numbered list of places, each with the code of the function it sits in and the line that matters marked `→`: answer from that code, and open a file only where a body was cut. A `resolved` place has diff --git a/plugins/axiomcode/skills/axiomcode/SKILL.md b/plugins/axiomcode/skills/axiomcode/SKILL.md index 2516f40bc..04ca8a048 100644 --- a/plugins/axiomcode/skills/axiomcode/SKILL.md +++ b/plugins/axiomcode/skills/axiomcode/SKILL.md @@ -1,13 +1,13 @@ --- name: axiomcode description: >- - Use for any why, what or where question about code — how a codebase works, where something lives, who calls it, what a change to it breaks, which tests cover an edit. Also use when resolving an issue or bug report, which names a symptom rather than a file. Examples: "How does X work?", "Where do I change Y?", "What calls this?", "What breaks if I change Z?", "Which tests do I run?", "Fix this issue". Search with grep as usual: after a grep the graph adds only what grep cannot know — which declaration each match reaches and the callers that never spell the name. Answers come from a resolved call graph, so they include callers that never spell the name — through an interface, an override, a callback, dependency injection or a config key — and every place comes with the code of the function it sits in. Call the MCP tools directly, no need to load this skill first: impact(name) for who calls it and what a change reaches (with no name: your uncommitted edits), path(start, end) for how A reaches B, tests() for the tests your edits reach. Only when those tools are not in your list, the same from the shell: `axiomcode impact `, `axiomcode path `, `axiomcode tests`. Java, TypeScript, Python, JavaScript, C#. + Use for any why, what or where question about code — how a codebase works, where something lives, who calls it, what a change to it breaks, which tests cover an edit. Also use when resolving an issue or bug report, which names a symptom rather than a file. Examples: "How does X work?", "Where do I change Y?", "What calls this?", "What breaks if I change Z?", "Which tests do I run?", "Fix this issue". Search with grep as usual: after a grep the graph adds only what grep cannot know — which declaration each match reaches and the callers that never spell the name. Answers come from a resolved call graph, so they include callers that never spell the name — through an interface, an override, a callback, dependency injection or a config key — and every place comes with the code of the function it sits in. Call the MCP tools directly, no need to load this skill first: impact(name) for who calls it and what a change reaches (with no name: your uncommitted edits), path(start, end) for how A reaches B, tests() for the tests your edits reach, context(task, source=True) for how something works as a step-by-step call flow with each step's code. Only when those tools are not in your list, the same from the shell: `axiomcode impact `, `axiomcode path `, `axiomcode tests`, `axiomcode context "" --source`. Java, TypeScript, Python, JavaScript, C#. --- # axiomcode Search with grep as usual; the graph answers what grep cannot. Use the MCP tools when they are in your list (in Claude Code -`mcp__plugin_axiomcode_axiomcode__impact`, `__path`, `__tests`); otherwise run +`mcp__plugin_axiomcode_axiomcode__impact`, `__path`, `__tests`, `__context`); otherwise run `/scripts/axiomcode ` from the repository root. Same answer either way. | the question | MCP tool | shell | @@ -17,6 +17,7 @@ Search with grep as usual; the graph answers what grep cannot. Use the MCP tools | what do my uncommitted edits reach? | `impact()` | `axiomcode impact` | | how does A reach B? | `path(start, end)` | `axiomcode path ` | | which tests do my edits need, and how do I run them? | `tests()` | `axiomcode tests` | +| how does this work, start to finish? | `context(task, source=True)` | `axiomcode context "" --source` | Names are written as in the code: `Owner.method`, `function`, `Type`, or `file.py:123` for the declaration at that line. There is no setup step: the first question builds the graph, and it refreshes itself after every edit. @@ -56,6 +57,14 @@ Example: `path(start="main", end="Ledger.put")`. The tests your uncommitted edits reach, each with its code, and a last line `run: ` that runs exactly those. Example: `tests()`. It is a lower bound: a test reached only through reflection or a service loader is not listed. +## context + +How something works, from a task in your own words: the files and callables the task touches and, for a +how-does-X-work question, the call flow step by step. Example: `context(task="how is an invoice settled", +source=True)` — source carries each step's code, so the flow is read without opening files. Only English task +words land (the graph's vocabulary is the code's identifiers); any language works once the task includes one +identifier as written in the code. + ## index `axiomcode index` builds the graph explicitly; `--lang`, `--src` and `--library` narrow it. Never re-run it on an diff --git a/plugins/axiomcode/skills/axiomcode/scripts/ax_blocks.py b/plugins/axiomcode/skills/axiomcode/scripts/ax_blocks.py index a46188007..34c4a44e8 100644 --- a/plugins/axiomcode/skills/axiomcode/scripts/ax_blocks.py +++ b/plugins/axiomcode/skills/axiomcode/scripts/ax_blocks.py @@ -168,7 +168,11 @@ def greppable(p): if other: out.append(f" {other} other line(s) grep matches for that name are NOT this declaration (another symbol of the same name, or text)") if not out: return None if len(places) > CAP: out.append(f"… {len(places) - CAP} more place(s) not shown — ask a narrower question to see them") - out += [x for x in foot if x.startswith(('run:', 'verified'))][:2] + # a bound: line is the answer saying where it stops being complete — an agent writing "only X reaches this" + # from these places needs it as much as the verified: line, so it is never tidied away here. + # run: stays LAST: the answer ends with the command to run, whatever else the foot carries. + kept = [x for x in foot if x.startswith(('verified', 'bound:'))][:3] + out += kept + [x for x in foot if x.startswith('run:')][:1] return out diff --git a/plugins/axiomcode/skills/axiomcode/scripts/ax_grep.py b/plugins/axiomcode/skills/axiomcode/scripts/ax_grep.py index 5e4d56383..a8bb8ec3b 100644 --- a/plugins/axiomcode/skills/axiomcode/scripts/ax_grep.py +++ b/plugins/axiomcode/skills/axiomcode/scripts/ax_grep.py @@ -196,6 +196,13 @@ def path(d, code): if ans: v = all(not a.get('unverified_hops') for a in ans) and v is not False foot = ev_foot(d) + [verified(v, sum(len(a.get('hops', [])) for a in ans) if ans else None)] if d.get('bound'): foot.append(f"bound: {d['bound']}") + # what reaches a [by name] caller reaches the target too whenever the by-name site is real: counted, or the + # upstream closure reads complete while every chain behind a by-name row is missing from it + bb = d.get('by_name_behind') + if bb: + foot.append(f"bound: {bb['methods']} more method(s) in {bb['files']} file(s) reach a [by name] caller above through" + " the graph's edges — callers of the target too if that by-name site is real; nearest: " + + ', '.join(f"{x['name']} ({x['at']})" for x in bb.get('nearest', []))) return rows, {}, foot diff --git a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode index 680a356a8..84766d19c 100755 --- a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode +++ b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode @@ -9,6 +9,9 @@ # how A reaches B: every hop of the call chain, with the code at each call. # axiomcode tests # the tests your uncommitted edits reach, and the command that runs exactly those. +# axiomcode context "" [--source] +# how something works, as a narrative: the files and callables the task touches and, for a +# how-does-X-work question, the call flow step by step; --source carries each step's code. # axiomcode index [] [--lang [,…]] [--src ] [--library [,…]] # build the graph (the first query builds it too). defaults to the current directory. # diff --git a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-context b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-context index 9bd4561de..9c19d8934 100755 --- a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-context +++ b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-context @@ -1198,7 +1198,16 @@ def main(argv): print(f" {n} — {len(rs)} call site(s)") for f, l, c, code in rs: print(f" {f}:{l}: {code}" + (f" [in {c}]" if c else '')) terms = task_terms(body) - if not terms: die("nothing to search for in that task description") + if not terms: + # the graph's vocabulary is the code's own identifiers, which are English words and names: a task + # written in another script matches nothing, and "nothing to search for" read as a parsing fault + # rather than a stated limit. A mixed task works — one identifier is enough to land. + if any(ord(c) > 127 and c.isalpha() for c in body or ''): + die("this task is written in a language the index cannot search: only English task words are supported," + " because the graph's vocabulary is the code's own identifiers." + " Rephrase the task in English, or keep your language and include one identifier as written in the" + " code (a mixed task works: the identifier lands it).") + die("nothing to search for in that task description") # what the question names that no graph here holds is said FIRST (#1571), and a scope that exists on disk but holds # no indexed file is a text scope, not a typo (#1382). Only a scope the index does not know is checked on disk. def holds(x): diff --git a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path index ce1c37721..51c36cbd1 100755 --- a/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path +++ b/plugins/axiomcode/skills/axiomcode/scripts/axiomcode-path @@ -1487,9 +1487,11 @@ def byname_sites(g, ids): on a receiver the engine could not type: `impact X` lists its caller as a lead, and the closure, which walks resolved edges only, left it out, so the two verbs gave different sets of direct callers and `path` gave no hint of the second one. The same rule as dl/impact.dl's `direct(... "by name")`: a callable target, a site that is not - a construction and not inside a mock's stub or verification, and not in the target's own body. Listed apart and - never walked: the name may belong to another method, and the by-name closure stays the search `path A B` runs - only when nothing resolved connects its two ends.""" + a construction and not inside a mock's stub or verification, and not in the target's own body. Listed apart, + not folded into the closure: the name may belong to another method, and the by-name closure stays the search + `path A B` runs only when nothing resolved connects its two ends. What reaches THEM is counted (byname_behind): + a method that reaches a by-name caller reaches the target too whenever the by-name site is real, and left + uncounted those callers-of-callers were missing from "everything that can reach X" with no line saying so.""" names = sorted({g.sym[i]['name'] for i in ids if i in g.sym and g.sym[i].get('method_id') and g.sym[i]['kind'] not in ('library', 'written', 'module') and g.sym[i].get('name')}) if not names or not g.has('unresolved_sites'): return [], set() @@ -1507,7 +1509,29 @@ def byname_sites(g, ids): return sorted(set(out), key=lambda x: (g.sym[x[0]]['is_test'], x[2] or '', x[3] or 0, g.sym[x[0]]['display'])), stubbed - {c for c, *_ in out} -def print_byname(g, found, limit): +def byname_behind(g, named, exclude, depth=40): + """the methods that reach a by-name caller through the graph's edges and are in neither the printed closure nor + the by-name list itself. Each one reaches the target exactly when the by-name site is real, so an upstream + closure that lists the by-name callers but not these is short by every chain behind them — on a spec question + ("only X can trigger this") they are the triggers the answer silently dropped. Counted and named nearest-first, + not folded in: their certainty is the by-name site's, not an edge's.""" + callers = {c for c, *_ in named} + if not callers: return [] + radj = collections.defaultdict(set) + for a, b, _ in g.edges(): radj[b].add(a) + seen = {c: 0 for c in callers}; fr = list(callers); d = 0 + while fr and d < depth: + d += 1; nxt = [] + for x in fr: + for y in radj.get(x, ()): + if y not in seen: seen[y] = d; nxt.append(y) + fr = nxt + out = [(m, dd) for m, dd in seen.items() if dd > 0 and m not in exclude and m in g.sym + and (not g.IN or g.under_in(g.sym[m]['file']))] + return sorted(out, key=lambda x: (x[1], g.sym[x[0]]['display'], x[0])) + + +def print_byname(g, found, limit, sel=None, behind=()): named, stubbed = found if stubbed: print(f" +{len(stubbed)} caller(s) only stub a method of this name on a mock (receiver not typed): they run none of it;" @@ -1520,6 +1544,16 @@ def print_byname(g, found, limit): for c, n, f, ln in named[:limit]: print(f" [by name] {g.disp(c)} {f}:{ln} — calls `{n}` (receiver not typed)") if len(named) > limit: print(f" … +{len(named) - limit} (--limit N)") + if behind: + # the examples: nearest NAMED methods first — an `` placeholder names nothing a reader can look up + show = sorted(behind, key=lambda x: ('<' in g.disp(x[0]), x[1], g.sym[x[0]]['display'], x[0]))[:3] + RESULT['by_name_behind'] = {'methods': len(behind), 'files': len({g.sym[m]['file'] for m, _ in behind}), + 'nearest': [{'name': g.disp(m), 'at': g.loc(m), 'hops': dd} for m, dd in show]} + tgt = f" is really `{sel}`" if sel else " resolves the way its name suggests" + print(f" behind them: {len(behind)} more method(s) in {len({g.sym[m]['file'] for m, _ in behind})} file(s)" + f" reach these by-name caller(s) through the graph's edges — callers of the target too if a by-name site{tgt};" + " nearest: " + ', '.join(f"{g.disp(m)} ({g.loc(m)})" for m, _ in show) + + (f" … +{len(behind) - 3}" if len(behind) > 3 else '')) def closure(g, sel, upstream, limit=40, depth=40): @@ -1580,11 +1614,12 @@ def closure(g, sel, upstream, limit=40, depth=40): if not upstream: print_boundary(g, ids, "it calls into libraries directly", "it also makes") named = byname_sites(g, ids) if upstream else ([], set()) + behind = byname_behind(g, named[0], {m for m, _ in rows} | set(ids) | {c for c, *_ in named[0]}, depth) if named[0] else [] if not rows: u = g.q(f"SELECT count(*) n FROM unresolved_sites WHERE caller_id IN ({','.join('?' * len(ids))})", *ids)[0]['n'] if not upstream else 0 print(" none — " + ("its body has %d unresolved call(s), so what it reaches is unknown, not nothing" % u if not upstream and u else "no resolved call " + ("into it; " if upstream else "out of it; ") + (f"{len(named[0])} unresolved site(s) write its name (below)" if named[0] else "an unresolved site elsewhere may still " + ("call it" if upstream else "be it")))) - print_byname(g, named, limit) + print_byname(g, named, limit, sel=sel, behind=behind) # AN EMPTY UPSTREAM CLOSURE IS THE MOST MISLEADING LINE THIS COMMAND CAN PRINT. For a live route handler, # a signal receiver or a CLI command the answer "0 methods reach it" is true of calls and false of the # program: the framework reaches it. Name the registration rather than leave the reader at a dead end. @@ -1659,7 +1694,7 @@ def closure(g, sel, upstream, limit=40, depth=40): print(f" {d:2} hop(s) {g.disp(m)} {g.loc(m)}{at}") if len(np_) > limit: print(f" … +{len(np_) - limit} production (--limit N)") if nt: print(f" +{len(nt)} test caller(s) within 2 hop(s) — `axiomcode test-impact` names them and the command that runs them") - print_byname(g, named, limit) + print_byname(g, named, limit, sel=sel, behind=behind) byhop = collections.Counter(d for _, d in rows); print(" by hop: " + ', '.join(f"{d}:{n}" for d, n in sorted(byhop.items()))) # most_common breaks a tie by insertion order, which is the order the closure rows arrived in: two files with the # same count then swap depending on which engine answered. Sort the tie by name so the line is stable either way. @@ -1705,13 +1740,16 @@ def closure(g, sel, upstream, limit=40, depth=40): # figures on a large tree, which reads as "unknown" and is skipped. The reader can only check the rows # actually in front of them, so the actionable number is the unresolved calls inside THOSE; the closure-wide # figure stays, in parentheses, as the honest total. + RESULT['unresolved_closure'] = u # 0 is a counted zero, not a missing value if u: shown_ids = [m for m, _ in (np_ if upstream else sorted(rows, key=lambda x: (x[1], g.sym[x[0]]['display'], x[0])))[:limit]] un = g.q("SELECT count(*) n FROM unresolved_sites WHERE caller_id IN (%s)" % ','.join('?' * (len(shown_ids) + len(ids))), *shown_ids, *ids)[0]['n'] if shown_ids or ids else 0 - print(f" bound: {un} unresolved call(s) inside the {len(shown_ids)} method(s) printed above" - + (f" ({u} across the whole closure)" if u != un else '') - + f" — the set is a lower bound; `path {sel}` for the chain") + RESULT['unresolved_inside'] = un + RESULT['bound'] = (f"{un} unresolved call(s) inside the {len(shown_ids)} method(s) printed above" + + (f" ({u} across the whole closure)" if u != un else '') + + f" — the set is a lower bound; `path {sel}` for the chain") + print(f" bound: {RESULT['bound']}") print(f" (chain from any one of them: path {sel}" + (")" if upstream else " — or the reverse)")) return 0 @@ -1888,13 +1926,20 @@ def path(g, a, b, show_all=False, limit=10, every=False, max_paths=20): if shown >= limit and not show_all: print(f" … +{len(hits) - shown} more targets (--all)"); break blind = {n for n, _ in (read_back(res['parent'], q, hits[0][0], srcs) or [])} u = g.q(f"SELECT count(*) n FROM unresolved_sites WHERE caller_id IN ({','.join('?' * len(blind))})", *blind)[0]['n'] if blind else 0 + # the JSON says what the prose says: verified true/false once hops were checked (null stays "nothing was + # checked"), and the unresolved count as a number, where 0 is a counted zero. An automation was parsing the + # prose to learn both because the fields sat at null beside a prose line stating them. + RESULT['verified'] = not (vbad or vlen) + RESULT['unresolved_inside'] = u print(f" verified: every printed hop is an edge in the graph and a second, independent traversal finds the same length" if not (vbad or vlen) else f" ✗ verification failed on {vbad} hop(s) / {vlen} length(s) — report this") if seen_tiers: print(" what the hops are:"); [print(l) for l in ax_edges.legend(seen_tiers)] print_boundary(g, [m for m, _ in hits[:limit]], "the target(s) call into libraries", "the target(s) also make") if every: every_route(g, res, q, srcs, dsts, max_paths, adj) else: print(f" (one shortest chain per reached target; --every for all the routes and every method on any of them)") - if u: print(f" bound: the methods on the nearest chain contain {u} unresolved call(s) — other chains may exist that the graph cannot see") + if u: + RESULT['bound'] = f"the methods on the nearest chain contain {u} unresolved call(s) — other chains may exist that the graph cannot see" + print(f" bound: {RESULT['bound']}") return 0 # nothing resolved either way: say so, then whether unresolved sites would connect them print(f"no chain of resolved calls connects {la} and {lb} in either direction (searched {len(g.edges())} edges, depth ≤ 40)") diff --git a/skills/axiomcode/SKILL.md b/skills/axiomcode/SKILL.md index ae477b924..38950cb10 100644 --- a/skills/axiomcode/SKILL.md +++ b/skills/axiomcode/SKILL.md @@ -56,6 +56,14 @@ Example: `path(start="main", end="Ledger.put")`. The tests your uncommitted edits reach, each with its code, and a last line `run: ` that runs exactly those. Example: `tests()`. It is a lower bound: a test reached only through reflection or a service loader is not listed. +## context + +How something works, from a task in your own words: the files and callables the task touches and, for a +how-does-X-work question, the call flow step by step. Example: `context(task="how is an invoice settled", +source=True)` — source carries each step's code, so the flow is read without opening files. Only English task +words land (the graph's vocabulary is the code's identifiers); any language works once the task includes one +identifier as written in the code. + ## index `axiomcode index` builds the graph explicitly; `--lang`, `--src` and `--library` narrow it. Never re-run it on an diff --git a/tests/cases/python/path-crosses-framework-and-byname/app/jobs.py b/tests/cases/python/path-crosses-framework-and-byname/app/jobs.py index e6007ef38..3aea44d51 100644 --- a/tests/cases/python/path-crosses-framework-and-byname/app/jobs.py +++ b/tests/cases/python/path-crosses-framework-and-byname/app/jobs.py @@ -12,3 +12,13 @@ def replay(source): def rewind(source): return source.reopen(3) + + +def scheduler(source): + # reaches settle only through replay's by-name site + return replay(source) + + +def rollback(source): + # reaches only rewind, whose by-name site names reopen, not settle + return rewind(source) diff --git a/tests/cases/python/path-crosses-framework-and-byname/case.json b/tests/cases/python/path-crosses-framework-and-byname/case.json index 609428ed6..62f0e0e52 100644 --- a/tests/cases/python/path-crosses-framework-and-byname/case.json +++ b/tests/cases/python/path-crosses-framework-and-byname/case.json @@ -19,6 +19,33 @@ {"why": "#1421: path '*' lists the untyped-receiver call impact lists as [by name]", "run": ["path", "*", "Ledger.settle"], "want": ["1 hop(s) nightly", "[by name] replay app/jobs.py:10 — calls `settle` (receiver not typed)"]}, + {"why": "a method that reaches a by-name caller through resolved calls is counted behind it: it reaches the target too whenever the by-name site is real, and uncounted it was missing from the closure with no line saying so", + "run": ["path", "*", "Ledger.settle"], + "want": ["behind them: 1 more method(s) in 1 file(s)", "scheduler"]}, + {"why": "CONTROL: a caller of a method whose by-name site names ANOTHER method is not behind the target's by-name callers", + "run": ["path", "*", "Ledger.settle"], + "avoid": ["rollback"]}, + {"why": "--json says what the prose says: verification and the unresolved count are typed fields, 0 a counted zero, never null beside a prose line stating them", + "run": ["path", "nightly", "Ledger.settle", "--json"], "stdout_json": true, + "want": ["\"verified\": true", "\"unresolved_inside\": 0"]}, + {"why": "a chain through a method with an unresolved call types the count and the bound the prose prints", + "run": ["path", "scheduler", "replay", "--json"], "stdout_json": true, + "want": ["\"verified\": true", "\"unresolved_inside\": 1", "\"bound\": \"the methods on the nearest chain contain 1 unresolved call(s)"]}, + {"why": "the closure answer types its verification, the closure-wide unresolved count and the by-name groups", + "run": ["path", "*", "Ledger.settle", "--json"], "stdout_json": true, + "want": ["\"verified\": true", "\"unresolved_closure\": 0", "\"by_name\"", "\"by_name_behind\""]}, + {"why": "a task written in another language names the lexical limit, not a parsing fault", + "run": ["context", "钱包转账的金额是怎么结算的"], "expect_error": true, + "want": ["only English task words are supported", "a mixed task works"], + "avoid": ["nothing to search for"]}, + {"why": "CONTROL: one identifier lands a task kept in another language", + "run": ["context", "订单 Ledger 怎么结算"], + "want": ["task terms:"], + "avoid": ["only English task words"]}, + {"why": "CONTROL: an English task with no content words keeps the plain empty-terms line", + "run": ["context", "it is"], "expect_error": true, + "want": ["nothing to search for in that task description"], + "avoid": ["only English task words"]}, {"why": "impact names the same by-name caller", "run": ["impact", "Ledger.settle"], "want": ["[by name] replay", "[resolved] nightly"]}, diff --git a/tests/front_door.py b/tests/front_door.py index 57fef9c2e..67566e8e3 100644 --- a/tests/front_door.py +++ b/tests/front_door.py @@ -12,7 +12,7 @@ path answer with numbered places and a fenced code block; after an edit, impact with no name starts with `your edits:`, and tests lists the test with its code and ends with a `run:` line. A verb off the surface is refused with the supported list. - b. the MCP server lists exactly impact, path and tests, each with at most two parameters, and a call to one + b. the MCP server lists exactly context, impact, path and tests, each with at most two parameters, and a call to one answers in the same shape. c. CONTROLS: the dispatcher run directly, bin/axiomcode with --json, and AXIOMCODE_RAW=1 give the old answer — no fenced block — for the same question. @@ -98,11 +98,13 @@ def main(): if fails: return 1 # ── a. the installed command, no flags ─────────────────────────────────────────────────────────────────── - for verb, args in (('find', ('how is the invoice total computed',)), ('context', ('the invoice total',)), + for verb, args in (('find', ('how is the invoice total computed',)), ('graph', ()), ('diff', ()), ('install', ())): rc, out, err = cli(repo, verb, *args) check(f'{verb}: off the surface, the installed command refuses it and names the supported verbs', rc != 0 and 'impact' in err and 'path' in err and 'tests' in err, (out + err)[:400]) + rc, out, err = cli(repo, 'context', 'how is the invoice total computed') + check('context: on the surface, the task in words answers with the flow', rc == 0 and 'pricing.py' in out, (out + err)[:400]) rc, out, err = cli(repo, 'impact', 'vat_rate') check('impact : its direct caller as a numbered place with its code', rc == 0 and places(out) and 'shop/pricing.py:6' in out and 'return net * (1 + vat_rate())' in out, out[:600] + err[-300:]) @@ -125,7 +127,7 @@ def main(): # ── b. the MCP server ────────────────────────────────────────────────────────────────────────────────────── got = mcp(repo, [('path', {'start': 'total', 'end': 'vat_rate'}), ('impact', {'name': 'vat_rate'})]) tools = {t['name']: list((t.get('inputSchema') or {}).get('properties', {})) for t in got.get(2, {}).get('tools', [])} - check('MCP tools/list is exactly impact, path and tests (search is grep\'s)', set(tools) == {'impact', 'path', 'tests'}, tools) + check('MCP tools/list is exactly context, impact, path and tests (search is grep\'s)', set(tools) == {'context', 'impact', 'path', 'tests'}, tools) check('MCP: every tool takes at most two parameters', bool(tools) and all(len(p) <= 2 for p in tools.values()), tools) text = lambda i: ''.join(c.get('text', '') for c in got.get(i, {}).get('content', [])) check('MCP path answers as numbered places with their code', places(text(3)), text(3)[:600]) diff --git a/tests/mcp.py b/tests/mcp.py index ac05f5e00..2870385e9 100644 --- a/tests/mcp.py +++ b/tests/mcp.py @@ -30,8 +30,9 @@ LAUNCHER = os.path.join(ROOT, 'bin', 'axiomcode.js') SERVER = os.path.join(ROOT, 'plugins', 'axiomcode', 'mcp', 'server.py') # THE SMALL SURFACE: four questions, each with at most two parameters and no options. The front-door answer is capped -# at ten places with the rest counted, so no tool is paged. -TOOLS = {'impact': ['name'], 'path': ['start', 'end'], 'tests': []} +# at ten places with the rest counted, so no tool is paged. context is the one narrative verb: a task in words, +# answered as the verb's own flow rather than as places. +TOOLS = {'context': ['task', 'source'], 'impact': ['name'], 'path': ['start', 'end'], 'tests': []} def exchange(cmd, cwd, env=None, workdir=None): @@ -116,7 +117,9 @@ def check_arguments(label, cmd, cwd, lax=False): # it had been narrowed (#1567): the repository is the session's, and there are no flags ('path', {'start': 'a', 'end': 'b', 'in_path': 'src'}, 'in_path: unexpected argument'), ('impact', {'name': 'A.f', 'repo': cwd}, 'repo: unexpected argument'), - ('tests', {'why': True}, 'why: unexpected argument')] + ('tests', {'why': True}, 'why: unexpected argument'), + ('context', {}, 'task'), + ('context', {'task': 'x', 'budget': 3}, 'budget: unexpected argument')] for name, args, field in wrong: res = call(cmd, cwd, name, args) text = ' '.join(c.get('text', '') for c in res.get('content', [])) @@ -125,7 +128,8 @@ def check_arguments(label, cmd, cwd, lax=False): bad.append(f"{label}: {name}({json.dumps(args)}) was not refused naming {field!r}: {res}") # the control: every parameter a tool declares still passes, including the ones the CLI's hints name right = [('impact', {'name': 'A.f'}), ('impact', {}), - ('path', {'start': 'a', 'end': 'b'}), ('tests', {})] + ('path', {'start': 'a', 'end': 'b'}), ('tests', {}), + ('context', {'task': 'x'}), ('context', {'task': 'x', 'source': True})] for name, args in right: res = call(cmd, cwd, name, args) text = ' '.join(c.get('text', '') for c in res.get('content', [])) @@ -144,7 +148,8 @@ def check_front_door(): cwd = os.getcwd() try: bad = [] - for call, want in ((lambda: server.impact('A.f'), ['impact', 'A.f', cwd]), + for call, want in ((lambda: server.context('how is a total computed'), ['context', 'how is a total computed', cwd]), + (lambda: server.impact('A.f'), ['impact', 'A.f', cwd]), (lambda: server.impact(''), ['impact', cwd]), (lambda: server.impact(), ['impact', cwd]), (lambda: server.path('a', 'b'), ['path', 'a', 'b', cwd]), diff --git a/tests/surfaces.py b/tests/surfaces.py index d7e5cb582..ea2e81f7d 100644 --- a/tests/surfaces.py +++ b/tests/surfaces.py @@ -15,7 +15,7 @@ that is in neither list fails, so exposing one is a decision rather than an accident (#1034). The agent-facing docs (both copies of SKILL.md, AGENTS.md, the Cursor rule, and the README's CLI section) name no old -MCP tool (`axiomcode_context` …) and no flag other than index's. +MCP tool (`axiomcode_context` …) and no flag other than index's setup flags and context's --source. python3 tests/surfaces.py """ @@ -29,17 +29,17 @@ MCP = os.path.join(PLUG, 'mcp', 'server.py') CLI = os.path.join(ROOT, 'bin', 'axiomcode') # the command an install puts on $PATH -PUBLIC = ['index', 'impact', 'path', 'tests'] +PUBLIC = ['index', 'impact', 'path', 'tests', 'context'] NO_MCP = {'index': 'setup, not a question: the first query through the MCP server builds the graph itself'} # dispatched, not advertised: verb -> why INTERNAL = { 'build': 'the old name of index', - 'context': 'search by task words; the orient hook and the suites call it, no caller-facing surface does', 'changed': 'impact with no name answers the same question at the front door; the edit hooks read it with --json', 'test-impact': 'what tests runs; its flags (--range, --staged, --why, …) serve scripts and the suites', } OLD_TOOLS = re.compile(r'\baxiomcode_(context|impact|path|changed|test_impact|graph|index|diff)\b') -INDEX_FLAGS = {'--lang', '--src', '--library'} +# the flags the surface may teach: index's setup flags, and context's one option (the flow with each step's code) +INDEX_FLAGS = {'--lang', '--src', '--library', '--source'} def dispatched():