Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
143 changes: 140 additions & 3 deletions plugins/axiomcode/skills/axiomcode/scripts/ax_nonsource.py
Original file line number Diff line number Diff line change
Expand Up @@ -65,16 +65,151 @@ def _classify(fp):
# callable: called (`note(`), quoted as a value (`"note"`, `'note'`), or qualified with `#`, `::` or `->`
# (`Owner#note`; a CSS `#note` selector is not one). A dotted `Owner.note` is matched as the qualified name itself and never reaches this rule. The
# rest are PROSE: counted and grep-able, not listed as places a rename breaks.
OUT_OF_SCOPE_EXT = {'.sh', '.bash', '.zsh', '.ksh', '.dl'}
#
# LOCKFILES AND MANIFEST LISTS. A lockfile is written by a package manager, never by hand, and every word in it is a
# package name, a version or a flag: `"optional": true`, `"debug": "^4.1.0"`. A manifest's metadata (name,
# description, keywords) and its dependency lists are the same: they name packages, not callables. A method named
# `debug`, `optional` or `string` matched there is a package or a word, never a binding, and in a JavaScript
# repository it outnumbered every real one (a context question's next step became a line of package-lock.json).
# The rest of a manifest (scripts, tasks, tool configuration) is kept: that is where a name can be referred to. A
# JSON manifest's KEYS are not: `"start":` names an npm script and `"testEnvironment":` an option, so only the values
# are searched. What is skipped is decided by where the word sits in the file's structure (the entry, table, block or
# element that holds it), not by the line alone: a one-line manifest holds its scripts and its dependencies together.
OUT_OF_SCOPE_EXT = {'.sh', '.bash', '.zsh', '.ksh', '.dl', '.lock', '.lockfile'}
SHELL_SHEBANG = re.compile(r'#!\s*\S*(?:/|\s)(?:env\s+)?(?:ba|z|k|da)?sh\b')
LOCKFILES = {'package-lock.json', 'npm-shrinkwrap.json', 'pnpm-lock.yaml', 'yarn.lock', 'bun.lock', 'deno.lock',
'packages.lock.json', 'project.assets.json', 'paket.lock', 'gradle.lockfile', 'poetry.lock', 'uv.lock',
'pipfile.lock', 'pdm.lock', 'conda-lock.yml', 'composer.lock', 'gemfile.lock', 'cargo.lock', 'go.sum'}
# JSON manifests: the top-level keys that describe the package or list other packages
_JSON_META = r'name|version|description|keywords|authors?|contributors|maintainers|license|homepage|repository|bugs|funding|private'
JSON_LIST_KEY = {
'package.json': re.compile(r'(?i)^(' + _JSON_META + r'|type|engines|os|cpu|publishConfig|workspaces|packageManager|'
r'overrides|resolutions|pnpm|\w*dependencies(Meta)?)$'),
'bower.json': re.compile(r'(?i)^(' + _JSON_META + r'|ignore|resolutions|\w*dependencies)$'),
'deno.json': re.compile(r'^(name|version|imports|scopes|importMap|lock|nodeModulesDir|vendor|workspace|patch|links)$'),
'composer.json': re.compile(r'(?i)^(' + _JSON_META + r'|type|support|require(-dev)?|conflict|replace|provide|suggest|'
r'repositories|minimum-stability|prefer-stable)$')}
JSON_LIST_KEY['deno.jsonc'] = JSON_LIST_KEY['deno.json']
# YAML manifests: the top-level blocks that list packages (a conda environment, a pnpm workspace and its catalog)
YAML_LIST_KEY = {
'environment.yml': re.compile(r'^(name|channels|dependencies|prefix)$'),
'pnpm-workspace.yaml': re.compile(r'^(packages|catalogs?|overrides|patchedDependencies|\w*BuiltDependencies|'
r'peerDependencyRules|allowedDeprecatedVersions|packageExtensions)$')}
YAML_LIST_KEY['environment.yaml'] = YAML_LIST_KEY['environment.yml']
# TOML manifests: the tables and keys that list packages
PY_DEP_TABLE = re.compile(r'^\[\s*(dependency-groups|project\.optional-dependencies|tool\.poetry(\.group\.[^\]]+)?\.(dev-)?dependencies|'
r'tool\.pdm\.dev-dependencies|tool\.uv)\s*\]')
PIPFILE_TABLE = re.compile(r'^\[\s*(packages|dev-packages|requires|source|[\w-]+-packages)\s*\]')
PY_DEP_KEY = re.compile(r'^\s*(dependencies|requires|dev-dependencies|optional-dependencies)\s*=')
# a file that is nothing but a list of packages
REQUIREMENTS = re.compile(r'^(requirements|constraints)[\w.-]*\.(txt|in)$')
# XML manifests: the elements that name a package (MSBuild, packages.config, .nuspec) and a POM's dependency blocks
XML_PKG_LINE = re.compile(r'<\s*(PackageReference|PackageVersion|package|dependency)\b[^>]*\b(Include|Update|id)\s*=')
POM_BLOCK = re.compile(r'<(/?)(dependencies|dependencyManagement|parent|exclusions)>')
POM_COORD = re.compile(r'^\s*<(groupId|artifactId|version|packaging|name|description|url|scope|type|classifier|optional|'
r'modelVersion|id|tags|authors|owners)>[^<]*</\1>\s*$')


def is_lockfile(rel):
"""a file a package manager writes: every word in it is a package, a version or a flag"""
return os.path.basename(rel).lower() in LOCKFILES


def _json_spans(text, list_key):
"""{line: [(start col, end col)]} of the strings a word is not matched in: each string under a top-level key that
`list_key` matches, and each key at any depth. A string-aware scan, so a minified manifest is split by entry too."""
out, depth, want_key, listed, line, bol, i, n = {}, 0, False, False, 1, 0, 0, len(text)
while i < n:
c = text[i]
if c == '"':
j = i + 1
while j < n and text[j] != '"': j += 2 if text[j] == '\\' else 1
k = j + 1
while k < n and text[k] in ' \t\r\n': k += 1
if depth == 1 and want_key: want_key, listed = False, bool(list_key.match(text[i + 1:j]))
if (k < n and text[k] == ":") or (depth >= 1 and listed):
out.setdefault(line, []).append((i - bol, j + 1 - bol))
nl = text.count('\n', i, j)
if nl: line += nl; bol = text.rindex('\n', i, j) + 1
i = j + 1
continue
if text.startswith('//', i): # a JSONC comment (deno.jsonc)
j = text.find('\n', i); i = n if j < 0 else j
continue
if c in '{[':
depth += 1
if depth == 1: want_key = c == '{'
elif c in '}]': depth -= 1
elif c == ',' and depth == 1: want_key = True
elif c == '\n': line += 1; bol = i + 1
i += 1
return out


def _unquoted(s):
"""a TOML line without its strings and its comment: what is left are the brackets that open and close a list"""
return re.sub(r'"(?:\\.|[^"\\])*"|\'[^\']*\'', '""', s).split('#', 1)[0]


def _toml_lines(text, table, key=None):
"""the lines inside a table `table` matches, and those of a `key = [ ... ]` list outside one"""
out, in_table, open_brackets = set(), False, 0
for i, ln in enumerate(text.split('\n'), 1):
if open_brackets > 0: # inside a `dependencies = [ ... ]` spread over lines
out.add(i); s = _unquoted(ln); open_brackets += s.count('[') - s.count(']'); continue
if ln.lstrip().startswith('['): in_table = bool(table.match(ln.strip())); continue
if in_table: out.add(i); continue
if key and key.match(ln):
out.add(i); s = _unquoted(ln); open_brackets = s.count('[') - s.count(']')
return out


def _yaml_lines(text, block):
"""the lines of the top-level YAML blocks `block` matches: the key's line and every indented or list line under it"""
out, inside = set(), False
for i, ln in enumerate(text.split('\n'), 1):
m = re.match(r'([\w.-]+)\s*:', ln)
if m: inside = bool(block.match(m.group(1)))
elif ln[:1] not in ('', ' ', '\t', '-', '#'): inside = False
if inside: out.add(i)
return out


def _xml_lines(base, text):
"""the lines of an XML manifest that name a package: a POM's dependency blocks and coordinates, a NuGet element"""
lines, out, depth = text.split('\n'), set(), 0
for i, ln in enumerate(lines, 1):
if base == 'pom.xml':
opened = depth > 0
for m in POM_BLOCK.finditer(ln): depth += -1 if m.group(1) else 1
if opened or depth > 0 or POM_COORD.match(ln): out.add(i)
elif XML_PKG_LINE.search(ln) or (base.endswith('.nuspec') and POM_COORD.match(ln)): out.add(i)
return out


def manifest_skip(rel, text):
"""-> skip(line, col): True where a word written there sits in a lockfile, a manifest's metadata or one of its
dependency lists (see LOCKFILES AND MANIFEST LISTS); None for any other file"""
base = os.path.basename(rel).lower()
if is_lockfile(rel) or REQUIREMENTS.match(base): return lambda _l, _c: True
if base in JSON_LIST_KEY:
spans = _json_spans(text, JSON_LIST_KEY[base])
return lambda l, c: any(a <= c < b for a, b in spans.get(l, ()))
if base in YAML_LIST_KEY: lines = _yaml_lines(text, YAML_LIST_KEY[base])
elif base == 'pyproject.toml': lines = _toml_lines(text, PY_DEP_TABLE, PY_DEP_KEY)
elif base == 'pipfile': lines = _toml_lines(text, PIPFILE_TABLE)
elif base in ('pom.xml', 'packages.config') or base.endswith(('.csproj', '.fsproj', '.vbproj', '.props', '.targets', '.nuspec')):
lines = _xml_lines(base, text)
else: return None
return lambda l, _c: l in lines
COMMON = re.compile(r'[a-z]+')
_QUOTES = '"\'`'
QUALIFIER = re.compile(r'[\w$)\]>](?:#|::|->)$') # `Owner#note`, `Owner::note`, `$obj->note`; not a CSS `#note` selector


def out_of_scope(rel, text):
"""a shell script or a Datalog file: a name matched there is never a binding (owner's rule)"""
if os.path.splitext(rel)[1].lower() in OUT_OF_SCOPE_EXT: return True
"""a shell script, a Datalog file or a lockfile: a name matched there is never a binding (owner's rule)"""
if os.path.splitext(rel)[1].lower() in OUT_OF_SCOPE_EXT or is_lockfile(rel): return True
return not os.path.splitext(rel)[1] and bool(SHELL_SHEBANG.match(text[:120]))


Expand Down Expand Up @@ -211,9 +346,11 @@ def hits(self, names):
try: text = open(os.path.join(self.repo, rel), errors='replace').read()
except OSError: continue
if not any(n in text for n in names) or out_of_scope(rel, text): continue
skip = manifest_skip(rel, text)
for i, line in enumerate(text.split('\n'), 1):
shaped = {}
for m in pat.finditer(line):
if skip and skip(i, m.start(1)): continue
n = m.group(1); hits.append((n, rel, i))
shaped[n] = shaped.get(n, False) or not is_common(n) or code_shaped(line, m.start(1), m.end(1))
self.prose.update((n, rel, i) for n, ok in shaped.items() if not ok)
Expand Down
13 changes: 10 additions & 3 deletions plugins/axiomcode/skills/axiomcode/scripts/ax_pages.py
Original file line number Diff line number Diff line change
Expand Up @@ -300,10 +300,17 @@ def next_context(text):
"step's BODY, not only the line shown, since the body is the explanation and the flow is only its spine "
"(`--source` prints it)." + gap)
m = re.search(r'^\s+(?:hop \d+|name only, no call path)\s+(\S+)\s+\(\d+ symbol\(s\)\)[^\n]*\n\s+-> ([^\n]+)', text, re.M)
if m:
f = m.group(1); syms = [x.strip() for x in m.group(2).split(',') if x.strip()][:2]
return (f"next: read {f} first — it holds {' and '.join(syms)}; then `impact <the one you will change>` for what a change "
"to it reaches. The other files are ranked context, not a reading list")
# --source prints each file's declarations as code (`name (file:line)` and its lines) instead of the `->` list
m = re.search(r'^\s+(?:hop \d+|name only, no call path)\s+(\S+)\s+\(\d+ symbol\(s\)\)[^\n]*\n((?:\s+\S+ \(\S+:\d+\)\n(?:\s+(?:\d+|) \| [^\n]*\n)*)+)',
text, re.M)
if not m: return ''
f = m.group(1); syms = [x.strip() for x in m.group(2).split(',') if x.strip()][:2]
return (f"next: read {f} first — it holds {' and '.join(syms)}; then `impact <the one you will change>` for what a change "
"to it reaches. The other files are ranked context, not a reading list")
syms = re.findall(r'^\s+(\S+) \(\S+:\d+\)$', m.group(2), re.M)[:2]
return (f"next: answer from the code of {m.group(1)} shown first above — {' and '.join(syms)}; then `impact <the one you will "
"change>` for what a change to it reaches. The other files are ranked context, not a reading list")

def next_changed(text):
if re.search(r'^(no change|no git base)', text, re.M): return ''
Expand Down
Loading
Loading