diff --git a/CHANGELOG.md b/CHANGELOG.md index f9b62a5919..f408b68fa3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -527,6 +527,8 @@ Intra-class call binding across five languages, new language-structure coverage, - Fix: a PHP `use` import written with a leading-backslash / fully-qualified prefix now resolves to its target definition instead of being dropped (#2661, thanks @ousamabenyounes). - Fix: an unresolved local JS/TS import (to a file absent from the scan) now emits a stable, portable `ref` target id instead of leaking a per-checkout absolute-path slug (#2457, thanks @rohit-jsfreaky). - Fix: `graphify benchmark` no longer crashes on a node whose label is `None` (#2674, thanks @Arthuro0103). +- Feat: `graphify vibe install` adds full-parity Mistral Vibe (mistralai/mistral-vibe) support — skill file with `user-invocable: true` frontmatter (exposes `/graphify` as a native slash command in vibe's autocomplete), `AGENTS.md` always-on section, and two `pre_tool` hooks in `.vibe/hooks.toml` (matchers `grep` and `read_file` — vibe's snake-cased tool names per `vibe/core/tools/base.py:get_name`) that nudge toward `graphify query` for search/read operations. Global scope honors `VIBE_HOME`, hook ownership is `name`-based (never touches user-authored hooks that happen to shell out to graphify), and `hooks.toml` merges preserve hand-authored comments via `tomlkit`. Verified end-to-end against vibe 2.24.0. +- Fix: `graphify vibe install` now also writes a writable `graphify-extract` subagent to `/agents/graphify-extract.toml` and the skill fragment dispatches Task calls with `subagent_type="graphify-extract"` (#2537 review, thanks @MusicalNinjaDad). Vibe's built-in `explore` subagent (`vibe/core/agents/models.py:EXPLORE`) is read-only — its `enabled_tools = ["grep", "read_file", "skill"]` cannot write chunk JSON — so semantic extraction inheriting it silently produced zero `.graphify_chunk_NN.json` files and the pipeline aborted. The new agent enables `write_file` scoped to `**/graphify-out/.graphify_chunk_*.json` only (no unbounded write grant), reuses the `explore` system prompt, and is registered/unregistered by `_install_vibe_extract_agent` / `_uninstall_vibe_extract_agent` symmetric to the hook flow. `uninstall_all` sweeps it alongside the rest. ## 0.9.40 (2026-08-11) diff --git a/README.md b/README.md index 5fa367209b..3986955df4 100644 --- a/README.md +++ b/README.md @@ -244,6 +244,7 @@ for example `graphify claude install --project` or `graphify codex install --pro | Cursor | `graphify cursor install` | | Devin CLI | `graphify devin install` | | Google Antigravity | `graphify antigravity install` | +| Mistral Vibe | `graphify install --platform vibe` | Codex users also need `multi_agent = true` under `[features]` in `~/.codex/config.toml` for parallel extraction. CodeBuddy uses the same Agent tool and PreToolUse hook mechanism as Claude Code. Factory Droid uses the `Task` tool for parallel subagent dispatch. OpenClaw and Aider use sequential extraction (parallel agent support is still early on those platforms). Trae uses the Agent tool for parallel subagent dispatch and does **not** support `PreToolUse` hooks, so AGENTS.md is the always-on mechanism. @@ -316,6 +317,7 @@ Run this once in your project after building a graph: | Pi coding agent | `graphify pi install` | | Devin CLI | `graphify devin install` | | Google Antigravity | `graphify antigravity install` | +| Mistral Vibe | `graphify vibe install` | This writes a small config file that tells your assistant to consult the knowledge graph for codebase questions, preferring scoped queries like `graphify query ""` over reading the full report or grepping raw files. @@ -797,6 +799,10 @@ graphify devin install # skill file + .windsurf/rules/graphify.md (D graphify devin uninstall graphify antigravity install # .agents/rules + .agents/workflows (Google Antigravity) graphify antigravity uninstall +graphify vibe install # ~/.vibe/skills + AGENTS.md + hooks.toml pre_tool guards (Mistral Vibe) +graphify vibe install --project # project-scoped: .vibe/skills + AGENTS.md + .vibe/hooks.toml +graphify vibe install --strict # block first raw read per session (matches claude --strict) +graphify vibe uninstall graphify extract ./docs # headless LLM extraction for CI (no IDE needed) graphify extract ./docs --backend gemini # explicit backend: gemini, kimi, claude, openai, deepseek, ollama, bedrock, or claude-cli diff --git a/graphify/__main__.py b/graphify/__main__.py index 5f5750b885..5111757dd9 100644 --- a/graphify/__main__.py +++ b/graphify/__main__.py @@ -88,6 +88,16 @@ _uninstall_kilo_plugin, _uninstall_opencode_plugin, _vscode_skill_destination, + _uninstall_vibe_hook, + _vibe_install, + _vibe_uninstall, + _install_vibe_hook, + _vibe_hooks_path, + _vibe_hook_entries, + _install_vibe_extract_agent, + _uninstall_vibe_extract_agent, + _vibe_extract_agent_path, + _vibe_agents_dir, claude_install, claude_uninstall, codebuddy_install, diff --git a/graphify/detect.py b/graphify/detect.py index 59ae39b227..0a4a5c62bc 100644 --- a/graphify/detect.py +++ b/graphify/detect.py @@ -952,6 +952,7 @@ def _is_regular_file(path: Path) -> bool: ".graphify", # graphify's own extraction cache — never index self-generated data ".obsidian", ".smart-env", # Obsidian vault metadata and plugin caches (#2493) ".worktrees", # git worktree convention (#947) — sibling checkouts, always redundant + ".vibe", # Mistral Vibe project-scope skill/agent/hook install dir (#2537) } # Large generated files that are never useful to extract diff --git a/graphify/install.py b/graphify/install.py index 36a4484c87..44d01a5b34 100644 --- a/graphify/install.py +++ b/graphify/install.py @@ -19,6 +19,7 @@ import os import platform import re +import shlex import shutil import stat import sys @@ -237,6 +238,11 @@ def _platform_skill_destination(platform_name: str, *, project: bool = False, pr # Global Antigravity skill dir (all workspaces): ~/.gemini/config/skills/ return Path.home() / ".gemini" / "config" / "skills" / "graphify" / "SKILL.md" + if platform_name == "vibe": + if project: + return (project_dir or Path(".")) / ".vibe" / "skills" / "graphify" / "SKILL.md" + return _vibe_home() / "skills" / "graphify" / "SKILL.md" + cfg = _PLATFORM_CONFIG[platform_name] if project: return (project_dir or Path(".")) / cfg["skill_dst"] @@ -679,6 +685,13 @@ def _register_always_on_block(target: Path, prefix: str, registration: str) -> N "skill_dst": Path(".config") / "devin" / "skills" / "graphify" / "SKILL.md", "claude_md": False, }, + "vibe": { + # Mistral Vibe (mistralai/mistral-vibe): reuses the `agents` references bundle. + "skill_file": "skill-vibe.md", + "skill_dst": Path(".vibe") / "skills" / "graphify" / "SKILL.md", + "claude_md": False, + "skill_refs": "agents", + }, } # CLI-only platform aliases, resolved to a real _PLATFORM_CONFIG key before # dispatch. `skills` is the friendly alias for the generic `agents` platform @@ -814,7 +827,7 @@ def _print_banner() -> None: print("\n" + "\n".join(lines) + "\n") except Exception: pass -def install(platform: str = "claude", *, project: bool = False, project_dir: Path | None = None) -> None: +def install(platform: str = "claude", *, project: bool = False, project_dir: Path | None = None, strict: bool = False) -> None: _print_banner() platform = _canonical_platform(platform) if platform == "gemini": @@ -823,12 +836,15 @@ def install(platform: str = "claude", *, project: bool = False, project_dir: Pat if platform == "cursor": _cursor_install(Path(".")) return + if platform == "vibe": + _vibe_install(project_dir, project=project, strict=strict) + return # On Windows, antigravity needs the PowerShell skill, not the bash one if platform == "antigravity" and sys.platform == "win32": platform = "antigravity-windows" if platform not in _PLATFORM_CONFIG: print( - f"error: unknown platform '{platform}'. Choose from: {', '.join(_PLATFORM_CONFIG)}, gemini, cursor", + f"error: unknown platform '{platform}'. Choose from: {', '.join(_PLATFORM_CONFIG)}, gemini, cursor, vibe", file=sys.stderr, ) sys.exit(1) @@ -958,11 +974,11 @@ def gemini_install(project_dir: Path | None = None, *, project: bool = False) -> def _refuse_to_modify(settings_path: Path) -> "NoReturn": """Abort a hook install rather than clobber a config file we can't parse (#2167).""" print( - f"[graphify] refusing to modify {settings_path}: not valid JSON " + f"[graphify] refusing to modify {settings_path}: not valid JSON or TOML " "(fix or move it and re-run)", file=sys.stderr, ) - sys.exit(1) + raise SystemExit(1) def _read_settings_for_merge(settings_path: Path) -> dict: """Load an existing settings/hooks JSON file for a read-modify-write merge. @@ -1727,6 +1743,188 @@ def _uninstall_codex_hook(project_dir: Path) -> None: existing["hooks"]["PreToolUse"] = filtered hooks_path.write_text(json.dumps(existing, indent=2), encoding="utf-8") print(f" .codex/hooks.json -> PreToolUse hook removed") + + + +_VIBE_HOOK_ENTRIES: "tuple[tuple[str, str, tuple[str, ...], str], ...]" = ( + ( + "graphify-nudge-search", + "grep", + ("hook-guard", "search"), + "Nudge toward `graphify query` instead of raw grep.", + ), + ( + "graphify-nudge-read", + "read_file", + ("hook-guard", "read"), + "Nudge toward `graphify query` instead of raw file reads.", + ), +) + + +_VIBE_OWNED_HOOK_NAMES: frozenset[str] = frozenset( + name for name, _, _, _ in _VIBE_HOOK_ENTRIES +) + + +def _vibe_home() -> Path: + """Resolve the vibe home dir, honoring vibe's own `VIBE_HOME` env var.""" + override = os.environ.get("VIBE_HOME") + if override: + return Path(override).expanduser() + return Path.home() / ".vibe" + + +def _vibe_hooks_path(project_dir: Path, *, project: bool) -> Path: + """Return the hooks.toml path vibe reads for the requested scope.""" + if project: + return (project_dir or Path(".")) / ".vibe" / "hooks.toml" + return _vibe_home() / "hooks.toml" + + +def _vibe_agents_md_path(project_dir: Path, *, project: bool) -> Path: + """Return the AGENTS.md path vibe merges for the requested scope.""" + if project: + return (project_dir or Path(".")) / "AGENTS.md" + return _vibe_home() / "AGENTS.md" + + +def _vibe_hook_entries(strict: bool = False) -> "list[dict]": + """Build the two pre_tool entries (grep + read) with shlex-safe commands.""" + exe = _resolve_graphify_exe() + entries: list[dict] = [] + for name, matcher, subcommand_parts, description in _VIBE_HOOK_ENTRIES: + argv: list[str] = [exe, *subcommand_parts] + if subcommand_parts == ("hook-guard", "read") and strict: + argv.append("--strict") + command = shlex.join(argv) + entries.append({ + "name": name, + "type": "pre_tool", + "match": matcher, + "command": command, + "timeout": 5.0, + "strict": False, + "description": description, + }) + return entries + + +def _install_vibe_hook(project_dir: Path, *, project: bool, strict: bool = False) -> None: + """Merge graphify pre_tool entries into vibe's hooks.toml (idempotent, name-owned).""" + try: + import tomlkit + from tomlkit.items import AoT + except ImportError as exc: + raise RuntimeError( + "graphify vibe install requires `tomlkit`. " + "Reinstall graphifyy (e.g. `uv tool install --reinstall graphifyy`)." + ) from exc + + hooks_path = _vibe_hooks_path(project_dir, project=project) + hooks_path.parent.mkdir(parents=True, exist_ok=True) + + if hooks_path.exists(): + try: + doc = tomlkit.parse(hooks_path.read_text(encoding="utf-8-sig")) + except Exception: + _refuse_to_modify(hooks_path) + else: + doc = tomlkit.document() + + existing = doc.get("hooks") + if existing is None: + table_array = tomlkit.aot() + doc["hooks"] = table_array + elif isinstance(existing, AoT): + table_array = existing + else: + _refuse_to_modify(hooks_path) + + kept: list = [] + for h in table_array: + if not hasattr(h, "get"): + _refuse_to_modify(hooks_path) + if h.get("name") in _VIBE_OWNED_HOOK_NAMES: + continue + kept.append(h) + + while len(table_array) > 0: + table_array.pop() + for entry in kept: + table_array.append(entry) + for entry in _vibe_hook_entries(strict=strict): + table = tomlkit.table() + for key, value in entry.items(): + table[key] = value + table_array.append(table) + + output = tomlkit.dumps(doc) + if hooks_path.exists(): + current = hooks_path.read_text(encoding="utf-8") + if _normalize_toml(current) == _normalize_toml(output): + print(f" {hooks_path} -> pre_tool hooks already registered (no change)") + return + backup = hooks_path.with_name(hooks_path.name + ".graphify-bak") + shutil.copy2(hooks_path, backup) + hooks_path.write_text(output, encoding="utf-8") + strict_note = " (strict)" if strict else "" + print(f" {hooks_path} -> pre_tool hooks registered (grep + read){strict_note}") + + +def _normalize_toml(text: str) -> str: + """Collapse tomlkit fresh-vs-parsed whitespace drift for idempotency checks.""" + return "\n".join(line for line in text.splitlines() if line.strip()) + "\n" + + +def _uninstall_vibe_hook(project_dir: Path, *, project: bool) -> None: + """Strip graphify pre_tool entries from vibe's hooks.toml (name-owned).""" + try: + import tomlkit + from tomlkit.items import AoT + except ImportError as exc: + raise RuntimeError( + "graphify vibe uninstall requires `tomlkit`. " + "Reinstall graphifyy (e.g. `uv tool install --reinstall graphifyy`)." + ) from exc + + hooks_path = _vibe_hooks_path(project_dir, project=project) + if not hooks_path.exists(): + return + try: + doc = tomlkit.parse(hooks_path.read_text(encoding="utf-8-sig")) + except Exception: + return + + table_array = doc.get("hooks") + if not isinstance(table_array, AoT): + return + + kept: list = [] + for h in table_array: + if not hasattr(h, "get"): + return + if h.get("name") in _VIBE_OWNED_HOOK_NAMES: + continue + kept.append(h) + + if len(kept) == len(table_array): + return + + while len(table_array) > 0: + table_array.pop() + for entry in kept: + table_array.append(entry) + if len(table_array) == 0: + del doc["hooks"] + + output = tomlkit.dumps(doc) + if not output.strip(): + hooks_path.unlink() + print(f" {hooks_path} -> removed (empty after graphify uninstall)") + else: + hooks_path.write_text(output, encoding="utf-8") + print(f" {hooks_path} -> pre_tool hooks removed") def _agents_install(project_dir: Path, platform: str, project: bool = False) -> None: """Write the graphify section to the local AGENTS.md for always-on platforms.""" target = (project_dir or Path(".")) / "AGENTS.md" @@ -1786,6 +1984,183 @@ def _amp_uninstall(project_dir: Path | None = None) -> None: if removed: print("skill removed") _agents_uninstall(project_dir or Path("."), platform="amp") +def _install_vibe_agents_md(target: Path) -> None: + """Write or refresh the graphify AGENTS.md section at target (idempotent).""" + target.parent.mkdir(parents=True, exist_ok=True) + body = _always_on("agents-md") + new_content = ( + _replace_or_append_section(target.read_text(encoding="utf-8"), _AGENTS_MD_MARKER, body) + if target.exists() else body + ) + if target.exists() and new_content == target.read_text(encoding="utf-8"): + print(f"graphify already configured in {target.resolve()} (no change)") + return + target.write_text(new_content, encoding="utf-8") + print(f"graphify section written to {target.resolve()}") + + +def _uninstall_vibe_agents_md(targets: list[Path]) -> None: + """Strip the graphify section from every AGENTS.md in targets (skips missing).""" + for target in targets: + if not target.exists(): + continue + cleaned = _remove_marker_section(target.read_text(encoding="utf-8"), _AGENTS_MD_MARKER) + if cleaned is None: + continue + if cleaned: + target.write_text(cleaned + "\n", encoding="utf-8") + print(f"graphify section removed from {target.resolve()}") + else: + target.unlink() + print(f"AGENTS.md was empty after removal - deleted {target.resolve()}") + + +_VIBE_EXTRACT_AGENT_NAME = "graphify-extract" + +# TOML body for the writable graphify-extract subagent. Vibe's AgentRegistry +# scans `/agents/*.toml`; the file stem becomes the agent name +# (`AgentProfile.from_toml` sets `name=path.stem`, see mistralai/mistral-vibe +# vibe/core/agents/models.py), so this file MUST be named +# `graphify-extract.toml`. Only the four whitelisted tools are enabled, and +# `write_file` is scoped to graphify's own chunk output so nothing else in the +# workspace is writable. The system prompt id `explore` keeps it lean — vibe's +# read-only subagent uses the same prompt. +_VIBE_EXTRACT_AGENT_TOML = """\ +# graphify-extract: writable subagent used by the graphify skill. +# Managed by `graphify vibe install` -- do not hand-edit; changes are overwritten. +# Named vs. the read-only `explore` builtin because Task-dispatched semantic +# extraction has to write one chunk JSON per invocation, which `explore` +# cannot do (its enabled_tools omit write_file). +display_name = "Graphify Extract" +description = "Writable subagent used by graphify's /graphify skill for semantic chunk extraction. Reads files, writes one graphify-out/.graphify_chunk_NN.json per Task, then exits." +safety = "safe" +agent_type = "subagent" +system_prompt_id = "explore" +enabled_tools = ["grep", "read_file", "write_file", "skill"] + +[tools.write_file] +permission = "always" +allowlist = ["**/graphify-out/.graphify_chunk_*.json"] + +[tools.read_file] +permission = "always" + +[tools.grep] +permission = "always" + +[tools.skill] +permission = "always" +""" + + +def _vibe_agents_dir(project_dir: Path, *, project: bool) -> Path: + """Return the vibe agents dir for the requested scope (mirror hooks.toml layout).""" + if project: + return (project_dir or Path(".")) / ".vibe" / "agents" + return _vibe_home() / "agents" + + +def _vibe_extract_agent_path(project_dir: Path, *, project: bool) -> Path: + """Full path of the graphify-extract.toml agent file for the requested scope.""" + return _vibe_agents_dir(project_dir, project=project) / f"{_VIBE_EXTRACT_AGENT_NAME}.toml" + + +def _install_vibe_extract_agent(project_dir: Path, *, project: bool) -> None: + """Write the writable graphify-extract subagent (idempotent). + + Without this, the graphify skill dispatches Task subagents that inherit + vibe's read-only `explore` profile and cannot write chunk JSON files, so + Part B extraction silently produces zero chunks (see #2537 review). + """ + target = _vibe_extract_agent_path(project_dir, project=project) + target.parent.mkdir(parents=True, exist_ok=True) + body = _VIBE_EXTRACT_AGENT_TOML + if target.exists() and target.read_text(encoding="utf-8") == body: + print(f" {target} -> graphify-extract agent already registered (no change)") + return + if target.exists(): + backup = target.with_name(target.name + ".graphify-bak") + shutil.copy2(target, backup) + target.write_text(body, encoding="utf-8") + print(f" {target} -> graphify-extract subagent registered") + + +def _uninstall_vibe_extract_agent(project_dir: Path, *, project: bool) -> None: + """Remove the graphify-extract.toml agent file (leave user-authored agents alone).""" + target = _vibe_extract_agent_path(project_dir, project=project) + if not target.exists(): + return + target.unlink() + print(f" {target} -> graphify-extract subagent removed") + # Prune the agents dir if empty and we own it (project scope only — the + # global VIBE_HOME/agents may hold unrelated user agents). + if project: + agents_dir = target.parent + try: + if agents_dir.is_dir() and not any(agents_dir.iterdir()): + agents_dir.rmdir() + except OSError: + pass + + +def _print_vibe_install_summary(strict: bool) -> None: + """Post-install summary + optional strict-mode note.""" + print() + print("Mistral Vibe will now check the knowledge graph before answering") + print("codebase questions and rebuild it after code changes.") + print("Use /graphify in vibe to build or update the graph.") + print("Semantic extraction dispatches to the `graphify-extract` subagent") + print("(installed alongside the skill; vibe's read-only `explore` cannot") + print("write chunk files).") + if strict: + print("Strict mode: the first raw file read per session is blocked until") + print("one `graphify query` runs (toggle with GRAPHIFY_HOOK_STRICT=0).") + + +def _vibe_install(project_dir: Path | None = None, *, project: bool = False, strict: bool = False) -> None: + """Full-parity Vibe install: skill + AGENTS.md always-on + hooks.toml + graphify-extract agent.""" + project_dir = project_dir or Path(".") + skill_dst = _copy_skill_file("vibe", project=project, project_dir=project_dir) + _install_vibe_agents_md(_vibe_agents_md_path(project_dir, project=project)) + _install_vibe_hook(project_dir, project=project, strict=strict) + _install_vibe_extract_agent(project_dir, project=project) + + if project: + _print_project_git_add_hint([ + _project_scope_root(skill_dst, project_dir), + project_dir / "AGENTS.md", + project_dir / ".vibe", + ]) + else: + # v8 stamps per-platform (#2694), not globally. _copy_skill_file already + # wrote the stamp for the vibe skill; nothing more to refresh here. + pass + + _print_vibe_install_summary(strict) + + +def _vibe_uninstall(project_dir: Path | None = None, *, project: bool = False, remove_user_skill: bool | None = None) -> None: + """Reverse of `_vibe_install`; scope rules mirror `gemini_uninstall` (#2215).""" + explicit_dir = project_dir is not None + project_dir = project_dir or Path(".") + if remove_user_skill is None: + remove_user_skill = not project and not explicit_dir + + agents_md_targets: list[Path] = [] + if project or (explicit_dir and not remove_user_skill): + _remove_skill_file("vibe", project=True, project_dir=project_dir) + _uninstall_vibe_hook(project_dir, project=True) + _uninstall_vibe_extract_agent(project_dir, project=True) + agents_md_targets.append(_vibe_agents_md_path(project_dir, project=True)) + if remove_user_skill: + _remove_skill_file("vibe", project=False) + _uninstall_vibe_hook(project_dir, project=False) + _uninstall_vibe_extract_agent(project_dir, project=False) + agents_md_targets.append(_vibe_agents_md_path(project_dir, project=False)) + + _uninstall_vibe_agents_md(agents_md_targets) + + def _agents_platform_install(project_dir: Path | None = None) -> None: """`graphify agents install`: skill into ~/.agents/skills + AGENTS.md. @@ -1829,6 +2204,8 @@ def _project_install(platform_name: str, project_dir: Path | None = None, strict elif platform_name == "codex": hint_paths.append(project_dir / ".codex") _print_project_git_add_hint(hint_paths) + elif platform_name == "vibe": + _vibe_install(project_dir, project=True, strict=strict) elif platform_name == "devin": skill_dst = _copy_skill_file("devin", project=True, project_dir=project_dir) _devin_rules_install(project_dir) @@ -1866,6 +2243,8 @@ def _project_uninstall(platform_name: str, project_dir: Path | None = None) -> N _agents_uninstall(project_dir, platform=platform_name) if platform_name == "codex": _uninstall_codex_hook(project_dir) + elif platform_name == "vibe": + _vibe_uninstall(project_dir, project=True) elif platform_name == "antigravity": _antigravity_uninstall(project_dir, project=True) elif platform_name == "devin": @@ -2052,6 +2431,7 @@ def uninstall_all(project_dir: Path | None = None, purge: bool = False) -> None: claude_uninstall(pd, remove_user_skill=True) codebuddy_uninstall(pd, remove_user_skill=True) gemini_uninstall(pd, remove_user_skill=True) + _vibe_uninstall(pd, project=True, remove_user_skill=True) vscode_uninstall(pd) _cursor_uninstall(pd) _kiro_uninstall(pd) @@ -2277,6 +2657,7 @@ def codebuddy_uninstall(project_dir: Path | None = None, *, project: bool = Fals "trae", "trae-cn", "uninstall", + "vibe", "vscode", }) @@ -2338,13 +2719,13 @@ def dispatch_install_cli(cmd: str) -> bool: if project_scope: _project_install(chosen_platform, Path("."), strict=strict) else: - if strict: + if strict and _canonical_platform(chosen_platform) != "vibe": print( "note: --strict applies to the project PreToolUse hook; run " "`graphify install --project --strict` or `graphify claude install --strict`.", file=sys.stderr, ) - install(platform=chosen_platform) + install(platform=chosen_platform, strict=strict) elif cmd == "uninstall": args = sys.argv[2:] purge = "--purge" in args @@ -2556,4 +2937,18 @@ def dispatch_install_cli(cmd: str) -> bool: else: print("Usage: graphify antigravity [install|uninstall]", file=sys.stderr) sys.exit(1) + elif cmd == "vibe": + subcmd = sys.argv[2] if len(sys.argv) > 2 else "" + project_scope = "--project" in sys.argv[3:] + strict = "--strict" in sys.argv[3:] + if subcmd == "install": + _vibe_install(Path("."), project=project_scope, strict=strict) + elif subcmd == "uninstall": + if project_scope: + _vibe_uninstall(Path("."), project=True) + else: + _vibe_uninstall() + else: + print("Usage: graphify vibe [install|uninstall] [--project] [--strict]", file=sys.stderr) + sys.exit(1) return True diff --git a/graphify/skill-vibe.md b/graphify/skill-vibe.md new file mode 100644 index 0000000000..41e754a47c --- /dev/null +++ b/graphify/skill-vibe.md @@ -0,0 +1,710 @@ +--- +name: graphify +description: "Use for any question about a codebase, its architecture, file relationships, or project content — especially when graphify-out/ exists, where the question should be treated as a graphify query first. Turns any input (code, docs, papers, images, videos) into a persistent knowledge graph with god nodes, community detection, and query/path/explain tools." +user-invocable: true +allowed-tools: + - bash + - read_file + - grep + - write_file + - edit + - task + - ask_user_question +--- + + +# /graphify + +Turn any folder of files into a navigable knowledge graph with community detection, an honest audit trail, and three outputs: interactive HTML, GraphRAG-ready JSON, and a plain-language GRAPH_REPORT.md. + +## Usage + +``` +/graphify # full pipeline on current directory (HTML viz; add --obsidian for a vault) +/graphify # full pipeline on specific path +/graphify https://github.com// # clone repo then run full pipeline on it +/graphify https://github.com// --branch # clone a specific branch +/graphify ... # clone multiple repos, build each, merge into one cross-repo graph +/graphify --mode deep # thorough extraction, richer INFERRED edges +/graphify --update # incremental - re-extract only new/changed files +/graphify --directed # build directed graph (preserves edge direction: source→target) +/graphify --whisper-model medium # use a larger Whisper model for better transcription accuracy +/graphify --cluster-only # rerun clustering on existing graph +/graphify --no-viz # skip visualization, just report + JSON +/graphify --html # (HTML is generated by default - this flag is a no-op) +/graphify --svg # also export graph.svg (embeds in Notion, GitHub) +/graphify --graphml # export graph.graphml (Gephi, yEd) +/graphify --neo4j # generate graphify-out/cypher.txt for Neo4j +/graphify --neo4j-push bolt://localhost:7687 # push directly to Neo4j +/graphify --falkordb # generate graphify-out/cypher.txt for FalkorDB +/graphify --falkordb-push falkordb://localhost:6379 # push directly to FalkorDB +/graphify --mcp # start MCP stdio server for agent access +/graphify --watch # watch folder, auto-rebuild on code changes (no LLM needed) +/graphify --wiki # build agent-crawlable wiki (index.md + one article per community) +/graphify --obsidian --obsidian-dir ~/vaults/my-project # write vault to custom path (e.g. existing vault) +/graphify add # fetch URL, save to ./raw, update graph +/graphify add --author "Name" # tag who wrote it +/graphify add --contributor "Name" # tag who added it to the corpus +/graphify query "" # BFS traversal - broad context +/graphify query "" --dfs # DFS - trace a specific path +/graphify query "" --budget 1500 # cap answer at N tokens +/graphify path "AuthModule" "Database" # shortest path between two concepts +/graphify explain "SwinTransformer" # plain-language explanation of a node +``` + +## What graphify is for + +Drop any folder of code, docs, papers, images, or video into graphify and get a queryable knowledge graph. Persistent across sessions, honest audit trail (EXTRACTED/INFERRED/AMBIGUOUS), community detection surfaces cross-document connections you wouldn't think to ask about. + +## What You Must Do When Invoked + +If the user invoked `/graphify --help` or `/graphify -h` (with no other arguments), print the contents of the `## Usage` section above verbatim and stop. Do not run any commands, do not detect files, do not default the path to `.`. Just print the Usage block and return. + +**Fast path — existing graph:** Before doing anything else, check whether `graphify-out/graph.json` exists. The expected location is `graphify-out/graph.json` relative to the **current working directory** (i.e. the project root where you are running commands). If it exists AND the user's request is a natural-language question about the codebase (e.g. "How does X work?", "What calls Y?", "Trace the data flow through Z") and NOT an explicit rebuild command (`--update`, `--cluster-only`, or a bare path/URL that implies fresh extraction): **skip Steps 1–5 entirely and jump straight to `## For /graphify query`.** Run `graphify query ""` immediately. Do not run detect. Do not check corpus size. Do not ask the user to narrow. The graph is already built — use it. + +If no path was given, use `.` (current directory). Do not ask the user for a path. + +If the path argument starts with `https://github.com/` or `http://github.com/`, treat it as a GitHub URL - run Step 0 before anything else, then continue with the resolved local path. + +Follow these steps in order. Do not skip steps. + +### Step 0 - GitHub repos and multi-path merge (only if a URL or several paths) + +Only when the path is one or more `https://github.com/...` URLs, or several local subfolders to merge. See `references/github-and-merge.md` for the clone, cross-repo merge, and monorepo flow, then continue with the resolved local path. A plain local path skips this step. + +### Step 1 - Ensure graphify is installed + +```bash +# Detect the correct Python interpreter (handles uv tool, pipx, venv, system installs) +PYTHON="" +GRAPHIFY_BIN=$(which graphify 2>/dev/null) +# 1. uv tool installs — most reliable on modern Mac/Linux +if [ -z "$PYTHON" ] && command -v uv >/dev/null 2>&1; then + _UV_PY=$(uv tool run --from graphifyy python -c "import sys; print(sys.executable)" 2>/dev/null) + if [ -n "$_UV_PY" ]; then PYTHON="$_UV_PY"; fi +fi +# 2. Read shebang from graphify binary (pipx and direct pip installs) +if [ -z "$PYTHON" ] && [ -n "$GRAPHIFY_BIN" ]; then + _SHEBANG=$(head -1 "$GRAPHIFY_BIN" | tr -d '#!') + case "$_SHEBANG" in + *[!a-zA-Z0-9/_.@-]*) ;; + *) "$_SHEBANG" -c "import graphify" 2>/dev/null && PYTHON="$_SHEBANG" ;; + esac +fi +# 3. Fall back to python3 +if [ -z "$PYTHON" ]; then PYTHON="python3"; fi +if ! "$PYTHON" -c "import graphify" 2>/dev/null; then + if command -v uv >/dev/null 2>&1; then + uv tool install --upgrade graphifyy -q 2>&1 | tail -3 + _UV_PY=$(uv tool run --from graphifyy python -c "import sys; print(sys.executable)" 2>/dev/null) + if [ -n "$_UV_PY" ]; then PYTHON="$_UV_PY"; fi + else + "$PYTHON" -m pip install graphifyy -q 2>/dev/null \ + || "$PYTHON" -m pip install graphifyy -q --break-system-packages 2>&1 | tail -3 + fi +fi +# Write interpreter path for all subsequent steps (persists across invocations) +mkdir -p graphify-out +"$PYTHON" -c "import sys; open('graphify-out/.graphify_python', 'w', encoding='utf-8').write(sys.executable)" +# Save scan root so `graphify update` (no args) knows where to look next time +echo "$(cd INPUT_PATH && pwd)" > graphify-out/.graphify_root +``` + +If the import succeeds, print nothing and move straight to Step 2. + +**In every subsequent bash block, replace `python3` with `$(cat graphify-out/.graphify_python)` to use the correct interpreter.** + +### Step 2 - Detect files + +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from graphify.detect import detect +from pathlib import Path +result = detect(Path('INPUT_PATH')) +print(json.dumps(result, ensure_ascii=False)) +" > graphify-out/.graphify_detect.json +``` + +Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: + +``` +Corpus: X files · ~Y words + code: N files (.py .ts .go ...) + docs: N files (.md .txt ...) + papers: N files (.pdf ...) + images: N files + video: N files (.mp4 .mp3 ...) +``` + +Omit any category with 0 files from the summary. + +Then act on it: +- If `total_files` is 0: stop with "No supported files found in [path]." +- If `skipped_sensitive` is non-empty: report the count and list the skipped file names, so a wrongly-flagged source or doc is visible and can be renamed or moved (#2106). +- If `total_words` > 2,000,000 OR `total_files` > 500: show the warning. Then compute the top 5 first-level subdirectories by file count: + - Read `scan_root` from the detect JSON (always an absolute path to the resolved INPUT_PATH). + - Concatenate all file lists across all types (`code`, `document`, `paper`, `image`, `video`). + - Filter out any path that starts with `scan_root + "/graphify-out/"` to exclude converted sidecars. + - For each file, strip the `scan_root` prefix and take the first path component. Files directly in `scan_root` with no subdirectory count as `(root)`. + - If all files are in `(root)` with no subdirectories, do not ask to narrow — no subfolders exist. Instead suggest `--no-cluster` to skip the expensive clustering step and proceed. + - Otherwise rank by count, show the top 5 with file counts, then ask which subfolder to run on. Wait for the user's answer before proceeding. +- Otherwise: proceed directly to Step 2.5 if video files were detected, or Step 3 if not. + +### Step 2.5 - Video and audio (only if video files detected) + +Skip this step entirely if `detect` returned zero `video` files. When the corpus has video or audio, see `references/transcribe.md` to transcribe them to text first, then treat the transcripts as doc files in Step 3. + +### Step 3 - Extract entities and relationships + +**Before starting:** note whether `--mode deep` was given. You must pass `DEEP_MODE=true` to every subagent in Step B2 if it was. Track this from the original invocation - do not lose it. + +This step has two parts: **structural extraction** (deterministic, free) and **semantic extraction** (LLM, costs tokens). + +> **graphify needs no API key. Never ask the user for one, and never block on one.** Code is extracted structurally (AST) with no LLM and no key at all — a code-only corpus (the common `/graphify .` on a repo) skips semantic extraction entirely, so it needs nothing here: go straight to Part A and skip Part B. Semantic extraction (only for docs, papers, and images) uses Gemini **only if** `GEMINI_API_KEY`/`GOOGLE_API_KEY` is already set; otherwise the host agent itself is the LLM. graphify does **not** read `ANTHROPIC_API_KEY`, `OPENAI_API_KEY`, or any other provider key. If you catch yourself about to prompt for, wait on, or stop because of a missing API key, that is a misread of this skill — proceed without one. + +**Before semantic extraction:** check whether `GEMINI_API_KEY` or `GOOGLE_API_KEY` is set. If neither is set, print this one-liner to the user: +> Tip: set `GEMINI_API_KEY` or `GOOGLE_API_KEY` to use Gemini for semantic extraction (`pip install 'graphifyy[gemini]'`). + +Print it once, then continue — do not wait for the user to supply a key. If `GEMINI_API_KEY` or `GOOGLE_API_KEY` IS set, use `graphify.llm.extract_corpus_parallel(files, backend="gemini")` for semantic extraction instead of dispatching subagents. The default Gemini model is `gemini-3-flash-preview`; set `GRAPHIFY_GEMINI_MODEL` or pass `--model` in headless CLI flows to override it. + +> **No other API keys are read.** When `GEMINI_API_KEY`/`GOOGLE_API_KEY` are unset, semantic extraction falls to the host agent itself — the running session is the LLM. On a host that dispatches subagents (e.g. Claude Code), dispatch them as written in Part B. On a host that runs the CLI directly in a terminal and cannot dispatch subagents, do not stall: a code-only corpus has no semantic work, so write the empty semantic file (Part B "Fast path") and continue to Part C; for a corpus with docs/papers/images, either set a Gemini key or extract those inline yourself, but in no case prompt for `ANTHROPIC_API_KEY` — that prompt is a misread of this skill. + +**Run Part A (AST) and Part B (semantic) in parallel. Dispatch all semantic subagents AND start AST extraction in the same message. Both can run simultaneously since they operate on different file types. Merge results in Part C as before.** + +Note: Parallelizing AST + semantic saves 5-15s on large corpora. AST is deterministic and fast; start it while subagents are processing docs/papers. + +#### Part A - Structural extraction for code files + +For any code files detected, run AST extraction in parallel with Part B subagents: + +```bash +$(cat graphify-out/.graphify_python) -c " +import sys, json +from graphify.extract import collect_files, extract +from pathlib import Path +import json + +code_files = [] +detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) +for f in detect.get('files', {}).get('code', []): + code_files.extend(collect_files(Path(f)) if Path(f).is_dir() else [Path(f)]) + +if code_files: + result = extract(code_files, cache_root=Path('INPUT_PATH')) + Path('graphify-out/.graphify_ast.json').write_text(json.dumps(result, indent=2, ensure_ascii=False), encoding=\"utf-8\") + print(f'AST: {len(result[\"nodes\"])} nodes, {len(result[\"edges\"])} edges') +else: + Path('graphify-out/.graphify_ast.json').write_text(json.dumps({'nodes':[],'edges':[],'input_tokens':0,'output_tokens':0}, ensure_ascii=False), encoding=\"utf-8\") + print('No code files - skipping AST extraction') +" +``` + +#### Part B - Semantic extraction (parallel subagents) + +**Fast path:** If detection found zero docs, papers, and images (code-only corpus), skip Part B entirely and go straight to Part C. AST handles code - there is nothing for semantic subagents to do. **First write an empty semantic file** so Part C's merge has its input (it reads `.graphify_semantic.json` unconditionally; without this a code-only run hits `FileNotFoundError`): + +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path +Path('graphify-out/.graphify_semantic.json').write_text(json.dumps({'nodes':[],'edges':[],'hyperedges':[],'input_tokens':0,'output_tokens':0}), encoding='utf-8') +" +``` + +**MANDATORY: You MUST use the Agent tool here. Reading files yourself one-by-one is forbidden - it is 5-10x slower. If you do not use the Agent tool you are doing this wrong.** + +Before dispatching subagents, print a timing estimate: +- Load `total_words` and file counts from `graphify-out/.graphify_detect.json` +- Estimate agents needed: `ceil(uncached_non_code_files / 22)` (chunk size is 20-25) +- Estimate time: ~45s per agent batch (they run in parallel, so total ≈ 45s × ceil(agents/parallel_limit)) +- Print: "Semantic extraction: ~N files → X agents, estimated ~Ys" + +**Step B0 - Check extraction cache first** + +Before dispatching any subagents, check which files already have cached extraction results: + +SPEC_PATH below is the **absolute** path of the `references/extraction-spec.md` that ships beside this SKILL.md — the same file Step B2 loads and hands to every subagent. It is the extraction prompt, so cache entries are attributed to it: when a graphify upgrade changes the prompt, entries produced by the old one are re-extracted instead of replayed, and unchanged prompts keep their entries (#1939). Substitute the real path in both Step B0 and Step B3 — pass the same one to each, and do not drop the argument. + +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from graphify.cache import check_semantic_cache +from pathlib import Path + +detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) +# Only content files go to semantic extraction. Code is already covered structurally +# by the AST pass (Part A); flattening every category here makes subagents re-read +# every source file (#1392). Video is transcribed to a document in Step 2.5 first. +all_files = [f for cat in ('document', 'paper', 'image') for f in detect['files'].get(cat, [])] + +cached_nodes, cached_edges, cached_hyperedges, uncached = check_semantic_cache(all_files, root='INPUT_PATH', prompt_file='SPEC_PATH') + +# Always (re)write the cache file: write hits, else DELETE any leftover from a prior +# run so Part C never merges a stale .graphify_cached.json (#1392). +if cached_nodes or cached_edges or cached_hyperedges: + Path('graphify-out/.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges, 'hyperedges': cached_hyperedges}, ensure_ascii=False), encoding=\"utf-8\") +else: + Path('graphify-out/.graphify_cached.json').unlink(missing_ok=True) +Path('graphify-out/.graphify_uncached.txt').write_text('\n'.join(uncached), encoding=\"utf-8\") +print(f'Cache: {len(all_files)-len(uncached)} files hit, {len(uncached)} files need extraction') +" +``` + +Only dispatch subagents for files listed in `graphify-out/.graphify_uncached.txt`. If all files are cached, skip to Part C directly. + +**Step B1 - Split into chunks** + +Load files from `graphify-out/.graphify_uncached.txt`. Split into chunks of 20-25 files each. Each image gets its own chunk (vision needs separate context). When splitting, group files from the same directory together so related artifacts land in the same chunk and cross-file relationships are more likely to be extracted. + +**Step B2 - Dispatch ALL subagents in a single message** + +> Uses the `Task` tool for parallel subagent dispatch. +> Call `Task` once per chunk — ALL in the same response so they run in parallel. +> **On Mistral Vibe, dispatch each `Task` with `subagent_type="graphify-extract"`.** This writable subagent is installed by `graphify vibe install` alongside the skill. Do NOT use vibe's built-in `explore` subagent — it is read-only (`enabled_tools = ["grep", "read_file", "skill"]`) and cannot write the chunk JSON files this pipeline needs. If `graphify-extract` is missing (older install, or you overrode `VIBE_HOME` after install), re-run `graphify vibe install` first. + +Pass the extraction prompt as the task description: + +``` +Task(subagent_type="graphify-extract", description="Your task is to perform the following. Follow the instructions below exactly.\n\n\n[extraction prompt, with FILE_LIST, CHUNK_NUM, TOTAL_CHUNKS, DEEP_MODE substituted]\n\n\nExecute this now. Output ONLY the structured JSON response.") +``` + +Each subagent writes its result to its own `graphify-out/.graphify_chunk_NN.json`. Collect results as each `Task` completes and parse each as JSON. + +CHUNK_PATH must be an **absolute** path — derive it before dispatching: +```bash +PROJECT_ROOT=$(pwd) # cwd — where Part C globs graphify-out/ (NOT .graphify_root/scan dir, #1392) +# Then for chunk N: CHUNK_PATH="${PROJECT_ROOT}/graphify-out/.graphify_chunk_0N.json" +``` + +Subagent prompt template: + +See `references/extraction-spec.md` for the exact subagent prompt (JSON schema, node-ID rules, confidence rubric, hyperedge, and vision rules). Load it only here, only when at least one chunk holds a doc, paper, or image; a pure-code corpus has skipped Part B and never reads it. Pass each subagent that prompt verbatim with FILE_LIST, CHUNK_NUM, TOTAL_CHUNKS, DEEP_MODE, and CHUNK_PATH substituted, and have it write the result to CHUNK_PATH. + +**Step B3 - Collect, cache, and merge** + +Wait for all subagents. For each result: +- Check that `graphify-out/.graphify_chunk_NN.json` exists on disk — this is the success signal +- If the file exists and contains valid JSON with `nodes` and `edges`, include it and save to cache +- If the file is missing, the subagent was likely dispatched as read-only (vibe's built-in `explore`, or Claude Code's Explore type) — print a warning: "chunk N missing from disk — subagent may have been read-only. On vibe, re-dispatch with `subagent_type=\"graphify-extract\"` (installed by `graphify vibe install`). On Claude Code, re-dispatch with the general-purpose agent." Do not silently skip. +- If a subagent failed or returned invalid JSON, print a warning and skip that chunk - do not abort + +If more than half the chunks failed or are missing, stop and tell the user to re-run. On vibe, ensure `subagent_type="graphify-extract"` (installed by `graphify vibe install`) is passed to every `Task` call — the default `explore` subagent is read-only and will silently produce zero chunks. On Claude Code, ensure `subagent_type="general-purpose"` is used. + +Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent call completes, read the real token counts from the Agent tool result's `usage` field and write them back into the chunk JSON before merging** — the chunk JSON itself always has placeholder zeros. Then run: +```bash +$(cat graphify-out/.graphify_python) -c " +import json, glob +from pathlib import Path + +chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +all_nodes, all_edges, all_hyperedges = [], [], [] +total_in, total_out = 0, 0 +for c in chunks: + d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + all_nodes += d.get('nodes', []) + all_edges += d.get('edges', []) + all_hyperedges += d.get('hyperedges', []) + total_in += d.get('input_tokens', 0) + total_out += d.get('output_tokens', 0) +Path('graphify-out/.graphify_semantic_new.json').write_text(json.dumps({ + 'nodes': all_nodes, 'edges': all_edges, 'hyperedges': all_hyperedges, + 'input_tokens': total_in, 'output_tokens': total_out, +}, indent=2, ensure_ascii=False), encoding=\"utf-8\") +print(f'Merged {len(chunks)} chunks: {total_in:,} in / {total_out:,} out tokens') +" +``` + +Save new results to cache. Pass the same SPEC_PATH as Step B0 — it stamps each entry with the prompt that produced it, and a write under a different prompt than the read lands where the next run won't look (#1939): +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from graphify.cache import save_semantic_cache +from pathlib import Path + +new = json.loads(Path('graphify-out/.graphify_semantic_new.json').read_text(encoding=\"utf-8\")) if Path('graphify-out/.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] +saved = save_semantic_cache(new.get('nodes', []), new.get('edges', []), new.get('hyperedges', []), root='INPUT_PATH', allowed_source_files=uncached, prompt_file='SPEC_PATH') +print(f'Cached {saved} files') +" +``` + +Merge cached + new results into `graphify-out/.graphify_semantic.json`: +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path + +cached = json.loads(Path('graphify-out/.graphify_cached.json').read_text(encoding=\"utf-8\")) if Path('graphify-out/.graphify_cached.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} +new = json.loads(Path('graphify-out/.graphify_semantic_new.json').read_text(encoding=\"utf-8\")) if Path('graphify-out/.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} + +all_nodes = cached['nodes'] + new.get('nodes', []) +all_edges = cached['edges'] + new.get('edges', []) +all_hyperedges = cached.get('hyperedges', []) + new.get('hyperedges', []) +seen = set() +deduped = [] +for n in all_nodes: + if n['id'] not in seen: + seen.add(n['id']) + deduped.append(n) + +merged = { + 'nodes': deduped, + 'edges': all_edges, + 'hyperedges': all_hyperedges, + 'input_tokens': new.get('input_tokens', 0), + 'output_tokens': new.get('output_tokens', 0), +} +Path('graphify-out/.graphify_semantic.json').write_text(json.dumps(merged, indent=2, ensure_ascii=False), encoding=\"utf-8\") +print(f'Extraction complete - {len(deduped)} nodes, {len(all_edges)} edges ({len(cached[\"nodes\"])} from cache, {len(new.get(\"nodes\",[]))} new)') +" +``` +Clean up temp files: `rm -f graphify-out/.graphify_cached.json graphify-out/.graphify_uncached.txt graphify-out/.graphify_semantic_new.json` + +#### Part C - Merge AST + semantic into final extraction + +```bash +$(cat graphify-out/.graphify_python) -c " +import sys, json +from pathlib import Path + +ast = json.loads(Path('graphify-out/.graphify_ast.json').read_text(encoding=\"utf-8\")) +sem = json.loads(Path('graphify-out/.graphify_semantic.json').read_text(encoding=\"utf-8\")) + +# Merge: AST nodes first, semantic nodes deduplicated by id +seen = {n['id'] for n in ast['nodes']} +merged_nodes = list(ast['nodes']) +for n in sem['nodes']: + if n['id'] not in seen: + merged_nodes.append(n) + seen.add(n['id']) + +merged_edges = ast['edges'] + sem['edges'] +merged_hyperedges = sem.get('hyperedges', []) +merged = { + 'nodes': merged_nodes, + 'edges': merged_edges, + 'hyperedges': merged_hyperedges, + 'input_tokens': sem.get('input_tokens', 0), + 'output_tokens': sem.get('output_tokens', 0), +} +Path('graphify-out/.graphify_extract.json').write_text(json.dumps(merged, indent=2, ensure_ascii=False), encoding=\"utf-8\") +total = len(merged_nodes) +edges = len(merged_edges) +print(f'Merged: {total} nodes, {edges} edges ({len(ast[\"nodes\"])} AST + {len(sem[\"nodes\"])} semantic)') +" +``` + +### Step 4 - Build graph, cluster, analyze, generate outputs + +**Before starting:** the code blocks below pass `directed=IS_DIRECTED` to `build_from_json()`. Replace `IS_DIRECTED` with `True` if `--directed` was given (builds a `DiGraph` preserving edge direction source→target), otherwise `False` (the default undirected `Graph`). Substitute it the same way you substitute `INPUT_PATH` — do not leave the literal `IS_DIRECTED` in the code. + +```bash +mkdir -p graphify-out +$(cat graphify-out/.graphify_python) -c " +import sys, json +from graphify.build import build_from_json +from graphify.cluster import cluster, score_all +from graphify.analyze import god_nodes, surprising_connections, suggest_questions +from graphify.report import generate +from graphify.export import to_json +from pathlib import Path + +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) +detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) + +# root= mirrors the --update runbook (#1361): relativize source_file to the same +# base so the full build and incremental --update never drift apart on re-extract. +G = build_from_json(extraction, root='INPUT_PATH', directed=IS_DIRECTED) +# Guard BEFORE any write: an empty extraction must not clobber a good graph.json / +# GRAPH_REPORT.md / analysis sidecar. Check immediately after build (#1392). +if G.number_of_nodes() == 0: + print('ERROR: Graph is empty - extraction produced no nodes.') + print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.') + raise SystemExit(1) +communities = cluster(G) +cohesion = score_all(G, communities) +tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)} +gods = god_nodes(G) +surprises = surprising_connections(G, communities) +labels = {cid: 'Community ' + str(cid) for cid in communities} +# Placeholder questions - regenerated with real labels in Step 5 +questions = suggest_questions(G, communities, labels) + +# Export FIRST and honor the #479 shrink-guard: to_json returns False (writing +# nothing) when the new graph is smaller than the existing graph.json. Only write +# GRAPH_REPORT.md + the analysis sidecar when the graph was actually written, so +# they never describe a graph that graph.json doesn't contain (#1392). +wrote = to_json(G, communities, 'graphify-out/graph.json') +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') + raise SystemExit(1) +report = generate(G, communities, cohesion, labels, gods, surprises, detection, tokens, 'INPUT_PATH', suggested_questions=questions) +Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") +analysis = { + 'communities': {str(k): v for k, v in communities.items()}, + 'cohesion': {str(k): v for k, v in cohesion.items()}, + 'gods': gods, + 'surprises': surprises, + 'questions': questions, +} +Path('graphify-out/.graphify_analysis.json').write_text(json.dumps(analysis, indent=2, ensure_ascii=False), encoding=\"utf-8\") +print(f'Graph: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges, {len(communities)} communities') +" +``` + +If this step prints `ERROR: Graph is empty`, stop and tell the user what happened - do not proceed to labeling or visualization. + +Replace INPUT_PATH with the actual path. + +### Step 4.5 - Graph health check (read-only integrity gate) + +A non-destructive diagnostic on the extraction, before labeling. It surfaces edge collapse, dangling/missing endpoints, and self-loops — the silent-corruption modes of incremental updates and AST/LLM id mismatches. Read-only; never aborts. + +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path +from graphify.diagnostics import diagnose_extraction, format_diagnostic_report + +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) +summary = diagnose_extraction(extraction, directed=IS_DIRECTED, root='INPUT_PATH') +print(format_diagnostic_report(summary)) +flags = [f'{summary[k]} {label}' for k, label in ( + ('dangling_endpoint_edges', 'dangling-endpoint edges'), + ('missing_endpoint_edges', 'missing-endpoint edges'), + ('self_loop_edges', 'self-loop edges'), + ('directed_same_endpoint_collapsed_edges', 'collapsed (directed) edges'), + ('undirected_same_endpoint_collapsed_edges', 'collapsed (undirected) edges'), +) if summary.get(k, 0)] +print('GRAPH HEALTH WARNING: ' + '; '.join(flags) + ' - graph may be incomplete/corrupt.' if flags else 'Graph health: OK (no dangling/missing/collapsed edges).') +" +``` + +Substitute `IS_DIRECTED` and `INPUT_PATH` as in Step 4. If a `GRAPH HEALTH WARNING` prints, surface it in the final summary (do not abort — the graph is still usable, but the integrity issue must be visible, per the Honesty Rules). + +### Step 5 - Label communities + +Read `graphify-out/.graphify_analysis.json`. For each community key, look at its node labels and write a 2-5 word plain-language name (e.g. "Attention Mechanism", "Training Pipeline", "Data Loading"). + +Then regenerate the report and save the labels for the visualizer: + +```bash +$(cat graphify-out/.graphify_python) -c " +import sys, json +from graphify.build import build_from_json +from graphify.cluster import score_all +from graphify.analyze import god_nodes, surprising_connections, suggest_questions +from graphify.report import generate +from pathlib import Path + +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) +detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) +analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text(encoding=\"utf-8\")) + +# root= as in Step 4 / the --update runbook (#1361) — same base for node-key parity. +G = build_from_json(extraction, root='INPUT_PATH', directed=IS_DIRECTED) +communities = {int(k): v for k, v in analysis['communities'].items()} +cohesion = {int(k): v for k, v in analysis['cohesion'].items()} +tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)} + +# LABELS - replace these with the names you chose above +labels = LABELS_DICT + +# Regenerate questions with real community labels (labels affect question phrasing) +questions = suggest_questions(G, communities, labels) + +report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) +Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") +Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +print('Report updated with community labels') +" +``` + +Replace `LABELS_DICT` with the actual dict you constructed (e.g. `{0: "Attention Mechanism", 1: "Training Pipeline"}`). +Replace INPUT_PATH with the actual path. + +### Step 6 - Generate Obsidian vault (opt-in) + HTML + +**Generate HTML always** (unless `--no-viz`). **Obsidian vault only if `--obsidian` was explicitly given** — skip it otherwise, it generates one file per node. + +If `--obsidian` was given: + +- If `--obsidian-dir ` was also given, pass it via `--dir`. Otherwise defaults to `graphify-out/obsidian`. + +```bash +graphify export obsidian +# or with custom dir: graphify export obsidian --dir ~/vaults/my-project +``` + +Generate the HTML graph (always, unless `--no-viz`): + +```bash +graphify export html # auto-aggregates to community view if graph > 5000 nodes +# or: graphify export html --no-viz +``` + +### Steps 6b-8 - Wiki, Neo4j, FalkorDB, SVG, GraphML, MCP, benchmark (only on their flags) + +These run only when their flag is present (`--wiki`, `--neo4j`/`--neo4j-push`, `--falkordb`/`--falkordb-push`, `--svg`, `--graphml`, `--mcp`) or, for the token-reduction benchmark, when `total_words` exceeds 5,000. A default run with no export flags skips all of them. See `references/exports.md` for each one. Run any `--wiki` export before Step 9 cleanup so `.graphify_labels.json` is still available. + +--- + +### Step 9 - Save manifest, update cost tracker, clean up, and report + +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path +from datetime import datetime, timezone +from graphify.detect import save_manifest + +# Save manifest for --update +detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) +extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) +# In --update mode, 'all_files' carries the full corpus; 'files' is the changed +# subset. Full-rebuild mode populates only 'files', so the fallback handles that. +# root= relativizes the manifest keys to the scan root (same base as the build), +# so the on-disk manifest is portable across clones/machines and a later --update +# matches cached files instead of missing every one (#1417). +# +# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output: +# a detected file whose chunk failed or was omitted must stay unstamped so the +# next --update re-queues it, otherwise it is marked done and its content is lost +# forever (#2015). This mirrors the library extract path exactly +# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the +# raw corpus. Code files are always stamped (AST is deterministic); only semantic +# types are gated on output. +from graphify.cli import _stamped_manifest_files +_corpus = detect.get('all_files') or detect['files'] +_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH')) +# Files dispatched this run (the changed subset) but NOT stamped above still carry +# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues +# them instead of reading them as unchanged (#1948). +_sem_types = ('document', 'paper', 'image') +_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl} +_stamped = {f for fl in _manifest_files.values() for f in fl} +_cleared = _dispatched - _stamped +# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root +# files newly excluded since last run are dropped rather than masquerading as +# deletions; untouched files' prior rows are still preserved (#1908). +_scan = {f for fl in _corpus.values() for f in fl} +save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None) + +# Update cumulative cost tracker +input_tok = extract.get('input_tokens', 0) +output_tok = extract.get('output_tokens', 0) + +cost_path = Path('graphify-out/cost.json') +if cost_path.exists(): + cost = json.loads(cost_path.read_text(encoding=\"utf-8\")) +else: + cost = {'runs': [], 'total_input_tokens': 0, 'total_output_tokens': 0} + +cost['runs'].append({ + 'date': datetime.now(timezone.utc).isoformat(), + 'input_tokens': input_tok, + 'output_tokens': output_tok, + 'files': detect.get('total_files', 0), +}) +cost['total_input_tokens'] += input_tok +cost['total_output_tokens'] += output_tok +cost_path.write_text(json.dumps(cost, indent=2, ensure_ascii=False), encoding=\"utf-8\") + +print(f'This run: {input_tok:,} input tokens, {output_tok:,} output tokens') +print(f'All time: {cost[\"total_input_tokens\"]:,} input, {cost[\"total_output_tokens\"]:,} output ({len(cost[\"runs\"])} runs)') +" +rm -f graphify-out/.graphify_detect.json graphify-out/.graphify_extract.json graphify-out/.graphify_ast.json graphify-out/.graphify_semantic.json graphify-out/.graphify_analysis.json +find graphify-out -maxdepth 1 -name '.graphify_chunk_*.json' -delete 2>/dev/null +rm -f graphify-out/.needs_update 2>/dev/null || true +``` + +Replace INPUT_PATH with the actual path (same value used in Steps 4-5) so the manifest is relativized to the scan root. + +Tell the user (omit the obsidian line unless --obsidian was given): +``` +Graph complete. Outputs in PATH_TO_DIR/graphify-out/ + + graph.html - interactive graph, open in browser + GRAPH_REPORT.md - audit report + graph.json - raw graph data + obsidian/ - Obsidian vault (only if --obsidian was given) +``` + +If graphify saved you time, consider supporting it: https://github.com/sponsors/safishamsi + +Replace PATH_TO_DIR with the actual absolute path of the directory that was processed. + +Then paste these sections from GRAPH_REPORT.md directly into the chat: +- God Nodes +- Surprising Connections +- Suggested Questions + +Do NOT paste the full report - just those three sections. Keep it concise. + +Then immediately offer to explore. Pick the single most interesting suggested question from the report - the one that crosses the most community boundaries or has the most surprising bridge node - and ask: + +> "The most interesting question this graph can answer: **[question]**. Want me to trace it?" + +If the user says yes, run `/graphify query "[question]"` on the graph and walk them through the answer using the graph structure - which nodes connect, which community boundaries get crossed, what the path reveals. Keep going as long as they want to explore. Each answer should end with a natural follow-up ("this connects to X - want to go deeper?") so the session feels like navigation, not a one-shot report. + +The graph is the map. Your job after the pipeline is to be the guide. + +--- + +## Interpreter guard for subcommands + +Before running any subcommand below (`--update`, `--cluster-only`, `query`, `path`, `explain`, `add`), check that `.graphify_python` exists. If it's missing (e.g. user deleted `graphify-out/`), re-resolve the interpreter first: + +```bash +if [ ! -f graphify-out/.graphify_python ]; then + GRAPHIFY_BIN=$(which graphify 2>/dev/null) + if [ -n "$GRAPHIFY_BIN" ]; then + PYTHON=$(head -1 "$GRAPHIFY_BIN" | tr -d '#!') + case "$PYTHON" in *[!a-zA-Z0-9/_.@-]*) PYTHON="python3" ;; esac + else + PYTHON="python3" + fi + mkdir -p graphify-out + "$PYTHON" -c "import sys; open('graphify-out/.graphify_python', 'w', encoding='utf-8').write(sys.executable)" +fi +``` + +## For --update and --cluster-only + +Both are non-default subcommands. `--update` re-extracts only new or changed files; `--cluster-only` reruns clustering on the existing graph. See `references/update.md` for both flows. + +--- + +## For /graphify query + +When `graphify-out/graph.json` already exists and the user asks a question about the corpus, answer from the graph rather than rebuilding it: + +```bash +graphify query "" +``` + +Before traversal, expand the question against the graph's own vocabulary so a wording mismatch does not collapse the answer to noise. If the `graphify query` CLI is unavailable, fall back to an inline NetworkX traversal of `graphify-out/graph.json`. Answer using only what the graph output contains, and quote `source_location` when citing a specific fact. For that vocab-expansion step, the BFS/DFS traversal modes, the `--budget` cap, the NetworkX fallback, `save-result` feedback, and the `/graphify path` and `/graphify explain` flows, see `references/query.md`. + +--- + +## For /graphify add and --watch + +Neither is part of the default build. When the user runs `/graphify add ` to fetch a URL into the corpus, or passes `--watch` to auto-rebuild on file changes, see `references/add-watch.md`. + +--- + +## For the commit hook and native AGENTS.md integration + +When the user asks to install the post-commit auto-rebuild hook or wire graphify into a project's AGENTS.md, see `references/hooks.md`. + +--- + +## Honesty Rules + +- Never invent an edge. If unsure, use AMBIGUOUS. +- Never skip the corpus check warning. +- Always show token cost in the report. +- Never hide cohesion scores behind symbols - show the raw number. +- Never run HTML viz on a graph with more than 5,000 nodes without warning the user. diff --git a/pyproject.toml b/pyproject.toml index 50bec7f3c9..936e78e6bf 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -5,7 +5,7 @@ build-backend = "setuptools.build_meta" [project] name = "graphifyy" version = "0.9.79" -description = "AI coding assistant skill (Claude Code, CodeBuddy, Codex, OpenCode, Kilo Code, Cursor, Gemini CLI, Aider, OpenClaw, Factory Droid, Trae, Hermes, Kiro, Pi, Devin CLI, Google Antigravity) - turn any folder of code, docs, papers, images, or videos into a queryable knowledge graph" +description = "AI coding assistant skill (Claude Code, CodeBuddy, Codex, OpenCode, Kilo Code, Cursor, Gemini CLI, Aider, OpenClaw, Factory Droid, Trae, Hermes, Kiro, Pi, Devin CLI, Google Antigravity, Mistral Vibe) - turn any folder of code, docs, papers, images, or videos into a queryable knowledge graph" readme = "README.md" license = "Apache-2.0" license-files = ["LICENSE", "LICENSE-MIT", "NOTICE"] @@ -42,6 +42,12 @@ dependencies = [ "tree-sitter-fortran>=0.6,<0.8", "tree-sitter-bash>=0.23,<0.27", "tree-sitter-json>=0.23,<0.26", + # tomlkit is a comment/format-preserving TOML round-tripper. `graphify vibe + # install` merges graphify pre_tool entries into the user's .vibe/hooks.toml + # (which may have hand-authored comments and other hooks), and stdlib alone + # cannot do this: tomllib is read-only, and it's only in 3.11+ while + # requires-python is 3.10. Pure Python, no C ext, tiny footprint. + "tomlkit>=0.13", ] [project.urls] @@ -155,7 +161,7 @@ include-package-data = false # under graphify/skills//references/, and the always-on injection blocks # under graphify/always_on/. There is no graphify/skills//SKILL.md in the # repo, so no SKILL.md glob is needed here. -graphify = ["skill.md", "skill-codex.md", "skill-opencode.md", "skill-kilo.md", "command-kilo.md", "skill-aider.md", "skill-amp.md", "skill-agents.md", "skill-copilot.md", "skill-claw.md", "skill-windows.md", "skill-droid.md", "skill-trae.md", "skill-kiro.md", "skill-vscode.md", "skill-pi.md", "skill-devin.md", "skills/*/references/*.md", "always_on/*.md"] +graphify = ["skill.md", "skill-codex.md", "skill-opencode.md", "skill-kilo.md", "command-kilo.md", "skill-aider.md", "skill-amp.md", "skill-agents.md", "skill-copilot.md", "skill-claw.md", "skill-windows.md", "skill-droid.md", "skill-trae.md", "skill-kiro.md", "skill-vscode.md", "skill-pi.md", "skill-devin.md", "skill-vibe.md", "skills/*/references/*.md", "always_on/*.md"] [tool.pytest.ini_options] testpaths = ["tests"] diff --git a/tests/test_detect.py b/tests/test_detect.py index 0994738f74..43eb71161c 100644 --- a/tests/test_detect.py +++ b/tests/test_detect.py @@ -3270,6 +3270,27 @@ def test_detect_prunes_venv_names_without_markers(tmp_path): assert not any(f"{os.sep}{name}{os.sep}" in f for f in all_files), f"{name} must stay pruned" +def test_detect_prunes_vibe_install_dir(tmp_path): + """#2537: .vibe/ holds graphify's own skill/agent/hook artifacts when + installed in project scope — scanning it wastes extraction on config files + instead of real code, so it must be pruned by default.""" + vibe = tmp_path / ".vibe" + (vibe / "skills" / "graphify").mkdir(parents=True) + (vibe / "skills" / "graphify" / "SKILL.md").write_text("# skill\n") + (vibe / "agents").mkdir() + (vibe / "agents" / "graphify-extract.toml").write_text("display_name = 'x'\n") + (vibe / "hooks.toml").write_text('[[hooks]]\nname = "x"\n') + (tmp_path / "main.py").write_text("def main():\n return 1\n") + + result = detect(tmp_path) + all_files = [f for files in result["files"].values() for f in files] + assert any("main.py" in f for f in all_files), "real source must still be scanned" + assert not any(".vibe" in f for f in all_files), ".vibe/ must be pruned (#2537)" + assert any(f"{os.sep}.vibe{os.sep}" in d for d in result["pruned_noise_dirs"]), ( + "pruned .vibe must be traceable in pruned_noise_dirs (#2537)" + ) + + @pytest.mark.parametrize( ("configured_out", "absolute", "symlink_target"), [ diff --git a/tests/test_vibe.py b/tests/test_vibe.py new file mode 100644 index 0000000000..ed1a49d3c2 --- /dev/null +++ b/tests/test_vibe.py @@ -0,0 +1,701 @@ +"""Vibe install lays down its full three-artifact always-on layer. + +Mistral Vibe (github.com/mistralai/mistral-vibe) is Agent Skills-compliant and +has a first-class pre_tool hook system in `.vibe/hooks.toml`, so `graphify vibe +install` needs Claude Code parity: skill (with `user-invocable: true` for +`/graphify` slash-command autocomplete), AGENTS.md always-on section, AND +hooks.toml pre_tool entries that nudge grep/read toward `graphify query`. + +Every test scopes HOME to `tmp_path/home` so a bare (non-project) install +lands in an isolated ~/.vibe rather than the developer's real home. +""" +import pytest + +import graphify.__main__ as m + + +@pytest.fixture +def home(tmp_path, monkeypatch): + """Redirect ~/.vibe to tmp_path/home/.vibe for the duration of the test.""" + fake_home = tmp_path / "home" + fake_home.mkdir() + monkeypatch.setattr("pathlib.Path.home", lambda: fake_home) + monkeypatch.setenv("HOME", str(fake_home)) + return fake_home + + +def test_project_install_writes_skill_agents_md_and_hooks(tmp_path, home): + m._vibe_install(tmp_path, project=True) + + skill = tmp_path / ".vibe" / "skills" / "graphify" / "SKILL.md" + references = tmp_path / ".vibe" / "skills" / "graphify" / "references" + agents_md = tmp_path / "AGENTS.md" + hooks_toml = tmp_path / ".vibe" / "hooks.toml" + + assert skill.exists(), "SKILL.md must be installed under .vibe/skills/" + assert references.is_dir(), "shared agents references bundle must be copied" + assert agents_md.exists(), "AGENTS.md always-on section must be written" + assert hooks_toml.exists(), "pre_tool hooks must be registered" + + +def test_skill_frontmatter_is_user_invocable(tmp_path, home): + """`user-invocable: true` is what makes /graphify show in vibe autocomplete.""" + m._vibe_install(tmp_path, project=True) + body = (tmp_path / ".vibe" / "skills" / "graphify" / "SKILL.md").read_text(encoding="utf-8") + assert body.startswith("---\n"), "SKILL.md must have YAML frontmatter" + assert "user-invocable: true" in body, "vibe needs this to expose /graphify" + assert "name: graphify" in body + + +def test_hooks_toml_shape_matches_vibe_pre_tool_schema(tmp_path, home): + """Vibe's HookConfig requires: name (str), type (HookType), command (str). + + Optional/defaulted: match, timeout, strict, description. This test locks + the full required contract so a regression that drops `name` (which Vibe + would reject at parse time) doesn't silently pass CI. + """ + import tomlkit + + m._vibe_install(tmp_path, project=True) + doc = tomlkit.parse((tmp_path / ".vibe" / "hooks.toml").read_text(encoding="utf-8")) + hooks = list(doc["hooks"]) + assert len(hooks) == 2, "one entry per matcher (grep, read_file)" + + matchers = {h["match"] for h in hooks} + assert matchers == {"grep", "read_file"}, ( + f"unexpected matchers: {matchers}; vibe registers the read tool as `read_file` " + f"and uses fnmatch — `read` alone never matches" + ) + + names = {h["name"] for h in hooks} + assert names == {"graphify-nudge-search", "graphify-nudge-read"}, ( + f"vibe HookConfig requires `name`; got {names}" + ) + + for h in hooks: + assert isinstance(h["name"], str) and h["name"], "name must be non-empty str" + assert h["type"] == "pre_tool" + assert isinstance(h["command"], str) and "graphify" in h["command"] + assert "hook-guard" in h["command"] + assert isinstance(h["timeout"], float) + assert h["strict"] is False + + +def test_strict_flag_flows_only_to_read_hook(tmp_path, home): + """--strict escalates read-guard blocking; grep-guard stays a nudge (matches claude).""" + import tomlkit + + m._vibe_install(tmp_path, project=True, strict=True) + doc = tomlkit.parse((tmp_path / ".vibe" / "hooks.toml").read_text(encoding="utf-8")) + hooks = {h["match"]: h["command"] for h in doc["hooks"]} + assert "--strict" in hooks["read_file"], "read guard must carry --strict when requested" + assert "--strict" not in hooks["grep"], "search guard must not block; nudge only" + + +def test_install_is_idempotent_no_backup_churn(tmp_path, home): + """Second install with same inputs must not rewrite files or drop .graphify-bak.""" + m._vibe_install(tmp_path, project=True) + before = (tmp_path / ".vibe" / "hooks.toml").read_bytes() + m._vibe_install(tmp_path, project=True) + after = (tmp_path / ".vibe" / "hooks.toml").read_bytes() + assert before == after, "hooks.toml must be byte-identical after re-install" + backups = list((tmp_path / ".vibe").glob("*.graphify-bak")) + assert backups == [], f"idempotent re-install must not create backups: {backups}" + + +def test_strict_flip_creates_backup_and_updates_command(tmp_path, home): + """When the effective hook command changes, backup the old file (matches gemini/claude).""" + m._vibe_install(tmp_path, project=True, strict=False) + m._vibe_install(tmp_path, project=True, strict=True) + backup = tmp_path / ".vibe" / "hooks.toml.graphify-bak" + assert backup.exists(), "flipping strict must preserve the previous hooks.toml" + body = (tmp_path / ".vibe" / "hooks.toml").read_text(encoding="utf-8") + assert "hook-guard read --strict" in body + + +def test_install_preserves_user_authored_hooks_and_comments(tmp_path, home): + """This is why we picked tomlkit over stdlib TOML writers.""" + (tmp_path / ".vibe").mkdir() + existing = ( + '# my custom hooks - keep these\n' + '[[hooks]]\n' + 'name = "my-guard"\n' + 'type = "pre_tool"\n' + 'match = "bash"\n' + 'command = "my-guard.sh"\n' + 'timeout = 10.0\n' + 'strict = true\n' + 'description = "user hook"\n' + ) + (tmp_path / ".vibe" / "hooks.toml").write_text(existing, encoding="utf-8") + + m._vibe_install(tmp_path, project=True) + + body = (tmp_path / ".vibe" / "hooks.toml").read_text(encoding="utf-8") + assert "# my custom hooks - keep these" in body, "user comment must survive" + assert 'name = "my-guard"' in body, "user hook must survive" + assert "hook-guard search" in body, "graphify hooks must be added" + assert "hook-guard read" in body + + +def test_uninstall_strips_only_graphify_entries(tmp_path, home): + """User's own [[hooks]] entries must remain after `vibe uninstall --project`.""" + (tmp_path / ".vibe").mkdir() + (tmp_path / ".vibe" / "hooks.toml").write_text( + '[[hooks]]\n' + 'name = "my-guard"\n' + 'type = "pre_tool"\n' + 'match = "bash"\n' + 'command = "my-guard.sh"\n' + 'timeout = 10.0\n' + 'strict = true\n' + 'description = "user hook"\n', + encoding="utf-8", + ) + m._vibe_install(tmp_path, project=True) + m._vibe_uninstall(tmp_path, project=True) + + hooks_toml = tmp_path / ".vibe" / "hooks.toml" + assert hooks_toml.exists(), "must leave user-authored hooks.toml behind" + body = hooks_toml.read_text(encoding="utf-8") + assert 'name = "my-guard"' in body, "user hook must survive uninstall" + assert "graphify" not in body, "no graphify entries may remain" + + +def test_uninstall_removes_empty_hooks_toml(tmp_path, home): + """If graphify was the only source of [[hooks]], uninstall must clean the empty file.""" + m._vibe_install(tmp_path, project=True) + m._vibe_uninstall(tmp_path, project=True) + assert not (tmp_path / ".vibe" / "hooks.toml").exists() + assert not (tmp_path / ".vibe" / "skills" / "graphify" / "SKILL.md").exists() + assert not (tmp_path / "AGENTS.md").exists() + + +def test_agents_md_section_uses_shared_marker(tmp_path, home): + """The AGENTS.md section must use the same `## graphify` marker as codex/opencode/aider + so the shared uninstall path (`_remove_marker_section`) matches and cleans it.""" + m._vibe_install(tmp_path, project=True) + body = (tmp_path / "AGENTS.md").read_text(encoding="utf-8") + assert body.startswith("## graphify"), "must open with the shared marker heading" + assert "graphify query" in body, "must nudge toward the query-first path" + + +def test_agents_md_section_replaced_in_place_not_duplicated(tmp_path, home): + """Re-install must not double-append the graphify section (regression #580 / #1688).""" + (tmp_path / "AGENTS.md").write_text( + "# project prelude\n\nsome existing content.\n\n" + "## graphify\n\nstale graphify block from an older version.\n", + encoding="utf-8", + ) + m._vibe_install(tmp_path, project=True) + body = (tmp_path / "AGENTS.md").read_text(encoding="utf-8") + assert body.count("## graphify") == 1, "must replace, not duplicate" + assert "stale graphify block from an older version" not in body, "old block must be replaced" + assert "# project prelude" in body, "user's prelude must survive" + + +def test_global_install_targets_home_dot_vibe(tmp_path, home): + """Bare `graphify vibe install` (no --project) must land under ~/.vibe.""" + m._vibe_install(project=False) + assert (home / ".vibe" / "skills" / "graphify" / "SKILL.md").exists() + assert (home / ".vibe" / "hooks.toml").exists() + assert (home / ".vibe" / "AGENTS.md").exists() + + +def test_platform_config_registers_vibe(): + """Vibe must be discoverable through the shared platform-config surface. + + Behavior-based: don't lock the internal `skill_refs` value (reusing the + agents bundle is an implementation choice that could change). Just verify + the platform is registered and that after install the expected reference + files land beside SKILL.md. + """ + from graphify.install import _PLATFORM_CONFIG, _CLI_INSTALL_COMMANDS + assert "vibe" in _PLATFORM_CONFIG + assert "vibe" in _CLI_INSTALL_COMMANDS + + +def test_install_lands_references_sidecar_alongside_skill(tmp_path, home): + """The references/ progressive-disclosure sidecar must ship with SKILL.md. + + Bundle-source coupling (which references bundle) is an internal choice; the + user-visible contract is that references exist and SKILL.md can point at + them. Regressions where the sidecar goes missing (packaging bug, wrong + skill_refs key) break the skill body's `references/hooks.md`-style links. + """ + m._vibe_install(tmp_path, project=True) + refs = tmp_path / ".vibe" / "skills" / "graphify" / "references" + assert refs.is_dir(), "references/ sidecar must land next to SKILL.md" + assert any(refs.glob("*.md")), "references/ must contain at least one .md" + + +def test_dispatch_cli_recognizes_vibe_subcommand(tmp_path, home, monkeypatch): + """`graphify vibe install --project` must reach `_vibe_install`.""" + import sys + monkeypatch.chdir(tmp_path) + monkeypatch.setattr(sys, "argv", ["graphify", "vibe", "install", "--project"]) + handled = m.dispatch_install_cli("vibe") + assert handled is True, "vibe dispatch must claim the command" + assert (tmp_path / ".vibe" / "skills" / "graphify" / "SKILL.md").exists() + assert (tmp_path / ".vibe" / "hooks.toml").exists() + + +def test_dispatch_bare_vibe_uninstall_removes_global_install(tmp_path, home, monkeypatch): + """Regression: `graphify vibe uninstall` (no --project) must clean ~/.vibe. + + The dispatch branch previously passed an explicit `project_dir` even for + bare uninstall, which forced `explicit_dir=True` in `_vibe_uninstall` and + resolved `remove_user_skill=False` -> project-only cleanup, leaving the + entire global install intact. This test locks the fix. + """ + import sys + monkeypatch.chdir(tmp_path) + + monkeypatch.setattr(sys, "argv", ["graphify", "vibe", "install"]) + m.dispatch_install_cli("vibe") + assert (home / ".vibe" / "skills" / "graphify" / "SKILL.md").exists() + assert (home / ".vibe" / "hooks.toml").exists() + assert (home / ".vibe" / "AGENTS.md").exists() + + monkeypatch.setattr(sys, "argv", ["graphify", "vibe", "uninstall"]) + m.dispatch_install_cli("vibe") + assert not (home / ".vibe" / "skills" / "graphify" / "SKILL.md").exists() + assert not (home / ".vibe" / "hooks.toml").exists() + assert not (home / ".vibe" / "AGENTS.md").exists() + + +def test_dispatch_platform_vibe_alias_goes_through_full_install(tmp_path, home, monkeypatch): + """`graphify install --platform vibe` must invoke the full-parity installer. + + The bare `install()` function had a skill-only default path; vibe needs the + three-artifact orchestrator (skill + AGENTS.md + hooks.toml). Regressions + here would ship a skill without hooks or AGENTS.md when users pick the + `--platform vibe` invocation form. + """ + monkeypatch.chdir(tmp_path) + m.install(platform="vibe", project=True, project_dir=tmp_path) + assert (tmp_path / ".vibe" / "skills" / "graphify" / "SKILL.md").exists() + assert (tmp_path / ".vibe" / "hooks.toml").exists() + assert (tmp_path / "AGENTS.md").exists() + + +def test_install_platform_vibe_propagates_strict_flag(tmp_path, home, monkeypatch): + """`graphify install --platform vibe --strict` (bare global) must propagate --strict. + + Regression: the top-level `install()` used to drop the strict kwarg for the + vibe branch, so bare-global install silently produced a non-strict read hook + while --project --strict worked. Unlike Claude Code (where --strict requires + a project hook), vibe's global hooks.toml at ~/.vibe/hooks.toml legitimately + accepts --strict, so the flag must flow through. + """ + import sys + monkeypatch.chdir(tmp_path) + monkeypatch.setattr(sys, "argv", ["graphify", "install", "--platform", "vibe", "--strict"]) + m.dispatch_install_cli("install") + + hooks_toml = home / ".vibe" / "hooks.toml" + assert hooks_toml.exists(), "bare global install must create hooks.toml" + body = hooks_toml.read_text(encoding="utf-8") + assert "hook-guard read --strict" in body, "read hook must carry --strict" + assert "hook-guard search --strict" not in body, "search hook must stay a nudge" + + +def test_uninstall_all_sweeps_vibe_global(tmp_path, home, monkeypatch): + """`graphify uninstall` (uninstall_all) must clean the global vibe install. + + Otherwise vibe hooks keep firing after the user removes graphify from all + other platforms. Regression lock: private helpers looked healthy but + `uninstall_all` was missing the vibe cleanup call. + """ + monkeypatch.chdir(tmp_path) + m._vibe_install(project=False) + assert (home / ".vibe" / "skills" / "graphify" / "SKILL.md").exists() + assert (home / ".vibe" / "hooks.toml").exists() + assert (home / ".vibe" / "AGENTS.md").exists() + + m.uninstall_all(project_dir=tmp_path, purge=False) + + assert not (home / ".vibe" / "skills" / "graphify" / "SKILL.md").exists() + assert not (home / ".vibe" / "hooks.toml").exists() + assert not (home / ".vibe" / "AGENTS.md").exists() + + +def test_vibe_home_env_var_redirects_global_install(tmp_path, monkeypatch): + """Global install must honor VIBE_HOME so users on non-default vibe setups don't get orphaned installs. + + Vibe's own harness_files/_harness_manager.py reads VIBE_HOME/skills, + VIBE_HOME/hooks.toml, and VIBE_HOME/AGENTS.md. Hardcoding ~/.vibe silently + breaks users who set VIBE_HOME=/opt/vibe-shared. + """ + custom_vibe_home = tmp_path / "custom-vibe-root" + real_home = tmp_path / "home" + real_home.mkdir() + + monkeypatch.setattr("pathlib.Path.home", lambda: real_home) + monkeypatch.setenv("HOME", str(real_home)) + monkeypatch.setenv("VIBE_HOME", str(custom_vibe_home)) + + m._vibe_install(project=False) + + assert (custom_vibe_home / "skills" / "graphify" / "SKILL.md").exists() + assert (custom_vibe_home / "hooks.toml").exists() + assert (custom_vibe_home / "AGENTS.md").exists() + assert not (real_home / ".vibe").exists() + + +def test_vibe_home_env_var_symmetric_uninstall(tmp_path, monkeypatch): + """Uninstall must honor VIBE_HOME too, or install/uninstall drift and hooks stay behind.""" + custom_vibe_home = tmp_path / "custom-vibe-root" + real_home = tmp_path / "home" + real_home.mkdir() + monkeypatch.setattr("pathlib.Path.home", lambda: real_home) + monkeypatch.setenv("HOME", str(real_home)) + monkeypatch.setenv("VIBE_HOME", str(custom_vibe_home)) + + m._vibe_install(project=False) + m._vibe_uninstall() + + assert not (custom_vibe_home / "skills" / "graphify" / "SKILL.md").exists() + assert not (custom_vibe_home / "hooks.toml").exists() + assert not (custom_vibe_home / "AGENTS.md").exists() + + +def test_install_preserves_user_hook_whose_command_mentions_graphify(tmp_path, home): + """A user hook that calls graphify for their own reason must survive install. + + Regression: the old filter was `"graphify" not in str(h.get("command"))`, + which deleted ANY hook whose command contained 'graphify' - including the + user's own tooling. The fix filters by hook `name` in a fixed set. + """ + (tmp_path / ".vibe").mkdir() + (tmp_path / ".vibe" / "hooks.toml").write_text( + '[[hooks]]\n' + 'name = "my-graphify-metrics"\n' + 'type = "post_tool"\n' + 'match = "*"\n' + 'command = "graphify metrics --log"\n' + 'timeout = 5.0\n' + 'strict = false\n' + 'description = "my own thing that happens to call graphify"\n', + encoding="utf-8", + ) + m._vibe_install(tmp_path, project=True) + body = (tmp_path / ".vibe" / "hooks.toml").read_text(encoding="utf-8") + assert 'name = "my-graphify-metrics"' in body, ( + "user hook whose command mentions 'graphify' must NOT be deleted" + ) + assert "hook-guard search" in body, "graphify hooks must still be added" + + +def test_uninstall_preserves_user_hook_whose_command_mentions_graphify(tmp_path, home): + """Same regression on the uninstall side: only remove graphify-owned entries by name.""" + m._vibe_install(tmp_path, project=True) + import tomlkit + from tomlkit.items import AoT + hooks_path = tmp_path / ".vibe" / "hooks.toml" + doc = tomlkit.parse(hooks_path.read_text(encoding="utf-8")) + user_hook = tomlkit.table() + user_hook["name"] = "my-graphify-log" + user_hook["type"] = "post_tool" + user_hook["match"] = "*" + user_hook["command"] = "graphify metrics --log" + user_hook["timeout"] = 5.0 + user_hook["strict"] = False + user_hook["description"] = "user thing" + assert isinstance(doc["hooks"], AoT) + doc["hooks"].append(user_hook) + hooks_path.write_text(tomlkit.dumps(doc), encoding="utf-8") + + m._vibe_uninstall(tmp_path, project=True) + + remaining = hooks_path.read_text(encoding="utf-8") + assert 'name = "my-graphify-log"' in remaining, "user hook must survive uninstall" + assert "hook-guard search" not in remaining, "graphify hooks must be removed" + + +def test_hook_command_shell_quotes_paths_with_metacharacters(monkeypatch): + """Hook commands must survive shell-metachar paths (macOS home with `(`, `$`, `;`). + + Vibe runs hooks via asyncio.create_subprocess_shell, so metachars in the + graphify exe path parse as shell operators. The previous "quote only if + the path has a space" heuristic left injection open for paths like + `/tmp/g;touch /tmp/pwned/graphify` or `~/Library/Application Support/...`. + shlex.join produces a shell-safe single-token quoting on POSIX. + """ + hostile_paths = [ + "/tmp/g;touch /tmp/pwned/graphify", + "/tmp/g$(whoami)/graphify", + "/tmp/g`whoami`/graphify", + "/Users/me/App (2)/bin/graphify", + "/tmp/with spaces/graphify", + "/tmp/g&background/graphify", + "/tmp/g|pipe/graphify", + '/tmp/g"quote/graphify', + "/tmp/g'squote/graphify", + ] + import shlex + import graphify.install as install_mod + for path in hostile_paths: + monkeypatch.setattr(install_mod, "_resolve_graphify_exe", lambda p=path: p) + entries = install_mod._vibe_hook_entries() + for e in entries: + tokens = shlex.split(e["command"]) + assert tokens[0] == path, ( + f"path {path!r} was not shell-quoted correctly; " + f"got tokens {tokens!r} from command {e['command']!r}" + ) + assert tokens[1] == "hook-guard" + assert tokens[2] in ("search", "read") + + +def test_install_refuses_when_hooks_is_plain_array_not_aot(tmp_path, home): + """`hooks = []` (plain Array) must be refused, not corrupted with appended tables. + + Regression: the old `isinstance(existing, list)` check accepted tomlkit's + Array (which passes the list-like check). Appending tables into a plain + array produced unparsable TOML that broke vibe on next load. + """ + (tmp_path / ".vibe").mkdir() + (tmp_path / ".vibe" / "hooks.toml").write_text('hooks = []\n', encoding="utf-8") + with pytest.raises(SystemExit): + m._vibe_install(tmp_path, project=True) + + +def test_install_refuses_when_hooks_is_inline_table(tmp_path, home): + (tmp_path / ".vibe").mkdir() + (tmp_path / ".vibe" / "hooks.toml").write_text( + 'hooks = { name = "bad", type = "pre_tool", command = "x" }\n', + encoding="utf-8", + ) + with pytest.raises(SystemExit): + m._vibe_install(tmp_path, project=True) + + +def test_install_refuses_when_hooks_is_scalar(tmp_path, home): + (tmp_path / ".vibe").mkdir() + (tmp_path / ".vibe" / "hooks.toml").write_text('hooks = "bad"\n', encoding="utf-8") + with pytest.raises(SystemExit): + m._vibe_install(tmp_path, project=True) + + +def test_install_refuses_when_hooks_element_is_not_table(tmp_path, home): + """`hooks = ["bad"]` must refuse rather than AttributeError on `.get()`. + + A `hooks = ["bad"]` file parses as an AoT-ish container to some tomlkit + versions but each element is a scalar. The pre-fix code called `.get()` + unconditionally on every element, which would AttributeError mid-write. + Now: element shape is checked and _refuse_to_modify is called on the first + non-mapping element. + """ + (tmp_path / ".vibe").mkdir() + (tmp_path / ".vibe" / "hooks.toml").write_text('hooks = ["bad"]\n', encoding="utf-8") + with pytest.raises(SystemExit): + m._vibe_install(tmp_path, project=True) + + +def test_install_refuses_when_toml_is_malformed(tmp_path, home): + """Malformed TOML must _refuse_to_modify, not silently reset the file.""" + (tmp_path / ".vibe").mkdir() + (tmp_path / ".vibe" / "hooks.toml").write_text( + '[[hooks\nunterminated table header\n', + encoding="utf-8", + ) + with pytest.raises(SystemExit): + m._vibe_install(tmp_path, project=True) + + +# --------------------------------------------------------------------------- +# graphify-extract writable subagent (Option A fix for #2537 review comment) +# +# Vibe's built-in `explore` subagent (vibe/core/agents/models.py:EXPLORE) is +# read-only — its enabled_tools are ["grep", "read_file", "skill"] with no +# write_file. When the graphify skill dispatches Task subagents that inherit +# `explore`, they cannot write .graphify_chunk_NN.json and Part B extraction +# silently produces zero chunks. `graphify vibe install` therefore installs +# a writable subagent profile at /agents/graphify-extract.toml, which +# the skill fragment now targets via subagent_type="graphify-extract". +# --------------------------------------------------------------------------- + + +def test_project_install_writes_graphify_extract_agent(tmp_path, home): + """Agent TOML lands under .vibe/agents/ so vibe's AgentRegistry discovers it.""" + m._vibe_install(tmp_path, project=True) + agent_toml = tmp_path / ".vibe" / "agents" / "graphify-extract.toml" + assert agent_toml.exists(), ( + "graphify-extract.toml must be installed under .vibe/agents/ — " + "without it, Task dispatch inherits vibe's read-only `explore` and " + "chunk writes silently fail (PR #2537 review comment)." + ) + + +def test_global_install_writes_graphify_extract_agent(tmp_path, home): + """Non-project install lands the agent under VIBE_HOME/agents/ (default ~/.vibe).""" + m._vibe_install() + agent_toml = home / ".vibe" / "agents" / "graphify-extract.toml" + assert agent_toml.exists() + + +def test_graphify_extract_agent_is_writable_subagent(tmp_path, home): + """Contract: safety=safe + agent_type=subagent + write_file enabled. + + Locks the four fields that make this agent usable by the skill: + - agent_type=subagent (required for Task dispatch to accept it) + - write_file in enabled_tools (the whole point — Explore lacks it) + - allowlist scoped to graphify's chunk output (no unbounded write grant) + - system_prompt_id=explore (reuse vibe's built-in lean prompt) + Anything drifting here breaks the fix without breaking the previous tests. + """ + import tomllib + + m._vibe_install(tmp_path, project=True) + agent_toml = tmp_path / ".vibe" / "agents" / "graphify-extract.toml" + data = tomllib.loads(agent_toml.read_text(encoding="utf-8")) + + assert data["agent_type"] == "subagent" + assert data["safety"] == "safe" + assert data["system_prompt_id"] == "explore" + assert "write_file" in data["enabled_tools"], ( + "graphify-extract must enable write_file — otherwise it degenerates " + "into the same read-only agent as vibe's built-in `explore`." + ) + assert "read_file" in data["enabled_tools"] + assert "grep" in data["enabled_tools"] + assert "skill" in data["enabled_tools"] + + write_cfg = data["tools"]["write_file"] + assert write_cfg["permission"] == "always" + assert write_cfg["allowlist"] == ["**/graphify-out/.graphify_chunk_*.json"], ( + "write_file must be scoped to graphify's chunk output only — a broader " + "allowlist would grant this subagent workspace-wide write access." + ) + + +def test_graphify_extract_install_is_idempotent(tmp_path, home): + """Re-installing with identical body must NOT create a .graphify-bak.""" + m._vibe_install(tmp_path, project=True) + agent_toml = tmp_path / ".vibe" / "agents" / "graphify-extract.toml" + mtime_before = agent_toml.stat().st_mtime + + m._vibe_install(tmp_path, project=True) + + backup = agent_toml.with_name(agent_toml.name + ".graphify-bak") + assert not backup.exists(), "no-op re-install must not churn a backup" + assert agent_toml.stat().st_mtime == mtime_before, ( + "no-op re-install must leave the file untouched" + ) + + +def test_graphify_extract_install_backs_up_hand_edited_file(tmp_path, home): + """A hand-edited agent file must be backed up before we overwrite it.""" + m._vibe_install(tmp_path, project=True) + agent_toml = tmp_path / ".vibe" / "agents" / "graphify-extract.toml" + agent_toml.write_text("# user hand-edit\ndisplay_name = \"Broken\"\n", encoding="utf-8") + + m._vibe_install(tmp_path, project=True) + + backup = agent_toml.with_name(agent_toml.name + ".graphify-bak") + assert backup.exists(), "hand-edited file must be preserved as .graphify-bak" + assert "user hand-edit" in backup.read_text(encoding="utf-8") + # The live file is refreshed to the canonical body. + assert "Graphify Extract" in agent_toml.read_text(encoding="utf-8") + + +def test_project_uninstall_removes_graphify_extract_agent(tmp_path, home): + """Symmetric project uninstall: the agent file is removed.""" + m._vibe_install(tmp_path, project=True) + agent_toml = tmp_path / ".vibe" / "agents" / "graphify-extract.toml" + assert agent_toml.exists() + + m._vibe_uninstall(tmp_path, project=True) + + assert not agent_toml.exists(), "graphify-extract.toml must be removed on uninstall" + + +def test_project_uninstall_prunes_empty_agents_dir(tmp_path, home): + """If we own the agents dir (project scope, only our file), prune it.""" + m._vibe_install(tmp_path, project=True) + agents_dir = tmp_path / ".vibe" / "agents" + assert agents_dir.is_dir() + + m._vibe_uninstall(tmp_path, project=True) + + assert not agents_dir.exists(), ( + "empty .vibe/agents/ must be pruned so uninstall leaves no trace" + ) + + +def test_project_uninstall_preserves_user_agents_in_shared_dir(tmp_path, home): + """Do NOT delete unrelated agent files a user co-located under .vibe/agents/.""" + m._vibe_install(tmp_path, project=True) + user_agent = tmp_path / ".vibe" / "agents" / "my-custom.toml" + user_agent.write_text("display_name = \"My Custom\"\n", encoding="utf-8") + + m._vibe_uninstall(tmp_path, project=True) + + assert user_agent.exists(), ( + "user-authored agents in the shared dir must survive graphify uninstall" + ) + # And the dir survives because it is non-empty. + assert (tmp_path / ".vibe" / "agents").is_dir() + + +def test_global_uninstall_removes_graphify_extract_but_leaves_home_agents_dir(tmp_path, home): + """Global scope: remove the file, but never touch VIBE_HOME/agents/ itself. + + A user may have unrelated custom agents in ~/.vibe/agents/ (that's exactly + what the dir is for — see vibe's AgentRegistry.search_paths). Pruning it + on graphify uninstall would silently delete their work. + """ + m._vibe_install() + agent_toml = home / ".vibe" / "agents" / "graphify-extract.toml" + assert agent_toml.exists() + + m._vibe_uninstall() + + assert not agent_toml.exists() + # Dir must still exist — even if empty, we do not prune the global agents dir. + assert (home / ".vibe" / "agents").exists(), ( + "global VIBE_HOME/agents/ must NEVER be pruned by graphify — the dir " + "belongs to the user and may hold unrelated custom agent profiles." + ) + + +def test_vibe_home_env_var_redirects_graphify_extract_agent(tmp_path, monkeypatch): + """VIBE_HOME redirects the agent file the same way it redirects skills/hooks.""" + custom_vibe_home = tmp_path / "custom-vibe-root" + monkeypatch.setenv("VIBE_HOME", str(custom_vibe_home)) + m._vibe_install() + assert (custom_vibe_home / "agents" / "graphify-extract.toml").exists() + + m._vibe_uninstall() + assert not (custom_vibe_home / "agents" / "graphify-extract.toml").exists() + + +def test_uninstall_all_sweeps_graphify_extract_agent(tmp_path, home, monkeypatch): + """`graphify uninstall` (no --platform) must sweep the vibe agent alongside hooks.""" + m._vibe_install() + agent_toml = home / ".vibe" / "agents" / "graphify-extract.toml" + assert agent_toml.exists() + + monkeypatch.chdir(tmp_path) + m.uninstall_all() + + assert not agent_toml.exists(), ( + "uninstall_all must remove graphify-extract.toml so vibe stops " + "advertising a subagent that no longer belongs to any installed skill." + ) + + +def test_skill_fragment_targets_graphify_extract_not_explore(tmp_path, home): + """The rendered skill body must dispatch subagent_type=graphify-extract. + + Guards against a fragment edit reverting to the read-only default that + prompted the #2537 review comment in the first place. + """ + m._vibe_install(tmp_path, project=True) + skill = (tmp_path / ".vibe" / "skills" / "graphify" / "SKILL.md").read_text(encoding="utf-8") + assert 'subagent_type="graphify-extract"' in skill, ( + "skill must instruct vibe to use the writable graphify-extract subagent" + ) + # The old Claude-Code-specific advice must NOT be the sole recovery hint on vibe. + assert "graphify-extract" in skill, "the fix must be visible in the skill body" diff --git a/tools/skillgen/expected/graphify__skill-vibe.md b/tools/skillgen/expected/graphify__skill-vibe.md new file mode 100644 index 0000000000..41e754a47c --- /dev/null +++ b/tools/skillgen/expected/graphify__skill-vibe.md @@ -0,0 +1,710 @@ +--- +name: graphify +description: "Use for any question about a codebase, its architecture, file relationships, or project content — especially when graphify-out/ exists, where the question should be treated as a graphify query first. Turns any input (code, docs, papers, images, videos) into a persistent knowledge graph with god nodes, community detection, and query/path/explain tools." +user-invocable: true +allowed-tools: + - bash + - read_file + - grep + - write_file + - edit + - task + - ask_user_question +--- + + +# /graphify + +Turn any folder of files into a navigable knowledge graph with community detection, an honest audit trail, and three outputs: interactive HTML, GraphRAG-ready JSON, and a plain-language GRAPH_REPORT.md. + +## Usage + +``` +/graphify # full pipeline on current directory (HTML viz; add --obsidian for a vault) +/graphify # full pipeline on specific path +/graphify https://github.com// # clone repo then run full pipeline on it +/graphify https://github.com// --branch # clone a specific branch +/graphify ... # clone multiple repos, build each, merge into one cross-repo graph +/graphify --mode deep # thorough extraction, richer INFERRED edges +/graphify --update # incremental - re-extract only new/changed files +/graphify --directed # build directed graph (preserves edge direction: source→target) +/graphify --whisper-model medium # use a larger Whisper model for better transcription accuracy +/graphify --cluster-only # rerun clustering on existing graph +/graphify --no-viz # skip visualization, just report + JSON +/graphify --html # (HTML is generated by default - this flag is a no-op) +/graphify --svg # also export graph.svg (embeds in Notion, GitHub) +/graphify --graphml # export graph.graphml (Gephi, yEd) +/graphify --neo4j # generate graphify-out/cypher.txt for Neo4j +/graphify --neo4j-push bolt://localhost:7687 # push directly to Neo4j +/graphify --falkordb # generate graphify-out/cypher.txt for FalkorDB +/graphify --falkordb-push falkordb://localhost:6379 # push directly to FalkorDB +/graphify --mcp # start MCP stdio server for agent access +/graphify --watch # watch folder, auto-rebuild on code changes (no LLM needed) +/graphify --wiki # build agent-crawlable wiki (index.md + one article per community) +/graphify --obsidian --obsidian-dir ~/vaults/my-project # write vault to custom path (e.g. existing vault) +/graphify add # fetch URL, save to ./raw, update graph +/graphify add --author "Name" # tag who wrote it +/graphify add --contributor "Name" # tag who added it to the corpus +/graphify query "" # BFS traversal - broad context +/graphify query "" --dfs # DFS - trace a specific path +/graphify query "" --budget 1500 # cap answer at N tokens +/graphify path "AuthModule" "Database" # shortest path between two concepts +/graphify explain "SwinTransformer" # plain-language explanation of a node +``` + +## What graphify is for + +Drop any folder of code, docs, papers, images, or video into graphify and get a queryable knowledge graph. Persistent across sessions, honest audit trail (EXTRACTED/INFERRED/AMBIGUOUS), community detection surfaces cross-document connections you wouldn't think to ask about. + +## What You Must Do When Invoked + +If the user invoked `/graphify --help` or `/graphify -h` (with no other arguments), print the contents of the `## Usage` section above verbatim and stop. Do not run any commands, do not detect files, do not default the path to `.`. Just print the Usage block and return. + +**Fast path — existing graph:** Before doing anything else, check whether `graphify-out/graph.json` exists. The expected location is `graphify-out/graph.json` relative to the **current working directory** (i.e. the project root where you are running commands). If it exists AND the user's request is a natural-language question about the codebase (e.g. "How does X work?", "What calls Y?", "Trace the data flow through Z") and NOT an explicit rebuild command (`--update`, `--cluster-only`, or a bare path/URL that implies fresh extraction): **skip Steps 1–5 entirely and jump straight to `## For /graphify query`.** Run `graphify query ""` immediately. Do not run detect. Do not check corpus size. Do not ask the user to narrow. The graph is already built — use it. + +If no path was given, use `.` (current directory). Do not ask the user for a path. + +If the path argument starts with `https://github.com/` or `http://github.com/`, treat it as a GitHub URL - run Step 0 before anything else, then continue with the resolved local path. + +Follow these steps in order. Do not skip steps. + +### Step 0 - GitHub repos and multi-path merge (only if a URL or several paths) + +Only when the path is one or more `https://github.com/...` URLs, or several local subfolders to merge. See `references/github-and-merge.md` for the clone, cross-repo merge, and monorepo flow, then continue with the resolved local path. A plain local path skips this step. + +### Step 1 - Ensure graphify is installed + +```bash +# Detect the correct Python interpreter (handles uv tool, pipx, venv, system installs) +PYTHON="" +GRAPHIFY_BIN=$(which graphify 2>/dev/null) +# 1. uv tool installs — most reliable on modern Mac/Linux +if [ -z "$PYTHON" ] && command -v uv >/dev/null 2>&1; then + _UV_PY=$(uv tool run --from graphifyy python -c "import sys; print(sys.executable)" 2>/dev/null) + if [ -n "$_UV_PY" ]; then PYTHON="$_UV_PY"; fi +fi +# 2. Read shebang from graphify binary (pipx and direct pip installs) +if [ -z "$PYTHON" ] && [ -n "$GRAPHIFY_BIN" ]; then + _SHEBANG=$(head -1 "$GRAPHIFY_BIN" | tr -d '#!') + case "$_SHEBANG" in + *[!a-zA-Z0-9/_.@-]*) ;; + *) "$_SHEBANG" -c "import graphify" 2>/dev/null && PYTHON="$_SHEBANG" ;; + esac +fi +# 3. Fall back to python3 +if [ -z "$PYTHON" ]; then PYTHON="python3"; fi +if ! "$PYTHON" -c "import graphify" 2>/dev/null; then + if command -v uv >/dev/null 2>&1; then + uv tool install --upgrade graphifyy -q 2>&1 | tail -3 + _UV_PY=$(uv tool run --from graphifyy python -c "import sys; print(sys.executable)" 2>/dev/null) + if [ -n "$_UV_PY" ]; then PYTHON="$_UV_PY"; fi + else + "$PYTHON" -m pip install graphifyy -q 2>/dev/null \ + || "$PYTHON" -m pip install graphifyy -q --break-system-packages 2>&1 | tail -3 + fi +fi +# Write interpreter path for all subsequent steps (persists across invocations) +mkdir -p graphify-out +"$PYTHON" -c "import sys; open('graphify-out/.graphify_python', 'w', encoding='utf-8').write(sys.executable)" +# Save scan root so `graphify update` (no args) knows where to look next time +echo "$(cd INPUT_PATH && pwd)" > graphify-out/.graphify_root +``` + +If the import succeeds, print nothing and move straight to Step 2. + +**In every subsequent bash block, replace `python3` with `$(cat graphify-out/.graphify_python)` to use the correct interpreter.** + +### Step 2 - Detect files + +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from graphify.detect import detect +from pathlib import Path +result = detect(Path('INPUT_PATH')) +print(json.dumps(result, ensure_ascii=False)) +" > graphify-out/.graphify_detect.json +``` + +Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: + +``` +Corpus: X files · ~Y words + code: N files (.py .ts .go ...) + docs: N files (.md .txt ...) + papers: N files (.pdf ...) + images: N files + video: N files (.mp4 .mp3 ...) +``` + +Omit any category with 0 files from the summary. + +Then act on it: +- If `total_files` is 0: stop with "No supported files found in [path]." +- If `skipped_sensitive` is non-empty: report the count and list the skipped file names, so a wrongly-flagged source or doc is visible and can be renamed or moved (#2106). +- If `total_words` > 2,000,000 OR `total_files` > 500: show the warning. Then compute the top 5 first-level subdirectories by file count: + - Read `scan_root` from the detect JSON (always an absolute path to the resolved INPUT_PATH). + - Concatenate all file lists across all types (`code`, `document`, `paper`, `image`, `video`). + - Filter out any path that starts with `scan_root + "/graphify-out/"` to exclude converted sidecars. + - For each file, strip the `scan_root` prefix and take the first path component. Files directly in `scan_root` with no subdirectory count as `(root)`. + - If all files are in `(root)` with no subdirectories, do not ask to narrow — no subfolders exist. Instead suggest `--no-cluster` to skip the expensive clustering step and proceed. + - Otherwise rank by count, show the top 5 with file counts, then ask which subfolder to run on. Wait for the user's answer before proceeding. +- Otherwise: proceed directly to Step 2.5 if video files were detected, or Step 3 if not. + +### Step 2.5 - Video and audio (only if video files detected) + +Skip this step entirely if `detect` returned zero `video` files. When the corpus has video or audio, see `references/transcribe.md` to transcribe them to text first, then treat the transcripts as doc files in Step 3. + +### Step 3 - Extract entities and relationships + +**Before starting:** note whether `--mode deep` was given. You must pass `DEEP_MODE=true` to every subagent in Step B2 if it was. Track this from the original invocation - do not lose it. + +This step has two parts: **structural extraction** (deterministic, free) and **semantic extraction** (LLM, costs tokens). + +> **graphify needs no API key. Never ask the user for one, and never block on one.** Code is extracted structurally (AST) with no LLM and no key at all — a code-only corpus (the common `/graphify .` on a repo) skips semantic extraction entirely, so it needs nothing here: go straight to Part A and skip Part B. Semantic extraction (only for docs, papers, and images) uses Gemini **only if** `GEMINI_API_KEY`/`GOOGLE_API_KEY` is already set; otherwise the host agent itself is the LLM. graphify does **not** read `ANTHROPIC_API_KEY`, `OPENAI_API_KEY`, or any other provider key. If you catch yourself about to prompt for, wait on, or stop because of a missing API key, that is a misread of this skill — proceed without one. + +**Before semantic extraction:** check whether `GEMINI_API_KEY` or `GOOGLE_API_KEY` is set. If neither is set, print this one-liner to the user: +> Tip: set `GEMINI_API_KEY` or `GOOGLE_API_KEY` to use Gemini for semantic extraction (`pip install 'graphifyy[gemini]'`). + +Print it once, then continue — do not wait for the user to supply a key. If `GEMINI_API_KEY` or `GOOGLE_API_KEY` IS set, use `graphify.llm.extract_corpus_parallel(files, backend="gemini")` for semantic extraction instead of dispatching subagents. The default Gemini model is `gemini-3-flash-preview`; set `GRAPHIFY_GEMINI_MODEL` or pass `--model` in headless CLI flows to override it. + +> **No other API keys are read.** When `GEMINI_API_KEY`/`GOOGLE_API_KEY` are unset, semantic extraction falls to the host agent itself — the running session is the LLM. On a host that dispatches subagents (e.g. Claude Code), dispatch them as written in Part B. On a host that runs the CLI directly in a terminal and cannot dispatch subagents, do not stall: a code-only corpus has no semantic work, so write the empty semantic file (Part B "Fast path") and continue to Part C; for a corpus with docs/papers/images, either set a Gemini key or extract those inline yourself, but in no case prompt for `ANTHROPIC_API_KEY` — that prompt is a misread of this skill. + +**Run Part A (AST) and Part B (semantic) in parallel. Dispatch all semantic subagents AND start AST extraction in the same message. Both can run simultaneously since they operate on different file types. Merge results in Part C as before.** + +Note: Parallelizing AST + semantic saves 5-15s on large corpora. AST is deterministic and fast; start it while subagents are processing docs/papers. + +#### Part A - Structural extraction for code files + +For any code files detected, run AST extraction in parallel with Part B subagents: + +```bash +$(cat graphify-out/.graphify_python) -c " +import sys, json +from graphify.extract import collect_files, extract +from pathlib import Path +import json + +code_files = [] +detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) +for f in detect.get('files', {}).get('code', []): + code_files.extend(collect_files(Path(f)) if Path(f).is_dir() else [Path(f)]) + +if code_files: + result = extract(code_files, cache_root=Path('INPUT_PATH')) + Path('graphify-out/.graphify_ast.json').write_text(json.dumps(result, indent=2, ensure_ascii=False), encoding=\"utf-8\") + print(f'AST: {len(result[\"nodes\"])} nodes, {len(result[\"edges\"])} edges') +else: + Path('graphify-out/.graphify_ast.json').write_text(json.dumps({'nodes':[],'edges':[],'input_tokens':0,'output_tokens':0}, ensure_ascii=False), encoding=\"utf-8\") + print('No code files - skipping AST extraction') +" +``` + +#### Part B - Semantic extraction (parallel subagents) + +**Fast path:** If detection found zero docs, papers, and images (code-only corpus), skip Part B entirely and go straight to Part C. AST handles code - there is nothing for semantic subagents to do. **First write an empty semantic file** so Part C's merge has its input (it reads `.graphify_semantic.json` unconditionally; without this a code-only run hits `FileNotFoundError`): + +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path +Path('graphify-out/.graphify_semantic.json').write_text(json.dumps({'nodes':[],'edges':[],'hyperedges':[],'input_tokens':0,'output_tokens':0}), encoding='utf-8') +" +``` + +**MANDATORY: You MUST use the Agent tool here. Reading files yourself one-by-one is forbidden - it is 5-10x slower. If you do not use the Agent tool you are doing this wrong.** + +Before dispatching subagents, print a timing estimate: +- Load `total_words` and file counts from `graphify-out/.graphify_detect.json` +- Estimate agents needed: `ceil(uncached_non_code_files / 22)` (chunk size is 20-25) +- Estimate time: ~45s per agent batch (they run in parallel, so total ≈ 45s × ceil(agents/parallel_limit)) +- Print: "Semantic extraction: ~N files → X agents, estimated ~Ys" + +**Step B0 - Check extraction cache first** + +Before dispatching any subagents, check which files already have cached extraction results: + +SPEC_PATH below is the **absolute** path of the `references/extraction-spec.md` that ships beside this SKILL.md — the same file Step B2 loads and hands to every subagent. It is the extraction prompt, so cache entries are attributed to it: when a graphify upgrade changes the prompt, entries produced by the old one are re-extracted instead of replayed, and unchanged prompts keep their entries (#1939). Substitute the real path in both Step B0 and Step B3 — pass the same one to each, and do not drop the argument. + +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from graphify.cache import check_semantic_cache +from pathlib import Path + +detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) +# Only content files go to semantic extraction. Code is already covered structurally +# by the AST pass (Part A); flattening every category here makes subagents re-read +# every source file (#1392). Video is transcribed to a document in Step 2.5 first. +all_files = [f for cat in ('document', 'paper', 'image') for f in detect['files'].get(cat, [])] + +cached_nodes, cached_edges, cached_hyperedges, uncached = check_semantic_cache(all_files, root='INPUT_PATH', prompt_file='SPEC_PATH') + +# Always (re)write the cache file: write hits, else DELETE any leftover from a prior +# run so Part C never merges a stale .graphify_cached.json (#1392). +if cached_nodes or cached_edges or cached_hyperedges: + Path('graphify-out/.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges, 'hyperedges': cached_hyperedges}, ensure_ascii=False), encoding=\"utf-8\") +else: + Path('graphify-out/.graphify_cached.json').unlink(missing_ok=True) +Path('graphify-out/.graphify_uncached.txt').write_text('\n'.join(uncached), encoding=\"utf-8\") +print(f'Cache: {len(all_files)-len(uncached)} files hit, {len(uncached)} files need extraction') +" +``` + +Only dispatch subagents for files listed in `graphify-out/.graphify_uncached.txt`. If all files are cached, skip to Part C directly. + +**Step B1 - Split into chunks** + +Load files from `graphify-out/.graphify_uncached.txt`. Split into chunks of 20-25 files each. Each image gets its own chunk (vision needs separate context). When splitting, group files from the same directory together so related artifacts land in the same chunk and cross-file relationships are more likely to be extracted. + +**Step B2 - Dispatch ALL subagents in a single message** + +> Uses the `Task` tool for parallel subagent dispatch. +> Call `Task` once per chunk — ALL in the same response so they run in parallel. +> **On Mistral Vibe, dispatch each `Task` with `subagent_type="graphify-extract"`.** This writable subagent is installed by `graphify vibe install` alongside the skill. Do NOT use vibe's built-in `explore` subagent — it is read-only (`enabled_tools = ["grep", "read_file", "skill"]`) and cannot write the chunk JSON files this pipeline needs. If `graphify-extract` is missing (older install, or you overrode `VIBE_HOME` after install), re-run `graphify vibe install` first. + +Pass the extraction prompt as the task description: + +``` +Task(subagent_type="graphify-extract", description="Your task is to perform the following. Follow the instructions below exactly.\n\n\n[extraction prompt, with FILE_LIST, CHUNK_NUM, TOTAL_CHUNKS, DEEP_MODE substituted]\n\n\nExecute this now. Output ONLY the structured JSON response.") +``` + +Each subagent writes its result to its own `graphify-out/.graphify_chunk_NN.json`. Collect results as each `Task` completes and parse each as JSON. + +CHUNK_PATH must be an **absolute** path — derive it before dispatching: +```bash +PROJECT_ROOT=$(pwd) # cwd — where Part C globs graphify-out/ (NOT .graphify_root/scan dir, #1392) +# Then for chunk N: CHUNK_PATH="${PROJECT_ROOT}/graphify-out/.graphify_chunk_0N.json" +``` + +Subagent prompt template: + +See `references/extraction-spec.md` for the exact subagent prompt (JSON schema, node-ID rules, confidence rubric, hyperedge, and vision rules). Load it only here, only when at least one chunk holds a doc, paper, or image; a pure-code corpus has skipped Part B and never reads it. Pass each subagent that prompt verbatim with FILE_LIST, CHUNK_NUM, TOTAL_CHUNKS, DEEP_MODE, and CHUNK_PATH substituted, and have it write the result to CHUNK_PATH. + +**Step B3 - Collect, cache, and merge** + +Wait for all subagents. For each result: +- Check that `graphify-out/.graphify_chunk_NN.json` exists on disk — this is the success signal +- If the file exists and contains valid JSON with `nodes` and `edges`, include it and save to cache +- If the file is missing, the subagent was likely dispatched as read-only (vibe's built-in `explore`, or Claude Code's Explore type) — print a warning: "chunk N missing from disk — subagent may have been read-only. On vibe, re-dispatch with `subagent_type=\"graphify-extract\"` (installed by `graphify vibe install`). On Claude Code, re-dispatch with the general-purpose agent." Do not silently skip. +- If a subagent failed or returned invalid JSON, print a warning and skip that chunk - do not abort + +If more than half the chunks failed or are missing, stop and tell the user to re-run. On vibe, ensure `subagent_type="graphify-extract"` (installed by `graphify vibe install`) is passed to every `Task` call — the default `explore` subagent is read-only and will silently produce zero chunks. On Claude Code, ensure `subagent_type="general-purpose"` is used. + +Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent call completes, read the real token counts from the Agent tool result's `usage` field and write them back into the chunk JSON before merging** — the chunk JSON itself always has placeholder zeros. Then run: +```bash +$(cat graphify-out/.graphify_python) -c " +import json, glob +from pathlib import Path + +chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +all_nodes, all_edges, all_hyperedges = [], [], [] +total_in, total_out = 0, 0 +for c in chunks: + d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + all_nodes += d.get('nodes', []) + all_edges += d.get('edges', []) + all_hyperedges += d.get('hyperedges', []) + total_in += d.get('input_tokens', 0) + total_out += d.get('output_tokens', 0) +Path('graphify-out/.graphify_semantic_new.json').write_text(json.dumps({ + 'nodes': all_nodes, 'edges': all_edges, 'hyperedges': all_hyperedges, + 'input_tokens': total_in, 'output_tokens': total_out, +}, indent=2, ensure_ascii=False), encoding=\"utf-8\") +print(f'Merged {len(chunks)} chunks: {total_in:,} in / {total_out:,} out tokens') +" +``` + +Save new results to cache. Pass the same SPEC_PATH as Step B0 — it stamps each entry with the prompt that produced it, and a write under a different prompt than the read lands where the next run won't look (#1939): +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from graphify.cache import save_semantic_cache +from pathlib import Path + +new = json.loads(Path('graphify-out/.graphify_semantic_new.json').read_text(encoding=\"utf-8\")) if Path('graphify-out/.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] +saved = save_semantic_cache(new.get('nodes', []), new.get('edges', []), new.get('hyperedges', []), root='INPUT_PATH', allowed_source_files=uncached, prompt_file='SPEC_PATH') +print(f'Cached {saved} files') +" +``` + +Merge cached + new results into `graphify-out/.graphify_semantic.json`: +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path + +cached = json.loads(Path('graphify-out/.graphify_cached.json').read_text(encoding=\"utf-8\")) if Path('graphify-out/.graphify_cached.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} +new = json.loads(Path('graphify-out/.graphify_semantic_new.json').read_text(encoding=\"utf-8\")) if Path('graphify-out/.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} + +all_nodes = cached['nodes'] + new.get('nodes', []) +all_edges = cached['edges'] + new.get('edges', []) +all_hyperedges = cached.get('hyperedges', []) + new.get('hyperedges', []) +seen = set() +deduped = [] +for n in all_nodes: + if n['id'] not in seen: + seen.add(n['id']) + deduped.append(n) + +merged = { + 'nodes': deduped, + 'edges': all_edges, + 'hyperedges': all_hyperedges, + 'input_tokens': new.get('input_tokens', 0), + 'output_tokens': new.get('output_tokens', 0), +} +Path('graphify-out/.graphify_semantic.json').write_text(json.dumps(merged, indent=2, ensure_ascii=False), encoding=\"utf-8\") +print(f'Extraction complete - {len(deduped)} nodes, {len(all_edges)} edges ({len(cached[\"nodes\"])} from cache, {len(new.get(\"nodes\",[]))} new)') +" +``` +Clean up temp files: `rm -f graphify-out/.graphify_cached.json graphify-out/.graphify_uncached.txt graphify-out/.graphify_semantic_new.json` + +#### Part C - Merge AST + semantic into final extraction + +```bash +$(cat graphify-out/.graphify_python) -c " +import sys, json +from pathlib import Path + +ast = json.loads(Path('graphify-out/.graphify_ast.json').read_text(encoding=\"utf-8\")) +sem = json.loads(Path('graphify-out/.graphify_semantic.json').read_text(encoding=\"utf-8\")) + +# Merge: AST nodes first, semantic nodes deduplicated by id +seen = {n['id'] for n in ast['nodes']} +merged_nodes = list(ast['nodes']) +for n in sem['nodes']: + if n['id'] not in seen: + merged_nodes.append(n) + seen.add(n['id']) + +merged_edges = ast['edges'] + sem['edges'] +merged_hyperedges = sem.get('hyperedges', []) +merged = { + 'nodes': merged_nodes, + 'edges': merged_edges, + 'hyperedges': merged_hyperedges, + 'input_tokens': sem.get('input_tokens', 0), + 'output_tokens': sem.get('output_tokens', 0), +} +Path('graphify-out/.graphify_extract.json').write_text(json.dumps(merged, indent=2, ensure_ascii=False), encoding=\"utf-8\") +total = len(merged_nodes) +edges = len(merged_edges) +print(f'Merged: {total} nodes, {edges} edges ({len(ast[\"nodes\"])} AST + {len(sem[\"nodes\"])} semantic)') +" +``` + +### Step 4 - Build graph, cluster, analyze, generate outputs + +**Before starting:** the code blocks below pass `directed=IS_DIRECTED` to `build_from_json()`. Replace `IS_DIRECTED` with `True` if `--directed` was given (builds a `DiGraph` preserving edge direction source→target), otherwise `False` (the default undirected `Graph`). Substitute it the same way you substitute `INPUT_PATH` — do not leave the literal `IS_DIRECTED` in the code. + +```bash +mkdir -p graphify-out +$(cat graphify-out/.graphify_python) -c " +import sys, json +from graphify.build import build_from_json +from graphify.cluster import cluster, score_all +from graphify.analyze import god_nodes, surprising_connections, suggest_questions +from graphify.report import generate +from graphify.export import to_json +from pathlib import Path + +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) +detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) + +# root= mirrors the --update runbook (#1361): relativize source_file to the same +# base so the full build and incremental --update never drift apart on re-extract. +G = build_from_json(extraction, root='INPUT_PATH', directed=IS_DIRECTED) +# Guard BEFORE any write: an empty extraction must not clobber a good graph.json / +# GRAPH_REPORT.md / analysis sidecar. Check immediately after build (#1392). +if G.number_of_nodes() == 0: + print('ERROR: Graph is empty - extraction produced no nodes.') + print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.') + raise SystemExit(1) +communities = cluster(G) +cohesion = score_all(G, communities) +tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)} +gods = god_nodes(G) +surprises = surprising_connections(G, communities) +labels = {cid: 'Community ' + str(cid) for cid in communities} +# Placeholder questions - regenerated with real labels in Step 5 +questions = suggest_questions(G, communities, labels) + +# Export FIRST and honor the #479 shrink-guard: to_json returns False (writing +# nothing) when the new graph is smaller than the existing graph.json. Only write +# GRAPH_REPORT.md + the analysis sidecar when the graph was actually written, so +# they never describe a graph that graph.json doesn't contain (#1392). +wrote = to_json(G, communities, 'graphify-out/graph.json') +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') + raise SystemExit(1) +report = generate(G, communities, cohesion, labels, gods, surprises, detection, tokens, 'INPUT_PATH', suggested_questions=questions) +Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") +analysis = { + 'communities': {str(k): v for k, v in communities.items()}, + 'cohesion': {str(k): v for k, v in cohesion.items()}, + 'gods': gods, + 'surprises': surprises, + 'questions': questions, +} +Path('graphify-out/.graphify_analysis.json').write_text(json.dumps(analysis, indent=2, ensure_ascii=False), encoding=\"utf-8\") +print(f'Graph: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges, {len(communities)} communities') +" +``` + +If this step prints `ERROR: Graph is empty`, stop and tell the user what happened - do not proceed to labeling or visualization. + +Replace INPUT_PATH with the actual path. + +### Step 4.5 - Graph health check (read-only integrity gate) + +A non-destructive diagnostic on the extraction, before labeling. It surfaces edge collapse, dangling/missing endpoints, and self-loops — the silent-corruption modes of incremental updates and AST/LLM id mismatches. Read-only; never aborts. + +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path +from graphify.diagnostics import diagnose_extraction, format_diagnostic_report + +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) +summary = diagnose_extraction(extraction, directed=IS_DIRECTED, root='INPUT_PATH') +print(format_diagnostic_report(summary)) +flags = [f'{summary[k]} {label}' for k, label in ( + ('dangling_endpoint_edges', 'dangling-endpoint edges'), + ('missing_endpoint_edges', 'missing-endpoint edges'), + ('self_loop_edges', 'self-loop edges'), + ('directed_same_endpoint_collapsed_edges', 'collapsed (directed) edges'), + ('undirected_same_endpoint_collapsed_edges', 'collapsed (undirected) edges'), +) if summary.get(k, 0)] +print('GRAPH HEALTH WARNING: ' + '; '.join(flags) + ' - graph may be incomplete/corrupt.' if flags else 'Graph health: OK (no dangling/missing/collapsed edges).') +" +``` + +Substitute `IS_DIRECTED` and `INPUT_PATH` as in Step 4. If a `GRAPH HEALTH WARNING` prints, surface it in the final summary (do not abort — the graph is still usable, but the integrity issue must be visible, per the Honesty Rules). + +### Step 5 - Label communities + +Read `graphify-out/.graphify_analysis.json`. For each community key, look at its node labels and write a 2-5 word plain-language name (e.g. "Attention Mechanism", "Training Pipeline", "Data Loading"). + +Then regenerate the report and save the labels for the visualizer: + +```bash +$(cat graphify-out/.graphify_python) -c " +import sys, json +from graphify.build import build_from_json +from graphify.cluster import score_all +from graphify.analyze import god_nodes, surprising_connections, suggest_questions +from graphify.report import generate +from pathlib import Path + +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) +detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) +analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text(encoding=\"utf-8\")) + +# root= as in Step 4 / the --update runbook (#1361) — same base for node-key parity. +G = build_from_json(extraction, root='INPUT_PATH', directed=IS_DIRECTED) +communities = {int(k): v for k, v in analysis['communities'].items()} +cohesion = {int(k): v for k, v in analysis['cohesion'].items()} +tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)} + +# LABELS - replace these with the names you chose above +labels = LABELS_DICT + +# Regenerate questions with real community labels (labels affect question phrasing) +questions = suggest_questions(G, communities, labels) + +report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) +Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") +Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +print('Report updated with community labels') +" +``` + +Replace `LABELS_DICT` with the actual dict you constructed (e.g. `{0: "Attention Mechanism", 1: "Training Pipeline"}`). +Replace INPUT_PATH with the actual path. + +### Step 6 - Generate Obsidian vault (opt-in) + HTML + +**Generate HTML always** (unless `--no-viz`). **Obsidian vault only if `--obsidian` was explicitly given** — skip it otherwise, it generates one file per node. + +If `--obsidian` was given: + +- If `--obsidian-dir ` was also given, pass it via `--dir`. Otherwise defaults to `graphify-out/obsidian`. + +```bash +graphify export obsidian +# or with custom dir: graphify export obsidian --dir ~/vaults/my-project +``` + +Generate the HTML graph (always, unless `--no-viz`): + +```bash +graphify export html # auto-aggregates to community view if graph > 5000 nodes +# or: graphify export html --no-viz +``` + +### Steps 6b-8 - Wiki, Neo4j, FalkorDB, SVG, GraphML, MCP, benchmark (only on their flags) + +These run only when their flag is present (`--wiki`, `--neo4j`/`--neo4j-push`, `--falkordb`/`--falkordb-push`, `--svg`, `--graphml`, `--mcp`) or, for the token-reduction benchmark, when `total_words` exceeds 5,000. A default run with no export flags skips all of them. See `references/exports.md` for each one. Run any `--wiki` export before Step 9 cleanup so `.graphify_labels.json` is still available. + +--- + +### Step 9 - Save manifest, update cost tracker, clean up, and report + +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path +from datetime import datetime, timezone +from graphify.detect import save_manifest + +# Save manifest for --update +detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) +extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) +# In --update mode, 'all_files' carries the full corpus; 'files' is the changed +# subset. Full-rebuild mode populates only 'files', so the fallback handles that. +# root= relativizes the manifest keys to the scan root (same base as the build), +# so the on-disk manifest is portable across clones/machines and a later --update +# matches cached files instead of missing every one (#1417). +# +# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output: +# a detected file whose chunk failed or was omitted must stay unstamped so the +# next --update re-queues it, otherwise it is marked done and its content is lost +# forever (#2015). This mirrors the library extract path exactly +# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the +# raw corpus. Code files are always stamped (AST is deterministic); only semantic +# types are gated on output. +from graphify.cli import _stamped_manifest_files +_corpus = detect.get('all_files') or detect['files'] +_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH')) +# Files dispatched this run (the changed subset) but NOT stamped above still carry +# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues +# them instead of reading them as unchanged (#1948). +_sem_types = ('document', 'paper', 'image') +_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl} +_stamped = {f for fl in _manifest_files.values() for f in fl} +_cleared = _dispatched - _stamped +# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root +# files newly excluded since last run are dropped rather than masquerading as +# deletions; untouched files' prior rows are still preserved (#1908). +_scan = {f for fl in _corpus.values() for f in fl} +save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None) + +# Update cumulative cost tracker +input_tok = extract.get('input_tokens', 0) +output_tok = extract.get('output_tokens', 0) + +cost_path = Path('graphify-out/cost.json') +if cost_path.exists(): + cost = json.loads(cost_path.read_text(encoding=\"utf-8\")) +else: + cost = {'runs': [], 'total_input_tokens': 0, 'total_output_tokens': 0} + +cost['runs'].append({ + 'date': datetime.now(timezone.utc).isoformat(), + 'input_tokens': input_tok, + 'output_tokens': output_tok, + 'files': detect.get('total_files', 0), +}) +cost['total_input_tokens'] += input_tok +cost['total_output_tokens'] += output_tok +cost_path.write_text(json.dumps(cost, indent=2, ensure_ascii=False), encoding=\"utf-8\") + +print(f'This run: {input_tok:,} input tokens, {output_tok:,} output tokens') +print(f'All time: {cost[\"total_input_tokens\"]:,} input, {cost[\"total_output_tokens\"]:,} output ({len(cost[\"runs\"])} runs)') +" +rm -f graphify-out/.graphify_detect.json graphify-out/.graphify_extract.json graphify-out/.graphify_ast.json graphify-out/.graphify_semantic.json graphify-out/.graphify_analysis.json +find graphify-out -maxdepth 1 -name '.graphify_chunk_*.json' -delete 2>/dev/null +rm -f graphify-out/.needs_update 2>/dev/null || true +``` + +Replace INPUT_PATH with the actual path (same value used in Steps 4-5) so the manifest is relativized to the scan root. + +Tell the user (omit the obsidian line unless --obsidian was given): +``` +Graph complete. Outputs in PATH_TO_DIR/graphify-out/ + + graph.html - interactive graph, open in browser + GRAPH_REPORT.md - audit report + graph.json - raw graph data + obsidian/ - Obsidian vault (only if --obsidian was given) +``` + +If graphify saved you time, consider supporting it: https://github.com/sponsors/safishamsi + +Replace PATH_TO_DIR with the actual absolute path of the directory that was processed. + +Then paste these sections from GRAPH_REPORT.md directly into the chat: +- God Nodes +- Surprising Connections +- Suggested Questions + +Do NOT paste the full report - just those three sections. Keep it concise. + +Then immediately offer to explore. Pick the single most interesting suggested question from the report - the one that crosses the most community boundaries or has the most surprising bridge node - and ask: + +> "The most interesting question this graph can answer: **[question]**. Want me to trace it?" + +If the user says yes, run `/graphify query "[question]"` on the graph and walk them through the answer using the graph structure - which nodes connect, which community boundaries get crossed, what the path reveals. Keep going as long as they want to explore. Each answer should end with a natural follow-up ("this connects to X - want to go deeper?") so the session feels like navigation, not a one-shot report. + +The graph is the map. Your job after the pipeline is to be the guide. + +--- + +## Interpreter guard for subcommands + +Before running any subcommand below (`--update`, `--cluster-only`, `query`, `path`, `explain`, `add`), check that `.graphify_python` exists. If it's missing (e.g. user deleted `graphify-out/`), re-resolve the interpreter first: + +```bash +if [ ! -f graphify-out/.graphify_python ]; then + GRAPHIFY_BIN=$(which graphify 2>/dev/null) + if [ -n "$GRAPHIFY_BIN" ]; then + PYTHON=$(head -1 "$GRAPHIFY_BIN" | tr -d '#!') + case "$PYTHON" in *[!a-zA-Z0-9/_.@-]*) PYTHON="python3" ;; esac + else + PYTHON="python3" + fi + mkdir -p graphify-out + "$PYTHON" -c "import sys; open('graphify-out/.graphify_python', 'w', encoding='utf-8').write(sys.executable)" +fi +``` + +## For --update and --cluster-only + +Both are non-default subcommands. `--update` re-extracts only new or changed files; `--cluster-only` reruns clustering on the existing graph. See `references/update.md` for both flows. + +--- + +## For /graphify query + +When `graphify-out/graph.json` already exists and the user asks a question about the corpus, answer from the graph rather than rebuilding it: + +```bash +graphify query "" +``` + +Before traversal, expand the question against the graph's own vocabulary so a wording mismatch does not collapse the answer to noise. If the `graphify query` CLI is unavailable, fall back to an inline NetworkX traversal of `graphify-out/graph.json`. Answer using only what the graph output contains, and quote `source_location` when citing a specific fact. For that vocab-expansion step, the BFS/DFS traversal modes, the `--budget` cap, the NetworkX fallback, `save-result` feedback, and the `/graphify path` and `/graphify explain` flows, see `references/query.md`. + +--- + +## For /graphify add and --watch + +Neither is part of the default build. When the user runs `/graphify add ` to fetch a URL into the corpus, or passes `--watch` to auto-rebuild on file changes, see `references/add-watch.md`. + +--- + +## For the commit hook and native AGENTS.md integration + +When the user asks to install the post-commit auto-rebuild hook or wire graphify into a project's AGENTS.md, see `references/hooks.md`. + +--- + +## Honesty Rules + +- Never invent an edge. If unsure, use AMBIGUOUS. +- Never skip the corpus check warning. +- Always show token cost in the report. +- Never hide cohesion scores behind symbols - show the raw number. +- Never run HTML viz on a graph with more than 5,000 nodes without warning the user. diff --git a/tools/skillgen/fragments/core/vibe.md b/tools/skillgen/fragments/core/vibe.md new file mode 100644 index 0000000000..41e754a47c --- /dev/null +++ b/tools/skillgen/fragments/core/vibe.md @@ -0,0 +1,710 @@ +--- +name: graphify +description: "Use for any question about a codebase, its architecture, file relationships, or project content — especially when graphify-out/ exists, where the question should be treated as a graphify query first. Turns any input (code, docs, papers, images, videos) into a persistent knowledge graph with god nodes, community detection, and query/path/explain tools." +user-invocable: true +allowed-tools: + - bash + - read_file + - grep + - write_file + - edit + - task + - ask_user_question +--- + + +# /graphify + +Turn any folder of files into a navigable knowledge graph with community detection, an honest audit trail, and three outputs: interactive HTML, GraphRAG-ready JSON, and a plain-language GRAPH_REPORT.md. + +## Usage + +``` +/graphify # full pipeline on current directory (HTML viz; add --obsidian for a vault) +/graphify # full pipeline on specific path +/graphify https://github.com// # clone repo then run full pipeline on it +/graphify https://github.com// --branch # clone a specific branch +/graphify ... # clone multiple repos, build each, merge into one cross-repo graph +/graphify --mode deep # thorough extraction, richer INFERRED edges +/graphify --update # incremental - re-extract only new/changed files +/graphify --directed # build directed graph (preserves edge direction: source→target) +/graphify --whisper-model medium # use a larger Whisper model for better transcription accuracy +/graphify --cluster-only # rerun clustering on existing graph +/graphify --no-viz # skip visualization, just report + JSON +/graphify --html # (HTML is generated by default - this flag is a no-op) +/graphify --svg # also export graph.svg (embeds in Notion, GitHub) +/graphify --graphml # export graph.graphml (Gephi, yEd) +/graphify --neo4j # generate graphify-out/cypher.txt for Neo4j +/graphify --neo4j-push bolt://localhost:7687 # push directly to Neo4j +/graphify --falkordb # generate graphify-out/cypher.txt for FalkorDB +/graphify --falkordb-push falkordb://localhost:6379 # push directly to FalkorDB +/graphify --mcp # start MCP stdio server for agent access +/graphify --watch # watch folder, auto-rebuild on code changes (no LLM needed) +/graphify --wiki # build agent-crawlable wiki (index.md + one article per community) +/graphify --obsidian --obsidian-dir ~/vaults/my-project # write vault to custom path (e.g. existing vault) +/graphify add # fetch URL, save to ./raw, update graph +/graphify add --author "Name" # tag who wrote it +/graphify add --contributor "Name" # tag who added it to the corpus +/graphify query "" # BFS traversal - broad context +/graphify query "" --dfs # DFS - trace a specific path +/graphify query "" --budget 1500 # cap answer at N tokens +/graphify path "AuthModule" "Database" # shortest path between two concepts +/graphify explain "SwinTransformer" # plain-language explanation of a node +``` + +## What graphify is for + +Drop any folder of code, docs, papers, images, or video into graphify and get a queryable knowledge graph. Persistent across sessions, honest audit trail (EXTRACTED/INFERRED/AMBIGUOUS), community detection surfaces cross-document connections you wouldn't think to ask about. + +## What You Must Do When Invoked + +If the user invoked `/graphify --help` or `/graphify -h` (with no other arguments), print the contents of the `## Usage` section above verbatim and stop. Do not run any commands, do not detect files, do not default the path to `.`. Just print the Usage block and return. + +**Fast path — existing graph:** Before doing anything else, check whether `graphify-out/graph.json` exists. The expected location is `graphify-out/graph.json` relative to the **current working directory** (i.e. the project root where you are running commands). If it exists AND the user's request is a natural-language question about the codebase (e.g. "How does X work?", "What calls Y?", "Trace the data flow through Z") and NOT an explicit rebuild command (`--update`, `--cluster-only`, or a bare path/URL that implies fresh extraction): **skip Steps 1–5 entirely and jump straight to `## For /graphify query`.** Run `graphify query ""` immediately. Do not run detect. Do not check corpus size. Do not ask the user to narrow. The graph is already built — use it. + +If no path was given, use `.` (current directory). Do not ask the user for a path. + +If the path argument starts with `https://github.com/` or `http://github.com/`, treat it as a GitHub URL - run Step 0 before anything else, then continue with the resolved local path. + +Follow these steps in order. Do not skip steps. + +### Step 0 - GitHub repos and multi-path merge (only if a URL or several paths) + +Only when the path is one or more `https://github.com/...` URLs, or several local subfolders to merge. See `references/github-and-merge.md` for the clone, cross-repo merge, and monorepo flow, then continue with the resolved local path. A plain local path skips this step. + +### Step 1 - Ensure graphify is installed + +```bash +# Detect the correct Python interpreter (handles uv tool, pipx, venv, system installs) +PYTHON="" +GRAPHIFY_BIN=$(which graphify 2>/dev/null) +# 1. uv tool installs — most reliable on modern Mac/Linux +if [ -z "$PYTHON" ] && command -v uv >/dev/null 2>&1; then + _UV_PY=$(uv tool run --from graphifyy python -c "import sys; print(sys.executable)" 2>/dev/null) + if [ -n "$_UV_PY" ]; then PYTHON="$_UV_PY"; fi +fi +# 2. Read shebang from graphify binary (pipx and direct pip installs) +if [ -z "$PYTHON" ] && [ -n "$GRAPHIFY_BIN" ]; then + _SHEBANG=$(head -1 "$GRAPHIFY_BIN" | tr -d '#!') + case "$_SHEBANG" in + *[!a-zA-Z0-9/_.@-]*) ;; + *) "$_SHEBANG" -c "import graphify" 2>/dev/null && PYTHON="$_SHEBANG" ;; + esac +fi +# 3. Fall back to python3 +if [ -z "$PYTHON" ]; then PYTHON="python3"; fi +if ! "$PYTHON" -c "import graphify" 2>/dev/null; then + if command -v uv >/dev/null 2>&1; then + uv tool install --upgrade graphifyy -q 2>&1 | tail -3 + _UV_PY=$(uv tool run --from graphifyy python -c "import sys; print(sys.executable)" 2>/dev/null) + if [ -n "$_UV_PY" ]; then PYTHON="$_UV_PY"; fi + else + "$PYTHON" -m pip install graphifyy -q 2>/dev/null \ + || "$PYTHON" -m pip install graphifyy -q --break-system-packages 2>&1 | tail -3 + fi +fi +# Write interpreter path for all subsequent steps (persists across invocations) +mkdir -p graphify-out +"$PYTHON" -c "import sys; open('graphify-out/.graphify_python', 'w', encoding='utf-8').write(sys.executable)" +# Save scan root so `graphify update` (no args) knows where to look next time +echo "$(cd INPUT_PATH && pwd)" > graphify-out/.graphify_root +``` + +If the import succeeds, print nothing and move straight to Step 2. + +**In every subsequent bash block, replace `python3` with `$(cat graphify-out/.graphify_python)` to use the correct interpreter.** + +### Step 2 - Detect files + +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from graphify.detect import detect +from pathlib import Path +result = detect(Path('INPUT_PATH')) +print(json.dumps(result, ensure_ascii=False)) +" > graphify-out/.graphify_detect.json +``` + +Replace INPUT_PATH with the actual path the user provided. Do NOT cat or print the JSON - read it silently and present a clean summary instead: + +``` +Corpus: X files · ~Y words + code: N files (.py .ts .go ...) + docs: N files (.md .txt ...) + papers: N files (.pdf ...) + images: N files + video: N files (.mp4 .mp3 ...) +``` + +Omit any category with 0 files from the summary. + +Then act on it: +- If `total_files` is 0: stop with "No supported files found in [path]." +- If `skipped_sensitive` is non-empty: report the count and list the skipped file names, so a wrongly-flagged source or doc is visible and can be renamed or moved (#2106). +- If `total_words` > 2,000,000 OR `total_files` > 500: show the warning. Then compute the top 5 first-level subdirectories by file count: + - Read `scan_root` from the detect JSON (always an absolute path to the resolved INPUT_PATH). + - Concatenate all file lists across all types (`code`, `document`, `paper`, `image`, `video`). + - Filter out any path that starts with `scan_root + "/graphify-out/"` to exclude converted sidecars. + - For each file, strip the `scan_root` prefix and take the first path component. Files directly in `scan_root` with no subdirectory count as `(root)`. + - If all files are in `(root)` with no subdirectories, do not ask to narrow — no subfolders exist. Instead suggest `--no-cluster` to skip the expensive clustering step and proceed. + - Otherwise rank by count, show the top 5 with file counts, then ask which subfolder to run on. Wait for the user's answer before proceeding. +- Otherwise: proceed directly to Step 2.5 if video files were detected, or Step 3 if not. + +### Step 2.5 - Video and audio (only if video files detected) + +Skip this step entirely if `detect` returned zero `video` files. When the corpus has video or audio, see `references/transcribe.md` to transcribe them to text first, then treat the transcripts as doc files in Step 3. + +### Step 3 - Extract entities and relationships + +**Before starting:** note whether `--mode deep` was given. You must pass `DEEP_MODE=true` to every subagent in Step B2 if it was. Track this from the original invocation - do not lose it. + +This step has two parts: **structural extraction** (deterministic, free) and **semantic extraction** (LLM, costs tokens). + +> **graphify needs no API key. Never ask the user for one, and never block on one.** Code is extracted structurally (AST) with no LLM and no key at all — a code-only corpus (the common `/graphify .` on a repo) skips semantic extraction entirely, so it needs nothing here: go straight to Part A and skip Part B. Semantic extraction (only for docs, papers, and images) uses Gemini **only if** `GEMINI_API_KEY`/`GOOGLE_API_KEY` is already set; otherwise the host agent itself is the LLM. graphify does **not** read `ANTHROPIC_API_KEY`, `OPENAI_API_KEY`, or any other provider key. If you catch yourself about to prompt for, wait on, or stop because of a missing API key, that is a misread of this skill — proceed without one. + +**Before semantic extraction:** check whether `GEMINI_API_KEY` or `GOOGLE_API_KEY` is set. If neither is set, print this one-liner to the user: +> Tip: set `GEMINI_API_KEY` or `GOOGLE_API_KEY` to use Gemini for semantic extraction (`pip install 'graphifyy[gemini]'`). + +Print it once, then continue — do not wait for the user to supply a key. If `GEMINI_API_KEY` or `GOOGLE_API_KEY` IS set, use `graphify.llm.extract_corpus_parallel(files, backend="gemini")` for semantic extraction instead of dispatching subagents. The default Gemini model is `gemini-3-flash-preview`; set `GRAPHIFY_GEMINI_MODEL` or pass `--model` in headless CLI flows to override it. + +> **No other API keys are read.** When `GEMINI_API_KEY`/`GOOGLE_API_KEY` are unset, semantic extraction falls to the host agent itself — the running session is the LLM. On a host that dispatches subagents (e.g. Claude Code), dispatch them as written in Part B. On a host that runs the CLI directly in a terminal and cannot dispatch subagents, do not stall: a code-only corpus has no semantic work, so write the empty semantic file (Part B "Fast path") and continue to Part C; for a corpus with docs/papers/images, either set a Gemini key or extract those inline yourself, but in no case prompt for `ANTHROPIC_API_KEY` — that prompt is a misread of this skill. + +**Run Part A (AST) and Part B (semantic) in parallel. Dispatch all semantic subagents AND start AST extraction in the same message. Both can run simultaneously since they operate on different file types. Merge results in Part C as before.** + +Note: Parallelizing AST + semantic saves 5-15s on large corpora. AST is deterministic and fast; start it while subagents are processing docs/papers. + +#### Part A - Structural extraction for code files + +For any code files detected, run AST extraction in parallel with Part B subagents: + +```bash +$(cat graphify-out/.graphify_python) -c " +import sys, json +from graphify.extract import collect_files, extract +from pathlib import Path +import json + +code_files = [] +detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) +for f in detect.get('files', {}).get('code', []): + code_files.extend(collect_files(Path(f)) if Path(f).is_dir() else [Path(f)]) + +if code_files: + result = extract(code_files, cache_root=Path('INPUT_PATH')) + Path('graphify-out/.graphify_ast.json').write_text(json.dumps(result, indent=2, ensure_ascii=False), encoding=\"utf-8\") + print(f'AST: {len(result[\"nodes\"])} nodes, {len(result[\"edges\"])} edges') +else: + Path('graphify-out/.graphify_ast.json').write_text(json.dumps({'nodes':[],'edges':[],'input_tokens':0,'output_tokens':0}, ensure_ascii=False), encoding=\"utf-8\") + print('No code files - skipping AST extraction') +" +``` + +#### Part B - Semantic extraction (parallel subagents) + +**Fast path:** If detection found zero docs, papers, and images (code-only corpus), skip Part B entirely and go straight to Part C. AST handles code - there is nothing for semantic subagents to do. **First write an empty semantic file** so Part C's merge has its input (it reads `.graphify_semantic.json` unconditionally; without this a code-only run hits `FileNotFoundError`): + +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path +Path('graphify-out/.graphify_semantic.json').write_text(json.dumps({'nodes':[],'edges':[],'hyperedges':[],'input_tokens':0,'output_tokens':0}), encoding='utf-8') +" +``` + +**MANDATORY: You MUST use the Agent tool here. Reading files yourself one-by-one is forbidden - it is 5-10x slower. If you do not use the Agent tool you are doing this wrong.** + +Before dispatching subagents, print a timing estimate: +- Load `total_words` and file counts from `graphify-out/.graphify_detect.json` +- Estimate agents needed: `ceil(uncached_non_code_files / 22)` (chunk size is 20-25) +- Estimate time: ~45s per agent batch (they run in parallel, so total ≈ 45s × ceil(agents/parallel_limit)) +- Print: "Semantic extraction: ~N files → X agents, estimated ~Ys" + +**Step B0 - Check extraction cache first** + +Before dispatching any subagents, check which files already have cached extraction results: + +SPEC_PATH below is the **absolute** path of the `references/extraction-spec.md` that ships beside this SKILL.md — the same file Step B2 loads and hands to every subagent. It is the extraction prompt, so cache entries are attributed to it: when a graphify upgrade changes the prompt, entries produced by the old one are re-extracted instead of replayed, and unchanged prompts keep their entries (#1939). Substitute the real path in both Step B0 and Step B3 — pass the same one to each, and do not drop the argument. + +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from graphify.cache import check_semantic_cache +from pathlib import Path + +detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) +# Only content files go to semantic extraction. Code is already covered structurally +# by the AST pass (Part A); flattening every category here makes subagents re-read +# every source file (#1392). Video is transcribed to a document in Step 2.5 first. +all_files = [f for cat in ('document', 'paper', 'image') for f in detect['files'].get(cat, [])] + +cached_nodes, cached_edges, cached_hyperedges, uncached = check_semantic_cache(all_files, root='INPUT_PATH', prompt_file='SPEC_PATH') + +# Always (re)write the cache file: write hits, else DELETE any leftover from a prior +# run so Part C never merges a stale .graphify_cached.json (#1392). +if cached_nodes or cached_edges or cached_hyperedges: + Path('graphify-out/.graphify_cached.json').write_text(json.dumps({'nodes': cached_nodes, 'edges': cached_edges, 'hyperedges': cached_hyperedges}, ensure_ascii=False), encoding=\"utf-8\") +else: + Path('graphify-out/.graphify_cached.json').unlink(missing_ok=True) +Path('graphify-out/.graphify_uncached.txt').write_text('\n'.join(uncached), encoding=\"utf-8\") +print(f'Cache: {len(all_files)-len(uncached)} files hit, {len(uncached)} files need extraction') +" +``` + +Only dispatch subagents for files listed in `graphify-out/.graphify_uncached.txt`. If all files are cached, skip to Part C directly. + +**Step B1 - Split into chunks** + +Load files from `graphify-out/.graphify_uncached.txt`. Split into chunks of 20-25 files each. Each image gets its own chunk (vision needs separate context). When splitting, group files from the same directory together so related artifacts land in the same chunk and cross-file relationships are more likely to be extracted. + +**Step B2 - Dispatch ALL subagents in a single message** + +> Uses the `Task` tool for parallel subagent dispatch. +> Call `Task` once per chunk — ALL in the same response so they run in parallel. +> **On Mistral Vibe, dispatch each `Task` with `subagent_type="graphify-extract"`.** This writable subagent is installed by `graphify vibe install` alongside the skill. Do NOT use vibe's built-in `explore` subagent — it is read-only (`enabled_tools = ["grep", "read_file", "skill"]`) and cannot write the chunk JSON files this pipeline needs. If `graphify-extract` is missing (older install, or you overrode `VIBE_HOME` after install), re-run `graphify vibe install` first. + +Pass the extraction prompt as the task description: + +``` +Task(subagent_type="graphify-extract", description="Your task is to perform the following. Follow the instructions below exactly.\n\n\n[extraction prompt, with FILE_LIST, CHUNK_NUM, TOTAL_CHUNKS, DEEP_MODE substituted]\n\n\nExecute this now. Output ONLY the structured JSON response.") +``` + +Each subagent writes its result to its own `graphify-out/.graphify_chunk_NN.json`. Collect results as each `Task` completes and parse each as JSON. + +CHUNK_PATH must be an **absolute** path — derive it before dispatching: +```bash +PROJECT_ROOT=$(pwd) # cwd — where Part C globs graphify-out/ (NOT .graphify_root/scan dir, #1392) +# Then for chunk N: CHUNK_PATH="${PROJECT_ROOT}/graphify-out/.graphify_chunk_0N.json" +``` + +Subagent prompt template: + +See `references/extraction-spec.md` for the exact subagent prompt (JSON schema, node-ID rules, confidence rubric, hyperedge, and vision rules). Load it only here, only when at least one chunk holds a doc, paper, or image; a pure-code corpus has skipped Part B and never reads it. Pass each subagent that prompt verbatim with FILE_LIST, CHUNK_NUM, TOTAL_CHUNKS, DEEP_MODE, and CHUNK_PATH substituted, and have it write the result to CHUNK_PATH. + +**Step B3 - Collect, cache, and merge** + +Wait for all subagents. For each result: +- Check that `graphify-out/.graphify_chunk_NN.json` exists on disk — this is the success signal +- If the file exists and contains valid JSON with `nodes` and `edges`, include it and save to cache +- If the file is missing, the subagent was likely dispatched as read-only (vibe's built-in `explore`, or Claude Code's Explore type) — print a warning: "chunk N missing from disk — subagent may have been read-only. On vibe, re-dispatch with `subagent_type=\"graphify-extract\"` (installed by `graphify vibe install`). On Claude Code, re-dispatch with the general-purpose agent." Do not silently skip. +- If a subagent failed or returned invalid JSON, print a warning and skip that chunk - do not abort + +If more than half the chunks failed or are missing, stop and tell the user to re-run. On vibe, ensure `subagent_type="graphify-extract"` (installed by `graphify vibe install`) is passed to every `Task` call — the default `explore` subagent is read-only and will silently produce zero chunks. On Claude Code, ensure `subagent_type="general-purpose"` is used. + +Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent call completes, read the real token counts from the Agent tool result's `usage` field and write them back into the chunk JSON before merging** — the chunk JSON itself always has placeholder zeros. Then run: +```bash +$(cat graphify-out/.graphify_python) -c " +import json, glob +from pathlib import Path + +chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +all_nodes, all_edges, all_hyperedges = [], [], [] +total_in, total_out = 0, 0 +for c in chunks: + d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + all_nodes += d.get('nodes', []) + all_edges += d.get('edges', []) + all_hyperedges += d.get('hyperedges', []) + total_in += d.get('input_tokens', 0) + total_out += d.get('output_tokens', 0) +Path('graphify-out/.graphify_semantic_new.json').write_text(json.dumps({ + 'nodes': all_nodes, 'edges': all_edges, 'hyperedges': all_hyperedges, + 'input_tokens': total_in, 'output_tokens': total_out, +}, indent=2, ensure_ascii=False), encoding=\"utf-8\") +print(f'Merged {len(chunks)} chunks: {total_in:,} in / {total_out:,} out tokens') +" +``` + +Save new results to cache. Pass the same SPEC_PATH as Step B0 — it stamps each entry with the prompt that produced it, and a write under a different prompt than the read lands where the next run won't look (#1939): +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from graphify.cache import save_semantic_cache +from pathlib import Path + +new = json.loads(Path('graphify-out/.graphify_semantic_new.json').read_text(encoding=\"utf-8\")) if Path('graphify-out/.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] +saved = save_semantic_cache(new.get('nodes', []), new.get('edges', []), new.get('hyperedges', []), root='INPUT_PATH', allowed_source_files=uncached, prompt_file='SPEC_PATH') +print(f'Cached {saved} files') +" +``` + +Merge cached + new results into `graphify-out/.graphify_semantic.json`: +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path + +cached = json.loads(Path('graphify-out/.graphify_cached.json').read_text(encoding=\"utf-8\")) if Path('graphify-out/.graphify_cached.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} +new = json.loads(Path('graphify-out/.graphify_semantic_new.json').read_text(encoding=\"utf-8\")) if Path('graphify-out/.graphify_semantic_new.json').exists() else {'nodes':[],'edges':[],'hyperedges':[]} + +all_nodes = cached['nodes'] + new.get('nodes', []) +all_edges = cached['edges'] + new.get('edges', []) +all_hyperedges = cached.get('hyperedges', []) + new.get('hyperedges', []) +seen = set() +deduped = [] +for n in all_nodes: + if n['id'] not in seen: + seen.add(n['id']) + deduped.append(n) + +merged = { + 'nodes': deduped, + 'edges': all_edges, + 'hyperedges': all_hyperedges, + 'input_tokens': new.get('input_tokens', 0), + 'output_tokens': new.get('output_tokens', 0), +} +Path('graphify-out/.graphify_semantic.json').write_text(json.dumps(merged, indent=2, ensure_ascii=False), encoding=\"utf-8\") +print(f'Extraction complete - {len(deduped)} nodes, {len(all_edges)} edges ({len(cached[\"nodes\"])} from cache, {len(new.get(\"nodes\",[]))} new)') +" +``` +Clean up temp files: `rm -f graphify-out/.graphify_cached.json graphify-out/.graphify_uncached.txt graphify-out/.graphify_semantic_new.json` + +#### Part C - Merge AST + semantic into final extraction + +```bash +$(cat graphify-out/.graphify_python) -c " +import sys, json +from pathlib import Path + +ast = json.loads(Path('graphify-out/.graphify_ast.json').read_text(encoding=\"utf-8\")) +sem = json.loads(Path('graphify-out/.graphify_semantic.json').read_text(encoding=\"utf-8\")) + +# Merge: AST nodes first, semantic nodes deduplicated by id +seen = {n['id'] for n in ast['nodes']} +merged_nodes = list(ast['nodes']) +for n in sem['nodes']: + if n['id'] not in seen: + merged_nodes.append(n) + seen.add(n['id']) + +merged_edges = ast['edges'] + sem['edges'] +merged_hyperedges = sem.get('hyperedges', []) +merged = { + 'nodes': merged_nodes, + 'edges': merged_edges, + 'hyperedges': merged_hyperedges, + 'input_tokens': sem.get('input_tokens', 0), + 'output_tokens': sem.get('output_tokens', 0), +} +Path('graphify-out/.graphify_extract.json').write_text(json.dumps(merged, indent=2, ensure_ascii=False), encoding=\"utf-8\") +total = len(merged_nodes) +edges = len(merged_edges) +print(f'Merged: {total} nodes, {edges} edges ({len(ast[\"nodes\"])} AST + {len(sem[\"nodes\"])} semantic)') +" +``` + +### Step 4 - Build graph, cluster, analyze, generate outputs + +**Before starting:** the code blocks below pass `directed=IS_DIRECTED` to `build_from_json()`. Replace `IS_DIRECTED` with `True` if `--directed` was given (builds a `DiGraph` preserving edge direction source→target), otherwise `False` (the default undirected `Graph`). Substitute it the same way you substitute `INPUT_PATH` — do not leave the literal `IS_DIRECTED` in the code. + +```bash +mkdir -p graphify-out +$(cat graphify-out/.graphify_python) -c " +import sys, json +from graphify.build import build_from_json +from graphify.cluster import cluster, score_all +from graphify.analyze import god_nodes, surprising_connections, suggest_questions +from graphify.report import generate +from graphify.export import to_json +from pathlib import Path + +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) +detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) + +# root= mirrors the --update runbook (#1361): relativize source_file to the same +# base so the full build and incremental --update never drift apart on re-extract. +G = build_from_json(extraction, root='INPUT_PATH', directed=IS_DIRECTED) +# Guard BEFORE any write: an empty extraction must not clobber a good graph.json / +# GRAPH_REPORT.md / analysis sidecar. Check immediately after build (#1392). +if G.number_of_nodes() == 0: + print('ERROR: Graph is empty - extraction produced no nodes.') + print('Possible causes: all files were skipped, binary-only corpus, or extraction failed.') + raise SystemExit(1) +communities = cluster(G) +cohesion = score_all(G, communities) +tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)} +gods = god_nodes(G) +surprises = surprising_connections(G, communities) +labels = {cid: 'Community ' + str(cid) for cid in communities} +# Placeholder questions - regenerated with real labels in Step 5 +questions = suggest_questions(G, communities, labels) + +# Export FIRST and honor the #479 shrink-guard: to_json returns False (writing +# nothing) when the new graph is smaller than the existing graph.json. Only write +# GRAPH_REPORT.md + the analysis sidecar when the graph was actually written, so +# they never describe a graph that graph.json doesn't contain (#1392). +wrote = to_json(G, communities, 'graphify-out/graph.json') +if not wrote: + print('ERROR: refused to shrink graphify-out/graph.json (existing graph has more nodes; #479).') + print('If this shrink is intentional (you deleted files), re-run a full build with --force.') + raise SystemExit(1) +report = generate(G, communities, cohesion, labels, gods, surprises, detection, tokens, 'INPUT_PATH', suggested_questions=questions) +Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") +analysis = { + 'communities': {str(k): v for k, v in communities.items()}, + 'cohesion': {str(k): v for k, v in cohesion.items()}, + 'gods': gods, + 'surprises': surprises, + 'questions': questions, +} +Path('graphify-out/.graphify_analysis.json').write_text(json.dumps(analysis, indent=2, ensure_ascii=False), encoding=\"utf-8\") +print(f'Graph: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges, {len(communities)} communities') +" +``` + +If this step prints `ERROR: Graph is empty`, stop and tell the user what happened - do not proceed to labeling or visualization. + +Replace INPUT_PATH with the actual path. + +### Step 4.5 - Graph health check (read-only integrity gate) + +A non-destructive diagnostic on the extraction, before labeling. It surfaces edge collapse, dangling/missing endpoints, and self-loops — the silent-corruption modes of incremental updates and AST/LLM id mismatches. Read-only; never aborts. + +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path +from graphify.diagnostics import diagnose_extraction, format_diagnostic_report + +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) +summary = diagnose_extraction(extraction, directed=IS_DIRECTED, root='INPUT_PATH') +print(format_diagnostic_report(summary)) +flags = [f'{summary[k]} {label}' for k, label in ( + ('dangling_endpoint_edges', 'dangling-endpoint edges'), + ('missing_endpoint_edges', 'missing-endpoint edges'), + ('self_loop_edges', 'self-loop edges'), + ('directed_same_endpoint_collapsed_edges', 'collapsed (directed) edges'), + ('undirected_same_endpoint_collapsed_edges', 'collapsed (undirected) edges'), +) if summary.get(k, 0)] +print('GRAPH HEALTH WARNING: ' + '; '.join(flags) + ' - graph may be incomplete/corrupt.' if flags else 'Graph health: OK (no dangling/missing/collapsed edges).') +" +``` + +Substitute `IS_DIRECTED` and `INPUT_PATH` as in Step 4. If a `GRAPH HEALTH WARNING` prints, surface it in the final summary (do not abort — the graph is still usable, but the integrity issue must be visible, per the Honesty Rules). + +### Step 5 - Label communities + +Read `graphify-out/.graphify_analysis.json`. For each community key, look at its node labels and write a 2-5 word plain-language name (e.g. "Attention Mechanism", "Training Pipeline", "Data Loading"). + +Then regenerate the report and save the labels for the visualizer: + +```bash +$(cat graphify-out/.graphify_python) -c " +import sys, json +from graphify.build import build_from_json +from graphify.cluster import score_all +from graphify.analyze import god_nodes, surprising_connections, suggest_questions +from graphify.report import generate +from pathlib import Path + +extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) +detection = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) +analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text(encoding=\"utf-8\")) + +# root= as in Step 4 / the --update runbook (#1361) — same base for node-key parity. +G = build_from_json(extraction, root='INPUT_PATH', directed=IS_DIRECTED) +communities = {int(k): v for k, v in analysis['communities'].items()} +cohesion = {int(k): v for k, v in analysis['cohesion'].items()} +tokens = {'input': extraction.get('input_tokens', 0), 'output': extraction.get('output_tokens', 0)} + +# LABELS - replace these with the names you chose above +labels = LABELS_DICT + +# Regenerate questions with real community labels (labels affect question phrasing) +questions = suggest_questions(G, communities, labels) + +report = generate(G, communities, cohesion, labels, analysis['gods'], analysis['surprises'], detection, tokens, 'INPUT_PATH', suggested_questions=questions) +Path('graphify-out/GRAPH_REPORT.md').write_text(report, encoding=\"utf-8\") +Path('graphify-out/.graphify_labels.json').write_text(json.dumps({str(k): v for k, v in labels.items()}, ensure_ascii=False), encoding=\"utf-8\") +print('Report updated with community labels') +" +``` + +Replace `LABELS_DICT` with the actual dict you constructed (e.g. `{0: "Attention Mechanism", 1: "Training Pipeline"}`). +Replace INPUT_PATH with the actual path. + +### Step 6 - Generate Obsidian vault (opt-in) + HTML + +**Generate HTML always** (unless `--no-viz`). **Obsidian vault only if `--obsidian` was explicitly given** — skip it otherwise, it generates one file per node. + +If `--obsidian` was given: + +- If `--obsidian-dir ` was also given, pass it via `--dir`. Otherwise defaults to `graphify-out/obsidian`. + +```bash +graphify export obsidian +# or with custom dir: graphify export obsidian --dir ~/vaults/my-project +``` + +Generate the HTML graph (always, unless `--no-viz`): + +```bash +graphify export html # auto-aggregates to community view if graph > 5000 nodes +# or: graphify export html --no-viz +``` + +### Steps 6b-8 - Wiki, Neo4j, FalkorDB, SVG, GraphML, MCP, benchmark (only on their flags) + +These run only when their flag is present (`--wiki`, `--neo4j`/`--neo4j-push`, `--falkordb`/`--falkordb-push`, `--svg`, `--graphml`, `--mcp`) or, for the token-reduction benchmark, when `total_words` exceeds 5,000. A default run with no export flags skips all of them. See `references/exports.md` for each one. Run any `--wiki` export before Step 9 cleanup so `.graphify_labels.json` is still available. + +--- + +### Step 9 - Save manifest, update cost tracker, clean up, and report + +```bash +$(cat graphify-out/.graphify_python) -c " +import json +from pathlib import Path +from datetime import datetime, timezone +from graphify.detect import save_manifest + +# Save manifest for --update +detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) +extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\")) +# In --update mode, 'all_files' carries the full corpus; 'files' is the changed +# subset. Full-rebuild mode populates only 'files', so the fallback handles that. +# root= relativizes the manifest keys to the scan root (same base as the build), +# so the on-disk manifest is portable across clones/machines and a later --update +# matches cached files instead of missing every one (#1417). +# +# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output: +# a detected file whose chunk failed or was omitted must stay unstamped so the +# next --update re-queues it, otherwise it is marked done and its content is lost +# forever (#2015). This mirrors the library extract path exactly +# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the +# raw corpus. Code files are always stamped (AST is deterministic); only semantic +# types are gated on output. +from graphify.cli import _stamped_manifest_files +_corpus = detect.get('all_files') or detect['files'] +_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH')) +# Files dispatched this run (the changed subset) but NOT stamped above still carry +# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues +# them instead of reading them as unchanged (#1948). +_sem_types = ('document', 'paper', 'image') +_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl} +_stamped = {f for fl in _manifest_files.values() for f in fl} +_cleared = _dispatched - _stamped +# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root +# files newly excluded since last run are dropped rather than masquerading as +# deletions; untouched files' prior rows are still preserved (#1908). +_scan = {f for fl in _corpus.values() for f in fl} +save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None) + +# Update cumulative cost tracker +input_tok = extract.get('input_tokens', 0) +output_tok = extract.get('output_tokens', 0) + +cost_path = Path('graphify-out/cost.json') +if cost_path.exists(): + cost = json.loads(cost_path.read_text(encoding=\"utf-8\")) +else: + cost = {'runs': [], 'total_input_tokens': 0, 'total_output_tokens': 0} + +cost['runs'].append({ + 'date': datetime.now(timezone.utc).isoformat(), + 'input_tokens': input_tok, + 'output_tokens': output_tok, + 'files': detect.get('total_files', 0), +}) +cost['total_input_tokens'] += input_tok +cost['total_output_tokens'] += output_tok +cost_path.write_text(json.dumps(cost, indent=2, ensure_ascii=False), encoding=\"utf-8\") + +print(f'This run: {input_tok:,} input tokens, {output_tok:,} output tokens') +print(f'All time: {cost[\"total_input_tokens\"]:,} input, {cost[\"total_output_tokens\"]:,} output ({len(cost[\"runs\"])} runs)') +" +rm -f graphify-out/.graphify_detect.json graphify-out/.graphify_extract.json graphify-out/.graphify_ast.json graphify-out/.graphify_semantic.json graphify-out/.graphify_analysis.json +find graphify-out -maxdepth 1 -name '.graphify_chunk_*.json' -delete 2>/dev/null +rm -f graphify-out/.needs_update 2>/dev/null || true +``` + +Replace INPUT_PATH with the actual path (same value used in Steps 4-5) so the manifest is relativized to the scan root. + +Tell the user (omit the obsidian line unless --obsidian was given): +``` +Graph complete. Outputs in PATH_TO_DIR/graphify-out/ + + graph.html - interactive graph, open in browser + GRAPH_REPORT.md - audit report + graph.json - raw graph data + obsidian/ - Obsidian vault (only if --obsidian was given) +``` + +If graphify saved you time, consider supporting it: https://github.com/sponsors/safishamsi + +Replace PATH_TO_DIR with the actual absolute path of the directory that was processed. + +Then paste these sections from GRAPH_REPORT.md directly into the chat: +- God Nodes +- Surprising Connections +- Suggested Questions + +Do NOT paste the full report - just those three sections. Keep it concise. + +Then immediately offer to explore. Pick the single most interesting suggested question from the report - the one that crosses the most community boundaries or has the most surprising bridge node - and ask: + +> "The most interesting question this graph can answer: **[question]**. Want me to trace it?" + +If the user says yes, run `/graphify query "[question]"` on the graph and walk them through the answer using the graph structure - which nodes connect, which community boundaries get crossed, what the path reveals. Keep going as long as they want to explore. Each answer should end with a natural follow-up ("this connects to X - want to go deeper?") so the session feels like navigation, not a one-shot report. + +The graph is the map. Your job after the pipeline is to be the guide. + +--- + +## Interpreter guard for subcommands + +Before running any subcommand below (`--update`, `--cluster-only`, `query`, `path`, `explain`, `add`), check that `.graphify_python` exists. If it's missing (e.g. user deleted `graphify-out/`), re-resolve the interpreter first: + +```bash +if [ ! -f graphify-out/.graphify_python ]; then + GRAPHIFY_BIN=$(which graphify 2>/dev/null) + if [ -n "$GRAPHIFY_BIN" ]; then + PYTHON=$(head -1 "$GRAPHIFY_BIN" | tr -d '#!') + case "$PYTHON" in *[!a-zA-Z0-9/_.@-]*) PYTHON="python3" ;; esac + else + PYTHON="python3" + fi + mkdir -p graphify-out + "$PYTHON" -c "import sys; open('graphify-out/.graphify_python', 'w', encoding='utf-8').write(sys.executable)" +fi +``` + +## For --update and --cluster-only + +Both are non-default subcommands. `--update` re-extracts only new or changed files; `--cluster-only` reruns clustering on the existing graph. See `references/update.md` for both flows. + +--- + +## For /graphify query + +When `graphify-out/graph.json` already exists and the user asks a question about the corpus, answer from the graph rather than rebuilding it: + +```bash +graphify query "" +``` + +Before traversal, expand the question against the graph's own vocabulary so a wording mismatch does not collapse the answer to noise. If the `graphify query` CLI is unavailable, fall back to an inline NetworkX traversal of `graphify-out/graph.json`. Answer using only what the graph output contains, and quote `source_location` when citing a specific fact. For that vocab-expansion step, the BFS/DFS traversal modes, the `--budget` cap, the NetworkX fallback, `save-result` feedback, and the `/graphify path` and `/graphify explain` flows, see `references/query.md`. + +--- + +## For /graphify add and --watch + +Neither is part of the default build. When the user runs `/graphify add ` to fetch a URL into the corpus, or passes `--watch` to auto-rebuild on file changes, see `references/add-watch.md`. + +--- + +## For the commit hook and native AGENTS.md integration + +When the user asks to install the post-commit auto-rebuild hook or wire graphify into a project's AGENTS.md, see `references/hooks.md`. + +--- + +## Honesty Rules + +- Never invent an edge. If unsure, use AMBIGUOUS. +- Never skip the corpus check warning. +- Always show token cost in the report. +- Never hide cohesion scores behind symbols - show the raw number. +- Never run HTML viz on a graph with more than 5,000 nodes without warning the user. diff --git a/tools/skillgen/gen.py b/tools/skillgen/gen.py index bda0ab2e4e..e3ce99cdb7 100644 --- a/tools/skillgen/gen.py +++ b/tools/skillgen/gen.py @@ -1249,7 +1249,9 @@ def monolith_roundtrip(platform: Platform) -> list[str]: if platform.bucket != "monolith": return [] if platform.roundtrip_ref is None: - return [f"[{platform.key}] monolith is missing roundtrip_ref"] + # Post-v8 monolith with no pristine baseline to freeze against (mirrors + # the `agents` post-v8 handling in _v8_baseline_ref). Nothing to check. + return [] rendered_lines = render(platform)[0].content.splitlines() # Strip trigger lines from the original — they are non-spec and their removal diff --git a/tools/skillgen/platforms.toml b/tools/skillgen/platforms.toml index 0856d22694..f4a28bacb8 100644 --- a/tools/skillgen/platforms.toml +++ b/tools/skillgen/platforms.toml @@ -203,3 +203,12 @@ bucket = "monolith" skill_dst = "graphify/skill-devin.md" monolith = "devin" roundtrip_ref = "47042beb05d1f6dd2186c0c499ae2840ce604ead:graphify/skill-devin.md" + +# Mistral Vibe (mistralai/mistral-vibe). Monolith bucket because vibe's SkillMetadata +# schema requires extra frontmatter fields (user-invocable, allowed-tools) that the +# split-bucket _render_frontmatter emits only name+description for. Post-v8 platform, +# so no roundtrip_ref (no v8 baseline to diff against). +[platform.vibe] +bucket = "monolith" +skill_dst = "graphify/skill-vibe.md" +monolith = "vibe" diff --git a/uv.lock b/uv.lock index 0e90baef91..48c7f7bd2e 100644 --- a/uv.lock +++ b/uv.lock @@ -17,6 +17,10 @@ resolution-markers = [ "python_full_version < '3.11'", ] +[options] +exclude-newer = "2026-10-05T11:51:12.173679Z" +exclude-newer-span = "P2D" + [[package]] name = "annotated-doc" version = "0.0.4" @@ -1096,7 +1100,7 @@ wheels = [ [[package]] name = "graphifyy" -version = "0.9.76" +version = "0.9.79" source = { editable = "." } dependencies = [ { name = "networkx", version = "3.4.2", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.11'" }, @@ -1105,6 +1109,7 @@ dependencies = [ { name = "numpy", version = "2.4.6", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version >= '3.13'" }, { name = "rapidfuzz" }, { name = "tomli", marker = "python_full_version < '3.11'" }, + { name = "tomlkit" }, { name = "tree-sitter" }, { name = "tree-sitter-bash" }, { name = "tree-sitter-c" }, @@ -1346,6 +1351,7 @@ requires-dist = [ { name = "tiktoken", marker = "extra == 'kimi'" }, { name = "tiktoken", marker = "extra == 'openai'" }, { name = "tomli", marker = "python_full_version < '3.11'", specifier = ">=2.0.1" }, + { name = "tomlkit", specifier = ">=0.13" }, { name = "tree-sitter", specifier = ">=0.23.0,<0.26" }, { name = "tree-sitter-bash", specifier = ">=0.23,<0.27" }, { name = "tree-sitter-c", specifier = ">=0.23,<0.25" }, @@ -4562,6 +4568,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/c7/18/c86eb8e0202e32dd3df50d43d7ff9854f8e0603945ff398974c1d91ac1ef/tomli_w-1.2.0-py3-none-any.whl", hash = "sha256:188306098d013b691fcadc011abd66727d3c414c571bb01b1a174ba8c983cf90", size = 6675, upload-time = "2025-01-15T12:07:22.074Z" }, ] +[[package]] +name = "tomlkit" +version = "0.15.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/94/96/e07752635b98536177fa1f37671c8f3cdde2e724c6bcf6034b2cfb571565/tomlkit-0.15.1.tar.gz", hash = "sha256:e25bbf38843005246210a12982776f27f99cb9be67160e14434d0c0d21ee1e97", size = 180129, upload-time = "2026-07-17T01:48:04.562Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/13/bc/8c13eb66537dce1d2bd3a57132902f38d0e7f5bb46fa9f4daed9fe9d76ee/tomlkit-0.15.1-py3-none-any.whl", hash = "sha256:177a05aece5a8ca5266fd3c448abb47b8d352f09d477d3ca8332db4d89b24304", size = 49449, upload-time = "2026-07-17T01:48:05.728Z" }, +] + [[package]] name = "tqdm" version = "4.67.3"