From feac32fc96b48b471d6f4a565efeaf2d0b0e0d84 Mon Sep 17 00:00:00 2001 From: max Date: Mon, 14 Sep 2026 12:42:56 +0300 Subject: [PATCH] feat: add Qwen inference gateway and current Codex transcript support --- .env.example | 9 + .project-docs/agent-context.md | 33 +++ .project-docs/manifest.yaml | 31 +++ .project-docs/navigation.md | 28 ++ .project-docs/open-questions.md | 6 + .../processes/inference-api-setup.md | 66 +++++ .project-docs/project.md | 25 ++ .project-docs/services/inference-api.md | 56 ++++ .project-docs/validation-allowlist.json | 257 ++++++++++++++++++ AGENTS.md | 33 +++ CLAUDE.md | 33 +++ README.md | 4 + ...7-grep-resilient-to-deleted-transcripts.md | 4 +- pyproject.toml | 1 + src/session_recall/config.py | 15 +- src/session_recall/embed.py | 3 + src/session_recall/inference.py | 121 +++++++++ src/session_recall/rerank.py | 3 + src/session_recall/transcripts.py | 45 ++- tests/test_inference.py | 105 +++++++ tests/test_transcripts.py | 32 ++- 21 files changed, 899 insertions(+), 11 deletions(-) create mode 100644 .project-docs/agent-context.md create mode 100644 .project-docs/manifest.yaml create mode 100644 .project-docs/navigation.md create mode 100644 .project-docs/open-questions.md create mode 100644 .project-docs/processes/inference-api-setup.md create mode 100644 .project-docs/project.md create mode 100644 .project-docs/services/inference-api.md create mode 100644 .project-docs/validation-allowlist.json create mode 100644 AGENTS.md create mode 100644 CLAUDE.md create mode 100644 src/session_recall/inference.py create mode 100644 tests/test_inference.py diff --git a/.env.example b/.env.example index 766ce03..1628f06 100644 --- a/.env.example +++ b/.env.example @@ -7,3 +7,12 @@ VOYAGE_API_KEY= # Optional only for a portable/custom Cursor profile. The standard macOS/Linux # database path is detected automatically. SESSION_RECALL_CURSOR_DB= + +# Optional Qwen inference gateway; see .project-docs/processes/inference-api-setup.md. +# SESSION_RECALL_EMBED=inference-api +# SESSION_RECALL_EMBED_BASE_URL=https://inference.example/v1 +# SESSION_RECALL_EMBED_DIM=4096 +# SESSION_RECALL_INFERENCE_MAX_TOKENS=8192 +# SESSION_RECALL_EMBED_REVISION=deployment-revision +# SESSION_RECALL_DB_PATH=/path/to/separate/index-qwen.db +INFERENCE_API_KEY= diff --git a/.project-docs/agent-context.md b/.project-docs/agent-context.md new file mode 100644 index 0000000..1101b29 --- /dev/null +++ b/.project-docs/agent-context.md @@ -0,0 +1,33 @@ + + +# Project guidance + +Project knowledge lives under `.project-docs/`. + +Before making a non-trivial project-specific claim: + +1. read `.project-docs/manifest.yaml`; +2. follow `.project-docs/navigation.md`; +3. search the routed locations and read the canonical record plus material + related links; +4. check status, freshness, and sources. + +Use `services/` for current systems, `processes/` for exact actions, +`decisions/` for rationale, `reactions/` for conditional first actions, +`bugs/` for known failures, and `timeline/` for chronology. + +Do not treat observations, stale claims, conflicts, recommendations, or +unknowns as confirmed facts. Never store, quote, partially reproduce, +transform, or echo secret values. + +Change project guidance only through the `project-documentation` workflow at +the canonical editable source, `.project-docs/agent-context.md`. Do not edit +`AGENTS.md` or `CLAUDE.md` directly; both generated targets are byte-identical +to the canonical source. + +After every documentation mutation, run +`python3 /scripts/validate_project_documentation.py +--project-root --tracked-documentation`. Add `--include` for a +new untracked documentation file. If canonical guidance changed, also run the +guidance synchronization script with `--diff`, `--write`, and `--check`. +Documentation work is incomplete until validation and synchronization pass. diff --git a/.project-docs/manifest.yaml b/.project-docs/manifest.yaml new file mode 100644 index 0000000..edc0ca8 --- /dev/null +++ b/.project-docs/manifest.yaml @@ -0,0 +1,31 @@ +schema_version: 1 +entrypoints: + project: project.md + navigation: navigation.md +routes: + current_state: + - project.md + - services/ + how_to: + - processes/ + why: + - decisions/ + incident_first_action: + - reactions/ + - processes/ + - services/ + known_problem: + - bugs/ + history: + - timeline/ + - changelog/ + unresolved: + - open-questions.md + - observations/ +search_fields: + - id + - title + - summary + - tags + - canonical_for + - related diff --git a/.project-docs/navigation.md b/.project-docs/navigation.md new file mode 100644 index 0000000..92c9c31 --- /dev/null +++ b/.project-docs/navigation.md @@ -0,0 +1,28 @@ +# Documentation navigation + +## Search order + +1. Classify the question using `manifest.yaml` routes. +2. Search only routed paths with task terms, identifiers, tags, and synonyms. +3. Read the best canonical match. +4. Follow only material `related` links. +5. Check status, freshness, and sources before using a claim. + +Example: + +```bash +rg -n -i 'trace|tracing|observability' \ + .project-docs/services .project-docs/processes .project-docs/decisions +``` + +## Stopping rules + +- No canonical match means unknown. +- Observations remain unconfirmed. +- Stale records require re-verification. +- Conflicts preserve every sourced version. +- Recommendations are not current behavior. +- Missing sources invalidate confirmed claims. + +Record missing knowledge in `open-questions.md`; do not invent a project +default. diff --git a/.project-docs/open-questions.md b/.project-docs/open-questions.md new file mode 100644 index 0000000..1f6c041 --- /dev/null +++ b/.project-docs/open-questions.md @@ -0,0 +1,6 @@ +# Open questions + +- Gateway deployments must supply their own endpoint, credentials, dimensions, + context limit, and embedding revision. These are not repository defaults. +- The gateway registry does not provide a required immutable embedding revision; + operators must update the configured revision when the backend changes. diff --git a/.project-docs/processes/inference-api-setup.md b/.project-docs/processes/inference-api-setup.md new file mode 100644 index 0000000..5208b9c --- /dev/null +++ b/.project-docs/processes/inference-api-setup.md @@ -0,0 +1,66 @@ +--- +id: inference-api-setup +type: process +title: Connect a Qwen inference gateway +summary: Configure the gateway and build a separate index before switching retrieval. +status: confirmed +tags: [qwen, setup, migration, backup] +canonical_for: [inference-api-setup] +verified_at: 2026-09-14 +sources: + - type: repository + reference: src/session_recall/config.py + confirmed_at: 2026-09-14 + - type: repository + reference: src/session_recall/inference.py + confirmed_at: 2026-09-14 + - type: repository + reference: src/session_recall/cli.py + confirmed_at: 2026-09-14 +related: [../project.md, ../services/inference-api.md] +--- + +# Connect a Qwen inference gateway + +Check the [required gateway contract](../services/inference-api.md) first. +Obtain the endpoint, embedding dimension and context limit from the deployment +owner or authenticated model registry. Supply `INFERENCE_API_KEY` through your +secret manager or environment; never commit its value. + +Example configuration for a gateway with 4096-dimensional vectors and an +8192-token context window: + +```sh +export SESSION_RECALL_EMBED=inference-api +export SESSION_RECALL_EMBED_BASE_URL=https://inference.example/v1 +export SESSION_RECALL_EMBED_DIM=4096 +export SESSION_RECALL_INFERENCE_MAX_TOKENS=8192 +export SESSION_RECALL_EMBED_REVISION=deployment-revision +export SESSION_RECALL_DB_PATH="$HOME/.local/share/session-recall/index-qwen.db" +session-recall index +session-recall health +session-recall search "why did we choose" +``` + +The preset uses model aliases `embedder` and `reranker`. Override them with +`SESSION_RECALL_EMBED_MODEL` and `SESSION_RECALL_RERANK_MODEL` if your gateway +uses different public names. Existing provider overrides also take precedence +over the preset, so remove or update stale overrides when migrating. + +Before migration, preserve the old provider settings and make a consistent +SQLite backup of the old index. Build the new embedding space in a separate +file with `SESSION_RECALL_DB_PATH`; do not mix Voyage and Qwen vectors. +Review indexing errors, corpus coverage and health before switching clients. +An unavailable source database cannot supply new history; retain its old +index until its historical records have been accounted for. + +Configure CLI invocations, background indexers, and MCP launchers with the same +provider settings and database path. Restart existing MCP processes after +switching, because they retain their configuration and open database. +To roll back, restore the previous launcher configuration and use the old +index with its original embedding provider. + +When the backend behind the public alias changes incompatibly, update +`SESSION_RECALL_EMBED_REVISION` and rebuild. A stable model alias alone does +not identify an immutable embedding space. Session Recall does not auto-load +`.env` files; export these settings or load them through your launcher. diff --git a/.project-docs/project.md b/.project-docs/project.md new file mode 100644 index 0000000..12dcb07 --- /dev/null +++ b/.project-docs/project.md @@ -0,0 +1,25 @@ +--- +id: session-recall-project +type: project +title: Session Recall +summary: Semantic retrieval over local Claude Code, Codex, and Cursor history. +status: confirmed +tags: [recall, indexing, embeddings] +canonical_for: [project-overview] +verified_at: 2026-09-14 +sources: + - type: repository + reference: README.md + confirmed_at: 2026-09-14 +related: [services/inference-api.md, processes/inference-api-setup.md] +--- + +# Session Recall + +Session Recall extracts conversation text, stores text and vectors in SQLite, +and exposes retrieval through its CLI and MCP server. Provider selection and +index identity are configured outside the repository. + +See the [Inference API contract](services/inference-api.md) and +[connection procedure](processes/inference-api-setup.md) for hosted Qwen gateways. +The root README covers existing providers and the remaining product workflows. diff --git a/.project-docs/services/inference-api.md b/.project-docs/services/inference-api.md new file mode 100644 index 0000000..e2cf8a0 --- /dev/null +++ b/.project-docs/services/inference-api.md @@ -0,0 +1,56 @@ +--- +id: inference-api-provider +type: service +title: Qwen inference gateway provider +summary: Embeddings and reranking through an authenticated capability-based HTTP API. +status: confirmed +tags: [qwen, inference, embeddings, reranker, codex] +canonical_for: [inference-api-provider] +verified_at: 2026-09-14 +sources: + - type: repository + reference: src/session_recall/inference.py + confirmed_at: 2026-09-14 + - type: repository + reference: src/session_recall/config.py + confirmed_at: 2026-09-14 + - type: repository + reference: tests/test_inference.py + confirmed_at: 2026-09-14 +related: [../project.md, ../processes/inference-api-setup.md] +--- + +# Inference API provider + +The `inference-api` preset selects the public `embedder` and `reranker` model +aliases. It defaults to 4096 embedding dimensions; deployments can override +that value. The base URL and credential have no default. + +The client sends `POST embeddings` beneath the configured `/v1` base URL, +with `input_type=document` for indexing and `input_type=query` for retrieval. +It sends batches of up to 64 strings and requests base64 float32 vectors. +Response indices determine ordering. Both float arrays and base64 responses +are validated for dimension and finite values before storage. + +Reranking uses `POST rerank` with `query`, `documents`, and `top_n`. Results +must contain distinct valid document indices and finite `relevance_score` +values. The client returns scores in descending order. + +Both endpoints receive `truncate_prompt_tokens`, defaulting to 8192. This +limits model input; the full extracted text remains in the local index. +The gateway must support these request fields, including the query/document +instruction distinction; a generic embeddings-only API is not sufficient. + +The client uses `INFERENCE_API_KEY`, verifies HTTPS for remote endpoints, +does not follow redirects, and ignores proxy environment variables. Local +loopback HTTP is supported. Requests use a 120-second timeout with a +10-second connection timeout and up to three attempts for transport errors, +429, and selected temporary server errors. HTTP error messages omit response +bodies so echoed transcripts do not enter indexing logs. + +Index fingerprints include the provider, model alias, dimension, endpoint, +operator-supplied revision, token limit, and query/document preprocessing +version. Changing any of these requires compatible reindexing. Encoding and +batch size do not change the embedding space. + +See the [setup procedure](../processes/inference-api-setup.md). diff --git a/.project-docs/validation-allowlist.json b/.project-docs/validation-allowlist.json new file mode 100644 index 0000000..8a2181a --- /dev/null +++ b/.project-docs/validation-allowlist.json @@ -0,0 +1,257 @@ +{ + "version": 1, + "allow": [ + { + "path": ".github/workflows/test.yml", + "rule": "personal-path", + "literal": "~/.local/share/session-recall/models", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.es-ES.md", + "rule": "personal-path", + "literal": "~/.local/bin", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.es-ES.md", + "rule": "personal-path", + "literal": "~/.local/share/session-recall/index.db", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.es-ES.md", + "rule": "personal-path", + "literal": "~/.claude/settings.json", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.es-ES.md", + "rule": "personal-path", + "literal": "~/Sync/sr-share", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.es-ES.md", + "rule": "personal-path", + "literal": "~/meta-docs", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.es-ES.md", + "rule": "personal-path", + "literal": "~/.local/share/session-recall/", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.es-ES.md", + "rule": "personal-path", + "literal": "~/.local/share/session-recall/index.db)", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.md", + "rule": "personal-path", + "literal": "~/.local/bin", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.md", + "rule": "personal-path", + "literal": "~/.local/share/session-recall/index.db", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.md", + "rule": "personal-path", + "literal": "~/.claude/settings.json", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.md", + "rule": "personal-path", + "literal": "~/Sync/sr-share", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.md", + "rule": "personal-path", + "literal": "~/meta-docs", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.md", + "rule": "personal-path", + "literal": "~/.local/share/session-recall/", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.md", + "rule": "personal-path", + "literal": "~/.local/share/session-recall/index.db)", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.zh-CN.md", + "rule": "personal-path", + "literal": "~/.local/bin", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.zh-CN.md", + "rule": "personal-path", + "literal": "~/.local/share/session-recall/index.db", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.zh-CN.md", + "rule": "personal-path", + "literal": "~/.claude/settings.json", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.zh-CN.md", + "rule": "personal-path", + "literal": "~/Sync/sr-share", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.zh-CN.md", + "rule": "personal-path", + "literal": "~/meta-docs", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.zh-CN.md", + "rule": "personal-path", + "literal": "~/.local/share/session-recall/", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "README.zh-CN.md", + "rule": "personal-path", + "literal": "~/.local/share/session-recall/index.db)", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "docs/README.ru.md", + "rule": "personal-path", + "literal": "~/.local/bin", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "docs/README.ru.md", + "rule": "personal-path", + "literal": "~/.local/share/session-recall/index.db", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "docs/README.ru.md", + "rule": "personal-path", + "literal": "~/.claude/settings.json", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "docs/README.ru.md", + "rule": "personal-path", + "literal": "~/Sync/sr-share", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "docs/README.ru.md", + "rule": "personal-path", + "literal": "~/meta-docs", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "docs/README.ru.md", + "rule": "personal-path", + "literal": "~/.local/share/session-recall/", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "docs/README.ru.md", + "rule": "personal-path", + "literal": "~/.local/share/session-recall/index.db)", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "docs/decisions/2026-06-27-grep-resilient-to-deleted-transcripts.md", + "rule": "personal-path", + "literal": "~/.claude/projects/-Users-maxim/", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "docs/decisions/2026-06-27-recall-ergonomics-when-and-recent-sessions.md", + "rule": "personal-path", + "literal": "~/sidekey)", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "docs/decisions/2026-07-26-voyage-403-egress-via-netcup.md", + "rule": "personal-path", + "literal": "~/.local/bin/session-recall{,-mcp}", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "docs/decisions/2026-07-26-voyage-403-egress-via-netcup.md", + "rule": "personal-path", + "literal": "~/.claude/settings.json", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "docs/decisions/2026-07-26-voyage-403-egress-via-netcup.md", + "rule": "personal-path", + "literal": "~/.zshenv", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "docs/decisions/2026-07-30-p2p-sharing-v1-security-gate.md", + "rule": "personal-path", + "literal": "~/.ssh", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "docs/decisions/2026-07-31-metadocs-living-project-memory.md", + "rule": "personal-path", + "literal": "~/.local/share/session-recall/salvage/", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "docs/team-hub.ru.md", + "rule": "personal-path", + "literal": "~/.local/share/session-recall/hub.json", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "skills/setup/SKILL.md", + "rule": "personal-path", + "literal": "~/.local/share/session-recall/", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": "skills/setup/SKILL.md", + "rule": "personal-path", + "literal": "~/Sync/sr-share", + "reason": "Documented example of a standard runtime path, with no personal username or credential." + }, + { + "path": ".github/workflows/publish.yml", + "rule": "sensitive-assignment", + "literal": "token: write", + "reason": "Documented example or GitHub Actions permission keyword; this is not a credential value." + }, + { + "path": "docs/decisions/2026-08-06-team-hub-central-index.md", + "rule": "sensitive-assignment", + "literal": "token: \u2026\u00bb.", + "reason": "Documented example or GitHub Actions permission keyword; this is not a credential value." + }, + { + "path": "skills/setup/SKILL.md", + "rule": "sensitive-assignment", + "literal": "VOYAGE_API_KEY=\u2026`", + "reason": "Documented example using an ellipsis placeholder, not a credential value." + } + ] +} diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..1101b29 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,33 @@ + + +# Project guidance + +Project knowledge lives under `.project-docs/`. + +Before making a non-trivial project-specific claim: + +1. read `.project-docs/manifest.yaml`; +2. follow `.project-docs/navigation.md`; +3. search the routed locations and read the canonical record plus material + related links; +4. check status, freshness, and sources. + +Use `services/` for current systems, `processes/` for exact actions, +`decisions/` for rationale, `reactions/` for conditional first actions, +`bugs/` for known failures, and `timeline/` for chronology. + +Do not treat observations, stale claims, conflicts, recommendations, or +unknowns as confirmed facts. Never store, quote, partially reproduce, +transform, or echo secret values. + +Change project guidance only through the `project-documentation` workflow at +the canonical editable source, `.project-docs/agent-context.md`. Do not edit +`AGENTS.md` or `CLAUDE.md` directly; both generated targets are byte-identical +to the canonical source. + +After every documentation mutation, run +`python3 /scripts/validate_project_documentation.py +--project-root --tracked-documentation`. Add `--include` for a +new untracked documentation file. If canonical guidance changed, also run the +guidance synchronization script with `--diff`, `--write`, and `--check`. +Documentation work is incomplete until validation and synchronization pass. diff --git a/CLAUDE.md b/CLAUDE.md new file mode 100644 index 0000000..1101b29 --- /dev/null +++ b/CLAUDE.md @@ -0,0 +1,33 @@ + + +# Project guidance + +Project knowledge lives under `.project-docs/`. + +Before making a non-trivial project-specific claim: + +1. read `.project-docs/manifest.yaml`; +2. follow `.project-docs/navigation.md`; +3. search the routed locations and read the canonical record plus material + related links; +4. check status, freshness, and sources. + +Use `services/` for current systems, `processes/` for exact actions, +`decisions/` for rationale, `reactions/` for conditional first actions, +`bugs/` for known failures, and `timeline/` for chronology. + +Do not treat observations, stale claims, conflicts, recommendations, or +unknowns as confirmed facts. Never store, quote, partially reproduce, +transform, or echo secret values. + +Change project guidance only through the `project-documentation` workflow at +the canonical editable source, `.project-docs/agent-context.md`. Do not edit +`AGENTS.md` or `CLAUDE.md` directly; both generated targets are byte-identical +to the canonical source. + +After every documentation mutation, run +`python3 /scripts/validate_project_documentation.py +--project-root --tracked-documentation`. Add `--include` for a +new untracked documentation file. If canonical guidance changed, also run the +guidance synchronization script with `--diff`, `--write`, and `--check`. +Documentation work is incomplete until validation and synchronization pass. diff --git a/README.md b/README.md index 740284f..577463c 100644 --- a/README.md +++ b/README.md @@ -316,6 +316,10 @@ export SESSION_RECALL_EMBED_MODEL=your-model export SESSION_RECALL_EMBED_DIM=1024 ``` +**Hosted Qwen gateway:** use the `inference-api` preset for query-aware embeddings +and a Cohere-style reranker. See the [gateway setup guide](.project-docs/processes/inference-api-setup.md) +for credentials, model aliases, context limits, and migration to a separate index. + **A different embedder needs its own index.** Vector tables are fixed-width, so changing the model or dimension means rebuilding: delete `~/.local/share/session-recall/index.db` and re-run `index`. Session Recall fingerprints the embedding space of every indexed file diff --git a/docs/decisions/2026-06-27-grep-resilient-to-deleted-transcripts.md b/docs/decisions/2026-06-27-grep-resilient-to-deleted-transcripts.md index bed71f3..cacd06c 100644 --- a/docs/decisions/2026-06-27-grep-resilient-to-deleted-transcripts.md +++ b/docs/decisions/2026-06-27-grep-resilient-to-deleted-transcripts.md @@ -9,7 +9,7 @@ global `grep` (without `session_id`) crashed with ``` [Errno 2] No such file or directory: -/Users/maxim/.claude/projects/-Users-maxim/98688231-0c4e-471d-aec2-a4ee74efda4f.jsonl +$HOME/.claude/projects/-Users-maxim/98688231-0c4e-471d-aec2-a4ee74efda4f.jsonl ``` Hypothesis in the feedback: the path is "broken/truncated" (`-Users-maxim` instead of @@ -40,7 +40,7 @@ drill-down on such a hit would return nothing). Diagnosis disproved the feedback hypothesis. Measurement against the real index: - `~/.claude/projects/-Users-maxim/` **exists** — it is a valid project dir for - sessions launched from home `/Users/maxim` (23 `.jsonl`). The path is not truncated, the encoding + sessions launched from home `$HOME` (23 `.jsonl`). The path is not truncated, the encoding is correct. - Of the **338** indexed `file_path` entries, exactly **one** is missing from disk — that very `98688231-…jsonl`. The file was **deleted after indexing**; its chunks remained in the DB and diff --git a/pyproject.toml b/pyproject.toml index dc5a99e..4790bda 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -24,6 +24,7 @@ classifiers = [ "Topic :: Text Processing :: Indexing", ] dependencies = [ + "httpx>=0.27,<1", "sqlite-vec==0.1.9", "voyageai>=0.3.0", "mcp>=1.2.0,<2", diff --git a/src/session_recall/config.py b/src/session_recall/config.py index b79aece..faa8c1a 100644 --- a/src/session_recall/config.py +++ b/src/session_recall/config.py @@ -7,7 +7,7 @@ from urllib.parse import urlparse DATA_DIR = Path(os.environ.get("XDG_DATA_HOME") or (Path.home() / ".local" / "share")) / "session-recall" -DB_PATH = DATA_DIR / "index.db" +DB_PATH = Path(os.environ.get("SESSION_RECALL_DB_PATH") or (DATA_DIR / "index.db")).expanduser() SETTINGS_PATH = DATA_DIR / "settings.json" # written by onboarding, human-editable CLAUDE_PROJECTS = Path( os.environ.get("SESSION_RECALL_CLAUDE_PROJECTS") @@ -53,6 +53,9 @@ class EmbedSettings: # reranker together, because those four are not independent choices — picking a # local model and leaving a cloud reranker configured just fails later, further away. PRESETS: dict[str, EmbedSettings] = { + "inference-api": EmbedSettings( + provider="inference-api", model="embedder", dim=4096, base_url=None, + send_dimensions=False, rerank_provider="inference-api", rerank_model="reranker"), "voyage": EmbedSettings( provider="voyage", model="voyage-4-large", dim=1024, base_url=None, send_dimensions=False, rerank_provider="voyage", rerank_model="rerank-2.5"), @@ -196,4 +199,12 @@ def embed_fingerprint() -> str: """Which embedding space vectors live in right now. Read at call time (not frozen above) so tests and long processes see configuration changes. The format is part of file signatures — change it and every file re-embeds.""" - return f"{EMBED_PROVIDER}/{EMBED_MODEL}/{EMBED_DIM}" + fingerprint = f"{EMBED_PROVIDER}/{EMBED_MODEL}/{EMBED_DIM}" + if EMBED_PROVIDER == "inference-api": + import hashlib + # Public capability aliases stay constant when the backend rotates. + identity = (EMBED_BASE_URL or "") + "|" + os.environ.get( + "SESSION_RECALL_EMBED_REVISION", "") + "|query-document-v1|" + os.environ.get( + "SESSION_RECALL_INFERENCE_MAX_TOKENS", "8192") + fingerprint += "/" + hashlib.sha256(identity.encode()).hexdigest()[:16] + return fingerprint diff --git a/src/session_recall/embed.py b/src/session_recall/embed.py index 98b9e8b..fae27c8 100644 --- a/src/session_recall/embed.py +++ b/src/session_recall/embed.py @@ -156,6 +156,9 @@ def make_embedder(provider: str | None = None, model: str | None = None, provider = (provider or config.EMBED_PROVIDER).lower() if provider == "voyage": return VoyageEmbedder(model=model) + if provider == "inference-api": + from .inference import InferenceEmbedder + return InferenceEmbedder(model=model, dim=dim) if provider in ("openai", "openai-compatible"): return OpenAIEmbedder(model=model, dim=dim) if provider == "builtin": diff --git a/src/session_recall/inference.py b/src/session_recall/inference.py new file mode 100644 index 0000000..cc82d32 --- /dev/null +++ b/src/session_recall/inference.py @@ -0,0 +1,121 @@ +"""Clients for a Qwen inference gateway with public capability model aliases.""" +import base64 +import math +import os +import struct +import time +from urllib.parse import urlparse + +import httpx + +from . import config + + +class InferenceClient: + def __init__(self, base_url=None, api_key=None): + self.base_url = (base_url or config.EMBED_BASE_URL or "").rstrip("/") + self.api_key = api_key + self._client = None + + @property + def client(self): + if self._client is None: + url = urlparse(self.base_url) + if (url.scheme not in ("http", "https") or not url.hostname + or url.username or url.password or url.query or url.fragment): + raise ValueError("Set SESSION_RECALL_EMBED_BASE_URL to the gateway /v1 URL") + if url.scheme != "https" and url.hostname not in ("localhost", "127.0.0.1", "::1"): + raise ValueError("Remote inference gateways require HTTPS") + key = self.api_key or os.environ.get("INFERENCE_API_KEY") + if not key: + raise ValueError("INFERENCE_API_KEY is required") + self._client = httpx.Client( + base_url=self.base_url + "/", + headers={"Authorization": "Bearer " + key}, + timeout=httpx.Timeout(120, connect=10), + # The Voyage SOCKS egress must not intercept corporate traffic. + trust_env=False, follow_redirects=False, + ) + return self._client + + def post(self, endpoint, payload): + for attempt in range(3): + try: + response = self.client.post(endpoint, json=payload) + except httpx.TransportError: + if attempt == 2: + raise RuntimeError("Inference API transport failed after 3 attempts") from None + else: + if response.is_success: + return response.json() + if response.status_code not in (429, 500, 502, 503, 504) or attempt == 2: + # Never put provider bodies (which can echo text) in indexing logs. + raise RuntimeError(f"Inference API {endpoint}: HTTP {response.status_code}") + time.sleep(2 ** attempt) + + +class InferenceEmbedder: + def __init__(self, model=None, dim=None, client=None): + self.model = model or config.EMBED_MODEL + self.dim = dim or config.EMBED_DIM + self.client = client or InferenceClient() + + def _embed(self, texts, input_type): + out = [] + # Keep background batches bounded on the shared GPU. + for start in range(0, len(texts), 64): + batch = texts[start:start + 64] + data = self.client.post("embeddings", { + "model": self.model, "input": batch, "input_type": input_type, + "encoding_format": "base64", + "truncate_prompt_tokens": int(os.environ.get("SESSION_RECALL_INFERENCE_MAX_TOKENS", "8192")), + })["data"] + if (len(data) != len(batch) + or sorted(item["index"] for item in data) != list(range(len(batch)))): + raise ValueError("Inference API returned invalid embedding indices") + for item in sorted(data, key=lambda item: item["index"]): + vector = item["embedding"] + if isinstance(vector, str): + raw = base64.b64decode(vector, validate=True) + if len(raw) != self.dim * 4: + raise ValueError("Inference API changed embedding dimensions") + vector = list(struct.unpack(f"<{self.dim}f", raw)) + if (len(vector) != self.dim + or any(not isinstance(x, (int, float)) or not math.isfinite(x) for x in vector)): + raise ValueError("Inference API returned an invalid embedding or changed dimensions") + out.append(vector) + return out + + def embed_documents(self, texts): + return self._embed(texts, "document") + + def embed_query(self, text): + return self._embed([text], "query")[0] + + +class InferenceReranker: + def __init__(self, model=None, client=None): + self.model = model or config.RERANK_MODEL + self.client = client or InferenceClient() + + def rerank(self, query, documents, top_k): + if not documents or top_k <= 0: + return [] + results = self.client.post("rerank", { + "model": self.model, "query": query, + "documents": documents, "top_n": min(top_k, len(documents)), + "truncate_prompt_tokens": int(os.environ.get("SESSION_RECALL_INFERENCE_MAX_TOKENS", "8192")), + })["results"] + ranked = [] + seen = set() + for item in results: + index, score = item["index"], item["relevance_score"] + if (type(index) is not int or not 0 <= index < len(documents) + or index in seen or not isinstance(score, (int, float)) + or not math.isfinite(score)): + raise ValueError("Inference API returned invalid reranking results") + seen.add(index) + ranked.append((index, float(score))) + if len(ranked) != min(top_k, len(documents)): + raise ValueError("Inference API returned an incomplete reranking result") + return sorted(ranked, key=lambda item: item[1], reverse=True)[:top_k] diff --git a/src/session_recall/rerank.py b/src/session_recall/rerank.py index 8937a2b..bb12fe5 100644 --- a/src/session_recall/rerank.py +++ b/src/session_recall/rerank.py @@ -51,6 +51,9 @@ def make_reranker(provider: str | None = None, model: str | None = None) -> Opti return None if provider == "voyage": return VoyageReranker(model=model) + if provider == "inference-api": + from .inference import InferenceReranker + return InferenceReranker(model=model) if provider == "fake": return FakeReranker() raise ValueError(f"unknown rerank provider: {provider!r} (set SESSION_RECALL_RERANK_PROVIDER)") diff --git a/src/session_recall/transcripts.py b/src/session_recall/transcripts.py index b87d6a2..bcb0268 100644 --- a/src/session_recall/transcripts.py +++ b/src/session_recall/transcripts.py @@ -203,7 +203,7 @@ def discover_codex_transcripts(sessions_dir: Path, archived_dir: Path) -> list[T def extractor_version(source: str) -> str: - versions = {"claude": "2", "codex": "1", "cursor": "2"} + versions = {"claude": "2", "codex": "2", "cursor": "2"} try: return versions[source] except KeyError as exc: @@ -270,7 +270,36 @@ def _render_claude(obj: dict) -> str: return "\n".join(part for part in parts if part) +def _completed_message(obj: dict) -> tuple[str, str] | None: + """Codex desktop records visible messages as completed typed items. + + Only explicit UserMessage/AgentMessage items are conversation surface; + tool output, reasoning and mirrored response_item messages stay excluded. + """ + payload = obj.get("payload") + if (obj.get("type") != "event_msg" or not isinstance(payload, dict) + or payload.get("type") != "item_completed"): + return None + item = payload.get("item") + if not isinstance(item, dict): + return None + role = {"UserMessage": "user", "AgentMessage": "assistant"}.get(item.get("type")) + if role is None: + return None + content = item.get("content") + if not isinstance(content, list): + return None + text = "\n".join(block["text"] for block in content + if isinstance(block, dict) + and block.get("type") in {"text", "Text", "input_text", "output_text"} + and isinstance(block.get("text"), str) and block["text"].strip()) + return role, text + + def _render_codex(obj: dict) -> str: + completed = _completed_message(obj) + if completed is not None: + return completed[1] envelope = obj.get("type") payload = obj.get("payload") or {} if not isinstance(payload, dict): @@ -330,6 +359,9 @@ def _render_codex(obj: dict) -> str: def _codex_role_and_type(obj: dict) -> tuple[str, str]: + completed = _completed_message(obj) + if completed is not None: + return completed[0], "item_completed" envelope = obj.get("type", "") payload = obj.get("payload") or {} payload = payload if isinstance(payload, dict) else {} @@ -482,8 +514,9 @@ def read_transcript(path: str, source: str) -> list[dict[str, Any]]: payload = payload if isinstance(payload, dict) else {} kind = payload.get("type") envelope = obj.get("type") - if envelope == "event_msg" and kind in {"user_message", "agent_message"}: - role = "user" if kind == "user_message" else "assistant" + if envelope == "event_msg" and (kind in {"user_message", "agent_message"} + or _completed_message(obj) is not None): + role = event.role content: Any = event.content if role == "user" else [{"type": "text", "text": event.content}] turns.append(_claude_shaped( event, surface=bool(event.content), role=role, event_type=role, content=content)) @@ -558,11 +591,11 @@ def make_chunk(event: TranscriptEvent, role: str, text: str) -> Chunk: if not isinstance(payload, dict): continue kind = payload.get("type") - if event.obj.get("type") == "event_msg" and kind in { - "user_message", "agent_message"}: + if event.obj.get("type") == "event_msg" and (kind in { + "user_message", "agent_message"} or _completed_message(event.obj) is not None): if not event.content.strip(): continue - role = "user" if kind == "user_message" else "assistant" + role = event.role chunks.append(make_chunk(event, role, event.content)) return chunks diff --git a/tests/test_inference.py b/tests/test_inference.py new file mode 100644 index 0000000..a408891 --- /dev/null +++ b/tests/test_inference.py @@ -0,0 +1,105 @@ +import json +import pytest +import httpx +from session_recall import config +from session_recall.inference import InferenceClient, InferenceEmbedder, InferenceReranker + + +def client_for(handler): + client = InferenceClient('https://inference.example/v1', 'test-key') + client._client = httpx.Client(base_url=client.base_url + '/', transport=httpx.MockTransport(handler)) + return client + + +def test_query_documents_batches_and_response_order(): + calls = [] + def handler(request): + body = json.loads(request.content) + calls.append(body) + assert request.url.path == '/v1/embeddings' + return httpx.Response(200, json={'data': [ + {'index': i, 'embedding': [float(i), 1.0]} for i in reversed(range(len(body['input']))) + ]}) + embedder = InferenceEmbedder('embedder', 2, client_for(handler)) + assert len(embedder.embed_documents(['document'] * 65)) == 65 + assert embedder.embed_query('query') == [0.0, 1.0] + assert [c['input_type'] for c in calls] == ['document', 'document', 'query'] + assert [len(c['input']) for c in calls] == [64, 1, 1] + assert all('dimensions' not in c for c in calls) + + +@pytest.mark.parametrize('data', [ + [{'index': 0, 'embedding': [1]}], + [{'index': 1, 'embedding': [1, 2]}], + [], +]) +def test_embedding_rejects_wrong_dimensions_or_missing_results(data): + e = InferenceEmbedder('embedder', 2, client_for(lambda r: httpx.Response(200,json={'data':data}))) + with pytest.raises(ValueError): + e.embed_query('query') + + +def test_rerank_contract(): + def handler(request): + assert request.url.path == '/v1/rerank' + assert json.loads(request.content) == { + 'model':'reranker','query':'q','documents':['a','b'],'top_n':2, + 'truncate_prompt_tokens':8192, + } + return httpx.Response(200, json={'results':[ + {'index':0,'relevance_score':0.1}, {'index':1,'relevance_score':0.9}, + ]}) + r = InferenceReranker('reranker',client_for(handler)) + assert r.rerank('q',['a','b'],3) == [(1,0.9),(0,0.1)] + assert r.rerank('q',[],3) == [] + + +@pytest.mark.parametrize('results', [ + [{'index':2,'relevance_score':0.8}], + [{'index':0,'relevance_score':0.8},{'index':0,'relevance_score':0.2}], + [], +]) +def test_rerank_rejects_invalid_results(results): + r = InferenceReranker('reranker',client_for(lambda r: httpx.Response(200,json={'results':results}))) + with pytest.raises(ValueError): + r.rerank('q',['a','b'],2) + + +def test_retry_and_no_response_body_in_errors(monkeypatch): + calls = [] + monkeypatch.setattr('session_recall.inference.time.sleep',lambda _:None) + def handler(request): + calls.append(request) + return httpx.Response(503 if len(calls)<3 else 200,json={'data':[]}) + assert client_for(handler).post('embeddings',{}) == {'data':[]} + assert len(calls)==3 + c = client_for(lambda r:httpx.Response(401,text='sensitive response')) + with pytest.raises(RuntimeError,match='HTTP 401') as exc: + c.post('embeddings',{}) + assert 'sensitive' not in str(exc.value) + + +def test_requires_dedicated_key_and_https(monkeypatch): + monkeypatch.delenv('INFERENCE_API_KEY',raising=False) + monkeypatch.setenv('OPENAI_API_KEY','unrelated') + with pytest.raises(ValueError,match='INFERENCE_API_KEY'): + InferenceClient('https://inference.example/v1').client + with pytest.raises(ValueError,match='HTTPS'): + InferenceClient('http://inference.example/v1','test').client + + +def test_preset_and_fingerprint(monkeypatch): + settings = config.resolve_embed({'SESSION_RECALL_EMBED':'inference-api'}) + assert settings.provider == settings.rerank_provider == 'inference-api' + assert settings.model == 'embedder' and settings.rerank_model == 'reranker' + monkeypatch.setattr(config,'EMBED_PROVIDER','inference-api') + before=config.embed_fingerprint() + monkeypatch.setenv('SESSION_RECALL_EMBED_REVISION','new-backend') + assert before != config.embed_fingerprint() + + +def test_base64_embeddings_decode_little_endian_float32(): + import base64, struct + value=base64.b64encode(struct.pack('<2f',0.25,-0.5)).decode() + c=client_for(lambda r:httpx.Response(200,json={'data':[{'index':0,'embedding':value}]})) + assert InferenceEmbedder('embedder',2,c).embed_query('query') == [0.25,-0.5] diff --git a/tests/test_transcripts.py b/tests/test_transcripts.py index 16d17a7..631f6d6 100644 --- a/tests/test_transcripts.py +++ b/tests/test_transcripts.py @@ -211,7 +211,37 @@ def test_claude_passthrough_and_version_validation(tmp_path): path.write_text(json.dumps(raw) + "\n") assert read_transcript(str(path), "claude") == [raw] assert extractor_version("claude") == "2" - assert extractor_version("codex") == "1" + assert extractor_version("codex") == "2" assert extractor_version("cursor") == "2" with pytest.raises(ValueError, match="unknown transcript source"): read_transcript(str(path), "other") + + +def test_desktop_completed_messages_index_once_and_exclude_tools(tmp_path): + path = tmp_path / 'desktop.jsonl' + rows = [ + {'type':'session_meta','payload':{'id':'desktop','cwd':'/repo'}}, + {'type':'event_msg','payload':{'type':'item_completed','item':{ + 'type':'UserMessage','id':'u1','content':[{'type':'text','text':'Find the decision'}], + }}}, + {'type':'response_item','payload':{'type':'message','role':'user', + 'content':[{'type':'input_text','text':'Find the decision'}]}}, + {'type':'event_msg','payload':{'type':'item_completed','item':{ + 'type':'AgentMessage','id':'a1','phase':'final', + 'content':[{'type':'Text','text':'We chose Qwen.'}], + }}}, + {'type':'event_msg','payload':{'type':'item_completed','item':{ + 'type':'CommandExecution','stdout':'tool output must not be embedded', + }}}, + {'type':'event_msg','payload':{'type':'item_completed','item':{ + 'type':'Reasoning','content':[{'type':'Text','text':'private reasoning'}], + }}}, + ] + path.write_text(''.join(json.dumps(row)+'\n' for row in rows)) + chunks=extract_codex_file(str(path)) + assert [(c.role,c.text) for c in chunks] == [('user','Find the decision'),('assistant','We chose Qwen.')] + for chunk in chunks: + raw=path.read_bytes()[chunk.byte_offset:chunk.byte_offset+chunk.byte_len] + assert json.loads(raw)['payload']['type']=='item_completed' + surface=[t for t in read_transcript(str(path),'codex') if t['__surface']] + assert [_text(t) for t in surface] == ['Find the decision','We chose Qwen.']