From a330444d36d8fce91d503847d7edd153e648cbf5 Mon Sep 17 00:00:00 2001 From: lazy Date: Wed, 23 Sep 2026 00:25:25 -0400 Subject: [PATCH] Clarify AI sidebar workspaces and repair empty Jev reviews --- .impeccable/config.json | 12 +- AGENTS.md | 4 +- CLAUDE.md | 2 + README.md | 4 +- WARP.md | 2 + backend/clinical/AGENTS.md | 2 + backend/clinical/ai_evidence_review.py | 10 +- backend/evidence_review/AGENTS.md | 1 + backend/evidence_review/runner.py | 6 +- backend/evidence_review/units.py | 38 ++++-- backend/tests/AGENTS.md | 1 + backend/tests/evidence_review/test_units.py | 17 +++ backend/tests/test_ai_evidence_review.py | 34 +++++ desktop/AGENTS.md | 2 +- desktop/scripts/ui-import-smoke.mjs | 37 ++++-- .../ai-backend/JEV_SIDEBAR_IMPLEMENTATION.md | 16 ++- viewer/assets/AGENTS.md | 4 +- viewer/assets/live/AGENTS.md | 10 +- viewer/assets/live/controller.ts | 1 + viewer/assets/live/evidence-panel.ts | 54 +++++--- viewer/assets/live/panel.ts | 124 +++++++++++++----- viewer/assets/live/presentation.ts | 18 +++ viewer/assets/radsysx-viewer.css | 59 ++++++++- viewer/scripts/AGENTS.md | 1 + viewer/scripts/test-evidence.mjs | 16 +++ viewer/scripts/test-live.mjs | 7 + 26 files changed, 372 insertions(+), 110 deletions(-) create mode 100644 viewer/assets/live/presentation.ts diff --git a/.impeccable/config.json b/.impeccable/config.json index bbe82d2..3ffadca 100644 --- a/.impeccable/config.json +++ b/.impeccable/config.json @@ -2,16 +2,6 @@ "detector": { "ignoreRules": [], "ignoreFiles": [], - "ignoreValues": [ - { - "rule": "side-tab", - "value": "*", - "files": [ - "viewer/assets/radsysx-viewer.css" - ], - "createdAt": "2026-09-23T02:43:10.069Z", - "reason": "Assistant design review: the pre-existing 2px muted #5c7c91 rule groups nested Jev claim judgments and execution receipts; it is not a thick decorative card accent. Verified in the rendered 280px sidebar and git blame." - } - ] + "ignoreValues": [] } } diff --git a/AGENTS.md b/AGENTS.md index 5155485..38d0e97 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -285,7 +285,7 @@ Last updated: 2026-09-22 - Favor rigorous, beautiful, professional solutions with high signal-to-noise. - Prefer Linux-native commands and paths. - Record durable behavior changes in this file or the nearest relevant child `AGENTS.md`. -- `.impeccable/config.json` records reasoned, file-scoped design-detector exceptions. The muted 2 px Jev judgment separator is intentional grouping, not a decorative card accent; its `side-tab` exception is limited to `viewer/assets/radsysx-viewer.css`. +- Keep sidebar information separated into Chat, Research and Jev review views. Use a quiet reading-room palette and progressive disclosure; technical receipts and full abstracts stay collapsed by default. Jev judgment groups use subtle horizontal separators; the old side-accent exception is removed. ## Child DOX Index @@ -339,7 +339,7 @@ The standalone [evidence-review runbook](backend/evidence_review/README.md) docu - See `roadmap/ai-backend/NIM_IMPLEMENTATION.md` for configuration, commands and dated acceptance. Keep provider keys separate and never silently fall back when the chosen model fails. - The sidebar **Settings → Research models** saves Gemini/NVIDIA NIM provider and exact model per signed account. Environment supplies the default until a saved choice exists. The dropdown includes the entire hosted NVIDIA catalog; catalog inclusion does not verify tool support or entitlement. Saving closes that account's sessions/jobs and requires reconnection; live voice selection and offline evidence CLI model flags remain separate. -- Keep the AI sidebar compact: one header row, inline data confirmation/connect, and media controls visible only during an active connection. Data attestation, transmission disclosure and separate microphone/image consent remain required. +- Keep the AI sidebar compact: show the text/research model near the header and separate Chat, Research and Jev review views. Optional voice setup is collapsed within Chat; data attestation and disclosure live by the composer. Keep media state visible while connected and preserve separate microphone/image consent. - Sidebar Jev reviews run through backend-owned `/api/ai/sidebar/*/evidence-reviews` contracts in research/pilot only, using backend-only `RADSYSX_TYPESAFE_AI_API_KEY`. They retain original answers, require independent text confirmation, and persist POSIX-private artifacts in `.ai-evidence/` beside the actual database (or absolute `RADSYSX_AI_EVIDENCE_DIR`). Ending voice preserves review; explicit cancel/account changes/logout stop it; source-history deletion cancels jobs and removes their artifacts. Clinical use and the qualified human evidence-quality study remain unapproved/pending. diff --git a/CLAUDE.md b/CLAUDE.md index d8a1bee..eb68b1f 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -86,3 +86,5 @@ NVIDIA NIM is available for explicit evidence evaluation and opt-in PubMed resea Typed **Send** and explicit **Research** work without Gemini Live/OpenAI Realtime. Confirm synthetic/deidentified content and choose the standard model in **Settings → Text & research models**. Only the question, bounded text history (chat only) and neutral viewer metadata are sent; image pixels require separate live sharing. **Connect voice** starts a separate voice conversation. See `roadmap/ai-backend/DESKTOP_AI_ACTIVATION.md`. ChatGPT/Codex subscription sign-in is available under desktop AI Settings for typed chat and public PubMed research. It uses isolated backend-owned Codex App Server and the OS keyring, not OpenAI API credentials or Realtime entitlement. Read `roadmap/ai-backend/CODEX_SUBSCRIPTION.md`; never copy the user's existing Codex auth. + +The AI sidebar separates Chat, Research and Jev review. Optional voice setup stays in Chat; the review workspace hides the composer and collapses abstracts/technical receipts. Sidebar evidence uses versioned intact cited passages and explicitly reports empty previews as unavailable. See `roadmap/ai-backend/JEV_SIDEBAR_IMPLEMENTATION.md` for the corrected workflow and live acceptance. diff --git a/README.md b/README.md index e6b1f2b..76dbb28 100644 --- a/README.md +++ b/README.md @@ -285,7 +285,7 @@ The OHIF sidebar offers `gemini-3.8-live-extended-thinking` and `gpt-realtime-2. 1. Run `npm run desktop` and open **Settings → API keys** in the **RadSysX AI** panel. Enter and save your own Gemini key, OpenAI key, or both. Saving refreshes provider availability without restarting; the first connection verifies provider access. Keys are sent only to the local backend for encrypted storage, and are never displayed again or saved in browser storage/history. Replacing or removing a key ends your active assistant sessions and tasks. 2. Open a synthetic/deidentified study. The key settings distinguish **your saved key**, an **app-configured key**, and **not configured**. Removing a saved key explicitly returns that provider to an app-configured key when one exists. -3. Select the provider, confirm the displayed content is synthetic or deidentified, then connect. If the initial catalog loaded before local sign-in, use **Retry assistant setup** first. This records your declaration, not an automated deidentification result. Changing provider ends the old session and requires fresh confirmation. +3. In **Chat → Voice**, select the provider, confirm the displayed content is synthetic or deidentified, then connect. If the initial catalog loaded before local sign-in, use **Retry setup** first. This records your declaration, not an automated deidentification result. Changing provider ends the old session and requires fresh confirmation. 4. After **Connected**, enable the microphone or send text to the voice provider. Independent text chat and research are described below. To discuss what is visible, click **Share active image** and wait for **image sharing on · image sent**. The panel names the active image; capture covers that selected viewport, at up to one JPEG frame per second. The whole app, sidebar and other windows are outside this capture. 5. Ask for a viewer action, image explanation, or public research. Review a proposed durable report save or deletion before applying it. Local-only DICOM must be imported/associated through the worklist before saving a report. @@ -293,7 +293,7 @@ The OHIF sidebar offers `gemini-3.8-live-extended-thinking` and `gpt-realtime-2. **ChatGPT subscription:** in **Settings**, choose **Sign in with ChatGPT**, continue in the official browser flow, then select **ChatGPT / Codex subscription** and an account-available model under Text & research models. Sign-in preserves your existing model until you save. Uses your plan’s Codex allowance for text/public PubMed research; Realtime voice remains separately API-key billed. Credentials stay in an isolated OS-keyring account. See the [subscription runbook](roadmap/ai-backend/CODEX_SUBSCRIPTION.md) for setup, limits and validation. -**Jev evidence review** is visible above the conversation: use **Review latest evidence with Jev** for the latest completed PubMed result, or **Saved research** to reopen a conversation. Preview the exact claims and abstracts, select claims, confirm public/synthetic text, then explicitly start the review. Jev shows abstract-support judgments and saved execution receipts; the original answer stays unchanged. It does not analyze image pixels. +The sidebar separates **Chat**, **Research**, and **Jev review**. Chat discusses the case; Research shows literature answers and collapsed source lists. Optional voice setup is inside Chat. From a completed PubMed result, choose **Review evidence with Jev** to enter a dedicated review view. Select the unchanged cited passages, inspect their abstracts, confirm public/synthetic text, then explicitly start Jev. Results show abstract-support judgments with execution receipts under details. Empty previews explain why Jev has not run and offer **Prepare review again**. It does not analyze image pixels. **Literature research** cards show the recorded provider/model and worker steps, including waiting for the model, searching PubMed and preparing the answer. Completed, timed-out and cancelled jobs stay visible in history. Model configuration alone does not mean a job is running. [Desktop activation evidence](roadmap/ai-backend/DESKTOP_AI_ACTIVATION.md) separates local checks from hosted-provider results. diff --git a/WARP.md b/WARP.md index 0676352..d639aee 100644 --- a/WARP.md +++ b/WARP.md @@ -97,3 +97,5 @@ NVIDIA NIM is available for explicit evidence evaluation and opt-in PubMed resea Typed **Send** and explicit **Research** work without Gemini Live/OpenAI Realtime. Confirm synthetic/deidentified content and choose the standard model in **Settings → Text & research models**. Only the question, bounded text history (chat only) and neutral viewer metadata are sent; image pixels require separate live sharing. **Connect voice** starts a separate voice conversation. See `roadmap/ai-backend/DESKTOP_AI_ACTIVATION.md`. ChatGPT/Codex subscription sign-in is available under desktop AI Settings for typed chat and public PubMed research through isolated, pinned Codex App Server. Subscription credentials stay in the OS keyring; Realtime remains API-key billed. Read `roadmap/ai-backend/CODEX_SUBSCRIPTION.md`. + +The AI sidebar separates Chat, Research and Jev review. Optional voice setup stays in Chat; the review workspace hides the composer and collapses abstracts/technical receipts. Sidebar evidence uses versioned intact cited passages and explicitly reports empty previews as unavailable. See `roadmap/ai-backend/JEV_SIDEBAR_IMPLEMENTATION.md` for the corrected workflow and live acceptance. diff --git a/backend/clinical/AGENTS.md b/backend/clinical/AGENTS.md index 1ccac45..35ce814 100644 --- a/backend/clinical/AGENTS.md +++ b/backend/clinical/AGENTS.md @@ -69,6 +69,8 @@ ## Explicit Jev evidence reviews +- New sidebar preparations use the versioned intact-passage extractor (`cited-passages-v2`) for trailing paragraph citations and comma-grouped source IDs. Preserve original offsets/text and rebuild the declared plan during evaluation. A preview with no executable pairs is unavailable with `no_reviewable_claims`; legacy ready/zero-pair summaries are projected the same way without rewriting artifacts. Explicit Prepare review again creates a fresh preview and requires renewed confirmation. + - Own `ai_evidence_contracts.py`, `ai_evidence_repository.py`, `ai_evidence_artifacts.py`, `ai_evidence_review.py` and their additive `ai_jev_reviews`/`ai_jev_operations` persistence. These focused contracts stay separate from live conversation messages and primary-model tool results. - Only signed unexpired `ai.run` actors in enabled research/pilot may prepare an owned completed PubMed research result. Freeze its original answer/source/context identity and recorded generation, retrieve original abstracts by fixed PMID endpoints, then require exact-preview public/synthetic confirmation and backend-issued claim selection before pinned `jev-1.13.0` inference. Saved-history review does not require a live connection. The preview is not patient-text approval. - One active review per actor, two globally; preparation is bounded to 20 seconds, evaluation/cleanup to 70 (the runner itself remains 60 plus five). Recheck authority/source identity before every external call. Account stop invalidates queued starts, even if they were waiting for the owner lock. Operation identity and starting state commit together; duplicate requests never schedule duplicate inference. diff --git a/backend/clinical/ai_evidence_review.py b/backend/clinical/ai_evidence_review.py index d11df66..ef9092e 100644 --- a/backend/clinical/ai_evidence_review.py +++ b/backend/clinical/ai_evidence_review.py @@ -80,9 +80,10 @@ def artifacts(self,row): def _summary(self,row): values={k:v for k,v in row.progress_json.items() if k in { 'generation','totalPairs','completedPairs','settledPairs','submittedAttempts','unknownUsageAttempts'}} + empty = row.status == 'ready' and not values.get('totalPairs') return EvidenceReviewSummary(review_id=row.id,session_id=row.session_id,tool_call_id=row.tool_id, - source_context_version=row.context_version,status=row.status,created_at=row.created_at, - updated_at=row.updated_at,reason=row.reason,**values) + source_context_version=row.context_version,status='unavailable' if empty else row.status,created_at=row.created_at, + updated_at=row.updated_at,reason='no_reviewable_claims' if empty else row.reason,**values) def list(self,actor,session_id): self.require(actor) @@ -247,11 +248,12 @@ async def _work(self,job): 'data_class':'public_literature','generation':job.row.progress_json['generation'], 'result':result.model_dump(mode='json'),'evidence':[e.model_dump(mode='json') for e in evidence], 'capture_exclusions':[e.model_dump(mode='json') for e in exclusions]},limits=self.limits) - plan=build_review_plan(snapshot,limits=self.limits) + plan=build_review_plan(snapshot,limits=self.limits,builder_version='cited-passages-v2') ref,digest=self.artifacts(job.row).create_preview(snapshot,plan,job.row.progress_json['generation']) self.repository.update_if_current(job.row.id,job.row.generation,preview_ref=ref,preview_hash=digest, progress_json={**job.row.progress_json,'totalPairs':len(plan.pairs)}) - status='ready' + status='ready' if plan.pairs else 'unavailable' + reason=None if plan.pairs else 'no_reviewable_claims' else: with self.artifacts(job.row).open_run() as store: job.store=store diff --git a/backend/evidence_review/AGENTS.md b/backend/evidence_review/AGENTS.md index 3361826..d812a0f 100644 --- a/backend/evidence_review/AGENTS.md +++ b/backend/evidence_review/AGENTS.md @@ -25,6 +25,7 @@ None. ## Evaluation inputs and providers - `sentence-citations-v1` keeps exact Unicode spans, compound sentences and separate cited abstracts. Ambiguous paragraph citation attachment is excluded; curated annotations must validate against original spans and sources. +- Sidebar preparation explicitly uses `cited-passages-v2`: retain a whole paragraph/list item when only its final sentence is cited, rather than guessing individual sentence attribution. Recognize exact single and comma-grouped `[s1, s2]` citation spans; each cited abstract still receives an independent judgment. The runner validates the declared supported builder and reconstructs the full immutable plan. Legacy CLI/default and saved v1 plans retain their original sentence rules; never silently reinterpret an existing preview. - `settings.py` reads only explicitly requested deployment settings. Clinical/unknown modes reject network evaluation; Jev does not require Gemini credentials. The normal application provider catalog is unchanged. - `typesafe.py` pins Jev 1.13.0 and validates all five-way probabilities, model identity and usage. Adapters issue one attempt; the runner owns retries/deadlines. Transport discards error bodies, rejects compression and bounds success bodies. diff --git a/backend/evidence_review/runner.py b/backend/evidence_review/runner.py index 4f163a1..faebdbb 100644 --- a/backend/evidence_review/runner.py +++ b/backend/evidence_review/runner.py @@ -46,7 +46,11 @@ async def evaluate_snapshot(snapshot: Snapshot, plan: ReviewPlan, *, adapter: Ev snapshot = load_snapshot(canonical_json(snapshot.model_dump(mode="json")), limits=limits) annotations = tuple(SpanAnnotation(start=u.start,end=u.end,citation_spans=u.citation_spans) for u in plan.units if u.origin == "curated") - expected = build_review_plan(snapshot,limits=limits,annotations=annotations) + versions = {u.builder_version for u in plan.units} + if len(versions) > 1: + raise ValueError("evaluation_input_mismatch") + expected = build_review_plan(snapshot,limits=limits,annotations=annotations, + **({'builder_version':next(iter(versions))} if versions else {})) expected = select_review_plan(expected, selected_unit_ids=selected_unit_ids) selection = ([u.unit_id for u in expected.units if u.unit_id in selected_unit_ids] if selected_unit_ids is not None else None) diff --git a/backend/evidence_review/units.py b/backend/evidence_review/units.py index 2f79603..76a3a23 100644 --- a/backend/evidence_review/units.py +++ b/backend/evidence_review/units.py @@ -8,7 +8,9 @@ from .serialization import canonical_json, sha256_bytes BUILDER_VERSION = "sentence-citations-v1" +PASSAGE_BUILDER_VERSION = "cited-passages-v2" _CITE = re.compile(r"\[(s[1-9][0-9]?)\]") +_CITE_GROUP = re.compile(r"\[(s[1-9][0-9]?(?:\s*,\s*s[1-9][0-9]?){0,19})\]") _ABBREVIATION = re.compile(r"\b(?:et al\.|e\.g\.|i\.e\.|Fig\.|Dr\.|vs\.)", re.I) @@ -16,7 +18,7 @@ def _identity(*parts): return sha256_bytes(canonical_json(parts)) -def _sentences(text, offset): +def _sentences(text, offset, citation_pattern=_CITE): protected = {i for match in _ABBREVIATION.finditer(text) for i in range(match.start(),match.end())} numbered = re.match(r"\s*\d+\.\s", text) if numbered: @@ -36,7 +38,7 @@ def _sentences(text, offset): while cursor < len(text): while cursor < len(text) and text[cursor].isspace(): cursor += 1 - citation = _CITE.match(text,cursor) + citation = citation_pattern.match(text,cursor) if not citation: break cursor = citation.end() @@ -59,28 +61,42 @@ def _sentences(text, offset): return spans -def _candidates(answer): +def _candidates(answer, *, group_passages=False): candidates = [] + citation_pattern = _CITE_GROUP if group_passages else _CITE for paragraph in re.finditer(r"\S(?:[\s\S]*?\S)?(?=\n\s*\n|\s*\Z)", answer): text, offset = paragraph.group(), paragraph.start() chunks = [(offset,text)] if re.search(r"(?m)^\s*(?:[-*]\s+|\d+\.\s+)", text): chunks = [(offset+m.start(),m.group()) for m in re.finditer(r"[^\n]+", text)] for position, chunk in chunks: - if chunk.lstrip().startswith("#") or (chunk.rstrip().endswith(":") and not _CITE.search(chunk)): + if chunk.lstrip().startswith("#") or (chunk.rstrip().endswith(":") and not citation_pattern.search(chunk)): candidates.append((position,position+len(chunk),"formatting")) continue - spans = _sentences(chunk,position) - with_cites = [index for index,(start,end) in enumerate(spans) if _CITE.search(answer[start:end])] + spans = _sentences(chunk,position,citation_pattern) + with_cites = [index for index,(start,end) in enumerate(spans) if citation_pattern.search(answer[start:end])] ambiguous = len(spans) > 1 and with_cites == [len(spans)-1] + if ambiguous and group_passages: + # Review the whole cited passage. Never guess which individual + # sentence the terminal citation supports or rewrite its text. + candidates.append((spans[0][0],spans[-1][1],None)) + continue for index,(start,end) in enumerate(spans): candidates.append((start,end,"ambiguous_citation" if ambiguous and index == len(spans)-1 else None)) return candidates -def build_review_plan(snapshot: Snapshot, *, limits: Limits, annotations: tuple[SpanAnnotation,...] = ()) -> ReviewPlan: +def build_review_plan(snapshot: Snapshot, *, limits: Limits, annotations: tuple[SpanAnnotation,...] = (), + builder_version: str = BUILDER_VERSION) -> ReviewPlan: + if builder_version not in {BUILDER_VERSION, PASSAGE_BUILDER_VERSION}: + raise ValueError("unsupported_unit_builder") snapshot = load_snapshot(canonical_json(snapshot.model_dump(mode="json")),limits=limits) answer = snapshot.result.summary + citation_pattern = _CITE_GROUP if builder_version == PASSAGE_BUILDER_VERSION else _CITE + def citations(start,end): + return tuple(CitationSpan(start=m.start(),end=m.end(),source_id=source_id) + for m in citation_pattern.finditer(answer,start,end) + for source_id in re.findall(r's[1-9][0-9]?',m.group(1))) sources = {s.id:s for s in snapshot.result.sources} evidence = {e.citation_id:e for e in snapshot.evidence} annotated = sorted(annotations,key=lambda a:a.start) @@ -89,11 +105,11 @@ def build_review_plan(snapshot: Snapshot, *, limits: Limits, annotations: tuple[ if not 0 <= item.start < item.end <= len(answer) or item.start < previous: raise ValueError("invalid_annotation_span") previous = item.end - actual = {(m.start(),m.end(),m.group(1)) for m in _CITE.finditer(answer,item.start,item.end)} + actual = {(c.start,c.end,c.source_id) for c in citations(item.start,item.end)} provided = {(c.start,c.end,c.source_id) for c in item.citation_spans} if not actual or provided != actual or any(c.source_id not in sources for c in item.citation_spans): raise ValueError("invalid_annotation_citation") - candidates = _candidates(answer) + candidates = _candidates(answer,group_passages=builder_version == PASSAGE_BUILDER_VERSION) # Explicit annotations override intersected automatic spans; remaining text # remains accounted for, conservatively unreviewed if a boundary was cut. for item in annotated: @@ -112,7 +128,7 @@ def build_review_plan(snapshot: Snapshot, *, limits: Limits, annotations: tuple[ eligible_count = 0 for start,end,reason in candidates: text = answer[start:end] - spans = tuple(CitationSpan(start=m.start(),end=m.end(),source_id=m.group(1)) for m in _CITE.finditer(answer,start,end)) + spans = citations(start,end) if reason == "formatting": coverage.append(CoverageItem(start=start,end=end,reason=reason)) continue @@ -127,7 +143,7 @@ def build_review_plan(snapshot: Snapshot, *, limits: Limits, annotations: tuple[ ids = list(dict.fromkeys(c.source_id for c in spans)) unit = ReviewUnit(unit_id=_identity(snapshot.snapshot_sha256,start,end),snapshot_id=snapshot.snapshot_id, text=text,start=start,end=end,citation_spans=spans,evidence_ids=tuple(evidence[i].evidence_id for i in ids if i in evidence), - builder_version=BUILDER_VERSION,origin="curated" if any(a.start==start and a.end==end for a in annotated) else "automatic") + builder_version=builder_version,origin="curated" if any(a.start==start and a.end==end for a in annotated) else "automatic") units.append(unit) pair_ids, reasons = [], [] for source_id in ids: diff --git a/backend/tests/AGENTS.md b/backend/tests/AGENTS.md index 9f2a5b6..38db863 100644 --- a/backend/tests/AGENTS.md +++ b/backend/tests/AGENTS.md @@ -43,6 +43,7 @@ The evidence-review CLI tests exercise real private artifacts and mocked capture - `test_ai_research_settings.py` covers signed owner-only model preferences, persistence/runtime resolution, session invalidation, catalog membership, provider readiness, mode/auth/origin gates and fixed no-store failures. Mock discovery and provider keys; live catalog checks are separate from these network-isolated tests. - `test_ai_evidence_review.py` composes the actual owned service, temporary database and private artifacts with HTTP-only fixtures. It verifies exact text confirmation/selection, payload isolation, ownership, capacity, idempotency/atomic starts, cancellation/backoff, expiry/settings changes, deletion, restart and unknown billing. Clear TypeSafe environment keys and block external sockets; synthetic fixture responses never establish live acceptance. +- Evidence regression coverage includes whole cited passages, grouped citation aliases evaluated separately, unchanged v1 plan behavior, and honest zero-pair status for new/legacy preparations. A ready fixture alone is insufficient: verify actual executable pairs and resolved-model assessments. - `test_ai_evidence_routes.py` exercises production router composition with isolated persistence and synthetic HTTP: all six ownership/auth/mode/origin gates, strict bounded bodies, fixed private failures, preview/start/idempotency, unchanged answer and source deletion. No real key or external network is permitted. - Pre-merge Live regressions cover privacy-class attestation revocation, required update context, unchanged sessions on oversized input, private credential-database failures and invalid mode rejection before research imports. CI includes all three Jev sidebar service/routes/provenance modules as well as the pure evaluator suite. diff --git a/backend/tests/evidence_review/test_units.py b/backend/tests/evidence_review/test_units.py index ce18d1b..98979bd 100644 --- a/backend/tests/evidence_review/test_units.py +++ b/backend/tests/evidence_review/test_units.py @@ -1,6 +1,23 @@ import pytest +def test_cited_passage_groups_terminal_reference_without_guessing_sentence_scope(snapshot_factory): + from backend.evidence_review.contracts import Limits + from backend.evidence_review.units import build_review_plan + passage = '**Finding:** Only seven studies assessed reproducibility. Standardized workflows were recommended. [s1]' + answer = 'Uncited introduction.\n\n' + passage + '\n\nThis is not image analysis.\n' + snapshot = snapshot_factory(answer=answer) + plan = build_review_plan(snapshot, limits=Limits(), builder_version='cited-passages-v2') + assert len(plan.pairs) == 1 + unit = plan.pairs[0].unit + assert unit.text == passage == answer[unit.start:unit.end] + assert unit.origin == 'automatic' and unit.builder_version == 'cited-passages-v2' + assert len(unit.citation_spans) == 1 + assert not build_review_plan(snapshot, limits=Limits()).pairs # Legacy plans remain reproducible. + with pytest.raises(ValueError): + build_review_plan(snapshot, limits=Limits(), builder_version='untrusted-version') + + @pytest.mark.parametrize("answer,expected", [ ("The response was 3.5 and toxicity increased [s1].",1), ("Smith et al. found an effect [s1].",1), diff --git a/backend/tests/test_ai_evidence_review.py b/backend/tests/test_ai_evidence_review.py index 8562cc3..a47b65f 100644 --- a/backend/tests/test_ai_evidence_review.py +++ b/backend/tests/test_ai_evidence_review.py @@ -101,6 +101,37 @@ async def scenario(): asyncio.run(scenario()) +def test_terminal_citation_passage_runs_and_preserves_whole_answer(review): + async def scenario(): + passage = '**Finding:** Synthetic finding. A second related statement. [s1]' + answer = 'Introduction without a citation.\n\n' + passage + '\n\nNot an image assessment.' + detail = await ready(review, summary=answer) + assert len(detail.claims) == 1 and detail.claims[0].text == passage + assert not review.http.submitted + result = await finish(review, detail) + assert result.status == 'completed' and result.completed_pairs == 1 + assert result.original_answer == answer + assert passage in review.http.submitted[0].content.decode() + assert result.assessments[0].resolved_model == 'jev-1.13.0' + asyncio.run(scenario()) + + +def test_empty_preparation_and_legacy_zero_pair_preview_are_not_ready(review): + async def scenario(): + from backend.clinical.ai_evidence_contracts import EvidencePrepareRequest + sid, tid = seed_research(review.live, summary='No inline citation exists.') + detail = await review.service.prepare(review.actor, sid, tid, EvidencePrepareRequest(idempotency_key='empty')) + await review.service.jobs[detail.review_id].task + detail = review.service.get(review.actor, detail.review_id) + assert detail.status == 'unavailable' and detail.reason == 'no_reviewable_claims' + assert not detail.claims and not review.http.submitted + row = review.service.repository.owned(review.actor, detail.review_id) + review.service.repository.update_if_current(row.id, row.generation, status='ready', reason=None) + assert review.service.get(review.actor, row.id).status == 'unavailable' + assert review.service.list(review.actor, sid).reviews[0].status == 'unavailable' + asyncio.run(scenario()) + + def test_foreign_actor_and_tampered_source_are_rejected(review): async def scenario(): detail=await ready(review) @@ -468,5 +499,8 @@ async def scenario(): detail=review.service.get(review.actor,detail.review_id) assert detail.status=='ready',detail.reason assert len(detail.abstracts)==2 and len({a.evidence_id for a in detail.abstracts})==2 + assert detail.total_pairs == 2 and len(detail.claims) == 1 assert not review.http.submitted + result = await finish(review, detail) + assert result.completed_pairs == 2 and result.status == 'completed' asyncio.run(scenario()) diff --git a/desktop/AGENTS.md b/desktop/AGENTS.md index 409ac90..f74d2f9 100644 --- a/desktop/AGENTS.md +++ b/desktop/AGENTS.md @@ -121,7 +121,7 @@ - The evidence-review smoke preserves strict pre-submission consent/focus checks. If OHIF remounts the dock after submission, reacquire the current panel and open the same saved review via its visible GET-only action; never infer failure or success from a detached element, relax request-count assertions, or silently resubmit. -- The evidence-review smoke also checks the visible Jev entry, the latest-result action, research activity cards, persisted progress and recorded dispatch models. Keep these fixture-only assertions separate from actual hosted-provider availability. +- The evidence-review smoke checks the separate Chat/Research/Jev workspaces, composer hidden during review, visible research-card Jev action, latest-result action, persisted progress and recorded dispatch models. Open reviews through the visible research action; internal review hosts live in the dedicated review workspace. Keep these fixture-only assertions separate from actual hosted-provider availability. - The evidence-review smoke first drives typed Send and explicit Research through the real sidebar before voice connects, verifies zero fixture Realtime providers, saved progress/model receipts and Jev eligibility, then ends text and exercises the existing voice/review flow. This uses isolated synthetic workers and does not prove hosted availability. diff --git a/desktop/scripts/ui-import-smoke.mjs b/desktop/scripts/ui-import-smoke.mjs index c8f48fa..06a6330 100644 --- a/desktop/scripts/ui-import-smoke.mjs +++ b/desktop/scripts/ui-import-smoke.mjs @@ -1298,6 +1298,8 @@ async function exerciseLiveViewer(providerId, keepOpen = false) { profile.inputSampleRate !== expectedRate || profile.outputSampleRate !== 24000 || !profile.screen || !profile.tools) { throw new Error('Synthetic provider profile did not match the selected audio/action contract'); } + panel.querySelector('[data-action="view-chat"]').click(); + panel.querySelector('[data-role="voice-options"]').open = true; const providerSelect = panel.querySelector('[data-role="provider"]'); if (providerSelect && !providerSelect.options.length) { if (button('connect').textContent !== 'Retry setup') throw new Error('Initial setup failure did not expose an actionable retry'); @@ -1308,7 +1310,7 @@ async function exerciseLiveViewer(providerId, keepOpen = false) { if (!providerSelect || providerSelect.disabled || !Array.from(providerSelect.options).some(option => option.value === providerId)) throw new Error(`Provider dropdown unavailable: ${JSON.stringify({ present: Boolean(providerSelect), disabled: providerSelect?.disabled, values: providerSelect ? Array.from(providerSelect.options).map(option => option.value) : [], status: panel.state.backendStatus, message: panel.querySelector('[data-role="status"]')?.textContent })}`); providerSelect.value = providerId; providerSelect.dispatchEvent(new Event('change', { bubbles: true })); await waitFor(() => providerSelect.value === providerId && panel.state.backendStatus === 'disconnected' && - panel.querySelector('[data-role="disclosure"]').textContent.includes(profile.label), 'Provider selection did not update the sidebar'); + providerSelect.title === expectedModel && providerSelect.selectedOptions[0]?.textContent.includes(profile.label), 'Provider selection did not update the sidebar'); const select = panel.querySelector('#radsysx-live-attestation'); if (select.value !== '') throw new Error('Provider selection did not require fresh data confirmation'); select.value = 'synthetic'; select.dispatchEvent(new Event('change', { bubbles: true })); @@ -2459,6 +2461,10 @@ async function exerciseEvidenceReview(phase, prior = {}) { }; const card = id => panel.querySelector(`[data-evidence-tool="${id}"]`); const button = (host,action) => host.querySelector(`[data-evidence-action="${action}"]`); + const openReview = id => { + panel.querySelector('[data-action="view-research"]').click(); + panel.querySelector(`[data-action="open-review"][data-id="${id}"]`).click(); + }; const counters = () => api('_fixture/evidence'); const sid = prior.sessionId ?? panel.state.backendSessionId; if (phase==='prepare') { @@ -2468,13 +2474,16 @@ async function exerciseEvidenceReview(phase, prior = {}) { const history = await api(`sidebar/sessions/${sid}`); assert(card('smoke-research-one'), 'Missing public PubMed fixture / Review evidence with Jev action'); assert(panel.querySelector('[aria-label="Jev evidence review"]') && !panel.querySelector('[data-action="review-latest"]').disabled, 'Prominent Jev action is missing or unavailable'); - assert(panel.querySelector('[data-role="research-tools"]').textContent.includes('Literature research'), 'Research activity is not visible above the transcript'); + panel.querySelector('[data-action="view-research"]').click(); + assert(!panel.querySelector('[data-role="research-view"]').hidden && panel.querySelector('[data-role="chat-view"]').hidden, 'Research and chat workspaces are not separated'); + assert(panel.querySelector('[data-role="research-tools"] [data-action="open-review"]'), 'Research result has no explicit Jev action'); assert(history.events.some(event=>event.kind==='research_progress' && event.stage==='searching_pubmed'), 'Research progress was not journaled through the broker'); assert(history.tools.filter(tool=>tool.name==='research_run').every(tool=>tool.research?.modelId), 'Research cards have no recorded dispatch model'); const original = history.tools.find(t=>t.toolCallId==='smoke-research-one').result; await wait(()=>!button(card('smoke-research-one'),'open').disabled,'Review button remained busy'); - button(card('smoke-research-one'),'open').click(); - await wait(()=>card('smoke-research-one').textContent.includes('Ready to review'),()=> 'Preview did not become ready: '+card('smoke-research-one').textContent); + openReview('smoke-research-one'); + await wait(()=>card('smoke-research-one').textContent.includes('Ready for your confirmation'),()=> 'Preview did not become ready: '+card('smoke-research-one').textContent); + assert(!panel.querySelector('[data-role="review-view"]').hidden && panel.querySelector('[data-role="composer"]').hidden, 'Review workspace did not hide the composer'); const host = card('smoke-research-one'); const foreground=getComputedStyle(host.querySelector('[data-evidence-detail] > p:not([data-evidence-message])')).color; assert(foreground!=='rgb(0, 0, 0)','Review text is unreadable on the dark card'); @@ -2516,7 +2525,7 @@ async function exerciseEvidenceReview(phase, prior = {}) { // through GET; pre-submission consent/focus still has the strict check above. if(current && !first.isConnected && !button(current,'open').disabled) { first=current; - if(!button(first,'open').hidden)button(first,'open').click(); + if(!button(first,'open').hidden)openReview('smoke-research-one'); } return first.isConnected && first.textContent.includes('Supported by this abstract'); },()=> 'Completed judgment did not appear: '+JSON.stringify({connected:first.isConnected,current:card('smoke-research-one')?.textContent})); @@ -2542,8 +2551,8 @@ async function exerciseEvidenceReview(phase, prior = {}) { const catalog=await api('sidebar/research-settings/models/nvidia_nim'); assert(catalog.models.length===82 && catalog.models.every(id=>[...modelSelect.options].some(o=>o.value===id)),'NVIDIA models were filtered'); panel.querySelector('[data-action="close-credentials"]').click(); - const second=card('smoke-research-two');panel.querySelector('[data-action="review-latest"]').click(); - await wait(()=>second.textContent.includes('Ready to review'),'Second review preview failed'); + const second=card('smoke-research-two');panel.querySelector('[data-action="view-review"]').click();panel.querySelector('[data-action="review-latest"]').click(); + await wait(()=>second.textContent.includes('Ready for your confirmation'),'Second review preview failed'); const confirmation=second.querySelector('[data-evidence-confirmation]');confirmation.value='synthetic';confirmation.dispatchEvent(new Event('change',{bubbles:true}));button(second,'start').click(); await wait(async()=> (await counters()).submitted===2,'Blocked review was not submitted'); panel.querySelector('[data-action="end"]').click(); @@ -2554,7 +2563,7 @@ async function exerciseEvidenceReview(phase, prior = {}) { assert((await api(`sidebar/evidence-reviews/${secondSummary.reviewId}`)).unknownUsageAttempts===1,'Cancelled submitted attempt lost unknown usage'); assert(!button(second,'prepare').hidden,'Cancelled review has no fresh preparation action'); button(second,'prepare').click(); - await wait(()=>second.textContent.includes('Ready to review'),'Fresh abstract preparation failed'); + await wait(()=>second.textContent.includes('Ready for your confirmation'),'Fresh abstract preparation failed'); const freshSummary=(await api(`sidebar/sessions/${sid}/evidence-reviews`)).reviews.find(r=>r.toolCallId==='smoke-research-two'); assert(freshSummary.reviewId!==secondSummary.reviewId,'Fresh preparation reused cancelled review'); assert((await api(`sidebar/evidence-reviews/${secondSummary.reviewId}`)).status==='cancelled','Fresh preparation replaced prior receipt'); @@ -2562,7 +2571,7 @@ async function exerciseEvidenceReview(phase, prior = {}) { assert((await counters()).submitted===2,'Fresh abstract preparation inferred before confirmation'); panel.querySelector('[data-action="history"]').click();await wait(()=>panel.querySelector(`[data-action="read-history"][data-id="${sid}"]`),'History action missing'); panel.querySelector(`[data-action="read-history"][data-id="${sid}"]`).click(); - await wait(()=>panel.querySelector('[data-role="status"]').textContent.includes('Viewing saved conversation'),'Saved history did not open'); + await wait(()=>panel.querySelector('[data-role="status"]').textContent.includes('Saved conversation'),'Saved history did not open'); await wait(()=>!button(card('smoke-research-one'),'open').disabled,'Saved review summaries did not load'); // Delay a genuine owned GET across a history switch, even after its abort. const other=await api('sidebar/sessions',{}); @@ -2574,7 +2583,7 @@ async function exerciseEvidenceReview(phase, prior = {}) { } return response; }; - button(card('smoke-research-one'),'open').click();await wait(()=>releaseOld,'Saved review GET was not captured'); + openReview('smoke-research-one');await wait(()=>releaseOld,'Saved review GET was not captured'); panel.querySelector('[data-action="history"]').click();await wait(()=>panel.querySelector(`[data-action="read-history"][data-id="${other.sessionId}"]`),'Other history missing'); panel.querySelector(`[data-action="read-history"][data-id="${other.sessionId}"]`).click();await wait(()=>!card('smoke-research-one'),'History did not switch'); window.fetch=originalFetch;releaseOld();await new Promise(resolve=>setTimeout(resolve,100)); @@ -2582,10 +2591,9 @@ async function exerciseEvidenceReview(phase, prior = {}) { panel.querySelector('[data-action="history"]').click();await wait(()=>panel.querySelector(`[data-action="read-history"][data-id="${sid}"]`),'Original history missing'); panel.querySelector(`[data-action="read-history"][data-id="${sid}"]`).click();await wait(()=>card('smoke-research-one') && !button(card('smoke-research-one'),'open').disabled,'Original review card did not reload'); await api(`sidebar/sessions/${other.sessionId}`,undefined,'DELETE'); - button(card('smoke-research-one'),'open').click();await wait(()=>card('smoke-research-one').textContent.includes('Supported by this abstract'),'Saved receipt was not restored'); + openReview('smoke-research-one');await wait(()=>card('smoke-research-one').textContent.includes('Supported by this abstract'),'Saved receipt was not restored'); assert((await counters()).submitted===2,'Reopen triggered inference'); - // Expand the real receipt for the retained synthetic screenshot. - card('smoke-research-one').querySelectorAll('details').forEach(node=>{if(node.querySelector('summary')?.textContent.includes('Execution receipt'))node.open=true;}); + // Keep technical details collapsed in the final reading-flow screenshot. card('smoke-research-one').scrollIntoView({block:'start'}); return {...prior,status:receipt.status,resolvedModel:receipt.assessments[0].resolvedModel,submitted:2,completedPairs:1,excludedSubmitted:false,unknownUsageAttempts:1,unchangedAnswer:true,reopenWithoutInference:true,repreparedWithoutInference:true,endVoiceIndependent:true,staleHistoryReplyDiscarded:true,geometry,nvidiaModelCount:catalog.models.length}; } @@ -2605,6 +2613,7 @@ async function exerciseTextWithoutVoice() { const sid=panel().state.backendSessionId; const chat=await api('sidebar/sessions/'+sid); if(chat.session.mode!=='text' || chat.session.liveUrl!==null)throw Error('Text request allocated a voice session'); + panel().querySelector('[data-action="view-research"]').click(); fill('Find public literature for synthetic research one.');panel().querySelector('[data-action="research"]').click(); await wait(async()=> (await api('sidebar/sessions/'+sid)).tools.some(t=>t.name==='research_run' && t.status==='completed')); await wait(()=> !panel().querySelector('[data-action="review-latest"]').disabled); @@ -2619,5 +2628,7 @@ async function exerciseTextWithoutVoice() { try {panel().querySelector(`[data-action="clear-history"][data-id="${sid}"]`).click();} finally {window.confirm=originalConfirm;} await wait(()=>!panel().state.backendSessionId); if(!panel().querySelector('[data-role="history"]').hidden)panel().querySelector('[data-action="history"]').click(); + panel().querySelector('[data-action="view-chat"]').click(); + panel().querySelector('[data-role="voice-options"]').open=true; return {chat:true,research:true,jevEligible:true,voiceConnections:0,modelRecorded:completed.tools.every(t=>Boolean(t.research?.modelId))}; } diff --git a/roadmap/ai-backend/JEV_SIDEBAR_IMPLEMENTATION.md b/roadmap/ai-backend/JEV_SIDEBAR_IMPLEMENTATION.md index 0815c6c..5020773 100644 --- a/roadmap/ai-backend/JEV_SIDEBAR_IMPLEMENTATION.md +++ b/roadmap/ai-backend/JEV_SIDEBAR_IMPLEMENTATION.md @@ -20,12 +20,12 @@ Each source receives its own judgment: Supported by this abstract, Partially sup - One active review per actor, two globally. Preparation is bounded to 20 seconds and evaluation/cleanup to 70 seconds. Existing 40-pair/20-abstract/input/attempt/retry limits remain enforced. - End voice or change viewport without cancelling a separate review. Explicit review cancellation, account settings changes, logout/expiry, disable and shutdown stop it. Committed pairs survive partial failures; submitted attempts without a receipt show unknown usage/billing. - Reopen/Refresh reads saved state without inference. Retry unfinished work requires renewed confirmation and reuses only exactly matching completed pairs. Restart interrupts work without replay. A confirmed operation interrupted before evaluator initialization can retry from its persisted selection only when the manifest is exactly the preparation-only checkpoint; initialized evaluations retain strict resume checks. Generation-guarded browser polling uses 1/2/4-second backoff and stops after 100 seconds until explicit refresh. -- Failed/interrupted preparation, cancelled/unavailable reviews or a preview with no eligible claims offers **Fetch abstracts again**. This explicitly creates a fresh owned preview, preserves previous records and requires fresh text confirmation before inference; reopening selects the latest preparation. +- Failed/interrupted preparation, cancelled/unavailable reviews or a preview with no eligible claims offers **Prepare review again**. This explicitly creates a fresh owned preview, preserves previous records and requires fresh text confirmation before inference; reopening selects the latest preparation. - Source-history deletion invalidates jobs, joins cancellation, removes private artifacts and then clears history. Failed cleanup retains a deletion marker for retry/startup; late callbacks cannot recreate deleted rows. Missing artifacts show unavailable, never implicit retrieval/inference. - Private immutable artifacts use `.ai-evidence/` beside the actual file-backed database, POSIX 0700 directories/0600 files and existing no-symlink/hash guarantees. Set absolute `RADSYSX_AI_EVIDENCE_DIR` for a non-file database. No private locator reaches browser DTOs. Preserve the private directory with database backups. - Direct `python backend/server.py` and package imports are supported. The pure evaluator and existing standalone CLI remain independently usable. -The AI sidebar uses a charcoal/slate reading-room palette with muted blue accents, subdued borders, dark scrollbars and no bright mint fills or status glow. Details expand inside research cards; permanent header space stays compact. The full NVIDIA research dropdown is retained. Native ChatNVIDIA still runs inside the bounded DeepAgents/LangGraph research lane; Jev itself uses the independent native evaluation service. +The AI sidebar uses a charcoal/slate reading-room palette with muted blue accents, subdued borders, dark scrollbars and no bright mint fills or status glow. Chat, Research and Jev review have separate workspaces; source abstracts and technical receipts stay collapsed until requested. The full NVIDIA research dropdown is retained. Native ChatNVIDIA still runs inside the bounded DeepAgents/LangGraph research lane; Jev itself uses the independent native evaluation service. ## Verification on 2026-09-22 @@ -74,3 +74,15 @@ The reviewer separately declined to establish diagnostic accuracy or repeat exte The user authorized PR creation and merge after implementation sign-off. Reviewing the still-open live-assistant foundation PR exposed missing context validation, privacy-class changes retaining earlier attestation, credential-database errors escaping the private error boundary, and inconsistent mode parsing. Nine regression cases reproduced these failures before the fixes. The combined pre-merge backend suite passed 563 tests; viewer 41 and desktop 17 tests, shared type checks and Python compilation passed. CI now includes the Jev sidebar service, routes and provenance suites. A case-insensitive HTML-escaping assertion covers lower/upper-case tags without the test-only filtering-regexp pattern flagged by CodeQL. Hosted checks and the merge outcome are recorded on the integration PR. The pre-merge Electron repeat also passed the full Jev workflow. Earlier repeats showed that its harness retained a detached OHIF panel after submission despite a completed backend review. It now reacquires the actual dock and opens the same saved review through GET after remount; exact consent/focus assertions before submission, inference counts, cancellation, deletion and narrow layout assertions remain intact. + +## Sidebar clarity and live recovery — 2026-09-23 + +The user reported an unstructured rail and a Ready review with zero claims. Two real extraction gaps were present: a multi-sentence passage with a terminal citation was conservatively excluded, and comma-grouped citation IDs produced no units. Preparation nevertheless advertised ready. New sidebar previews use `cited-passages-v2`: the full cited passage is one review unit, preserving exact offsets/text, with each cited abstract judged independently. No individual-sentence attribution is invented. The runner rebuilds the versioned plan; old CLI/default and saved v1 plans retain their prior semantics. Zero executable pairs are unavailable, including legacy ready/zero-pair rows; explicit Prepare review again makes a fresh preview and clears consent. + +The rail now separates Chat, Research and Jev review. The text/research model is named in the header; optional voice setup is collapsed in Chat. Research cards show formatted answers, one Jev action, and collapsed sources/receipts. The review workspace hides the composer and unrelated results, shows preparation versus inference plainly, and includes only source abstracts attached to selected passages. Hashes, probabilities and original-answer copies remain under details. The old side-accent separator and its design-detector exception were removed; judgment groups use horizontal separators. + +Native Electron acceptance used the actual desktop origin and the saved public radiomics answer from subscription acceptance. Its old empty review displayed No reviewable passages; Prepare review again produced one intact passage linked to PubMed 42719846. After inspecting the public text and selecting public-literature confirmation, Start Jev review completed with **supported**, resolved **jev-1.13.0**, and reported **1,101 input / 60 output tokens**. Backend GET confirmed one completed assessment and an exactly unchanged original answer. This used real NCBI and TypeSafe, with no image or patient content. It establishes execution, not clinical validity or general judgment accuracy. The ignored private receipt is `tmp/codex-acceptance/jev-sidebar-receipt.json`. + +Focused validation: 248 backend/evaluator tests, 50 viewer tests, viewer type checking and the production viewer build passed. The isolated synthetic Electron evidence smoke also passed: independent text/research with zero Realtime connections, workspace separation, exact selection/consent, keyboard focus, saved receipts, cancellation with unknown usage, voice independence, history/deletion, the complete fixture NVIDIA dropdown, and no horizontal overflow at 280px. Native inspection and one real public Jev request were exercised separately from fixtures. The final synthetic screenshot keeps the judgment visible and technical details collapsed. Hosted CI results are recorded on the fix PR. + +Design audit: the changed rail has no remaining detector findings; no new suppressions were added. The pre-existing grid-background advisory belongs to the unrelated local-viewer illustration and remains unchanged and unsuppressed. Representative body, subdued-copy and navigation contrasts range from 6.52:1 to 11.74:1. Full screen-reader, mobile/touch and clinical validation were not performed. diff --git a/viewer/assets/AGENTS.md b/viewer/assets/AGENTS.md index e8bedc9..2e336cf 100644 --- a/viewer/assets/AGENTS.md +++ b/viewer/assets/AGENTS.md @@ -46,8 +46,8 @@ - `viewer/assets/live/AGENTS.md`: typed Live controller, PCM media, semantic OHIF adapter, and sidebar. - The live sidebar uses compact connection/media controls and a Settings overlay containing account-owned research provider/model dropdowns and API-key inputs. Keep CSS aligned with the typed panel and its hidden-state contract; media controls do not occupy space before connection. -- Evidence review styles belong inside the existing scrollable research card. No additional permanent header region; checkbox/select controls, long hashes/source text and receipts must fit the 280 px sidebar without horizontal overflow. +- Evidence review uses its own scrollable workspace, reached from a research result or the Jev review navigation button. Hide the composer during review; checkbox/select controls, long source text and collapsed receipts must fit a 280 px sidebar without horizontal overflow. - The radiologist-facing AI sidebar uses the restrained reading-room palette in `radsysx-viewer.css`: charcoal/slate surfaces, muted blue actions, dark scrollbars and no luminous status glow. Scope the theme to `.radsysx-live-shell` and its controls; do not tint diagnostic canvases or change image presentation. -- The Jev entry and research activity cards live inside the conversation scroll region, not in an expanded permanent header. Keep long model IDs wrapped and restrained reading-room colors; evidence eligibility, confirmation and receipts remain owned by the typed controller. +- Chat, Research and Jev review each have one visible workspace within the shared scroll region. Keep long model IDs wrapped and restrained reading-room colors; evidence eligibility, confirmation and receipts remain owned by the typed controller. diff --git a/viewer/assets/live/AGENTS.md b/viewer/assets/live/AGENTS.md index 9d3cd62..f605be3 100644 --- a/viewer/assets/live/AGENTS.md +++ b/viewer/assets/live/AGENTS.md @@ -34,11 +34,11 @@ - Coordinate the desktop synthetic Live smoke after the viewer build. Unit/fake-provider evidence is separate from authenticated Gemini, microphone/speaker hardware, clinical capture, and Orthanc-backed persistence acceptance. - **Settings → Text & research models** loads backend-owned current selection and provider model catalogs. Preserve the exact saved model on discovery failure, show all returned NVIDIA IDs (no filtering or truncation), and offer explicit refresh/reload. Catalog replies use generation guards so a late response cannot replace a newer provider choice. Save only an explicit loaded choice; clear attestation and stop the active connection before saving, then refresh availability. The backend terminates all of the account's sessions/jobs. Keep unsaved dropdown choice intact when API-key status changes and update Gemini research configuration readiness from the credential receipt. No research preference lives in browser storage. -- Sidebar controls are compact: brand, voice-provider selector, Settings and history share one row; data confirmation and Connect share a row with the provider transmission disclosure underneath. Show microphone/image/stop/end controls only during connecting/ready/reconnecting, with microphone/sharing still disabled before backend readiness. Keep status errors visible and preserve the conversation/composer space. The exact voice model is the provider selector tooltip. +- Sidebar controls are compact: brand, Settings and history share the header, followed by the active text/research model and Chat / Research / Jev review navigation. Optional voice setup is a collapsed disclosure in Chat; active voice remains visible across views. Data confirmation and transmission disclosure sit beside the composer, which is hidden during review. Only acknowledged readiness enables microphone/sharing. The exact voice model remains the provider selector tooltip. ## Explicit saved-result Jev review -- `evidence.ts` owns independent prepare/start/retry/cancel and saved-result GET polling; `evidence-panel.ts` renders exact frozen claims/abstracts, exclusions, five abstract-scoped judgments and execution receipts inside completed PubMed research cards. It never feeds judgments into conversation, reports or viewer actions. +- `evidence.ts` owns independent prepare/start/retry/cancel and saved-result GET polling; `evidence-panel.ts` renders exact frozen claims/abstracts, exclusions, five abstract-scoped judgments and execution receipts in a dedicated review workspace reached from completed PubMed research cards. It never feeds judgments into conversation, reports or viewer actions. - Require separate public-literature/synthetic text confirmation and a nonempty backend-issued claim selection. Image attestation is insufficient. No edited prose or browser evidence payload is accepted. Clear consent on selection/preview/session/account change and every attempted start/retry; retain an uncertain operation's key for retransmission. - Guard every awaited response by generation, even if abort is ignored. One poll at a time, 1/2/4-second backoff, 100-second limit and explicit GET-only Refresh. End voice and collapse preserve background reviews. Session changes/unmount stop subscriptions; saved history/remount reads results without inference. Settings/account changes invalidate outstanding replies and consent. - Keep one stable host per tool. Transcript and sibling tool updates must preserve focused selection/confirmation controls; status and receipt regions update independently. Native labels/details and escaped text keep controls accessible. Configuration appears as one Settings row; only a saved resolved-model receipt establishes completion. Never reduce NVIDIA catalog options to accommodate review UI. @@ -47,9 +47,9 @@ - User preference (2026-09-22): the AI sidebar is for radiologists in a reading room. Use an understated charcoal/slate palette with muted blue accents, readable subdued text, soft borders and dark native scrollbars. Avoid bright mint/white button fills, neon glows and distracting animation. Preserve visible keyboard focus, clear states and compact conversation space. Theme only the AI surface, preserving diagnostic image rendering. -- Failed/interrupted preparation, cancelled/unavailable reviews and previews without eligible claims expose an explicit **Fetch abstracts again** action. Create a new owned preparation without modifying prior records or making Jev calls; clear consent and require confirmation for its new preview. Reopen selects the newest saved review for that tool. Refresh remains GET-only, and uncertain preparation retransmission keeps its operation key. +- Failed/interrupted preparation, cancelled/unavailable reviews and previews without eligible claims expose an explicit **Prepare review again** action. Create a new owned preparation without modifying prior records or making Jev calls; clear consent and require confirmation for its new preview. Reopen selects the newest saved review for that tool. Refresh remains GET-only, and uncertain preparation retransmission keeps its operation key. -- Keep the compact Jev entry visible in the scrollable conversation before any research result, explaining the prerequisite and offering Saved research. Review latest evidence opens the latest eligible PubMed card through the existing exact-preview workflow; it never starts inference or bypasses text confirmation. Research cards precede the transcript and display backend-recorded provider/model and allowlisted worker stages; unknown historic models stay unknown. Terminal status overrides stale progress. Ending voice refreshes owned tool receipts after closing; if unavailable, pending tasks show status unconfirmed rather than continuing to claim they are running. +- Keep Chat transcript, Research results and Jev review in separate persistent workspaces. Research cards show the question, status, formatted answer and one Jev action; source lists and execution details are collapsed. Jev review hides the composer and unrelated results, explains preparation versus actual inference, and shows only abstracts linked to selected passages. Empty previews never claim readiness. Workspace navigation does not infer or reset confirmation. Reopening, account/session/selection changes clear confirmation; GET refresh never infers. Terminal status overrides stale progress; ending voice refreshes owned receipts or shows status unconfirmed. ## Text without a voice connection @@ -58,5 +58,7 @@ ## ChatGPT / Codex subscription +- `presentation.ts` formats escaped paragraphs, bold spans, inline code and lists only. It never enables generated HTML, images or links. The original answer/claim bytes remain unchanged in backend evidence artifacts. In the Research composer, Enter runs research; in Chat, Enter sends text. Both retain the same unsent draft and attestation requirements. + - `subscription.ts` owns Settings login/status/sign-out and bounded GET-only polling. Keep the OAuth URL transient and validate HTTPS official hosts; never handle tokens/passwords or open model-provided destinations. Show confirmed account/plan, preserve the existing text/research model until explicit selection/save, and distinguish subscription allowance from voice API billing. Stop polling and discard stale responses after settings closure/disposal. Account changes stop owned work and clear attestation before dispatch. - Codex models come from the authenticated backend catalog. Recorded provider/model receipts must label ChatGPT/Codex subscription; no voice/audio session is created for text. Research uses the backend PubMed dynamic tool and retains the existing explicit Jev preview/confirmation. diff --git a/viewer/assets/live/controller.ts b/viewer/assets/live/controller.ts index 786bb09..389d029 100644 --- a/viewer/assets/live/controller.ts +++ b/viewer/assets/live/controller.ts @@ -6,6 +6,7 @@ import { OHIFAdapter } from './ohif.js'; import { EventGate, TranscriptStore, object, parseEvent, request, safeUrl, toolFromWire, type AIResearchSettings, type AIResearchModels, type ResearchProviderId, type AICredentialStatusResponse, type Attestation, type ProviderId, type ProviderProfile, type AudioChunk, type CaptureRequest, type Citation, type DesktopCapture, type Json, type ServerEvent, type Session, type SavedConversation, type Tool } from './protocol.js'; export class LiveController { + sidebarView: 'chat' | 'research' | 'review' = 'chat'; readonly evidence = new EvidenceController(() => this.emit()); readonly subscription = new SubscriptionController(() => this.emit(), async () => { this.evidence.dispose(); this.requireAttestation(); diff --git a/viewer/assets/live/evidence-panel.ts b/viewer/assets/live/evidence-panel.ts index 73e3592..8ce42a0 100644 --- a/viewer/assets/live/evidence-panel.ts +++ b/viewer/assets/live/evidence-panel.ts @@ -1,5 +1,6 @@ import { escape, object, type EvidenceLabel, type EvidenceReviewDetail, type EvidenceReviewSummary, type Tool } from './protocol.js'; import { EvidenceController } from './evidence.js'; +import { answerMarkup, inlineMarkup } from './presentation.js'; const labels: Record = { supported: 'Supported by this abstract', partially_supported: 'Partially supported', contradicted: 'Contradicted by this abstract', mixed: 'Mixed', not_addressed: 'Not addressed' }; const statuses: Record = { preparing: 'Preparing', ready: 'Ready to review', reviewing: 'Reviewing', completed: 'Completed', partial: 'Partially completed', failed: 'Failed', cancelled: 'Cancelled', interrupted: 'Interrupted', unavailable: 'Unavailable' }; @@ -14,41 +15,48 @@ export function evidenceEligibility(tool: Tool): string | null { return null; } export function renderEvidenceSummary(summary: EvidenceReviewSummary): string { - return `Jev · ${escape(statuses[summary.status] ?? summary.status)}${summary.completedPairs} of ${summary.totalPairs} judgments completed · ${summary.settledPairs} pairs settled${summary.unknownUsageAttempts ? `${summary.unknownUsageAttempts} attempts with unknown usage/billing` : ''}`; + const empty = summary.status === 'ready' && !summary.totalPairs || summary.reason === 'no_reviewable_claims'; + const title = empty ? 'No reviewable passages' : summary.status === 'ready' ? 'Ready for your confirmation' : `Jev · ${statuses[summary.status] ?? summary.status}`; + const progress = empty ? 'Nothing has been sent to Jev.' : summary.status === 'ready' ? `${summary.totalPairs} abstract check${summary.totalPairs === 1 ? '' : 's'} prepared. Jev has not run yet.` : summary.status === 'preparing' ? 'Fetching the cited abstracts. Jev has not run yet.' : `${summary.completedPairs} of ${summary.totalPairs} abstract checks completed`; + return `${escape(title)}${escape(progress)}${summary.unknownUsageAttempts ? `${summary.unknownUsageAttempts} attempts with unknown usage/billing` : ''}`; } /** Public text is escaped; receipts distinguish execution from abstract support. */ -export function renderEvidenceDetail(detail: EvidenceReviewDetail): string { - const abstracts = detail.abstracts.map(source => `
${escape(source.citationId)} · ${escape(source.title)} · ${escape(source.completeness)} +export function renderEvidenceDetail(detail: EvidenceReviewDetail, selection?: Set): string { + const selected = selection ?? new Set(detail.selectedUnitIds ?? detail.claims.filter(claim => claim.eligible).map(claim => claim.unitId)); + const sourceIds = new Set(detail.claims.filter(claim => selected.has(claim.unitId)).flatMap(claim => claim.evidenceIds)); + const sources = detail.abstracts.filter(source => sourceIds.has(source.evidenceId)); + const abstracts = sources.map(source => `
${escape(source.citationId)} · ${escape(source.title)} ${canonicalPubMed(source.url) ? `PubMed ${escape(source.pmid)} ↗` : ''}

Abstract fetched for this review · ${escape(source.retrievedAt)}. It may differ from the text seen during research.

${source.completeness !== 'complete' ? '

Not reviewed · complete evidence is unavailable.

' : ''} ${source.sections.map(section => `
${section.label ? `${escape(section.label)}` : ''}

${escape(section.text)}

`).join('')} -
${row('Abstract SHA-256', source.textSha256)}${row('Extraction version', source.extractionVersion)}
`).join(''); +
Source provenance
${row('Abstract SHA-256', source.textSha256)}${row('Extraction version', source.extractionVersion)}
`).join(''); const judgments = detail.assessments.map(assessment => { const claim = detail.claims.find(value => value.unitId === assessment.unitId); const source = detail.abstracts.find(value => value.evidenceId === assessment.evidenceId); - return `

${escape(claim?.text ?? 'Claim unavailable')}

${assessment.label ? labels[assessment.label] : 'Not reviewed'} + return `
${assessment.label ? labels[assessment.label] : 'Not reviewed'}
${answerMarkup(claim?.text ?? 'Claim unavailable')}

${escape(source?.citationId ?? assessment.evidenceId)} · ${escape(source?.title ?? 'Source unavailable')}${assessment.reason ? ` · ${escape(assessment.reason)}` : ''}

Execution receipt · ${assessment.reused ? 'Reused' : 'New'}
${row('Pair ID', assessment.pairId)}${row('Status',assessment.status)}${row('Requested reviewer',assessment.requestedModel)}${row('Resolved reviewer',assessment.resolvedModel)}${row('Rubric',assessment.rubricVersion)}${row('Rubric SHA-256',assessment.rubricSha256)}${row('Answer SHA-256',assessment.answerSha256)}${row('Abstract SHA-256',assessment.abstractSha256)}${row('Request SHA-256',assessment.requestSha256)}${row('Attempt IDs',assessment.attemptIds.join(', '))}
${assessment.probabilities ? `
Model probabilities

Model output; not a probability of clinical truth.

${Object.entries(assessment.probabilities).map(([key,value]) => row(labels[key as EvidenceLabel],value)).join('')}
` : ''}
`; }).join(''); - return `
Original answer · unchanged

${escape(detail.originalAnswer ?? '')}

- ${judgments}
${abstracts}
+ return `${judgments}
Source abstracts · ${sources.length}

These are the original abstracts linked to the selected passages.

${abstracts}
+
Original answer · unchanged
${answerMarkup(detail.originalAnswer ?? '')}
${detail.exclusions.length ? `
Not reviewed · exclusions (${detail.exclusions.length})
    ${detail.exclusions.map(exclusion => `
  • ${escape(detail.claims.find(c => c.unitId === exclusion.unitId)?.text ?? exclusion.citationId ?? 'Source')} · ${escape(exclusion.reason)}
  • `).join('')}
` : ''} -
Review receipt and attempts
${row('Review ID',detail.reviewId)}${row('Source context version',detail.sourceContextVersion)}${row('Generation provider',detail.generation.providerId)}${row('Generation model',detail.generation.modelId)}${row('Generation recorded',detail.generation.recordedAt)}${row('Reviewer',detail.modelId)}${row('Preview SHA-256',detail.previewSha256)}${row('Answer SHA-256',detail.answerSha256)}${row('Snapshot SHA-256',detail.snapshotSha256)}${row('Created',detail.createdAt)}${row('Updated',detail.updatedAt)}${row('Reason',detail.reason)}${row('Submitted attempts',detail.submittedAttempts)}${row('Unknown usage/billing',detail.unknownUsageAttempts)}
+
Execution details
${row('Review ID',detail.reviewId)}${row('Source context version',detail.sourceContextVersion)}${row('Generation provider',detail.generation.providerId)}${row('Generation model',detail.generation.modelId)}${row('Generation recorded',detail.generation.recordedAt)}${row('Reviewer',detail.modelId)}${row('Preview SHA-256',detail.previewSha256)}${row('Answer SHA-256',detail.answerSha256)}${row('Snapshot SHA-256',detail.snapshotSha256)}${row('Created',detail.createdAt)}${row('Updated',detail.updatedAt)}${row('Reason',detail.reason)}${row('Submitted attempts',detail.submittedAttempts)}${row('Unknown usage/billing',detail.unknownUsageAttempts)}
${detail.earlierAttemptCount ? `

${detail.earlierAttemptCount} earlier attempts retained in private storage; latest ${detail.attempts.length} shown.

` : ''} ${detail.attempts.map(attempt => `
Attempt ${escape(attempt.attemptId)}
${row('Pair ID',attempt.pairId)}${row('Submitted',attempt.submitted)}${row('Started',attempt.startedAt)}${row('Ended',attempt.endedAt)}${row('Reason',attempt.reason)}${row('Request SHA-256',attempt.requestSha256)}${row('Reported usage',attempt.usage ? JSON.stringify(attempt.usage) : 'Unknown usage/billing')}
`).join('')}
`; } /** One stable host per research card; only status/receipt regions change during polling. */ -export function mountEvidencePanel(host: HTMLElement, controller: EvidenceController): {update(): void; dispose(): void} { +export function mountEvidencePanel(host: HTMLElement, controller: EvidenceController, onClose?: () => void): {update(): void; dispose(): void} { const toolId = host.dataset.evidenceTool!; host.classList.add('radsysx-evidence'); host.innerHTML = `
-

-
- Jev evidence review -

Check claims for support, contradictions and gaps in their cited abstracts.

- + + +
A second set of hands.
Discuss the image, change a view, or research a question.
-
+
-
+ +
+

${credentialSettingsMarkup()} `; @@ -140,12 +160,13 @@ export function registerPanel(controller: LiveController): void { const button = (event.target as Element).closest('button'); if (!button || button.disabled) return; const action = button.dataset.action; - if (action === 'connect') void (controller.providers.length ? controller.connect(this.attestation) : controller.initialize()); + if (action?.startsWith('view-')) { this.showView(action.slice(5) as 'chat' | 'research' | 'review'); } + else if (action === 'connect') void (controller.providers.length ? controller.connect(this.attestation) : controller.initialize()); else if (action === 'voice') void controller.toggleMicrophone(); else if (action === 'stop-speaking') controller.stopSpeaking(); else if (action === 'share') void controller.toggleSharing(); else if (action === 'end' || action === 'end-text') void controller.end(); - else if (action === 'research') void controller.research(this.attestation); + else if (action === 'research') { this.showView('research'); void controller.research(this.attestation); } else if (action === 'credentials') { void controller.showCredentials(); this.button('close-credentials').focus(); } else if (action === 'close-credentials') { this.clearKeyInputs(); controller.closeCredentials(); this.button('credentials').focus(); } else if (action === 'refresh-research-models') void controller.loadResearchModels(true); @@ -159,15 +180,16 @@ export function registerPanel(controller: LiveController): void { else if (action === 'history') void controller.showHistory(); else if (action === 'review-latest') { const tool = [...controller.tools.values()].slice(-12).reverse().find(tool => !evidenceEligibility(tool)); - if (tool) void controller.evidence.openTool(tool.id).then(() => this.evidenceCards.get(tool.id)?.article.scrollIntoView({ block: 'nearest' })); + if (tool) this.openReview(tool.id); } + else if (action === 'open-review' && button.dataset.id) this.openReview(button.dataset.id); else if (action === 'toggle-mention') { this.mentionOpen = !this.mentionOpen; this.render(); } else if (action === 'attach' && button.dataset.id) { controller.selected.add(button.dataset.id); this.mentionOpen = false; this.render(); } else if (action === 'remove' && button.dataset.id) { controller.selected.delete(button.dataset.id); this.render(); } else if (action === 'approve') void controller.decide(button.dataset.id!, true); else if (action === 'decline') void controller.decide(button.dataset.id!, false); else if (action === 'cancel') void controller.cancel(button.dataset.id!); - else if (action === 'read-history') void controller.readHistory(button.dataset.id!); + else if (action === 'read-history') void controller.readHistory(button.dataset.id!).then(() => this.showView([...controller.tools.values()].some(tool => tool.name === 'research_run') ? 'research' : 'chat')); else if (action === 'clear-history') { // Clearing durable history is explicit and separate from ending a call. if (window.confirm('Clear this saved conversation and its tool history?')) void controller.clearHistory(button.dataset.id!); @@ -179,8 +201,8 @@ export function registerPanel(controller: LiveController): void { this.node('provider').addEventListener('change', event => void controller.selectProvider((event.target as HTMLSelectElement).value as ProviderId)); this.querySelector('#radsysx-live-attestation')!.addEventListener('change', event => { this.attestation = (event.target as HTMLSelectElement).value as Attestation || undefined; }); this.querySelector('textarea')!.addEventListener('input', event => { controller.draft = (event.target as HTMLTextAreaElement).value; if (/(^|\s)@$/.test(controller.draft)) { this.mentionOpen = true; this.render(); } }); - this.querySelector('textarea')!.addEventListener('keydown', event => { if (event.key === 'Enter' && !event.shiftKey && !event.isComposing) { event.preventDefault(); void controller.sendText(this.attestation); } }); - this.querySelector('.radsysx-ai-composer')!.addEventListener('submit', event => { event.preventDefault(); void controller.sendText(this.attestation); }); + this.querySelector('textarea')!.addEventListener('keydown', event => { if (event.key === 'Enter' && !event.shiftKey && !event.isComposing) { event.preventDefault(); this.submit(); } }); + this.querySelector('.radsysx-ai-composer')!.addEventListener('submit', event => { event.preventDefault(); this.submit(); }); this.querySelectorAll('form[data-credential-provider]').forEach(form => { form.addEventListener('submit', event => { event.preventDefault(); @@ -205,7 +227,15 @@ export function registerPanel(controller: LiveController): void { void controller.evidence.selectSession(controller.evidenceSessionId); this.unsubscribe = controller.subscribe(() => this.render()); } - disconnectedCallback(): void { controller.subscription.stop(); controller.evidence.dispose(); this.evidenceCards.forEach(card => card.review?.dispose()); this.evidenceCards.clear(); this.node('tools').replaceChildren(); this.clearKeyInputs(); this.unsubscribe?.(); this.unsubscribe = undefined; } + disconnectedCallback(): void { controller.subscription.stop(); controller.evidence.dispose(); this.evidenceCards.forEach(card => { card.review?.dispose(); card.article.remove(); card.reviewHost?.remove(); }); this.evidenceCards.clear(); this.clearKeyInputs(); this.unsubscribe?.(); this.unsubscribe = undefined; } + private showView(view: 'chat' | 'research' | 'review'): void { + controller.sidebarView = view; this.render(); this.node('conversation').scrollTop = 0; + } + private openReview(id: string): void { this.showView('review'); void controller.evidence.openTool(id); } + private submit(): void { + if (controller.sidebarView === 'research') void controller.research(this.attestation); + else void controller.sendText(this.attestation); + } private clearKeyInputs(): void { this.querySelectorAll('input[data-key-provider]').forEach(input => { input.value = ''; }); } private renderCredentials(): void { if (this.credentialInputEpoch !== controller.credentialInputEpoch) { this.credentialInputEpoch = controller.credentialInputEpoch; this.clearKeyInputs(); } @@ -277,17 +307,19 @@ export function registerPanel(controller: LiveController): void { providerSelect.value = controller.providerId; providerSelect.disabled = controller.credentialsBusy || controller.status === 'loading' || !controller.providers.length; providerSelect.title = controller.model; - this.node('disclosure').textContent = `Text · ${controller.session?.mode === 'text' && controller.status === 'text_ready' ? controller.session.modelId : controller.researchSettings?.modelId ?? 'model in Settings'}. Sends your question and neutral case/series metadata; no image pixels. Optional voice · ${controller.provider?.label ?? 'choose a provider'}.`; + this.node('text-model').textContent = `Text & research · ${controller.session?.mode === 'text' && controller.status === 'text_ready' ? controller.session.modelId : controller.researchSettings?.modelId ?? 'Choose a model in Settings'}`; + this.node('disclosure').textContent = 'Sends your question and neutral case/series context. No image pixels.'; this.button('end-text').hidden = controller.session?.mode !== 'text' || controller.status !== 'text_ready'; this.button('research').disabled = controller.textBusy || controller.credentialsBusy || controller.status === 'loading'; this.querySelector('[aria-label="Send message"]')!.disabled = controller.textBusy || controller.credentialsBusy || controller.status === 'loading'; this.dataset.connection = controller.status; const active = ['connecting', 'ready', 'reconnecting'].includes(controller.status); + this.node('voice-state').textContent = active ? controller.ready ? 'Connected' : 'Connecting' : 'Optional'; this.node('setup').hidden = active; this.node('session-controls').hidden = !active; this.button('connect').disabled = controller.credentialsBusy || controller.status === 'loading'; this.button('connect').textContent = controller.providers.length ? 'Connect voice' : 'Retry setup'; - this.node('status').textContent = controller.message; + this.node('status').textContent = controller.message === 'Viewing saved conversation. Audio and image frames are not recorded.' ? 'Saved conversation' : controller.message; this.node('status').hidden = !controller.message || controller.message === 'Confirm the displayed data to begin.'; this.node('voice-label').textContent = controller.audio.listening ? 'Mic on' : 'Mic off'; this.node('interaction').textContent = controller.ready && controller.interaction === 'IN_PROGRESS' ? 'Thinking and working…' : controller.ready ? 'Connected · interrupt anytime' : active ? 'Connecting…' : ''; @@ -309,7 +341,7 @@ export function registerPanel(controller: LiveController): void { const signature = JSON.stringify(transcripts); if (signature !== this.threadSignature) { const thread = this.node('thread'); const scrollContainer = this.node('conversation'); const scroll = scrollContainer.scrollHeight - scrollContainer.scrollTop - scrollContainer.clientHeight < 80; - thread.innerHTML = transcripts.length ? transcripts.map(item => `
${item.role === 'user' ? 'You' : 'RadSysX AI'}
${escape(item.text)}
`).join('') : '
Start a conversation to see its transcript.
'; + thread.innerHTML = transcripts.length ? transcripts.map(item => `
${item.role === 'user' ? 'You' : 'RadSysX AI'}
${item.role === 'user' ? escape(item.text) : answerMarkup(item.text)}
`).join('') : '
Discuss this case.Ask a question or describe a finding. Voice is optional; images are not sent with text.
'; if (scroll) scrollContainer.scrollTop = scrollContainer.scrollHeight; this.threadSignature = signature; } @@ -325,31 +357,53 @@ export function registerPanel(controller: LiveController): void { const eligible = [...visibleTools].reverse().find(tool => !evidenceEligibility(tool)); const researching = visibleTools.some(tool => tool.name === 'research_run' && ['pending', 'running'].includes(tool.status)); this.button('review-latest').disabled = !eligible || controller.evidence.busy || controller.credentialsBusy; - this.node('evidence-next').textContent = eligible ? 'Preview the selected claims and original abstracts before anything is sent to Jev.' : researching ? 'Research is in progress below. Jev review becomes available when cited abstracts arrive.' : 'Enter a public literature question and choose Research. Then review its cited abstracts here.'; + this.node('evidence-next').textContent = eligible ? 'Preview the exact passages and their cited abstracts before sending to Jev.' : researching ? 'Research is in progress. Review becomes available when cited abstracts arrive.' : 'Start with a literature question in Research, or open saved research.'; const visibleIds = new Set(visibleTools.map(tool => tool.id)); - for (const [id, card] of this.evidenceCards) if (!visibleIds.has(id)) { card.review?.dispose(); card.article.remove(); this.evidenceCards.delete(id); } + for (const [id, card] of this.evidenceCards) if (!visibleIds.has(id)) { card.review?.dispose(); card.reviewHost?.remove(); card.article.remove(); this.evidenceCards.delete(id); } for (const tool of visibleTools) { let card = this.evidenceCards.get(tool.id); if (!card) { const article = document.createElement('article'); article.className = 'radsysx-live-tool'; - const body = document.createElement('div'); body.className = 'radsysx-live-tool-body'; article.append(body); this.node(['research_run', 'text_chat'].includes(tool.name) ? 'research-tools' : 'tools').append(article); + const body = document.createElement('div'); body.className = 'radsysx-live-tool-body'; article.append(body); this.node(tool.name === 'research_run' ? 'research-tools' : 'tools').append(article); card = { article, body, signature: '' }; this.evidenceCards.set(tool.id, card); } - const signature = JSON.stringify([tool, controller.historical]); + const reviewSummary = controller.evidence.reviewForTool(tool.id); + const signature = JSON.stringify([tool, controller.historical, reviewSummary]); if (signature !== card.signature) { card.signature = signature; const pending = !controller.historical && !['completed', 'failed', 'cancelled', 'declined', 'rejected', 'denied', 'interrupted', 'outcome_unknown'].includes(tool.status); - card.body.innerHTML = `
${tool.name === 'research_run' ? 'Literature research' : escape(tool.name.replace(/_/g, ' '))}${escape(tool.status)}
${['research_run', 'text_chat'].includes(tool.name) ? renderResearchActivity(tool) : `${tool.approval ? 'Review this action' : 'Details'}
${escape(JSON.stringify(tool.args, null, 2))}
`}${tool.name === 'text_chat' && tool.status === 'completed' ? '' : renderToolResult(tool.result)}${tool.approval ? `
` : pending ? `` : ''}`; + card.body.innerHTML = `

${tool.name === 'research_run' ? escape(tool.args.query || 'Literature research') : escape(tool.name.replace(/_/g, ' '))}

${['research_run', 'text_chat'].includes(tool.name) ? renderResearchActivity(tool) : `${tool.approval ? 'Review this action' : 'Details'}
${escape(JSON.stringify(tool.args, null, 2))}
`}${tool.name === 'text_chat' && tool.status === 'completed' ? '' : renderToolResult(tool.result)}${tool.approval ? `
` : pending ? `` : ''}${!evidenceEligibility(tool) ? `` : ''}`; if (tool.name === 'research_run' && !card.review) { const reason = evidenceEligibility(tool); if (!reason) { - const host = document.createElement('section'); host.dataset.evidenceTool = tool.id; card.article.append(host); - card.review = mountEvidencePanel(host, controller.evidence); + const host = document.createElement('section'); host.dataset.evidenceTool = tool.id; this.node('review-panels').append(host); card.reviewHost = host; + card.review = mountEvidencePanel(host, controller.evidence, () => this.showView('research')); } else { const note = document.createElement('p'); note.textContent = reason; card.body.append(note); } } + const reviewLink = card.body.querySelector('[data-action="open-review"]'); + if (reviewLink && reviewSummary) reviewLink.textContent = reviewSummary.status === 'completed' + ? `View Jev result · ${reviewSummary.completedPairs} checked →` + : reviewSummary.status === 'reviewing' ? 'Jev is reviewing · View progress →' : 'Open Jev review →'; } card.review?.update(); + card.article.hidden = tool.name === 'text_chat' && tool.status === 'completed'; + if (card.reviewHost) card.reviewHost.hidden = !controller.evidence.open || controller.evidence.detail?.toolCallId !== tool.id; + } + const view = controller.sidebarView; + for (const name of ['chat', 'research', 'review'] as const) { + this.node(`${name}-view`).hidden = view !== name; + this.querySelector(`nav [data-action="view-${name}"]`)!.setAttribute('aria-pressed', String(view === name)); } + this.node('composer').hidden = view === 'review'; + this.node('voice-options').hidden = view !== 'chat' && !active; + this.button('research').hidden = view !== 'research'; + this.querySelector('[aria-label="Send message"]')!.hidden = view === 'research'; + this.querySelector('.radsysx-live-hint')!.textContent = view === 'research' ? 'Enter to research' : 'Enter to send'; + textarea.placeholder = view === 'research' ? 'Enter a public literature question…' : 'Ask about this case or describe a finding…'; + this.node('research-empty').hidden = visibleTools.some(tool => tool.name === 'research_run'); + this.node('research-count').textContent = allTools.filter(tool => tool.name === 'research_run').length ? String(allTools.filter(tool => tool.name === 'research_run').length) : ''; + this.node('evidence-entry').hidden = Boolean(controller.evidence.open && controller.evidence.detail); + this.node('review-message').textContent = controller.evidence.open && !controller.evidence.detail ? controller.evidence.busy ? 'Preparing the review…' : controller.evidence.message : ''; this.node('sources').innerHTML = controller.citations.length ? 'Sources' + controller.citations.map(source => `${escape(source.title)} ↗`).join('') : ''; if (this.suggestionSignature !== controller.suggestionsHtml) { this.suggestionSignature = controller.suggestionsHtml; this.node('suggestions').replaceChildren(); diff --git a/viewer/assets/live/presentation.ts b/viewer/assets/live/presentation.ts new file mode 100644 index 0000000..b093ba3 --- /dev/null +++ b/viewer/assets/live/presentation.ts @@ -0,0 +1,18 @@ +import { escape } from './protocol.js'; + +/** A deliberately small, escaped text formatter: no HTML, images or generated links. */ +export function inlineMarkup(value: string): string { + return escape(value) + .replace(/\*\*([^*\n]+)\*\*/g, '$1') + .replace(/`([^`\n]+)`/g, '$1'); +} + +export function answerMarkup(text: string): string { + return text.split(/\n\s*\n/).filter(value => value.trim()).map(block => { + const lines = block.split('\n'); + if (lines.every(line => /^\s*(?:[-*]|\d+\.)\s+/.test(line))) { + return `
    ${lines.map(line => `
  • ${inlineMarkup(line.replace(/^\s*(?:[-*]|\d+\.)\s+/, ''))}
  • `).join('')}
`; + } + return `

${lines.map(line => inlineMarkup(line.replace(/^#{1,6}\s+/, ''))).join('
')}

`; + }).join(''); +} diff --git a/viewer/assets/radsysx-viewer.css b/viewer/assets/radsysx-viewer.css index 1f18c39..7736788 100644 --- a/viewer/assets/radsysx-viewer.css +++ b/viewer/assets/radsysx-viewer.css @@ -963,7 +963,7 @@ body.radsysx-local-viewer [data-radsysx-local-upload-modal="true"] button:focus- .radsysx-live-credentials button { border: 1px solid #3a5264; border-radius: 5px; padding: 7px 9px; background: #1e3242; color: #c9d9e5; font-size: 11px; cursor: pointer; } .radsysx-live-credentials button[type="submit"] { background: #314e60; color: #e0e8ef; border-color: #314e60; } -/* Saved evidence stays inside the scrollable research card, below its unchanged answer. */ +/* Saved evidence has a dedicated scrollable review workspace. */ .radsysx-evidence { color: #bdcdd8; margin-top: 0.6rem; border-top: 1px solid #3a4d5c; padding-top: 0.6rem; font-size: 0.72rem; line-height: 1.5; overflow-wrap: anywhere; min-width: 0; } .radsysx-evidence [data-evidence-status] { display: grid; gap: 0.15rem; margin-bottom: 0.4rem; } .radsysx-evidence [data-evidence-detail] { margin-top: 0.5rem; } @@ -974,7 +974,7 @@ body.radsysx-local-viewer [data-radsysx-local-upload-modal="true"] button:focus- .radsysx-evidence select { width: 100%; min-width: 0; background: #111e29; color: #c9d9e5; border: 1px solid #496071; border-radius: 0.35rem; padding: 0.45rem; font-size: 0.7rem; } .radsysx-evidence-text { white-space: pre-wrap; overflow-wrap: anywhere; } .radsysx-evidence .radsysx-evidence-text { color: #c5d2dd; font-size: 0.72rem; } -.radsysx-evidence-judgment { border-left: 2px solid #5c7c91; padding-left: 0.5rem; margin: 0.8rem 0; } +.radsysx-evidence-judgment { border-top: 1px solid #30424f; padding-top: 0.8rem; margin: 0.8rem 0; } .radsysx-evidence a { color: #9dbbd0; text-decoration: underline; } .radsysx-evidence dt { margin-top: 0.5rem; color: #96adbd; } .radsysx-evidence dd { margin: 0; font-size: 0.65rem; overflow-wrap: anywhere; } @@ -999,3 +999,58 @@ body.radsysx-local-viewer [data-radsysx-local-upload-modal="true"] button:focus- .radsysx-live-shell .radsysx-ai-mention-option span { color: #c3d3df; } .radsysx-live-shell button:hover:not(:disabled) { border-color: #607e92; } .radsysx-live-primary:hover:not(:disabled), .radsysx-live-shell .radsysx-ai-send-button:hover:not(:disabled) { background: #3b5b6e; } + +/* Distinct workspaces keep conversation, literature, and review from competing. */ +.radsysx-live-shell { gap: 0.5rem; background: #101820; } +.radsysx-live-header { padding: 0.85rem 0.85rem 0; } +.radsysx-sidebar-model { margin: 0 0.85rem 0.25rem; color: #99adbd; font-size: 0.7rem; overflow-wrap: anywhere; } +.radsysx-sidebar-tabs { display: flex; gap: 0.2rem; margin: 0 0.75rem; padding: 0.2rem; background: #0b1219; border: 1px solid #293b48; border-radius: 0.5rem; } +.radsysx-sidebar-tabs button { flex: 1; min-width: 0; border: 1px solid transparent; border-radius: 0.35rem; background: transparent; color: #96aaba; padding: 0.55rem 0.25rem; font-size: 0.75rem; cursor: pointer; } +.radsysx-sidebar-tabs button[aria-pressed="true"] { background: #263744; color: #d6e0e7; border-color: #405665; } +.radsysx-sidebar-tabs span { color: #a7bdcc; font-size: 0.65rem; } +.radsysx-voice-options { border: 0; background: transparent; padding: 0.2rem 0; } +.radsysx-voice-options > summary { display: flex; justify-content: space-between; cursor: pointer; color: #a0b4c3; font-size: 0.72rem; padding: 0.3rem 0; } +.radsysx-voice-options > summary::before { content: '›'; margin-right: 0.4rem; } +.radsysx-voice-options[open] > summary::before { content: '⌄'; } +.radsysx-voice-options > summary span { margin-left: auto; color: #8298a9; } +.radsysx-voice-options .radsysx-live-controls { margin: 0.5rem 0 0; flex-wrap: wrap; } +.radsysx-section-heading { margin: 0.65rem 0.85rem 1rem; } +.radsysx-section-heading h3 { margin: 0; font-size: 0.95rem; font-weight: 550; color: #d0dce5; } +.radsysx-section-heading p { font-size: 0.72rem; color: #8ea4b5; margin: 0.25rem 0 0; } +.radsysx-section-heading button { padding: 0.4rem 0; border: 0; background: transparent; color: #a5bfd2; font-size: 0.72rem; text-decoration: underline; cursor: pointer; } +.radsysx-live-tool { padding: 0.9rem; margin-bottom: 0.8rem; border-color: #2c3e4c; background: #151f29; } +.radsysx-live-tool h4 { font-size: 0.8rem; line-height: 1.5; font-weight: 550; margin: 0 0 0.5rem; color: #d0dce5; overflow-wrap: anywhere; } +.radsysx-live-tool-body { display: flex; flex-direction: column; } +.radsysx-live-tool-body > .radsysx-execution { order: 3; } +.radsysx-live-tool .radsysx-research-model { color: #8ea4b5; font-size: 0.68rem; margin: 0 0 0.3rem; } +.radsysx-live-tool [role="status"] { font-size: 0.7rem; margin: 0 0 0.5rem; color: #a9beca; } +.radsysx-live-tool summary { font-size: 0.72rem; } +.radsysx-live-tool .radsysx-review-link { align-self: flex-start; margin: 0.65rem 0; } +.radsysx-live-research-result { font-size: 0.8rem; line-height: 1.65; } +.radsysx-live-research-result p { white-space: normal; margin: 0.7rem 0; } +.radsysx-live-research-result ul { padding-left: 1.1rem; } +.radsysx-live-research-result a { font-size: 0.73rem; margin: 0.5rem 0; } +.radsysx-live-shell .radsysx-ai-message[data-role="assistant"] .radsysx-ai-message-body { white-space: normal; } +.radsysx-ai-message-body p { margin: 0.6rem 0; } +.radsysx-live-shell .radsysx-ai-composer { margin-top: 0.2rem; padding: 0.6rem; } +.radsysx-data-confirmation { display: flex; align-items: center; gap: 0.5rem; font-size: 0.66rem; color: #94aabb; margin-bottom: 0.5rem; } +.radsysx-data-confirmation select { flex: 1; min-width: 0; border: 1px solid #344b5b; background: #111d27; color: #bdcdd8; border-radius: 0.3rem; padding: 0.4rem; font-size: 0.68rem; } +.radsysx-live-shell [data-role="disclosure"] { margin: 0.45rem 0 0; font-size: 0.64rem; line-height: 1.5; color: #8b9fac; } +.radsysx-live-shell [data-action="end-text"] { border: 0; background: transparent; font-size: 0.65rem; padding: 0.25rem 0; } +.radsysx-live-shell [data-role="review-panels"] { padding: 0 0.85rem 1rem; } +.radsysx-evidence { border: 0; margin: 0; padding: 0; font-size: 0.76rem; line-height: 1.6; } +.radsysx-evidence [data-evidence-status] { padding: 0.7rem 0; border-bottom: 1px solid #2c3e4c; margin-bottom: 0.85rem; } +.radsysx-evidence [data-evidence-status] strong { font-size: 0.9rem; color: #d1dce5; font-weight: 550; } +.radsysx-evidence [data-evidence-status] span, .radsysx-evidence-scope { font-size: 0.7rem; color: #96adbd; } +.radsysx-evidence details { margin: 0.7rem 0; } +.radsysx-evidence summary { padding: 0.3rem 0; color: #a9c0d1; font-size: 0.74rem; } +.radsysx-evidence fieldset { border: 0; padding: 0; } +.radsysx-evidence legend { position: absolute; width: 1px; height: 1px; overflow: hidden; clip-path: inset(50%); } +.radsysx-evidence fieldset label { padding: 0.7rem; background: #182630; border: 1px solid #344b5b; border-radius: 0.4rem; margin: 0.6rem 0; } +.radsysx-evidence [data-evidence-confirm] { margin: 1rem 0; } +.radsysx-evidence [data-evidence-confirm] p { color: #9aafbd; font-size: 0.72rem; } +.radsysx-evidence [data-evidence-guidance]:empty, .radsysx-evidence [data-evidence-message]:empty { display: none; } +.radsysx-evidence button { border: 1px solid #3a5161; background: #1c2e3b; color: #c0d1dd; border-radius: 0.4rem; padding: 0.5rem 0.7rem; font-size: 0.73rem; cursor: pointer; } +.radsysx-evidence [data-evidence-action="refresh"], .radsysx-evidence [data-evidence-action="close"] { background: transparent; border-color: transparent; padding-left: 0.25rem; padding-right: 0.25rem; } +.radsysx-evidence-verdict { display: block; margin-bottom: 0.5rem; font-size: 0.86rem; font-weight: 550; color: #c9dbe5; } +.radsysx-evidence-judgment > .radsysx-evidence-text { white-space: normal; } diff --git a/viewer/scripts/AGENTS.md b/viewer/scripts/AGENTS.md index be91037..9caecf8 100644 --- a/viewer/scripts/AGENTS.md +++ b/viewer/scripts/AGENTS.md @@ -44,6 +44,7 @@ - `test:live` compiles once and runs both `test-live.mjs` and `test-evidence.mjs`. Review tests use synthetic wire fixtures and mock timers to prove confirmation/selection, stable uncertain idempotency, old-response rejection, bounded polling, no inference on reopen/refresh and escaped abstract-scoped presentation. The guarded Electron smoke owns actual DOM/focus/layout acceptance. - Research presentation checks distinguish model waiting, PubMed search and terminal timeout/cancellation; recorded model identity must remain unknown when absent. Ending-session tests require backend terminal receipts or explicit unconfirmed status, never a stale running card. +- Presentation tests cover escaped formatting, selected-source previews, collapsed technical details and zero-pair review explanations. Native desktop checks must also inspect all three workspace views and the real preparation/confirmation/completion path; markup tests alone do not establish usability. - Text-controller regressions prove Send/Research without voice or audio allocation, preserved draft/operation identity on uncertain submission, and stale polling rejection after End. Keep these separate from actual hosted-model acceptance. diff --git a/viewer/scripts/test-evidence.mjs b/viewer/scripts/test-evidence.mjs index e47b58e..782ecfe 100644 --- a/viewer/scripts/test-evidence.mjs +++ b/viewer/scripts/test-evidence.mjs @@ -31,6 +31,22 @@ function reviewFixture(overrides = {}) { }; } const response = value => new Response(JSON.stringify(value)); + +test('empty previews explain that Jev has not run, and show no misleading ready count', () => { + const html = renderEvidenceSummary(reviewFixture({status:'ready',totalPairs:0,completedPairs:0})); + assert.match(html,/No reviewable passages/); assert.match(html,/Nothing has been sent/); + assert.doesNotMatch(html,/Ready|0 of 0/); + const ready = renderEvidenceSummary(reviewFixture({status:'ready',completedPairs:0})); + assert.match(ready,/Ready for your confirmation/); assert.match(ready,/Jev has not run yet/); +}); + +test('review preview includes only selected source abstracts and keeps details collapsed', () => { + const detail = reviewFixture(); + detail.abstracts.push({...detail.abstracts[0],evidenceId:'unrelated',title:'UNRELATED_ABSTRACT'}); + const html = renderEvidenceDetail(detail); + assert.doesNotMatch(html,/UNRELATED_ABSTRACT|]* open/); + assert.match(html,/Source abstracts · 1/); assert.match(html,/Source provenance/); +}); const flush = async () => { for (let i=0;i<20;i++) await Promise.resolve(); }; function harness(detail=reviewFixture({status:'ready',selectedUnitIds:null,assessments:[],attempts:[]})) { const calls=[]; let current=detail; diff --git a/viewer/scripts/test-live.mjs b/viewer/scripts/test-live.mjs index 1ccccb8..dccf0d8 100644 --- a/viewer/scripts/test-live.mjs +++ b/viewer/scripts/test-live.mjs @@ -7,6 +7,13 @@ import { OHIFAdapter } from '../.cache/live-runtime/ohif.js'; import { LiveAudio } from '../.cache/live-runtime/audio.js'; import { LiveController } from '../.cache/live-runtime/controller.js'; import { credentialSettingsMarkup, renderToolResult, renderResearchActivity, researchStatus } from '../.cache/live-runtime/panel.js'; +import { answerMarkup } from '../.cache/live-runtime/presentation.js'; + +test('answer presentation formats prose without enabling HTML, images or generated links', () => { + const html = answerMarkup('**Finding:** A result [s1].\n\n- First\n- Second\n\n\n[jump](javascript:alert(1))'); + assert.match(html, /Finding:<\/strong>/); assert.match(html, /
  • First/); + assert.match(html, /<img/); assert.doesNotMatch(html, /