From 77e6343d6afcd7ebff03f476aef362c6f585e369 Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Fri, 2 Oct 2026 01:34:01 +0200 Subject: [PATCH 01/19] doc-links: doc-comment references become MENTIONS edges, C# first A reference written in a doc comment is extracted with the definition it documents (internal/cbm/doclink.c; the C# scanner doclink_cs.c) and resolved per file in the pipeline (src/pipeline/doc_links.c, doc_links_cs.c, doc_links_msbuild.c). A reference that names a definition of the repository with certainty becomes a MENTIONS edge from the documented definition (the File node for a file-level comment) with the properties via, syntax, tier, line and count. Any other reference becomes a row of the new doc_link_unresolved table with its reason: missing, ambiguous, external, test_only_target, graph_gap, unparseable or below_bar_tier. Nothing is guessed. index_status reports the layer in a doc_links block. C# binds a cref by the compiler's lookup rule: the first scope with the name decides, a cref never binds an inherited member, System is outside the repository. Assemblies come from project files, and MSBuild global usings (, ImplicitUsings, Directory.Build.props/.targets, imports, conditions) are evaluated from the project files' content as data the indexer has already read; the layer opens no file. Two declarations of a name in different assemblies make the reference ambiguous. Each file's scope is stored line-free with its LSP surface ("dl"), so an incremental run re-resolves the files a change can reach and falls back to a full run for a change that can re-route references anywhere (a project file, a global using, a deleted file). The scanner and the resolver treat the repository as untrusted input: bounded nesting, no input-driven recursion, work that grows linearly (held by test seams, never by a clock), no silent limits, and a damaged or foreign stored scope fails closed. Signed-off-by: Martin Vogel --- Makefile.cbm | 7 +- README.md | 4 +- docs/EVALUATION_PLAN.md | 9 +- internal/cbm/cbm.c | 7 +- internal/cbm/cbm.h | 40 + internal/cbm/doclink.c | 442 ++ internal/cbm/doclink.h | 215 + internal/cbm/doclink_cs.c | 3653 +++++++++++ internal/cbm/extract_defs.c | 86 +- internal/cbm/result_compact.c | 7 + src/cli/cli.c | 4 +- src/mcp/mcp.c | 321 +- src/mcp/mcp_internal.h | 5 + src/pipeline/doc_links.c | 789 +++ src/pipeline/doc_links.h | 231 + src/pipeline/doc_links_cs.c | 5995 +++++++++++++++++ src/pipeline/doc_links_msbuild.c | 3863 +++++++++++ src/pipeline/doc_links_msbuild.h | 146 + src/pipeline/lsp_surface.c | 22 +- src/pipeline/pass_parallel.c | 19 +- src/pipeline/pipeline.c | 96 +- src/pipeline/pipeline_incremental.c | 400 +- src/pipeline/pipeline_internal.h | 39 +- src/store/store.c | 530 ++ src/store/store.h | 93 + tests/test_doc_mentions.c | 9206 +++++++++++++++++++++++++++ tests/test_doc_mentions_helpers.h | 414 ++ tests/test_edge_structural.c | 1 + tests/test_lang_contract.c | 1 + tests/test_main.c | 4 + 30 files changed, 26608 insertions(+), 41 deletions(-) create mode 100644 internal/cbm/doclink.c create mode 100644 internal/cbm/doclink.h create mode 100644 internal/cbm/doclink_cs.c create mode 100644 src/pipeline/doc_links.c create mode 100644 src/pipeline/doc_links.h create mode 100644 src/pipeline/doc_links_cs.c create mode 100644 src/pipeline/doc_links_msbuild.c create mode 100644 src/pipeline/doc_links_msbuild.h create mode 100644 tests/test_doc_mentions.c create mode 100644 tests/test_doc_mentions_helpers.h diff --git a/Makefile.cbm b/Makefile.cbm index d76269f666..3c34559142 100644 --- a/Makefile.cbm +++ b/Makefile.cbm @@ -350,6 +350,8 @@ EXTRACTION_SRCS = \ $(CBM_DIR)/extract_channels.c \ $(CBM_DIR)/extract_k8s.c \ $(CBM_DIR)/extract_dbt.c \ + $(CBM_DIR)/doclink.c \ + $(CBM_DIR)/doclink_cs.c \ $(CBM_DIR)/sql_values.c \ $(CBM_DIR)/helpers.c \ $(CBM_DIR)/result_compact.c \ @@ -454,6 +456,9 @@ PIPELINE_SRCS = \ src/pipeline/pass_configures.c \ src/pipeline/pass_configlink.c \ src/pipeline/pass_doclinks.c \ + src/pipeline/doc_links.c \ + src/pipeline/doc_links_cs.c \ + src/pipeline/doc_links_msbuild.c \ src/pipeline/pass_route_nodes.c \ src/pipeline/pass_enrichment.c \ src/pipeline/pass_envscan.c \ @@ -681,7 +686,7 @@ TEST_DISCOVER_SRCS = \ TEST_GRAPH_BUFFER_SRCS = tests/test_graph_buffer.c -TEST_PIPELINE_SRCS = tests/test_registry.c tests/test_pipeline.c tests/test_importance.c tests/test_cross_repo.c tests/test_fqn.c tests/test_route_canon.c tests/test_path_alias.c tests/test_configlink.c tests/test_doclinks.c tests/test_infrascan.c tests/test_worker_pool.c tests/test_parallel.c tests/test_index_resilience.c tests/test_index_format.c tests/test_call_reference_contract.c tests/repro/repro_call_scope_usages.c tests/repro/repro_call_argument_usages.c tests/repro/repro_reference_precision.c tests/repro/repro_lexical_binding_precision.c tests/repro/repro_call_argument_matrix_a.c tests/repro/repro_call_argument_matrix_b.c tests/repro/repro_call_node_behaviors.c tests/repro/repro_language_registry.c tests/repro/repro_call_node_manifest.c tests/repro/repro_lsp_ordered_signatures.c tests/repro/repro_lsp_ordered_local.c tests/repro/repro_ts_overload_return_chains.c tests/repro/repro_harness_cleanup.c tests/repro/repro_runner_filter.c +TEST_PIPELINE_SRCS = tests/test_registry.c tests/test_pipeline.c tests/test_importance.c tests/test_cross_repo.c tests/test_fqn.c tests/test_route_canon.c tests/test_path_alias.c tests/test_configlink.c tests/test_doclinks.c tests/test_doc_mentions.c tests/test_infrascan.c tests/test_worker_pool.c tests/test_parallel.c tests/test_index_resilience.c tests/test_index_format.c tests/test_call_reference_contract.c tests/repro/repro_call_scope_usages.c tests/repro/repro_call_argument_usages.c tests/repro/repro_reference_precision.c tests/repro/repro_lexical_binding_precision.c tests/repro/repro_call_argument_matrix_a.c tests/repro/repro_call_argument_matrix_b.c tests/repro/repro_call_node_behaviors.c tests/repro/repro_language_registry.c tests/repro/repro_call_node_manifest.c tests/repro/repro_lsp_ordered_signatures.c tests/repro/repro_lsp_ordered_local.c tests/repro/repro_ts_overload_return_chains.c tests/repro/repro_harness_cleanup.c tests/repro/repro_runner_filter.c TEST_WATCHER_SRCS = tests/test_watcher.c diff --git a/README.md b/README.md index f644775a73..0de469ac42 100644 --- a/README.md +++ b/README.md @@ -740,7 +740,9 @@ mode. `index_status` keeps describing the published graph and its freshness. ### Edge Types -`CONTAINS_PACKAGE`, `CONTAINS_FOLDER`, `CONTAINS_FILE`, `DEFINES`, `DEFINES_METHOD`, `IMPORTS`, `CALLS`, `CALL_REFERENCE`, `HTTP_CALLS`, `ASYNC_CALLS`, `IMPLEMENTS`, `HANDLES`, `USAGE`, `CONFIGURES`, `REFERENCES_FILE`, `WRITES`, `MEMBER_OF`, `TESTS`, `USES_TYPE`, `FILE_CHANGES_WITH` +`CONTAINS_PACKAGE`, `CONTAINS_FOLDER`, `CONTAINS_FILE`, `DEFINES`, `DEFINES_METHOD`, `IMPORTS`, `CALLS`, `CALL_REFERENCE`, `HTTP_CALLS`, `ASYNC_CALLS`, `IMPLEMENTS`, `HANDLES`, `USAGE`, `CONFIGURES`, `REFERENCES_FILE`, `MENTIONS`, `WRITES`, `MEMBER_OF`, `TESTS`, `USES_TYPE`, `FILE_CHANGES_WITH` + +`MENTIONS` links a documented definition to the code its doc comment references (C# ``, ``, ``, ``), one edge per pair with `via`, `syntax`, `tier` (`exact` or `unique`), first `line` and `count`. A reference is bound only when it names exactly one definition under the language's own lookup rules; the rest are counted by reason in `index_status` under `doc_links` (samples with `diagnostics='full'`). ### Qualified Names diff --git a/docs/EVALUATION_PLAN.md b/docs/EVALUATION_PLAN.md index 5f8bb532cc..efd56207db 100644 --- a/docs/EVALUATION_PLAN.md +++ b/docs/EVALUATION_PLAN.md @@ -311,15 +311,16 @@ SHA, and — **not just totals** — a **per-type breakdown**: - `node-types.json` — a histogram of **node count by label** (`Function`, `Method`, `Class`, `Interface`, `Type`/`Enum`/`Struct`, `Field`, `Variable`, `Route`, `Module`, `Section`, `Macro`, `File`, `Folder`) + total. -- `edge-types.json` — a histogram of **edge count by type, with one entry for every one of the 32 +- `edge-types.json` — a histogram of **edge count by type, with one entry for every one of the 33 edge types, including those that came back `0`**. The canonical set is the indexer's own - `ALL_EDGE_TYPES[]` (26 intra-repo types — `tests/test_lang_contract.c`): `CALLS`, `ASYNC_CALLS`, + `ALL_EDGE_TYPES[]` (27 intra-repo types — `tests/test_lang_contract.c`): `CALLS`, `ASYNC_CALLS`, `HTTP_CALLS`, `GRPC_CALLS`, `GRAPHQL_CALLS`, `TRPC_CALLS`, `DEFINES`, `DEFINES_METHOD`, `IMPLEMENTS`, `INHERITS`, `OVERRIDE`, `DECORATES`, `IMPORTS`, `HANDLES`, `CONFIGURES`, `DEPENDS_ON`, `USAGE`, `DATA_FLOWS`, `SEMANTICALLY_RELATED`, `SIMILAR_TO`, `TESTS`, `TESTS_FILE`, `INFRA_MAPS`, - `FILE_CHANGES_WITH`, `CONTAINS_FILE`, `CONTAINS_FOLDER` — **plus the 6 cross-repo types** from the + `FILE_CHANGES_WITH`, `CONTAINS_FILE`, `CONTAINS_FOLDER`, `MENTIONS` (doc comment -> referenced + code) — **plus the 6 cross-repo types** from the cross-repo pass (`CROSS_HTTP_CALLS`, `CROSS_ASYNC_CALLS`, `CROSS_GRPC_CALLS`, `CROSS_GRAPHQL_CALLS`, - `CROSS_TRPC_CALLS`, `CROSS_CHANNEL`) and a total. The writer **emits the full 32-type list and + `CROSS_TRPC_CALLS`, `CROSS_CHANNEL`) and a total. The writer **emits the full 33-type list and back-fills missing types with `0`** rather than recording only the types that appeared. These come straight from `query_graph` (`MATCH (n) RETURN labels(n), count(*)` and diff --git a/internal/cbm/cbm.c b/internal/cbm/cbm.c index a39a93ae43..b351e5f2ed 100644 --- a/internal/cbm/cbm.c +++ b/internal/cbm/cbm.c @@ -6,7 +6,8 @@ #include "foundation/mem_events.h" // waste sanitizer: the bound allocators bypass every observer #include "foundation/log.h" // cbm_log_warn -- extract.lsp.skipped #include "cbm.h" -#include "arena.h" // CBMArena, cbm_arena_init/alloc/strdup/destroy +#include "arena.h" // CBMArena, cbm_arena_init/alloc/strdup/destroy +#include "doclink.h" // cbm_doclinks_extract: doc-comment references + doc-link scope #include "helpers.h" #include "lang_specs.h" #include "extract_unified.h" @@ -3405,6 +3406,10 @@ static CBMFileResult *extract_file_ex_body(const char *source, int source_len, C result->imports_count = result->imports.count; + /* Doc-comment references of the documented definitions, and the file's + * doc-link scope: both read the complete docstrings and the tree. */ + cbm_doclinks_extract(&ctx); + // Accumulate profiling counters atomic_fetch_add(&total_parse_ns, t1 - t0); atomic_fetch_add(&total_extract_ns, t2 - t1); diff --git a/internal/cbm/cbm.h b/internal/cbm/cbm.h index 0058f0a511..4ed02f1ddd 100644 --- a/internal/cbm/cbm.h +++ b/internal/cbm/cbm.h @@ -574,6 +574,26 @@ typedef struct { int cap; } CBMFieldTypeArray; +/* One reference found in a definition's complete doc comment, or in the file's + * own doc (doclink.h has the syntax table and the per-language parsers). + * Resolution happens later, per file, in the pipeline + * (src/pipeline/doc_links.c). */ +typedef struct { + const char *source_qn; // QN of the documented definition (the edge source); + // for a file-level reference the file's module QN + const char *raw; // the reference as written, markup entities decoded + uint32_t line; // 1-based source line of the reference + uint32_t def_line; // 1-based start line of the documented definition + uint16_t syntax; // CBMDocLinkSyntax (doclink.h) + uint16_t flags; // CBM_DOCLINK_FLAG_* (doclink.h); set by the driver +} CBMDocLink; + +typedef struct { + CBMDocLink *items; + int count; + int cap; +} CBMDocLinkArray; + // Full extraction result for one file. typedef struct CBMFileResult { CBMArena arena; // owns local memory; composites may also retain child arenas below @@ -668,6 +688,14 @@ typedef struct CBMFileResult { * the Rust inner docs (//!). NULL for other languages and undocumented * files. */ const char *module_doc; + + /* Doc-comment references of this file's definitions (doclink.h), and the + * file's doc-link scope: a language-tagged text blob with what OTHER + * files' doc-link resolution needs from this file (C#: namespaces, + * usings, type and member declarations). A pure function of the file's + * bytes; NULL for languages without a scope scanner. */ + CBMDocLinkArray doc_links; + const char *doc_scope; } CBMFileResult; // --- Enclosing function cache --- @@ -759,6 +787,18 @@ typedef struct { * POD section index. */ void *doc_memo; void *doc_pod_index; + /* Start line of every doc text doc_run_text built, keyed by the returned + * pointer (doclink.c), so a definition's doc references get exact source + * lines. NULL until the first doc; allocated in `scratch`. */ + void *doc_lines; + /* A per-file slot for the language's doc-link hooks (parse_doc and + * scan_scope, internal/cbm/doclink_.c): state that has to survive + * between the hook calls of ONE file, e.g. an index of the file's doc + * sections. NULL at the start of every file. What a hook stores here must + * be allocated in `scratch` (or `arena`), so that it ends with the file. + * The core never reads, interprets or frees it. It is the only place for + * such state: a static or thread-local cache is not. */ + void *doclink_state; } CBMExtractCtx; // --- Public API --- diff --git a/internal/cbm/doclink.c b/internal/cbm/doclink.c new file mode 100644 index 0000000000..1b55ab792e --- /dev/null +++ b/internal/cbm/doclink.c @@ -0,0 +1,442 @@ +/* + * doclink.c — doc-comment references, extraction half: the language table, + * the doc-line side map and the per-file driver. See doclink.h. + */ +#include "doclink.h" + +#include "arena.h" +#include "foundation/constants.h" +#include "foundation/mem_core.h" + +#include +#include +#include + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +static _Atomic int doc_alloc_fail_after[CBM_DOCLINK_ALLOC_KINDS]; + +void cbm_doclink_test_fail_alloc_after(int kind, int nth) { + if (kind >= 0 && kind < CBM_DOCLINK_ALLOC_KINDS) { + atomic_store(&doc_alloc_fail_after[kind], nth); + } +} + +bool cbm_doclink_test_fail_alloc(int kind) { + if (kind < 0 || kind >= CBM_DOCLINK_ALLOC_KINDS) { + return false; + } + int n = atomic_load(&doc_alloc_fail_after[kind]); + while (n > 0) { + if (atomic_compare_exchange_weak(&doc_alloc_fail_after[kind], &n, n - 1)) { + return n == 1; + } + } + return false; +} + +void cbm_doclink_test_reset_alloc(void) { + for (int i = 0; i < CBM_DOCLINK_ALLOC_KINDS; i++) { + atomic_store(&doc_alloc_fail_after[i], 0); + } +} + +static _Atomic uint64_t doc_work_copied; +static _Atomic uint64_t doc_work_parse_input; +static _Atomic uint64_t doc_work_cleaned; + +void cbm_doclink_test_doc_work_reset(void) { + atomic_store(&doc_work_copied, 0); + atomic_store(&doc_work_parse_input, 0); + atomic_store(&doc_work_cleaned, 0); +} + +void cbm_doclink_test_doc_work(uint64_t *copied, uint64_t *parse_input, uint64_t *cleaned) { + *copied = atomic_load(&doc_work_copied); + *parse_input = atomic_load(&doc_work_parse_input); + *cleaned = atomic_load(&doc_work_cleaned); +} + +void cbm_doclink_test_note_doc_work(uint64_t copied, uint64_t parse_input, uint64_t cleaned) { + atomic_fetch_add(&doc_work_copied, copied); + atomic_fetch_add(&doc_work_parse_input, parse_input); + atomic_fetch_add(&doc_work_cleaned, cleaned); +} +#endif + +/* ── Link families ─────────────────────────────────────────────────── + * + * One row per CBMDocLinkSyntax value, generated from the family list in + * doclink.h (language, name, external, and the ship gate). */ + +static const CBMDocLinkFamily DOCLINK_FAMILIES[CBM_DOCLINK_SYNTAX_COUNT] = { +#define DOCLINK_FAMILY_ROW(id, lang, name, external, ships) [id] = {lang, name, external, ships}, + CBM_DOCLINK_FAMILY_LIST(DOCLINK_FAMILY_ROW) +#undef DOCLINK_FAMILY_ROW +}; + +const CBMDocLinkFamily *cbm_doclink_family(int syntax) { + if (syntax <= CBM_DOCLINK_NONE || syntax >= CBM_DOCLINK_SYNTAX_COUNT || + !DOCLINK_FAMILIES[syntax].name) { + return NULL; + } + return &DOCLINK_FAMILIES[syntax]; +} + +const char *cbm_doclink_syntax_name(int syntax) { + const CBMDocLinkFamily *f = cbm_doclink_family(syntax); + return f ? f->name : ""; +} + +bool cbm_doclink_syntax_is_external(int syntax) { + const CBMDocLinkFamily *f = cbm_doclink_family(syntax); + return f && f->external; +} + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +enum { DOCLINK_SHIP_AS_TABLE = 0, DOCLINK_SHIP_ON, DOCLINK_SHIP_OFF }; +static _Atomic unsigned char doclink_ship_override[CBM_DOCLINK_SYNTAX_COUNT]; + +void cbm_doclink_test_set_ships(int syntax, bool ships) { + if (cbm_doclink_family(syntax)) { + atomic_store(&doclink_ship_override[syntax], ships ? DOCLINK_SHIP_ON : DOCLINK_SHIP_OFF); + } +} + +void cbm_doclink_test_reset_ships(void) { + for (int i = 0; i < CBM_DOCLINK_SYNTAX_COUNT; i++) { + atomic_store(&doclink_ship_override[i], DOCLINK_SHIP_AS_TABLE); + } +} +#endif + +bool cbm_doclink_syntax_ships(int syntax) { + const CBMDocLinkFamily *f = cbm_doclink_family(syntax); + if (!f) { + return false; + } +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + unsigned char forced = atomic_load(&doclink_ship_override[syntax]); + if (forced != DOCLINK_SHIP_AS_TABLE) { + return forced == DOCLINK_SHIP_ON; + } +#endif + return f->ships; +} + +/* ── Language table ──────────────────────────────────────────────── */ + +/* One row per language with a doc-reference parser: everything the extraction + * half needs from a language leg. */ +typedef struct { + CBMLanguage lang; + /* One documented definition's doc text -> tokens (cbm_doclinks_push). */ + void (*parse_doc)(CBMExtractCtx *ctx, const CBMDefinition *def, const char *doc, + uint32_t doc_line); + /* The file's scope blob, or NULL: the language's resolver needs none. */ + const char *(*scan_scope)(CBMExtractCtx *ctx); + /* The tag line that blob starts with, and the blob as it is persisted + * (line numbers dropped). NULL: the blob is persisted as it is. */ + const char *scope_tag; + char *(*portable_scope)(const char *scope); + /* Labels that twin another definition of the same name and line (C# + * fields and constants are also emitted as module-level Variables with + * the same doc): only the first label carries the doc's references. */ + const char *twin_label; + const char *twin_of; +} doclink_lang_t; + +static const doclink_lang_t DOCLINK_LANGS[] = { + {.lang = CBM_LANG_CSHARP, + .parse_doc = cbm_doclink_cs_parse_doc, + .scan_scope = cbm_doclink_cs_scan_scope, + .scope_tag = CBM_DOCLINK_CS_SCOPE_TAG, + .portable_scope = cbm_doclink_cs_portable_scope, + .twin_label = "Variable", + .twin_of = "Field"}, + /* MSBuild project files: no references, but a scope the C# resolver reads */ + {.lang = CBM_LANG_XML, + .parse_doc = cbm_doclink_cs_project_parse_doc, + .scan_scope = cbm_doclink_cs_project_scan_scope, + .scope_tag = CBM_DOCLINK_CS_SCOPE_TAG, + .portable_scope = cbm_doclink_cs_portable_scope}, +}; + +static const doclink_lang_t *doclink_lang(CBMLanguage lang) { + for (size_t i = 0; i < sizeof(DOCLINK_LANGS) / sizeof(DOCLINK_LANGS[0]); i++) { + if (DOCLINK_LANGS[i].lang == lang) { + return &DOCLINK_LANGS[i]; + } + } + return NULL; +} + +bool cbm_doclink_lang_supported(CBMLanguage lang) { + return doclink_lang(lang) != NULL; +} + +/* ── Doc-line side map ───────────────────────────────────────────── */ + +typedef struct { + const char *doc; + uint32_t line; + int token_first; + int token_count; + bool parsed; +} doc_line_ent_t; + +typedef struct { + doc_line_ent_t *items; + int count; + int cap; + bool sorted; +} doc_line_map_t; + +enum { DOC_LINE_MAP_INIT = 64 }; + +static CBMArena *doclink_scratch(CBMExtractCtx *ctx) { + return ctx->scratch ? ctx->scratch : ctx->arena; +} + +void cbm_doclink_note_doc_line(CBMExtractCtx *ctx, const char *doc, uint32_t line) { + if (!ctx || !doc || !cbm_doclink_lang_supported(ctx->language)) { + return; + } + CBMArena *a = doclink_scratch(ctx); + doc_line_map_t *m = (doc_line_map_t *)ctx->doc_lines; + if (!m) { + m = (doc_line_map_t *)cbm_arena_alloc(a, sizeof(*m)); + if (!m) { + return; + } + memset(m, 0, sizeof(*m)); + ctx->doc_lines = m; + } + if (m->count >= m->cap) { + int ncap = m->cap ? m->cap * PAIR_LEN : DOC_LINE_MAP_INIT; + doc_line_ent_t *grown = (doc_line_ent_t *)cbm_arena_alloc(a, (size_t)ncap * sizeof(*grown)); + if (!grown) { + return; /* the reference keeps its definition's line instead */ + } + if (m->count > 0) { + memcpy(grown, m->items, (size_t)m->count * sizeof(*grown)); + } + m->items = grown; + m->cap = ncap; + } + m->items[m->count++] = (doc_line_ent_t){.doc = doc, .line = line}; + m->sorted = false; +} + +static int doc_line_cmp(const void *a, const void *b) { + uintptr_t x = (uintptr_t)((const doc_line_ent_t *)a)->doc; + uintptr_t y = (uintptr_t)((const doc_line_ent_t *)b)->doc; + return (x > y) - (x < y); +} + +/* Per-file metadata for this exact immutable doc pointer, never for a line. */ +static doc_line_ent_t *doc_info_of(CBMExtractCtx *ctx, const char *doc) { + doc_line_map_t *m = (doc_line_map_t *)ctx->doc_lines; + if (!m || m->count == 0) { + return NULL; + } + if (!m->sorted) { + qsort(m->items, (size_t)m->count, sizeof(m->items[0]), doc_line_cmp); + m->sorted = true; + } + int lo = 0; + int hi = m->count - SKIP_ONE; + uintptr_t key = (uintptr_t)doc; + while (lo <= hi) { + int mid = lo + ((hi - lo) / PAIR_LEN); + uintptr_t v = (uintptr_t)m->items[mid].doc; + if (v == key) { + return &m->items[mid]; + } + if (v < key) { + lo = mid + SKIP_ONE; + } else { + hi = mid - SKIP_ONE; + } + } + return NULL; +} + +/* A doc not built by doc_run_text keeps the definition's own line. */ +static uint32_t doc_line_of(CBMExtractCtx *ctx, const char *doc) { + const doc_line_ent_t *info = doc_info_of(ctx, doc); + return info ? info->line : 0; +} + +/* ── Persisted scope ─────────────────────────────────────────────── */ + +char *cbm_doclink_portable_scope(const char *scope) { + if (!scope) { + return NULL; + } + for (size_t i = 0; i < sizeof(DOCLINK_LANGS) / sizeof(DOCLINK_LANGS[0]); i++) { + const doclink_lang_t *L = &DOCLINK_LANGS[i]; + if (!L->scope_tag || !L->portable_scope) { + continue; + } + size_t tl = strlen(L->scope_tag); + if (strncmp(scope, L->scope_tag, tl) == 0 && (scope[tl] == '\n' || scope[tl] == '\0')) { + return L->portable_scope(scope); + } + } + return cbm_mem_strdup(CBM_MEM_CLASS_OTHER, scope); +} + +/* ── Driver ──────────────────────────────────────────────────────── */ + +enum { DOCLINK_INIT_CAP = 16 }; + +void cbm_doclinks_push(CBMDocLinkArray *arr, CBMArena *a, CBMDocLink link) { + if (arr->count >= arr->cap) { + int ncap = arr->cap ? arr->cap * PAIR_LEN : DOCLINK_INIT_CAP; + CBMDocLink *grown; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (cbm_doclink_test_fail_alloc(CBM_DOCLINK_ALLOC_TOKENS)) { + grown = NULL; + } else +#endif + { + grown = (CBMDocLink *)cbm_arena_alloc(a, (size_t)ncap * sizeof(*grown)); + } + if (!grown) { + return; + } + if (arr->count > 0) { + memcpy(grown, arr->items, (size_t)arr->count * sizeof(*grown)); + } + arr->items = grown; + arr->cap = ncap; + } + arr->items[arr->count++] = link; +} + +typedef struct { + uint32_t line; + const char *name; +} doclink_twin_key_t; + +static int twin_key_cmp(const void *a, const void *b) { + const doclink_twin_key_t *x = (const doclink_twin_key_t *)a; + const doclink_twin_key_t *y = (const doclink_twin_key_t *)b; + if (x->line != y->line) { + return x->line < y->line ? -1 : 1; + } + return strcmp(x->name, y->name); +} + +/* The (line, name) keys of the definitions a twin label duplicates, sorted; + * NULL when there are none. */ +static doclink_twin_key_t *twin_keys(CBMExtractCtx *ctx, const char *label, int *out_count) { + *out_count = 0; + const CBMDefArray *defs = &ctx->result->defs; + int n = 0; + for (int i = 0; i < defs->count; i++) { + const CBMDefinition *d = &defs->items[i]; + if (d->label && d->name && strcmp(d->label, label) == 0) { + n++; + } + } + if (n == 0) { + return NULL; + } + doclink_twin_key_t *keys = (doclink_twin_key_t *)cbm_arena_alloc( + doclink_scratch(ctx), (size_t)n * sizeof(doclink_twin_key_t)); + if (!keys) { + return NULL; + } + int k = 0; + for (int i = 0; i < defs->count; i++) { + const CBMDefinition *d = &defs->items[i]; + if (d->label && d->name && strcmp(d->label, label) == 0) { + keys[k].line = d->start_line; + keys[k].name = d->name; + k++; + } + } + qsort(keys, (size_t)k, sizeof(keys[0]), twin_key_cmp); + *out_count = k; + return keys; +} + +void cbm_doclinks_extract(CBMExtractCtx *ctx) { + if (!ctx || !ctx->result) { + return; + } + const doclink_lang_t *L = doclink_lang(ctx->language); + if (!L) { + return; + } + int twin_count = 0; + doclink_twin_key_t *twins = L->twin_label ? twin_keys(ctx, L->twin_of, &twin_count) : NULL; + CBMDefArray *defs = &ctx->result->defs; + for (int i = 0; i < defs->count; i++) { + const CBMDefinition *d = &defs->items[i]; + if (!d->docstring || !d->docstring[0] || !d->qualified_name || !d->name) { + continue; + } + if (twins && d->label && strcmp(d->label, L->twin_label) == 0) { + doclink_twin_key_t key = {d->start_line, d->name}; + if (bsearch(&key, twins, (size_t)twin_count, sizeof(twins[0]), twin_key_cmp)) { + continue; /* the Field twin carries these references */ + } + } + doc_line_ent_t *info = doc_info_of(ctx, d->docstring); + uint32_t doc_line = info ? info->line : 0; + bool reusable = ctx->language == CBM_LANG_CSHARP && info && info->line > 0; + CBMDocLinkArray *links = &ctx->result->doc_links; + if (reusable && info->parsed) { + /* C# lexical tokens depend only on the shared text and its line. + * Bind each replay to this definition. Indices survive array + * growth; no token pointer is kept across a push. */ + for (int k = 0; k < info->token_count; k++) { + CBMDocLink link = links->items[info->token_first + k]; + link.source_qn = d->qualified_name; + link.def_line = d->start_line; + cbm_doclinks_push(links, ctx->arena, link); + } + } else { + int first = links->count; + bool complete = false; + if (reusable) { + complete = cbm_doclink_cs_parse_doc_checked(ctx, d, d->docstring, doc_line); + } else { + L->parse_doc(ctx, d, d->docstring, doc_line ? doc_line : d->start_line); + } + if (reusable && complete) { + info->token_first = first; + info->token_count = links->count - first; + info->parsed = true; + } + } + } + /* The file's own doc: its references belong to the file. The parser gets + * a definition-shaped stand-in for the file; what it pushes is marked, and + * the resolving half takes the File node as the source. No qualified name + * for that node is computed on this side. */ + const char *file_doc = ctx->result->module_doc; + if (file_doc && file_doc[0]) { + CBMDocLinkArray *arr = &ctx->result->doc_links; + int first = arr->count; + CBMDefinition file_def = { + .name = ctx->rel_path, + .qualified_name = ctx->module_qn ? ctx->module_qn : "", + .label = "File", + .file_path = ctx->rel_path, + .start_line = SKIP_ONE, + .end_line = SKIP_ONE, + .docstring = file_doc, + }; + uint32_t doc_line = doc_line_of(ctx, file_doc); + L->parse_doc(ctx, &file_def, file_doc, doc_line ? doc_line : file_def.start_line); + for (int i = first; i < arr->count; i++) { + arr->items[i].flags |= CBM_DOCLINK_FLAG_FILE; + } + } + if (L->scan_scope) { + ctx->result->doc_scope = L->scan_scope(ctx); + } +} diff --git a/internal/cbm/doclink.h b/internal/cbm/doclink.h new file mode 100644 index 0000000000..960fdc329f --- /dev/null +++ b/internal/cbm/doclink.h @@ -0,0 +1,215 @@ +/* + * doclink.h — doc-comment references, extraction half. + * + * A definition's complete doc comment (extract_defs.c stores it whole) can + * name other code: C# ``, Java `{@link ...}`, Rust intra-doc + * links, and so on. Extraction only FINDS those references; whether one names + * a graph node is decided later, per file, by the pipeline's resolver + * (src/pipeline/doc_links.c), which turns them into MENTIONS edges or + * doc_link_unresolved rows. + * + * A language leg plugs in at two places of this half, and nowhere else: + * - its link families: lines of CBM_DOCLINK_FAMILY_LIST below (enum value, + * name, ship gate); + * - one row of the language table in doclink.c: + * parse_doc one documented definition's doc text -> CBMDocLink + * tokens + * scan_scope the file's doc-link SCOPE: a language-tagged text blob + * with what other files' resolution needs from this file + * (C#: namespaces, usings, type and member + * declarations). A pure function of the file's bytes, + * persisted with the file's LSP surface, and so + * identical on full and incremental runs. Optional. + * scope_tag, the blob's tag line and its persisted form (line + * portable_scope numbers dropped). Optional. + * State a language's hooks need between their calls for one file lives in + * ctx->doclink_state (cbm.h), never in a static or thread-local. + * A language without a row produces no tokens and no scope. The resolving + * half's hooks are a cbm_doclink_resolver_t (src/pipeline/doc_links.h). + */ +#ifndef CBM_DOCLINK_H +#define CBM_DOCLINK_H + +#include "cbm.h" + +#include +#include +#include + +/* Every link family, across languages: one line per (language, reference + * form), and the ONE place a language leg adds its families -- + * + * X(enum value, language, name, external, ships) + * + * name the "syntax" MENTIONS-edge property and doc_link_unresolved + * column, so it is public surface; with the language it is the + * tier a release audit judges. Names repeat across languages + * (`see` is C#'s, Java's, Kotlin's ...): the value, not the name, + * identifies the family, and a row's language is its file's. + * external outside the repository by construction (a URL): never looked up + * ships the SHIP GATE. true: a resolved reference becomes a MENTIONS + * edge. false: the tier is below the audit's bar; its references + * still resolve, but a resolved one stays a doc_link_unresolved + * row with reason below_bar_tier (an unresolved one keeps its own + * reason). Nothing else about the family changes. + * + * The enum below and the family table (doclink.c) are both generated from this + * list, so a value cannot exist without its name and its gate. */ +#define CBM_DOCLINK_FAMILY_LIST(X) \ + /* csharp */ \ + X(CBM_DOCLINK_CS_SEE, CBM_LANG_CSHARP, "see", false, true) \ + X(CBM_DOCLINK_CS_SEEALSO, CBM_LANG_CSHARP, "seealso", false, true) \ + X(CBM_DOCLINK_CS_EXCEPTION, CBM_LANG_CSHARP, "exception", false, true) \ + X(CBM_DOCLINK_CS_INHERITDOC, CBM_LANG_CSHARP, "inheritdoc", false, true) \ + /* any language (CBM_LANG_COUNT): a URL is never an edge */ \ + X(CBM_DOCLINK_HREF, CBM_LANG_COUNT, "href", true, false) + +typedef enum { + CBM_DOCLINK_NONE = 0, +#define CBM_DOCLINK_FAMILY_VALUE(id, lang, name, external, ships) id, + CBM_DOCLINK_FAMILY_LIST(CBM_DOCLINK_FAMILY_VALUE) +#undef CBM_DOCLINK_FAMILY_VALUE + CBM_DOCLINK_SYNTAX_COUNT +} CBMDocLinkSyntax; + +/* CBMDocLink.flags. */ +enum { + /* The reference is written in the FILE's own doc (CBMFileResult.module_doc: + * a Go package comment, Rust `//!` inner docs), not in a definition's. Its + * edge source is the file's File node, which the resolving half looks up + * from the file it is resolving; `source_qn` is not the source then. Set + * by the driver (cbm_doclinks_extract) on what a parser pushes for the + * file-level doc -- a parser never sets it. */ + CBM_DOCLINK_FLAG_FILE = 1, +}; + +/* One link family: a line of CBM_DOCLINK_FAMILY_LIST. */ +typedef struct { + CBMLanguage lang; /* the language whose doc comments write it; CBM_LANG_COUNT: any */ + const char *name; + bool external; + bool ships; +} CBMDocLinkFamily; + +/* The family of a syntax value; NULL for a value that names none. */ +const CBMDocLinkFamily *cbm_doclink_family(int syntax); + +/* "see", "seealso", "exception", "inheritdoc", "href"; "" for an unknown value. */ +const char *cbm_doclink_syntax_name(int syntax); + +/* True for a syntax whose reference is outside the repository by construction + * (a URL): the resolver records it as `external` without a lookup. */ +bool cbm_doclink_syntax_is_external(int syntax); + +/* True when resolved references of this family become edges (the ship gate). */ +bool cbm_doclink_syntax_ships(int syntax); + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +/* Test seam: ship or hold back one family whatever the table says, until + * cbm_doclink_test_reset_ships. Test builds only. */ +void cbm_doclink_test_set_ships(int syntax, bool ships); +void cbm_doclink_test_reset_ships(void); +#endif + +/* True when `lang` has a doc-reference parser. */ +bool cbm_doclink_lang_supported(CBMLanguage lang); + +/* Remember that the doc text `doc` (as returned by the doc-comment extractor) + * starts at 1-based source line `line`. Called by doc_run_text; a no-op for + * languages without a parser. */ +void cbm_doclink_note_doc_line(CBMExtractCtx *ctx, const char *doc, uint32_t line); + +/* Harvest the references of every documented definition into + * ctx->result->doc_links and the file's scope into ctx->result->doc_scope. + * Called once per file at the end of extraction, with the tree still alive. + * + * A file-level doc (ctx->result->module_doc) goes through the same parse_doc + * hook once more: `def` is then a stand-in for the FILE (label "File", name + * and file_path the relative path, qualified_name the file's module QN, + * start_line 1) and `doc_line` the doc's first line. The parser fills its + * tokens exactly as for a definition; the driver marks them + * CBM_DOCLINK_FLAG_FILE afterwards. */ +void cbm_doclinks_extract(CBMExtractCtx *ctx); + +/* Append one token (copies nothing: `raw` must live in the result arena). */ +void cbm_doclinks_push(CBMDocLinkArray *arr, CBMArena *a, CBMDocLink link); + +/* The scope as persisted with the file's LSP surface: line numbers zeroed, so + * an edit that only moves lines leaves it (and the surface hash) unchanged. + * Other files' references never read a file's lines; only the file's own + * references do, and those are resolved from its fresh extraction. Returns a + * memory-core block (release with cbm_free(CBM_MEM_CLASS_OTHER, p)), NULL on + * allocation failure. */ +char *cbm_doclink_portable_scope(const char *scope); + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +/* Fail one selected allocation attempt, then automatically disarm. */ +enum { + CBM_DOCLINK_ALLOC_SPAN, + CBM_DOCLINK_ALLOC_TEXT, + CBM_DOCLINK_ALLOC_VALUE, + CBM_DOCLINK_ALLOC_TOKENS, + CBM_DOCLINK_ALLOC_KINDS, +}; +void cbm_doclink_test_fail_alloc_after(int kind, int nth); +bool cbm_doclink_test_fail_alloc(int kind); +void cbm_doclink_test_reset_alloc(void); + +/* Actual comment memcpy bytes, input bytes submitted to the C# lexical parser, + * and bytes allocated for cleaned reference values; separate from token output. */ +void cbm_doclink_test_doc_work_reset(void); +void cbm_doclink_test_doc_work(uint64_t *copied, uint64_t *parse_input, uint64_t *cleaned); +void cbm_doclink_test_note_doc_work(uint64_t copied, uint64_t parse_input, uint64_t cleaned); +#endif + +/* ── C# (doclink_cs.c) ─────────────────────────────────────────────── */ + +void cbm_doclink_cs_parse_doc(CBMExtractCtx *ctx, const CBMDefinition *def, const char *doc, + uint32_t doc_line); +/* False if allocating a value or appending a token failed. Only a complete + * lexical result can be shared with another definition. */ +bool cbm_doclink_cs_parse_doc_checked(CBMExtractCtx *ctx, const CBMDefinition *def, const char *doc, + uint32_t doc_line); +const char *cbm_doclink_cs_scan_scope(CBMExtractCtx *ctx); +char *cbm_doclink_cs_portable_scope(const char *scope); + +/* MSBuild project files (*.csproj, *.props, *.targets) set the global usings + * of a C# project, so they have a scope blob too: the C# resolver evaluates + * those blobs and never opens a file. The blob carries the C# tag. The scan + * is gated by the file's name: any other XML file costs nothing and has no + * blob. A project file holds no doc references: its parse_doc does nothing. */ +void cbm_doclink_cs_project_parse_doc(CBMExtractCtx *ctx, const CBMDefinition *def, const char *doc, + uint32_t doc_line); +const char *cbm_doclink_cs_project_scan_scope(CBMExtractCtx *ctx); + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +/* Test seam: what the C# scope scans since the last reset cost -- the text + * positions, brace-stack entries and modifier children visited, and the bytes + * they took from the scratch arena. A test holds these against the size of its input, so that a + * scan whose cost grows faster than its input fails without a clock. Test + * builds only. */ +void cbm_doclink_cs_test_cost_reset(void); +void cbm_doclink_cs_test_cost(uint64_t *text_steps, uint64_t *scratch_bytes); +/* Completed C# scope-builder bytes since the cost reset, including any embedded + * terminator. A direct-extraction test compares this with the returned C string. */ +uint64_t cbm_doclink_cs_test_scope_bytes(void); +/* Import-index construction and a one-shot candidate-buffer failure. Query + * visits remain in the pipeline's existing test_work counter. */ +uint64_t cbm_doclink_cs_test_index_work(void); +void cbm_doclink_cs_test_fail_candidate_alloc(bool enabled); +bool cbm_doclink_cs_test_candidate_alloc_failed(void); +#endif + +/* Normalize one C# parameter type as written in a declaration or a cref + * parameter list: attributes, ref/out/in/params/this/scoped modifiers, type + * arguments, namespaces, nullable markers and a trailing parameter name are + * dropped, BCL names map to their keyword (Int32 -> int), array and pointer + * suffixes stay. "?" (a type nothing is known about) when nothing is left, when + * the text is too long to be a type, or when the result does not fit `out`: + * never a cut name. Writes a NUL-terminated string and returns its length. */ +size_t cbm_doclink_cs_norm_type(const char *in, size_t len, char *out, size_t cap); + +/* The C# scope blob starts with this tag line. */ +#define CBM_DOCLINK_CS_SCOPE_TAG "cs1" + +#endif /* CBM_DOCLINK_H */ diff --git a/internal/cbm/doclink_cs.c b/internal/cbm/doclink_cs.c new file mode 100644 index 0000000000..354debc294 --- /dev/null +++ b/internal/cbm/doclink_cs.c @@ -0,0 +1,3653 @@ +/* + * doclink_cs.c — C# doc-comment references and the C# doc-link scope. + * + * Tokens: the cref of , , and , and the + * href of / (external). / name the + * definition's own parameters and are not references; pulls doc + * text from another file. + * + * Scope blob (one record per line, tab-separated, first line "cs1"; records + * in document order, so a member follows its type): + * X from to lines whose declarations could not be + * placed (the braces stop pairing, a + * namespace cannot be named, a block nests + * past a limit): a definition there has no + * known scope + * R id parent start end name + * namespace region: `name` as its + * declaration writes it (`A.B`), inside the + * region `parent` (0 = the file) + * U region kind alias target + * using: n namespace, s static, a alias; + * a `g` after it for a `global using` + * T region start end kind outer name tparams bases + * type: c class, s struct, i interface, e + * enum, r record class, t record struct, d + * delegate; then `p` when it is declared + * partial, then `!` when a parse error hides + * some of its members. `outer` is the ordinal + * of the enclosing type's T record (its + * position among the file's T records), `-` + * for none. tparams ','-joined, bases + * '|'-joined as written + * M start kind explicit type name tparams sig + * member of the type with ordinal `type`: c + * callable (method, constructor, primary + * constructor), v field (constant, enum + * member), p property (record parameter), e + * event, o operator (name: its token, or + * `implicit` / `explicit`), x indexer (name + * `this`); then `s` for what `using static` + * brings in (static or const, an enum's + * member, no extension method; a static + * constructor too). sig: '|'-joined normalized + * parameter types, '-' for v, p and e + * Q name a type that is declared, but whose + * namespace or outer type could not be + * established + * A record names its outer type and its owner by ordinal, never by a path, so + * the blob grows with the file and not with the nesting. Everything is + * derived from the tree alone, so the blob is a pure function of the file's + * bytes. Nesting is the tree's while the tree has no parse error, and the + * braces' otherwise (see "Braces" below). + */ +#include "doclink.h" + +#include "arena.h" +#include "foundation/constants.h" +#include "foundation/mem_core.h" +#include "tree_sitter/api.h" + +#include +#include +#include +#include +#include + +/* ── Small arena string builder ──────────────────────────────────── */ + +typedef struct { + CBMArena *a; + char *buf; + size_t len; + size_t cap; + bool failed; +} cs_sb_t; + +enum { CS_SB_INIT = 1024, CS_UINT_DIGITS = 16, CS_NAME_MAX = 512 }; + +static void sb_reserve(cs_sb_t *sb, size_t extra) { + if (sb->failed || sb->len + extra + SKIP_ONE <= sb->cap) { + return; + } + size_t ncap = sb->cap ? sb->cap : CS_SB_INIT; + while (ncap < sb->len + extra + SKIP_ONE) { + ncap *= PAIR_LEN; + } + char *grown = (char *)cbm_arena_alloc(sb->a, ncap); + if (!grown) { + sb->failed = true; + return; + } + if (sb->len > 0) { + memcpy(grown, sb->buf, sb->len); + } + sb->buf = grown; + sb->cap = ncap; +} + +static void sb_putn(cs_sb_t *sb, const char *s, size_t n) { + sb_reserve(sb, n); + if (sb->failed) { + return; + } + memcpy(sb->buf + sb->len, s, n); + sb->len += n; + sb->buf[sb->len] = '\0'; +} + +static void sb_puts(cs_sb_t *sb, const char *s) { + sb_putn(sb, s, strlen(s)); +} + +static void sb_putc(cs_sb_t *sb, char c) { + sb_putn(sb, &c, SKIP_ONE); +} + +static void sb_putu(cs_sb_t *sb, uint32_t v) { + char tmp[CS_UINT_DIGITS]; + int n = snprintf(tmp, sizeof(tmp), "%u", v); + if (n > 0) { + sb_putn(sb, tmp, (size_t)n); + } +} + +/* ── Doc-comment references ──────────────────────────────────────── */ + +static int cs_tag_syntax(const char *name, size_t len) { + static const struct { + const char *name; + int syntax; + } tags[] = { + {"see", CBM_DOCLINK_CS_SEE}, + {"seealso", CBM_DOCLINK_CS_SEEALSO}, + {"exception", CBM_DOCLINK_CS_EXCEPTION}, + {"inheritdoc", CBM_DOCLINK_CS_INHERITDOC}, + }; + for (size_t i = 0; i < sizeof(tags) / sizeof(tags[0]); i++) { + if (strlen(tags[i].name) == len && memcmp(tags[i].name, name, len) == 0) { + return tags[i].syntax; + } + } + return CBM_DOCLINK_NONE; +} + +static bool cs_attr_name_char(char c) { + return isalnum((unsigned char)c) || c == '_' || c == ':' || c == '.' || c == '-'; +} + +/* Decode the five XML entities, drop the comment prefix (`///`, ` * `) that + * follows a line break inside a value, collapse whitespace, trim. */ +static const char *cs_clean_value(CBMArena *a, const char *v, size_t n) { + char *out; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (cbm_doclink_test_fail_alloc(CBM_DOCLINK_ALLOC_VALUE)) { + out = NULL; + } else +#endif + { + out = (char *)cbm_arena_alloc(a, n + SKIP_ONE); + } +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (out) { + cbm_doclink_test_note_doc_work(0, 0, n + SKIP_ONE); + } +#endif + if (!out) { + return NULL; + } + static const struct { + const char *ent; + char ch; + } ents[] = {{"<", '<'}, {">", '>'}, {"&", '&'}, {""", '"'}, {"'", '\''}}; + size_t w = 0; + bool space = false; + size_t i = 0; + while (i < n) { + char c = v[i]; + if (c == '\n' || c == '\r') { + i++; + while (i < n && (v[i] == ' ' || v[i] == '\t' || v[i] == '\r' || v[i] == '\n')) { + i++; + } + static const char line_doc[] = "///"; + size_t ld = sizeof(line_doc) - SKIP_ONE; + if (i + ld <= n && memcmp(v + i, line_doc, ld) == 0) { + i += ld; /* the next `///` line of the same comment */ + } else if (i < n && v[i] == '*' && !(i + SKIP_ONE < n && v[i + SKIP_ONE] == '/')) { + i++; /* a block comment's leading star */ + } + space = true; + continue; + } + if (c == ' ' || c == '\t') { + space = true; + i++; + continue; + } + if (space && w > 0) { + out[w++] = ' '; + } + space = false; + if (c == '&') { + bool decoded = false; + for (size_t e = 0; e < sizeof(ents) / sizeof(ents[0]); e++) { + size_t el = strlen(ents[e].ent); + if (i + el <= n && memcmp(v + i, ents[e].ent, el) == 0) { + out[w++] = ents[e].ch; + i += el; + decoded = true; + break; + } + } + if (decoded) { + continue; + } + } + out[w++] = c; + i++; + } + out[w] = '\0'; + return out; +} + +bool cbm_doclink_cs_parse_doc_checked(CBMExtractCtx *ctx, const CBMDefinition *def, const char *doc, + uint32_t doc_line) { +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + cbm_doclink_test_note_doc_work(0, strlen(doc), 0); +#endif + bool complete = true; + const char *p = doc; + const char *counted = doc; + uint32_t line = doc_line; + while ((p = strchr(p, '<')) != NULL) { + /* what stands in an XML comment or a CDATA section is text, not + * markup: a cref written there is no reference */ + static const char comment_open[] = "" : "]]>"); + if (!end) { + break; /* it does not end: the rest of the text is inside it */ + } + p = end + 3; + continue; + } + const char *tag = p; + const char *q = p + SKIP_ONE; + while (*q == ' ' || *q == '\t') { + q++; + } + const char *name = q; + while (isalpha((unsigned char)*q)) { + q++; + } + int syntax = cs_tag_syntax(name, (size_t)(q - name)); + if (syntax == CBM_DOCLINK_NONE || + !(*q == ' ' || *q == '\t' || *q == '\n' || *q == '\r' || *q == '/' || *q == '>')) { + p = tag + SKIP_ONE; + continue; + } + const char *cref = NULL; + size_t cref_len = 0; + const char *href = NULL; + size_t href_len = 0; + bool closed = false; + while (*q) { + if (*q == '>') { + closed = true; + q++; + break; + } + if (*q == '<') { + break; /* malformed: another tag starts first */ + } + if (*q == '"' || *q == '\'') { + const char *end = strchr(q + SKIP_ONE, *q); + if (!end) { + break; + } + q = end + SKIP_ONE; + continue; + } + if (!cs_attr_name_char(*q)) { + q++; + continue; + } + const char *an = q; + while (cs_attr_name_char(*q)) { + q++; + } + size_t an_len = (size_t)(q - an); + const char *r = q; + while (*r == ' ' || *r == '\t' || *r == '\n' || *r == '\r') { + r++; + } + if (*r != '=') { + continue; + } + r++; + while (*r == ' ' || *r == '\t' || *r == '\n' || *r == '\r') { + r++; + } + if (*r != '"' && *r != '\'') { + q = r; + continue; + } + const char *vend = strchr(r + SKIP_ONE, *r); + if (!vend) { + break; + } + const char *val = r + SKIP_ONE; + size_t val_len = (size_t)(vend - val); + if (an_len == 4 && memcmp(an, "cref", 4) == 0) { + cref = val; + cref_len = val_len; + } else if (an_len == 4 && memcmp(an, "href", 4) == 0) { + href = val; + href_len = val_len; + } + q = vend + SKIP_ONE; + } + if (!closed) { + p = tag + SKIP_ONE; + continue; + } + for (; counted < tag; counted++) { + if (*counted == '\n') { + line++; + } + } + const char *raw = NULL; + int tok_syntax = CBM_DOCLINK_NONE; + if (cref) { + raw = cs_clean_value(ctx->arena, cref, cref_len); + tok_syntax = syntax; + } else if (href && (syntax == CBM_DOCLINK_CS_SEE || syntax == CBM_DOCLINK_CS_SEEALSO)) { + raw = cs_clean_value(ctx->arena, href, href_len); + tok_syntax = CBM_DOCLINK_HREF; + } + if (tok_syntax != CBM_DOCLINK_NONE && !raw) { + complete = false; + } + if (raw && raw[0]) { + CBMDocLink link = { + .source_qn = def->qualified_name, + .raw = raw, + .line = line, + .def_line = def->start_line, + .syntax = (uint16_t)tok_syntax, + }; + int before = ctx->result->doc_links.count; + cbm_doclinks_push(&ctx->result->doc_links, ctx->arena, link); + if (ctx->result->doc_links.count == before) { + complete = false; + } + } + p = q; + } + return complete; +} + +void cbm_doclink_cs_parse_doc(CBMExtractCtx *ctx, const CBMDefinition *def, const char *doc, + uint32_t doc_line) { + (void)cbm_doclink_cs_parse_doc_checked(ctx, def, doc, doc_line); +} + +/* ── Parameter type normalization ────────────────────────────────── */ + +typedef struct { + char *s; + size_t len; +} cs_str_t; + +static bool cs_ident_start(char c) { + return isalpha((unsigned char)c) || c == '_' || c == '@'; +} + +static bool cs_ident_char(char c) { + return isalnum((unsigned char)c) || c == '_'; +} + +/* Remove [start, end) from s. */ +static void cs_cut(cs_str_t *t, size_t start, size_t end) { + memmove(t->s + start, t->s + end, t->len - end + SKIP_ONE); + t->len -= end - start; +} + +static void cs_trim(cs_str_t *t) { + size_t b = 0; + while (b < t->len && isspace((unsigned char)t->s[b])) { + b++; + } + if (b > 0) { + cs_cut(t, 0, b); + } + while (t->len > 0 && isspace((unsigned char)t->s[t->len - SKIP_ONE])) { + t->s[--t->len] = '\0'; + } +} + +/* Index just past the bracket group opened at s[i] (one of < { [ ( ), + * matching nested groups of any of those kinds; t->len when unbalanced. */ +static size_t cs_group_end(const cs_str_t *t, size_t i) { + int depth = 0; + for (size_t k = i; k < t->len; k++) { + char c = t->s[k]; + if (c == '<' || c == '{' || c == '[' || c == '(') { + depth++; + } else if (c == '>' || c == '}' || c == ']' || c == ')') { + depth--; + if (depth == 0) { + return k + SKIP_ONE; + } + } + } + return t->len; +} + +/* Leading parameter attribute lists: [NotNull] [In] T x. */ +static void cs_drop_attributes(cs_str_t *t) { + cs_trim(t); + while (t->len > 0 && t->s[0] == '[') { + size_t e = cs_group_end(t, 0); + cs_cut(t, 0, e); + cs_trim(t); + } +} + +/* Nullable / Nullable{X} (optionally System.- or global::System.- + * qualified) -> X. */ +static void cs_unwrap_nullable(cs_str_t *t) { + static const char *const prefixes[] = {"global::System.Nullable", "System.Nullable", + "Nullable"}; + /* one pass: what an unwrapped group leaves at its place is looked at + * again (Nullable>), the text before it never is */ + for (size_t i = 0; i < t->len;) { + bool unwrapped = false; + bool boundary = i == 0 || !(cs_ident_char(t->s[i - SKIP_ONE]) || + t->s[i - SKIP_ONE] == '.' || t->s[i - SKIP_ONE] == ':'); + for (size_t pi = 0; boundary && !unwrapped && pi < sizeof(prefixes) / sizeof(prefixes[0]); + pi++) { + size_t pl = strlen(prefixes[pi]); + if (i + pl > t->len || memcmp(t->s + i, prefixes[pi], pl) != 0) { + continue; + } + size_t j = i + pl; + while (j < t->len && t->s[j] == ' ') { + j++; + } + if (j >= t->len || (t->s[j] != '<' && t->s[j] != '{')) { + continue; + } + size_t e = cs_group_end(t, j); + if (e > t->len || e <= j + SKIP_ONE) { + continue; + } + /* keep the inner text */ + size_t inner_len = e - j - PAIR_LEN; + memmove(t->s + i, t->s + j + SKIP_ONE, inner_len); + memmove(t->s + i + inner_len, t->s + e, t->len - e + SKIP_ONE); + t->len = i + inner_len + (t->len - e); + unwrapped = true; + } + if (!unwrapped) { + i++; + } + } +} + +static bool cs_word_at(const cs_str_t *t, size_t i, const char *w) { + size_t wl = strlen(w); + if (i + wl > t->len || memcmp(t->s + i, w, wl) != 0) { + return false; + } + if (i > 0 && cs_ident_char(t->s[i - SKIP_ONE])) { + return false; + } + return i + wl < t->len && isspace((unsigned char)t->s[i + wl]); +} + +static void cs_drop_modifiers(cs_str_t *t) { + static const char *const mods[] = {"ref", "out", "in", "params", + "this", "scoped", "readonly", "final"}; + for (size_t i = 0; i < t->len;) { + bool cut = false; + for (size_t m = 0; m < sizeof(mods) / sizeof(mods[0]); m++) { + if (cs_word_at(t, i, mods[m])) { + size_t e = i + strlen(mods[m]); + while (e < t->len && isspace((unsigned char)t->s[e])) { + e++; + } + cs_cut(t, i, e); + cut = true; + break; + } + } + if (!cut) { + i++; + } + } +} + +/* Doc-ID arity markers: `1, ``0. */ +static void cs_drop_backtick_arity(cs_str_t *t) { + for (size_t i = 0; i < t->len;) { + if (t->s[i] != '`') { + i++; + continue; + } + size_t e = i; + while (e < t->len && t->s[e] == '`') { + e++; + } + size_t d = e; + while (d < t->len && isdigit((unsigned char)t->s[d])) { + d++; + } + if (d > e) { + cs_cut(t, i, d); + } else { + i = e; + } + } +} + +enum { CS_NORM_WORK = 512 }; + +/* Remove every <...> and {...} group. An opener pairs with the next closer of + * its own kind that no unpaired opener stands before (`<` with `>`, `{` with + * `}`), so groups nest; what does not pair stays. Two passes over the text: + * pair, then copy what is outside. */ +static void cs_drop_type_args(cs_str_t *t) { + uint16_t past[CS_NORM_WORK]; /* for a paired opener: index just past its closer */ + uint16_t open[CS_NORM_WORK]; + size_t depth = 0; + if (t->len >= CS_NORM_WORK) { + return; + } + for (size_t i = 0; i < t->len; i++) { + char c = t->s[i]; + past[i] = 0; + if (c == '<' || c == '{') { + open[depth++] = (uint16_t)i; + } else if (depth > 0 && ((c == '>' && t->s[open[depth - SKIP_ONE]] == '<') || + (c == '}' && t->s[open[depth - SKIP_ONE]] == '{'))) { + past[open[--depth]] = (uint16_t)(i + SKIP_ONE); + } + } + size_t w = 0; + for (size_t i = 0; i < t->len;) { + if (past[i]) { + i = past[i]; + } else { + t->s[w++] = t->s[i++]; + } + } + t->s[w] = '\0'; + t->len = w; +} + +static const char *cs_bcl_alias(const char *base, size_t len) { + static const struct { + const char *bcl; + const char *kw; + } map[] = { + {"Int32", "int"}, {"Int64", "long"}, {"Int16", "short"}, {"Byte", "byte"}, + {"SByte", "sbyte"}, {"UInt32", "uint"}, {"UInt64", "ulong"}, {"UInt16", "ushort"}, + {"Single", "float"}, {"Double", "double"}, {"Decimal", "decimal"}, {"Boolean", "bool"}, + {"Char", "char"}, {"String", "string"}, {"Object", "object"}, {"IntPtr", "nint"}, + {"UIntPtr", "nuint"}, {"Void", "void"}, + }; + for (size_t i = 0; i < sizeof(map) / sizeof(map[0]); i++) { + if (strlen(map[i].bcl) == len && memcmp(map[i].bcl, base, len) == 0) { + return map[i].kw; + } + } + return NULL; +} + +size_t cbm_doclink_cs_norm_type(const char *in, size_t len, char *out, size_t cap) { + if (!out || cap == 0) { + return 0; + } + out[0] = '\0'; + char work[CS_NORM_WORK]; + if (!in || len == 0 || len >= sizeof(work)) { + snprintf(out, cap, "?"); + return strlen(out); + } + memcpy(work, in, len); + work[len] = '\0'; + cs_str_t t = {work, len}; + cs_drop_attributes(&t); + cs_unwrap_nullable(&t); + cs_drop_modifiers(&t); + cs_drop_backtick_arity(&t); + cs_drop_type_args(&t); + /* "..." -> "[]" (varargs spelling) */ + for (size_t i = 0; i + 2 < t.len; i++) { + if (t.s[i] == '.' && t.s[i + 1] == '.' && t.s[i + 2] == '.') { + t.s[i] = '['; + t.s[i + 1] = ']'; + cs_cut(&t, i + 2, i + 3); + } + } + while (t.len > 0 && t.s[t.len - SKIP_ONE] == '@') { + t.s[--t.len] = '\0'; + } + cs_trim(&t); + /* A trailing identifier after whitespace is the parameter name. */ + size_t last_ws = 0; + bool have_ws = false; + for (size_t i = 0; i < t.len; i++) { + if (isspace((unsigned char)t.s[i])) { + last_ws = i; + have_ws = true; + } + } + if (have_ws) { + size_t s0 = last_ws + SKIP_ONE; + bool ident = s0 < t.len && cs_ident_start(t.s[s0]); + for (size_t i = s0 + SKIP_ONE; ident && i < t.len; i++) { + ident = cs_ident_char(t.s[i]); + } + if (ident) { + t.s[last_ws] = '\0'; + t.len = last_ws; + } + } + /* Drop all whitespace and trailing ?/! markers. */ + size_t w = 0; + for (size_t i = 0; i < t.len; i++) { + if (!isspace((unsigned char)t.s[i])) { + t.s[w++] = t.s[i]; + } + } + t.s[w] = '\0'; + t.len = w; + while (t.len > 0 && (t.s[t.len - SKIP_ONE] == '?' || t.s[t.len - SKIP_ONE] == '!')) { + t.s[--t.len] = '\0'; + } + /* base, then array ([] [,]) and pointer (*) suffixes */ + size_t suf = t.len; + while (suf > 0 && t.s[suf - SKIP_ONE] == '*') { + suf--; + } + for (;;) { + if (suf > 0 && t.s[suf - SKIP_ONE] == ']') { + size_t k = suf - SKIP_ONE; + while (k > 0 && t.s[k - SKIP_ONE] == ',') { + k--; + } + if (k > 0 && t.s[k - SKIP_ONE] == '[') { + suf = k - SKIP_ONE; + continue; + } + } + break; + } + size_t base_end = suf; + while (base_end > 0 && t.s[base_end - SKIP_ONE] == '?') { + base_end--; + } + size_t base_start = 0; + for (size_t i = 0; i < base_end; i++) { + if (t.s[i] == '.') { + base_start = i + SKIP_ONE; + } else if (t.s[i] == ':' && i + SKIP_ONE < base_end && t.s[i + SKIP_ONE] == ':') { + base_start = i + PAIR_LEN; + } + } + if (base_start < base_end && t.s[base_start] == '@') { + base_start++; + } + size_t blen = base_end > base_start ? base_end - base_start : 0; + if (blen == 0) { + snprintf(out, cap, "?"); + return strlen(out); + } + const char *kw = cs_bcl_alias(t.s + base_start, blen); + int n = kw ? snprintf(out, cap, "%s%.*s", kw, (int)(t.len - suf), t.s + suf) + : snprintf(out, cap, "%.*s%.*s", (int)blen, t.s + base_start, (int)(t.len - suf), + t.s + suf); + if (n < 0 || (size_t)n >= cap) { + /* it does not fit: an unknown type, never a cut one (two long names + * cut to one prefix would compare equal) */ + snprintf(out, cap, "?"); + } + return strlen(out); +} + +/* ── Scope scan ──────────────────────────────────────────────────── */ + +/* An entry of the brace list: a structural brace, or the nesting depth a + * preprocessor branch starts from. */ +typedef struct { + uint32_t pos; /* byte offset */ + uint32_t row; /* 0-based */ + int match; /* index of the paired brace, CBM_NOT_FOUND when unpaired */ + int alias; /* the first branch's brace this one stands in for, or CBM_NOT_FOUND */ + int depth; /* nesting depth just after this entry */ + char kind; /* '{', '}', or '#' for a branch mark */ +} cs_brace_t; + +/* A declaration keyword (namespace, class, struct ...). */ +typedef struct { + uint32_t end; /* first byte after the keyword */ + uint32_t row; /* 0-based */ + char kind; /* N namespace; c s i e r as for types */ + bool partial; /* the word before it is `partial` */ +} cs_head_t; + +/* An open brace while the braces are paired. The open braces form a stack + * that is never copied: every open brace is one entry that names the one + * below it, so "the stack as it stood at the #if" is a single index, however + * deep the nesting, and going back to it costs nothing. */ +typedef struct { + int brace; /* its entry in the brace list */ + int below; /* the open brace under it, or CBM_NOT_FOUND */ +} cs_open_t; + +/* An open #if while the braces are paired. */ +typedef struct { + int at_if; /* the top open brace where the #if stands (CBM_NOT_FOUND: none) */ + int n_if; + int end1; /* ... and where its first branch ended; valid once has_end1 */ + int n_end1; + int first_open; /* first open-brace entry allocated in the current branch */ + bool has_end1; /* an #else was seen */ + int mark; /* its entry in the brace list */ +} cs_pp_t; + +enum { + CS_OWNER_NONE = -1, /* not in a type */ + CS_OWNER_LEXICAL = -2, /* found in an error node: whichever type's braces hold it */ + CS_ITEM_TEXT = -3, /* frame of a declaration read from the text */ + CS_PP_IF = 1, + CS_PP_ELSE, + CS_PP_ENDIF, + CS_PP_MAX = 32, /* nested #if */ + CS_CHAR_LITERAL_MAX = 12, /* '\U0010FFFF' */ + CS_RAW_QUOTES = 3, /* """ */ + CS_LEX_MAX_NEST = 64, /* interpolated strings inside interpolation holes */ + /* How deep a file may nest what the scope records. A block past either + * limit is not placed: its types are named in Q records, what is + * documented inside has no scope (an X range). Deeper than any program; + * the limits keep a lookup's walk over the enclosing types and namespaces + * bounded. */ + CS_MAX_TYPE_NEST = 64, /* types inside types */ + CS_MAX_NS_SEGMENTS = 64, /* segments of a namespace's full name */ +}; + +/* One declaration of the file. `node` is the declaration (a field's + * declaration for each of its declarators); a declaration read from the + * text, where the tree has no node for it, has none. */ +typedef struct { + TSNode node; + TSNode name; + TSNode tparams; + TSNode params; + const char *text_name; /* text declarations; M: an operator's or indexer's name */ + const char *text_tparams; /* text declarations: "T,U" */ + uint32_t start; /* byte offset: document order */ + uint32_t line; /* 1-based */ + int brace; /* text declarations: the brace opening the block, or CBM_NOT_FOUND */ + int tree_idx; /* position among the tree's items; CS_ITEM_TEXT for a text one */ + int owner; /* M: tree_idx of the type the tree nests it in, or CS_OWNER_* */ + char tag; /* N namespace block, F file-scoped namespace, U using, T type, + M member */ + char kind; /* T: c s i e r t d; M: c v p e o x */ + bool explicit_impl; + bool is_static; /* M: what `using static` brings in */ + bool partial; /* T: declared `partial` */ + bool broken; /* T: a parse error sits among its members */ + bool from_text; /* read from the text, not from a declaration node */ +} cs_item_t; + +typedef struct { + CBMExtractCtx *ctx; + CBMArena *tmp; /* names, paths, items, tokens: nothing of it outlives the scan */ + cs_sb_t sb; + cs_item_t *items; + int nitems; + int cap_items; + cs_brace_t *braces; + int nbraces; + int cap_braces; + cs_head_t *heads; + int nheads; + int cap_heads; + cs_open_t *open; /* brace pairing: every brace that was ever open */ + int nopen; + int cap_open; + int top; /* the innermost open brace (index into open), or CBM_NOT_FOUND */ + int sp; /* how many are open */ + cs_pp_t pp[CS_PP_MAX]; + int npp; + uint32_t row_pos; /* row cursor: the row of byte row_pos is `row` */ + uint32_t row; + uint64_t cost_steps; /* text positions and brace-stack entries visited */ + uint64_t cost_bytes; /* bytes taken from the scratch arena */ + bool failed; /* out of memory */ + bool lexical; /* the tree has parse errors: nesting is read from the braces */ + uint32_t untrusted; /* byte offset from which the braces do not pair up */ + uint32_t untrusted_row; /* its 0-based row */ + int next_region; + int types_out; /* T records written: a type's ordinal is its position among them */ + uint32_t root_end_byte; + uint32_t root_end_line; +} cs_scan_t; + +enum { + CS_ITEMS_INIT = 256, + CS_BRACES_INIT = 1024, + CS_HEADS_INIT = 64, + CS_HEADER_SCAN_MAX = 4096, /* bytes from a type's name to its `{` or `;` */ + CS_TPARAMS_SCAN_MAX = 1024, +}; + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +static _Atomic uint64_t cs_cost_text_steps; +static _Atomic uint64_t cs_cost_scratch_bytes; +static _Atomic uint64_t cs_cost_scope_bytes; + +void cbm_doclink_cs_test_cost_reset(void) { + atomic_store(&cs_cost_text_steps, 0); + atomic_store(&cs_cost_scratch_bytes, 0); + atomic_store(&cs_cost_scope_bytes, 0); +} + +void cbm_doclink_cs_test_cost(uint64_t *text_steps, uint64_t *scratch_bytes) { + *text_steps = atomic_load(&cs_cost_text_steps); + *scratch_bytes = atomic_load(&cs_cost_scratch_bytes); +} + +uint64_t cbm_doclink_cs_test_scope_bytes(void) { + return atomic_load(&cs_cost_scope_bytes); +} +#endif + +/* Hand a cost to the test seam (nothing in a product build). */ +static void cs_cost_add(uint64_t steps, uint64_t bytes) { +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + atomic_fetch_add(&cs_cost_text_steps, steps); + atomic_fetch_add(&cs_cost_scratch_bytes, bytes); +#else + (void)steps; + (void)bytes; +#endif +} + +/* Hand a finished scan's cost to the test seam. */ +static void cs_cost_publish(const cs_scan_t *s) { + cs_cost_add(s->cost_steps, s->cost_bytes); +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (!s->failed && !s->sb.failed && s->sb.buf) { + atomic_fetch_add(&cs_cost_scope_bytes, s->sb.len); + } +#endif +} + +static bool cs_kind_is(TSNode n, const char *kind) { + return strcmp(ts_node_type(n), kind) == 0; +} + +static TSNode cs_field(TSNode n, const char *field) { + return ts_node_child_by_field_name(n, field, (uint32_t)strlen(field)); +} + +/* The children of a node, in order. ts_node_child(n, i) walks from the first + * child on every call, so a loop over i costs the square of the child count + * (a parameter list, a base list and a class body are as long as the file + * makes them); a cursor steps from one child to the next. */ +typedef struct { + TSTreeCursor cur; + bool started; + bool done; +} cs_kids_t; + +static cs_kids_t cs_kids(TSNode parent) { + return (cs_kids_t){.cur = ts_tree_cursor_new(parent)}; +} + +/* The next child, named or not; false after the last one. */ +static bool cs_kids_next(cs_kids_t *k, TSNode *out) { + if (k->done) { + return false; + } + bool moved = k->started ? ts_tree_cursor_goto_next_sibling(&k->cur) + : ts_tree_cursor_goto_first_child(&k->cur); + k->started = true; + if (!moved) { + k->done = true; + return false; + } + *out = ts_tree_cursor_current_node(&k->cur); + return true; +} + +/* The next NAMED child; false after the last one. */ +static bool cs_kids_next_named(cs_kids_t *k, TSNode *out) { + while (cs_kids_next(k, out)) { + if (ts_node_is_named(*out)) { + return true; + } + } + return false; +} + +static void cs_kids_end(cs_kids_t *k) { + ts_tree_cursor_delete(&k->cur); +} + +/* The first named child of `n` of kind `kind`; a null node when it has none. */ +static TSNode cs_child_of_kind(TSNode n, const char *kind) { + TSNode found = {0}; + cs_kids_t k = cs_kids(n); + TSNode c; + while (cs_kids_next_named(&k, &c)) { + if (strcmp(ts_node_type(c), kind) == 0) { + found = c; + break; + } + } + cs_kids_end(&k); + return found; +} + +/* A declaration's type_parameter_list: a named field in some grammar + * versions, an unnamed child in others. */ +static TSNode cs_type_params(TSNode decl) { + TSNode tp = cs_field(decl, "type_parameters"); + if (!ts_node_is_null(tp)) { + return tp; + } + return cs_child_of_kind(decl, "type_parameter_list"); +} + +/* Node text without whitespace and without verbatim '@' markers, appended to + * sb. `global::` prefixes are dropped. Text that is too long to be a type + * name, or that holds a scope separator or non-whitespace control byte, is + * written as "?" (an unresolvable name: the declaring type then counts as + * having an open hierarchy). The whole field is replaced before any copy. */ +static void cs_put_text_nows(cs_scan_t *s, TSNode n) { + const char *src = s->ctx->source; + uint32_t a = ts_node_start_byte(n); + uint32_t b = ts_node_end_byte(n); + if (b - a > 8 && memcmp(src + a, "global::", 8) == 0) { + a += 8; + } + bool ok = b >= a && b - a <= CS_NAME_MAX; + for (uint32_t i = a; ok && i < b; i++) { + unsigned char c = (unsigned char)src[i]; + ok = c != '|' && c != ';' && c != '{' && c != '}' && + !((c < 0x20 && !isspace(c)) || c == 0x7f); + } + if (!ok) { + sb_putc(&s->sb, '?'); + return; + } + for (uint32_t i = a; i < b; i++) { + char c = src[i]; + if (isspace((unsigned char)c) || c == '@') { + continue; + } + sb_putc(&s->sb, c); + } +} + +static void *cs_tmp_alloc(cs_scan_t *s, size_t n) { + s->cost_bytes += n; + void *p = cbm_arena_alloc(s->tmp, n ? n : SKIP_ONE); + if (!p) { + s->failed = true; + } + return p; +} + +/* Copy of source bytes [a, b) without whitespace and '@'. NULL unless it is a + * (dotted) identifier of sane length: an error-recovered parse can hand back a + * "name" spanning arbitrary code, which must not become a declaration. */ +static char *cs_ident_dup(cs_scan_t *s, uint32_t a, uint32_t b) { + const char *src = s->ctx->source; + if (b <= a || b - a > CS_NAME_MAX) { + return NULL; + } + char *out = (char *)cs_tmp_alloc(s, (size_t)(b - a) + SKIP_ONE); + if (!out) { + return NULL; + } + size_t w = 0; + for (uint32_t i = a; i < b; i++) { + unsigned char c = (unsigned char)src[i]; + if (isspace(c) || c == '@') { + continue; + } + if (!(isalnum(c) || c == '_' || c == '.' || c >= 0x80)) { + return NULL; /* not an identifier */ + } + out[w++] = (char)c; + } + out[w] = '\0'; + return w > 0 ? out : NULL; +} + +static char *cs_name_dup(cs_scan_t *s, TSNode n) { + if (ts_node_is_null(n)) { + return NULL; + } + return cs_ident_dup(s, ts_node_start_byte(n), ts_node_end_byte(n)); +} + +static uint32_t cs_line(TSNode n) { + return ts_node_start_point(n).row + TS_LINE_OFFSET; +} + +static uint32_t cs_end_line(TSNode n) { + return ts_node_end_point(n).row + TS_LINE_OFFSET; +} + +/* type_parameter_list -> "T,U" into sb (nothing when absent). */ +static void cs_put_tparams(cs_scan_t *s, TSNode list) { + if (ts_node_is_null(list)) { + return; + } + bool first = true; + cs_kids_t k = cs_kids(list); + TSNode tp; + while (cs_kids_next_named(&k, &tp)) { + if (!cs_kind_is(tp, "type_parameter")) { + continue; + } + TSNode nm = cs_field(tp, "name"); + if (ts_node_is_null(nm)) { + continue; + } + if (!first) { + sb_putc(&s->sb, ','); + } + cs_put_text_nows(s, nm); + first = false; + } + cs_kids_end(&k); +} + +static void cs_put_sig(cs_scan_t *s, TSNode params) { + if (ts_node_is_null(params)) { + return; + } + const char *src = s->ctx->source; + bool first = true; + cs_kids_t k = cs_kids(params); + TSNode p; + while (cs_kids_next(&k, &p)) { + /* A parameter is a `parameter` node -- except the `params` one, which + * the grammar leaves inline in the list: its type is the list's own + * "type" field. */ + TSNode ty = {0}; + if (cs_kind_is(p, "parameter")) { + ty = cs_field(p, "type"); + } else { + const char *field = ts_tree_cursor_current_field_name(&k.cur); + if (!field || strcmp(field, "type") != 0) { + continue; + } + ty = p; + } + char norm[CBM_SZ_256]; + if (ts_node_is_null(ty)) { + snprintf(norm, sizeof(norm), "?"); + } else { + uint32_t a = ts_node_start_byte(ty); + uint32_t b = ts_node_end_byte(ty); + cbm_doclink_cs_norm_type(src + a, (size_t)(b - a), norm, sizeof(norm)); + } + if (!first) { + sb_putc(&s->sb, '|'); + } + sb_puts(&s->sb, norm); + first = false; + } + cs_kids_end(&k); +} + +static bool cs_has_child_kind(TSNode n, const char *kind) { + return !ts_node_is_null(cs_child_of_kind(n, kind)); +} + +static char cs_type_kind(const char *k) { + if (strcmp(k, "class_declaration") == 0) { + return 'c'; + } + if (strcmp(k, "struct_declaration") == 0) { + return 's'; + } + if (strcmp(k, "interface_declaration") == 0) { + return 'i'; + } + if (strcmp(k, "enum_declaration") == 0) { + return 'e'; + } + if (strcmp(k, "record_declaration") == 0) { + return 'r'; /* a record class; cs_decl_kind tells a `record struct` apart */ + } + if (strcmp(k, "record_struct_declaration") == 0) { + return 't'; + } + if (strcmp(k, "delegate_declaration") == 0) { + return 'd'; + } + return 0; +} + +/* True when the anonymous token `word` is a direct child of `n`, or sits in + * one of its `modifier` children (`partial`, `static`). */ +static bool cs_has_word(TSNode n, const char *word) { + bool found = false; + uint64_t steps = 0; + cs_kids_t k = cs_kids(n); + TSNode c; + while (!found && cs_kids_next(&k, &c)) { + steps++; + const char *t = ts_node_type(c); + if (!ts_node_is_named(c)) { + found = strcmp(t, word) == 0; + } else if (strcmp(t, "modifier") == 0 && ts_node_child_count(c) > 0) { + found = strcmp(ts_node_type(ts_node_child(c, 0)), word) == 0; + } + } + cs_kids_end(&k); + cs_cost_add(steps, 0); + return found; +} + +/* The kind of a type declaration node: c class, s struct, i interface, e + * enum, r record class, t record struct, d delegate; 0 for any other node. */ +static char cs_decl_kind(TSNode decl) { + char kind = cs_type_kind(ts_node_type(decl)); + return (kind == 'r' && cs_has_word(decl, "struct")) ? 't' : kind; +} + +/* ── Tokens: the block structure the tree lost ─────────────────────── + * + * Error recovery closes blocks early and late, reports a block namespace as + * a file-scoped one, or gives up on a declaration and leaves its header as + * loose tokens in an error node. On the C# bench corpus 1,805 of 32,686 files + * have parse errors, and in 230 of them a namespace or type has a tree extent + * that is not its brace extent (the API reference files among them) -- which + * moves every declaration after the error into the wrong namespace or out of + * its outer type. A file whose tree has errors therefore takes its nesting + * from the text instead: the braces say where blocks begin and end, the + * declaration keywords say what the blocks are. The parser's own tokens + * cannot serve: once it loses the thread inside a string it lexes code as + * string content. So the braces are scanned here, with the lexical grammar a + * brace scanner needs -- comments, character and string literals in all + * their forms, interpolation holes, preprocessor branches. On the 30,881 + * error-free corpus files this scanner yields exactly the parser's brace + * tokens. Where the braces still do not pair up, the rest of the file is not + * placed at all. */ + +static uint32_t cs_row_of(cs_scan_t *s, uint32_t pos) { + const char *src = s->ctx->source; + if (pos < s->row_pos) { + s->row_pos = 0; + s->row = 0; + } + for (uint32_t i = s->row_pos; i < pos; i++) { + s->row += src[i] == '\n'; + } + s->row_pos = pos; + return s->row; +} + +/* A new entry of the brace list; its index, or CBM_NOT_FOUND. */ +static int cs_brace_push(cs_scan_t *s, uint32_t pos, char kind) { + if (s->failed) { + return CBM_NOT_FOUND; + } + if (s->nbraces >= s->cap_braces) { + int ncap = s->cap_braces ? s->cap_braces * PAIR_LEN : CS_BRACES_INIT; + cs_brace_t *grown = (cs_brace_t *)cs_tmp_alloc(s, (size_t)ncap * sizeof(*grown)); + if (!grown) { + return CBM_NOT_FOUND; + } + if (s->nbraces > 0) { + memcpy(grown, s->braces, (size_t)s->nbraces * sizeof(*grown)); + } + s->braces = grown; + s->cap_braces = ncap; + } + s->braces[s->nbraces] = (cs_brace_t){.pos = pos, + .row = cs_row_of(s, pos), + .match = CBM_NOT_FOUND, + .alias = CBM_NOT_FOUND, + .kind = kind}; + return s->nbraces++; +} + +/* Brace `b` is open now: it goes on top of the open braces. */ +static bool cs_open_push(cs_scan_t *s, int b) { + if (s->nopen >= s->cap_open) { + int ncap = s->cap_open ? s->cap_open * PAIR_LEN : CS_HEADS_INIT; + cs_open_t *grown = (cs_open_t *)cs_tmp_alloc(s, (size_t)ncap * sizeof(*grown)); + if (!grown) { + return false; + } + if (s->nopen > 0) { + memcpy(grown, s->open, (size_t)s->nopen * sizeof(*grown)); + } + s->open = grown; + s->cap_open = ncap; + } + s->open[s->nopen] = (cs_open_t){.brace = b, .below = s->top}; + s->top = s->nopen++; + s->sp++; + return true; +} + +static void cs_mark_untrusted_at(cs_scan_t *s, uint32_t pos, uint32_t row) { + if (pos < s->untrusted) { + s->untrusted = pos; + s->untrusted_row = row; + } +} + +static void cs_mark_untrusted(cs_scan_t *s, int brace) { + cs_mark_untrusted_at(s, s->braces[brace].pos, s->braces[brace].row); +} + +/* The scanner reports a brace: pair it. */ +static void cs_lex_brace(void *ud, uint32_t pos, bool open) { + cs_scan_t *s = (cs_scan_t *)ud; + int b = cs_brace_push(s, pos, open ? '{' : '}'); + if (b < 0) { + return; + } + if (open) { + (void)cs_open_push(s, b); + } else if (s->top < 0) { + cs_mark_untrusted(s, b); /* closes nothing */ + } else { + int o = s->open[s->top].brace; + s->top = s->open[s->top].below; + s->sp--; + if (s->braces[o].match < 0) { + s->braces[o].match = b; /* the first branch's close stands */ + } + s->braces[b].match = o; + } + s->braces[b].depth = s->sp; +} + +/* A later branch of a conditional ended. Where it leaves the same blocks + * open as the first one did, its open braces stand in for the first + * branch's (`class X : A {` / `#else` / `class X : B {` share one closing + * brace). Where the branches disagree the first one stands alone: a file + * can balance per configuration only (`#if A {` ... `#if A }`), and the + * braces left over at the end say whether this one does. + * + * Only entries opened in the current branch may acquire aliases. The first + * branch can replace a deep pre-existing stack; every empty later branch + * starts from that old stack. Walking it again would cost depth per branch. */ +static void cs_branch_merge(cs_scan_t *s, const cs_pp_t *f) { + if (s->sp != f->n_end1) { + return; + } + int a = s->top; + int b = f->end1; + while (a != b && a >= f->first_open && b >= 0) { + s->cost_steps++; + s->braces[s->open[a].brace].alias = s->open[b].brace; + a = s->open[a].below; + b = s->open[b].below; + } +} + +/* The scanner reports #if / #else (or #elif) / #endif. Every branch starts + * from the nesting the #if started from; after the #endif the first + * branch's result stands. */ +static void cs_lex_branch(void *ud, uint32_t pos, int what) { + cs_scan_t *s = (cs_scan_t *)ud; + int mark = cs_brace_push(s, pos, '#'); + if (mark < 0) { + return; + } + if (what == CS_PP_IF) { + if (s->npp >= CS_PP_MAX) { + cs_mark_untrusted(s, mark); + } else { + s->pp[s->npp++] = + (cs_pp_t){.at_if = s->top, .n_if = s->sp, .first_open = s->nopen, .mark = mark}; + } + } else if (s->npp > 0) { + cs_pp_t *f = &s->pp[s->npp - SKIP_ONE]; + if (f->has_end1) { + cs_branch_merge(s, f); + } else if (what == CS_PP_ELSE) { + f->end1 = s->top; + f->n_end1 = s->sp; + f->has_end1 = true; + } + if (what == CS_PP_ELSE) { + s->top = f->at_if; + s->sp = f->n_if; + f->first_open = s->nopen; + } else if (what == CS_PP_ENDIF) { + if (f->has_end1) { + s->top = f->end1; + s->sp = f->n_end1; + } + s->npp--; + } + } + s->braces[mark].depth = s->sp; +} + +/* The scanner reports a declaration keyword; `partial` when the word before + * it is that modifier. */ +static void cs_lex_keyword(void *ud, uint32_t kw_end, char kind, bool partial) { + cs_scan_t *s = (cs_scan_t *)ud; + if (s->failed) { + return; + } + if (s->nheads >= s->cap_heads) { + int ncap = s->cap_heads ? s->cap_heads * PAIR_LEN : CS_HEADS_INIT; + cs_head_t *grown = (cs_head_t *)cs_tmp_alloc(s, (size_t)ncap * sizeof(*grown)); + if (!grown) { + return; + } + if (s->nheads > 0) { + memcpy(grown, s->heads, (size_t)s->nheads * sizeof(*grown)); + } + s->heads = grown; + s->cap_heads = ncap; + } + s->heads[s->nheads++] = + (cs_head_t){.end = kw_end, .row = cs_row_of(s, kw_end), .kind = kind, .partial = partial}; +} + +/* --- the scanner ---------------------------------------------------- */ + +typedef struct { + const char *src; + uint32_t n; + void *ud; + uint64_t steps; /* positions looked at */ + int nest; /* interpolation holes the scan is inside of */ + uint32_t stop_pos; /* where the scan gave up (`stopped`) */ + bool stopped; /* holes nested deeper than CS_LEX_MAX_NEST: the rest is not read */ + bool prev_sep; /* the previous token was ':' or ',' */ + bool prev_partial; /* the previous token was `partial` (or `partial record`) */ +} cs_lex_t; + +static bool cs_lex_word(unsigned char c) { + return isalnum(c) || c == '_' || c >= 0x80; +} + +/* The declaration a keyword starts: N for `namespace`, a type kind, or 0. */ +static char cs_keyword_kind(const char *text, uint32_t len) { + static const struct { + const char *word; + char kind; + } words[] = {{"namespace", 'N'}, {"class", 'c'}, {"struct", 's'}, + {"interface", 'i'}, {"enum", 'e'}, {"record", 'r'}}; + for (size_t i = 0; i < sizeof(words) / sizeof(words[0]); i++) { + if (strlen(words[i].word) == len && memcmp(words[i].word, text, len) == 0) { + return words[i].kind; + } + } + return 0; +} + +static uint32_t cs_lex_code(cs_lex_t *lx, uint32_t i, bool hole); + +/* The code of an interpolation hole from src[i]: just past the brace that + * closes it. A hole can hold the next interpolated string, and each such + * level costs stack: past CS_LEX_MAX_NEST -- deeper than any program -- the + * scan stops for good, and what follows in the file is not placed. */ +static uint32_t cs_lex_hole(cs_lex_t *lx, uint32_t i) { + if (lx->nest >= CS_LEX_MAX_NEST) { + if (!lx->stopped) { + lx->stopped = true; + lx->stop_pos = i; + } + return lx->n; + } + lx->nest++; + uint32_t end = cs_lex_code(lx, i, true); + lx->nest--; + return end; +} + +/* Length of the run of `c` at src[i]. */ +static uint32_t cs_lex_run(cs_lex_t *lx, uint32_t i, char c) { + uint32_t j = i; + while (j < lx->n && lx->src[j] == c) { + j++; + } + lx->steps += (uint64_t)(j - i) + SKIP_ONE; + return j - i; +} + +/* 'x', '\n', 'A': past the literal, or past the lone quote when it is + * not one. */ +static uint32_t cs_lex_char(const cs_lex_t *lx, uint32_t i) { + uint32_t j = i + SKIP_ONE; + if (j < lx->n && lx->src[j] == '\\') { + j += PAIR_LEN; + } + while (j < lx->n && lx->src[j] != '\'' && lx->src[j] != '\n' && j - i < CS_CHAR_LITERAL_MAX) { + j++; + } + return (j < lx->n && lx->src[j] == '\'') ? j + SKIP_ONE : i + SKIP_ONE; +} + +/* "..." with backslash escapes; an unterminated one ends with its line. */ +static uint32_t cs_lex_string(const cs_lex_t *lx, uint32_t i) { + uint32_t j = i + SKIP_ONE; + while (j < lx->n) { + char c = lx->src[j]; + if (c == '\\') { + j += PAIR_LEN; + } else if (c == '"') { + return j + SKIP_ONE; + } else if (c == '\n') { + return j; + } else { + j++; + } + } + return lx->n; +} + +/* @"..." : a quote is written twice. `i` is at the opening quote. */ +static uint32_t cs_lex_verbatim(const cs_lex_t *lx, uint32_t i) { + uint32_t j = i + SKIP_ONE; + while (j < lx->n) { + if (lx->src[j] != '"') { + j++; + } else if (j + SKIP_ONE < lx->n && lx->src[j + SKIP_ONE] == '"') { + j += PAIR_LEN; + } else { + return j + SKIP_ONE; + } + } + return lx->n; +} + +/* """...""" (`quotes` >= 3 of them): ends with a run of at least as many. + * With `dollars`, a run of that many `{` opens a hole of code. */ +static uint32_t cs_lex_raw(cs_lex_t *lx, uint32_t i, uint32_t quotes, uint32_t dollars) { + uint32_t j = i + quotes; + while (j < lx->n) { + char c = lx->src[j]; + if (c == '"') { + uint32_t run = cs_lex_run(lx, j, '"'); + if (run >= quotes) { + return j + run; + } + j += run; + } else if (dollars > 0 && c == '{') { + uint32_t run = cs_lex_run(lx, j, '{'); + j += run; + if (run >= dollars) { + j = cs_lex_hole(lx, j); + j += cs_lex_run(lx, j, '}'); /* the rest of the closing run */ + } + } else { + j++; + } + } + return lx->n; +} + +/* $"..." / $@"..." : text with {holes} of code; {{ and }} are literal braces. + * `i` is at the opening quote. */ +static uint32_t cs_lex_interpolated(cs_lex_t *lx, uint32_t i, bool verbatim) { + uint32_t j = i + SKIP_ONE; + while (j < lx->n) { + char c = lx->src[j]; + if (c == '"') { + if (verbatim && j + SKIP_ONE < lx->n && lx->src[j + SKIP_ONE] == '"') { + j += PAIR_LEN; + continue; + } + return j + SKIP_ONE; + } + if (c == '\\' && !verbatim) { + j += PAIR_LEN; + } else if (c == '{' || c == '}') { + if (j + SKIP_ONE < lx->n && lx->src[j + SKIP_ONE] == c) { + j += PAIR_LEN; + } else if (c == '{') { + j = cs_lex_hole(lx, j + SKIP_ONE); + } else { + j++; + } + } else if (c == '\n' && !verbatim) { + return j; + } else { + j++; + } + } + return lx->n; +} + +/* A literal that starts with a quote, `@`, or `$` at src[i]; returns i itself + * when there is none there. A run of `$` is measured once: where it starts + * no literal the scan goes on behind it (or at its last `$`, when that one + * starts an ordinary interpolated string), never at its second character. */ +static uint32_t cs_lex_literal(cs_lex_t *lx, uint32_t i) { + const char *src = lx->src; + uint32_t n = lx->n; + char c = src[i]; + if (c == '"') { + uint32_t q = cs_lex_run(lx, i, '"'); + if (q >= CS_RAW_QUOTES) { + return cs_lex_raw(lx, i, q, 0); + } + return q == PAIR_LEN ? i + PAIR_LEN : cs_lex_string(lx, i); + } + if (c == '@' && i + SKIP_ONE < n && src[i + SKIP_ONE] == '"') { + return cs_lex_verbatim(lx, i + SKIP_ONE); + } + if (c == '@' && i + PAIR_LEN < n && src[i + SKIP_ONE] == '$' && src[i + PAIR_LEN] == '"') { + return cs_lex_interpolated(lx, i + PAIR_LEN, true); + } + if (c == '$') { + uint32_t d = cs_lex_run(lx, i, '$'); + uint32_t j = i + d; + if (j < n && src[j] == '"') { + uint32_t q = cs_lex_run(lx, j, '"'); + if (q >= CS_RAW_QUOTES) { + return cs_lex_raw(lx, j, q, d); + } + return (d == SKIP_ONE) ? cs_lex_interpolated(lx, j, false) : j - SKIP_ONE; + } + if (j + SKIP_ONE < n && src[j] == '@' && src[j + SKIP_ONE] == '"') { + return (d == SKIP_ONE) ? cs_lex_interpolated(lx, j + SKIP_ONE, true) : j - SKIP_ONE; + } + return j; + } + return i; +} + +static bool cs_lex_is(const char *src, uint32_t w, uint32_t len, const char *word) { + return strlen(word) == len && memcmp(src + w, word, len) == 0; +} + +/* A `#` directive at the start of a line: reports conditional branches and + * returns the end of the line. */ +static uint32_t cs_lex_directive(cs_lex_t *lx, uint32_t i, bool hole) { + const char *src = lx->src; + uint32_t j = i + SKIP_ONE; + while (j < lx->n && (src[j] == ' ' || src[j] == '\t')) { + j++; + } + uint32_t w = j; + while (j < lx->n && isalpha((unsigned char)src[j])) { + j++; + } + uint32_t len = j - w; + if (!hole) { + if (cs_lex_is(src, w, len, "if")) { + cs_lex_branch(lx->ud, i, CS_PP_IF); + } else if (cs_lex_is(src, w, len, "else") || cs_lex_is(src, w, len, "elif")) { + cs_lex_branch(lx->ud, i, CS_PP_ELSE); + } else if (cs_lex_is(src, w, len, "endif")) { + cs_lex_branch(lx->ud, i, CS_PP_ENDIF); + } + } + while (j < lx->n && src[j] != '\n') { + j++; + } + return j; +} + +/* A word at src[i] (an identifier, keyword or number, with an optional `@`): + * reports a declaration keyword and returns its end; i when there is none. + * *partial is set when the word leaves the `partial` modifier standing for + * the next keyword: `partial` itself, or `record` after it (`partial record + * struct`). */ +static uint32_t cs_lex_identifier(cs_lex_t *lx, uint32_t i, bool hole, bool *partial) { + static const char modifier[] = "partial"; + const char *src = lx->src; + bool verbatim = src[i] == '@'; + uint32_t a = verbatim ? i + SKIP_ONE : i; + uint32_t b = a; + while (b < lx->n && cs_lex_word((unsigned char)src[b])) { + b++; + } + if (b == a) { + return i; + } + char kind = (verbatim || hole) ? 0 : cs_keyword_kind(src + a, b - a); + if (kind && !lx->prev_sep) { + cs_lex_keyword(lx->ud, b, kind, lx->prev_partial); + } + *partial = !verbatim && !hole && + ((kind == 'r' && lx->prev_partial) || + (b - a == sizeof(modifier) - SKIP_ONE && memcmp(src + a, modifier, b - a) == 0)); + return b; +} + +/* Past a comment at src[i], or i when there is none. */ +static uint32_t cs_lex_comment(const cs_lex_t *lx, uint32_t i) { + const char *src = lx->src; + uint32_t n = lx->n; + if (src[i] != '/' || i + SKIP_ONE >= n) { + return i; + } + if (src[i + SKIP_ONE] == '/') { + while (i < n && src[i] != '\n') { + i++; + } + return i; + } + if (src[i + SKIP_ONE] == '*') { + i += PAIR_LEN; + while (i + SKIP_ONE < n && !(src[i] == '*' && src[i + SKIP_ONE] == '/')) { + i++; + } + return i + PAIR_LEN <= n ? i + PAIR_LEN : n; + } + return i; +} + +/* Scan code from src[i]. In a `hole` (the code of an interpolation) nothing + * is reported and the scan returns just past the brace that closes it. */ +static uint32_t cs_lex_code(cs_lex_t *lx, uint32_t i, bool hole) { + const char *src = lx->src; + uint32_t n = lx->n; + int depth = 0; + bool line_start = !hole; + while (i < n) { + unsigned char c = (unsigned char)src[i]; + lx->steps++; + if (isspace(c)) { + line_start = line_start || c == '\n'; + i++; + continue; + } + if (c == '#' && line_start) { + i = cs_lex_directive(lx, i, hole); + continue; + } + line_start = false; + uint32_t e = cs_lex_comment(lx, i); + if (e > i) { + i = e; + continue; + } + bool sep = c == ':' || c == ','; + bool partial = false; + if (c == '\'') { + e = cs_lex_char(lx, i); + } else if (c == '{' || c == '}') { + if (hole && c == '}' && depth == 0) { + return i + SKIP_ONE; + } + if (hole) { + depth += c == '{' ? SKIP_ONE : -SKIP_ONE; + } else { + cs_lex_brace(lx->ud, i, c == '{'); + } + e = i + SKIP_ONE; + } else { + e = (c == '"' || c == '$' || c == '@') ? cs_lex_literal(lx, i) : i; + if (e == i && (cs_lex_word(c) || c == '@')) { + e = cs_lex_identifier(lx, i, hole, &partial); + } + if (e == i) { + e = i + SKIP_ONE; + } + } + lx->prev_sep = sep; + lx->prev_partial = partial; + i = e; + } + return n; +} + +/* Scan the file's braces and declaration keywords, pair the braces. The + * first brace without a partner (or conditional whose branches disagree) + * starts the untrusted part of the file. */ +static void cs_scan_tokens(cs_scan_t *s) { + cs_lex_t lx = {.src = s->ctx->source, .n = s->root_end_byte, .ud = s}; + (void)cs_lex_code(&lx, 0, false); + s->cost_steps += lx.steps; + if (s->failed) { + return; + } + if (lx.stopped) { + cs_mark_untrusted_at(s, lx.stop_pos, cs_row_of(s, lx.stop_pos)); + } + /* the outermost brace that is still open */ + int unpaired = CBM_NOT_FOUND; + for (int o = s->top; o >= 0; o = s->open[o].below) { + unpaired = s->open[o].brace; + } + if (unpaired >= 0) { + cs_mark_untrusted(s, unpaired); + } + /* a later branch's brace closes where the first branch's does */ + for (int i = 0; i < s->nbraces; i++) { + cs_brace_t *b = &s->braces[i]; + int a = b->alias; + for (int hops = 0; a >= 0 && b->match < 0 && hops < CS_PP_MAX; hops++) { + b->match = s->braces[a].match; + a = s->braces[a].alias; + } + } +} + +/* Index of the first entry of the brace list at or after byte `pos`. */ +static int cs_brace_lower(const cs_scan_t *s, uint32_t pos) { + int lo = 0; + int hi = s->nbraces; + while (lo < hi) { + int mid = lo + ((hi - lo) / PAIR_LEN); + if (s->braces[mid].pos < pos) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + return lo; +} + +/* The open brace AT byte `pos`, or CBM_NOT_FOUND. */ +static int cs_open_brace_at(const cs_scan_t *s, uint32_t pos) { + int b = cs_brace_lower(s, pos); + return (b < s->nbraces && s->braces[b].pos == pos && s->braces[b].kind == '{') ? b + : CBM_NOT_FOUND; +} + +/* Brace nesting depth at byte `pos`. */ +static int cs_depth_at(const cs_scan_t *s, uint32_t pos) { + int i = cs_brace_lower(s, pos); + return i > 0 ? s->braces[i - SKIP_ONE].depth : 0; +} + +/* The extent of the block opened by brace `b`. false when it has no partner. */ +static bool cs_brace_extent(const cs_scan_t *s, int b, uint32_t *end, uint32_t *end_line, + int *inner_depth) { + if (b < 0 || b >= s->nbraces || s->braces[b].kind != '{' || s->braces[b].match < 0) { + return false; + } + const cs_brace_t *close = &s->braces[s->braces[b].match]; + *end = close->pos + SKIP_ONE; + *end_line = close->row + TS_LINE_OFFSET; + *inner_depth = s->braces[b].depth; + return true; +} + +/* ── Text ────────────────────────────────────────────────────────── */ + +/* Past whitespace and comments. */ +static uint32_t cs_skip_space(const char *src, uint32_t i, uint32_t n) { + for (;;) { + while (i < n && isspace((unsigned char)src[i])) { + i++; + } + if (i + SKIP_ONE >= n || src[i] != '/') { + return i; + } + if (src[i + SKIP_ONE] == '/') { + while (i < n && src[i] != '\n') { + i++; + } + } else if (src[i + SKIP_ONE] == '*') { + i += PAIR_LEN; + while (i + SKIP_ONE < n && !(src[i] == '*' && src[i + SKIP_ONE] == '/')) { + i++; + } + i = i + PAIR_LEN <= n ? i + PAIR_LEN : n; + } else { + return i; + } + } +} + +static bool cs_word_char(unsigned char c) { + return isalnum(c) || c == '_' || c >= 0x80; +} + +/* Recovery may omit punctuation before the name node. Start at the actual + * namespace header, not at that recovered node's first byte. */ +static uint32_t cs_namespace_name_start(const cs_scan_t *s, TSNode node) { + static const char keyword[] = "namespace"; + const uint32_t width = (uint32_t)(sizeof(keyword) - SKIP_ONE); + uint32_t a = ts_node_start_byte(node); + uint32_t n = s->root_end_byte; + const char *src = s->ctx->source; + if (a > n || n - a < width || memcmp(src + a, keyword, width) != 0) { + return n; + } + a += width; + uint32_t start = cs_skip_space(src, a, n); + return start > a || (a < n && src[a] == '@') ? start : n; +} + +/* Namespace names have nonempty identifier segments. Keep the scanner's + * existing Unicode-byte support, while allowing trivia between tokens and + * verbatim markers only at segment starts. NULL names remain unplaced items. */ +static char *cs_namespace_name_dup(cs_scan_t *s, uint32_t a, uint32_t b) { + if (b <= a || b > s->root_end_byte || b - a > CS_NAME_MAX) { + return NULL; + } + const char *src = s->ctx->source; + char *out = (char *)cs_tmp_alloc(s, (size_t)(b - a) + SKIP_ONE); + if (!out) { + return NULL; + } + size_t w = 0; + uint32_t i = a; + for (;;) { + i = cs_skip_space(src, i, b); + if (i < b && src[i] == '@') { + i++; + } + if (i == b || !cs_word_char((unsigned char)src[i]) || isdigit((unsigned char)src[i])) { + return NULL; + } + while (i < b && cs_word_char((unsigned char)src[i])) { + out[w++] = src[i++]; + } + i = cs_skip_space(src, i, b); + if (i == b) { + out[w] = '\0'; + return out; + } + if (src[i] != '.') { + return NULL; + } + out[w++] = '.'; + i++; + } +} + +/* Keep malformed dotted candidates until their known delimiter. Dropping an + * invalid file-scoped header would place its declarations in the root. The + * next recorded declaration bounds this scan even when no delimiter exists. */ +static uint32_t cs_namespace_header_end(cs_scan_t *s, uint32_t a, uint32_t stop) { + const char *src = s->ctx->source; + uint32_t i = a; + while (i < stop) { + uint32_t next = cs_skip_space(src, i, stop); + if (next != i) { + i = next; + continue; + } + unsigned char c = (unsigned char)src[i]; + if (!cs_word_char(c) && c != '@' && c != '.') { + break; + } + s->cost_steps++; + i++; + } + return i; +} + +/* End of the identifier starting at src[i] (i itself when there is none); + * `dotted` accepts a qualified name. */ +static uint32_t cs_ident_end(const char *src, uint32_t i, uint32_t n, bool dotted) { + uint32_t e = i; + if (e < n && src[e] == '@') { + e++; + } + uint32_t first = e; + while (e < n && (cs_word_char((unsigned char)src[e]) || + (dotted && src[e] == '.' && e > first && e + SKIP_ONE < n && + cs_word_char((unsigned char)src[e + SKIP_ONE])))) { + e++; + } + if (e == first || isdigit((unsigned char)src[first])) { + return i; + } + return e; +} + +/* Words that follow `class` / `record` ... without being a declared name. */ +static bool cs_not_a_name(const char *name) { + static const char *const words[] = {"class", "struct", "interface", "enum", "record", + "where", "in", "is", "as", "when", + "and", "or", "not", "with", "switch"}; + for (size_t i = 0; i < sizeof(words) / sizeof(words[0]); i++) { + if (strcmp(name, words[i]) == 0) { + return true; + } + } + return false; +} + +/* The type parameters written at src[*pos] (``): their names + * ','-joined, "" when there are none. *pos moves past the list. NULL when the + * list cannot be read. Nothing at or past `stop` is read. */ +static const char *cs_text_tparams(cs_scan_t *s, uint32_t *pos, uint32_t stop) { + const char *src = s->ctx->source; + uint32_t n = s->root_end_byte; + uint32_t i = cs_skip_space(src, *pos, n); + if (i >= n || src[i] != '<') { + return ""; + } + if (i >= stop) { + return NULL; /* the list is the next declaration's */ + } + uint32_t limit = stop - i > CS_TPARAMS_SCAN_MAX ? i + CS_TPARAMS_SCAN_MAX : stop; + char *out = (char *)cs_tmp_alloc(s, (size_t)(limit - i) + SKIP_ONE); + if (!out) { + return NULL; + } + size_t w = 0; + int square = 0; + uint32_t word_s = 0; + uint32_t word_e = 0; /* the last identifier of the current parameter */ + for (uint32_t k = i + SKIP_ONE; k < limit; k++) { + char c = src[k]; + s->cost_steps++; + if (c == '[') { + square++; + } else if (c == ']') { + square--; + } else if (square > 0) { + continue; /* an attribute on the parameter */ + } else if (c == ',' || c == '>') { + if (word_e == word_s) { + return NULL; + } + if (w > 0) { + out[w++] = ','; + } + memcpy(out + w, src + word_s, word_e - word_s); + w += word_e - word_s; + word_s = word_e = 0; + if (c == '>') { + out[w] = '\0'; + *pos = k + SKIP_ONE; + return out; + } + } else if (cs_word_char((unsigned char)c)) { + uint32_t e = cs_ident_end(src, k, limit, false); + if (e == k) { + return NULL; + } + word_s = k; + word_e = e; + k = e - SKIP_ONE; + } else if (!isspace((unsigned char)c)) { + return NULL; + } + } + return NULL; +} + +/* From the end of a type's name and type parameters to the `{` that opens + * its body (returned as a brace index) or the `;` that ends a body-less + * declaration (*bodyless). CBM_NOT_FOUND with *bodyless false when neither is + * found: then this was no declaration. Nothing at or past `stop` is read. */ +static int cs_text_body(cs_scan_t *s, uint32_t from, uint32_t stop, bool *bodyless) { + const char *src = s->ctx->source; + uint32_t limit = from + CS_HEADER_SCAN_MAX < stop ? from + CS_HEADER_SCAN_MAX : stop; + int round = 0; + *bodyless = false; + for (uint32_t i = from; i < limit; i++) { + char c = src[i]; + s->cost_steps++; + if (c == '/' && i + SKIP_ONE < limit && + (src[i + SKIP_ONE] == '/' || src[i + SKIP_ONE] == '*')) { + uint32_t past = cs_skip_space(src, i, limit); + if (past <= i) { + return CBM_NOT_FOUND; + } + i = past - SKIP_ONE; + continue; + } + if (c == '(') { + round++; + } else if (c == ')') { + round--; + } else if (round > 0) { + continue; + } else if (c == ';') { + *bodyless = true; + return CBM_NOT_FOUND; + } else if (c == '{') { + return cs_open_brace_at(s, i); + } else if (c == '}' || c == '=') { + return CBM_NOT_FOUND; + } + } + return CBM_NOT_FOUND; +} + +/* ── Collection ──────────────────────────────────────────────────── */ + +/* A new item; returns its index or CBM_NOT_FOUND when memory ran out. */ +static int cs_item_new(cs_scan_t *s, char tag) { + if (s->failed) { + return CBM_NOT_FOUND; + } + if (s->nitems >= s->cap_items) { + int ncap = s->cap_items ? s->cap_items * PAIR_LEN : CS_ITEMS_INIT; + cs_item_t *grown = (cs_item_t *)cs_tmp_alloc(s, (size_t)ncap * sizeof(*grown)); + if (!grown) { + return CBM_NOT_FOUND; + } + if (s->nitems > 0) { + memcpy(grown, s->items, (size_t)s->nitems * sizeof(*grown)); + } + s->items = grown; + s->cap_items = ncap; + } + cs_item_t *it = &s->items[s->nitems]; + memset(it, 0, sizeof(*it)); + it->tag = tag; + it->owner = CS_OWNER_NONE; + it->brace = CBM_NOT_FOUND; + it->tree_idx = s->nitems; + return s->nitems++; +} + +static int cs_item_of_node(cs_scan_t *s, char tag, TSNode node) { + int i = cs_item_new(s, tag); + if (i >= 0) { + s->items[i].node = node; + s->items[i].start = ts_node_start_byte(node); + s->items[i].line = cs_line(node); + } + return i; +} + +/* True when the first parameter of `params` is written with `this`: an + * extension method. The text before the parameter's type says so, whatever + * shape the grammar gives the modifier. */ +static bool cs_first_param_this(const cs_scan_t *s, TSNode params) { + static const char word[] = "this"; + if (ts_node_is_null(params)) { + return false; + } + TSNode p = cs_child_of_kind(params, "parameter"); + if (ts_node_is_null(p)) { + return false; + } + TSNode ty = cs_field(p, "type"); + const char *src = s->ctx->source; + uint32_t a = ts_node_start_byte(p); + uint32_t b = ts_node_is_null(ty) ? ts_node_end_byte(p) : ts_node_start_byte(ty); + uint32_t wl = (uint32_t)(sizeof(word) - SKIP_ONE); + for (uint32_t i = a; i + wl <= b; i++) { + if (memcmp(src + i, word, wl) == 0 && + (i == a || !cs_word_char((unsigned char)src[i - SKIP_ONE])) && + (i + wl == b || !cs_word_char((unsigned char)src[i + wl]))) { + return true; + } + } + return false; +} + +/* True for a member `using static` brings in: one declared `static` or + * `const`, an enum's member -- and no extension method. */ +static bool cs_member_static(const cs_scan_t *s, TSNode decl, TSNode params) { + if (cs_kind_is(decl, "enum_member_declaration")) { + return true; + } + if (!cs_has_word(decl, "static") && !cs_has_word(decl, "const")) { + return false; + } + return !cs_first_param_this(s, params); +} + +static void cs_member_new(cs_scan_t *s, TSNode decl, char kind, bool explicit_impl, bool is_static, + int owner, TSNode name, TSNode tparams, TSNode params) { + if (ts_node_is_null(name)) { + return; + } + int i = cs_item_of_node(s, 'M', decl); + if (i < 0) { + return; + } + cs_item_t *it = &s->items[i]; + it->kind = kind; + it->explicit_impl = explicit_impl; + it->is_static = is_static; + it->owner = owner; + it->name = name; + it->tparams = tparams; + it->params = params; +} + +/* A member that has no identifier for a name: an operator (`name` is its + * token: +, ==, true, implicit ...) or an indexer (`this`). The graph has no + * node for either; the record says that the type declares one. */ +static void cs_member_unnamed(cs_scan_t *s, TSNode decl, char kind, int owner, const char *name, + TSNode params) { + int i = cs_item_of_node(s, 'M', decl); + if (i < 0) { + return; + } + cs_item_t *it = &s->items[i]; + it->kind = kind; + it->owner = owner; + it->text_name = name; + it->params = params; +} + +/* Every variable_declarator name under a field / event field declaration. */ +static void cs_collect_declarators(cs_scan_t *s, TSNode decl, char kind, int owner) { + TSNode null_node = {0}; + /* Modifiers belong to this declaration and are shared by all its names. */ + bool is_static = cs_member_static(s, decl, null_node); + cs_kids_t outer = cs_kids(decl); + TSNode vd; + while (cs_kids_next_named(&outer, &vd)) { + if (!cs_kind_is(vd, "variable_declaration")) { + continue; + } + cs_kids_t inner = cs_kids(vd); + TSNode d; + while (cs_kids_next_named(&inner, &d)) { + if (cs_kind_is(d, "variable_declarator")) { + cs_member_new(s, decl, kind, false, is_static, owner, cs_field(d, "name"), + null_node, null_node); + } + } + cs_kids_end(&inner); + } + cs_kids_end(&outer); +} + +/* A member whose header did not parse: an error node among its own parts, or + * (for a callable) an error anywhere outside its body. Its name and signature + * cannot be trusted then -- `public safe extern int M();` comes back as a + * method named `extern`. */ +static bool cs_header_broken(TSNode decl, bool callable) { + TSNode body = callable ? cs_field(decl, "body") : (TSNode){0}; + bool broken = false; + cs_kids_t k = cs_kids(decl); + TSNode ch; + while (!broken && cs_kids_next(&k, &ch)) { + if (!ts_node_is_null(body) && ts_node_eq(ch, body)) { + continue; + } + broken = cs_kind_is(ch, "ERROR") || ts_node_is_missing(ch) || + (callable && ts_node_has_error(ch)); + } + cs_kids_end(&k); + return broken; +} + +static void cs_collect_member(cs_scan_t *s, TSNode c, const char *k, int owner) { + TSNode null_node = {0}; + bool is_operator = strcmp(k, "operator_declaration") == 0; + bool is_conversion = strcmp(k, "conversion_operator_declaration") == 0; + bool is_indexer = strcmp(k, "indexer_declaration") == 0; + bool callable = strcmp(k, "method_declaration") == 0 || + strcmp(k, "constructor_declaration") == 0 || is_operator || is_conversion; + bool member = callable || is_indexer || strcmp(k, "property_declaration") == 0 || + strcmp(k, "field_declaration") == 0 || + strcmp(k, "event_field_declaration") == 0 || + strcmp(k, "event_declaration") == 0 || strcmp(k, "enum_member_declaration") == 0; + if (member && ts_node_has_error(c) && cs_header_broken(c, callable)) { + if (owner >= 0) { + s->items[owner].broken = true; /* a member of it is hidden */ + } + return; + } + if (strcmp(k, "method_declaration") == 0) { + TSNode params = cs_field(c, "parameters"); + cs_member_new(s, c, 'c', cs_has_child_kind(c, "explicit_interface_specifier"), + cs_member_static(s, c, params), owner, cs_field(c, "name"), cs_type_params(c), + params); + } else if (strcmp(k, "constructor_declaration") == 0) { + TSNode params = cs_field(c, "parameters"); + cs_member_new(s, c, 'c', false, cs_member_static(s, c, params), owner, cs_field(c, "name"), + null_node, params); + } else if (strcmp(k, "property_declaration") == 0) { + cs_member_new(s, c, 'p', cs_has_child_kind(c, "explicit_interface_specifier"), + cs_member_static(s, c, null_node), owner, cs_field(c, "name"), null_node, + null_node); + } else if (strcmp(k, "field_declaration") == 0) { + cs_collect_declarators(s, c, 'v', owner); + } else if (strcmp(k, "event_field_declaration") == 0) { + cs_collect_declarators(s, c, 'e', owner); + } else if (strcmp(k, "event_declaration") == 0) { + cs_member_new(s, c, 'e', cs_has_child_kind(c, "explicit_interface_specifier"), + cs_member_static(s, c, null_node), owner, cs_field(c, "name"), null_node, + null_node); + } else if (strcmp(k, "enum_member_declaration") == 0) { + cs_member_new(s, c, 'v', false, cs_member_static(s, c, null_node), owner, + cs_field(c, "name"), null_node, null_node); + } else if (is_operator) { + /* an anonymous token's type is its text */ + TSNode op = cs_field(c, "operator"); + if (!ts_node_is_null(op)) { + cs_member_unnamed(s, c, 'o', owner, ts_node_type(op), cs_field(c, "parameters")); + } + } else if (is_conversion) { + cs_member_unnamed(s, c, 'o', owner, cs_has_word(c, "implicit") ? "implicit" : "explicit", + cs_field(c, "parameters")); + } else if (is_indexer) { + cs_member_unnamed(s, c, 'x', owner, "this", cs_field(c, "parameters")); + } +} + +/* A node whose children are being collected. */ +typedef struct { + cs_kids_t kids; + int owner; /* the item of the type whose members the node holds, or CS_OWNER_* */ +} cs_walk_t; + +typedef struct { + cs_walk_t *frames; + int count; + int cap; +} cs_walk_stack_t; + +/* Go into `node`: its children are collected next. false when memory ran out. */ +static bool cs_walk_push(cs_scan_t *s, cs_walk_stack_t *w, TSNode node, int owner) { + if (ts_node_is_null(node)) { + return true; + } + if (w->count >= w->cap) { + int ncap = w->cap ? w->cap * PAIR_LEN : CS_HEADS_INIT; + cs_walk_t *grown = (cs_walk_t *)cs_tmp_alloc(s, (size_t)ncap * sizeof(*grown)); + if (!grown) { + return false; + } + if (w->count > 0) { + memcpy(grown, w->frames, (size_t)w->count * sizeof(*grown)); + } + w->frames = grown; + w->cap = ncap; + } + w->frames[w->count++] = (cs_walk_t){.kids = cs_kids(node), .owner = owner}; + return true; +} + +/* One child `c` of a collected node: record what it declares and go into it + * where declarations can sit. false when memory ran out. */ +static bool cs_collect_child(cs_scan_t *s, cs_walk_stack_t *w, TSNode c, int owner) { + const char *k = ts_node_type(c); + if (strcmp(k, "ERROR") == 0) { + if (owner >= 0) { + s->items[owner].broken = true; + } + return cs_walk_push(s, w, c, CS_OWNER_LEXICAL); + } + if (strncmp(k, "preproc_", 8) == 0 || strcmp(k, "declaration_list") == 0 || + strcmp(k, "enum_member_declaration_list") == 0) { + return cs_walk_push(s, w, c, owner); + } + if (strcmp(k, "using_directive") == 0) { + (void)cs_item_of_node(s, 'U', c); + return true; + } + bool file_scoped = strcmp(k, "file_scoped_namespace_declaration") == 0; + if (file_scoped || strcmp(k, "namespace_declaration") == 0) { + int ni = cs_item_of_node(s, file_scoped ? 'F' : 'N', c); + if (ni < 0) { + return false; + } + s->items[ni].name = cs_field(c, "name"); + /* a file-scoped namespace holds nothing itself -- unless it is a + * block namespace recovery could not parse, whose declarations + * then sit in an error node under it */ + TSNode body = file_scoped ? (TSNode){0} : cs_field(c, "body"); + return cs_walk_push(s, w, ts_node_is_null(body) ? c : body, CS_OWNER_NONE); + } + char tk = cs_decl_kind(c); + if (tk) { + int ti = cs_item_of_node(s, 'T', c); + if (ti < 0) { + return false; + } + s->items[ti].kind = tk; + s->items[ti].partial = cs_has_word(c, "partial"); + s->items[ti].name = cs_field(c, "name"); + s->items[ti].tparams = cs_type_params(c); + return tk == 'd' || cs_walk_push(s, w, cs_field(c, "body"), ti); + } + if (owner != CS_OWNER_NONE) { + cs_collect_member(s, c, k, owner); + } + return true; +} + +/* Collect the declarations under `root` in document order. `owner` is the + * item of the type whose members a node holds (CS_OWNER_NONE outside a type). + * Preprocessor blocks and error nodes are looked into: a declaration that + * parsed is a declaration wherever recovery left it, and its placement is + * checked against the braces afterwards. What sits in an error node belongs + * to whichever block's braces hold it. + * + * The walk keeps its own stack: how deep declarations nest is the file's + * choice, and must not be this thread's stack depth. */ +static void cs_collect(cs_scan_t *s, TSNode root, int owner) { + cs_walk_stack_t w = {0}; + bool ok = !s->failed && cs_walk_push(s, &w, root, owner); + while (w.count > 0) { + /* a child may push a frame, which moves the array: no pointer into it + * is kept across cs_collect_child */ + TSNode c; + if (!ok || s->failed || !cs_kids_next_named(&w.frames[w.count - SKIP_ONE].kids, &c)) { + cs_kids_end(&w.frames[w.count - SKIP_ONE].kids); + w.count--; + continue; + } + ok = cs_collect_child(s, &w, c, w.frames[w.count - SKIP_ONE].owner); + } +} + +/* ── Declarations the tree has no node for ───────────────────────── */ + +static int cs_u32_cmp(const void *a, const void *b) { + uint32_t x = *(const uint32_t *)a; + uint32_t y = *(const uint32_t *)b; + return (x > y) - (x < y); +} + +static int cs_item_start_cmp(const void *a, const void *b) { + const cs_item_t *x = (const cs_item_t *)a; + const cs_item_t *y = (const cs_item_t *)b; + if (x->start != y->start) { + return x->start < y->start ? -1 : 1; + } + /* declarators of one field declaration keep their order */ + return (x->tree_idx > y->tree_idx) - (x->tree_idx < y->tree_idx); +} + +/* Read the declaration a keyword starts from the text and add it as an item, + * unless the tree already has a node for it (`parsed`: the name positions of + * the tree's namespaces and types). A header ends where the next declaration + * keyword stands (`stop`): reading past it would read the file once per + * keyword. */ +static void cs_text_item(cs_scan_t *s, const cs_head_t *h, uint32_t stop, const uint32_t *parsed, + int nparsed) { + const char *src = s->ctx->source; + uint32_t n = s->root_end_byte; + uint32_t a = cs_skip_space(src, h->end, n); + if (a == h->end) { + return; /* the keyword runs into something: not a declaration */ + } + bool is_ns = h->kind == 'N'; + if (bsearch(&a, parsed, (size_t)nparsed, sizeof(uint32_t), cs_u32_cmp)) { + return; + } + char *name = NULL; + int brace = CBM_NOT_FOUND; + const char *tparams = ""; + char tag = 'T'; + if (is_ns) { + uint32_t after = cs_namespace_header_end(s, a, stop); + name = cs_namespace_name_dup(s, a, after); + if (name && src[a] != '@' && cs_not_a_name(name)) { + name = NULL; /* bare keywords stay unplaced; verbatim identifiers are names */ + } + if (after < n && src[after] == '{') { + brace = cs_open_brace_at(s, after); + tag = 'N'; + } else if (after < n && src[after] == ';') { + tag = 'F'; + } else { + return; + } + if (tag == 'N' && brace < 0) { + return; + } + /* An invalid name must still push an unplaced namespace frame. */ + } else { + uint32_t b = cs_ident_end(src, a, n, false); + name = cs_ident_dup(s, a, b); + if (!name || cs_not_a_name(name)) { + return; + } + uint32_t pos = b; + tparams = cs_text_tparams(s, &pos, stop); + bool bodyless = false; + brace = tparams ? cs_text_body(s, pos, stop, &bodyless) : CBM_NOT_FOUND; + if (!tparams || (brace < 0 && !bodyless)) { + return; + } + } + int i = cs_item_new(s, tag); + if (i < 0) { + return; + } + cs_item_t *it = &s->items[i]; + it->from_text = true; + it->tree_idx = CS_ITEM_TEXT; + it->partial = h->partial; + it->kind = is_ns ? 0 : h->kind; + it->text_name = name; + it->text_tparams = tparams; + it->brace = brace; + it->start = a; /* the name: after the modifiers, inside the same braces */ + it->line = h->row + TS_LINE_OFFSET; + it->broken = true; +} + +/* Add the declarations only the token stream shows and put all items in + * document order. */ +static void cs_add_text_items(cs_scan_t *s) { + if (s->failed || s->nheads == 0) { + return; + } + int tree_items = s->nitems; + uint32_t *parsed = + (uint32_t *)cs_tmp_alloc(s, (size_t)(tree_items + SKIP_ONE) * sizeof(uint32_t)); + if (!parsed) { + return; + } + int nparsed = 0; + for (int i = 0; i < tree_items; i++) { + const cs_item_t *it = &s->items[i]; + if (it->tag == 'N' || it->tag == 'F') { + parsed[nparsed++] = cs_namespace_name_start(s, it->node); + } else if (it->tag == 'T' && !ts_node_is_null(it->name)) { + parsed[nparsed++] = ts_node_start_byte(it->name); + } + } + qsort(parsed, (size_t)nparsed, sizeof(uint32_t), cs_u32_cmp); + for (int h = 0; h < s->nheads && !s->failed; h++) { + uint32_t stop = h + SKIP_ONE < s->nheads ? s->heads[h + SKIP_ONE].end : s->root_end_byte; + cs_text_item(s, &s->heads[h], stop, parsed, nparsed); + } + if (s->nitems > tree_items) { + qsort(s->items, (size_t)s->nitems, sizeof(cs_item_t), cs_item_start_cmp); + } +} + +/* ── Emission ────────────────────────────────────────────────────── */ + +static void cs_put_bases(cs_scan_t *s, TSNode type_decl, char kind) { + if (kind == 'e' || kind == 'd') { + return; /* an enum's base is its underlying integral type */ + } + TSNode bl = cs_child_of_kind(type_decl, "base_list"); + if (ts_node_is_null(bl)) { + return; + } + bool first = true; + cs_kids_t k = cs_kids(bl); + TSNode b; + while (cs_kids_next_named(&k, &b)) { + if (cs_kind_is(b, "primary_constructor_base_type")) { + TSNode ty = cs_field(b, "type"); + if (ts_node_is_null(ty) && ts_node_named_child_count(b) > 0) { + ty = ts_node_named_child(b, 0); + } + b = ty; + } else if (cs_kind_is(b, "argument_list") || cs_kind_is(b, "comment")) { + continue; + } + if (ts_node_is_null(b)) { + continue; + } + if (!first) { + sb_putc(&s->sb, '|'); + } + cs_put_text_nows(s, b); + first = false; + } + cs_kids_end(&k); +} + +/* What an M record says of its member. */ +typedef struct { + uint32_t line; + char kind; + bool explicit_impl; + bool is_static; + int type; /* the ordinal of the T record it belongs to */ + const char *name; +} cs_member_rec_t; + +/* A member record. A callable, an operator and an indexer carry their + * parameter types; every other member a '-'. */ +static void cs_emit_member(cs_scan_t *s, const cs_member_rec_t *m, TSNode tparams, TSNode params) { + if (!m->name || m->type < 0) { + return; + } + sb_puts(&s->sb, "M\t"); + sb_putu(&s->sb, m->line); + sb_putc(&s->sb, '\t'); + sb_putc(&s->sb, m->kind); + if (m->is_static) { + sb_putc(&s->sb, 's'); + } + sb_putc(&s->sb, '\t'); + sb_putc(&s->sb, m->explicit_impl ? '1' : '0'); + sb_putc(&s->sb, '\t'); + sb_putu(&s->sb, (uint32_t)m->type); + sb_putc(&s->sb, '\t'); + sb_puts(&s->sb, m->name); + sb_putc(&s->sb, '\t'); + cs_put_tparams(s, tparams); + sb_putc(&s->sb, '\t'); + if (m->kind == 'c' || m->kind == 'o' || m->kind == 'x') { + cs_put_sig(s, params); + } else { + sb_putc(&s->sb, '-'); + } + sb_putc(&s->sb, '\n'); +} + +static void cs_emit_using(cs_scan_t *s, TSNode u, int region) { + bool is_global = false; + bool is_static = false; + bool is_alias = false; + cs_kids_t k = cs_kids(u); + TSNode ch; + while (cs_kids_next(&k, &ch)) { + if (ts_node_is_named(ch)) { + continue; + } + const char *t = ts_node_type(ch); + if (strcmp(t, "global") == 0) { + is_global = true; + } else if (strcmp(t, "static") == 0) { + is_static = true; + } else if (strcmp(t, "=") == 0) { + is_alias = true; + } + } + cs_kids_end(&k); + TSNode alias = is_alias ? cs_field(u, "name") : (TSNode){0}; + TSNode target = {0}; + k = cs_kids(u); + while (cs_kids_next_named(&k, &ch)) { + if (is_alias && ts_node_eq(ch, alias)) { + continue; + } + if (cs_kind_is(ch, "comment")) { + continue; + } + target = ch; + } + cs_kids_end(&k); + if (ts_node_is_null(target)) { + return; + } + /* what the directive brings in, and whom it serves: `global using` (of a + * namespace, of a type's static members, of an alias alike) is in scope in + * every file of the project */ + sb_puts(&s->sb, "U\t"); + sb_putu(&s->sb, (uint32_t)region); + sb_putc(&s->sb, '\t'); + sb_putc(&s->sb, is_alias ? 'a' : (is_static ? 's' : 'n')); + if (is_global) { + sb_putc(&s->sb, 'g'); + } + sb_putc(&s->sb, '\t'); + if (is_alias && !ts_node_is_null(alias)) { + cs_put_text_nows(s, alias); + } else { + sb_putc(&s->sb, '-'); + } + sb_putc(&s->sb, '\t'); + cs_put_text_nows(s, target); + sb_putc(&s->sb, '\n'); +} + +/* Lines [from, to] hold declarations that could not be placed. */ +static void cs_emit_unplaced(cs_scan_t *s, uint32_t from, uint32_t to) { + sb_puts(&s->sb, "X\t"); + sb_putu(&s->sb, from); + sb_putc(&s->sb, '\t'); + sb_putu(&s->sb, to); + sb_putc(&s->sb, '\n'); +} + +/* A namespace region record: the name as its declaration writes it (`A.B`), + * under the region `parent`. Returns the new region id. */ +static int cs_emit_region(cs_scan_t *s, int parent, uint32_t start, uint32_t end, + const char *name) { + int id = s->next_region++; + sb_puts(&s->sb, "R\t"); + sb_putu(&s->sb, (uint32_t)id); + sb_putc(&s->sb, '\t'); + sb_putu(&s->sb, (uint32_t)parent); + sb_putc(&s->sb, '\t'); + sb_putu(&s->sb, start); + sb_putc(&s->sb, '\t'); + sb_putu(&s->sb, end); + sb_putc(&s->sb, '\t'); + sb_puts(&s->sb, name); + sb_putc(&s->sb, '\n'); + return id; +} + +/* A type record, the type's ordinal being `ord`. Its outer type is named by + * that type's ordinal, never by a path: a record's size does not grow with + * the nesting. */ +static void cs_emit_type(cs_scan_t *s, const cs_item_t *it, int region, uint32_t end_line, + int outer, const char *name, bool incomplete, int ord) { + sb_puts(&s->sb, "T\t"); + sb_putu(&s->sb, (uint32_t)region); + sb_putc(&s->sb, '\t'); + sb_putu(&s->sb, it->line); + sb_putc(&s->sb, '\t'); + sb_putu(&s->sb, end_line); + sb_putc(&s->sb, '\t'); + sb_putc(&s->sb, it->kind); + if (it->partial) { + sb_putc(&s->sb, 'p'); + } + if (incomplete) { + sb_putc(&s->sb, '!'); + } + sb_putc(&s->sb, '\t'); + if (outer >= 0) { + sb_putu(&s->sb, (uint32_t)outer); + } else { + sb_putc(&s->sb, '-'); + } + sb_putc(&s->sb, '\t'); + sb_puts(&s->sb, name); + sb_putc(&s->sb, '\t'); + if (it->from_text) { + /* its header did not parse: the bases are unknown, which the "?" + * records as a hierarchy that cannot be followed */ + sb_puts(&s->sb, it->text_tparams); + sb_puts(&s->sb, it->kind == 'e' ? "\t\n" : "\t?\n"); + return; + } + cs_put_tparams(s, it->tparams); + sb_putc(&s->sb, '\t'); + cs_put_bases(s, it->node, it->kind); + sb_putc(&s->sb, '\n'); + /* A parameter list on the type itself is its primary constructor + * (`record R(int X)`, `class C(int x)`): a constructor the graph has no + * node for. A record's parameters are properties as well. */ + bool record = it->kind == 'r' || it->kind == 't'; + if (!(record || it->kind == 'c' || it->kind == 's')) { + return; + } + TSNode pl = cs_child_of_kind(it->node, "parameter_list"); + if (ts_node_is_null(pl)) { + return; + } + TSNode null_node = {0}; + cs_member_rec_t ctor = {.line = it->line, .kind = 'c', .type = ord, .name = name}; + cs_emit_member(s, &ctor, null_node, pl); + cs_kids_t params = cs_kids(pl); + TSNode p; + while (record && cs_kids_next_named(¶ms, &p)) { + if (cs_kind_is(p, "parameter")) { + cs_member_rec_t prop = {.line = cs_line(p), + .kind = 'p', + .type = ord, + .name = cs_name_dup(s, cs_field(p, "name"))}; + cs_emit_member(s, &prop, null_node, null_node); + } + } + cs_kids_end(¶ms); +} + +/* An open block while the items are placed. */ +typedef struct { + int item; /* tree_idx of its declaration; CBM_NOT_FOUND for the file itself */ + uint32_t end; /* first byte after it */ + int inner_depth; /* brace depth of what it holds (files with parse errors) */ + int region; /* namespace region in effect inside */ + int ns_segments; /* segments of the namespace in effect inside */ + int type; /* a type's block: the ordinal of its T record (CBM_NOT_FOUND: none) */ + int type_depth; /* types that enclose what it holds */ + bool is_type; + bool bad; /* its own place or name is unknown: so is everything inside */ +} cs_frame_t; + +/* The open blocks. A block is pushed only on a placed one, and a placed + * block adds a type level or at least one namespace segment, so the two + * nesting limits bound the stack: the root, the placed blocks, and one + * unplaced block on top. */ +enum { CS_FRAMES = CS_MAX_TYPE_NEST + CS_MAX_NS_SEGMENTS + PAIR_LEN }; + +typedef struct { + cs_frame_t frames[CS_FRAMES]; + int sp; +} cs_stack_t; + +static void cs_frame_push(cs_scan_t *s, cs_stack_t *st, cs_frame_t f) { + if (st->sp >= CS_FRAMES) { + s->failed = true; /* cannot happen (see CS_FRAMES); no scope rather than a wrong one */ + return; + } + st->frames[st->sp++] = f; +} + +/* Segments of a dotted name. */ +static int cs_segments(const char *name) { + int n = SKIP_ONE; + for (const char *p = name; *p; p++) { + n += *p == '.'; + } + return n; +} + +/* What follows a namespace's name in a file with parse errors: the brace + * that opens its block (returned), or the `;` of a file-scoped namespace + * (*file_scoped). Recovery reports a block namespace it could not parse as a + * file-scoped one whose content is an error node; the text says which it is. + * Anything else -- `namespace A` / `#else` / `namespace B` / `#endif` / `{` + * -- is a namespace this scan cannot name: CBM_NOT_FOUND, not file-scoped. */ +static int cs_namespace_brace(const cs_scan_t *s, TSNode name, bool *file_scoped) { + const char *src = s->ctx->source; + uint32_t after = cs_skip_space(src, ts_node_end_byte(name), s->root_end_byte); + *file_scoped = after < s->root_end_byte && src[after] == ';'; + return cs_open_brace_at(s, after); +} + +static void cs_place_namespace(cs_scan_t *s, cs_stack_t *st, const cs_item_t *it, bool trusted) { + const cs_frame_t top = st->frames[st->sp - SKIP_ONE]; + if (top.bad) { + return; /* inside a block that is not placed: nothing is, and nothing nests */ + } + const char *name = it->from_text ? it->text_name : NULL; + if (!it->from_text && !ts_node_is_null(it->name)) { + name = cs_namespace_name_dup(s, cs_namespace_name_start(s, it->node), + ts_node_end_byte(it->name)); + } + int brace = it->brace; + bool file_scoped = it->tag == 'F'; + if (!it->from_text && s->lexical) { + /* the text decides what kind of namespace declaration this is */ + file_scoped = false; + brace = ts_node_is_null(it->name) ? CBM_NOT_FOUND + : cs_namespace_brace(s, it->name, &file_scoped); + } + /* a file-scoped namespace is the first declaration of its file */ + bool ok = trusted && !top.is_type && name && (!file_scoped || st->sp == SKIP_ONE); + uint32_t end = s->root_end_byte; + uint32_t end_line = s->root_end_line; + int inner = top.inner_depth; + if (!file_scoped) { + bool known = false; + if (brace >= 0) { + known = cs_brace_extent(s, brace, &end, &end_line, &inner); + } else if (!s->lexical && !it->from_text) { + end = ts_node_end_byte(it->node); + end_line = cs_end_line(it->node); + known = true; + } + if (!known) { + ok = false; + end = UINT32_MAX; /* where it ends is unknown: nothing after it is placed */ + } + } + /* a namespace nested past the limit is not placed */ + int segments = name ? top.ns_segments + cs_segments(name) : 0; + ok = ok && segments <= CS_MAX_NS_SEGMENTS; + if (!ok && it->start < s->untrusted) { + cs_emit_unplaced(s, it->line, end == UINT32_MAX ? s->root_end_line : end_line); + } + int region = ok ? cs_emit_region(s, top.region, it->line, end_line, name) : top.region; + cs_frame_push(s, st, + (cs_frame_t){.item = it->tree_idx, + .end = end, + .inner_depth = inner, + .region = region, + .ns_segments = ok ? segments : top.ns_segments, + .type = CBM_NOT_FOUND, + .type_depth = 0, + .is_type = false, + .bad = !ok}); +} + +static void cs_put_quarantine(cs_scan_t *s, const char *name) { + sb_puts(&s->sb, "Q\t"); + sb_puts(&s->sb, name); + sb_putc(&s->sb, '\n'); +} + +static void cs_place_type(cs_scan_t *s, cs_stack_t *st, const cs_item_t *it, bool trusted) { + const cs_frame_t top = st->frames[st->sp - SKIP_ONE]; + const char *name = it->from_text ? it->text_name : cs_name_dup(s, it->name); + if (top.bad) { + /* declared inside a block that is not placed: its name must not + * resolve to anything else; nothing nests under it */ + if (name) { + cs_put_quarantine(s, name); + } + return; + } + bool block = it->brace >= 0; + uint32_t end = it->start; + uint32_t end_line = it->line; + int inner = 0; + bool paired = true; + bool whole = false; /* the tree's extent is the block's extent */ + if (it->from_text) { + paired = !block || cs_brace_extent(s, it->brace, &end, &end_line, &inner); + } else { + TSNode body = it->kind == 'd' ? (TSNode){0} : cs_field(it->node, "body"); + block = !ts_node_is_null(body); + end = ts_node_end_byte(it->node); + end_line = cs_end_line(it->node); + whole = true; + if (block && s->lexical) { + uint32_t node_end = end; + paired = cs_brace_extent(s, cs_open_brace_at(s, ts_node_start_byte(body)), &end, + &end_line, &inner); + whole = paired && end == node_end; + } + } + /* a type nested past the limit is not placed */ + bool ok = trusted && paired && name && top.type_depth < CS_MAX_TYPE_NEST; + int ord = CBM_NOT_FOUND; + if (ok) { + ord = s->types_out++; + cs_emit_type(s, it, top.region, end_line, top.type, name, it->broken || !whole, ord); + } else if (name) { + /* declared, but where is unknown: its name must not resolve to + * anything else either */ + cs_put_quarantine(s, name); + } + if (!ok && block && it->start < s->untrusted) { + cs_emit_unplaced(s, it->line, paired ? end_line : s->root_end_line); + } + if (block) { + cs_frame_push(s, st, + (cs_frame_t){.item = it->tree_idx, + .end = paired ? end : UINT32_MAX, + .inner_depth = inner, + .region = top.region, + .ns_segments = top.ns_segments, + .type = ord, + .type_depth = top.type_depth + SKIP_ONE, + .is_type = true, + .bad = !ok}); + } +} + +/* Place every collected declaration in the block that holds it and write + * its record. */ +static void cs_emit_items(cs_scan_t *s) { + cs_stack_t *st = (cs_stack_t *)cs_tmp_alloc(s, sizeof(*st)); + if (!st) { + return; + } + st->sp = 0; + st->frames[st->sp++] = (cs_frame_t){.item = CBM_NOT_FOUND, + .end = UINT32_MAX, + .inner_depth = 0, + .region = 0, + .ns_segments = 0, + .type = CBM_NOT_FOUND, + .type_depth = 0, + .is_type = false, + .bad = false}; + for (int i = 0; i < s->nitems && !s->failed && !s->sb.failed; i++) { + const cs_item_t *it = &s->items[i]; + /* leave the blocks that ended; with parse errors also those the + * braces say this item is not in (an #else branch re-opening the + * block its #if branch opened) */ + int depth = s->lexical ? cs_depth_at(s, it->start) : 0; + while (st->sp > SKIP_ONE && + (st->frames[st->sp - SKIP_ONE].end <= it->start || + (s->lexical && st->frames[st->sp - SKIP_ONE].inner_depth > depth))) { + st->sp--; + } + const cs_frame_t top = st->frames[st->sp - SKIP_ONE]; + bool trusted = !top.bad; + if (trusted && s->lexical) { + trusted = it->start < s->untrusted && depth == top.inner_depth; + } + switch (it->tag) { + case 'U': + if (trusted && !top.is_type) { + cs_emit_using(s, it->node, top.region); + } + break; + case 'N': + case 'F': + cs_place_namespace(s, st, it, trusted); + break; + case 'T': + cs_place_type(s, st, it, trusted); + break; + case 'M': + /* the braces decide whose member it is when the tree has errors */ + if (trusted && top.is_type && (s->lexical || it->owner == top.item)) { + cs_member_rec_t rec = {.line = cs_line(it->node), + .kind = it->kind, + .explicit_impl = it->explicit_impl, + .is_static = it->is_static, + .type = top.type, + .name = it->text_name ? it->text_name + : cs_name_dup(s, it->name)}; + cs_emit_member(s, &rec, it->tparams, it->params); + } + break; + default: + break; + } + } +} + +/* Field positions (0-based, tag included) that hold line numbers. */ +static bool cs_line_field(char tag, int field) { + switch (tag) { + case 'R': + return field == 3 || field == 4; + case 'T': + return field == 2 || field == 3; + case 'M': + return field == 1; + case 'X': + return field == 1 || field == 2; + default: + return false; + } +} + +char *cbm_doclink_cs_portable_scope(const char *scope) { + size_t n = strlen(scope); + char *out = (char *)cbm_alloc(CBM_MEM_CLASS_OTHER, n + SKIP_ONE); + if (!out) { + return NULL; + } + size_t w = 0; + const char *p = scope; + while (*p) { + const char *nl = strchr(p, '\n'); + size_t len = nl ? (size_t)(nl - p) : strlen(p); + char tag = p[0]; + int field = 0; + size_t i = 0; + while (i < len) { + size_t fend = i; + while (fend < len && p[fend] != '\t') { + fend++; + } + if (cs_line_field(tag, field)) { + out[w++] = '0'; + } else { + memcpy(out + w, p + i, fend - i); + w += fend - i; + } + i = fend; + if (i < len) { + out[w++] = '\t'; + i++; + field++; + } + } + if (nl) { + out[w++] = '\n'; + p = nl + SKIP_ONE; + } else { + p += len; + } + } + out[w] = '\0'; + return out; +} + +const char *cbm_doclink_cs_scan_scope(CBMExtractCtx *ctx) { + if (ts_node_is_null(ctx->root)) { + return NULL; + } + cs_scan_t s = {.ctx = ctx, + .tmp = ctx->scratch ? ctx->scratch : ctx->arena, + .sb = {.a = ctx->arena}, + .top = CBM_NOT_FOUND, + .untrusted = UINT32_MAX, + .next_region = SKIP_ONE}; + s.root_end_byte = ctx->source_len > 0 ? (uint32_t)ctx->source_len : 0; + s.root_end_line = cs_end_line(ctx->root); + s.lexical = ts_node_has_error(ctx->root); + if (s.lexical) { + cs_scan_tokens(&s); + } + /* the root itself is an error node when nothing of the file parsed */ + cs_collect(&s, ctx->root, cs_kind_is(ctx->root, "ERROR") ? CS_OWNER_LEXICAL : CS_OWNER_NONE); + cs_add_text_items(&s); + sb_puts(&s.sb, CBM_DOCLINK_CS_SCOPE_TAG "\n"); + cs_emit_items(&s); + if (s.untrusted != UINT32_MAX) { + cs_emit_unplaced(&s, s.untrusted_row + TS_LINE_OFFSET, s.root_end_line); + } + cs_cost_publish(&s); + if (s.failed || s.sb.failed || !s.sb.buf) { + return NULL; + } + return s.sb.buf; +} + +/* ── MSBuild project files ─────────────────────────────────────────── + * + * A C# project's global usings come from its MSBuild files: the project + * file, the Directory.Build.props / .targets above it, and what those import + * (R1). The resolver must not open any of them: what it would read is a path + * the index never looked at (a symbolic link out of the repository, a named + * pipe), and it would read it again on every run. So a project file has a + * scope blob like a source file, and the resolver evaluates blobs. + * + * Project blob: the tag line "cs1", then one record per line. A field is + * empty for an attribute that is not there, else '=' and its text, XML + * entities decoded, with \\ \t \n \r escaped. + * P sdk state the first record. state '-': read; + * '!': a project file that could not be + * read; '>': one larger than a project + * file is (CSX_MAX_PROJECT_BYTES), not + * read. Nothing follows the last two + * I group-cond cond project sdk ; old blobs repeat a group's + * condition in the first field + * B cond start of an , condition once + * J (empty) cond project sdk a child import; shares the B condition + * E end of the import group + * G cond ; its properties follow + * V cond name value a property; value '?' when it is not + * plain text + * H cond that has items; + * they follow + * N cond include remove static alias + * K name a property set inside a construct that + * is not evaluated () + * Y a that is not evaluated (in a + * , with Update, with metadata + * elements) + * C what another construct that is not evaluated + * Document order is evaluation order. Nothing else of the file is in the + * blob (targets, other items), so an edit there leaves it as it is. */ + +enum { CSX_EOF = 0, CSX_OPEN, CSX_EMPTY, CSX_CLOSE, CSX_TEXT, CSX_BAD }; + +/* The size past which a file is no project file to this reader. */ +enum { CSX_MAX_PROJECT_BYTES = 1000000 }; + +enum { + CSX_A_CONDITION = 0, + CSX_A_PROJECT, + CSX_A_SDK, + CSX_A_INCLUDE, + CSX_A_REMOVE, + CSX_A_UPDATE, + CSX_A_STATIC, + CSX_A_ALIAS, + CSX_A_COUNT +}; + +typedef struct { + uint32_t s; + uint32_t e; + bool has; +} csx_span_t; + +typedef struct { + int kind; + csx_span_t name; /* element name without a namespace prefix */ + csx_span_t text; /* CSX_TEXT */ + bool raw; /* CSX_TEXT of a CDATA section: no entities in it */ + csx_span_t attr[CSX_A_COUNT]; +} csx_tok_t; + +typedef struct { + const char *src; + uint32_t n; + uint32_t i; +} csx_t; + +/* Index of `lit` in src[from, limit), or `limit`. */ +static uint32_t csx_find(const csx_t *x, uint32_t from, uint32_t limit, const char *lit) { + size_t ll = strlen(lit); + for (uint32_t k = from; k + ll <= limit; k++) { + if (x->src[k] == lit[0] && memcmp(x->src + k, lit, ll) == 0) { + return k; + } + } + return limit; +} + +static bool csx_name_char(unsigned char c) { + return isalnum(c) || c == '_' || c == ':' || c == '.' || c == '-' || c >= 0x80; +} + +static csx_span_t csx_local(const csx_t *x, uint32_t s, uint32_t e) { + for (uint32_t k = e; k > s; k--) { + if (x->src[k - SKIP_ONE] == ':') { + s = k; + break; + } + } + return (csx_span_t){.s = s, .e = e, .has = true}; +} + +static bool csx_is(const csx_t *x, csx_span_t v, const char *word) { + return v.has && strlen(word) == v.e - v.s && memcmp(x->src + v.s, word, v.e - v.s) == 0; +} + +/* The attributes the blob keeps. MSBuild reads attribute names without + * regard to case. */ +static int csx_attr_index(const char *name, size_t len) { + static const char *const names[CSX_A_COUNT] = {"condition", "project", "sdk", "include", + "remove", "update", "static", "alias"}; + for (int a = 0; a < CSX_A_COUNT; a++) { + if (strlen(names[a]) != len) { + continue; + } + size_t k = 0; + while (k < len && tolower((unsigned char)name[k]) == names[a][k]) { + k++; + } + if (k == len) { + return a; + } + } + return CBM_NOT_FOUND; +} + +/* Past markup that is no element at src[x->i] ('<' stands there): a comment, + * a processing instruction, a declaration. false when it does not end. A + * CDATA section is text: *cdata, and the cursor stays. */ +static bool csx_skip_markup(csx_t *x, bool *skipped, bool *cdata) { + const char *s = x->src; + uint32_t rest = x->n - x->i; + *skipped = true; + *cdata = false; + if (rest >= 4 && memcmp(s + x->i, ""); + x->i = e + 3; + return e < x->n; + } + if (rest >= 9 && memcmp(s + x->i, "= PAIR_LEN && s[x->i + SKIP_ONE] == '?') { + uint32_t e = csx_find(x, x->i + PAIR_LEN, x->n, "?>"); + x->i = e + PAIR_LEN; + return e < x->n; + } + if (rest >= PAIR_LEN && s[x->i + SKIP_ONE] == '!') { + /* , with an internal subset up to "]>" */ + uint32_t e = csx_find(x, x->i + PAIR_LEN, x->n, ">"); + uint32_t sub = csx_find(x, x->i + PAIR_LEN, e, "["); + if (sub < e) { + uint32_t close = csx_find(x, sub, x->n, "]>"); + e = close < x->n ? close + SKIP_ONE : x->n; + } + x->i = e + SKIP_ONE; + return e < x->n; + } + *skipped = false; + return true; +} + +/* The attributes of a start tag from src[p]; the tag's kind (CSX_OPEN, + * CSX_EMPTY) or CSX_BAD. Moves the cursor past the tag. */ +static int csx_attributes(csx_t *x, uint32_t p, csx_tok_t *t) { + const char *s = x->src; + for (;;) { + while (p < x->n && isspace((unsigned char)s[p])) { + p++; + } + if (p >= x->n) { + return CSX_BAD; + } + if (s[p] == '>') { + x->i = p + SKIP_ONE; + return CSX_OPEN; + } + if (s[p] == '/' && p + SKIP_ONE < x->n && s[p + SKIP_ONE] == '>') { + x->i = p + PAIR_LEN; + return CSX_EMPTY; + } + uint32_t as = p; + while (p < x->n && csx_name_char((unsigned char)s[p])) { + p++; + } + if (p == as) { + return CSX_BAD; + } + csx_span_t an = csx_local(x, as, p); + while (p < x->n && isspace((unsigned char)s[p])) { + p++; + } + if (p >= x->n || s[p] != '=') { + return CSX_BAD; + } + p++; + while (p < x->n && isspace((unsigned char)s[p])) { + p++; + } + if (p >= x->n || (s[p] != '"' && s[p] != '\'')) { + return CSX_BAD; + } + char quote = s[p++]; + uint32_t vs = p; + while (p < x->n && s[p] != quote) { + p++; + } + if (p >= x->n) { + return CSX_BAD; + } + int ai = csx_attr_index(s + an.s, an.e - an.s); + if (ai >= 0) { + t->attr[ai] = (csx_span_t){.s = vs, .e = p, .has = true}; + } + p++; + } +} + +/* The next token of the document. Every byte is passed once. */ +static void csx_next(csx_t *x, csx_tok_t *t) { + memset(t, 0, sizeof(*t)); + const char *s = x->src; + for (;;) { + if (x->i >= x->n) { + t->kind = CSX_EOF; + return; + } + if (s[x->i] != '<') { + uint32_t a = x->i; + while (x->i < x->n && s[x->i] != '<') { + x->i++; + } + t->kind = CSX_TEXT; + t->text = (csx_span_t){.s = a, .e = x->i, .has = true}; + return; + } + bool skipped = false; + bool cdata = false; + if (!csx_skip_markup(x, &skipped, &cdata)) { + t->kind = CSX_BAD; + return; + } + if (cdata) { + uint32_t e = csx_find(x, x->i + 9, x->n, "]]>"); + if (e >= x->n) { + t->kind = CSX_BAD; + return; + } + t->kind = CSX_TEXT; + t->raw = true; + t->text = (csx_span_t){.s = x->i + 9, .e = e, .has = true}; + x->i = e + 3; + return; + } + if (!skipped) { + break; + } + } + uint32_t p = x->i + SKIP_ONE; + bool close = p < x->n && s[p] == '/'; + if (close) { + p++; + } + uint32_t ns = p; + while (p < x->n && csx_name_char((unsigned char)s[p])) { + p++; + } + if (p == ns) { + t->kind = CSX_BAD; + return; + } + t->name = csx_local(x, ns, p); + if (!close) { + t->kind = csx_attributes(x, p, t); + return; + } + while (p < x->n && isspace((unsigned char)s[p])) { + p++; + } + if (p >= x->n || s[p] != '>') { + t->kind = CSX_BAD; + return; + } + x->i = p + SKIP_ONE; + t->kind = CSX_CLOSE; +} + +/* One character of a field: escaped where it would break the line format. A + * control character has no place in XML text; it is written as a space. */ +static void csx_put_char(cs_sb_t *sb, unsigned char c) { + if (c == '\\') { + sb_puts(sb, "\\\\"); + } else if (c == '\t') { + sb_puts(sb, "\\t"); + } else if (c == '\n') { + sb_puts(sb, "\\n"); + } else if (c == '\r') { + sb_puts(sb, "\\r"); + } else { + sb_putc(sb, c < 0x20 ? ' ' : (char)c); + } +} + +/* A code point as UTF-8. */ +static void csx_put_codepoint(cs_sb_t *sb, uint32_t cp) { + if (cp < 0x80) { + csx_put_char(sb, (unsigned char)cp); + } else if (cp < 0x800) { + sb_putc(sb, (char)(0xC0 | (cp >> 6))); + sb_putc(sb, (char)(0x80 | (cp & 0x3F))); + } else if (cp < 0x10000) { + sb_putc(sb, (char)(0xE0 | (cp >> 12))); + sb_putc(sb, (char)(0x80 | ((cp >> 6) & 0x3F))); + sb_putc(sb, (char)(0x80 | (cp & 0x3F))); + } else { + sb_putc(sb, (char)(0xF0 | ((cp >> 18) & 0x07))); + sb_putc(sb, (char)(0x80 | ((cp >> 12) & 0x3F))); + sb_putc(sb, (char)(0x80 | ((cp >> 6) & 0x3F))); + sb_putc(sb, (char)(0x80 | (cp & 0x3F))); + } +} + +/* The entity at src[i] ('&' stands there): its code point and the index past + * it; 0 when there is none. */ +static uint32_t csx_entity(const csx_t *x, uint32_t i, uint32_t end, uint32_t *cp) { + static const struct { + const char *ent; + char ch; + } ents[] = {{"<", '<'}, {">", '>'}, {"&", '&'}, {""", '"'}, {"'", '\''}}; + const char *s = x->src; + for (size_t e = 0; e < sizeof(ents) / sizeof(ents[0]); e++) { + size_t el = strlen(ents[e].ent); + if (i + el <= end && memcmp(s + i, ents[e].ent, el) == 0) { + *cp = (unsigned char)ents[e].ch; + return i + (uint32_t)el; + } + } + if (i + PAIR_LEN < end && s[i + SKIP_ONE] == '#') { + bool hex = s[i + PAIR_LEN] == 'x' || s[i + PAIR_LEN] == 'X'; + uint32_t k = i + PAIR_LEN + (hex ? SKIP_ONE : 0); + uint32_t v = 0; + uint32_t digits = 0; + while (k < end && digits < CBM_SZ_8 && + (hex ? isxdigit((unsigned char)s[k]) : isdigit((unsigned char)s[k]))) { + unsigned char d = (unsigned char)s[k]; + uint32_t dv = isdigit(d) ? (uint32_t)(d - '0') : (uint32_t)(tolower(d) - 'a') + 10U; + v = (v * (hex ? 16U : 10U)) + dv; + k++; + digits++; + } + if (digits > 0 && k < end && s[k] == ';' && v > 0 && v <= 0x10FFFF) { + *cp = v; + return k + SKIP_ONE; + } + } + return 0; +} + +/* The text of `v`, entities decoded (unless `raw`), escaped. */ +static void csx_put_text(cs_sb_t *sb, const csx_t *x, csx_span_t v, bool raw) { + for (uint32_t i = v.s; i < v.e;) { + uint32_t cp = 0; + uint32_t past = (!raw && x->src[i] == '&') ? csx_entity(x, i, v.e, &cp) : 0; + if (past) { + csx_put_codepoint(sb, cp); + i = past; + } else { + csx_put_char(sb, (unsigned char)x->src[i]); + i++; + } + } +} + +/* A field: a tab, then nothing for an absent attribute, else '=' and its text. */ +static void csx_put_field(cs_sb_t *sb, const csx_t *x, csx_span_t v) { + sb_putc(sb, '\t'); + if (v.has) { + sb_putc(sb, '='); + csx_put_text(sb, x, v, false); + } +} + +/* What a project file's scan keeps between tokens. */ +typedef struct { + csx_t x; + cs_sb_t out; + cs_sb_t value; /* a property's text so far */ + int depth; /* open elements */ + char group; /* the child of the cursor is in: G H i c, or 0 */ + csx_span_t gcond; /* its Condition */ + bool h_written; /* the ItemGroup's H record is out */ + bool i_written; /* the ImportGroup's B record is out */ + bool prop_open; /* a property element is open ... */ + bool prop_complex; /* ... and holds elements, not just text */ + bool using_open; /* a with content is open ... */ + bool using_complex; /* ... and holds metadata elements */ + csx_tok_t pending; /* the open property's or 's start tag */ + int choose_props; /* in a : the depth of a 's children, or -1 */ +} csx_scan_t; + +static void csx_put_using(csx_scan_t *p, const csx_tok_t *t) { + if (!p->h_written) { + sb_putc(&p->out, 'H'); + csx_put_field(&p->out, &p->x, p->gcond); + sb_putc(&p->out, '\n'); + p->h_written = true; + } + sb_putc(&p->out, 'N'); + csx_put_field(&p->out, &p->x, t->attr[CSX_A_CONDITION]); + csx_put_field(&p->out, &p->x, t->attr[CSX_A_INCLUDE]); + csx_put_field(&p->out, &p->x, t->attr[CSX_A_REMOVE]); + csx_put_field(&p->out, &p->x, t->attr[CSX_A_STATIC]); + csx_put_field(&p->out, &p->x, t->attr[CSX_A_ALIAS]); + sb_putc(&p->out, '\n'); +} + +static void csx_put_import(csx_scan_t *p, const csx_tok_t *t, bool grouped) { + if (grouped && !p->i_written) { + sb_putc(&p->out, 'B'); + csx_put_field(&p->out, &p->x, p->gcond); + sb_putc(&p->out, '\n'); + p->i_written = true; + } + sb_putc(&p->out, grouped ? 'J' : 'I'); + csx_put_field(&p->out, &p->x, (csx_span_t){0}); + csx_put_field(&p->out, &p->x, t->attr[CSX_A_CONDITION]); + csx_put_field(&p->out, &p->x, t->attr[CSX_A_PROJECT]); + csx_put_field(&p->out, &p->x, t->attr[CSX_A_SDK]); + sb_putc(&p->out, '\n'); +} + +/* The open property ends: its record. */ +static void csx_put_property(csx_scan_t *p) { + const csx_tok_t *t = &p->pending; + sb_putc(&p->out, 'V'); + csx_put_field(&p->out, &p->x, t->attr[CSX_A_CONDITION]); + sb_putc(&p->out, '\t'); + sb_putn(&p->out, p->x.src + t->name.s, t->name.e - t->name.s); + sb_putc(&p->out, '\t'); + if (p->prop_complex) { + sb_putc(&p->out, '?'); + } else { + sb_putc(&p->out, '='); + if (p->value.len > 0) { + sb_putn(&p->out, p->value.buf, p->value.len); + } + } + sb_putc(&p->out, '\n'); +} + +/* An element inside a : what it could set is named, not evaluated. */ +static void csx_choose_child(csx_scan_t *p, const csx_tok_t *t) { + if (p->choose_props >= 0 && p->depth == p->choose_props) { + sb_puts(&p->out, "K\t"); + sb_putn(&p->out, p->x.src + t->name.s, t->name.e - t->name.s); + sb_putc(&p->out, '\n'); + } else if (csx_is(&p->x, t->name, "PropertyGroup") && p->choose_props < 0 && + t->kind == CSX_OPEN) { + p->choose_props = p->depth + SKIP_ONE; + } else if (csx_is(&p->x, t->name, "Using")) { + sb_puts(&p->out, "Y\n"); + } else if (csx_is(&p->x, t->name, "Import")) { + sb_puts(&p->out, "C\tImport\n"); + } +} + +/* A child of starts. */ +static void csx_project_child(csx_scan_t *p, const csx_tok_t *t) { + p->group = 0; + p->gcond = t->attr[CSX_A_CONDITION]; + if (csx_is(&p->x, t->name, "Import")) { + csx_put_import(p, t, false); + } else if (csx_is(&p->x, t->name, "PropertyGroup")) { + sb_putc(&p->out, 'G'); + csx_put_field(&p->out, &p->x, p->gcond); + sb_putc(&p->out, '\n'); + p->group = 'G'; + } else if (csx_is(&p->x, t->name, "ItemGroup")) { + p->group = 'H'; + p->h_written = false; + } else if (csx_is(&p->x, t->name, "ImportGroup")) { + p->group = 'i'; + p->i_written = false; + } else if (csx_is(&p->x, t->name, "Choose")) { + sb_puts(&p->out, "C\tChoose\n"); + p->group = 'c'; + p->choose_props = CBM_NOT_FOUND; + } + if (t->kind == CSX_EMPTY) { + p->group = 0; + } +} + +/* A grandchild of starts. */ +static void csx_group_child(csx_scan_t *p, const csx_tok_t *t) { + if (p->group == 'G') { + p->pending = *t; + p->value.len = 0; + p->prop_complex = false; + if (t->kind == CSX_EMPTY) { + csx_put_property(p); + } else { + p->prop_open = true; + } + } else if (p->group == 'H' && csx_is(&p->x, t->name, "Using")) { + if (t->attr[CSX_A_UPDATE].has) { + sb_puts(&p->out, "Y\n"); + } else if (t->kind == CSX_EMPTY) { + csx_put_using(p, t); + } else { + p->pending = *t; + p->using_open = true; + p->using_complex = false; + } + } else if (p->group == 'i' && csx_is(&p->x, t->name, "Import")) { + csx_put_import(p, t, true); + } +} + +/* A start tag below the root. */ +static void csx_element(csx_scan_t *p, const csx_tok_t *t) { + if (p->depth == SKIP_ONE) { + csx_project_child(p, t); + } else if (p->group == 'c') { + csx_choose_child(p, t); + } else if (p->depth == PAIR_LEN) { + csx_group_child(p, t); + } else { + p->prop_complex = p->prop_complex || p->prop_open; + p->using_complex = p->using_complex || p->using_open; + } + if (t->kind == CSX_OPEN) { + p->depth++; + } +} + +/* An end tag: `depth` is already the depth outside the element. */ +static void csx_element_end(csx_scan_t *p) { + if (p->depth == PAIR_LEN && p->prop_open) { + csx_put_property(p); + p->prop_open = false; + } else if (p->depth == PAIR_LEN && p->using_open) { + if (p->using_complex) { + sb_puts(&p->out, "Y\n"); + } else { + csx_put_using(p, &p->pending); + } + p->using_open = false; + } + if (p->depth == SKIP_ONE) { + if (p->group == 'i' && p->i_written) { + sb_puts(&p->out, "E\n"); + } + p->group = 0; + } + if (p->group == 'c' && p->choose_props == p->depth + SKIP_ONE) { + p->choose_props = CBM_NOT_FOUND; + } +} + +static bool csx_ci_suffix(const char *s, const char *sfx) { + size_t n = s ? strlen(s) : 0; + size_t sl = strlen(sfx); + if (n < sl) { + return false; + } + for (size_t i = 0; i < sl; i++) { + if (tolower((unsigned char)s[n - sl + i]) != sfx[i]) { + return false; + } + } + return true; +} + +void cbm_doclink_cs_project_parse_doc(CBMExtractCtx *ctx, const CBMDefinition *def, const char *doc, + uint32_t doc_line) { + /* a project file's comments hold no references to code */ + (void)ctx; + (void)def; + (void)doc; + (void)doc_line; +} + +const char *cbm_doclink_cs_project_scan_scope(CBMExtractCtx *ctx) { + /* The gate is the file's name, before a byte of it is looked at: only a + * *.csproj, *.props or *.targets file can be an MSBuild project file of a + * C# project. Every other XML file -- there are many, and large ones -- + * costs nothing here. */ + bool named_project = csx_ci_suffix(ctx->rel_path, ".csproj"); + if (!ctx->source || !(named_project || csx_ci_suffix(ctx->rel_path, ".props") || + csx_ci_suffix(ctx->rel_path, ".targets"))) { + return NULL; + } + /* A project file past the size a project file has is not read either, + * and its blob says so: what it holds is unknown, not absent. */ + if (ctx->source_len > CSX_MAX_PROJECT_BYTES) { + return cbm_arena_strdup(ctx->arena, CBM_DOCLINK_CS_SCOPE_TAG "\nP\t\t>\n"); + } + /* a *.csproj marks its directory as a C# project even when it cannot be + * read; a *.props or *.targets file has a blob only as an MSBuild + * */ + csx_scan_t p = { + .x = {.src = ctx->source, .n = ctx->source_len > 0 ? (uint32_t)ctx->source_len : 0}, + .out = {.a = ctx->arena}, + .value = {.a = ctx->scratch ? ctx->scratch : ctx->arena}, + .choose_props = CBM_NOT_FOUND}; + static const char bom[] = "\xEF\xBB\xBF"; + if (p.x.n >= 3 && memcmp(p.x.src, bom, 3) == 0) { + p.x.i = 3; + } + bool is_project = false; + bool bad = false; + csx_tok_t t; + for (;;) { + csx_next(&p.x, &t); + if (t.kind == CSX_EOF || t.kind == CSX_BAD) { + bad = t.kind == CSX_BAD || p.depth != 0; + break; + } + if (t.kind == CSX_TEXT) { + if (p.prop_open && p.depth == 3) { + csx_put_text(&p.value, &p.x, t.text, t.raw); + } + continue; + } + if (t.kind == CSX_CLOSE) { + if (p.depth == 0) { + bad = true; + break; + } + p.depth--; + csx_element_end(&p); + continue; + } + if (p.depth > 0) { + csx_element(&p, &t); + continue; + } + if (is_project) { + bad = true; /* a second root element */ + break; + } + if (!csx_is(&p.x, t.name, "Project")) { + break; /* XML, but no MSBuild file */ + } + is_project = true; + sb_puts(&p.out, CBM_DOCLINK_CS_SCOPE_TAG "\nP"); + csx_put_field(&p.out, &p.x, t.attr[CSX_A_SDK]); + sb_puts(&p.out, "\t-\n"); + if (t.kind == CSX_OPEN) { + p.depth++; + } + } + cs_cost_add(p.x.i, 0); + if ((bad && (is_project || named_project)) || (!is_project && named_project)) { + p.out.len = 0; + sb_puts(&p.out, CBM_DOCLINK_CS_SCOPE_TAG "\nP\t\t!\n"); + } else if (!is_project) { + return NULL; + } + if (p.out.failed || p.value.failed || !p.out.buf) { + return NULL; + } + return p.out.buf; +} diff --git a/internal/cbm/extract_defs.c b/internal/cbm/extract_defs.c index 281b38359a..67a2e925f1 100644 --- a/internal/cbm/extract_defs.c +++ b/internal/cbm/extract_defs.c @@ -1,5 +1,6 @@ #include "cbm.h" -#include "arena.h" // CBMArena, cbm_arena_alloc/strdup/sprintf +#include "arena.h" // CBMArena, cbm_arena_alloc/strdup/sprintf +#include "doclink.h" // cbm_doclink_note_doc_line #include "helpers.h" #include "lang_specs.h" #include "foundation/constants.h" @@ -1361,6 +1362,7 @@ typedef struct { doc_span_t *items; /* trivia directly before the anchor, in source order */ int count; int cap; + bool failed; /* a missing span must not become shared documentation */ bool code_before; /* a non-trivia sibling precedes items[0] */ uint32_t code_erow; /* ... its effective end row */ uint32_t code_eb; /* ... its end byte (Kotlin gap scan) */ @@ -1488,8 +1490,17 @@ static doc_span_t doc_span_of(TSNode n, const char *src, uint8_t kind) { static void doc_push_span(CBMArena *a, doc_trivia_t *t, const doc_span_t *sp) { if (t->count == t->cap) { int ncap = t->cap ? t->cap * DOC_SPAN_GROW : DOC_SPAN_INIT_CAP; - doc_span_t *grown = (doc_span_t *)cbm_arena_alloc(a, (size_t)ncap * sizeof(doc_span_t)); + doc_span_t *grown; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (cbm_doclink_test_fail_alloc(CBM_DOCLINK_ALLOC_SPAN)) { + grown = NULL; + } else +#endif + { + grown = (doc_span_t *)cbm_arena_alloc(a, (size_t)ncap * sizeof(doc_span_t)); + } if (!grown) { + t->failed = true; return; } if (t->count > 0) { @@ -2029,11 +2040,15 @@ static const char *doc_run_text(CBMExtractCtx *ctx, const doc_trivia_t *t, int f int kept = 0; bool words = false; size_t total = 0; + uint32_t first_row = 0; for (int k = first; k <= last; k++) { const doc_span_t *sp = &t->items[k]; if (!doc_span_kept(src, sp, go_directives)) { continue; } + if (kept == 0) { + first_row = sp->srow; + } total += (size_t)(sp->eb - sp->sb) + SKIP_ONE; words = words || doc_has_words(src + sp->sb, sp->eb - sp->sb); kept++; @@ -2041,7 +2056,15 @@ static const char *doc_run_text(CBMExtractCtx *ctx, const doc_trivia_t *t, int f if (kept == 0 || !words) { return NULL; } - char *buf = (char *)cbm_arena_alloc(ctx->arena, total + SKIP_ONE); + char *buf; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (cbm_doclink_test_fail_alloc(CBM_DOCLINK_ALLOC_TEXT)) { + buf = NULL; + } else +#endif + { + buf = (char *)cbm_arena_alloc(ctx->arena, total + SKIP_ONE); + } if (!buf) { return NULL; } @@ -2060,9 +2083,16 @@ static const char *doc_run_text(CBMExtractCtx *ctx, const doc_trivia_t *t, int f buf[w++] = '\n'; } memcpy(buf + w, src + sp->sb, eb - sp->sb); +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + cbm_doclink_test_note_doc_work(eb - sp->sb, 0, 0); +#endif w += eb - sp->sb; } buf[w] = '\0'; + /* Doc-link references need their source lines: the text's line k is the + * source line first_row + k (one comment per line, or a block keeping + * its own newlines). */ + cbm_doclink_note_doc_line(ctx, buf, first_row + SKIP_ONE); return buf; } @@ -2144,12 +2174,19 @@ static const char *doc_from_trivia(CBMExtractCtx *ctx, const doc_trivia_t *t, return doc_run_text(ctx, t, doc_run_first(lang, t, near), near, lang == CBM_LANG_GO); } -static const char *doc_for_anchor(CBMExtractCtx *ctx, TSNode anchor) { +static const char *doc_for_anchor_status(CBMExtractCtx *ctx, TSNode anchor, bool *complete) { doc_trivia_t t; doc_collect_trivia(ctx, anchor, &t); + if (complete) { + *complete = !t.failed; + } return doc_from_trivia(ctx, &t, ts_node_start_point(anchor).row); } +static const char *doc_for_anchor(CBMExtractCtx *ctx, TSNode anchor) { + return doc_for_anchor_status(ctx, anchor, NULL); +} + static bool doc_kind_is(TSNode n, const char *kind) { return !ts_node_is_null(n) && strcmp(ts_node_type(n), kind) == 0; } @@ -2636,11 +2673,19 @@ static const char *extract_docstring(CBMExtractCtx *ctx, TSNode node, const char } /* Doc of a Field, Variable, enum member or Macro (code languages only). */ -static const char *extract_member_docstring(CBMExtractCtx *ctx, TSNode node) { +static const char *extract_member_docstring_status(CBMExtractCtx *ctx, TSNode node, + bool *complete) { if (!doc_lang_member_docs(ctx->language)) { + if (complete) { + *complete = true; + } return NULL; } - return doc_for_anchor(ctx, doc_anchor(ctx, node)); + return doc_for_anchor_status(ctx, doc_anchor(ctx, node), complete); +} + +static const char *extract_member_docstring(CBMExtractCtx *ctx, TSNode node) { + return extract_member_docstring_status(ctx, node, NULL); } /* Go package comment: the comment group touching `package`, directives @@ -7462,8 +7507,8 @@ static void extract_elixir_call(CBMExtractCtx *ctx, TSNode node, const CBMLangSp * from `name` only where a language scopes a variable below the module — Nix, * whose binding names are attrpaths (`a.b.c = …` is name `c`, QN suffix `a.b.c`). * Pass NULL to use `name` for both. */ -static void push_var_def_qn(CBMExtractCtx *ctx, const char *name, const char *qn_name, - TSNode node) { +static void push_var_def_qn_doc(CBMExtractCtx *ctx, const char *name, const char *qn_name, + TSNode node, const char *doc) { if (!name || !name[0] || strcmp(name, "_") == 0) { return; } @@ -7485,10 +7530,18 @@ static void push_var_def_qn(CBMExtractCtx *ctx, const char *name, const char *qn def.start_line = ts_node_start_point(node).row + TS_LINE_OFFSET; def.end_line = ts_node_end_point(node).row + TS_LINE_OFFSET; def.is_exported = cbm_is_exported(name, ctx->language); - def.docstring = extract_member_docstring(ctx, node); + def.docstring = doc; cbm_defs_push(&ctx->result->defs, a, def); } +static void push_var_def_qn(CBMExtractCtx *ctx, const char *name, const char *qn_name, + TSNode node) { + if (!name || !name[0] || strcmp(name, "_") == 0) { + return; + } + push_var_def_qn_doc(ctx, name, qn_name, node, extract_member_docstring(ctx, node)); +} + static void push_var_def(CBMExtractCtx *ctx, const char *name, TSNode node) { push_var_def_qn(ctx, name, NULL, node); } @@ -7558,6 +7611,14 @@ static void extract_csharp_vars(CBMExtractCtx *ctx, TSNode node, CBMArena *a) { push_var_def(ctx, fname, node); return; } + /* All declarators have this field as their documentation anchor. Keep one + * immutable arena string, while each variable retains its own identity. + * Failed or absent collection keeps the original per-variable lookup. */ + bool complete = false; + const char *doc = extract_member_docstring_status(ctx, node, &complete); + if (!complete) { + doc = NULL; + } uint32_t n = ts_node_named_child_count(node); for (uint32_t i = 0; i < n; i++) { TSNode child = ts_node_named_child(node, i); @@ -7573,7 +7634,12 @@ static void extract_csharp_vars(CBMExtractCtx *ctx, TSNode node, CBMArena *a) { id = cbm_find_child_by_kind(decl, "identifier"); } if (!ts_node_is_null(id)) { - push_var_def(ctx, cbm_node_text(a, id, ctx->source), decl); + const char *name = cbm_node_text(a, id, ctx->source); + if (doc) { + push_var_def_qn_doc(ctx, name, NULL, decl, doc); + } else { + push_var_def(ctx, name, decl); + } } } } diff --git a/internal/cbm/result_compact.c b/internal/cbm/result_compact.c index c97400fece..9691b6c961 100644 --- a/internal/cbm/result_compact.c +++ b/internal/cbm/result_compact.c @@ -394,6 +394,12 @@ static void cr_walk(cr_ctx_t *c, CBMFileResult *r) { cr_str(c, &r->error_msg); cr_str(c, &r->error_ranges); cr_str(c, &r->module_doc); + cr_array(c, (void **)&r->doc_links.items, r->doc_links.count, sizeof(CBMDocLink)); + for (int i = 0; i < r->doc_links.count && r->doc_links.items; i++) { + cr_str(c, &r->doc_links.items[i].source_qn); + cr_str(c, &r->doc_links.items[i].raw); + } + cr_str(c, &r->doc_scope); cr_blob(c, (const void **)&r->source, r->source ? (size_t)r->source_len + SKIP_ONE : 0); } @@ -533,6 +539,7 @@ void cbm_result_compact(CBMFileResult *result) { tmp.infra_bindings.cap = tmp.infra_bindings.count; tmp.channels.cap = tmp.channels.count; tmp.field_types.cap = tmp.field_types.count; + tmp.doc_links.cap = tmp.doc_links.count; /* A composite kept its per-unit results only so shallow-copied strings * stayed valid; every string is now a copy of its own. */ diff --git a/src/cli/cli.c b/src/cli/cli.c index df51e723b4..db1c155596 100644 --- a/src/cli/cli.c +++ b/src/cli/cli.c @@ -1566,8 +1566,8 @@ static const char skill_content[] = "## Edge Types\n" "CALLS, HTTP_CALLS, ASYNC_CALLS, DATA_FLOWS, IMPORTS, DEFINES, DEFINES_METHOD,\n" "HANDLES, IMPLEMENTS, OVERRIDE, USAGE, CALL_REFERENCE, CONFIGURES, REFERENCES_FILE,\n" - "FILE_CHANGES_WITH, SIMILAR_TO, SEMANTICALLY_RELATED, CONTAINS_FILE, CONTAINS_FOLDER,\n" - "CONTAINS_PACKAGE\n" + "MENTIONS (doc comment -> referenced code), FILE_CHANGES_WITH, SIMILAR_TO,\n" + "SEMANTICALLY_RELATED, CONTAINS_FILE, CONTAINS_FOLDER, CONTAINS_PACKAGE\n" "\n" "## Cypher Examples (for query_graph)\n" "```\n" diff --git a/src/mcp/mcp.c b/src/mcp/mcp.c index 1fe774c712..1dc548fbf4 100644 --- a/src/mcp/mcp.c +++ b/src/mcp/mcp.c @@ -741,8 +741,9 @@ static const tool_def_t TOOLS[] = { "\"project\"]}"}, {"index_status", - "Project readiness, counts, root, and coverage gaps. diagnostics adds coverage rows; verbose " - "adds Git paths. Best-effort only; verify cited paths with check_index_coverage.", + "Project readiness, counts, root, coverage gaps, and doc_links (doc-comment references: " + "MENTIONS edges, unresolved by reason). diagnostics adds coverage rows; verbose adds Git " + "paths. Best-effort only; verify cited paths with check_index_coverage.", "{\"type\":\"object\",\"properties\":{\"project\":{\"type\":\"string\"}," "\"verbose\":{\"type\":\"boolean\",\"default\":false,\"description\":\"Add worktree/" "shadow Git paths for index-location debugging.\"}," @@ -6177,6 +6178,308 @@ static char *handle_query_graph(cbm_mcp_server_t *srv, const char *args) { return res; } +/* Doc-comment references (index_status.doc_links): MENTIONS edges, the + * unresolved rows per reason, and the layer's status. "error" when the + * generation recorded a failed doc-link layer, when the doc_link_unresolved + * table is missing (an index from before the layer: reindex), or when it + * cannot be read. Samples (diagnostics=full) are rows ordered by reason, + * path and line. */ +#ifdef CBM_ENABLE_TEST_SEAMS +static atomic_int mcp_doc_links_sample_alloc_countdown = ATOMIC_VAR_INIT(CBM_NOT_FOUND); +static atomic_bool mcp_doc_links_sample_alloc_failed = ATOMIC_VAR_INIT(false); + +void cbm_mcp_doc_links_test_fail_sample_alloc_after(int successful_copies) { + atomic_store_explicit(&mcp_doc_links_sample_alloc_failed, false, memory_order_relaxed); + atomic_store_explicit(&mcp_doc_links_sample_alloc_countdown, + successful_copies < 0 ? CBM_NOT_FOUND : successful_copies, + memory_order_relaxed); +} + +bool cbm_mcp_doc_links_test_sample_alloc_failed(void) { + return atomic_load_explicit(&mcp_doc_links_sample_alloc_failed, memory_order_relaxed); +} +#endif + +/* The seam rejects an actual sample string allocation in yyjson's pool, + * independently of when that pool needs another backing allocation. */ +static yyjson_mut_val *doc_links_sample_string(yyjson_mut_doc *doc, const char *source, + size_t length) { + if (!doc || !source) { + return NULL; + } +#ifdef CBM_ENABLE_TEST_SEAMS + int remaining = + atomic_load_explicit(&mcp_doc_links_sample_alloc_countdown, memory_order_relaxed); + while (remaining >= 0) { + int next = remaining == 0 ? CBM_NOT_FOUND : remaining - 1; + if (atomic_compare_exchange_weak_explicit(&mcp_doc_links_sample_alloc_countdown, &remaining, + next, memory_order_relaxed, + memory_order_relaxed)) { + if (remaining == 0) { + atomic_store_explicit(&mcp_doc_links_sample_alloc_failed, true, + memory_order_relaxed); + return NULL; + } + break; + } + } +#endif + return yyjson_mut_strncpy(doc, source, length); +} + +typedef struct { + char text[CBM_DOC_LINK_PREVIEW_LONG_BYTES + 1]; + size_t length; + size_t included; + bool truncated; + bool escaped; +} doc_link_preview_t; + +/* A strict scalar decoder. Invalid input is represented one byte at a time + * by the caller, including incomplete sequences at the end of the source. */ +static size_t doc_link_scalar(const unsigned char *text, size_t length, uint32_t *scalar) { + if (!length) { + return 0; + } + unsigned char first = text[0]; + if (first < 0x80) { + *scalar = first; + return 1; + } + size_t width; + uint32_t value; + if (first >= 0xC2 && first <= 0xDF) { + width = 2; + value = first & 0x1F; + } else if (first >= 0xE0 && first <= 0xEF) { + width = 3; + value = first & 0x0F; + } else if (first >= 0xF0 && first <= 0xF4) { + width = 4; + value = first & 0x07; + } else { + return 0; + } + if (length < width || (first == 0xE0 && text[1] < 0xA0) || (first == 0xED && text[1] > 0x9F) || + (first == 0xF0 && text[1] < 0x90) || (first == 0xF4 && text[1] > 0x8F)) { + return 0; + } + for (size_t i = 1; i < width; i++) { + if ((text[i] & 0xC0) != 0x80) { + return 0; + } + value = (value << 6) | (text[i] & 0x3F); + } + *scalar = value; + return width; +} + +static bool doc_link_visible_escape(uint32_t scalar) { + return scalar < 0x20 || scalar == 0x7F || (scalar >= 0x200B && scalar <= 0x200F) || + (scalar >= 0x202A && scalar <= 0x202E) || (scalar >= 0x2066 && scalar <= 0x2069) || + (scalar >= 0xE0000 && scalar <= 0xE007F); +} + +/* Output tokens are complete scalars or visible escapes. Lookahead bytes do + * not count as included source, and the shared transport encoder is not used. */ +static bool doc_link_format_preview(const cbm_doc_link_preview_text_t *source, size_t budget, + doc_link_preview_t *out) { + memset(out, 0, sizeof(*out)); + if (!source->text || budget >= sizeof(out->text) || + source->length > budget + CBM_DOC_LINK_PREVIEW_LOOKAHEAD || + source->original_bytes < source->length) { + return false; + } + const unsigned char *text = (const unsigned char *)source->text; + bool reserved = (source->length >= 7 && memcmp(text, "@bytes:", 7) == 0) || + (source->length >= 6 && memcmp(text, "@utf8:", 6) == 0); + while (out->included < source->length && out->length < budget) { + const unsigned char *at = text + out->included; + uint32_t scalar = 0; + size_t consumed = doc_link_scalar(at, source->length - out->included, &scalar); + char escape[16]; + const char *token = (const char *)at; + size_t bytes = consumed; + bool escaped = false; + if (!consumed) { + int written = snprintf(escape, sizeof(escape), "\\x%02X", (unsigned int)at[0]); + if (written < 0 || (size_t)written >= sizeof(escape)) { + return false; + } + consumed = 1; + bytes = (size_t)written; + token = escape; + escaped = true; + } else if (scalar == '\\') { + token = "\\\\"; + bytes = 2; + escaped = true; + } else if (doc_link_visible_escape(scalar) || (out->included == 0 && reserved)) { + int written = snprintf(escape, sizeof(escape), "\\u{%04X}", (unsigned int)scalar); + if (written < 0 || (size_t)written >= sizeof(escape)) { + return false; + } + bytes = (size_t)written; + token = escape; + escaped = true; + } + if (bytes > budget - out->length) { + break; + } + memcpy(out->text + out->length, token, bytes); + out->length += bytes; + out->included += consumed; + out->escaped = out->escaped || escaped; + } + out->text[out->length] = '\0'; + out->truncated = out->included < source->original_bytes; + return true; +} + +/* Neither array is attached to the report until every row and metadata entry + * has been constructed. The JSON document owns all intermediate allocations. */ +static bool doc_link_preview_samples(yyjson_mut_doc *doc, const cbm_doc_link_preview_row_t *rows, + int count, yyjson_mut_val **samples_out, + yyjson_mut_val **metadata_out) { + yyjson_mut_val *samples = yyjson_mut_arr(doc); + yyjson_mut_val *metadata = yyjson_mut_arr(doc); + if (!samples || !metadata) { + return false; + } + static const char *const names[] = {"rel_path", "syntax", "raw", "reason"}; + int emitted = 0; + for (int i = 0; i < count; i++) { + if (!rows[i].rel_path.original_bytes) { + continue; + } + const cbm_doc_link_preview_text_t *fields[] = {&rows[i].rel_path, &rows[i].syntax, + &rows[i].raw, &rows[i].reason}; + yyjson_mut_val *sample = yyjson_mut_obj(doc); + if (!sample) { + return false; + } + for (int field = 0; field < 4; field++) { + size_t budget = field == 0 || field == 2 ? CBM_DOC_LINK_PREVIEW_LONG_BYTES + : CBM_DOC_LINK_PREVIEW_SHORT_BYTES; + doc_link_preview_t preview; + if (!doc_link_format_preview(fields[field], budget, &preview)) { + return false; + } + yyjson_mut_val *value = doc_links_sample_string(doc, preview.text, preview.length); + if (!value || !yyjson_mut_obj_add_val(doc, sample, names[field], value) || + (field == 0 && !yyjson_mut_obj_add_int(doc, sample, "line", rows[i].line))) { + return false; + } + if (preview.truncated || preview.escaped) { + yyjson_mut_val *entry = yyjson_mut_obj(doc); + if (!entry || !yyjson_mut_obj_add_int(doc, entry, "sample_index", emitted) || + !yyjson_mut_obj_add_str(doc, entry, "field", names[field]) || + !yyjson_mut_obj_add_uint(doc, entry, "original_bytes", + fields[field]->original_bytes) || + !yyjson_mut_obj_add_uint(doc, entry, "included_source_bytes", + preview.included) || + !yyjson_mut_obj_add_bool(doc, entry, "truncated", preview.truncated) || + !yyjson_mut_obj_add_bool(doc, entry, "escaped", preview.escaped) || + !yyjson_mut_arr_append(metadata, entry)) { + return false; + } + } + } + if (!yyjson_mut_arr_append(samples, sample)) { + return false; + } + emitted++; + } + *samples_out = samples; + *metadata_out = yyjson_mut_arr_size(metadata) ? metadata : NULL; + return true; +} + +static bool add_doc_links_report(yyjson_mut_doc *doc, yyjson_mut_val *root, cbm_store_t *store, + const char *project, bool with_samples) { + int mentions = cbm_store_count_edges_by_type(store, project, "MENTIONS"); + cbm_doc_link_reason_count_t *reasons = NULL; + int nreasons = 0; + cbm_doc_link_row_t *unused_samples = NULL; + int unused_count = 0; + bool present = false; + int rc = cbm_store_doc_links_summary(store, project, &reasons, &nreasons, &unused_samples, + &unused_count, 0, &present); + cbm_doc_link_preview_row_t *rows = NULL; + int count = 0; + yyjson_mut_val *samples = NULL; + yyjson_mut_val *metadata = NULL; + bool sample_failed = with_samples && rc != CBM_STORE_OK; + if (with_samples && rc == CBM_STORE_OK) { + bool preview_present = false; + int preview_rc = + cbm_store_doc_links_preview(store, project, &rows, &count, &preview_present); + sample_failed = preview_rc != CBM_STORE_OK || preview_present != present; + if (!sample_failed) { + sample_failed = !doc_link_preview_samples(doc, rows, count, &samples, &metadata); + } + } + bool error = rc != CBM_STORE_OK || mentions < 0 || !present || sample_failed; + bool built = false; + yyjson_mut_val *dl = yyjson_mut_obj(doc); + yyjson_mut_val *unresolved = yyjson_mut_obj(doc); + if (!dl || !unresolved) { + goto cleanup; + } + for (int i = 0; i < nreasons; i++) { + if (strcmp(reasons[i].reason, "error") == 0) { + error = true; + continue; + } + yyjson_mut_val *key = yyjson_mut_strcpy(doc, reasons[i].reason); + yyjson_mut_val *number = yyjson_mut_int(doc, reasons[i].count); + if (!key || !number || !yyjson_mut_obj_add(unresolved, key, number)) { + goto cleanup; + } + } + if (!yyjson_mut_obj_add_int(doc, dl, "mentions", mentions < 0 ? 0 : mentions) || + !yyjson_mut_obj_add_val(doc, dl, "unresolved", unresolved) || + !yyjson_mut_obj_add_str(doc, dl, "status", error ? "error" : "ok")) { + goto cleanup; + } + if (error) { + const char *hint = + sample_failed && rc == CBM_STORE_OK + ? "Doc-link samples could not be prepared; retry index_status." + : (!present && rc == CBM_STORE_OK + ? "This index predates doc-comment links; re-run " + "index_repository(repo_path=...)." + : "The doc-link layer failed or its table could not be read; re-run " + "index_repository(repo_path=...)."); + if (!yyjson_mut_obj_add_str(doc, dl, "hint", hint)) { + goto cleanup; + } + } + if (with_samples && !sample_failed) { + if (!yyjson_mut_obj_add_val(doc, dl, "samples", samples)) { + goto cleanup; + } + if (metadata && + (!yyjson_mut_obj_add_val(doc, dl, "samples_preview", metadata) || + !yyjson_mut_obj_add_str(doc, dl, "samples_preview_note", + "Sample text is shown as previews; complete values remain " + "in doc_link_unresolved."))) { + goto cleanup; + } + } + /* A failed append leaves this whole report unattached. The caller discards + * the document and uses its existing MCP allocation-error response. */ + built = yyjson_mut_obj_add_val(doc, root, "doc_links", dl); +cleanup: + if (error && (rc != CBM_STORE_OK || sample_failed)) { + cbm_log_warn("index_status.doc_links", "project", project, "reason", "read_failed"); + } + cbm_store_free_doc_link_reasons(reasons, nreasons); + cbm_store_free_doc_links(unused_samples, unused_count); + cbm_store_free_doc_link_previews(rows, count); + return built; +} + /* Indexing-coverage report (#963), attached to index_status: the best-effort * signal from the separate index_coverage table (coverage is metadata ABOUT * the graph, stored outside it). Full per-project list, capped generously. */ @@ -6872,7 +7175,12 @@ static char *handle_index_status(cbm_mcp_server_t *srv, const char *args) { free(diagnostics); yyjson_mut_doc *doc = yyjson_mut_doc_new(NULL); - yyjson_mut_val *root = yyjson_mut_obj(doc); + yyjson_mut_val *root = doc ? yyjson_mut_obj(doc) : NULL; + if (!root) { + yyjson_mut_doc_free(doc); + free(project); + return mcp_result_from_json(args, NULL); + } yyjson_mut_doc_set_root(doc, root); if (project) { @@ -6902,9 +7210,16 @@ static char *handle_index_status(cbm_mcp_server_t *srv, const char *args) { } add_coverage_report(doc, root, store, project, have_proj_info ? proj_info.indexed_at : NULL, coverage_samples); + bool doc_links_built = + add_doc_links_report(doc, root, store, project, coverage_samples == COVERAGE_FILE_CAP); safe_str_free(&proj_info.name); safe_str_free(&proj_info.indexed_at); safe_str_free(&proj_info.root_path); + if (!doc_links_built) { + yyjson_mut_doc_free(doc); + free(project); + return mcp_result_from_json(args, NULL); + } if (counts_unreadable) { const char *hint; if (nodes < 0 && edges < 0) { diff --git a/src/mcp/mcp_internal.h b/src/mcp/mcp_internal.h index 784e071456..ead1fca389 100644 --- a/src/mcp/mcp_internal.h +++ b/src/mcp/mcp_internal.h @@ -13,6 +13,11 @@ typedef bool (*cbm_mcp_quarantine_test_hook_fn)(void *context, const char *step); typedef bool (*cbm_mcp_command_test_hook_fn)(void *context, const char *command); #ifdef CBM_ENABLE_TEST_SEAMS +/* One-shot sample JSON string-copy allocation failure, not a backing-pool + * growth event: 0 is next, 4 is fifth, -1 disables. Setting clears consumption. */ +void cbm_mcp_doc_links_test_fail_sample_alloc_after(int successful_copies); +bool cbm_mcp_doc_links_test_sample_alloc_failed(void); + typedef void (*cbm_mcp_auto_index_count_test_hook_fn)(void *context); #endif diff --git a/src/pipeline/doc_links.c b/src/pipeline/doc_links.c new file mode 100644 index 0000000000..5de452e337 --- /dev/null +++ b/src/pipeline/doc_links.c @@ -0,0 +1,789 @@ +/* + * doc_links.c — doc-comment references -> MENTIONS edges: the language- + * independent core (run state, per-file resolution, edge collapse, rows). + * See doc_links.h; the C# resolver is doc_links_cs.c. + */ +#include "pipeline/doc_links.h" + +#include "doclink.h" +#include "foundation/arena.h" +#include "foundation/compat.h" /* cbm_clock_gettime */ +#include "foundation/constants.h" +#include "foundation/log.h" +#include "foundation/mem_core.h" +#include "result_spill.h" /* a parked result's header: is there a scope to read back? */ +#include "yyjson/yyjson.h" + +#include +#include +#include +#include +#include + +/* ── Reasons ─────────────────────────────────────────────────────── */ + +static const char *const DOCLINK_REASON_NAMES[CBM_DOCLINK_REASON_COUNT] = { + [CBM_DOCLINK_REASON_MISSING] = "missing", + [CBM_DOCLINK_REASON_AMBIGUOUS] = "ambiguous", + [CBM_DOCLINK_REASON_EXTERNAL] = "external", + [CBM_DOCLINK_REASON_TEST_ONLY] = "test_only_target", + [CBM_DOCLINK_REASON_NOT_INDEXED] = "not_indexed", + [CBM_DOCLINK_REASON_GRAPH_GAP] = "graph_gap", + [CBM_DOCLINK_REASON_UNPARSEABLE] = "unparseable", + [CBM_DOCLINK_REASON_BELOW_BAR] = "below_bar_tier", +}; + +const char *cbm_doclink_reason_name(int reason) { + if (reason < 0 || reason >= CBM_DOCLINK_REASON_COUNT) { + return "missing"; + } + return DOCLINK_REASON_NAMES[reason]; +} + +/* ── Resolver table ──────────────────────────────────────────────── */ + +/* One pointer per language with a resolver: the one line a language leg adds + * to this file. */ +static const cbm_doclink_resolver_t *const DOCLINK_RESOLVERS[] = { + &cbm_doclink_cs_resolver, +}; + +enum { DOCLINK_RESOLVER_COUNT = sizeof(DOCLINK_RESOLVERS) / sizeof(DOCLINK_RESOLVERS[0]) }; + +static bool resolver_has_lang(const cbm_doclink_resolver_t *R, CBMLanguage lang) { + for (int i = 0; i < R->lang_count && i < CBM_DOCLINK_RESOLVER_LANGS; i++) { + if (R->langs[i] == lang) { + return true; + } + } + return false; +} + +static int resolver_slot(CBMLanguage lang) { + for (int i = 0; i < DOCLINK_RESOLVER_COUNT; i++) { + if (resolver_has_lang(DOCLINK_RESOLVERS[i], lang)) { + return i; + } + } + return CBM_NOT_FOUND; +} + +/* True when the scope blob's tag line is `tag`. */ +static bool scope_has_tag(const char *scope, const char *tag) { + if (!scope || !tag) { + return false; + } + size_t tl = strlen(tag); + return strncmp(scope, tag, tl) == 0 && (scope[tl] == '\n' || scope[tl] == '\0'); +} + +/* The resolver whose scope blobs carry this blob's tag line; NULL for none. */ +static const cbm_doclink_resolver_t *resolver_of_scope(const char *scope) { + for (int i = 0; i < DOCLINK_RESOLVER_COUNT; i++) { + if (scope_has_tag(scope, DOCLINK_RESOLVERS[i]->scope_tag)) { + return DOCLINK_RESOLVERS[i]; + } + } + return NULL; +} + +/* ── Incremental scope rules ─────────────────────────────────────── */ + +bool cbm_doclinks_is_scope_input(const char *rel_path) { + for (int i = 0; rel_path && i < DOCLINK_RESOLVER_COUNT; i++) { + if (DOCLINK_RESOLVERS[i]->scope_input && DOCLINK_RESOLVERS[i]->scope_input(rel_path)) { + return true; + } + } + return false; +} + +int cbm_doclinks_scope_delta(const char *stored, const char *fresh, cbm_doclink_name_fn removed, + void *ud) { + if (!stored && !fresh) { + return CBM_DOCLINK_DELTA_LOCAL; + } + if (!stored || !fresh) { + return CBM_DOCLINK_DELTA_GLOBAL; /* a scope appeared, or left with its file */ + } + const cbm_doclink_resolver_t *R = resolver_of_scope(stored); + if (!R || R != resolver_of_scope(fresh) || !R->scope_delta) { + return CBM_DOCLINK_DELTA_GLOBAL; + } + return R->scope_delta(stored, fresh, removed, ud); +} + +/* ── Run state ───────────────────────────────────────────────────── */ + +typedef struct { + cbm_doc_link_row_t *items; + int count; +} doclink_file_rows_t; + +struct cbm_doclinks { + const cbm_file_info_t *files; /* borrowed: the run's file list */ + int file_count; + const char *project; /* borrowed: names the File node of a file-level source */ + void *index[DOCLINK_RESOLVER_COUNT]; + doclink_file_rows_t *rows; /* per run file */ + CBMArena arena; /* scope copies handed to the resolvers */ + _Atomic bool failed; + _Atomic int64_t edges; + _Atomic int64_t mentions; + _Atomic int64_t self_mentions; + _Atomic int64_t local_refs; + _Atomic int64_t no_source; + _Atomic int64_t reasons[CBM_DOCLINK_REASON_COUNT]; + /* what the layer cost, for the doc_links.timing log line (never a gate) */ + int64_t build_ns; /* the per-language indexes */ + _Atomic int64_t resolve_ns; /* summed over the files, i.e. over the workers */ + _Atomic int64_t resolved_files; /* files that had references */ + _Atomic int64_t resolved_tokens; /* references handed to a resolver */ +}; + +static const char *itoa64(int64_t v, char *buf, size_t n) { + snprintf(buf, n, "%lld", (long long)v); + return buf; +} + +static int64_t doclinks_now_ns(void) { + struct timespec ts; + cbm_clock_gettime(CLOCK_MONOTONIC, &ts); + return ((int64_t)ts.tv_sec * 1000000000LL) + (int64_t)ts.tv_nsec; +} + +static bool want_doc_scope(const CBMFileResult *header) { + return header->doc_scope != NULL; /* a parked header's pointer is only a presence bit */ +} + +static int file_cmp(const void *a, const void *b) { + return strcmp(((const cbm_doclink_file_t *)a)->rel_path, + ((const cbm_doclink_file_t *)b)->rel_path); +} + +/* True when file `i`'s result is parked on disk with a scope: an acquire that + * handed nothing out then means a failed read, not "no scope". A header that + * cannot be peeked counts as one. */ +static bool parked_scope_unread(const cbm_pipeline_ctx_t *ctx, CBMFileResult **cache, int i) { + if ((cache && cache[i]) || !ctx || !ctx->spill || !cbm_result_spill_has(ctx->spill, i)) { + return false; + } + CBMFileResult header; + return !cbm_result_spill_peek_header(ctx->spill, i, &header) || want_doc_scope(&header); +} + +/* The scope of run file `i`, copied into the run's arena; NULL when the file + * has none. *why is set when it has one that cannot be had (the copy fails, + * or its parked result does not load): a scope dropped silently would take + * the file's declarations out of the index, and references to them would + * read as missing. */ +static const char *run_file_scope(cbm_doclinks_t *dl, const cbm_pipeline_ctx_t *ctx, + CBMFileResult **cache, int i, const char **why) { + bool loaded = false; + CBMFileResult *r = cbm_pipeline_result_acquire(ctx, cache, i, want_doc_scope, &loaded); + const char *scope = NULL; + if (r && r->doc_scope) { + scope = cbm_arena_strdup(&dl->arena, r->doc_scope); + if (!scope) { + *why = "alloc"; + } + } else if (!r && parked_scope_unread(ctx, cache, i)) { + *why = "scope_unreadable"; + } + cbm_pipeline_result_release(r, loaded); + return scope; +} + +/* Build one language's index over this run's files of that language plus the + * base scopes tagged for it. Returns false, with *why, when a scope cannot be + * read or copied or the index cannot be built: the caller fails the layer. */ +static bool build_language(cbm_doclinks_t *dl, int slot, const cbm_pipeline_ctx_t *ctx, + const cbm_file_info_t *files, int file_count, CBMFileResult **cache, + const cbm_doclink_scope_t *base, int base_count, const cbm_gbuf_t *graph, + const char **why) { + const cbm_doclink_resolver_t *R = DOCLINK_RESOLVERS[slot]; + int cap = 0; + for (int i = 0; i < file_count; i++) { + cap += resolver_has_lang(R, files[i].language); + } + for (int i = 0; i < base_count; i++) { + cap += scope_has_tag(base[i].scope, R->scope_tag); + } + if (cap == 0) { + return true; + } + cbm_doclink_file_t *lf = + (cbm_doclink_file_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, (size_t)cap * sizeof(*lf)); + if (!lf) { + *why = "alloc"; + return false; + } + const char *failed = NULL; + int n = 0; + for (int i = 0; i < file_count; i++) { + if (!resolver_has_lang(R, files[i].language)) { + continue; + } + const char *scope = run_file_scope(dl, ctx, cache, i, &failed); + lf[n++] = + (cbm_doclink_file_t){.rel_path = files[i].rel_path, .scope = scope, .run_file = i}; + } + for (int i = 0; i < base_count; i++) { + if (!scope_has_tag(base[i].scope, R->scope_tag)) { + continue; + } + const char *rel_path = cbm_arena_strdup(&dl->arena, base[i].rel_path); + const char *scope = cbm_arena_strdup(&dl->arena, base[i].scope); + if (!rel_path || !scope) { + failed = "alloc"; + } + lf[n++] = + (cbm_doclink_file_t){.rel_path = rel_path, .scope = scope, .run_file = CBM_NOT_FOUND}; + } + if (failed) { + cbm_free(CBM_MEM_CLASS_OTHER, lf); + *why = failed; + return false; + } + qsort(lf, (size_t)n, sizeof(*lf), file_cmp); + cbm_doclink_build_in_t in = { + .ctx = ctx, .graph = graph, .files = lf, .file_count = n, .run_file_count = file_count}; + dl->index[slot] = R->build(&in); + cbm_free(CBM_MEM_CLASS_OTHER, lf); + if (!dl->index[slot]) { + *why = "index"; + } + return dl->index[slot] != NULL; +} + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +static atomic_bool doclinks_test_fail_build; + +void cbm_doclinks_test_fail_build_once(void) { + atomic_store(&doclinks_test_fail_build, true); +} +#endif + +cbm_doclinks_t *cbm_doclinks_build(const cbm_pipeline_ctx_t *ctx, const cbm_file_info_t *files, + int file_count, CBMFileResult **cache, + const cbm_doclink_scope_t *base, int base_count, + const cbm_gbuf_t *graph) { +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (atomic_exchange(&doclinks_test_fail_build, false)) { + cbm_log_error("doc_links.error", "phase", "build", "reason", "alloc"); + return NULL; + } +#endif + cbm_doclinks_t *dl = (cbm_doclinks_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*dl)); + if (!dl) { + cbm_log_error("doc_links.error", "phase", "build", "reason", "alloc"); + return NULL; + } + dl->files = files; + dl->file_count = file_count; + dl->project = ctx ? ctx->project_name : NULL; + dl->rows = file_count > 0 + ? (doclink_file_rows_t *)cbm_calloc( + CBM_MEM_CLASS_OTHER, (size_t)file_count * sizeof(doclink_file_rows_t)) + : NULL; + cbm_arena_init(&dl->arena); + if (file_count > 0 && !dl->rows) { + cbm_log_error("doc_links.error", "phase", "build", "reason", "alloc"); + cbm_doclinks_free(dl); + return NULL; + } + int64_t started = doclinks_now_ns(); + for (int s = 0; s < DOCLINK_RESOLVER_COUNT; s++) { + const char *why = "alloc"; + if (!build_language(dl, s, ctx, files, file_count, cache, base, base_count, graph, &why)) { + cbm_log_error("doc_links.error", "phase", "index_build", "reason", why); + cbm_doclinks_free(dl); + return NULL; + } + } + dl->build_ns = doclinks_now_ns() - started; + return dl; +} + +/* ── Per-file resolution ─────────────────────────────────────────── */ + +typedef struct { + int64_t src; + int64_t tgt; + uint32_t line; + uint16_t syntax; + bool exact; +} doclink_mention_t; + +static int mention_cmp(const void *a, const void *b) { + const doclink_mention_t *x = (const doclink_mention_t *)a; + const doclink_mention_t *y = (const doclink_mention_t *)b; + if (x->src != y->src) { + return x->src < y->src ? -1 : 1; + } + if (x->tgt != y->tgt) { + return x->tgt < y->tgt ? -1 : 1; + } + if (x->line != y->line) { + return x->line < y->line ? -1 : 1; + } + return (int)x->syntax - (int)y->syntax; +} + +/* Row memory is CBM_MEM_CLASS_STORE everywhere (the store reads and frees the + * same rows): cbm_store_free_doc_links is its one release. */ +static char *dup_or_empty(const char *s) { + return cbm_mem_strdup(CBM_MEM_CLASS_STORE, s ? s : ""); +} + +static void mark_failed(cbm_doclinks_t *dl, const char *rel, const char *why) { + if (!atomic_exchange(&dl->failed, true)) { + cbm_log_error("doc_links.error", "phase", "resolve", "path", rel ? rel : "", "reason", why); + } +} + +/* Emit one MENTIONS edge per (source, target): first line, its syntax, the + * mention count, and tier exact when any mention bound exactly. */ +static void emit_mentions(cbm_doclinks_t *dl, doclink_mention_t *m, int n, cbm_gbuf_t *edge_out) { + qsort(m, (size_t)n, sizeof(*m), mention_cmp); + int i = 0; + while (i < n) { + int j = i; + bool exact = false; + while (j < n && m[j].src == m[i].src && m[j].tgt == m[i].tgt) { + exact = exact || m[j].exact; + j++; + } + char props[CBM_SZ_256]; + snprintf(props, sizeof(props), + "{\"via\":\"doc_comment\",\"syntax\":\"%s\",\"tier\":\"%s\",\"line\":%u," + "\"count\":%d}", + cbm_doclink_syntax_name(m[i].syntax), exact ? "exact" : "unique", m[i].line, + j - i); + cbm_gbuf_insert_edge(edge_out, m[i].src, m[i].tgt, "MENTIONS", props); + atomic_fetch_add_explicit(&dl->edges, 1, memory_order_relaxed); + i = j; + } +} + +void cbm_doclinks_resolve_file(cbm_doclinks_t *dl, int file_idx, const CBMFileResult *result, + const cbm_gbuf_t *graph, cbm_gbuf_t *edge_out) { + if (!dl || !result || file_idx < 0 || file_idx >= dl->file_count || + result->doc_links.count == 0 || !result->doc_links.items) { + return; + } + int64_t started = doclinks_now_ns(); + const cbm_file_info_t *fi = &dl->files[file_idx]; + int slot = resolver_slot(fi->language); + const cbm_doclink_resolver_t *R = slot >= 0 ? DOCLINK_RESOLVERS[slot] : NULL; + const void *index = slot >= 0 ? dl->index[slot] : NULL; + int n = result->doc_links.count; + doclink_mention_t *mentions = + (doclink_mention_t *)cbm_alloc(CBM_MEM_CLASS_OTHER, (size_t)n * sizeof(*mentions)); + cbm_doc_link_row_t *rows = + (cbm_doc_link_row_t *)cbm_calloc(CBM_MEM_CLASS_STORE, (size_t)n * sizeof(*rows)); + if (!mentions || !rows) { + cbm_free(CBM_MEM_CLASS_OTHER, mentions); + cbm_free(CBM_MEM_CLASS_STORE, rows); + mark_failed(dl, fi->rel_path, "alloc"); + return; + } + int nm = 0; + int nr = 0; + /* the file's own node: looked up once, and only when a reference of the + * file-level doc resolves */ + const cbm_gbuf_node_t *file_node = NULL; + bool file_node_looked_up = false; + for (int i = 0; i < n; i++) { + const CBMDocLink *link = &result->doc_links.items[i]; + cbm_doclink_outcome_t out = {.kind = CBM_DOCLINK_UNRESOLVED, + .reason = CBM_DOCLINK_REASON_MISSING}; + if (cbm_doclink_syntax_is_external(link->syntax)) { + out.reason = CBM_DOCLINK_REASON_EXTERNAL; + } else if (!R || !index) { + continue; /* no resolver for this language: nothing to say */ + } else { + R->resolve(index, file_idx, link, graph, &out); + } + if (out.kind == CBM_DOCLINK_LOCAL) { + atomic_fetch_add_explicit(&dl->local_refs, 1, memory_order_relaxed); + continue; + } + if (out.kind == CBM_DOCLINK_EDGE && out.target) { + const cbm_gbuf_node_t *src = NULL; + if (link->flags & CBM_DOCLINK_FLAG_FILE) { + /* written in the file's own doc: the source is the File node */ + if (!file_node_looked_up) { + file_node = cbm_pipeline_file_node(graph, dl->project, fi->rel_path); + file_node_looked_up = true; + } + src = file_node; + } else { + src = cbm_gbuf_find_by_qn(graph, link->source_qn); + } + if (!src) { + atomic_fetch_add_explicit(&dl->no_source, 1, memory_order_relaxed); + continue; + } + if (src->id == out.target->id) { + atomic_fetch_add_explicit(&dl->mentions, 1, memory_order_relaxed); + atomic_fetch_add_explicit(&dl->self_mentions, 1, memory_order_relaxed); + continue; + } + if (cbm_doclink_syntax_ships(link->syntax)) { + atomic_fetch_add_explicit(&dl->mentions, 1, memory_order_relaxed); + mentions[nm++] = (doclink_mention_t){.src = src->id, + .tgt = out.target->id, + .line = link->line, + .syntax = link->syntax, + .exact = out.exact}; + continue; + } + /* the ship gate: resolved, but this link family is below the + * bar -- a row, not an edge */ + out.reason = CBM_DOCLINK_REASON_BELOW_BAR; + } + int reason = out.reason; + if (reason < 0 || reason >= CBM_DOCLINK_REASON_COUNT) { + reason = CBM_DOCLINK_REASON_MISSING; + } + cbm_doc_link_row_t *row = &rows[nr]; + row->rel_path = dup_or_empty(fi->rel_path); + row->line = (int)link->line; + row->syntax = dup_or_empty(cbm_doclink_syntax_name(link->syntax)); + row->raw = dup_or_empty(link->raw); + row->reason = dup_or_empty(cbm_doclink_reason_name(reason)); + nr++; + if (!row->rel_path || !row->syntax || !row->raw || !row->reason) { + mark_failed(dl, fi->rel_path, "alloc"); + break; + } + atomic_fetch_add_explicit(&dl->reasons[reason], 1, memory_order_relaxed); + } + if (nm > 0) { + emit_mentions(dl, mentions, nm, edge_out); + } + cbm_free(CBM_MEM_CLASS_OTHER, mentions); + atomic_fetch_add_explicit(&dl->resolve_ns, doclinks_now_ns() - started, memory_order_relaxed); + atomic_fetch_add_explicit(&dl->resolved_files, 1, memory_order_relaxed); + atomic_fetch_add_explicit(&dl->resolved_tokens, n, memory_order_relaxed); + if (nr == 0) { + cbm_free(CBM_MEM_CLASS_STORE, rows); + return; + } + dl->rows[file_idx].items = rows; + dl->rows[file_idx].count = nr; +} + +/* ── Rows ────────────────────────────────────────────────────────── */ + +void cbm_doclinks_take_rows(cbm_doclinks_t *dl, cbm_doc_link_row_t **rows, int *count, + bool *failed) { + *rows = NULL; + *count = 0; + if (failed) { + *failed = dl ? atomic_load(&dl->failed) : true; + } + if (!dl) { + return; + } + int total = 0; + for (int i = 0; i < dl->file_count; i++) { + total += dl->rows[i].count; + } + cbm_doc_link_row_t *all = + total > 0 + ? (cbm_doc_link_row_t *)cbm_alloc(CBM_MEM_CLASS_STORE, (size_t)total * sizeof(*all)) + : NULL; + if (total > 0 && !all) { + mark_failed(dl, NULL, "alloc"); + if (failed) { + *failed = true; + } + return; /* the per-file rows are released by cbm_doclinks_free */ + } + int w = 0; + for (int i = 0; i < dl->file_count; i++) { + if (dl->rows[i].count > 0) { + memcpy(all + w, dl->rows[i].items, (size_t)dl->rows[i].count * sizeof(*all)); + w += dl->rows[i].count; + } + cbm_free(CBM_MEM_CLASS_STORE, dl->rows[i].items); + dl->rows[i].items = NULL; + dl->rows[i].count = 0; + } + *rows = all; + *count = w; + char b[8][CBM_SZ_32]; + cbm_log_info("doc_links.done", "edges", itoa64(atomic_load(&dl->edges), b[0], sizeof(b[0])), + "mentions", itoa64(atomic_load(&dl->mentions), b[1], sizeof(b[1])), "self", + itoa64(atomic_load(&dl->self_mentions), b[2], sizeof(b[2])), "local", + itoa64(atomic_load(&dl->local_refs), b[3], sizeof(b[3])), "unresolved", + itoa64(w, b[4], sizeof(b[4])), "no_source", + itoa64(atomic_load(&dl->no_source), b[5], sizeof(b[5]))); + cbm_log_info( + "doc_links.unresolved", "missing", + itoa64(atomic_load(&dl->reasons[CBM_DOCLINK_REASON_MISSING]), b[0], sizeof(b[0])), + "ambiguous", + itoa64(atomic_load(&dl->reasons[CBM_DOCLINK_REASON_AMBIGUOUS]), b[1], sizeof(b[1])), + "external", + itoa64(atomic_load(&dl->reasons[CBM_DOCLINK_REASON_EXTERNAL]), b[2], sizeof(b[2])), + "test_only_target", + itoa64(atomic_load(&dl->reasons[CBM_DOCLINK_REASON_TEST_ONLY]), b[3], sizeof(b[3])), + "graph_gap", + itoa64(atomic_load(&dl->reasons[CBM_DOCLINK_REASON_GRAPH_GAP]), b[4], sizeof(b[4])), + "unparseable", + itoa64(atomic_load(&dl->reasons[CBM_DOCLINK_REASON_UNPARSEABLE]), b[5], sizeof(b[5])), + "not_indexed", + itoa64(atomic_load(&dl->reasons[CBM_DOCLINK_REASON_NOT_INDEXED]), b[6], sizeof(b[6])), + "below_bar_tier", + itoa64(atomic_load(&dl->reasons[CBM_DOCLINK_REASON_BELOW_BAR]), b[7], sizeof(b[7]))); + /* resolve_cpu_ms is the time inside the resolvers added up over the + * files: with N workers the wall share is about a N-th of it */ + enum { NS_PER_MS = 1000000 }; + cbm_log_info("doc_links.timing", "index_build_ms", + itoa64(dl->build_ns / NS_PER_MS, b[0], sizeof(b[0])), "resolve_cpu_ms", + itoa64(atomic_load(&dl->resolve_ns) / NS_PER_MS, b[1], sizeof(b[1])), "files", + itoa64(atomic_load(&dl->resolved_files), b[2], sizeof(b[2])), "references", + itoa64(atomic_load(&dl->resolved_tokens), b[3], sizeof(b[3]))); +} + +void cbm_doclinks_free_rows(cbm_doc_link_row_t *rows, int count) { + cbm_store_free_doc_links(rows, count); +} + +void cbm_doclinks_free(cbm_doclinks_t *dl) { + if (!dl) { + return; + } + for (int s = 0; s < DOCLINK_RESOLVER_COUNT; s++) { + if (dl->index[s]) { + DOCLINK_RESOLVERS[s]->destroy(dl->index[s]); + } + } + if (dl->rows) { + for (int i = 0; i < dl->file_count; i++) { + cbm_doclinks_free_rows(dl->rows[i].items, dl->rows[i].count); + } + cbm_free(CBM_MEM_CLASS_OTHER, dl->rows); + } + cbm_arena_destroy(&dl->arena); + cbm_free(CBM_MEM_CLASS_OTHER, dl); +} + +/* ── Pipeline bracket ────────────────────────────────────────────── */ + +void cbm_doclinks_begin(cbm_pipeline_ctx_t *ctx, const cbm_file_info_t *files, int file_count, + CBMFileResult **cache) { + if (!ctx) { + return; + } + ctx->doc_links = cbm_doclinks_build(ctx, files, file_count, cache, ctx->doc_link_base, + ctx->doc_link_base_count, ctx->gbuf); + if (!ctx->doc_links) { + ctx->doc_links_failed = true; /* build logged doc_links.error */ + } +} + +void cbm_doclinks_end(cbm_pipeline_ctx_t *ctx) { + if (!ctx) { + return; + } + cbm_doc_link_row_t *rows = NULL; + int count = 0; + bool failed = ctx->doc_links_failed; + if (ctx->doc_links) { + bool run_failed = false; + cbm_doclinks_take_rows(ctx->doc_links, &rows, &count, &run_failed); + failed = failed || run_failed; + cbm_doclinks_free(ctx->doc_links); + ctx->doc_links = NULL; + } + cbm_pipeline_set_doc_link_rows(ctx->pipeline, rows, count, failed); + ctx->doc_links_failed = false; +} + +int cbm_pipeline_pass_doc_links(cbm_pipeline_ctx_t *ctx, const cbm_file_info_t *files, + int file_count) { + if (!ctx) { + return 0; + } + cbm_doclinks_begin(ctx, files, file_count, ctx->result_cache); + if (file_count > 0 && !ctx->result_cache) { + /* Without the result cache this route has nothing to read back. */ + cbm_log_error("doc_links.error", "phase", "sequential", "reason", "no_result_cache"); + ctx->doc_links_failed = true; + } + for (int i = 0; ctx->doc_links && ctx->result_cache && i < file_count; i++) { + bool loaded = false; + CBMFileResult *r = cbm_pipeline_result_acquire(ctx, ctx->result_cache, i, NULL, &loaded); + if (r) { + cbm_doclinks_resolve_file(ctx->doc_links, i, r, ctx->gbuf, ctx->gbuf); + } + cbm_pipeline_result_release(r, loaded); + } + cbm_doclinks_end(ctx); + return 0; +} + +/* ── Incremental carry-forward ───────────────────────────────────── */ + +static bool copy_row(cbm_doc_link_row_t *dst, const cbm_doc_link_row_t *src) { + dst->rel_path = dup_or_empty(src->rel_path); + dst->line = src->line; + dst->syntax = dup_or_empty(src->syntax); + dst->raw = dup_or_empty(src->raw); + dst->reason = dup_or_empty(src->reason); + return dst->rel_path && dst->syntax && dst->raw && dst->reason; +} + +int cbm_doclinks_merge_rows(const cbm_doc_link_row_t *old_rows, int old_count, + const CBMHashTable *replaced, const cbm_doc_link_row_t *fresh, + int fresh_count, cbm_doc_link_row_t **out, int *out_count) { + *out = NULL; + *out_count = 0; + int cap = old_count + fresh_count; + if (cap == 0) { + return 0; + } + cbm_doc_link_row_t *rows = + (cbm_doc_link_row_t *)cbm_calloc(CBM_MEM_CLASS_STORE, (size_t)cap * sizeof(*rows)); + if (!rows) { + return CBM_NOT_FOUND; + } + int n = 0; + for (int i = 0; i < old_count; i++) { + const cbm_doc_link_row_t *r = &old_rows[i]; + if (!r->rel_path || !r->rel_path[0]) { + continue; /* a previous generation's error marker */ + } + if (replaced && cbm_ht_get(replaced, r->rel_path)) { + continue; /* re-extracted or deleted: the fresh rows replace them */ + } + if (!copy_row(&rows[n++], r)) { + cbm_doclinks_free_rows(rows, n); + return CBM_NOT_FOUND; + } + } + for (int i = 0; i < fresh_count; i++) { + if (!copy_row(&rows[n++], &fresh[i])) { + cbm_doclinks_free_rows(rows, n); + return CBM_NOT_FOUND; + } + } + *out = rows; + *out_count = n; + return 0; +} + +/* ── Stored scopes ───────────────────────────────────────────────── */ + +/* The surface writer appends "dl" as the LAST top-level key, and a JSON + * string cannot contain an unescaped quote, so the last `"dl":"` in the row + * is that key. Only this tail is parsed: a row's "lsp" array is megabytes of + * definitions nobody needs here (the C# bench corpus holds ~1.5 GB of them). */ +int cbm_doclinks_scope_from_surface_json(const char *defs_json, char **out) { + *out = NULL; + if (!defs_json) { + return 0; + } + static const char key[] = "\"dl\":\""; + const size_t kl = sizeof(key) - SKIP_ONE; + size_t n = strlen(defs_json); + const char *hit = NULL; + for (size_t i = n; i >= kl; i--) { + if (defs_json[i - kl] == '"' && memcmp(defs_json + i - kl, key, kl) == 0) { + hit = defs_json + i - kl; + break; + } + } + if (!hit) { + return 0; + } + size_t tail = n - (size_t)(hit - defs_json); + char *buf = (char *)cbm_alloc(CBM_MEM_CLASS_OTHER, tail + PAIR_LEN); + if (!buf) { + return CBM_NOT_FOUND; + } + buf[0] = '{'; + memcpy(buf + SKIP_ONE, hit, tail + SKIP_ONE); + yyjson_doc *doc = yyjson_read(buf, tail + SKIP_ONE, 0); + yyjson_val *root = doc ? yyjson_doc_get_root(doc) : NULL; + const char *scope = root ? yyjson_get_str(yyjson_obj_get(root, "dl")) : NULL; + int rc = 0; + if (!scope) { + rc = CBM_NOT_FOUND; /* the key is there but not readable: a corrupt row */ + } else { + *out = cbm_mem_strdup(CBM_MEM_CLASS_OTHER, scope); + rc = *out ? 0 : CBM_NOT_FOUND; + } + yyjson_doc_free(doc); + cbm_free(CBM_MEM_CLASS_OTHER, buf); + return rc; +} + +int cbm_doclinks_scopes_from_surfaces(const cbm_lsp_surface_row_t *rows, int row_count, + const CBMHashTable *skip, cbm_doclink_scope_t **out, + int *count) { + *out = NULL; + *count = 0; + if (!rows || row_count <= 0) { + return 0; + } + int64_t started = doclinks_now_ns(); + cbm_doclink_scope_t *scopes = NULL; + int n = 0; + int cap = 0; + for (int i = 0; i < row_count; i++) { + const cbm_lsp_surface_row_t *row = &rows[i]; + if (!row->defs_json || !row->rel_path || (skip && cbm_ht_get(skip, row->rel_path))) { + continue; + } + char *scope = NULL; + if (cbm_doclinks_scope_from_surface_json(row->defs_json, &scope) != 0) { + cbm_doclinks_free_scopes(scopes, n); + return CBM_NOT_FOUND; + } + if (!scope) { + continue; /* no scope: a language without a scope scanner */ + } + if (n >= cap) { + int ncap = cap ? cap * PAIR_LEN : CBM_SZ_64; + cbm_doclink_scope_t *grown = (cbm_doclink_scope_t *)cbm_realloc( + CBM_MEM_CLASS_OTHER, scopes, (size_t)ncap * sizeof(*grown)); + if (!grown) { + cbm_free(CBM_MEM_CLASS_OTHER, scope); + cbm_doclinks_free_scopes(scopes, n); + return CBM_NOT_FOUND; + } + scopes = grown; + cap = ncap; + } + scopes[n].rel_path = cbm_mem_strdup(CBM_MEM_CLASS_OTHER, row->rel_path); + scopes[n].scope = scope; + n++; + if (!scopes[n - SKIP_ONE].rel_path) { + cbm_doclinks_free_scopes(scopes, n); + return CBM_NOT_FOUND; + } + } + *out = scopes; + *count = n; + /* an incremental run pays this instead of extracting the files again */ + char b[3][CBM_SZ_32]; + cbm_log_info("doc_links.stored_scopes", "rows", itoa64(row_count, b[0], sizeof(b[0])), "scopes", + itoa64(n, b[1], sizeof(b[1])), "elapsed_ms", + itoa64((doclinks_now_ns() - started) / 1000000, b[2], sizeof(b[2]))); + return 0; +} + +void cbm_doclinks_free_scopes(cbm_doclink_scope_t *scopes, int count) { + if (!scopes) { + return; + } + for (int i = 0; i < count; i++) { + cbm_free(CBM_MEM_CLASS_OTHER, (char *)scopes[i].rel_path); + cbm_free(CBM_MEM_CLASS_OTHER, (char *)scopes[i].scope); + } + cbm_free(CBM_MEM_CLASS_OTHER, scopes); +} diff --git a/src/pipeline/doc_links.h b/src/pipeline/doc_links.h new file mode 100644 index 0000000000..f44241678b --- /dev/null +++ b/src/pipeline/doc_links.h @@ -0,0 +1,231 @@ +/* + * doc_links.h — doc-comment references -> MENTIONS edges. + * + * Extraction (internal/cbm/doclink.h) leaves every documented definition's + * references in CBMFileResult.doc_links and every file's doc-link scope in + * CBMFileResult.doc_scope. This layer resolves them in the per-file resolve + * phase, beside CALLS: + * + * cbm_doclinks_build once per run, after the run's nodes exist: the + * per-language indexes over every file's scope + * (this run's results, plus the stored scopes of + * files an incremental run did not re-extract) + * cbm_doclinks_resolve_file per file, concurrently: MENTIONS edges into the + * caller's edge buffer, unresolved rows into the + * run's slot for that file + * cbm_doclinks_take_rows the run's unresolved rows, for publication into + * doc_link_unresolved + * + * Resolution is guess-free: a reference becomes an edge only when it names + * one entity EXACTLY (qualified name, doc ID, alias) or UNIQUELY at the first + * scope level of the language's own lookup rules. Everything else is a row + * with a reason. One MENTIONS edge per (documented definition, target): + * {"via":"doc_comment","syntax":..,"tier":..,"line":,"count":n}. + * + * Ship gate: a link family (doclink.h) whose tier is below the audit's bar + * does not ship. Its references still resolve, but a resolved one is written + * as a row with reason below_bar_tier instead of an edge; an unresolved one + * keeps its own reason. + * + * Per-language hooks: a cbm_doclink_resolver_t (C#: doc_links_cs.c). + */ +#ifndef CBM_PIPELINE_DOC_LINKS_H +#define CBM_PIPELINE_DOC_LINKS_H + +#include "pipeline/pipeline_internal.h" +#include "store/store.h" + +/* doc_link_unresolved.reason values. */ +typedef enum { + CBM_DOCLINK_REASON_MISSING = 0, /* named code that is not in the graph anywhere */ + CBM_DOCLINK_REASON_AMBIGUOUS, /* several entities, or an overload group */ + CBM_DOCLINK_REASON_EXTERNAL, /* outside the repository (URL, BCL, open scope) */ + CBM_DOCLINK_REASON_TEST_ONLY, /* product code naming a test-only entity */ + CBM_DOCLINK_REASON_NOT_INDEXED, /* on disk, but not in the graph */ + CBM_DOCLINK_REASON_GRAPH_GAP, /* declared in source, but without a graph node */ + CBM_DOCLINK_REASON_UNPARSEABLE, /* reference syntax not understood */ + CBM_DOCLINK_REASON_BELOW_BAR, /* resolved, but its link family does not ship + * (cbm_doclink_syntax_ships): below_bar_tier */ + CBM_DOCLINK_REASON_COUNT +} cbm_doclink_reason_t; + +const char *cbm_doclink_reason_name(int reason); + +typedef enum { + CBM_DOCLINK_EDGE = 0, /* target set: one MENTIONS edge */ + CBM_DOCLINK_UNRESOLVED, /* reason set: one doc_link_unresolved row */ + /* Neither an edge nor a row: what the parser took for a reference is no + * reference to code elsewhere. It names the definition's own parameter or + * type parameter, or it is text the language's doc tool renders as plain + * text, which only the resolver can tell (it needs the index). Counted as + * `local` in the doc_links.done log line, and nowhere in index_status. */ + CBM_DOCLINK_LOCAL, +} cbm_doclink_kind_t; + +typedef struct { + cbm_doclink_kind_t kind; + const cbm_gbuf_node_t *target; + bool exact; /* tier: "exact" (qualified/doc ID/alias) vs "unique" (scope lookup) */ + int reason; +} cbm_doclink_outcome_t; + +/* A file's stored doc-link scope (incremental runs: files not re-extracted). */ +typedef struct cbm_doclink_scope { + const char *rel_path; + const char *scope; +} cbm_doclink_scope_t; + +typedef struct cbm_doclinks cbm_doclinks_t; + +/* The resolve-phase bracket every pipeline route uses: begin builds the run + * state into ctx->doc_links (over ctx->doc_link_base for an incremental run); + * resolve_worker / the sequential pass call cbm_doclinks_resolve_file; end + * hands the rows to ctx->pipeline (cbm_pipeline_set_doc_link_rows) and frees + * the state. A failed build is logged and recorded, never silent. */ +void cbm_doclinks_begin(cbm_pipeline_ctx_t *ctx, const cbm_file_info_t *files, int file_count, + CBMFileResult **cache); +void cbm_doclinks_end(cbm_pipeline_ctx_t *ctx); + +/* Sequential pipelines: begin + resolve every file into ctx->gbuf + end. */ +int cbm_pipeline_pass_doc_links(cbm_pipeline_ctx_t *ctx, const cbm_file_info_t *files, + int file_count); + +/* Rows of the next generation on an incremental run: the previous rows of the + * files not re-extracted (paths in `replaced` are dropped, as is a previous + * error marker) followed by this run's rows. Strings are copied; free with + * cbm_doclinks_free_rows. Returns 0, -1 on allocation failure. */ +int cbm_doclinks_merge_rows(const cbm_doc_link_row_t *old_rows, int old_count, + const CBMHashTable *replaced, const cbm_doc_link_row_t *fresh, + int fresh_count, cbm_doc_link_row_t **out, int *out_count); + +/* Build the run's resolver state. `files[0..file_count)` / `cache` are the + * files extracted in this run (results read through the spill contract); + * `base` holds the scopes of the files it did not re-extract (NULL/0 on a full + * run); `graph` is the run's complete node set, read-only from here on. + * Returns NULL only when allocation failed (logged). */ +cbm_doclinks_t *cbm_doclinks_build(const cbm_pipeline_ctx_t *ctx, const cbm_file_info_t *files, + int file_count, CBMFileResult **cache, + const cbm_doclink_scope_t *base, int base_count, + const cbm_gbuf_t *graph); + +/* Resolve files[file_idx]'s references. Concurrent calls for different files + * are safe. Edges go to `edge_out` (the worker's buffer, or the graph itself + * on a sequential run). */ +void cbm_doclinks_resolve_file(cbm_doclinks_t *dl, int file_idx, const CBMFileResult *result, + const cbm_gbuf_t *graph, cbm_gbuf_t *edge_out); + +/* Move the run's unresolved rows out (file order, then line order) together + * with the failure flag; the caller frees them with cbm_doclinks_free_rows. + * Logs one summary line. */ +void cbm_doclinks_take_rows(cbm_doclinks_t *dl, cbm_doc_link_row_t **rows, int *count, + bool *failed); +void cbm_doclinks_free_rows(cbm_doc_link_row_t *rows, int count); +void cbm_doclinks_free(cbm_doclinks_t *dl); + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +/* Test seam: the next cbm_doclinks_build fails the way an allocation failure + * does (logged, NULL), so a test can follow a failed doc-link layer through + * publication and index_status. Test builds only. */ +void cbm_doclinks_test_fail_build_once(void); +#endif + +/* The stored doc-link scopes of a previous generation (lsp_surface rows, + * key "dl"), excluding `skip` paths. Strings are heap-owned by *out; free + * with cbm_doclinks_free_scopes. Returns 0, or -1 when a row is malformed. */ +int cbm_doclinks_scopes_from_surfaces(const cbm_lsp_surface_row_t *rows, int row_count, + const CBMHashTable *skip, cbm_doclink_scope_t **out, + int *count); +void cbm_doclinks_free_scopes(cbm_doclink_scope_t *scopes, int count); +/* The "dl" scope of one surface JSON: *out is a heap copy, or NULL when the + * row has none. Returns 0, -1 when the key is unreadable or memory ran out. */ +int cbm_doclinks_scope_from_surface_json(const char *defs_json, char **out); + +/* ── Per-language resolver hooks ─────────────────────────────────── */ + +typedef struct { + const char *rel_path; + const char *scope; /* the file's doc-link scope blob (NULL when it has none) */ + int run_file; /* index into the run's files[], or -1 for a base file */ +} cbm_doclink_file_t; + +/* What `build` is handed. LIFETIMES: the struct and its `files` array are + * valid only during the `build` call (the array is freed right after it): a + * resolver must not keep either pointer. The `rel_path` and `scope` strings + * the array points to stay valid until `destroy`, so an index may keep those + * without copying them. */ +typedef struct { + const cbm_pipeline_ctx_t *ctx; + const cbm_gbuf_t *graph; + const cbm_doclink_file_t *files; /* every file of the language, sorted by rel_path */ + int file_count; + int run_file_count; /* size of the run's files[] (run_file indexes into it) */ +} cbm_doclink_build_in_t; + +/* Receives one name a changed file no longer declares. false when it cannot be + * recorded: the hook then reports failure, and the caller fails closed. */ +typedef bool (*cbm_doclink_name_fn)(void *ud, const char *name, size_t len); + +/* What a changed file's scope means for the files an incremental run does NOT + * re-extract. */ +typedef enum { + CBM_DOCLINK_DELTA_LOCAL = 0, /* nothing outside the file resolves differently, except + * unresolved references that name a reported name */ + CBM_DOCLINK_DELTA_GLOBAL, /* any file may resolve differently: not repairable + * file by file */ +} cbm_doclink_delta_t; + +enum { CBM_DOCLINK_RESOLVER_LANGS = 4 }; + +/* Everything the resolving half needs from a language leg: one of these, and + * its pointer in doc_links.c's resolver table. */ +typedef struct { + /* The languages whose files it resolves, in ONE index: a leg whose + * references cross language values (TypeScript, TSX and JavaScript) lists + * them all. No language may be listed by two resolvers. */ + CBMLanguage langs[CBM_DOCLINK_RESOLVER_LANGS]; + int lang_count; + /* Tag line of its scope blobs (doclink.h); NULL: it has none. */ + const char *scope_tag; + /* The language's project-wide index; NULL on allocation failure. */ + void *(*build)(const cbm_doclink_build_in_t *in); + void (*destroy)(void *index); + /* Resolve one reference of run file `run_file` (its own scope is in the + * index). Thread-safe: the index is read-only after build. */ + void (*resolve)(const void *index, int run_file, const CBMDocLink *link, + const cbm_gbuf_t *graph, cbm_doclink_outcome_t *out); + /* Incremental runs; both optional. + * scope_input true for a file that is no source of the language and has + * no scope blob, but sets the scope of its files. A change + * to one is GLOBAL. (C# needs none: its MSBuild project + * files have scope blobs of their own.) + * scope_delta compare a changed file's stored and fresh scope blobs (both + * of this language) and report the names it no longer + * declares through `removed`. Returns a cbm_doclink_delta_t, + * or -1 on failure. Without the hook every change is GLOBAL. */ + bool (*scope_input)(const char *rel_path); + int (*scope_delta)(const char *stored, const char *fresh, cbm_doclink_name_fn removed, + void *ud); +} cbm_doclink_resolver_t; + +extern const cbm_doclink_resolver_t cbm_doclink_cs_resolver; + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +/* Test seam (doc_links_cs.c): the scope levels and overloads the C# resolver + * looked at since the last reset. A test holds it against the size of its + * input, so that a lookup whose cost grows faster than its input fails + * without a clock. Test builds only. */ +void cbm_doclink_cs_test_work_reset(void); +uint64_t cbm_doclink_cs_test_work(void); +#endif + +/* True when `rel_path` is a scope input of some language (scope_input). */ +bool cbm_doclinks_is_scope_input(const char *rel_path); + +/* The scope delta of one changed file (`fresh` set) or deleted file (`fresh` + * NULL): a cbm_doclink_delta_t, or -1 on failure. A file with no scope before + * and after is LOCAL; a scope that appears, disappears or changes its + * language is GLOBAL; otherwise the language's scope_delta hook decides. */ +int cbm_doclinks_scope_delta(const char *stored, const char *fresh, cbm_doclink_name_fn removed, + void *ud); + +#endif /* CBM_PIPELINE_DOC_LINKS_H */ diff --git a/src/pipeline/doc_links_cs.c b/src/pipeline/doc_links_cs.c new file mode 100644 index 0000000000..16216d9ce1 --- /dev/null +++ b/src/pipeline/doc_links_cs.c @@ -0,0 +1,5995 @@ +/* + * doc_links_cs.c — C# cref resolution for doc_links.h. + * + * The project-wide index is built from every C# file's doc-link scope and + * every MSBuild project file's (internal/cbm/doclink_cs.c). Nothing is read + * from the disk. Graph nodes are only looked up by qualified name, never by + * line, so the closure-repair route (whose unchanged nodes are line-less + * proxies) resolves exactly as a full build does. + * + * Assemblies. The index does not know which files a project compiles; it + * goes by where a file stands. A project is a directory that holds a + * *.csproj -- one whose SDK compiles something: a project file of the + * NoTargets or Traversal SDK only runs build steps, and is the project of no + * file. A file belongs to the nearest project at or above it. An + * assembly is named by its project file: projects whose project files have + * the same name are one assembly (a reference source beside its + * implementation, the flavours of one library). A directory that holds + * several project files is an assembly of its own: which of them a file + * there is compiled into is not known. A file no project file stands above + * is of a shared tree -- the largest directory around it with no project + * file in or below it. A shared tree belongs to no assembly: the projects + * that include its files compile them. (A repository without any project + * file is one tree.) + * + * Entities. A type entity is (scope, name, generic arity, owner): the scope + * is its namespace, or its outer type's entity, so `Outer.Inner` and + * `Outer.Inner` are two types. The owner of a top-level type is its + * assembly: the declarations of one full name in one assembly are one type + * with several declarations (its parts, a stub beside the implementation, + * one per flavour), and declarations in two assemblies are two types, which + * see nothing of each other. The shared trees' declarations have no + * assembly. A complete declaration there is a type of its tree. The parts of + * a `partial` type there are one entity whatever tree they stand in, and + * they are parts of every assembly's type of that name whose implementation + * is itself declared partial: what a reference sees of a partial type is its + * own assembly's parts and the shared trees'. Seen without an assembly's own + * parts -- from a shared tree, or from an assembly that has none --, a name + * those parts declare is ambiguous: it is neither bound to what else has + * that name nor missing. + * + * Contracts. A declaration in a project directory named `ref` is a stub: + * `ref` is the .NET convention for reference-assembly sources. Inside one + * assembly the stub stands behind the implementation: the implementation's + * node binds, the stub's only where the implementation has none (its file + * did not parse, it lacks the member, or a parse error hides it). An + * assembly whose every declaration of a type is a stub holds only the + * contract, and the rule joins a contract to the one implementation it can + * belong to: the declaration of that full name and arity the shared trees + * hold, when they hold exactly one and no assembly holds another + * implementation of the name (entity_twins). The stub is then no second + * type; a binding that exists only through the join is never exact. + * Nothing is chosen between two implementations, two assemblies or two + * shared trees. + * + * What a reference sees of a name: its own assembly's type -- for a file of + * a shared tree, what its own tree declares. Else every shared tree's and + * every other assembly's type of that name alike: one binds, several are + * ambiguous. Which assemblies a project references is not known, so nothing + * chooses between two of them -- no stub, no nearer directory. + * + * Alternatives. Complete declarations of one type in two or more projects + * (the flavours of one assembly) are alternatives: the one of the + * referencing file's own project binds, and from anywhere else the + * reference is ambiguous. The same holds for one member declared in two or + * more projects or shared trees, and for the parts of a type in two or more + * shared trees. The parts of a partial type in one place are no + * alternatives: the type's node is the part's that stands nearest to the + * reference (a choice of presentation: every part is the type). + * + * Nodes. A declaration owns the node at . when it is the last + * declaration of that path in its file (the extractor keeps one node per + * qualified name and the last declaration wins, so `Foo` / `Foo` in one + * file leave one of them without a node: a graph gap, never a fallback to + * the other). The same holds for members: overloads share one node; `Foo.X` + * and `Foo.X` share one, and it belongs to the later declaration. + * Operators, indexers, events, delegates, implicit and primary constructors + * have no node at all: a reference to one that is declared is a graph gap. + * + * Lookup is the C# compiler's cref binding, and it guesses nothing: the first + * scope level that has the name decides, and more than one candidate there + * is ambiguous. + * - a simple name: the documented method's type parameters; then for every + * enclosing type, innermost first: its type parameters, its own nested + * types and members; then for every enclosing namespace, innermost + * first: the types and namespaces in it, then (where a namespace + * declaration of the file stands) that declaration's aliases and after + * them its usings. The file's own usings and the project's global ones + * belong to the outermost level. A using of an inner namespace + * declaration is therefore asked before an outer namespace + * - a qualified name: its first segment is looked up as a simple name that + * names a type or a namespace; every further segment is a member of what + * the one before named. There is no second try from another level, and + * no fully-qualified shortcut past a nearer namespace of that name + * - inherited members are not looked up: the compiler does not consider + * them in a cref, in a class or in an interface, by a simple name or + * through the derived type's name. Such a reference is a row + * (CS_BIND_INHERITED says what else it could be) + * - arity: a name without type arguments is the arity-0 type. A generic + * type of that name never stands in (no cross-arity fallback); type + * arguments select the types and the generic methods of that arity. A + * method name without type arguments takes methods of any arity, the + * non-generic ones first + * - the level that has the name decides also when a parameter list + * follows: a written signature binds the overload of THAT level with + * exactly these parameter types; a type variable matches by its position + * only (`{K}` ... `(K)` against `` ... `(TKey)`); only when no + * overload matches does one with a parameter type nothing is known about + * count, and several of those are ambiguous. Without a parameter list an + * overload group is ambiguous + * - constructors: a parameter list on a type's name (`Foo(int)`, + * `Ns.Foo(int)`, `Outer.Inner(int)`), `Foo.Foo`, and `Foo(int)` written + * inside a generic `Foo`; a bare `Foo` is the type; no constructor is + * inherited + * - `using static` brings in a type's nested types and its static members, + * extension methods excepted + * - explicit interface implementations are not addressable by name + * - visibility: product code never binds a test-only declaration + * (test_only_target, no fallback), and a namespace that only test code + * declares is no name in its scope; a test program does not bind another + * program's global-namespace test type + * + * Reasons. + * - external is structural: a name through a namespace the repository does + * not declare (a using of one, a qualifier that is one, an open + * hierarchy: a base outside the repository), a keyword alias the + * repository does not declare, a project whose global usings could not + * be evaluated. There is no list of well-known outside names; the one + * namespace treated as never the repository's own is `System` with what + * is under it (CS_STANDARD_ROOT) + * - missing: a name every scope level was asked for, in declared + * namespaces only; a name whose level has no overload with the written + * parameters; an inherited member named through a derived type or by a + * simple name (the compiler binds no such cref) when the hierarchy is + * the repository's + * - graph_gap: declared, and no node (see Nodes); a namespace; what the + * parser could not place (a type whose members a parse error hides + * answers a member it does not show with graph_gap; a name some file + * declares without an establishable scope is a gap wherever it is + * written) + * - ambiguous: several candidates at the level that has the name; two + * assemblies' (or two shared trees') types of one name; the flavours of + * one type or member seen from outside their projects; a name that parts + * of the type out of the reference's view declare; also an overload + * group -- or one project's complete declarations of one type -- larger + * than CS_MAX_OVERLOADS that would have to be compared, and a name more + * than CS_MAX_FOREIGN assemblies declare. The index's log line + * doc_links.cs.ambiguous counts the references by these causes + * - unparseable: a reference longer than CS_REF_BUF, with more segments + * than CS_MAX_SEGS or more parameters than CS_MAX_PARAMS + * No limit decides silently: every one of them ends in one of these rows. + */ +#include "pipeline/doc_links.h" + +#include "doclink.h" +#include "helpers.h" /* cbm_fqn_module_source_lang */ +#include "pipeline/doc_links_msbuild.h" +#include "foundation/arena.h" +#include "foundation/constants.h" +#include "foundation/hash_table.h" +#include "foundation/log.h" +#include "foundation/mem_core.h" + +#include +#include +#include +#include +#include + +enum { + CS_MAX_SEGS = 16, /* segments of a written reference */ + CS_MAX_PARAMS = 24, /* its parameters */ + CS_MAX_SUPERS = 64, /* supertypes one lookup follows */ + CS_MAX_OVERLOADS = 256, /* declarations of one name one lookup compares */ + CS_MAX_FOREIGN = 64, /* other owners' types of one name one lookup compares */ + CS_MAX_NEST = 64, /* the scanner's nesting limits (types, namespace segments) */ + CS_KEY_BUF = 2048, /* a node's qualified name */ + CS_NAME_BUF = 513, /* one name (the scanner's CS_NAME_MAX and its terminator) */ + CS_PARAM_BUF = 256, /* one normalized parameter type */ + CS_REF_BUF = 1024, /* a written reference */ + CS_ARITY_NONE = -1, + CS_NONE = -1, + CS_AMBIGUOUS = -2, + /* An entity's owner: an assembly (>= 0); CS_POOL, the shared trees' parts + * of a partial type; below it, the one shared tree that holds a complete + * declaration (shared_owner()). */ + CS_POOL = -1, + CS_FIELDS = 9, /* the most fields a scope record has (T) */ + CS_MAX_ROOTS = 3, /* Enum, ValueType, Object */ + /* 0: the compiler's rule, inherited members are never bound. 1: where the + * compiler's lookup binds nothing, the nearest supertype that has the + * member binds (a class's base classes, an interface's base interfaces, + * then the implicit roots the repository declares). Never a different + * binding than the compiler's: only one where it has none. */ + CS_BIND_INHERITED = 0, +}; + +/* ── Index data ──────────────────────────────────────────────────── */ + +/* A namespace of the repository: the global one (id 0), every declared one, + * and every namespace above a declared one. */ +typedef struct { + int parent; + const char *name; /* its own segment */ + bool declared; /* a file declares it; else it is only above a declared one */ + bool prod; /* product code declares it, or a namespace under it */ + bool standard; /* `System`, or a namespace under it (see CS_STANDARD_ROOT) */ +} cs_ns_t; + +/* The one namespace no repository owns by declaring it. `System` is the + * standard library's namespace by the language standard, and user code adds + * to it (polyfills) without owning it: a name of `System`, or of a namespace + * under it, that the repository does not have is the standard library's -- + * outside -- also where the repository declares that namespace. No other + * root is treated so, and no list of names stands behind this: every other + * namespace is the repository's as soon as a file declares it. */ +static const char CS_STANDARD_ROOT[] = "System"; + +/* The IDs below retain the original directive order. Two written directives + * remain two candidates even when they name the same target. */ +typedef struct { + const char *name; + int id; +} cs_alias_id_t; + +typedef struct { + int scope; /* namespace or entity, according to the array */ + int id; +} cs_using_id_t; + +typedef struct { + cs_alias_id_t *aliases; /* by (name, directive ID) */ + size_t naliases; + cs_using_id_t *namespaces; /* by (namespace, directive ID) */ + size_t nnamespaces; + cs_using_id_t *entities; /* by (view entity, directive ID), at most two per using */ + size_t nentities; + bool ready; /* published only after all targets and arrays are complete */ +} cs_using_index_t; + +typedef struct { + const char *name; + int scope; + bool top; /* a namespace's top-level type, else an entity's named declaration */ +} cs_name_scope_t; + +typedef struct { + int parent; /* region index; CS_NONE for the file's own region 0 */ + int ns; + uint32_t start; /* its lines (see the note on lines below) */ + uint32_t end; + int u_lo; /* its usings: [u_lo, u_hi) of the file's */ + int u_hi; + cs_using_index_t using_index; + bool open; /* a using of its own, or of a region around it, names something + * outside the repository */ +} cs_region_t; + +typedef struct { + int region; + char kind; /* n namespace, s static, a alias */ + bool global; + const char *alias; + const char *target; /* as written */ + int ns; /* what it names: a namespace (CS_NONE: none of the repository) */ + int ent; /* ... or a type (CS_NONE; CS_AMBIGUOUS) */ + bool joined; /* ... one type only because a contract is joined to it */ +} cs_using_t; + +typedef struct { + int region; + uint32_t start; + uint32_t end; + char kind; /* c s i e r t d */ + bool partial; + bool incomplete; /* a parse error hides some of its members */ + bool owns_node; /* the last declaration of its path in the file */ + int outer; /* the enclosing type (index in the file), or CS_NONE */ + int depth; + const char *name; + const char *bases; + int arity; + int entity; + int gid; /* types of one path in the file share it */ + const cbm_gbuf_node_t *node; /* its graph node, when it owns one */ +} cs_type_t; + +/* Line numbers (start, end) are meaningful only in a file re-extracted by + * this run: the persisted scope of every other file carries zeros (so an edit + * that only moves lines keeps the file's surface). Lines are therefore read + * only for the SOURCE file's own context, never for a target. */ +typedef struct { + uint32_t start; + int type; /* the declaration it belongs to (index into the file's types) */ + int arity; /* generic method arity */ + char kind; /* c method or constructor, v field or enum member, p property, e event, + o operator, x indexer */ + bool explicit_impl; + bool is_static; /* what `using static` brings in: static, and no extension method */ + const char *name; + const char *sig; /* '|'-joined parameter types; NULL for v, p and e */ + int nparams; /* how many `sig` holds */ + const cbm_gbuf_node_t *node; /* the node a reference to it binds; NULL: none */ +} cs_member_t; + +/* A type parameter: of a type (owner = its index) or of a generic method + * (owner = the file's type count + the member's index). */ +typedef struct { + int owner; + int pos; + const char *name; +} cs_tparam_t; + +typedef struct { + uint32_t from; + uint32_t to; +} cs_span_lines_t; + +typedef struct { + const char *rel_path; + const char *module_qn; + bool is_test; + bool is_ref; /* of a project directory named `ref`: a reference assembly's source */ + int unit; + cs_span_lines_t *unplaced; /* line ranges without a scope: sorted, disjoint */ + int nunplaced; + cs_region_t *regions; + int nregions; + cs_using_t *usings; /* by region */ + int nusings; + cs_type_t *types; /* document order: index = the T record's ordinal */ + int ntypes; + cs_member_t *members; /* document order */ + int nmembers; + int *members_by_start; + cs_tparam_t *tparams; /* by (owner, name) */ + int ntparams; +} cs_file_t; + +typedef struct { + int file; + int type; +} cs_decl_t; + +/* Where a declaration stands for choosing the one a reference binds: with a + * node before without, an implementation before a reference assembly's stub, + * product code and test code apart. Entries of one class are in path order. */ +enum { CS_CLS_TEST = 1, CS_CLS_REF = 2, CS_CLS_NO_NODE = 4, CS_CLS_COUNT = 8 }; + +typedef struct { + int file; + int type; + unsigned char cls; +} cs_bind_t; + +/* A complete (not `partial`) declaration outside a reference assembly's + * source, by the project or shared tree it stands in. */ +typedef struct { + int unit; + int file; + int type; +} cs_full_t; + +typedef struct { + int ns; /* a top-level type's namespace; CS_NONE for a nested type */ + int parent; /* a nested type's outer entity; CS_NONE */ + int owner; /* a top-level type's: its assembly, CS_POOL, or a shared tree */ + int twin; /* the shared trees' parts that are parts of this type too; CS_NONE */ + /* For the shared trees' declaration, what the assemblies make of it: */ + bool used; /* an assembly's type has it as its twin */ + int stub; /* the assembly's type that has only stubs and takes it as its + * implementation; CS_NONE: none does, CS_AMBIGUOUS: several do */ + bool rival_known; /* `rival` was asked for */ + bool rival; /* an assembly holds an implementation of the name that is no part + * of it (see rival_implementation) */ + const char *name; + int arity; + char kind; + cs_decl_t *decls; + int ndecls; + int dcap; + int *units; /* the projects and shared trees that declare it: sorted, unique */ + int nunits; + int *bases; + int nbases; + cs_bind_t *binds; /* its declarations, by (class, file, type) */ + /* Its complete declarations, when two or more projects (or shared trees) + * hold one: alternatives, by (unit, file, type). NULL when at most one + * does. */ + cs_full_t *fulls; + int nfulls; + int full_units; /* how many units hold a complete declaration */ + int full_units_prod; /* ... one that is not test code */ + bool open; /* a base of its own is outside the repository */ + bool open_any; /* ... or one of a supertype's is */ + bool incomplete; /* a declaration of it has members a parse error hides */ + bool any_prod; + bool all_test; + bool has_impl; /* a declaration outside a reference assembly's source */ + bool impl_partial; /* ... that is declared partial */ + bool shared_parts; /* the shared trees' parts of a partial type, or nested in them */ + bool joined; /* stubs only, joined to the shared trees' one implementation (twin) */ +} cs_entity_t; + +/* A type by the scope that declares it: `scope` is a namespace for a + * top-level type and the outer entity for a nested one. */ +typedef struct { + int scope; + const char *name; + int arity; + int owner; /* the entity's (a nested type has its outer type's: 0 here) */ + int ent; +} cs_named_t; + +/* A member by the entity that declares it. */ +typedef struct { + int ent; + int file; + int midx; + unsigned char group; /* 0 value or event, 1 callable, 2 operator or indexer */ + unsigned char cls; +} cs_mref_t; + +/* A name that an assembly's own parts of a type declare beyond the shared + * trees' declaration `twin` they belong to: a nested type the shared trees + * do not have (`ent`), or members (`ent` and `arity` CS_NONE; one entry for + * all of a name, `prod`: one of them is no test declaration). */ +typedef struct { + int twin; + const char *name; + int arity; + int ent; + bool prod; +} cs_extra_t; + +/* A project (a directory with a *.csproj and what is below it, up to the + * next one), or a shared tree. */ +typedef struct { + const char *dir; + int group; /* its assembly; CS_NONE for a shared tree */ + bool is_ref; /* a project directory named `ref` */ + cs_using_t *usings; /* global usings: its files' and its project files' */ + int nusings; + int cap; + cs_using_index_t using_index; + bool open; /* a global using could not be evaluated, or names something outside */ +} cs_unit_t; + +/* A directory that holds project files. */ +typedef struct { + int count; /* its *.csproj files */ + const char *stem; /* the name of the first, without the extension */ +} cs_pdir_t; + +/* Why a reference is ambiguous. The row says `ambiguous`; the index's log + * line says how many references of each kind there were. */ +typedef enum { + CS_WHY_SCOPE = 0, /* by the language's rules: candidates of one scope level, overloads */ + CS_WHY_SHARED, /* two or more shared trees declare the type */ + CS_WHY_ASSEMBLIES, /* two or more assemblies do */ + CS_WHY_FLAVOURS, /* declarations of one type or member in several projects of one + assembly, seen from outside them */ + CS_WHY_PARTS, /* declared by a part of the type the reference does not see */ + CS_WHY_LIMIT, /* more candidates than one lookup compares */ + CS_WHY_COUNT, +} cs_why_t; + +/* Counted while references are resolved (by every worker at once). */ +typedef struct { + _Atomic uint64_t ambiguous[CS_WHY_COUNT]; +} cs_stats_t; + +typedef struct { + CBMArena arena; + bool oom; /* an index allocation failed */ + const char *project; + cs_file_t *files; + int nfiles; + int *run_to_file; + int run_count; + cs_ns_t *nss; + int nnss; + int nscap; + CBMHashTable *ns_by_key; /* "\x1f" -> id + 1 */ + cs_entity_t *ents; + int nents; + int ecap; + CBMHashTable *ent_by_key; + cs_named_t *tops; /* by (namespace, name, arity, entity) */ + int ntops; + cs_named_t *kids; /* by (outer entity, name, arity, entity) */ + int nkids; + cs_mref_t *mrefs; /* by (entity, name, group, signature, class, file, order) */ + int nmrefs; + cs_extra_t *extras; /* by (twin, name, arity, entity): a name's members first */ + int nextras; + cs_name_scope_t *name_scopes; /* distinct (name, kind, scope) reverse postings */ + size_t nname_scopes; + bool name_scopes_ready; + /* "P\x1f": the entity declares an operator or indexer of + * that name outside test code; "A...": it declares one at all */ + CBMHashTable *specials; + CBMHashTable *type_names; /* every type's simple name */ + CBMHashTable *quarantine; /* names of types declared where no scope is known */ + CBMHashTable *quarantine_test; /* the same, declared by test code only */ + CBMHashTable *unit_by_dir; /* dir -> unit + 1 */ + cs_unit_t *units; + int nunits; + int ucap; + cbm_msb_t *msb; /* the repository's MSBuild project files */ + CBMHashTable *project_dirs; /* directory -> cs_pdir_t: the ones that hold a *.csproj */ + const char **projects; /* the *.csproj files, in path order */ + int nprojects; + CBMHashTable *project_above; /* directories with a *.csproj in or below them */ + CBMHashTable *group_by_stem; /* project file name -> assembly + 1 */ + int ngroups; /* assemblies */ + int nshared; /* shared trees */ + cs_stats_t *stats; + bool bind_inherited; /* CS_BIND_INHERITED */ +} cs_index_t; + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +/* Test seam: scope levels, supertypes and overloads the resolver looked at, + * and the directories it walked for the files' projects, since the last + * reset. A test holds it against the size of its input. */ +static _Atomic uint64_t cs_work_counter; +/* New import index construction is measured separately from lookup work. */ +static _Atomic uint64_t cs_index_work_counter; +static _Atomic bool cs_fail_candidate_alloc; +static _Atomic bool cs_candidate_alloc_failed; + +void cbm_doclink_cs_test_fail_candidate_alloc(bool enabled) { + atomic_store(&cs_candidate_alloc_failed, false); + atomic_store(&cs_fail_candidate_alloc, enabled); +} + +bool cbm_doclink_cs_test_candidate_alloc_failed(void) { + return atomic_load(&cs_candidate_alloc_failed); +} + +void cbm_doclink_cs_test_work_reset(void) { + atomic_store(&cs_work_counter, 0); + atomic_store(&cs_index_work_counter, 0); +} + +uint64_t cbm_doclink_cs_test_work(void) { + return atomic_load(&cs_work_counter); +} + +uint64_t cbm_doclink_cs_test_index_work(void) { + return atomic_load(&cs_index_work_counter); +} + +static void cs_work(uint64_t n) { + atomic_fetch_add_explicit(&cs_work_counter, n, memory_order_relaxed); +} + +static void cs_index_work(uint64_t n) { + atomic_fetch_add_explicit(&cs_index_work_counter, n, memory_order_relaxed); +} +#else +static void cs_work(uint64_t n) { + (void)n; +} + +static void cs_index_work(uint64_t n) { + (void)n; +} +#endif + +/* ── Small helpers ───────────────────────────────────────────────── */ + +/* Index memory. A failed allocation is remembered: whatever a caller does + * with its NULL, the build as a whole reports failure instead of handing out + * an index that silently lacks a declaration. */ +static void *ix_alloc(cs_index_t *ix, size_t n) { + void *p = cbm_arena_alloc(&ix->arena, n ? n : SKIP_ONE); + ix->oom = ix->oom || !p; + return p; +} + +static void *ix_zalloc(cs_index_t *ix, size_t n) { + void *p = ix_alloc(ix, n); + if (p) { + memset(p, 0, n ? n : SKIP_ONE); + } + return p; +} + +static char *ix_strndup(cs_index_t *ix, const char *s, size_t n) { + char *p = cbm_arena_strndup(&ix->arena, s, n); + ix->oom = ix->oom || !p; + return p; +} + +static char *ix_strdup(cs_index_t *ix, const char *s) { + char *p = cbm_arena_strdup(&ix->arena, s ? s : ""); + ix->oom = ix->oom || !p; + return p; +} + +/* Add `key` to a name set; false when memory ran out. */ +static bool ht_mark(cs_index_t *ix, CBMHashTable *ht, const char *key) { + if (cbm_ht_get(ht, key)) { + return true; + } + char *k = ix_strdup(ix, key); + if (!k) { + return false; + } + cbm_ht_set(ht, k, (void *)k); + return true; +} + +/* A type name declared in `f` at a place no namespace or outer type could be + * established for. Product code never binds test-only declarations, so a name + * only test files quarantine blocks references from test files only. */ +static bool quarantine_name(cs_index_t *ix, const cs_file_t *f, const char *name) { + return ht_mark(ix, f->is_test ? ix->quarantine_test : ix->quarantine, name); +} + +static bool cs_is_test_path(const char *rel) { + /* C# test code: a directory named tests/test, or a *.Tests / *.UnitTests / + * *.FunctionalTests project directory (the prototype's rule). */ + const char *p = rel; + for (;;) { + const char *slash = strchr(p, '/'); + if (!slash) { + return false; + } + size_t n = (size_t)(slash - p); + char seg[CBM_SZ_256]; + if (n < sizeof(seg)) { + for (size_t i = 0; i < n; i++) { + seg[i] = (char)tolower((unsigned char)p[i]); + } + seg[n] = '\0'; + if (strcmp(seg, "tests") == 0 || strcmp(seg, "test") == 0) { + return true; + } + static const char *const sfx[] = {".tests", ".unittests", ".functionaltests"}; + for (size_t k = 0; k < sizeof(sfx) / sizeof(sfx[0]); k++) { + size_t sl = strlen(sfx[k]); + if (n >= sl && strcmp(seg + n - sl, sfx[k]) == 0) { + return true; + } + } + } + p = slash + SKIP_ONE; + } +} + +static bool cs_ci_suffix(const char *s, const char *sfx) { + size_t n = strlen(s); + size_t sl = strlen(sfx); + if (n < sl) { + return false; + } + for (size_t i = 0; i < sl; i++) { + if (tolower((unsigned char)s[n - sl + i]) != sfx[i]) { + return false; + } + } + return true; +} + +/* Split `s` in place at tabs into at most `max` fields; returns the count. */ +static int split_fields(char *s, char **out, int max) { + int n = 0; + out[n++] = s; + for (char *p = s; *p && n < max; p++) { + if (*p == '\t') { + *p = '\0'; + out[n++] = p + SKIP_ONE; + } + } + return n; +} + +static int count_list(const char *s, char sep) { + if (!s || !s[0]) { + return 0; + } + int n = 1; + for (const char *p = s; *p; p++) { + n += *p == sep; + } + return n; +} + +/* A decimal index of a scope record: its value when it is one below `limit`, + * else CS_NONE. A scope comes from the store: nothing in it is trusted to be + * in range. */ +static int field_index(const char *s, int limit) { + if (!s[0] || strlen(s) > CBM_SZ_8) { + return CS_NONE; + } + int v = 0; + for (const char *p = s; *p; p++) { + if (!isdigit((unsigned char)*p)) { + return CS_NONE; + } + v = (v * 10) + (*p - '0'); + } + return v < limit ? v : CS_NONE; +} + +/* ── Namespaces ──────────────────────────────────────────────────── */ + +static bool ns_key(char *key, size_t cap, int parent, const char *seg, size_t len) { + if (len == 0 || len >= CS_NAME_BUF) { + return false; + } + int kl = snprintf(key, cap, "%d\x1f%.*s", parent, (int)len, seg); + return kl > 0 && (size_t)kl < cap; +} + +/* The namespace `seg` directly under `parent`, or CS_NONE. */ +static int ns_find(const cs_index_t *ix, int parent, const char *seg, size_t len) { + char key[CS_NAME_BUF + CBM_SZ_16]; + if (!ns_key(key, sizeof(key), parent, seg, len)) { + return CS_NONE; + } + intptr_t v = (intptr_t)cbm_ht_get(ix->ns_by_key, key); + return v > 0 ? (int)(v - SKIP_ONE) : CS_NONE; +} + +/* The same, created when it is not there yet. CS_NONE for a name that is + * none, and when memory ran out. */ +static int ns_make(cs_index_t *ix, int parent, const char *seg, size_t len) { + int found = ns_find(ix, parent, seg, len); + char key[CS_NAME_BUF + CBM_SZ_16]; + if (found >= 0 || !ns_key(key, sizeof(key), parent, seg, len)) { + return found; + } + if (ix->nnss >= ix->nscap) { + int ncap = ix->nscap ? ix->nscap * PAIR_LEN : CBM_SZ_256; + cs_ns_t *grown = (cs_ns_t *)ix_alloc(ix, (size_t)ncap * sizeof(cs_ns_t)); + if (!grown) { + return CS_NONE; + } + if (ix->nnss > 0) { + memcpy(grown, ix->nss, (size_t)ix->nnss * sizeof(cs_ns_t)); + } + ix->nss = grown; + ix->nscap = ncap; + } + char *k = ix_strdup(ix, key); + char *name = ix_strndup(ix, seg, len); + if (!k || !name) { + return CS_NONE; + } + int id = ix->nnss++; + ix->nss[id] = (cs_ns_t){.parent = parent, .name = name}; + cbm_ht_set(ix->ns_by_key, k, (void *)(intptr_t)(id + SKIP_ONE)); + return id; +} + +/* The namespace the dotted `path` names under `from`, every segment created + * on the way; CS_NONE for a bad name or when memory ran out. */ +static int ns_make_path(cs_index_t *ix, int from, const char *path) { + int ns = from; + for (const char *p = path; ns >= 0 && *p;) { + const char *dot = strchr(p, '.'); + size_t n = dot ? (size_t)(dot - p) : strlen(p); + ns = ns_make(ix, ns, p, n); + p = dot ? dot + SKIP_ONE : p + n; + } + return ns; +} + +/* ── Scope blob parsing ──────────────────────────────────────────── */ + +static int tparam_cmp(const void *a, const void *b) { + const cs_tparam_t *x = (const cs_tparam_t *)a; + const cs_tparam_t *y = (const cs_tparam_t *)b; + if (x->owner != y->owner) { + return x->owner < y->owner ? -1 : 1; + } + int c = strcmp(x->name, y->name); + return c ? c : (x->pos > y->pos) - (x->pos < y->pos); +} + +/* Position of the type parameter `name` of `owner` in `f`, or CS_NONE. */ +static int tparam_find(const cs_file_t *f, int owner, const char *name, size_t len) { + int lo = 0; + int hi = f->ntparams; + while (lo < hi) { + int mid = lo + ((hi - lo) / PAIR_LEN); + const cs_tparam_t *t = &f->tparams[mid]; + int c = t->owner != owner ? (t->owner < owner ? -1 : 1) : strncmp(t->name, name, len); + if (c == 0 && t->name[len] != '\0') { + c = 1; + } + if (c < 0) { + lo = mid + SKIP_ONE; + } else if (c > 0) { + hi = mid; + } else { + return t->pos; + } + } + return CS_NONE; +} + +static int span_cmp(const void *a, const void *b) { + const cs_span_lines_t *x = (const cs_span_lines_t *)a; + const cs_span_lines_t *y = (const cs_span_lines_t *)b; + if (x->from != y->from) { + return x->from < y->from ? -1 : 1; + } + return (x->to > y->to) - (x->to < y->to); +} + +typedef struct { + uint32_t start; + bool plain; /* no generic method: those stand first among one line's members */ + int idx; +} cs_start_key_t; + +static int start_key_cmp(const void *a, const void *b) { + const cs_start_key_t *x = (const cs_start_key_t *)a; + const cs_start_key_t *y = (const cs_start_key_t *)b; + if (x->start != y->start) { + return x->start < y->start ? -1 : 1; + } + if (x->plain != y->plain) { + return x->plain ? 1 : -1; + } + return (x->idx > y->idx) - (x->idx < y->idx); +} + +static int using_region_cmp(const void *a, const void *b) { + const cs_using_t *x = (const cs_using_t *)a; + const cs_using_t *y = (const cs_using_t *)b; + if (x->region != y->region) { + return x->region < y->region ? -1 : 1; + } + /* document order within a region: the targets are fields of one buffer, + * in the order they were read */ + return (x->target > y->target) - (x->target < y->target); +} + +/* What a scan of the blob's lines found, to size the file's arrays. */ +typedef struct { + int regions; + int usings; + int types; + int members; + int unplaced; + int tparams; +} cs_counts_t; + +static void count_records(const char *buf, cs_counts_t *n) { + memset(n, 0, sizeof(*n)); + n->regions = SKIP_ONE; /* region 0: the file */ + for (const char *p = buf; *p;) { + const char *nl = strchr(p, '\n'); + switch (*p) { + case 'R': + n->regions++; + break; + case 'U': + n->usings++; + break; + case 'T': + n->types++; + break; + case 'M': + n->members++; + break; + case 'X': + n->unplaced++; + break; + default: + break; + } + /* an upper bound for the type parameters: one per record that can + * have a list, and one more per comma */ + if (*p == 'T' || *p == 'M') { + n->tparams++; + for (const char *q = p; *q && *q != '\n'; q++) { + n->tparams += *q == ','; + } + } + if (!nl) { + break; + } + p = nl + SKIP_ONE; + } +} + +/* Add the ','-joined type parameters `list` of `owner`. The list is cut at + * its commas in place, so every name is a string of its own afterwards. */ +static void add_tparams(cs_file_t *f, int owner, char *list) { + int pos = 0; + for (char *p = list; p && *p;) { + char *e = strchr(p, ','); + if (e) { + *e = '\0'; + } + f->tparams[f->ntparams++] = (cs_tparam_t){.owner = owner, .pos = pos++, .name = p}; + p = e ? e + SKIP_ONE : NULL; + } +} + +static bool parse_region(cs_index_t *ix, cs_file_t *f, char **fld, int n, int *next_region) { + /* R id parent start end name */ + if (n != 6) { + return false; + } + int id = field_index(fld[1], f->nregions); + int parent = field_index(fld[2], f->nregions); + if (id != *next_region || parent < 0 || parent >= id) { + return false; /* regions are numbered in order, each under an earlier one */ + } + (*next_region)++; + int ns = ns_make_path(ix, f->regions[parent].ns, fld[5]); + if (ns < 0) { + return false; + } + ix->nss[ns].declared = true; + /* product code declares it, and with it every namespace above: one that + * is marked has its upper ones marked already */ + for (int up = ns; !f->is_test && up > 0 && !ix->nss[up].prod; up = ix->nss[up].parent) { + ix->nss[up].prod = true; + } + f->regions[id] = (cs_region_t){.parent = parent, + .ns = ns, + .start = (uint32_t)strtoul(fld[3], NULL, 10), + .end = (uint32_t)strtoul(fld[4], NULL, 10)}; + return true; +} + +static bool parse_using(cs_file_t *f, char **fld, int n, int regions_seen) { + /* U region kind alias target */ + if (n != 5) { + return false; + } + int region = field_index(fld[1], regions_seen); + char kind = fld[2][0]; + bool global = kind && fld[2][1] == 'g'; + if (region < 0 || !kind || !strchr("nsa", kind) || (fld[2][1] && !(global && !fld[2][2]))) { + return false; + } + f->usings[f->nusings++] = (cs_using_t){.region = region, + .kind = kind, + .global = global, + .alias = strcmp(fld[3], "-") == 0 ? "" : fld[3], + .target = fld[4], + .ns = CS_NONE, + .ent = CS_NONE}; + return true; +} + +static bool parse_type(cs_file_t *f, char **fld, int n, int regions_seen) { + /* T region start end kind outer name tparams bases */ + if (n != 9) { + return false; + } + cs_type_t *t = &f->types[f->ntypes]; + memset(t, 0, sizeof(*t)); + t->region = field_index(fld[1], regions_seen); + t->start = (uint32_t)strtoul(fld[2], NULL, 10); + t->end = (uint32_t)strtoul(fld[3], NULL, 10); + t->kind = fld[4][0]; + const char *flags = t->kind ? fld[4] + SKIP_ONE : ""; + t->partial = flags[0] == 'p'; + flags += t->partial; + t->incomplete = flags[0] == '!'; + flags += t->incomplete; + bool top_level = strcmp(fld[5], "-") == 0; + t->outer = top_level ? CS_NONE : field_index(fld[5], f->ntypes); + t->depth = t->outer >= 0 ? f->types[t->outer].depth + SKIP_ONE : 0; + t->name = fld[6]; + t->bases = fld[8]; + t->arity = count_list(fld[7], ','); + t->entity = CS_NONE; + size_t nl = strlen(t->name); + if (t->region < 0 || !t->kind || !strchr("csiertd", t->kind) || flags[0] || + (t->outer < 0 && !top_level) || nl == 0 || nl >= CS_NAME_BUF || t->depth >= CS_MAX_NEST) { + return false; + } + add_tparams(f, f->ntypes, fld[7]); + f->ntypes++; + return true; +} + +static bool parse_member(cs_file_t *f, char **fld, int n) { + /* M start kind explicit type name tparams sig */ + if (n != 8) { + return false; + } + cs_member_t *m = &f->members[f->nmembers]; + memset(m, 0, sizeof(*m)); + m->start = (uint32_t)strtoul(fld[1], NULL, 10); + m->kind = fld[2][0]; + m->is_static = m->kind && fld[2][1] == 's'; + m->explicit_impl = fld[3][0] == '1'; + m->type = field_index(fld[4], f->ntypes); + m->name = fld[5]; + m->arity = count_list(fld[6], ','); + bool callable_like = m->kind == 'c' || m->kind == 'o' || m->kind == 'x'; + m->sig = callable_like ? fld[7] : NULL; + m->nparams = (m->sig && m->sig[0]) ? count_list(m->sig, '|') : 0; + size_t nl = strlen(m->name); + if (m->type < 0 || !m->kind || !strchr("cvpeox", m->kind) || + fld[2][m->is_static ? PAIR_LEN : SKIP_ONE] || nl == 0 || nl >= CS_NAME_BUF) { + return false; + } + /* owners of type parameters: the types come first, the members after + * the LAST type the blob has, whose count is known only at the end */ + add_tparams(f, -(f->nmembers + PAIR_LEN), fld[6]); + f->nmembers++; + return true; +} + +/* Everything the records of one blob say, into `f`. false for a blob this + * code did not write (or one the store damaged), and when memory ran out. */ +static bool parse_records(cs_index_t *ix, cs_file_t *f, char *buf) { + int next_region = SKIP_ONE; + bool first = true; + for (char *line = buf; line && *line;) { + char *nl = strchr(line, '\n'); + if (nl) { + *nl = '\0'; + } + bool ok = true; + if (first) { + first = false; + ok = strcmp(line, CBM_DOCLINK_CS_SCOPE_TAG) == 0; + } else { + char *fld[CS_FIELDS + SKIP_ONE]; + int n = split_fields(line, fld, CS_FIELDS + SKIP_ONE); + switch (line[0]) { + case 'R': + ok = parse_region(ix, f, fld, n, &next_region); + break; + case 'U': + ok = parse_using(f, fld, n, next_region); + break; + case 'T': + ok = parse_type(f, fld, n, next_region); + break; + case 'M': + ok = parse_member(f, fld, n); + break; + case 'X': + ok = n == 3; + if (ok) { + f->unplaced[f->nunplaced++] = + (cs_span_lines_t){.from = (uint32_t)strtoul(fld[1], NULL, 10), + .to = (uint32_t)strtoul(fld[2], NULL, 10)}; + } + break; + case 'Q': + ok = n == PAIR_LEN && fld[1][0] && quarantine_name(ix, f, fld[1]); + break; + default: + ok = false; + break; + } + } + if (!ok) { + return false; + } + line = nl ? nl + SKIP_ONE : NULL; + } + return !first && next_region == f->nregions; +} + +/* Sort the usings by region, give every region its range, and close the + * unplaced ranges into sorted, disjoint ones. */ +static void finish_ranges(cs_file_t *f) { + if (f->nusings > 1) { + qsort(f->usings, (size_t)f->nusings, sizeof(cs_using_t), using_region_cmp); + } + int u = 0; + for (int r = 0; r < f->nregions; r++) { + f->regions[r].u_lo = u; + while (u < f->nusings && f->usings[u].region == r) { + u++; + } + f->regions[r].u_hi = u; + } + if (f->nunplaced > 1) { + qsort(f->unplaced, (size_t)f->nunplaced, sizeof(cs_span_lines_t), span_cmp); + int w = 0; + for (int i = 1; i < f->nunplaced; i++) { + if (f->unplaced[i].from <= f->unplaced[w].to) { + if (f->unplaced[i].to > f->unplaced[w].to) { + f->unplaced[w].to = f->unplaced[i].to; + } + } else { + f->unplaced[++w] = f->unplaced[i]; + } + } + f->nunplaced = w + SKIP_ONE; + } +} + +/* The type parameters by (owner, name), and the members in line order -- of + * the members that start at one line, the generic methods first. + * false when memory ran out. */ +static bool finish_lookups(cs_index_t *ix, cs_file_t *f) { + /* a member's parameters were filed under -(index + 2) while the type + * count was still growing */ + for (int i = 0; i < f->ntparams; i++) { + if (f->tparams[i].owner < 0) { + f->tparams[i].owner = f->ntypes + (-f->tparams[i].owner - PAIR_LEN); + } + } + if (f->ntparams > 1) { + qsort(f->tparams, (size_t)f->ntparams, sizeof(cs_tparam_t), tparam_cmp); + } + if (f->nmembers == 0) { + return true; + } + f->members_by_start = (int *)ix_alloc(ix, (size_t)f->nmembers * sizeof(int)); + cs_start_key_t *keys = (cs_start_key_t *)cbm_alloc( + CBM_MEM_CLASS_OTHER, (size_t)f->nmembers * sizeof(cs_start_key_t)); + if (!f->members_by_start || !keys) { + cbm_free(CBM_MEM_CLASS_OTHER, keys); + ix->oom = true; + return false; + } + for (int i = 0; i < f->nmembers; i++) { + const cs_member_t *m = &f->members[i]; + keys[i] = (cs_start_key_t){ + .start = m->start, .plain = !(m->kind == 'c' && m->arity > 0), .idx = i}; + } + qsort(keys, (size_t)f->nmembers, sizeof(*keys), start_key_cmp); + for (int i = 0; i < f->nmembers; i++) { + f->members_by_start[i] = keys[i].idx; + } + cbm_free(CBM_MEM_CLASS_OTHER, keys); + return true; +} + +static bool parse_scope(cs_index_t *ix, cs_file_t *f, const char *blob) { + char *buf = ix_strdup(ix, blob); + if (!buf) { + return false; + } + cs_counts_t n; + count_records(buf, &n); + f->nregions = n.regions; + f->regions = (cs_region_t *)ix_zalloc(ix, (size_t)n.regions * sizeof(cs_region_t)); + f->usings = (cs_using_t *)ix_alloc(ix, (size_t)n.usings * sizeof(cs_using_t)); + f->types = (cs_type_t *)ix_alloc(ix, (size_t)n.types * sizeof(cs_type_t)); + f->members = (cs_member_t *)ix_alloc(ix, (size_t)n.members * sizeof(cs_member_t)); + f->unplaced = (cs_span_lines_t *)ix_alloc(ix, (size_t)n.unplaced * sizeof(cs_span_lines_t)); + f->tparams = (cs_tparam_t *)ix_alloc(ix, (size_t)n.tparams * sizeof(cs_tparam_t)); + if (ix->oom) { + return false; + } + f->regions[0].parent = CS_NONE; + if (!parse_records(ix, f, buf)) { + return false; + } + finish_ranges(f); + return finish_lookups(ix, f); +} + +/* ── Units (C# projects) and their global usings ─────────────────── */ + +/* Nothing here reads the disk: a project file is one the index holds (its + * path among the resolver's files, its content in its scope blob). What + * discovery did not take -- a symbolic link, a named pipe -- is no project + * file. */ + +static const char CS_PROJECT_EXT[] = ".csproj"; + +/* The name of a project directory that holds a reference assembly's source: + * the .NET convention (/ref beside /src). It tells the + * stub from the implementation inside one assembly, and nothing else. */ +static const char CS_REF_DIR[] = "ref"; + +/* Length of the directory above the one of length `len` in `path`. */ +static size_t parent_dir_len(const char *path, size_t len) { + while (len > 0 && path[len - SKIP_ONE] != '/') { + len--; + } + return len > 0 ? len - SKIP_ONE : 0; +} + +/* Take the project files out of the resolver's file list: every MSBuild blob + * goes to the evaluator, every *.csproj marks its directory as a project and + * names it. false when memory ran out. */ +static bool collect_projects(cs_index_t *ix, const cbm_doclink_build_in_t *in) { + int n = 0; + for (int i = 0; i < in->file_count; i++) { + n += cs_ci_suffix(in->files[i].rel_path, CS_PROJECT_EXT); + } + ix->projects = (const char **)ix_alloc(ix, (size_t)n * sizeof(char *)); + if (!ix->projects) { + return false; + } + for (int i = 0; i < in->file_count; i++) { + const cbm_doclink_file_t *src = &in->files[i]; + if (cbm_msb_is_project_scope(src->scope) && + !cbm_msb_add(ix->msb, src->rel_path, src->scope)) { + return false; + } + /* a project file of an SDK that compiles nothing (it runs build steps + * or builds other projects) is the project of no source file */ + if (!cs_ci_suffix(src->rel_path, CS_PROJECT_EXT) || + !cbm_msb_compiles(ix->msb, src->rel_path)) { + continue; + } + const char *slash = strrchr(src->rel_path, '/'); + const char *name = slash ? slash + SKIP_ONE : src->rel_path; + char *dir = ix_strndup(ix, src->rel_path, slash ? (size_t)(slash - src->rel_path) : 0); + char *path = ix_strdup(ix, src->rel_path); + if (!dir || !path) { + return false; + } + ix->projects[ix->nprojects++] = path; /* the file list is in path order */ + cs_pdir_t *pd = (cs_pdir_t *)cbm_ht_get(ix->project_dirs, dir); + if (pd) { + pd->count++; + continue; + } + pd = (cs_pdir_t *)ix_zalloc(ix, sizeof(*pd)); + char *stem = ix_strndup(ix, name, strlen(name) - (sizeof(CS_PROJECT_EXT) - SKIP_ONE)); + if (!pd || !stem) { + return false; + } + pd->count = SKIP_ONE; + pd->stem = stem; + cbm_ht_set(ix->project_dirs, dir, pd); + /* the directory and every one above it has a project in or below + * it; one that is marked has its upper ones marked already */ + for (size_t len = strlen(dir);; len = parent_dir_len(dir, len)) { + char *above = ix_strndup(ix, dir, len); + if (!above) { + return false; + } + if (cbm_ht_get(ix->project_above, above)) { + break; + } + cbm_ht_set(ix->project_above, above, above); + if (len == 0) { + break; + } + } + } + return true; +} + +/* The assembly of a project directory: the one its project file names -- + * projects whose project files have the same name are one assembly. A + * directory with several project files is an assembly of its own (which of + * them compiles a file there is not known). CS_NONE when memory ran out. */ +static int group_of(cs_index_t *ix, const cs_pdir_t *pd) { + if (pd->count != SKIP_ONE) { + return ix->ngroups++; + } + intptr_t g = (intptr_t)cbm_ht_get(ix->group_by_stem, pd->stem); + if (g <= 0) { + g = (intptr_t)ix->ngroups + SKIP_ONE; + ix->ngroups++; + cbm_ht_set(ix->group_by_stem, pd->stem, (void *)g); + } + return (int)(g - SKIP_ONE); +} + +/* The unit of directory `dir`, made on first sight: a project when `pd` says + * which project files the directory holds, else a shared tree. CS_NONE when + * memory ran out. */ +static int unit_get(cs_index_t *ix, const char *dir, const cs_pdir_t *pd) { + intptr_t v = (intptr_t)cbm_ht_get(ix->unit_by_dir, dir); + if (v > 0) { + return (int)(v - SKIP_ONE); + } + if (ix->nunits >= ix->ucap) { + int ncap = ix->ucap ? ix->ucap * PAIR_LEN : CBM_SZ_64; + cs_unit_t *grown = (cs_unit_t *)ix_alloc(ix, (size_t)ncap * sizeof(cs_unit_t)); + if (!grown) { + return CS_NONE; + } + if (ix->nunits > 0) { + memcpy(grown, ix->units, (size_t)ix->nunits * sizeof(cs_unit_t)); + } + ix->units = grown; + ix->ucap = ncap; + } + char *key = ix_strdup(ix, dir); + if (!key) { + return CS_NONE; + } + int id = ix->nunits++; + cs_unit_t *u = &ix->units[id]; + memset(u, 0, sizeof(*u)); + u->dir = key; + u->group = CS_NONE; + if (pd) { + const char *base = strrchr(key, '/'); + u->is_ref = strcmp(base ? base + SKIP_ONE : key, CS_REF_DIR) == 0; + u->group = group_of(ix, pd); + } else { + ix->nshared++; + } + cbm_ht_set(ix->unit_by_dir, key, (void *)(intptr_t)(id + SKIP_ONE)); + return id; +} + +/* The tree of the project-less directory rel[0, len): the length of its root + * -- the largest directory around it that has no project in or below it; the + * directory itself when a project stands below it (*whole stays unset: what + * is under it is not all of its tree). `memo` is what `dir_unit` holds for + * the directory the walk for a project stopped at: a tree's root when that + * directory is of a known tree, and then everything under it is of that tree + * too. `dir` is scratch for the paths asked. */ +static size_t tree_root(const cs_index_t *ix, const char *rel, size_t len, intptr_t memo, char *dir, + bool *whole) { + if (memo < CS_NONE) { + *whole = true; + return (size_t)(-(memo + PAIR_LEN)); + } + memcpy(dir, rel, len); + dir[len] = '\0'; + if (cbm_ht_get(ix->project_above, dir)) { + return len; + } + *whole = true; + size_t root = len; + while (root > 0) { + cs_work(SKIP_ONE); + size_t up = parent_dir_len(rel, root); + dir[up] = '\0'; + if (cbm_ht_get(ix->project_above, dir)) { + break; + } + root = up; + } + return root; +} + +/* The unit a file belongs to: its project -- the nearest directory at or + * above its own that holds a *.csproj. A file no project file stands above + * is of a shared tree (sources that projects elsewhere compile in): which + * programs it is part of is not known, and all such files of a repository + * are not one program. Its unit is the tree: the largest directory around it + * that has no project in or below it -- the file's own directory when a + * project stands below that. (A repository without any project file is one + * tree.) Every directory walked on the way up is remembered in `dir_unit`: + * its project's unit + 1; for one no project stands at or above, -1, or + * -(length of its tree's root + 2) when it has no project below it either. + * So a directory is walked once, for its project and for its tree, however + * many files it and the directories under it hold. CS_NONE when memory ran + * out. */ +static int unit_of(cs_index_t *ix, CBMHashTable *dir_unit, const char *rel) { + const char *slash = strrchr(rel, '/'); + size_t len = slash ? (size_t)(slash - rel) : 0; + char *dir = (char *)cbm_alloc(CBM_MEM_CLASS_OTHER, len + SKIP_ONE); + if (!dir) { + ix->oom = true; + return CS_NONE; + } + memcpy(dir, rel, len); + size_t cur = len; + int unit = CS_NONE; + intptr_t memo = 0; + bool projectless = false; + bool failed = false; + for (;;) { + cs_work(SKIP_ONE); + dir[cur] = '\0'; + memo = (intptr_t)cbm_ht_get(dir_unit, dir); + if (memo > 0) { + unit = (int)(memo - SKIP_ONE); + break; + } + const cs_pdir_t *pd = (const cs_pdir_t *)cbm_ht_get(ix->project_dirs, dir); + if (pd) { + unit = unit_get(ix, dir, pd); + failed = unit < 0; + break; + } + if (memo < 0 || cur == 0) { + projectless = true; + break; + } + cur = parent_dir_len(rel, cur); + } + bool whole = false; + size_t root = projectless ? tree_root(ix, rel, len, memo, dir, &whole) : 0; + for (size_t l = len; !failed;) { + memcpy(dir, rel, l); + dir[l] = '\0'; + if (!cbm_ht_get(dir_unit, dir)) { + char *key = ix_strdup(ix, dir); + if (!key) { + failed = true; + break; + } + intptr_t known = unit + SKIP_ONE; + if (projectless) { + known = (whole && l >= root) ? -(intptr_t)(root + PAIR_LEN) : CS_NONE; + } + cbm_ht_set(dir_unit, key, (void *)known); + } + if (l <= cur) { + break; + } + l = parent_dir_len(rel, l); + } + if (projectless && !failed) { + memcpy(dir, rel, root); + dir[root] = '\0'; + unit = unit_get(ix, dir, NULL); + } + cbm_free(CBM_MEM_CLASS_OTHER, dir); + return failed ? CS_NONE : unit; +} + +static bool unit_add_using(cs_index_t *ix, cs_unit_t *u, char kind, const char *alias, + const char *target) { + if (u->nusings >= u->cap) { + int ncap = u->cap ? u->cap * PAIR_LEN : CBM_SZ_8; + cs_using_t *grown = (cs_using_t *)ix_alloc(ix, (size_t)ncap * sizeof(cs_using_t)); + if (!grown) { + return false; + } + if (u->nusings > 0) { + memcpy(grown, u->usings, (size_t)u->nusings * sizeof(cs_using_t)); + } + u->usings = grown; + u->cap = ncap; + } + char *a = ix_strdup(ix, alias); + char *t = ix_strdup(ix, target); + if (!a || !t) { + return false; + } + u->usings[u->nusings++] = (cs_using_t){ + .kind = kind, .global = true, .alias = a, .target = t, .ns = CS_NONE, .ent = CS_NONE}; + return true; +} + +static int unit_using_cmp(const void *a, const void *b) { + const cs_using_t *x = (const cs_using_t *)a; + const cs_using_t *y = (const cs_using_t *)b; + if (x->kind != y->kind) { + return x->kind < y->kind ? -1 : 1; + } + int c = strcmp(x->alias, y->alias); + return c ? c : strcmp(x->target, y->target); +} + +/* What the project files could not tell, over all of them. */ +typedef struct { + int unevaluable; + int outside; +} cs_msb_totals_t; + +/* Global usings of every unit: the `global using` directives of its files + * (of a namespace, of a type's static members, of an alias alike) and the + * items of every project file in its directory -- all of them, in + * path order, so the union does not depend on how a directory is listed. + * false when memory ran out. */ +static bool units_collect_usings(cs_index_t *ix, cs_msb_totals_t *totals) { + for (int fi = 0; fi < ix->nfiles; fi++) { + const cs_file_t *f = &ix->files[fi]; + for (int u = 0; f->unit >= 0 && u < f->nusings; u++) { + const cs_using_t *us = &f->usings[u]; + if (us->global && + !unit_add_using(ix, &ix->units[f->unit], us->kind, us->alias, us->target)) { + return false; + } + } + } + cbm_msb_eval_context_t *eval_context = cbm_msb_eval_context_new(ix->msb); + if (!eval_context) { + return false; + } + for (int p = 0; p < ix->nprojects; p++) { + const char *slash = strrchr(ix->projects[p], '/'); + char *dir = ix_strndup(ix, ix->projects[p], slash ? (size_t)(slash - ix->projects[p]) : 0); + intptr_t unit = dir ? (intptr_t)cbm_ht_get(ix->unit_by_dir, dir) : 0; + if (unit <= 0) { + continue; /* a project directory without a C# file */ + } + cs_unit_t *u = &ix->units[unit - SKIP_ONE]; + if (!cbm_msb_has(ix->msb, ix->projects[p])) { + /* a *.csproj the extractor has no blob for: what it sets is not known */ + totals->unevaluable++; + u->open = true; + continue; + } + cbm_msb_result_t res; + if (!cbm_msb_eval_context_eval(eval_context, ix->projects[p], &res)) { + cbm_msb_eval_context_free(eval_context); + return false; + } + bool ok = true; + for (int k = 0; ok && k < res.count; k++) { + ok = unit_add_using(ix, u, res.usings[k].kind, res.usings[k].alias, + res.usings[k].target); + } + u->open = u->open || res.open; + totals->unevaluable += res.unevaluable; + totals->outside += res.outside; + cbm_msb_result_free(&res); + if (!ok) { + cbm_msb_eval_context_free(eval_context); + return false; + } + } + cbm_msb_eval_context_free(eval_context); + for (int ui = 0; ui < ix->nunits; ui++) { + cs_unit_t *u = &ix->units[ui]; + if (u->nusings < PAIR_LEN) { + continue; + } + qsort(u->usings, (size_t)u->nusings, sizeof(cs_using_t), unit_using_cmp); + int w = 0; + for (int i = 1; i < u->nusings; i++) { + if (unit_using_cmp(&u->usings[w], &u->usings[i]) != 0) { + u->usings[++w] = u->usings[i]; + } + } + u->nusings = w + SKIP_ONE; + } + return !ix->oom; +} + +/* ── Entities ────────────────────────────────────────────────────── */ + +/* The owner of a complete declaration in the shared tree `unit`. */ +static int shared_owner(int unit) { + return -(unit + PAIR_LEN); +} + +/* The entity of this scope, name, arity and owner, created on first sight. A + * top-level type has a namespace `ns` and an `owner`; a nested type has its + * outer entity `parent` (and the owner 0: its outer type's is the one that + * counts). CS_NONE when memory ran out. */ +static int entity_get(cs_index_t *ix, int ns, int parent, int owner, const char *name, int arity, + char kind) { + char key[CS_NAME_BUF + CBM_SZ_64]; + int kl = snprintf(key, sizeof(key), "%c%d\x1f%d\x1f%d\x1f%s", parent >= 0 ? 'E' : 'N', + parent >= 0 ? parent : ns, owner, arity, name); + if (kl < 0 || kl >= (int)sizeof(key)) { + ix->oom = true; /* cannot be: parse_type bounds the name */ + return CS_NONE; + } + intptr_t v = (intptr_t)cbm_ht_get(ix->ent_by_key, key); + if (v > 0) { + return (int)(v - SKIP_ONE); + } + if (ix->nents >= ix->ecap) { + int ncap = ix->ecap ? ix->ecap * PAIR_LEN : CBM_SZ_1K; + cs_entity_t *grown = + (cs_entity_t *)cbm_alloc(CBM_MEM_CLASS_OTHER, (size_t)ncap * sizeof(cs_entity_t)); + if (!grown) { + ix->oom = true; + return CS_NONE; + } + if (ix->nents > 0) { + memcpy(grown, ix->ents, (size_t)ix->nents * sizeof(cs_entity_t)); + } + cbm_free(CBM_MEM_CLASS_OTHER, ix->ents); + ix->ents = grown; + ix->ecap = ncap; + } + char *k = ix_strdup(ix, key); + if (!k) { + return CS_NONE; + } + int id = ix->nents++; + cs_entity_t *e = &ix->ents[id]; + memset(e, 0, sizeof(*e)); + e->ns = parent >= 0 ? CS_NONE : ns; + e->parent = parent >= 0 ? parent : CS_NONE; + e->owner = owner; + e->twin = CS_NONE; + e->stub = CS_NONE; + e->shared_parts = parent >= 0 ? ix->ents[parent].shared_parts : owner == CS_POOL; + e->name = name; + e->arity = arity; + e->kind = kind; + e->all_test = true; + cbm_ht_set(ix->ent_by_key, k, (void *)(intptr_t)(id + SKIP_ONE)); + return id; +} + +static bool entity_add_decl(cs_index_t *ix, cs_entity_t *e, int file, int type) { + if (e->ndecls >= e->dcap) { + int ncap = e->dcap ? e->dcap * PAIR_LEN : CBM_SZ_2; + cs_decl_t *grown = (cs_decl_t *)ix_alloc(ix, (size_t)ncap * sizeof(cs_decl_t)); + if (!grown) { + return false; + } + if (e->ndecls > 0) { + memcpy(grown, e->decls, (size_t)e->ndecls * sizeof(cs_decl_t)); + } + e->decls = grown; + e->dcap = ncap; + } + e->decls[e->ndecls++] = (cs_decl_t){.file = file, .type = type}; + const cs_file_t *f = &ix->files[file]; + e->any_prod = e->any_prod || !f->is_test; + e->all_test = e->all_test && f->is_test; + e->has_impl = e->has_impl || !f->is_ref; + e->impl_partial = e->impl_partial || (!f->is_ref && f->types[type].partial); + e->incomplete = e->incomplete || f->types[type].incomplete; + return true; +} + +/* A top-level type as one shared tree declares it. false when it does not + * fit (cannot be: parse_type bounds the name). */ +static bool tree_type_key(char *key, size_t cap, const cs_file_t *f, const cs_type_t *t) { + int n = snprintf(key, cap, "%d\x1f%d\x1f%d\x1f%s", f->regions[t->region].ns, f->unit, t->arity, + t->name); + return n > 0 && (size_t)n < cap; +} + +/* The top-level types the shared trees declare partial, by tree: into + * `partials` (keys in `keys`). false when memory ran out. */ +static bool tree_partials(const cs_index_t *ix, CBMHashTable *partials, CBMArena *keys) { + char key[CS_NAME_BUF + CBM_SZ_64]; + for (int fi = 0; fi < ix->nfiles; fi++) { + const cs_file_t *f = &ix->files[fi]; + for (int ti = 0; ix->units[f->unit].group < 0 && ti < f->ntypes; ti++) { + const cs_type_t *t = &f->types[ti]; + if (t->outer >= 0 || !t->partial || !tree_type_key(key, sizeof(key), f, t) || + cbm_ht_get(partials, key)) { + continue; + } + char *k = cbm_arena_strdup(keys, key); + if (!k) { + return false; + } + cbm_ht_set(partials, k, k); + } + } + return true; +} + +/* The owner of a top-level type declared in `f`: the file's assembly. For a + * file of a shared tree: the shared trees' parts of that name when the + * declaration is partial -- or stands in a tree that has partial ones of the + * name: one tree's declarations of a name are one type --, else the tree. */ +static int decl_owner(const cs_index_t *ix, const cs_file_t *f, const cs_type_t *t, + const CBMHashTable *partials) { + int group = ix->units[f->unit].group; + if (group >= 0) { + return group; + } + char key[CS_NAME_BUF + CBM_SZ_64]; + bool parts = t->partial || (tree_type_key(key, sizeof(key), f, t) && cbm_ht_get(partials, key)); + return parts ? CS_POOL : shared_owner(f->unit); +} + +/* The entity of every declared type. An outer type stands before the types + * it holds, so its entity is known when theirs is asked for. false when + * memory ran out. */ +static bool build_entities(cs_index_t *ix) { + CBMHashTable *partials = cbm_ht_create(CBM_SZ_1K); + CBMArena keys; + cbm_arena_init(&keys); + bool ok = partials && tree_partials(ix, partials, &keys); + for (int fi = 0; ok && fi < ix->nfiles; fi++) { + cs_file_t *f = &ix->files[fi]; + for (int ti = 0; ok && ti < f->ntypes; ti++) { + cs_type_t *t = &f->types[ti]; + int ent = t->outer >= 0 + ? entity_get(ix, CS_NONE, f->types[t->outer].entity, 0, t->name, t->arity, + t->kind) + : entity_get(ix, f->regions[t->region].ns, CS_NONE, + decl_owner(ix, f, t, partials), t->name, t->arity, t->kind); + ok = ent >= 0 && entity_add_decl(ix, &ix->ents[ent], fi, ti) && + ht_mark(ix, ix->type_names, t->name); + t->entity = ent; + } + } + cbm_ht_free(partials); + cbm_arena_destroy(&keys); + ix->oom = ix->oom || !ok; + return ok; +} + +static int int_cmp(const void *a, const void *b) { + int x = *(const int *)a; + int y = *(const int *)b; + return (x > y) - (x < y); +} + +static int full_cmp(const void *a, const void *b) { + const cs_full_t *x = (const cs_full_t *)a; + const cs_full_t *y = (const cs_full_t *)b; + if (x->unit != y->unit) { + return x->unit < y->unit ? -1 : 1; + } + if (x->file != y->file) { + return x->file < y->file ? -1 : 1; + } + return (x->type > y->type) - (x->type < y->type); +} + +/* The complete declarations of every entity that has them in two or more + * projects (or shared trees): the flavours of one type, alternatives of each + * other. A reference assembly's stubs stand behind and are not + * counted. false when memory ran out. */ +static bool entity_fulls(cs_index_t *ix) { + for (int i = 0; i < ix->nents; i++) { + cs_entity_t *e = &ix->ents[i]; + int n = 0; + for (int d = 0; d < e->ndecls; d++) { + const cs_file_t *f = &ix->files[e->decls[d].file]; + n += !f->is_ref && !f->types[e->decls[d].type].partial; + } + if (n < PAIR_LEN) { + continue; + } + cs_full_t *fulls = (cs_full_t *)ix_alloc(ix, (size_t)n * sizeof(cs_full_t)); + if (!fulls) { + return false; + } + int w = 0; + for (int d = 0; d < e->ndecls; d++) { + const cs_file_t *f = &ix->files[e->decls[d].file]; + if (!f->is_ref && !f->types[e->decls[d].type].partial) { + fulls[w++] = (cs_full_t){ + .unit = f->unit, .file = e->decls[d].file, .type = e->decls[d].type}; + } + } + qsort(fulls, (size_t)n, sizeof(cs_full_t), full_cmp); + int units = 0; + int prod = 0; + for (int k = 0; k < n;) { + int unit = fulls[k].unit; + bool product = false; + for (; k < n && fulls[k].unit == unit; k++) { + product = product || !ix->files[fulls[k].file].is_test; + } + units++; + prod += product; + } + if (units >= PAIR_LEN) { + e->fulls = fulls; + e->nfulls = n; + e->full_units = units; + e->full_units_prod = prod; + } + } + return true; +} + +/* The projects and shared trees that declare each entity: sorted, unique. */ +static bool entity_units(cs_index_t *ix) { + for (int i = 0; i < ix->nents; i++) { + cs_entity_t *e = &ix->ents[i]; + e->units = (int *)ix_alloc(ix, (size_t)e->ndecls * sizeof(int)); + if (!e->units) { + return false; + } + for (int d = 0; d < e->ndecls; d++) { + e->units[d] = ix->files[e->decls[d].file].unit; + } + qsort(e->units, (size_t)e->ndecls, sizeof(int), int_cmp); + int w = 0; + for (int d = 1; d < e->ndecls; d++) { + if (e->units[d] != e->units[w]) { + e->units[++w] = e->units[d]; + } + } + e->nunits = e->ndecls > 0 ? w + SKIP_ONE : 0; + } + return true; +} + +static bool ent_in_unit(const cs_entity_t *e, int unit) { + return bsearch(&unit, e->units, (size_t)e->nunits, sizeof(int), int_cmp) != NULL; +} + +/* ── Types by scope and name ─────────────────────────────────────── */ + +/* Order: scope, name, arity, then the owner -- the shared trees' (the + * directories' complete declarations, then the parts: CS_POOL) before the + * assemblies' -- then the entity. */ +static int named_key_cmp(const cs_named_t *e, int scope, const char *name, int arity) { + if (e->scope != scope) { + return e->scope < scope ? -1 : 1; + } + int c = strcmp(e->name, name); + if (c) { + return c; + } + return (e->arity > arity) - (e->arity < arity); +} + +static int named_cmp(const void *a, const void *b) { + const cs_named_t *x = (const cs_named_t *)a; + const cs_named_t *y = (const cs_named_t *)b; + int c = named_key_cmp(x, y->scope, y->name, y->arity); + if (c) { + return c; + } + if (x->owner != y->owner) { + return x->owner < y->owner ? -1 : 1; + } + return (x->ent > y->ent) - (x->ent < y->ent); +} + +/* The entry of `owner` within [lo, hi) (one scope, name and arity: sorted by + * owner, and an owner has one entity there), or CS_NONE. */ +static int named_of_owner(const cs_named_t *arr, int lo, int hi, int owner) { + while (lo < hi) { + int mid = lo + ((hi - lo) / PAIR_LEN); + if (arr[mid].owner < owner) { + lo = mid + SKIP_ONE; + } else if (arr[mid].owner > owner) { + hi = mid; + } else { + return mid; + } + } + return CS_NONE; +} + +/* Where the assemblies' entries start within [lo, hi): before it stand the + * shared trees'. */ +static int named_first_assembly(const cs_named_t *arr, int lo, int hi) { + while (lo < hi) { + int mid = lo + ((hi - lo) / PAIR_LEN); + if (arr[mid].owner < 0) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + return lo; +} + +/* [return, *hi) of arr[0..n): the entries of this scope, name and arity. */ +static int named_range(const cs_named_t *arr, int n, int scope, const char *name, int arity, + int *hi) { + int lo = 0; + int end = n; + while (lo < end) { + int mid = lo + ((end - lo) / PAIR_LEN); + if (named_key_cmp(&arr[mid], scope, name, arity) < 0) { + lo = mid + SKIP_ONE; + } else { + end = mid; + } + } + int a = lo; + end = n; + while (a < end) { + int mid = a + ((end - a) / PAIR_LEN); + if (named_key_cmp(&arr[mid], scope, name, arity) <= 0) { + a = mid + SKIP_ONE; + } else { + end = mid; + } + } + *hi = a; + return lo; +} + +/* The two tables: top-level types by namespace, nested types by their outer + * entity. */ +static bool build_named(cs_index_t *ix) { + for (int i = 0; i < ix->nents; i++) { + ix->ntops += ix->ents[i].parent < 0; + } + ix->nkids = ix->nents - ix->ntops; + ix->tops = (cs_named_t *)ix_alloc(ix, (size_t)ix->ntops * sizeof(cs_named_t)); + ix->kids = (cs_named_t *)ix_alloc(ix, (size_t)ix->nkids * sizeof(cs_named_t)); + if (!ix->tops || !ix->kids) { + return false; + } + int t = 0; + int k = 0; + for (int i = 0; i < ix->nents; i++) { + const cs_entity_t *e = &ix->ents[i]; + cs_named_t row = {.scope = e->parent >= 0 ? e->parent : e->ns, + .name = e->name, + .arity = e->arity, + .owner = e->owner, + .ent = i}; + if (e->parent >= 0) { + ix->kids[k++] = row; + } else { + ix->tops[t++] = row; + } + } + if (ix->ntops > 1) { + qsort(ix->tops, (size_t)ix->ntops, sizeof(cs_named_t), named_cmp); + } + if (ix->nkids > 1) { + qsort(ix->kids, (size_t)ix->nkids, sizeof(cs_named_t), named_cmp); + } + return true; +} + +/* True when an assembly among tops[lo, hi) -- the assemblies' types of the + * shared declaration `shared`'s name -- holds an implementation of the name + * that is no part of that declaration: a complete type, or parts where the + * shared declaration is no partial type. Every contract of the name asks; + * the assemblies are walked for the first, and the answer is kept with the + * shared declaration. */ +static bool rival_implementation(cs_index_t *ix, int lo, int hi, int shared) { + cs_entity_t *s = &ix->ents[shared]; + if (!s->rival_known) { + bool pool = s->owner == CS_POOL; + for (int i = lo; !s->rival && i < hi; i++) { + const cs_entity_t *e = &ix->ents[ix->tops[i].ent]; + s->rival = e->has_impl && !(pool && e->impl_partial); + } + s->rival_known = true; + } + return s->rival; +} + +/* The shared trees' declaration that belongs to an assembly's type: its + * twin. Two rules give a top-level type one: + * - parts: a type an assembly's implementation declares `partial` has the + * shared trees' parts of that name (CS_POOL) as parts of it -- what a + * reference sees of a partial type is its own assembly's parts and the + * shared trees'. A complete type has no further parts; + * - contract: an assembly whose every declaration of the type stands in a + * project directory named `ref` holds only the contract. `ref` is the + * .NET convention for reference-assembly sources, and the rule joins a + * contract to the one implementation it can belong to: the declaration + * the shared trees hold of that full name and arity, when they hold + * exactly one (the parts of one tree count as one) and no assembly + * holds another implementation of it. The stub is then no second type. + * Nothing is chosen between two trees' declarations, two + * implementations or two assemblies: no join there. + * A type nested in a type that has a twin has the one of its name nested in + * the twin. An entity is made after its outer type's, so one pass in order + * sees every outer type first. What the twin says of the type (test code, + * hidden members) is folded into it. */ +static void entity_twins(cs_index_t *ix) { + for (int i = 0; i < ix->nents; i++) { + cs_entity_t *e = &ix->ents[i]; + int hi = 0; + if (e->parent >= 0) { + int outer = ix->ents[e->parent].twin; + if (outer < 0) { + continue; + } + int lo = named_range(ix->kids, ix->nkids, outer, e->name, e->arity, &hi); + e->twin = lo < hi ? ix->kids[lo].ent : CS_NONE; + e->joined = e->twin >= 0 && ix->ents[e->parent].joined; + } else { + if (e->owner < 0 || (e->has_impl && !e->impl_partial)) { + continue; + } + int lo = named_range(ix->tops, ix->ntops, e->ns, e->name, e->arity, &hi); + int split = named_first_assembly(ix->tops, lo, hi); + int at = CS_NONE; + if (e->has_impl) { + at = named_of_owner(ix->tops, lo, split, CS_POOL); + } else if (split - lo == SKIP_ONE && ix->ents[ix->tops[lo].ent].nunits == SKIP_ONE && + !rival_implementation(ix, split, hi, ix->tops[lo].ent)) { + at = lo; + e->joined = true; + } + e->twin = at >= 0 ? ix->tops[at].ent : CS_NONE; + } + if (e->twin >= 0) { + cs_entity_t *parts = &ix->ents[e->twin]; + e->incomplete = e->incomplete || parts->incomplete; + e->any_prod = e->any_prod || parts->any_prod; + e->all_test = e->all_test && parts->all_test; + parts->used = true; + if (!e->has_impl) { + parts->stub = parts->stub == CS_NONE ? i : CS_AMBIGUOUS; + } + } + } +} + +/* ── Node binding ────────────────────────────────────────────────── */ + +static bool label_is_callable(const char *l) { + return l && (strcmp(l, "Method") == 0 || strcmp(l, "Function") == 0); +} + +static bool label_is_value(const char *l) { + return l && (strcmp(l, "Field") == 0 || strcmp(l, "Variable") == 0 || + strcmp(l, "Property") == 0 || strcmp(l, "Constant") == 0); +} + +/* .. of type `t` into `buf`, its length into *len. + * false when it does not fit: such a declaration has no node to be found. */ +static bool type_qn(const cs_file_t *f, int t, char *buf, size_t cap, size_t *len) { + size_t ml = strlen(f->module_qn); + size_t total = ml; + for (int x = t; x >= 0; x = f->types[x].outer) { + total += strlen(f->types[x].name) + SKIP_ONE; + } + if (total >= cap) { + return false; + } + size_t w = total; + buf[w] = '\0'; + for (int x = t; x >= 0; x = f->types[x].outer) { + size_t nl = strlen(f->types[x].name); + w -= nl; + memcpy(buf + w, f->types[x].name, nl); + buf[--w] = '.'; + } + memcpy(buf, f->module_qn, ml); + *len = total; + return true; +} + +/* The node a member's own qualified name has, when it is one a reference to + * a member of this kind can bind. */ +static const cbm_gbuf_node_t *member_node_at(const cbm_gbuf_t *g, const cs_member_t *m, char *qn, + size_t type_len, size_t cap) { + size_t nl = strlen(m->name); + if (m->kind == 'o' || m->kind == 'x' || type_len + nl + PAIR_LEN > cap) { + return NULL; /* operators and indexers have no node */ + } + qn[type_len] = '.'; + memcpy(qn + type_len + SKIP_ONE, m->name, nl + SKIP_ONE); + const cbm_gbuf_node_t *n = cbm_gbuf_find_by_qn(g, qn); + qn[type_len] = '\0'; + if (!n) { + return NULL; + } + return (m->kind == 'c' ? label_is_callable(n->label) : label_is_value(n->label)) ? n : NULL; +} + +/* Scratch tables of the node pass: cleared for every file. */ +typedef struct { + CBMHashTable *names; /* "\x1f" -> index + 1 */ + CBMArena keys; + int *last; /* per gid: the last type that has it */ + int cap_last; +} cs_node_pass_t; + +/* The path group of every type of the file (types with one path -- `Foo` and + * `Foo`, and what is nested in them under one name -- share the node + * .), and which declaration owns that node: the last one. */ +static bool assign_gids(cs_index_t *ix, cs_file_t *f, cs_node_pass_t *np) { + if (f->ntypes > np->cap_last) { + cbm_free(CBM_MEM_CLASS_OTHER, np->last); + np->last = (int *)cbm_alloc(CBM_MEM_CLASS_OTHER, (size_t)f->ntypes * sizeof(int)); + np->cap_last = np->last ? f->ntypes : 0; + if (!np->last) { + ix->oom = true; + return false; + } + } + cbm_ht_clear(np->names); + cbm_arena_rewind(&np->keys); /* the cleared table held the only pointers into it */ + int gids = 0; + for (int t = 0; t < f->ntypes; t++) { + cs_type_t *ty = &f->types[t]; + char key[CS_NAME_BUF + CBM_SZ_16]; + snprintf(key, sizeof(key), "%d\x1f%s", ty->outer >= 0 ? f->types[ty->outer].gid : CS_NONE, + ty->name); + intptr_t v = (intptr_t)cbm_ht_get(np->names, key); + if (v > 0) { + ty->gid = (int)(v - SKIP_ONE); + } else { + char *k = cbm_arena_strdup(&np->keys, key); + if (!k) { + ix->oom = true; + return false; + } + ty->gid = gids++; + cbm_ht_set(np->names, k, (void *)(intptr_t)(ty->gid + SKIP_ONE)); + } + np->last[ty->gid] = t; + } + for (int t = 0; t < f->ntypes; t++) { + f->types[t].owns_node = np->last[f->types[t].gid] == t; + } + return true; +} + +/* The graph node of every type and member of the file that has one a + * reference may bind. A member's node .. belongs to the + * last declaration of that qualified name in the file: overloads of one type + * share it, but when `Foo` and `Foo` both declare the member, the earlier + * type's member has none of its own (the same rule as for the types). No + * node either where the slot holds a member of the other kind. */ +static bool bind_nodes(cs_index_t *ix, cs_file_t *f, const cbm_gbuf_t *g, cs_node_pass_t *np) { + if (!assign_gids(ix, f, np)) { + return false; + } + char qn[CS_KEY_BUF]; + size_t len = 0; + for (int t = 0; t < f->ntypes; t++) { + cs_type_t *ty = &f->types[t]; + if (ty->owns_node && ty->kind != 'd' && type_qn(f, t, qn, sizeof(qn), &len)) { + const cbm_gbuf_node_t *n = cbm_gbuf_find_by_qn(g, qn); + ty->node = (n && cbm_label_is_type_like(n->label)) ? n : NULL; + } + } + /* which member is the last of its (path, name) */ + cbm_ht_clear(np->names); + cbm_arena_rewind(&np->keys); + for (int m = 0; m < f->nmembers; m++) { + char key[(CS_NAME_BUF) + CBM_SZ_16]; + snprintf(key, sizeof(key), "%d\x1f%s", f->types[f->members[m].type].gid, + f->members[m].name); + const char *k = cbm_ht_get_key(np->names, key); + if (!k) { + k = cbm_arena_strdup(&np->keys, key); + if (!k) { + ix->oom = true; + return false; + } + } + cbm_ht_set(np->names, k, (void *)(intptr_t)(m + SKIP_ONE)); + } + int cached = CS_NONE; + bool fits = false; + for (int m = 0; m < f->nmembers; m++) { + cs_member_t *mem = &f->members[m]; + char key[(CS_NAME_BUF) + CBM_SZ_16]; + snprintf(key, sizeof(key), "%d\x1f%s", f->types[mem->type].gid, mem->name); + intptr_t last = (intptr_t)cbm_ht_get(np->names, key); + if (last <= 0 || + f->types[f->members[last - SKIP_ONE].type].entity != f->types[mem->type].entity) { + continue; /* a declaration of another type took the name's node */ + } + if (mem->type != cached) { + cached = mem->type; + fits = type_qn(f, cached, qn, sizeof(qn), &len); + } + mem->node = fits ? member_node_at(g, mem, qn, len, sizeof(qn)) : NULL; + } + return true; +} + +static unsigned char decl_class(const cs_file_t *f, bool has_node) { + return (unsigned char)((has_node ? 0 : CS_CLS_NO_NODE) | (f->is_ref ? CS_CLS_REF : 0) | + (f->is_test ? CS_CLS_TEST : 0)); +} + +static int bind_cmp(const void *a, const void *b) { + const cs_bind_t *x = (const cs_bind_t *)a; + const cs_bind_t *y = (const cs_bind_t *)b; + if (x->cls != y->cls) { + return x->cls < y->cls ? -1 : 1; + } + if (x->file != y->file) { + return x->file < y->file ? -1 : 1; + } + return (x->type > y->type) - (x->type < y->type); +} + +/* Every entity's declarations by (class, path). */ +static bool build_binds(cs_index_t *ix) { + for (int i = 0; i < ix->nents; i++) { + cs_entity_t *e = &ix->ents[i]; + e->binds = (cs_bind_t *)ix_alloc(ix, (size_t)e->ndecls * sizeof(cs_bind_t)); + if (!e->binds) { + return false; + } + for (int d = 0; d < e->ndecls; d++) { + const cs_file_t *f = &ix->files[e->decls[d].file]; + e->binds[d] = + (cs_bind_t){.file = e->decls[d].file, + .type = e->decls[d].type, + .cls = decl_class(f, f->types[e->decls[d].type].node != NULL)}; + } + if (e->ndecls > 1) { + qsort(e->binds, (size_t)e->ndecls, sizeof(cs_bind_t), bind_cmp); + } + } + return true; +} + +/* ── Members by entity and name ──────────────────────────────────── */ + +/* One member for sorting: the sort keys carry their own comparison data (no + * global sort context). */ +typedef struct { + cs_mref_t ref; + const char *name; + const char *sig; +} cs_mkey_t; + +static int mkey_cmp(const void *a, const void *b) { + const cs_mkey_t *x = (const cs_mkey_t *)a; + const cs_mkey_t *y = (const cs_mkey_t *)b; + if (x->ref.ent != y->ref.ent) { + return x->ref.ent < y->ref.ent ? -1 : 1; + } + int c = strcmp(x->name, y->name); + if (c) { + return c; + } + if (x->ref.group != y->ref.group) { + return x->ref.group < y->ref.group ? -1 : 1; + } + c = strcmp(x->sig, y->sig); + if (c) { + return c; + } + if (x->ref.cls != y->ref.cls) { + return x->ref.cls < y->ref.cls ? -1 : 1; + } + if (x->ref.file != y->ref.file) { + return x->ref.file < y->ref.file ? -1 : 1; + } + return (x->ref.midx > y->ref.midx) - (x->ref.midx < y->ref.midx); +} + +/* The key of the operators and indexers named `name` that `ent` declares: + * those outside test code (`prod`), or all of them. false when it does not + * fit (cannot be for a declared one: parse_member bounds the name). */ +static bool special_key(char *key, size_t cap, bool prod, int ent, const char *name) { + int n = snprintf(key, cap, "%c%d\x1f%s", prod ? 'P' : 'A', ent, name); + return n > 0 && (size_t)n < cap; +} + +/* Note an operator or indexer of `ent`. false when memory ran out. */ +static bool special_mark(cs_index_t *ix, int ent, const char *name, unsigned char cls) { + char key[CS_NAME_BUF + CBM_SZ_64]; + if (!special_key(key, sizeof(key), false, ent, name)) { + return true; + } + if (!ht_mark(ix, ix->specials, key)) { + return false; + } + key[0] = 'P'; + return (cls & CS_CLS_TEST) != 0 || ht_mark(ix, ix->specials, key); +} + +/* Every member a name can address, by (entity, name, group, signature, + * class, path, order). Explicit interface implementations are left out: no + * name addresses one. */ +static bool build_mrefs(cs_index_t *ix) { + size_t total = 0; + for (int fi = 0; fi < ix->nfiles; fi++) { + total += (size_t)ix->files[fi].nmembers; + } + cs_mkey_t *keys = + (cs_mkey_t *)cbm_alloc(CBM_MEM_CLASS_OTHER, (total ? total : SKIP_ONE) * sizeof(cs_mkey_t)); + ix->mrefs = (cs_mref_t *)ix_alloc(ix, total * sizeof(cs_mref_t)); + if (!keys || !ix->mrefs) { + cbm_free(CBM_MEM_CLASS_OTHER, keys); + ix->oom = true; + return false; + } + size_t n = 0; + for (int fi = 0; fi < ix->nfiles; fi++) { + const cs_file_t *f = &ix->files[fi]; + for (int m = 0; m < f->nmembers; m++) { + const cs_member_t *mem = &f->members[m]; + if (mem->explicit_impl) { + continue; + } + unsigned char group = mem->kind == 'c' + ? SKIP_ONE + : ((mem->kind == 'o' || mem->kind == 'x') ? PAIR_LEN : 0); + unsigned char cls = decl_class(f, mem->node != NULL); + int ent = f->types[mem->type].entity; + if (group == PAIR_LEN && !special_mark(ix, ent, mem->name, cls)) { + cbm_free(CBM_MEM_CLASS_OTHER, keys); + return false; + } + keys[n++] = + (cs_mkey_t){.ref = {.ent = ent, .file = fi, .midx = m, .group = group, .cls = cls}, + .name = mem->name, + .sig = mem->sig ? mem->sig : ""}; + } + } + if (n > 1) { + qsort(keys, n, sizeof(cs_mkey_t), mkey_cmp); + } + for (size_t i = 0; i < n; i++) { + ix->mrefs[i] = keys[i].ref; + } + ix->nmrefs = (int)n; + cbm_free(CBM_MEM_CLASS_OTHER, keys); + return true; +} + +static const cs_member_t *mref_member(const cs_index_t *ix, const cs_mref_t *r) { + return &ix->files[r->file].members[r->midx]; +} + +/* Order of a member entry against a key (entity, name, and when `group` is + * not negative: group and signature). */ +static int mref_key_cmp(const cs_index_t *ix, const cs_mref_t *r, int ent, const char *name, + int group, const char *sig) { + if (r->ent != ent) { + return r->ent < ent ? -1 : 1; + } + const cs_member_t *m = mref_member(ix, r); + int c = strcmp(m->name, name); + if (c || group < 0) { + return c; + } + if (r->group != group) { + return (int)r->group < group ? -1 : 1; + } + return strcmp(m->sig ? m->sig : "", sig); +} + +/* [return, *hi): the entity's members named `name`; with `group` >= 0 only + * those of that group with exactly the signature `sig`. */ +static int mref_range(const cs_index_t *ix, int ent, const char *name, int group, const char *sig, + int *hi) { + int lo = 0; + int end = ix->nmrefs; + while (lo < end) { + int mid = lo + ((end - lo) / PAIR_LEN); + if (mref_key_cmp(ix, &ix->mrefs[mid], ent, name, group, sig) < 0) { + lo = mid + SKIP_ONE; + } else { + end = mid; + } + } + int a = lo; + end = ix->nmrefs; + while (a < end) { + int mid = a + ((end - a) / PAIR_LEN); + if (mref_key_cmp(ix, &ix->mrefs[mid], ent, name, group, sig) <= 0) { + a = mid + SKIP_ONE; + } else { + end = mid; + } + } + *hi = a; + return lo; +} + +/* ── What an assembly's own parts add to a shared declaration ────── */ + +static int extra_key_cmp(const cs_extra_t *x, int twin, const char *name, int arity) { + if (x->twin != twin) { + return x->twin < twin ? -1 : 1; + } + int c = strcmp(x->name, name); + if (c) { + return c; + } + return (x->arity > arity) - (x->arity < arity); +} + +static int extra_cmp(const void *a, const void *b) { + const cs_extra_t *x = (const cs_extra_t *)a; + const cs_extra_t *y = (const cs_extra_t *)b; + int c = extra_key_cmp(x, y->twin, y->name, y->arity); + return c ? c : (x->ent > y->ent) - (x->ent < y->ent); +} + +/* The shared trees' declaration beside which the type `ent` stands: a type + * nested in an assembly's own parts of a type, implemented there, that the + * shared trees' declaration of that type does not have. CS_NONE for every + * other type. */ +static int extra_type_twin(const cs_index_t *ix, int ent) { + const cs_entity_t *e = &ix->ents[ent]; + if (e->parent < 0 || !e->has_impl || e->twin >= 0) { + return CS_NONE; + } + return ix->ents[e->parent].twin; +} + +/* The same for a member: one an assembly's own parts of a type declare. A + * stub does not count: it declares what the implementation declares. */ +static int extra_member_twin(const cs_index_t *ix, const cs_mref_t *r) { + return (r->cls & CS_CLS_REF) ? CS_NONE : ix->ents[r->ent].twin; +} + +/* The names the assemblies' own parts declare beyond the shared trees' + * declarations they belong to, by that declaration and name: what a + * reference that sees the shared declaration without those parts cannot tell + * from what is not there (unseen_part_declares). One walk over the types and + * the members, so that no lookup walks the assemblies. false when memory ran + * out. */ +static bool build_extras(cs_index_t *ix) { + size_t n = 0; + for (int i = 0; i < ix->nents; i++) { + n += extra_type_twin(ix, i) >= 0; + } + for (int i = 0; i < ix->nmrefs; i++) { + n += extra_member_twin(ix, &ix->mrefs[i]) >= 0; + } + cs_extra_t *arr = (cs_extra_t *)ix_alloc(ix, n * sizeof(cs_extra_t)); + if (!arr) { + return false; + } + size_t w = 0; + for (int i = 0; i < ix->nents; i++) { + int twin = extra_type_twin(ix, i); + if (twin >= 0) { + arr[w++] = (cs_extra_t){ + .twin = twin, .name = ix->ents[i].name, .arity = ix->ents[i].arity, .ent = i}; + } + } + for (int i = 0; i < ix->nmrefs; i++) { + const cs_mref_t *r = &ix->mrefs[i]; + int twin = extra_member_twin(ix, r); + if (twin >= 0) { + arr[w++] = (cs_extra_t){.twin = twin, + .name = mref_member(ix, r)->name, + .arity = CS_NONE, + .ent = CS_NONE, + .prod = !(r->cls & CS_CLS_TEST)}; + } + } + if (n > 1) { + qsort(arr, n, sizeof(cs_extra_t), extra_cmp); + } + /* a name's members are one entry */ + size_t out = 0; + for (size_t i = 0; i < n; i++) { + if (out > 0 && arr[i].ent < 0 && arr[out - SKIP_ONE].ent < 0 && + extra_key_cmp(&arr[out - SKIP_ONE], arr[i].twin, arr[i].name, arr[i].arity) == 0) { + arr[out - SKIP_ONE].prod = arr[out - SKIP_ONE].prod || arr[i].prod; + } else { + arr[out++] = arr[i]; + } + } + ix->extras = arr; + ix->nextras = (int)out; + return true; +} + +/* [return, *hi): the extras of this shared declaration, name and arity + * (CS_NONE: the name's members). */ +static int extra_range(const cs_index_t *ix, int twin, const char *name, int arity, int *hi) { + int lo = 0; + int end = ix->nextras; + while (lo < end) { + int mid = lo + ((end - lo) / PAIR_LEN); + if (extra_key_cmp(&ix->extras[mid], twin, name, arity) < 0) { + lo = mid + SKIP_ONE; + } else { + end = mid; + } + } + int a = lo; + end = ix->nextras; + while (a < end) { + int mid = a + ((end - a) / PAIR_LEN); + if (extra_key_cmp(&ix->extras[mid], twin, name, arity) <= 0) { + a = mid + SKIP_ONE; + } else { + end = mid; + } + } + *hi = a; + return lo; +} + +/* ── Import candidate indexes ───────────────────────────────────── */ + +static int alias_id_cmp(const void *a, const void *b) { + const cs_alias_id_t *x = (const cs_alias_id_t *)a; + const cs_alias_id_t *y = (const cs_alias_id_t *)b; + cs_index_work(SKIP_ONE); + int order = strcmp(x->name, y->name); + return order ? order : (x->id > y->id) - (x->id < y->id); +} + +static int using_id_cmp(const void *a, const void *b) { + const cs_using_id_t *x = (const cs_using_id_t *)a; + const cs_using_id_t *y = (const cs_using_id_t *)b; + cs_index_work(SKIP_ONE); + if (x->scope != y->scope) { + return (x->scope > y->scope) - (x->scope < y->scope); + } + return (x->id > y->id) - (x->id < y->id); +} + +static int name_scope_cmp(const void *a, const void *b) { + const cs_name_scope_t *x = (const cs_name_scope_t *)a; + const cs_name_scope_t *y = (const cs_name_scope_t *)b; + cs_index_work(SKIP_ONE); + int order = strcmp(x->name, y->name); + if (order) { + return order; + } + if (x->top != y->top) { + return x->top ? 1 : -1; + } + return (x->scope > y->scope) - (x->scope < y->scope); +} + +static void *import_array(cs_index_t *ix, size_t n, size_t size) { + if (n > SIZE_MAX / size) { + ix->oom = true; + return NULL; + } + return n ? ix_alloc(ix, n * size) : NULL; +} + +/* One repository-wide pass, never one member-table expansion per import. + * Include test-only names and extras: the ordinary resolver, not this + * candidate filter, decides visibility, arity and unseen-part ambiguity. */ +static bool build_name_scopes(cs_index_t *ix) { + size_t n = 0; + const int counts[] = {ix->ntops, ix->nkids, ix->nmrefs, ix->nextras}; + for (size_t i = 0; i < sizeof(counts) / sizeof(counts[0]); i++) { + if ((size_t)counts[i] > SIZE_MAX - n) { + ix->oom = true; + return false; + } + n += (size_t)counts[i]; + } + cs_name_scope_t *rows = (cs_name_scope_t *)import_array(ix, n, sizeof(*rows)); + if (n && !rows) { + return false; + } + size_t w = 0; + for (int i = 0; i < ix->ntops; i++) { + cs_index_work(SKIP_ONE); + rows[w++] = + (cs_name_scope_t){.name = ix->tops[i].name, .scope = ix->tops[i].scope, .top = true}; + } + for (int i = 0; i < ix->nkids; i++) { + cs_index_work(SKIP_ONE); + rows[w++] = (cs_name_scope_t){.name = ix->kids[i].name, .scope = ix->kids[i].scope}; + } + for (int i = 0; i < ix->nmrefs; i++) { + cs_index_work(SKIP_ONE); + rows[w++] = (cs_name_scope_t){.name = mref_member(ix, &ix->mrefs[i])->name, + .scope = ix->mrefs[i].ent}; + } + for (int i = 0; i < ix->nextras; i++) { + cs_index_work(SKIP_ONE); + rows[w++] = (cs_name_scope_t){.name = ix->extras[i].name, .scope = ix->extras[i].twin}; + } + if (w > 1) { + qsort(rows, w, sizeof(*rows), name_scope_cmp); + } + size_t out = 0; + for (size_t i = 0; i < w; i++) { + if (!out || name_scope_cmp(&rows[out - SKIP_ONE], &rows[i])) { + rows[out++] = rows[i]; + } + } + ix->name_scopes = rows; + ix->nname_scopes = out; + ix->name_scopes_ready = true; + return true; +} + +/* Build only after these directives have resolved. A static import is + * associated with its own entity and its twin, without copying either + * entity's declarations into every region that imports it. */ +static bool build_using_index(cs_index_t *ix, const cs_using_t *us, int n, + cs_using_index_t *index) { + cs_using_index_t out = {0}; + for (int i = 0; i < n; i++) { + cs_index_work(SKIP_ONE); + const cs_using_t *u = &us[i]; + if (u->kind == 'a') { + out.naliases++; + } else if (u->kind == 'n' && u->ns >= 0) { + out.nnamespaces++; + } else if (u->kind == 's' && u->ent >= 0) { + size_t extra = ix->ents[u->ent].twin >= 0 ? PAIR_LEN : SKIP_ONE; + if (extra > SIZE_MAX - out.nentities) { + ix->oom = true; + return false; + } + out.nentities += extra; + } + } + out.aliases = (cs_alias_id_t *)import_array(ix, out.naliases, sizeof(*out.aliases)); + out.namespaces = (cs_using_id_t *)import_array(ix, out.nnamespaces, sizeof(*out.namespaces)); + out.entities = (cs_using_id_t *)import_array(ix, out.nentities, sizeof(*out.entities)); + if (ix->oom) { + return false; + } + size_t a = 0; + size_t ns = 0; + size_t e = 0; + for (int i = 0; i < n; i++) { + cs_index_work(SKIP_ONE); + const cs_using_t *u = &us[i]; + if (u->kind == 'a') { + out.aliases[a++] = (cs_alias_id_t){.name = u->alias, .id = i}; + } else if (u->kind == 'n' && u->ns >= 0) { + out.namespaces[ns++] = (cs_using_id_t){.scope = u->ns, .id = i}; + } else if (u->kind == 's' && u->ent >= 0) { + out.entities[e++] = (cs_using_id_t){.scope = u->ent, .id = i}; + int twin = ix->ents[u->ent].twin; + if (twin >= 0) { + out.entities[e++] = (cs_using_id_t){.scope = twin, .id = i}; + } + } + } + if (a > 1) { + qsort(out.aliases, a, sizeof(*out.aliases), alias_id_cmp); + } + if (ns > 1) { + qsort(out.namespaces, ns, sizeof(*out.namespaces), using_id_cmp); + } + if (e > 1) { + qsort(out.entities, e, sizeof(*out.entities), using_id_cmp); + } + out.ready = true; + *index = out; + return true; +} + +/* ── Choosing the declaration a reference binds ──────────────────── */ + +/* Number of leading directories two paths share. */ +static int common_dir_prefix(const char *a, const char *b) { + int n = 0; + for (;;) { + const char *sa = strchr(a, '/'); + const char *sb = strchr(b, '/'); + if (!sa || !sb) { + return n; + } + size_t la = (size_t)(sa - a); + if (la != (size_t)(sb - b) || memcmp(a, b, la) != 0) { + return n; + } + n++; + a = sa + SKIP_ONE; + b = sb + SKIP_ONE; + } +} + +/* Length of the first `dirs` directories of `path`, each with its '/'. */ +static size_t dir_prefix_len(const char *path, int dirs) { + const char *p = path; + for (int i = 0; i < dirs; i++) { + const char *slash = strchr(p, '/'); + if (!slash) { + break; + } + p = slash + SKIP_ONE; + } + return (size_t)(p - path); +} + +/* Of the declarations [lo, hi) -- one class, so in path order -- the one a + * reference written in file `src` binds (a choice of presentation among the + * parts of one type, made without a scan): the one in `src` itself; else the + * first one in the directory that shares the longest leading path with it; + * else the first. */ +static int nearest_bind(const cs_index_t *ix, const cs_bind_t *arr, int lo, int hi, int src) { + int a = lo; + int b = hi; + while (a < b) { + int mid = a + ((b - a) / PAIR_LEN); + if (arr[mid].file < src) { + a = mid + SKIP_ONE; + } else { + b = mid; + } + } + if (a < hi && arr[a].file == src) { + return a; + } + const char *sp = ix->files[src].rel_path; + int left = a > lo ? common_dir_prefix(ix->files[arr[a - SKIP_ONE].file].rel_path, sp) : 0; + int right = a < hi ? common_dir_prefix(ix->files[arr[a].file].rel_path, sp) : 0; + int best = left > right ? left : right; + if (best == 0) { + return lo; + } + /* the entries before `a` that share those directories are the last ones + * before it: the first of them */ + size_t plen = dir_prefix_len(sp, best); + int x = lo; + int y = a; + while (x < y) { + int mid = x + ((y - x) / PAIR_LEN); + if (strncmp(ix->files[arr[mid].file].rel_path, sp, plen) < 0) { + x = mid + SKIP_ONE; + } else { + y = mid; + } + } + return x; +} + +/* [return, *hi): the declarations of class `cls` among arr[0, n), which is + * sorted by class first. */ +static int class_range(const cs_bind_t *arr, int n, unsigned char cls, int *out_hi) { + int a = 0; + int b = n; + while (a < b) { + int mid = a + ((b - a) / PAIR_LEN); + if (arr[mid].cls < cls) { + a = mid + SKIP_ONE; + } else { + b = mid; + } + } + int first = a; + b = n; + while (a < b) { + int mid = a + ((b - a) / PAIR_LEN); + if (arr[mid].cls <= cls) { + a = mid + SKIP_ONE; + } else { + b = mid; + } + } + *out_hi = a; + return first; +} + +/* True when a declaration in file `fa` is the better one to bind than one in + * `fb`, for a reference written in `src`: its own file, then the longer + * shared path, then the path order. */ +static bool file_nearer(const cs_index_t *ix, int fa, int fb, int src) { + if ((fa == src) != (fb == src)) { + return fa == src; + } + const char *sp = ix->files[src].rel_path; + int pa = common_dir_prefix(ix->files[fa].rel_path, sp); + int pb = common_dir_prefix(ix->files[fb].rel_path, sp); + return pa != pb ? pa > pb : fa < fb; +} + +/* ── Resolution context ──────────────────────────────────────────── */ + +enum { CS_OK = 0, CS_UNRES, CS_LOCAL }; + +typedef struct { + int st; + const cbm_gbuf_node_t *node; + bool exact; + int reason; + cs_why_t why; /* of an ambiguous one */ +} cs_res_t; + +typedef struct { + const cs_index_t *ix; + int file; + const cs_file_t *f; + int type; /* the innermost type declaration around the definition, or CS_NONE */ + int member; /* the documented member, when it has type parameters; CS_NONE */ + int region; /* the namespace declaration around it */ + int unit; /* the file's project, or its directory in a shared tree */ + int group; /* its assembly; CS_NONE in a shared tree */ + bool prod; /* product code: test-only declarations are not bound */ + bool glob; /* the reference starts with global:: */ + /* The namespace declaration whose own aliases and usings are not asked + * (CS_NONE: none): a using directive's target is resolved without the + * directives beside it. */ + int skip_region; + /* Set when a namespace this context does not see was passed over (NULL: + * nobody asks). */ + bool *passed_over; +} cs_ctx_t; + +static cs_res_t res_edge(const cbm_gbuf_node_t *n, bool exact) { + return (cs_res_t){.st = CS_OK, .node = n, .exact = exact}; +} + +static cs_res_t res_unres(int reason) { + return (cs_res_t){.st = CS_UNRES, .reason = reason}; +} + +static cs_res_t res_ambiguous(cs_why_t why) { + return (cs_res_t){.st = CS_UNRES, .reason = CBM_DOCLINK_REASON_AMBIGUOUS, .why = why}; +} + +static bool res_is(const cs_res_t *r, int reason) { + return r->st == CS_UNRES && r->reason == reason; +} + +/* What choosing among declarations came to. */ +typedef enum { CS_PICK_NODE = 0, CS_PICK_GAP, CS_PICK_AMBIGUOUS } cs_pick_t; + +/* An entity and the shared trees' declaration that belongs to it: what a + * reference sees of one type is in at most these two. */ +typedef struct { + int ent[PAIR_LEN]; + int n; +} cs_view_t; + +static cs_view_t view_of(const cs_index_t *ix, int ent) { + int twin = ix->ents[ent].twin; + return (cs_view_t){.ent = {ent, twin}, .n = twin >= 0 ? PAIR_LEN : SKIP_ONE}; +} + +/* The node of the nearest of `e`'s own declarations of one class -- an + * implementation's, or (`stub`) a reference assembly's stub's -- that has + * one; never a test declaration's for product code. NULL when none has. */ +static const cbm_gbuf_node_t *class_node(const cs_ctx_t *c, const cs_entity_t *e, bool stub) { + const cs_index_t *ix = c->ix; + const cs_bind_t *best = NULL; + for (int test = 0; test <= (c->prod ? 0 : SKIP_ONE); test++) { + int b = 0; + int a = + class_range(e->binds, e->ndecls, + (unsigned char)((stub ? CS_CLS_REF : 0) | (test ? CS_CLS_TEST : 0)), &b); + if (a >= b) { + continue; + } + const cs_bind_t *at = &e->binds[nearest_bind(ix, e->binds, a, b, c->file)]; + if (!best || file_nearer(ix, at->file, best->file, c->file)) { + best = at; + } + } + return best ? ix->files[best->file].types[best->type].node : NULL; +} + +/* True when one of `e`'s own declarations is in view, with a node or + * without: for product code one that is not test code. */ +static bool declared_in_view(const cs_ctx_t *c, const cs_entity_t *e) { + for (int cls = 0; cls < CS_CLS_COUNT; cls++) { + int b = 0; + if (!(c->prod && (cls & CS_CLS_TEST)) && + class_range(e->binds, e->ndecls, (unsigned char)cls, &b) < b) { + return true; + } + } + return false; +} + +/* ── Visibility and candidates ───────────────────────────────────── */ + +static bool ent_visible(const cs_ctx_t *c, int ent) { + const cs_index_t *ix = c->ix; + const cs_entity_t *e = &ix->ents[ent]; + if (c->prod) { + return e->any_prod; + } + /* test code: a global-namespace test type (and what it holds) is local + * to its own program */ + int top = ent; + while (ix->ents[top].parent >= 0) { + top = ix->ents[top].parent; + } + if (ix->ents[top].all_test && ix->ents[top].ns == 0) { + return ent_in_unit(&ix->ents[top], c->unit); + } + return true; +} + +/* What a name was found to be at one scope level. */ +typedef struct { + char kind; /* T type, M member(s) of entity `id`, N namespace, L type parameter, X outside */ + int id; +} cs_cand_t; + +typedef struct { + cs_cand_t first; + int n; /* distinct candidates of the deciding level; > 1 is ambiguous */ + bool invisible; /* a level had only what this code may not bind (test code) */ + bool exact; /* decided by an alias */ + bool in_namespace; /* decided by an enclosing namespace's own types and namespaces */ + bool joined; /* one type only because a contract is joined to its implementation */ + cs_why_t why; /* what made several of them, when it was not the scope's rules */ +} cs_found_t; + +/* The result for a name with several candidates. */ +static cs_res_t found_ambiguous(const cs_found_t *fd) { + return res_ambiguous(fd->why); +} + +static void found_add(cs_found_t *fd, char kind, int id) { + if (fd->n > 0 && fd->first.kind == kind && fd->first.id == id) { + return; + } + if (fd->n == 0) { + fd->first = (cs_cand_t){.kind = kind, .id = id}; + } + fd->n++; +} + +static void found_add_type(const cs_ctx_t *c, cs_found_t *fd, int ent) { + if (ent_visible(c, ent)) { + found_add(fd, 'T', ent); + } else { + fd->invisible = true; + } +} + +/* The namespace `seg` under `parent` as this context sees it. For product + * code a namespace that only test code declares does not exist: it is of no + * program product code is compiled with, so its name neither stands in the + * way of what the scope has further out nor makes a name under it the + * repository's. CS_NONE when none is in view. That one was passed over is + * noted in the context (what the reference names may then be test code's: + * resolve_ref asks). */ +static int ns_in_view(const cs_ctx_t *c, int parent, const char *seg, size_t len) { + int child = ns_find(c->ix, parent, seg, len); + if (child >= 0 && c->prod && !c->ix->nss[child].prod) { + if (c->passed_over) { + *c->passed_over = true; + } + return CS_NONE; + } + return child; +} + +/* Every type among tops[lo, hi) -- other owners' types of one name -- is a + * candidate: one binds, several are ambiguous (`why`). Nothing chooses + * between two of them. */ +static void add_owners(const cs_ctx_t *c, int lo, int hi, cs_why_t why, cs_found_t *fd) { + if (hi - lo > CS_MAX_FOREIGN) { + fd->n += PAIR_LEN; /* more of them than one lookup compares: ambiguous */ + fd->why = CS_WHY_LIMIT; + return; + } + int before = fd->n; + for (int i = lo; i < hi; i++) { + found_add_type(c, fd, c->ix->tops[i].ent); + } + if (fd->n - before > SKIP_ONE) { + fd->why = why; + } +} + +/* The entry among tops[lo, split) -- the shared trees' declarations of one + * name -- that is the context's own: what its own tree declares (a complete + * type, or a part of the shared trees' partial one). CS_NONE for a file of a + * project, and when the tree declares none. */ +static int own_shared(const cs_ctx_t *c, int lo, int split) { + const cs_index_t *ix = c->ix; + if (c->group >= 0 || c->unit < 0 || lo >= split) { + return CS_NONE; + } + int mine = named_of_owner(ix->tops, lo, split, shared_owner(c->unit)); + int last = split - SKIP_ONE; /* the parts stand last among the shared trees' */ + if (mine < 0 && ix->tops[last].owner == CS_POOL && + ent_in_unit(&ix->ents[ix->tops[last].ent], c->unit)) { + mine = last; + } + return mine; +} + +/* The top-level types of one namespace, name and arity a reference from this + * context can mean: its own assembly's type (for a file of a shared tree: + * what its own tree declares). Else every shared tree's and every other + * assembly's type of that name alike: one binds, several are ambiguous. An + * assembly's type that has the shared trees' declaration as its twin is that + * declaration seen from the assembly, and no second type. */ +static void add_top_types(const cs_ctx_t *c, int ns, const char *name, int arity, cs_found_t *fd) { + const cs_named_t *arr = c->ix->tops; + int hi = 0; + int lo = named_range(arr, c->ix->ntops, ns, name, arity, &hi); + if (lo >= hi) { + return; + } + int split = named_first_assembly(arr, lo, hi); + int mine = c->group >= 0 ? named_of_owner(arr, split, hi, c->group) : own_shared(c, lo, split); + if (mine >= 0) { + found_add_type(c, fd, arr[mine].ent); + return; + } + int before = fd->n; + add_owners(c, lo, split, CS_WHY_SHARED, fd); + int shared = fd->n - before; + if (hi - split > CS_MAX_FOREIGN) { + fd->n += PAIR_LEN; /* more of them than one lookup compares: ambiguous */ + fd->why = CS_WHY_LIMIT; + return; + } + bool joined = false; + for (int i = split; i < hi; i++) { + const cs_entity_t *e = &c->ix->ents[arr[i].ent]; + if (e->twin < 0 || !ent_visible(c, e->twin)) { + found_add_type(c, fd, arr[i].ent); + } else { + joined = joined || e->joined; + } + } + if (fd->n - before > SKIP_ONE && shared < PAIR_LEN) { + fd->why = CS_WHY_ASSEMBLIES; + } + /* the name is one type only because a stub was joined to it */ + fd->joined = fd->joined || (joined && fd->n - before == SKIP_ONE); +} + +/* The type of this name and arity nested in `outer`: declared in that type + * itself, or in the shared trees' declaration that belongs to it. */ +static void add_nested_types(const cs_ctx_t *c, int outer, const char *name, int arity, + cs_found_t *fd) { + const cs_index_t *ix = c->ix; + cs_view_t v = view_of(ix, outer); + for (int k = 0; k < v.n; k++) { + int hi = 0; + int lo = named_range(ix->kids, ix->nkids, v.ent[k], name, arity, &hi); + if (lo < hi) { + /* one entity per outer type; the one of the type itself has the + * twin's as its own twin */ + found_add_type(c, fd, ix->kids[lo].ent); + /* a contract's nested type that only the joined implementation has */ + fd->joined = fd->joined || (k > 0 && ix->ents[outer].joined); + return; + } + } +} + +static void add_types(const cs_ctx_t *c, bool top, int scope, const char *name, int arity, + cs_found_t *fd) { + if (top) { + add_top_types(c, scope, name, arity, fd); + } else { + add_nested_types(c, scope, name, arity, fd); + } +} + +/* What a lookup asks one scope for. A parameter list is no part of it: the + * compiler finds the name first and matches the overloads afterwards. */ +typedef struct { + const char *name; + int arity; /* written type arguments; CS_ARITY_NONE when none */ + bool types_only; /* a qualifier: a namespace or a type */ + bool statics; /* through `using static`: static members only */ + bool ctors; /* the type's own name: its constructors (`statics`: the static one) */ + char kind; /* 0, or the only member kind a doc ID names: c v p e */ +} cs_query_t; + +static int type_arity(int written) { + return written > 0 ? written : 0; +} + +/* Members of one type under one key: those the entity declares and those the + * shared trees' parts that belong to it declare -- at most two ranges of the + * member table. */ +typedef struct { + int lo[PAIR_LEN]; + int hi[PAIR_LEN]; + int n; + int total; +} cs_mspans_t; + +/* The members of `ent` named `name`; with `group` >= 0 only those of that + * group with exactly the signature `sig`. */ +static cs_mspans_t mref_spans(const cs_index_t *ix, int ent, const char *name, int group, + const char *sig) { + cs_mspans_t sp = {0}; + cs_view_t v = view_of(ix, ent); + for (int k = 0; k < v.n; k++) { + int hi = 0; + int lo = mref_range(ix, v.ent[k], name, group, sig, &hi); + if (lo < hi) { + sp.lo[sp.n] = lo; + sp.hi[sp.n] = hi; + sp.n++; + sp.total += hi - lo; + } + } + return sp; +} + +/* The members `ent` declares under the query's name. A type's own name names + * its constructors, which no lookup by name finds. */ +static cs_mspans_t member_spans(const cs_ctx_t *c, int ent, const cs_query_t *q) { + if (!q->ctors && strcmp(q->name, c->ix->ents[ent].name) == 0) { + return (cs_mspans_t){0}; + } + return mref_spans(c->ix, ent, q->name, CS_NONE, NULL); +} + +/* True when the member takes part in a lookup of this query: methods of any + * arity for a name without type arguments, of that arity with them; every + * other member only without them. */ +static bool member_viable(const cs_ctx_t *c, const cs_mref_t *r, const cs_query_t *q) { + const cs_member_t *m = mref_member(c->ix, r); + if (r->group == PAIR_LEN || (q->kind && m->kind != q->kind)) { + return false; /* an operator or indexer has no identifier */ + } + if (q->ctors) { + return m->kind == 'c' && m->is_static == q->statics; + } + if (q->statics && !m->is_static) { + return false; + } + return m->kind == 'c' ? (q->arity <= 0 || m->arity == q->arity) : q->arity <= 0; +} + +static bool mref_visible(const cs_ctx_t *c, const cs_mref_t *r) { + return !(c->prod && (r->cls & CS_CLS_TEST)); +} + +enum { CS_HAS_NONE = 0, CS_HAS_VISIBLE, CS_HAS_INVISIBLE }; + +/* Does `ent` itself declare a member the query finds? A group larger than a + * lookup compares counts as there (and comes out ambiguous). */ +static int members_named(const cs_ctx_t *c, int ent, const cs_query_t *q) { + const cs_index_t *ix = c->ix; + cs_mspans_t sp = member_spans(c, ent, q); + if (sp.total > CS_MAX_OVERLOADS) { + return CS_HAS_VISIBLE; + } + int state = CS_HAS_NONE; + for (int s = 0; s < sp.n; s++) { + for (int i = sp.lo[s]; i < sp.hi[s]; i++) { + cs_work(SKIP_ONE); + if (!member_viable(c, &ix->mrefs[i], q)) { + continue; + } + if (mref_visible(c, &ix->mrefs[i])) { + return CS_HAS_VISIBLE; + } + state = CS_HAS_INVISIBLE; + } + } + return state; +} + +/* True when `ent` -- or the shared trees' declaration that belongs to it -- + * declares an operator or an indexer by this name (its token, or `this`) + * that the context may see: asked of what was noted when the members were + * listed (special_mark), whatever the number of such members. */ +static bool special_declared(const cs_ctx_t *c, int ent, const char *name) { + const cs_index_t *ix = c->ix; + cs_view_t v = view_of(ix, ent); + char key[CS_NAME_BUF + CBM_SZ_64]; + for (int k = 0; k < v.n; k++) { + cs_work(SKIP_ONE); + if (special_key(key, sizeof(key), c->prod, v.ent[k], name) && + cbm_ht_get(ix->specials, key)) { + return true; + } + } + return false; +} + +/* The supertypes of `ent`, nearest first: a class's base classes, an + * interface's base interfaces. At most CS_MAX_SUPERS; *more when the + * hierarchy goes on behind them. (The compiler's cref lookup never comes + * here: see CS_BIND_INHERITED.) */ +static int supers_of(const cs_index_t *ix, int ent, int *out, bool *more) { + int n = 0; + *more = false; + bool iface = ix->ents[ent].kind == 'i'; + int cur = ent; + for (int head = -SKIP_ONE; head < n; head++) { + if (head >= 0) { + cur = out[head]; + } + /* the bases its own declarations write, and those the shared trees' + * parts of it write */ + cs_view_t v = view_of(ix, cur); + for (int k = 0; k < v.n; k++) { + const cs_entity_t *e = &ix->ents[v.ent[k]]; + for (int b = 0; b < e->nbases; b++) { + char bk = ix->ents[e->bases[b]].kind; + if (iface ? bk != 'i' : !(bk == 'c' || bk == 'r')) { + continue; + } + bool seen = e->bases[b] == ent; + for (int j = 0; !seen && j < n; j++) { + seen = out[j] == e->bases[b]; + } + if (seen) { + continue; + } + if (n >= CS_MAX_SUPERS) { + *more = true; + return n; + } + out[n++] = e->bases[b]; + } + } + } + return n; +} + +/* ── Reference syntax ────────────────────────────────────────────── */ + +typedef struct { + char name[CS_NAME_BUF]; + int arity; /* CS_ARITY_NONE: no type arguments written */ + const char *targs; /* the type arguments as written, between the brackets; NULL: none */ + size_t targs_len; +} cs_seg_t; + +enum { CS_OP_NAME = 16 }; + +typedef struct { + char text[CS_REF_BUF]; /* the reference, trimmed: the segments' type arguments are in it */ + cs_seg_t segs[CS_MAX_SEGS]; + int nsegs; + bool has_params; + char params[CS_MAX_PARAMS][CS_PARAM_BUF]; + int nparams; + char sig[CS_MAX_PARAMS * CS_PARAM_BUF]; /* the parameters as a declaration's signature */ + bool sig_unknown; /* a parameter type nothing is known about */ + char docid; /* 0, or T M P F E N; O: DocFX's overload group */ + bool glob; /* starts at the global namespace */ + bool op; /* an operator, a conversion or an indexer */ + char op_name[CS_OP_NAME]; /* the name the scope records it under */ + bool maybe_indexer; /* `Item(...)`: a method of that name, else the indexer */ + bool keyword; /* a keyword alias, rewritten to its System type */ +} cs_ref_t; + +static bool is_open_bracket(char c) { + return c == '<' || c == '{' || c == '[' || c == '('; +} + +static bool is_close_bracket(char c) { + return c == '>' || c == '}' || c == ']' || c == ')'; +} + +/* Index just past the bracket group opened at s[i] (< { [ ( nest together). */ +static size_t group_end(const char *s, size_t n, size_t i) { + int depth = 0; + for (size_t k = i; k < n; k++) { + if (is_open_bracket(s[k])) { + depth++; + } else if (is_close_bracket(s[k]) && --depth == 0) { + return k + SKIP_ONE; + } + } + return n; +} + +/* Count top-level comma-separated items of s[0..n). */ +static int count_top(const char *s, size_t n) { + bool any = false; + int items = 1; + int depth = 0; + for (size_t i = 0; i < n; i++) { + char c = s[i]; + if (is_open_bracket(c)) { + depth++; + } else if (is_close_bracket(c)) { + depth--; + } else if (c == ',' && depth == 0) { + items++; + } + any = any || !isspace((unsigned char)c); + } + return any ? items : 0; +} + +/* An identifier as the scanner takes one: a letter of any script, a digit + * after the first position, an underscore. */ +static bool ident_ok(const char *s) { + if (strcmp(s, "#ctor") == 0 || strcmp(s, "#cctor") == 0) { + return true; + } + unsigned char first = (unsigned char)s[0]; + if (!(isalpha(first) || first == '_' || first >= CBM_SZ_128)) { + return false; + } + for (const char *p = s + SKIP_ONE; *p; p++) { + unsigned char ch = (unsigned char)*p; + if (!(isalnum(ch) || ch == '_' || ch >= CBM_SZ_128)) { + return false; + } + } + return true; +} + +/* One dotted segment: `Name`, `Name{T,U}`, `Name`, `Name``2`. */ +static bool parse_seg(const char *s, size_t n, cs_seg_t *out) { + while (n > 0 && isspace((unsigned char)*s)) { + s++; + n--; + } + while (n > 0 && isspace((unsigned char)s[n - SKIP_ONE])) { + n--; + } + out->arity = CS_ARITY_NONE; + out->targs = NULL; + out->targs_len = 0; + size_t name_end = n; + for (size_t i = 0; i < n; i++) { + if (s[i] == '`') { + name_end = i; + size_t d = i; + while (d < n && s[d] == '`') { + d++; + } + out->arity = atoi(s + d); + break; + } + if (s[i] == '{' || s[i] == '<') { + name_end = i; + size_t e = group_end(s, n, i); + size_t inner = e > i + PAIR_LEN ? e - i - PAIR_LEN : 0; + out->arity = count_top(s + i + SKIP_ONE, inner); + out->targs = s + i + SKIP_ONE; + out->targs_len = inner; + break; + } + } + const char *name = s; + if (name_end > 0 && name[0] == '@') { + name++; + name_end--; + } + if (name_end == 0 || name_end >= sizeof(out->name)) { + return false; + } + memcpy(out->name, name, name_end); + out->name[name_end] = '\0'; + return ident_ok(out->name); +} + +/* Split a dotted path (dots inside type-argument groups do not split). */ +static bool parse_path(const char *s, size_t n, cs_seg_t *segs, int *nsegs) { + *nsegs = 0; + size_t start = 0; + int depth = 0; + for (size_t i = 0; i <= n; i++) { + char c = i < n ? s[i] : '.'; + if (is_open_bracket(c)) { + depth++; + } else if (is_close_bracket(c)) { + depth--; + } else if (c == '.' && depth == 0) { + if (*nsegs >= CS_MAX_SEGS || !parse_seg(s + start, i - start, &segs[*nsegs])) { + return false; + } + (*nsegs)++; + start = i + SKIP_ONE; + } + } + return *nsegs > 0; +} + +static const char *const CS_KEYWORD_TYPES[][2] = { + {"int", "Int32"}, {"string", "String"}, {"object", "Object"}, {"bool", "Boolean"}, + {"byte", "Byte"}, {"sbyte", "SByte"}, {"short", "Int16"}, {"ushort", "UInt16"}, + {"uint", "UInt32"}, {"long", "Int64"}, {"ulong", "UInt64"}, {"float", "Single"}, + {"double", "Double"}, {"decimal", "Decimal"}, {"char", "Char"}, {"nint", "IntPtr"}, + {"nuint", "UIntPtr"}, {"void", "Void"}, +}; + +static const char *keyword_type(const char *s) { + for (size_t i = 0; i < sizeof(CS_KEYWORD_TYPES) / sizeof(CS_KEYWORD_TYPES[0]); i++) { + if (strcmp(s, CS_KEYWORD_TYPES[i][0]) == 0) { + return CS_KEYWORD_TYPES[i][1]; + } + } + return NULL; +} + +/* The token a metadata operator name stands for (`op_Addition` -> `+`): the + * name the scope records an operator under. NULL for a name that is none. */ +static const char *operator_token(const char *name, size_t len) { + static const struct { + const char *meta; + const char *token; + } ops[] = { + {"Addition", "+"}, + {"UnaryPlus", "+"}, + {"Subtraction", "-"}, + {"UnaryNegation", "-"}, + {"Multiply", "*"}, + {"Division", "/"}, + {"Modulus", "%"}, + {"BitwiseAnd", "&"}, + {"BitwiseOr", "|"}, + {"ExclusiveOr", "^"}, + {"LeftShift", "<<"}, + {"RightShift", ">>"}, + {"UnsignedRightShift", ">>>"}, + {"Equality", "=="}, + {"Inequality", "!="}, + {"LessThan", "<"}, + {"GreaterThan", ">"}, + {"LessThanOrEqual", "<="}, + {"GreaterThanOrEqual", ">="}, + {"LogicalNot", "!"}, + {"OnesComplement", "~"}, + {"Increment", "++"}, + {"Decrement", "--"}, + {"True", "true"}, + {"False", "false"}, + {"Implicit", "implicit"}, + {"Explicit", "explicit"}, + {"CheckedAddition", "+"}, + {"CheckedSubtraction", "-"}, + {"CheckedMultiply", "*"}, + {"CheckedDivision", "/"}, + {"CheckedUnaryNegation", "-"}, + {"CheckedIncrement", "++"}, + {"CheckedDecrement", "--"}, + {"CheckedExplicit", "explicit"}, + }; + for (size_t i = 0; i < sizeof(ops) / sizeof(ops[0]); i++) { + if (strlen(ops[i].meta) == len && memcmp(ops[i].meta, name, len) == 0) { + return ops[i].token; + } + } + return NULL; +} + +static bool starts_word(const char *s, const char *word) { + size_t n = strlen(word); + return strncmp(s, word, n) == 0 && !isalnum((unsigned char)s[n]) && s[n] != '_'; +} + +static const char *skip_blanks(const char *s) { + while (*s == ' ') { + s++; + } + return s; +} + +/* The operator, conversion or indexer written at `q` (the start of a + * segment): its recorded name into `name`. false when `q` starts none. */ +static bool operator_at(const char *q, char name[CS_OP_NAME]) { + static const char op_prefix[] = "op_"; + q = skip_blanks(q); + bool implicit = starts_word(q, "implicit"); + if (implicit || starts_word(q, "explicit")) { + if (!starts_word(skip_blanks(q + strlen("implicit")), "operator")) { + return false; + } + snprintf(name, CS_OP_NAME, "%s", implicit ? "implicit" : "explicit"); + return true; + } + if (starts_word(q, "operator")) { + const char *t = skip_blanks(q + strlen("operator")); + if (starts_word(t, "checked")) { + t = skip_blanks(t + strlen("checked")); + } + size_t n = 0; + while (t[n] && t[n] != '(' && t[n] != ' ' && n + SKIP_ONE < CS_OP_NAME) { + n++; + } + snprintf(name, CS_OP_NAME, "%.*s", (int)n, n > 0 ? t : "?"); + return true; + } + if (starts_word(q, "this")) { + const char *t = skip_blanks(q + strlen("this")); + if (*t != '[' && *t != '\0') { + return false; + } + snprintf(name, CS_OP_NAME, "this"); + return true; + } + size_t pl = sizeof(op_prefix) - SKIP_ONE; + if (strncmp(q, op_prefix, pl) == 0 && isupper((unsigned char)q[pl])) { + size_t n = 0; + while (isalnum((unsigned char)q[pl + n])) { + n++; + } + const char *token = operator_token(q + pl, n); + if (!token) { + return false; /* `op_Custom`: an identifier like any other */ + } + snprintf(name, CS_OP_NAME, "%s", token); + return true; + } + return false; +} + +/* Where the reference's last segment starts an operator, a conversion or an + * indexer: its offset in `s`, or CS_NONE. */ +static int operator_start(const char *s, char name[CS_OP_NAME]) { + int depth = 0; + for (size_t i = 0; s[i]; i++) { + if ((i == 0 || (s[i - SKIP_ONE] == '.' && depth == 0)) && operator_at(s + i, name)) { + return (int)i; + } + if (is_open_bracket(s[i])) { + depth++; + } else if (is_close_bracket(s[i])) { + depth--; + } + } + return CS_NONE; +} + +/* A doc ID writes a type parameter by its position: `0 (of the type and the + * types around it), ``0 (of the method). Such a parameter is kept as it + * stands (without a by-reference mark): its position is what is compared. */ +static bool slot_param(const char *s, size_t n, char *out, size_t cap) { + while (n > 0 && isspace((unsigned char)*s)) { + s++; + n--; + } + while (n > 0 && (isspace((unsigned char)s[n - SKIP_ONE]) || s[n - SKIP_ONE] == '@')) { + n--; + } + size_t ticks = 0; + while (ticks < n && s[ticks] == '`') { + ticks++; + } + if (ticks == 0 || ticks > PAIR_LEN || ticks >= n || !isdigit((unsigned char)s[ticks])) { + return false; + } + size_t d = ticks; + while (d < n && isdigit((unsigned char)s[d])) { + d++; + } + for (size_t k = d; k < n; k++) { + if (!strchr("[],*", s[k])) { + return false; + } + } + if (n >= cap) { + return false; + } + memcpy(out, s, n); + out[n] = '\0'; + return true; +} + +/* The written parameter list s[from, to) into the reference: every parameter + * normalized, and all of them as one signature. false for more parameters + * than a reference may have. */ +static bool parse_params(cs_ref_t *r, const char *s, size_t from, size_t to) { + bool any = false; + for (size_t i = from; i < to && !any; i++) { + any = !isspace((unsigned char)s[i]); + } + size_t start = from; + int depth = 0; + size_t w = 0; + for (size_t i = from; any && i <= to; i++) { + char c = i < to ? s[i] : ','; + if (is_open_bracket(c)) { + depth++; + } else if (is_close_bracket(c)) { + depth--; + } else if (c == ',' && depth == 0) { + if (r->nparams >= CS_MAX_PARAMS) { + return false; + } + char *p = r->params[r->nparams++]; + if (!slot_param(s + start, i - start, p, CS_PARAM_BUF)) { + (void)cbm_doclink_cs_norm_type(s + start, i - start, p, CS_PARAM_BUF); + } + size_t pl = strlen(p); + r->sig_unknown = r->sig_unknown || strchr(p, '?') != NULL; + if (w > 0) { + r->sig[w++] = '|'; + } + memcpy(r->sig + w, p, pl); + w += pl; + start = i + SKIP_ONE; + } + } + r->sig[w] = '\0'; + return true; +} + +/* A written reference into its parts. false for one this code does not + * understand -- and for one longer than CS_REF_BUF, which is never cut and + * resolved by what is left of it. */ +static bool parse_cref(const char *raw, cs_ref_t *r) { + static const char global_prefix[] = "global::"; + r->nsegs = 0; + r->nparams = 0; + r->has_params = r->sig_unknown = r->glob = r->op = r->maybe_indexer = r->keyword = false; + r->docid = 0; + r->sig[0] = '\0'; + r->op_name[0] = '\0'; + const char *in = raw ? raw : ""; + while (isspace((unsigned char)*in)) { + in++; + } + size_t n = strlen(in); + while (n > 0 && isspace((unsigned char)in[n - SKIP_ONE])) { + n--; + } + if (n == 0 || n >= sizeof(r->text)) { + return false; + } + memcpy(r->text, in, n); + r->text[n] = '\0'; + char *s = r->text; + if (n > PAIR_LEN && s[1] == ':' && strchr("TMPFENO!", s[0])) { + char k = s[0]; + if (k == '!') { + return false; /* the compiler's error marker: it could not bind the reference */ + } + s += PAIR_LEN; + while (isspace((unsigned char)*s)) { + s++; + } + if (k == 'O') { /* DocFX's overload group: a member without a signature */ + char *paren = strchr(s, '('); + if (paren) { + *paren = '\0'; + } + } + r->docid = k; + r->glob = true; /* a doc ID is a full name */ + } + size_t gl = sizeof(global_prefix) - SKIP_ONE; + if (strncmp(s, global_prefix, gl) == 0) { + r->glob = true; + s += gl; + } + n = strlen(s); + int op = operator_start(s, r->op_name); + if (op >= 0) { + r->op = true; + return op == 0 || parse_path(s, (size_t)op - SKIP_ONE, r->segs, &r->nsegs); + } + /* `Path(params)`: the first top-level parenthesis starts the list. */ + size_t paren = n; + int depth = 0; + for (size_t i = 0; i < n; i++) { + char c = s[i]; + if (c == '<' || c == '{' || c == '[') { + depth++; + } else if (c == '>' || c == '}' || c == ']') { + depth--; + } else if (c == '(' && depth == 0) { + paren = i; + break; + } + } + if (!parse_path(s, paren, r->segs, &r->nsegs)) { + return false; + } + if (paren < n) { + r->has_params = true; + size_t close = group_end(s, n, paren); + bool closed = close > paren && s[close - SKIP_ONE] == ')'; + if (!parse_params(r, s, paren + SKIP_ONE, closed ? close - SKIP_ONE : n)) { + return false; + } + } + r->maybe_indexer = r->has_params && strcmp(r->segs[r->nsegs - SKIP_ONE].name, "Item") == 0; + return true; +} + +/* ── Results ─────────────────────────────────────────────────────── */ + +/* Where the complete declarations of `e` in units before `unit` end. */ +static int fulls_before(const cs_entity_t *e, int unit) { + int lo = 0; + int hi = e->nfulls; + while (lo < hi) { + int mid = lo + ((hi - lo) / PAIR_LEN); + if (e->fulls[mid].unit < unit) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + return lo; +} + +/* The complete declaration of `e` that stands in the referencing file's own + * project (or shared tree) and that it may bind: one with a node when + * there is one, the nearest of several. NULL when there is none -- and when + * that unit holds more of them than one lookup compares (*too_many). */ +static const cs_full_t *own_full(const cs_ctx_t *c, const cs_entity_t *e, bool *too_many) { + const cs_index_t *ix = c->ix; + int lo = fulls_before(e, c->unit); + int hi = fulls_before(e, c->unit + SKIP_ONE); + if (hi - lo > CS_MAX_OVERLOADS) { + *too_many = true; + return NULL; + } + const cs_full_t *best = NULL; + bool best_node = false; + for (int i = lo; i < hi; i++) { + const cs_full_t *d = &e->fulls[i]; + if (c->prod && ix->files[d->file].is_test) { + continue; + } + bool node = ix->files[d->file].types[d->type].node != NULL; + if (!best || (node && !best_node) || + (node == best_node && file_nearer(ix, d->file, best->file, c->file))) { + best = d; + best_node = node; + } + } + return best; +} + +/* The one assembly that has only stubs of the type and takes the shared + * trees' declaration `ent` as its implementation; CS_NONE when there is + * none, or more than one. Its stubs are the type's stubs: where the + * implementation has no node, or a parse error hides a member of it, the + * stub's stands in -- as it does inside one assembly. */ +static int stub_user(const cs_index_t *ix, int ent) { + int stub = ix->ents[ent].stub; + return stub >= 0 ? stub : CS_NONE; +} + +/* The implementation's node among one entity's own declarations. What is an + * alternative is not chosen between: + * - complete declarations of the type in two or more projects (the + * flavours of one assembly's type): the one of the referencing file's own + * project is meant -- and when that one has no node, no other flavour's + * stands in for it; + * - parts in two or more shared trees, which may or may not be compiled + * together: the ones of the referencing file's own tree. + * From anywhere else the type is AMBIGUOUS (*why). GAP when no + * implementation in view has a node. */ +static cs_pick_t impl_node(const cs_ctx_t *c, const cs_entity_t *e, const cbm_gbuf_node_t **node, + cs_why_t *why) { + if ((c->prod ? e->full_units_prod : e->full_units) >= PAIR_LEN) { + bool too_many = false; + const cs_full_t *own = own_full(c, e, &too_many); + if (!own) { + *why = too_many ? CS_WHY_LIMIT : CS_WHY_FLAVOURS; + return CS_PICK_AMBIGUOUS; + } + *node = c->ix->files[own->file].types[own->type].node; + return *node ? CS_PICK_NODE : CS_PICK_GAP; + } + if (e->shared_parts && e->nunits >= PAIR_LEN && !ent_in_unit(e, c->unit)) { + *why = CS_WHY_SHARED; + return CS_PICK_AMBIGUOUS; + } + *node = class_node(c, e, false); + return *node ? CS_PICK_NODE : CS_PICK_GAP; +} + +/* The type's node for a reference from this context: an implementation's -- + * of the type's own declarations, then of the shared trees' declaration that + * belongs to it -- and a reference assembly's stub's only when no + * implementation has one. Never a test declaration for product code. A + * visible type whose eligible declarations all lack a node is a graph gap. A + * node that is the type's only because a contract is joined to its + * implementation is never an exact binding. */ +static cs_res_t type_result(const cs_ctx_t *c, int ent, bool exact) { + const cs_index_t *ix = c->ix; + cs_view_t v = view_of(ix, ent); + const cbm_gbuf_node_t *node = NULL; + bool in_view = false; + for (int k = 0; k < v.n; k++) { + const cs_entity_t *e = &ix->ents[v.ent[k]]; + cs_why_t why = CS_WHY_SCOPE; + cs_pick_t p = impl_node(c, e, &node, &why); + if (p == CS_PICK_NODE) { + return res_edge(node, exact && !(k > 0 && ix->ents[ent].joined)); + } + if (p == CS_PICK_AMBIGUOUS) { + return res_ambiguous(why); + } + in_view = in_view || declared_in_view(c, e); + } + for (int k = 0; k < v.n; k++) { + node = class_node(c, &ix->ents[v.ent[k]], true); + if (node) { + return res_edge(node, exact); + } + } + int stubs = stub_user(ix, ent); + node = stubs >= 0 ? class_node(c, &ix->ents[stubs], true) : NULL; + if (node) { + return res_edge(node, false); + } + return res_unres(in_view ? CBM_DOCLINK_REASON_GRAPH_GAP : CBM_DOCLINK_REASON_TEST_ONLY); +} + +/* Bind the members in `sp` -- one name, group and signature: the + * declarations of one member. An implementation's node before a reference + * assembly's stub's; never a test declaration for product code; the nearest + * of several. Declarations of the member in two or more projects (or shared + * trees) are alternatives: the one of the referencing file's own is meant, + * and from anywhere else none can be chosen. More declarations than + * one lookup compares are ambiguous as well. `view`: the type the members + * were looked up in (CS_NONE: none to speak of) -- a member a contract has + * only through the implementation joined to it is never an exact binding. */ +static cs_res_t bind_members(const cs_ctx_t *c, const cs_mspans_t *sp, bool exact, int view) { + const cs_index_t *ix = c->ix; + if (sp->total > CS_MAX_OVERLOADS) { + return res_ambiguous(CS_WHY_LIMIT); + } + int first_unit = CS_NONE; + bool several = false; + bool own = false; + bool visible = false; + for (int s = 0; s < sp->n; s++) { + for (int i = sp->lo[s]; i < sp->hi[s]; i++) { + const cs_mref_t *r = &ix->mrefs[i]; + if (!mref_visible(c, r)) { + continue; + } + visible = true; + if (r->cls & CS_CLS_REF) { + continue; + } + int unit = ix->files[r->file].unit; + own = own || unit == c->unit; + several = several || (first_unit >= 0 && unit != first_unit); + first_unit = first_unit >= 0 ? first_unit : unit; + } + } + if (!visible) { + return res_unres(CBM_DOCLINK_REASON_TEST_ONLY); + } + if (several && !own) { + return res_ambiguous(CS_WHY_FLAVOURS); + } + /* an implementation's node (the own one of alternatives), else a stub's */ + const cs_mref_t *best[PAIR_LEN] = {NULL, NULL}; + for (int s = 0; s < sp->n; s++) { + for (int i = sp->lo[s]; i < sp->hi[s]; i++) { + const cs_mref_t *r = &ix->mrefs[i]; + bool stub = (r->cls & CS_CLS_REF) != 0; + if (!mref_visible(c, r) || (r->cls & CS_CLS_NO_NODE) || + (!stub && several && ix->files[r->file].unit != c->unit)) { + continue; + } + if (!best[stub] || file_nearer(ix, r->file, best[stub]->file, c->file)) { + best[stub] = r; + } + } + } + const cs_mref_t *pick = best[0] ? best[0] : best[SKIP_ONE]; + if (!pick) { + return res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); + } + bool through_join = view >= 0 && ix->ents[view].joined && pick->ent != view; + return res_edge(mref_member(ix, pick)->node, exact && !through_join); +} + +/* Bind the type's members of one name, group and signature. A member whose + * implementation has no node is its stub's, where one assembly's stubs stand + * for this declaration (stub_user): never an exact binding. */ +static cs_res_t bind_signature(const cs_ctx_t *c, int ent, const char *name, int group, + const char *sig, bool exact) { + cs_mspans_t sp = mref_spans(c->ix, ent, name, group, sig); + cs_res_t res = bind_members(c, &sp, exact, ent); + int stubs = res_is(&res, CBM_DOCLINK_REASON_GRAPH_GAP) ? stub_user(c->ix, ent) : CS_NONE; + if (stubs >= 0) { + cs_mspans_t of_stubs = mref_spans(c->ix, stubs, name, group, sig); + cs_res_t stub = bind_members(c, &of_stubs, false, CS_NONE); + if (stub.st == CS_OK) { + return stub; + } + } + return res; +} + +/* Which written segments name the declaring type and the member: a written + * type argument stands for the type parameter at its position there. */ +typedef struct { + int type_seg; /* CS_NONE: the type is not written (a simple name in scope) */ + int member_seg; /* CS_NONE: the member is not written (a constructor by its type) */ +} cs_where_t; + +/* How a found name is used. */ +typedef struct { + const cs_ref_t *r; + bool has_params; /* a parameter list follows the name */ + cs_where_t where; + bool exact; /* tier of a member bound through the written path */ + bool exact_type; /* ... and of a type the whole path names */ +} cs_use_t; + +static size_t suffix_start(const char *s, size_t n) { + size_t e = n; + while (e > 0 && strchr("[],*", s[e - SKIP_ONE])) { + e--; + } + return e; +} + +/* Position of `name` among the ','-separated type arguments of a segment. */ +static int targ_position(const cs_seg_t *seg, const char *name, size_t len) { + int pos = 0; + int depth = 0; + size_t start = 0; + for (size_t i = 0; seg->targs && i <= seg->targs_len; i++) { + char ch = i < seg->targs_len ? seg->targs[i] : ','; + if (is_open_bracket(ch)) { + depth++; + } else if (is_close_bracket(ch)) { + depth--; + } else if (ch == ',' && depth == 0) { + const char *a = seg->targs + start; + size_t al = i - start; + while (al > 0 && isspace((unsigned char)*a)) { + a++; + al--; + } + while (al > 0 && isspace((unsigned char)a[al - SKIP_ONE])) { + al--; + } + if (al == len && memcmp(a, name, len) == 0) { + return pos; + } + pos++; + start = i + SKIP_ONE; + } + } + return CS_NONE; +} + +/* The position a doc ID's `N (*method false) or ``N (*method true) names; + * CS_NONE for any other text. */ +static int slot_of(const char *s, size_t n, bool *method) { + size_t ticks = 0; + while (ticks < n && s[ticks] == '`') { + ticks++; + } + if (ticks == 0 || ticks > PAIR_LEN || ticks == n) { + return CS_NONE; + } + int pos = 0; + for (size_t i = ticks; i < n; i++) { + if (!isdigit((unsigned char)s[i]) || pos > CBM_SZ_4K) { + return CS_NONE; + } + pos = (pos * CBM_DECIMAL_BASE) + (s[i] - '0'); + } + *method = ticks == PAIR_LEN; + return pos; +} + +/* A written type variable and a declared one are the same when they stand at + * the same position of the same list: the method's own list, the declaring + * type's, its outer type's, and so on. A doc ID writes the position itself; + * there the positions of a type's list count on from its outer types'. */ +static bool same_type_variable(const cs_ctx_t *c, const cs_ref_t *r, cs_where_t w, + const cs_mref_t *mr, const char *written, size_t wl, + const char *declared, size_t dl) { + const cs_file_t *f = &c->ix->files[mr->file]; + bool slot_method = false; + int slot = slot_of(written, wl, &slot_method); + int pos = tparam_find(f, f->ntypes + mr->midx, declared, dl); + if (pos >= 0) { + if (slot >= 0) { + return slot_method && slot == pos; + } + return w.member_seg >= 0 && targ_position(&r->segs[w.member_seg], written, wl) == pos; + } + int seg = w.type_seg; + for (int t = f->members[mr->midx].type; t >= 0; t = f->types[t].outer, seg--) { + pos = tparam_find(f, t, declared, dl); + if (pos < 0) { + continue; + } + if (slot >= 0) { + int before = 0; + for (int o = f->types[t].outer; o >= 0; o = f->types[o].outer) { + before += f->types[o].arity; + } + return !slot_method && slot == before + pos; + } + return seg >= 0 && targ_position(&r->segs[seg], written, wl) == pos; + } + return false; +} + +enum { CS_FIT_NO = 0, CS_FIT_UNKNOWN, CS_FIT_EXACT }; + +/* How the written parameter list fits a declared signature: EXACT when every + * parameter type is the declared one (the same text, or the same type + * variable), UNKNOWN when the rest are types nothing is known about on + * either side ("?"), NO otherwise. */ +static int sig_fit(const cs_ctx_t *c, const cs_ref_t *r, cs_where_t w, const cs_mref_t *mr) { + const char *sig = mref_member(c->ix, mr)->sig; + /* counted when the declaration was read: a declared signature is not + * walked for every reference that cannot mean it */ + int nd = mref_member(c->ix, mr)->nparams; + if (nd != r->nparams) { + return CS_FIT_NO; + } + cs_work((uint64_t)nd); + int fit = CS_FIT_EXACT; + const char *p = sig; + for (int i = 0; i < nd; i++) { + const char *e = strchr(p, '|'); + size_t dn = e ? (size_t)(e - p) : strlen(p); + const char *a = r->params[i]; + size_t an = strlen(a); + if (!(an == dn && memcmp(a, p, an) == 0)) { + size_t ab = suffix_start(a, an); + size_t db = suffix_start(p, dn); + bool same_suffix = (an - ab) == (dn - db) && memcmp(a + ab, p + db, an - ab) == 0; + if (same_suffix && same_type_variable(c, r, w, mr, a, ab, p, db)) { + /* the same type variable */ + } else if ((an == SKIP_ONE && a[0] == '?') || (dn == SKIP_ONE && p[0] == '?')) { + fit = CS_FIT_UNKNOWN; + } else { + return CS_FIT_NO; + } + } + p = e ? e + SKIP_ONE : p + dn; + } + return fit; +} + +typedef enum { + CS_MB_NONE = 0, /* the entity declares no such member */ + CS_MB_INVISIBLE, /* only its test declarations do, and the reference is product code */ + CS_MB_MISMATCH, /* it declares the name; nothing of it takes the written parameters */ + CS_MB_RES, /* *res is the answer */ +} cs_mb_t; + +/* The selected members of one side (generic callables, non-generic ones, + * values): how many, and whether they are one declaration's worth -- one + * signature and arity. */ +typedef struct { + int n; + const char *sig; + int arity; + bool many; +} cs_side_t; + +static void side_add(cs_side_t *s, const cs_member_t *m) { + const char *sig = m->sig ? m->sig : ""; + if (s->n == 0) { + s->sig = sig; + s->arity = m->arity; + } else if (strcmp(s->sig, sig) != 0 || s->arity != m->arity) { + s->many = true; + } + s->n++; +} + +/* The members the query finds in `ent`, for a reference without a parameter + * list: one member binds; a method name without type arguments means the + * non-generic methods when there are any; an overload group, and a method + * beside a value of the same name, are ambiguous. */ +static cs_mb_t members_plain(const cs_ctx_t *c, int ent, const cs_query_t *q, bool exact, + cs_res_t *res) { + const cs_index_t *ix = c->ix; + cs_mspans_t sp = member_spans(c, ent, q); + if (sp.total == 0) { + return CS_MB_NONE; + } + if (sp.total > CS_MAX_OVERLOADS) { + *res = res_ambiguous(CS_WHY_LIMIT); /* more than one lookup compares */ + return CS_MB_RES; + } + cs_side_t plain = {0}; + cs_side_t generic = {0}; + cs_side_t values = {0}; + bool invisible = false; + for (int s = 0; s < sp.n; s++) { + for (int i = sp.lo[s]; i < sp.hi[s]; i++) { + cs_work(SKIP_ONE); + const cs_mref_t *mr = &ix->mrefs[i]; + if (!member_viable(c, mr, q)) { + continue; + } + if (!mref_visible(c, mr)) { + invisible = true; + continue; + } + const cs_member_t *m = mref_member(ix, mr); + side_add(m->kind != 'c' ? &values : (m->arity > 0 ? &generic : &plain), m); + } + } + const cs_side_t *calls = plain.n > 0 ? &plain : &generic; + if (calls->n + values.n == 0) { + return invisible ? CS_MB_INVISIBLE : CS_MB_NONE; + } + if ((calls->n > 0 && values.n > 0) || calls->many) { + *res = res_ambiguous(CS_WHY_SCOPE); + } else if (values.n > 0) { + *res = bind_signature(c, ent, q->name, 0, "", exact); + } else { + *res = bind_signature(c, ent, q->name, SKIP_ONE, calls->sig, exact); + } + return CS_MB_RES; +} + +/* The same for a reference WITH a parameter list: the overload with exactly + * the written parameter types (the same text, or the same type variables); + * only when there is none, one whose difference is a type nothing is known + * about. A non-generic match stands before a generic one; several distinct + * matches are ambiguous. A value never takes a parameter list. */ +static cs_mb_t members_signed(const cs_ctx_t *c, int ent, const cs_query_t *q, const cs_use_t *u, + cs_res_t *res) { + const cs_index_t *ix = c->ix; + cs_mspans_t sp = member_spans(c, ent, q); + if (sp.total == 0) { + return CS_MB_NONE; + } + if (sp.total > CS_MAX_OVERLOADS) { + /* too many to compare one by one: the written text itself can still + * be looked up */ + cs_mspans_t written = mref_spans(ix, ent, q->name, SKIP_ONE, u->r->sig); + *res = (written.total > 0 && !u->r->sig_unknown) ? bind_members(c, &written, u->exact, ent) + : res_ambiguous(CS_WHY_LIMIT); + return CS_MB_RES; + } + bool named = false; + bool invisible = false; + for (int pass = CS_FIT_EXACT; pass >= CS_FIT_UNKNOWN; pass--) { + cs_side_t plain = {0}; + cs_side_t generic = {0}; + for (int s = 0; s < sp.n; s++) { + for (int i = sp.lo[s]; i < sp.hi[s]; i++) { + cs_work(SKIP_ONE); + const cs_mref_t *mr = &ix->mrefs[i]; + if (!member_viable(c, mr, q)) { + continue; + } + if (!mref_visible(c, mr)) { + invisible = true; + continue; + } + named = true; + const cs_member_t *m = mref_member(ix, mr); + if (m->kind == 'c' && sig_fit(c, u->r, u->where, mr) == pass) { + side_add(m->arity > 0 ? &generic : &plain, m); + } + } + } + const cs_side_t *s = plain.n > 0 ? &plain : &generic; + if (s->n > 0) { + *res = s->many ? res_ambiguous(CS_WHY_SCOPE) + : bind_signature(c, ent, q->name, SKIP_ONE, s->sig, u->exact); + return CS_MB_RES; + } + } + if (named) { + return CS_MB_MISMATCH; + } + return invisible ? CS_MB_INVISIBLE : CS_MB_NONE; +} + +static cs_mb_t members_result(const cs_ctx_t *c, int ent, const cs_query_t *q, const cs_use_t *u, + cs_res_t *res) { + return u->has_params ? members_signed(c, ent, q, u, res) + : members_plain(c, ent, q, u->exact, res); +} + +/* True when a part of the type `ent` that this reference does not see + * declares `name`. `ent` is a shared trees' declaration that assemblies have + * as a part of their type, seen here without one of them: the reference + * stands in a shared tree, or in an assembly that has no part of its own. + * One of those assemblies' own parts declares an implementation of that name + * -- a nested type that the shared trees do not have, or (`types_only` + * unset) a member. Which assembly the reference is compiled into is not + * known, so the name is neither bound nor missing. A stub does not count: it + * declares what the implementation declares. The assemblies are not walked: + * what their parts add is looked up by the name (build_extras). *why: when + * more assemblies declare a nested type of the name than one lookup compares + * (and none of those compared is in view), that is said. */ +static bool unseen_part_declares(const cs_ctx_t *c, int ent, const char *name, int arity, + bool types_only, cs_why_t *why) { + const cs_index_t *ix = c->ix; + int hi = 0; + int lo = extra_range(ix, ent, name, type_arity(arity), &hi); + for (int i = lo; i < hi; i++) { + if (i - lo >= CS_MAX_FOREIGN) { + *why = CS_WHY_LIMIT; + return true; + } + cs_work(SKIP_ONE); + if (ent_visible(c, ix->extras[i].ent)) { + return true; + } + } + if (types_only) { + return false; + } + lo = extra_range(ix, ent, name, CS_NONE, &hi); + return lo < hi && (ix->extras[lo].prod || !c->prod); +} + +/* The reason for a member a type does not show (`name`, when it has one). */ +static cs_res_t member_unfound(const cs_ctx_t *c, int ent, const char *name, int arity) { + const cs_entity_t *e = &c->ix->ents[ent]; + cs_why_t why = CS_WHY_PARTS; + if (name && unseen_part_declares(c, ent, name, arity, false, &why)) { + return res_ambiguous(why); + } + if (e->incomplete) { + return res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); /* a parse error hides members */ + } + return res_unres(e->open_any ? CBM_DOCLINK_REASON_EXTERNAL : CBM_DOCLINK_REASON_MISSING); +} + +/* A member the implementation does not show because a parse error hides it, + * looked up in the stubs that stand for this declaration (stub_user). true + * when a stub binds it: never an exact binding. */ +static bool stub_member(const cs_ctx_t *c, int ent, const cs_query_t *q, const cs_use_t *u, + cs_res_t *res) { + int stubs = c->ix->ents[ent].incomplete ? stub_user(c->ix, ent) : CS_NONE; + cs_res_t of_stubs = res_unres(CBM_DOCLINK_REASON_MISSING); + cs_use_t via = *u; + via.exact = false; + via.exact_type = false; + if (stubs < 0 || members_result(c, stubs, q, &via, &of_stubs) != CS_MB_RES || + of_stubs.st != CS_OK) { + return false; + } + of_stubs.exact = false; + *res = of_stubs; + return true; +} + +/* True when `ent` declares an instance constructor without parameters. */ +static bool declares_default_ctor(const cs_ctx_t *c, int ent) { + const cs_index_t *ix = c->ix; + cs_mspans_t sp = mref_spans(ix, ent, ix->ents[ent].name, SKIP_ONE, ""); + for (int s = 0; s < sp.n; s++) { + for (int i = sp.lo[s]; i < sp.hi[s]; i++) { + if (!mref_member(ix, &ix->mrefs[i])->is_static) { + return true; + } + } + } + return false; +} + +/* A constructor of `ent`. A type has the instance constructors its source + * writes (a primary constructor among them: the scope records it), and ones + * no source writes: the parameterless one of a struct, and of a class or + * record that writes none; a record's copy constructor. Those are declared + * and have no node. No constructor is inherited. */ +static cs_res_t ctor_result(const cs_ctx_t *c, int ent, const cs_use_t *u) { + const cs_entity_t *e = &c->ix->ents[ent]; + cs_query_t q = {.name = e->name, .arity = CS_ARITY_NONE, .ctors = true, .kind = 'c'}; + cs_res_t res = res_unres(CBM_DOCLINK_REASON_MISSING); + cs_mb_t st = members_result(c, ent, &q, u, &res); + bool value_type = e->kind == 's' || e->kind == 't'; + bool record = e->kind == 'r' || e->kind == 't'; + if (st == CS_MB_RES) { + /* a struct has its parameterless constructor beside the one it writes */ + bool second = + !u->has_params && value_type && res.st == CS_OK && !declares_default_ctor(c, ent); + return second ? res_ambiguous(CS_WHY_SCOPE) : res; + } + if (st == CS_MB_INVISIBLE) { + return res_unres(CBM_DOCLINK_REASON_TEST_ONLY); + } + if (e->incomplete) { + return res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); + } + bool supplied = value_type || (st == CS_MB_NONE && (e->kind == 'c' || e->kind == 'r')); + if (supplied && (!u->has_params || u->r->nparams == 0)) { + return res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); + } + if (record && u->has_params && u->r->nparams == SKIP_ONE && + strcmp(u->r->params[0], e->name) == 0) { + return res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); + } + return res_unres(CBM_DOCLINK_REASON_MISSING); +} + +/* The static constructor of `ent` (a doc ID's `#cctor`). */ +static cs_res_t static_ctor_result(const cs_ctx_t *c, int ent, bool exact) { + cs_query_t q = {.name = c->ix->ents[ent].name, + .arity = CS_ARITY_NONE, + .statics = true, + .ctors = true, + .kind = 'c'}; + cs_res_t res = res_unres(CBM_DOCLINK_REASON_MISSING); + cs_mb_t st = members_plain(c, ent, &q, exact, &res); + if (st == CS_MB_RES) { + return res; + } + return st == CS_MB_INVISIBLE ? res_unres(CBM_DOCLINK_REASON_TEST_ONLY) + : member_unfound(c, ent, NULL, CS_ARITY_NONE); +} + +/* ── Scope lookup ────────────────────────────────────────────────── */ + +/* The types nested in `ent` and the members it declares itself, by the + * query. A doc ID's member kind names a member, never a nested type. */ +static void level_entity(const cs_ctx_t *c, int ent, const cs_query_t *q, cs_found_t *fd) { + /* the shared trees' declaration seen without an assembly's own parts of + * the type: a name those parts declare is not told from what is here */ + cs_why_t why = CS_WHY_PARTS; + if (c->ix->ents[ent].used && + unseen_part_declares(c, ent, q->name, q->arity, q->types_only, &why)) { + fd->n += PAIR_LEN; + fd->why = why; + return; + } + if (!q->kind) { + add_types(c, false, ent, q->name, type_arity(q->arity), fd); + } + if (q->types_only) { + return; + } + int has = members_named(c, ent, q); + if (has == CS_HAS_VISIBLE) { + found_add(fd, 'M', ent); + } else if (has == CS_HAS_INVISIBLE) { + fd->invisible = true; + } +} + +/* What an alias was resolved to, as a candidate. */ +static void add_alias_target(const cs_ctx_t *c, const cs_using_t *u, cs_found_t *fd) { + if (u->ent >= 0) { + found_add_type(c, fd, u->ent); + fd->joined = fd->joined || u->joined; + } else if (u->ent == CS_AMBIGUOUS) { + fd->n += PAIR_LEN; + } else if (u->ns >= 0) { + found_add(fd, 'N', u->ns); + } else { + found_add(fd, 'X', CS_NONE); + } +} + +static void level_aliases(const cs_ctx_t *c, const cs_using_t *us, int n, + const cs_using_index_t *index, const cs_query_t *q, cs_found_t *fd) { + /* an alias names a type or a namespace: no candidate for `Name{T}` */ + if (q->arity > 0) { + return; + } + if (!index || !index->ready) { + for (int i = 0; i < n; i++) { + cs_work(SKIP_ONE); + if (us[i].kind == 'a' && strcmp(us[i].alias, q->name) == 0) { + add_alias_target(c, &us[i], fd); + } + } + return; + } + size_t lo = 0; + size_t hi = index->naliases; + while (lo < hi) { + cs_work(SKIP_ONE); + size_t mid = lo + (hi - lo) / PAIR_LEN; + if (strcmp(index->aliases[mid].name, q->name) < 0) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + for (size_t i = lo; i < index->naliases; i++) { + cs_work(SKIP_ONE); + if (strcmp(index->aliases[i].name, q->name)) { + break; + } + add_alias_target(c, &us[index->aliases[i].id], fd); + } +} + +/* The original resolver is shared by the full scan and candidate path. */ +static void level_using(const cs_ctx_t *c, const cs_using_t *u, const cs_query_t *q, + bool a_type_name, cs_found_t *fd) { + cs_work(SKIP_ONE); + if (u->kind == 'n' && u->ns >= 0 && a_type_name) { + /* a using brings a namespace's types, not the namespaces in it */ + add_types(c, true, u->ns, q->name, type_arity(q->arity), fd); + } else if (u->kind == 's' && u->ent >= 0) { + cs_query_t statics = *q; + statics.statics = true; + level_entity(c, u->ent, &statics, fd); + } +} + +static size_t name_scope_range(const cs_index_t *ix, const char *name, size_t *end) { + size_t lo = 0; + size_t hi = ix->nname_scopes; + while (lo < hi) { + cs_work(SKIP_ONE); + size_t mid = lo + (hi - lo) / PAIR_LEN; + if (strcmp(ix->name_scopes[mid].name, name) < 0) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + size_t first = lo; + hi = ix->nname_scopes; + while (lo < hi) { + cs_work(SKIP_ONE); + size_t mid = lo + (hi - lo) / PAIR_LEN; + if (strcmp(ix->name_scopes[mid].name, name) <= 0) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + *end = lo; + return first; +} + +/* IDs are accumulated before any result is added: if growth fails, a full + * scan can retry without duplicating a partially resolved candidate set. */ +enum { CS_CANDIDATE_STACK = 32 }; +typedef struct { + int local[CS_CANDIDATE_STACK]; + int *ids; + size_t n; + size_t cap; +} cs_candidate_ids_t; + +static bool candidate_add(cs_candidate_ids_t *ids, int id) { + if (ids->n == ids->cap) { + if (ids->cap > SIZE_MAX / PAIR_LEN / sizeof(int)) { + return false; + } + size_t cap = ids->cap * PAIR_LEN; + int *grown = NULL; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + bool fail = atomic_exchange(&cs_fail_candidate_alloc, false); + if (fail) { + atomic_store(&cs_candidate_alloc_failed, true); + } else +#endif + { + grown = (int *)cbm_alloc(CBM_MEM_CLASS_OTHER, cap * sizeof(int)); + } + if (!grown) { + return false; + } + memcpy(grown, ids->ids, ids->n * sizeof(int)); + cs_work(ids->n); + if (ids->ids != ids->local) { + cbm_free(CBM_MEM_CLASS_OTHER, ids->ids); + } + ids->ids = grown; + ids->cap = cap; + } + ids->ids[ids->n++] = id; + return true; +} + +static bool candidate_scope(const cs_using_id_t *rows, size_t n, int scope, + cs_candidate_ids_t *ids) { + size_t lo = 0; + size_t hi = n; + while (lo < hi) { + cs_work(SKIP_ONE); + size_t mid = lo + (hi - lo) / PAIR_LEN; + if (rows[mid].scope < scope) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + for (size_t i = lo; i < n; i++) { + cs_work(SKIP_ONE); + if (rows[i].scope != scope) { + break; + } + if (!candidate_add(ids, rows[i].id)) { + return false; + } + } + return true; +} + +static int candidate_id_cmp(const void *a, const void *b) { + cs_work(SKIP_ONE); + return int_cmp(a, b); +} + +static void level_usings(const cs_ctx_t *c, const cs_using_t *us, int n, + const cs_using_index_t *index, const cs_query_t *q, cs_found_t *fd) { + if (!n) { + return; + } + bool a_type_name = cbm_ht_get(c->ix->type_names, q->name) != NULL; + bool indexed = index && index->ready && c->ix->name_scopes_ready; + if (indexed && !index->nentities && (!index->nnamespaces || !a_type_name)) { + return; /* no static candidates, and no namespace can contribute this name */ + } + size_t hi = 0; + size_t lo = indexed ? name_scope_range(c->ix, q->name, &hi) : 0; + /* A common name in a large repository must not make a small scope walk + * more postings than it has directives. The baseline scan stays cheap. */ + if (!indexed || hi - lo >= (size_t)n) { + for (int i = 0; i < n; i++) { + level_using(c, &us[i], q, a_type_name, fd); + } + return; + } + cs_candidate_ids_t ids = {.cap = CS_CANDIDATE_STACK}; + ids.ids = ids.local; + bool complete = true; + for (size_t i = lo; complete && i < hi; i++) { + cs_work(SKIP_ONE); + const cs_name_scope_t *row = &c->ix->name_scopes[i]; + complete = row->top + ? candidate_scope(index->namespaces, index->nnamespaces, row->scope, &ids) + : candidate_scope(index->entities, index->nentities, row->scope, &ids); + } + if (complete) { + if (ids.n > 1) { + qsort(ids.ids, ids.n, sizeof(int), candidate_id_cmp); + } + for (size_t i = 0; i < ids.n; i++) { + cs_work(SKIP_ONE); + /* Own and twin postings may select the SAME directive twice. + * Distinct directive IDs must still be resolved separately. */ + if (!i || ids.ids[i] != ids.ids[i - SKIP_ONE]) { + level_using(c, &us[ids.ids[i]], q, a_type_name, fd); + } + } + } + if (ids.ids != ids.local) { + cbm_free(CBM_MEM_CLASS_OTHER, ids.ids); + } + if (!complete) { + for (int i = 0; i < n; i++) { + level_using(c, &us[i], q, a_type_name, fd); + } + } +} + +/* True when the lookup is settled by what the last level added. A level that + * had only invisible candidates does not bind: the lookup goes on, and + * remembers. */ +static bool settled(cs_found_t *fd, bool *invisible) { + *invisible = *invisible || fd->invisible; + fd->invisible = false; + return fd->n > 0; +} + +/* Look a simple name up the way the compiler does in a cref: the documented + * method's type parameters; for every enclosing type its type parameters and + * what it declares itself; for every enclosing namespace its types and + * namespaces, and where a namespace declaration of the file stands, that + * declaration's aliases and then its usings. The first level that has the + * name decides. One step per enclosing type and per enclosing namespace: the + * scanner bounds both nestings. *statics: the name came through a `using + * static`. */ +static void lookup(const cs_ctx_t *c, const cs_query_t *q, cs_found_t *fd, bool *statics) { + const cs_index_t *ix = c->ix; + const cs_file_t *f = c->f; + size_t nl = strlen(q->name); + bool invisible = false; + memset(fd, 0, sizeof(*fd)); + *statics = false; + if (q->arity <= 0 && c->member >= 0 && + tparam_find(f, f->ntypes + c->member, q->name, nl) >= 0) { + found_add(fd, 'L', CS_NONE); + return; + } + for (int t = c->type; t >= 0; t = f->types[t].outer) { + cs_work(SKIP_ONE); + if (q->arity <= 0 && tparam_find(f, t, q->name, nl) >= 0) { + found_add(fd, 'L', CS_NONE); + return; + } + level_entity(c, f->types[t].entity, q, fd); + if (settled(fd, &invisible)) { + return; + } + } + int reg = c->region; + for (int ns = f->regions[reg].ns;; ns = ix->nss[ns].parent) { + cs_work(SKIP_ONE); + add_types(c, true, ns, q->name, type_arity(q->arity), fd); + int child = q->arity > 0 ? CS_NONE : ns_in_view(c, ns, q->name, nl); + if (child >= 0) { + found_add(fd, 'N', child); + } + /* a declaration of this namespace stands here: its directives are + * asked, unless they are the peers of the directive being resolved */ + bool here = reg >= 0 && f->regions[reg].ns == ns; + bool asked = here && reg != c->skip_region; + const cs_using_t *us = asked ? f->usings + f->regions[reg].u_lo : NULL; + int nus = asked ? f->regions[reg].u_hi - f->regions[reg].u_lo : 0; + const cs_using_index_t *usi = asked ? &f->regions[reg].using_index : NULL; + const cs_unit_t *unit = (ns == 0 && c->unit >= 0) ? &ix->units[c->unit] : NULL; + if (fd->n > 0) { + /* C# rejects an alias beside a type or namespace of its name */ + cs_found_t alias = {0}; + level_aliases(c, us, nus, usi, q, &alias); + if (unit) { + level_aliases(c, unit->usings, unit->nusings, &unit->using_index, q, &alias); + } + fd->n += alias.n > 0; + fd->in_namespace = true; + } + if (settled(fd, &invisible)) { + return; + } + level_aliases(c, us, nus, usi, q, fd); + if (unit) { + level_aliases(c, unit->usings, unit->nusings, &unit->using_index, q, fd); + } + if (settled(fd, &invisible)) { + fd->exact = true; + return; + } + level_usings(c, us, nus, usi, q, fd); + if (unit) { + level_usings(c, unit->usings, unit->nusings, &unit->using_index, q, fd); + } + if (settled(fd, &invisible)) { + *statics = fd->first.kind == 'M'; + return; + } + if (here) { + reg = f->regions[reg].parent; + } + if (ns == 0) { + break; + } + } + fd->invisible = invisible; +} + +/* ── Paths ───────────────────────────────────────────────────────── */ + +typedef enum { + CS_TP_TYPE = 0, /* the path names the type `ent` */ + CS_TP_NS, /* ... the namespace `ns` */ + CS_TP_PARTIAL, /* the type `ent` was reached; it has no type by the next segment */ + CS_TP_NSFAIL, /* the namespace `ns` was reached; it has nothing by the next segment */ + CS_TP_RES, /* `res` is the answer */ +} cs_tp_kind_t; + +typedef struct { + cs_tp_kind_t st; + int ent; + int ns; + bool exact; /* the path itself is a qualified name (or an alias, or absolute) */ + bool in_ns; /* its first segment is of an enclosing namespace (or absolute): one + * more segment makes a qualified name of it */ + const cs_seg_t *next; /* CS_TP_PARTIAL: the segment the type has nothing by */ + bool joined; /* a segment is one type only because a contract is joined to its + * implementation: nothing bound through the path is exact */ + cs_res_t res; +} cs_tp_t; + +static cs_tp_t tp_res(cs_res_t r) { + return (cs_tp_t){.st = CS_TP_RES, .res = r}; +} + +/* Follow segs[from, n) from where `cur` stands: each is a member of what the + * one before named -- a namespace or type of a namespace, a type nested in a + * type. There is no second try from anywhere else. */ +static cs_tp_t path_from(const cs_ctx_t *c, cs_tp_t cur, const cs_seg_t *segs, int from, int n) { + for (int i = from; i < n; i++) { + cs_found_t fd = {0}; + if (cur.st == CS_TP_NS) { + int child = segs[i].arity > 0 + ? CS_NONE + : ns_in_view(c, cur.ns, segs[i].name, strlen(segs[i].name)); + add_types(c, true, cur.ns, segs[i].name, type_arity(segs[i].arity), &fd); + if (child >= 0 && fd.n == 0 && !fd.invisible) { + cur.ns = child; + continue; + } + fd.n += child >= 0; + } else { + add_types(c, false, cur.ent, segs[i].name, type_arity(segs[i].arity), &fd); + } + if (fd.n > SKIP_ONE) { + return tp_res(found_ambiguous(&fd)); + } + if (fd.n == 0) { + if (fd.invisible) { + return tp_res(res_unres(CBM_DOCLINK_REASON_TEST_ONLY)); + } + cur.st = cur.st == CS_TP_NS ? CS_TP_NSFAIL : CS_TP_PARTIAL; + cur.next = &segs[i]; + return cur; + } + cur.st = CS_TP_TYPE; + cur.ent = fd.first.id; + if (fd.joined) { + cur.joined = true; + cur.exact = false; + cur.in_ns = false; + } + } + return cur; +} + +/* An open scope: a using in view names a namespace or type the repository + * does not declare, or the project's global usings could not be evaluated. + * A name found nowhere may come from there. */ +static bool scope_open(const cs_ctx_t *c) { + if (c->unit >= 0 && c->ix->units[c->unit].open) { + return true; + } + return c->f->regions[c->region].open; +} + +/* Why a simple name was found at no scope level. */ +static cs_res_t simple_unfound(const cs_ctx_t *c) { + const cs_index_t *ix = c->ix; + if (scope_open(c)) { + return res_unres(CBM_DOCLINK_REASON_EXTERNAL); + } + bool hidden = false; + for (int t = c->type; t >= 0; t = c->f->types[t].outer) { + const cs_entity_t *e = &ix->ents[c->f->types[t].entity]; + if (e->open_any) { + return res_unres(CBM_DOCLINK_REASON_EXTERNAL); /* a base outside the repository */ + } + hidden = hidden || e->incomplete; + } + /* An enclosing type with members a parse error hides: the name may be + * one of them. */ + return res_unres(hidden ? CBM_DOCLINK_REASON_GRAPH_GAP : CBM_DOCLINK_REASON_MISSING); +} + +/* The path segs[0, n): a type, a namespace, or how far it got. The first + * segment is a simple name that names a type or a namespace in scope; + * `global::` and a doc ID start at the global namespace instead. `qualifier`: + * the path qualifies a name that follows it. */ +static cs_tp_t resolve_path(const cs_ctx_t *c, const cs_seg_t *segs, int n, bool qualifier) { + if (c->glob) { + return path_from(c, (cs_tp_t){.st = CS_TP_NS, .ns = 0, .exact = true, .in_ns = true}, segs, + 0, n); + } + if (n <= 0) { + return tp_res(res_unres(CBM_DOCLINK_REASON_UNPARSEABLE)); + } + cs_query_t q = {.name = segs[0].name, .arity = segs[0].arity, .types_only = true}; + cs_found_t fd; + bool statics = false; + lookup(c, &q, &fd, &statics); + if (fd.n > SKIP_ONE) { + return tp_res(found_ambiguous(&fd)); + } + if (fd.n == 0) { + if (fd.invisible) { + return tp_res(res_unres(CBM_DOCLINK_REASON_TEST_ONLY)); + } + if (n == SKIP_ONE && !qualifier) { + return tp_res(simple_unfound(c)); + } + /* a qualifier that is no namespace and no type in scope: a type of + * the repository that is not in scope here, or a namespace the + * repository does not declare */ + return tp_res(res_unres(cbm_ht_get(c->ix->type_names, segs[0].name) + ? CBM_DOCLINK_REASON_MISSING + : CBM_DOCLINK_REASON_EXTERNAL)); + } + /* exact: a name through an alias, and a qualified name -- two segments + * or more, the first found among an enclosing namespace's own types and + * namespaces. A name found by its simple name alone, or through a using, + * is what the scope makes of it. */ + bool exact = (fd.exact || (n > SKIP_ONE && fd.in_namespace)) && !fd.joined; + bool in_ns = (fd.exact || fd.in_namespace) && !fd.joined; + switch (fd.first.kind) { + case 'T': + return path_from(c, + (cs_tp_t){.st = CS_TP_TYPE, + .ent = fd.first.id, + .exact = exact, + .in_ns = in_ns, + .joined = fd.joined}, + segs, SKIP_ONE, n); + case 'N': + return path_from( + c, (cs_tp_t){.st = CS_TP_NS, .ns = fd.first.id, .exact = exact, .in_ns = in_ns}, segs, + SKIP_ONE, n); + case 'L': + /* a type parameter: nothing is a member of one */ + return tp_res(n == SKIP_ONE ? (cs_res_t){.st = CS_LOCAL} + : res_unres(CBM_DOCLINK_REASON_MISSING)); + default: + return tp_res(res_unres(CBM_DOCLINK_REASON_EXTERNAL)); + } +} + +/* A namespace as a reference's target: declared, and without a node. */ +static cs_res_t namespace_result(void) { + return res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); +} + +/* The reason for a path that ended at a namespace without the next name: a + * namespace the repository declares has no such type (missing); one it only + * has namespaces under, and the global one, hold nothing of the repository + * (the name is outside) -- and so is what a standard-library namespace lacks, + * declared or not. */ +static cs_res_t namespace_unfound(const cs_ctx_t *c, int ns) { + const cs_ns_t *n = &c->ix->nss[ns]; + return res_unres((n->declared && !n->standard) ? CBM_DOCLINK_REASON_MISSING + : CBM_DOCLINK_REASON_EXTERNAL); +} + +/* The reason for a path that did not come to a type or a namespace. */ +static cs_res_t path_unfound(const cs_ctx_t *c, const cs_tp_t *tp) { + if (tp->st == CS_TP_RES) { + return tp->res; + } + if (tp->st == CS_TP_NSFAIL) { + return namespace_unfound(c, tp->ns); + } + return member_unfound(c, tp->ent, tp->next ? tp->next->name : NULL, + tp->next ? tp->next->arity : CS_ARITY_NONE); +} + +/* ── Members of a type ───────────────────────────────────────────── */ + +/* The implicit roots a type of this kind derives from without naming them, + * as far as the repository declares them. */ +static int implicit_roots(const cs_ctx_t *c, char kind, int *out) { + static const char *const enum_roots[] = {"Enum", "ValueType", "Object", NULL}; + static const char *const value_roots[] = {"ValueType", "Object", NULL}; + static const char *const object_root[] = {"Object", NULL}; + static const char system_ns[] = "System"; + const char *const *roots = object_root; + if (kind == 'e') { + roots = enum_roots; + } else if (kind == 's' || kind == 't') { + roots = value_roots; + } else if (kind == 'i' || kind == 'd') { + return 0; + } + int sys = ns_find(c->ix, 0, system_ns, sizeof(system_ns) - SKIP_ONE); + int n = 0; + for (int i = 0; sys >= 0 && roots[i]; i++) { + cs_found_t fd = {0}; + add_types(c, true, sys, roots[i], 0, &fd); + if (fd.n == SKIP_ONE) { + out[n++] = fd.first.id; + } + } + return n; +} + +/* NOT the compiler's rule (see CS_BIND_INHERITED): where the compiler's + * lookup has bound nothing, the member `seg` of the nearest supertype of + * `ent` that has one. false when the index does not bind inherited members, + * and when no supertype in view has the name. */ +static bool inherited_result(const cs_ctx_t *c, int ent, const cs_seg_t *seg, const cs_use_t *u, + char kind, cs_res_t *res) { + const cs_index_t *ix = c->ix; + if (!ix->bind_inherited) { + return false; + } + int sup[CS_MAX_SUPERS + CS_MAX_ROOTS]; + bool more = false; + int n = supers_of(ix, ent, sup, &more); + n += implicit_roots(c, ix->ents[ent].kind, sup + n); + cs_query_t q = {.name = seg->name, .arity = seg->arity, .kind = kind}; + for (int i = 0; i < n; i++) { + cs_work(SKIP_ONE); + cs_found_t fd = {0}; + level_entity(c, sup[i], &q, &fd); + if (fd.n > SKIP_ONE) { + *res = found_ambiguous(&fd); + return true; + } + if (fd.n == SKIP_ONE && fd.first.kind == 'T' && !u->has_params) { + *res = type_result(c, fd.first.id, u->exact); + return true; + } + cs_mb_t st = fd.n == SKIP_ONE && fd.first.kind == 'M' + ? members_result(c, sup[i], &q, u, res) + : CS_MB_NONE; + if (st == CS_MB_RES) { + return true; + } + if (st == CS_MB_INVISIBLE || fd.invisible) { + *res = res_unres(CBM_DOCLINK_REASON_TEST_ONLY); + return true; + } + } + if (more) { + *res = res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); /* a hierarchy past CS_MAX_SUPERS */ + return true; + } + return false; +} + +/* A property or an event by its accessor's name (get_X, set_X; add_X, + * remove_X). The compiler binds the accessor; the graph keeps an accessor in + * its property or event, so that is the node. An indexer's accessors + * (get_Item, set_Item) have none. */ +static bool accessor_result(const cs_ctx_t *c, int ent, const cs_seg_t *seg, bool exact, + cs_res_t *res) { + static const struct { + const char *prefix; + char kind; + } acc[] = {{"get_", 'p'}, {"set_", 'p'}, {"add_", 'e'}, {"remove_", 'e'}}; + for (size_t a = 0; seg->arity <= 0 && a < sizeof(acc) / sizeof(acc[0]); a++) { + size_t al = strlen(acc[a].prefix); + if (strncmp(seg->name, acc[a].prefix, al) != 0 || !seg->name[al]) { + continue; + } + cs_query_t q = {.name = seg->name + al, .arity = CS_ARITY_NONE, .kind = acc[a].kind}; + cs_mb_t st = members_plain(c, ent, &q, exact, res); + if (st == CS_MB_INVISIBLE) { + *res = res_unres(CBM_DOCLINK_REASON_TEST_ONLY); + } else if (st != CS_MB_RES && acc[a].kind == 'p' && strcmp(q.name, "Item") == 0 && + special_declared(c, ent, "this")) { + *res = res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); + } else if (st != CS_MB_RES) { + continue; + } + return true; + } + return false; +} + +/* `seg` as a member of the type `ent`, looked up in `ent` itself: a member + * it declares, a type nested in it (with a parameter list: that type's + * constructor), its own name (`Foo.Foo`: its constructor), a property or + * event by its accessor's name. `kind` is the member kind a doc ID names. */ +static cs_res_t member_in(const cs_ctx_t *c, int ent, const cs_seg_t *seg, const cs_use_t *u, + char kind) { + const cs_entity_t *e = &c->ix->ents[ent]; + cs_res_t res = res_unres(CBM_DOCLINK_REASON_MISSING); + cs_use_t ctor_use = *u; + ctor_use.where.member_seg = CS_NONE; + if (strcmp(seg->name, "#ctor") == 0) { + return ctor_result(c, ent, &ctor_use); + } + if (strcmp(seg->name, "#cctor") == 0) { + return static_ctor_result(c, ent, u->exact); + } + cs_query_t q = {.name = seg->name, .arity = seg->arity, .kind = kind}; + cs_found_t fd = {0}; + level_entity(c, ent, &q, &fd); + if (fd.n > SKIP_ONE) { + return found_ambiguous(&fd); + } + if (fd.n == SKIP_ONE && fd.first.kind == 'T') { + if (!u->has_params) { + return type_result(c, fd.first.id, u->exact_type && !fd.joined); + } + ctor_use.where.type_seg = u->where.member_seg; /* the nested type is the last segment */ + ctor_use.exact = u->exact_type && !fd.joined; + return ctor_result(c, fd.first.id, &ctor_use); + } + if (fd.n == SKIP_ONE) { + cs_mb_t st = members_result(c, ent, &q, u, &res); + if (st == CS_MB_RES) { + return res; + } + if (st == CS_MB_INVISIBLE) { + return res_unres(CBM_DOCLINK_REASON_TEST_ONLY); + } + /* it has the name, and nothing of it takes the written parameters */ + } else { + if (fd.invisible) { + return res_unres(CBM_DOCLINK_REASON_TEST_ONLY); + } + /* `Foo.Foo`: the constructor -- unless it is a bare `Foo{T}.Foo` + * written on that generic type itself, which the compiler leaves + * unbound */ + bool own_name = !kind && seg->arity <= 0 && strcmp(seg->name, e->name) == 0; + bool on_type = c->type >= 0 && c->f->types[c->type].entity == ent; + if (own_name && (u->has_params || e->arity == 0 || !on_type)) { + return ctor_result(c, ent, &ctor_use); + } + if (!kind && accessor_result(c, ent, seg, u->exact, &res)) { + return res; + } + if (u->r->maybe_indexer && (!kind || kind == 'p') && special_declared(c, ent, "this")) { + return res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); /* `Item(int)`: the indexer */ + } + } + if (stub_member(c, ent, &q, u, &res) || inherited_result(c, ent, seg, u, kind, &res)) { + return res; + } + return member_unfound(c, ent, seg->name, seg->arity); +} + +/* ── Simple names, qualified names, operators, doc IDs ───────────── */ + +/* A simple name no scope level has. */ +static cs_res_t simple_fallback(const cs_ctx_t *c, const cs_ref_t *r, const cs_use_t *u) { + const cs_file_t *f = c->f; + const cs_seg_t *seg = &r->segs[0]; + cs_res_t res = res_unres(CBM_DOCLINK_REASON_MISSING); + /* `Foo(int)` written in a generic `Foo`: no type `Foo` is in scope, + * and the compiler takes the constructor of the type the reference + * stands in */ + if (r->has_params && seg->arity <= 0 && c->type >= 0 && + strcmp(f->types[c->type].name, seg->name) == 0) { + cs_use_t ctor_use = *u; + ctor_use.where = (cs_where_t){.type_seg = CS_NONE, .member_seg = CS_NONE}; + return ctor_result(c, f->types[c->type].entity, &ctor_use); + } + cs_query_t q = {.name = seg->name, .arity = seg->arity}; + for (int t = c->type; t >= 0; t = f->types[t].outer) { + int ent = f->types[t].entity; + if (accessor_result(c, ent, seg, u->exact, &res)) { + return res; + } + if (r->maybe_indexer && special_declared(c, ent, "this")) { + return res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); /* `Item(int)`: the indexer */ + } + if (stub_member(c, ent, &q, u, &res) || inherited_result(c, ent, seg, u, 0, &res)) { + return res; + } + } + return simple_unfound(c); +} + +static cs_res_t resolve_simple(const cs_ctx_t *c, const cs_ref_t *r) { + const cs_seg_t *seg = &r->segs[0]; + cs_query_t q = {.name = seg->name, .arity = seg->arity}; + cs_found_t fd; + bool statics = false; + lookup(c, &q, &fd, &statics); + if (fd.n > SKIP_ONE) { + return found_ambiguous(&fd); + } + q.statics = statics; /* through a `using static`: its static members are meant */ + bool exact = fd.exact && !fd.joined; + cs_use_t u = {.r = r, + .has_params = r->has_params, + .where = {.type_seg = CS_NONE, .member_seg = 0}, + .exact = exact}; + if (fd.n == 0) { + return fd.invisible ? res_unres(CBM_DOCLINK_REASON_TEST_ONLY) : simple_fallback(c, r, &u); + } + cs_res_t res = res_unres(CBM_DOCLINK_REASON_MISSING); + switch (fd.first.kind) { + case 'L': + /* a type parameter of the definition: no reference to code elsewhere */ + return r->has_params ? res : (cs_res_t){.st = CS_LOCAL}; + case 'N': + return r->has_params ? res : namespace_result(); + case 'T': + if (!r->has_params) { + return type_result(c, fd.first.id, exact); + } + if (fd.exact) { + return res; /* the compiler matches no parameter list against an alias */ + } + /* a parameter list on a type's name: its constructor */ + u.where = (cs_where_t){.type_seg = 0, .member_seg = CS_NONE}; + return ctor_result(c, fd.first.id, &u); + case 'M': { + cs_mb_t st = members_result(c, fd.first.id, &q, &u, &res); + if (st == CS_MB_RES) { + return res; + } + if (st == CS_MB_INVISIBLE) { + return res_unres(CBM_DOCLINK_REASON_TEST_ONLY); + } + /* the level that has the name decides: nothing further out is asked */ + if (stub_member(c, fd.first.id, &q, &u, &res) || + inherited_result(c, fd.first.id, seg, &u, 0, &res)) { + return res; + } + return member_unfound(c, fd.first.id, seg->name, seg->arity); + } + default: + return res_unres(CBM_DOCLINK_REASON_EXTERNAL); + } +} + +/* `Path.Last`: the last segment is a member of what the path before it + * names. Of a type: member_in. Of a namespace: a type (with a parameter + * list: its constructor) or a namespace. */ +static cs_res_t resolve_qualified(const cs_ctx_t *c, const cs_ref_t *r, char kind, + bool has_params) { + int n = r->nsegs; + const cs_seg_t *last = &r->segs[n - SKIP_ONE]; + cs_tp_t tp = resolve_path(c, r->segs, n - SKIP_ONE, true); + cs_use_t u = {.r = r, + .has_params = has_params, + .where = {.type_seg = n - PAIR_LEN, .member_seg = n - SKIP_ONE}, + .exact = tp.exact, + .exact_type = tp.exact || tp.in_ns}; + if (tp.st == CS_TP_TYPE) { + return member_in(c, tp.ent, last, &u, kind); + } + if (tp.st != CS_TP_NS) { + return path_unfound(c, &tp); + } + cs_found_t fd = {0}; + if (!kind) { + add_types(c, true, tp.ns, last->name, type_arity(last->arity), &fd); + } + bool is_ns = + !kind && last->arity <= 0 && ns_in_view(c, tp.ns, last->name, strlen(last->name)) >= 0; + if (fd.n + (is_ns ? SKIP_ONE : 0) > SKIP_ONE) { + return found_ambiguous(&fd); + } + if (is_ns) { + return has_params ? res_unres(CBM_DOCLINK_REASON_MISSING) : namespace_result(); + } + if (fd.n == 0) { + return fd.invisible ? res_unres(CBM_DOCLINK_REASON_TEST_ONLY) : namespace_unfound(c, tp.ns); + } + u.exact_type = u.exact_type && !fd.joined; + if (!has_params) { + return type_result(c, fd.first.id, u.exact_type); + } + u.where = (cs_where_t){.type_seg = n - SKIP_ONE, .member_seg = CS_NONE}; + u.exact = u.exact_type; + return ctor_result(c, fd.first.id, &u); +} + +/* An operator, a conversion or an indexer: of the type the path before it + * names, or of the nearest enclosing type that declares one. None has a + * node: one that is declared is a graph gap. */ +static cs_res_t resolve_operator(const cs_ctx_t *c, const cs_ref_t *r) { + if (r->nsegs > 0) { + cs_tp_t tp = resolve_path(c, r->segs, r->nsegs, true); + if (tp.st == CS_TP_NS) { + return namespace_unfound(c, tp.ns); + } + if (tp.st != CS_TP_TYPE) { + return path_unfound(c, &tp); + } + return special_declared(c, tp.ent, r->op_name) + ? res_unres(CBM_DOCLINK_REASON_GRAPH_GAP) + : member_unfound(c, tp.ent, r->op_name, CS_ARITY_NONE); + } + for (int t = c->type; t >= 0; t = c->f->types[t].outer) { + if (special_declared(c, c->f->types[t].entity, r->op_name)) { + return res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); + } + } + return c->type >= 0 ? member_unfound(c, c->f->types[c->type].entity, r->op_name, CS_ARITY_NONE) + : res_unres(CBM_DOCLINK_REASON_MISSING); +} + +/* A doc ID names its target by its full name, from the global namespace: + * `T:` a type, `N:` a namespace, `M:` `P:` `F:` `E:` a member of the kind + * the letter says. An `M:` without parentheses is the overload without + * parameters; DocFX's `O:` is the whole group. */ +static cs_res_t resolve_docid(const cs_ctx_t *c, const cs_ref_t *r) { + cs_ctx_t abs = *c; /* what stands around the reference is no part of the name */ + abs.type = CS_NONE; + abs.member = CS_NONE; + if (r->docid == 'T' || r->docid == 'N') { + cs_tp_t tp = resolve_path(&abs, r->segs, r->nsegs, false); + if (tp.st == CS_TP_RES) { + return tp.res; + } + if (r->docid == 'N') { + if (tp.st == CS_TP_NS) { + return namespace_result(); + } + /* a type, or a namespace the repository does not have */ + return res_unres(tp.st == CS_TP_NSFAIL ? CBM_DOCLINK_REASON_EXTERNAL + : CBM_DOCLINK_REASON_MISSING); + } + if (tp.st == CS_TP_TYPE) { + return type_result(&abs, tp.ent, !tp.joined); + } + return tp.st == CS_TP_NS ? res_unres(CBM_DOCLINK_REASON_MISSING) : path_unfound(&abs, &tp); + } + if (r->nsegs < PAIR_LEN) { + return res_unres(CBM_DOCLINK_REASON_UNPARSEABLE); + } + switch (r->docid) { + case 'M': + return resolve_qualified(&abs, r, 'c', true); + case 'O': + return resolve_qualified(&abs, r, 'c', false); + case 'P': + return resolve_qualified(&abs, r, 'p', r->has_params); + case 'E': + return resolve_qualified(&abs, r, 'e', false); + default: + return resolve_qualified(&abs, r, 'v', false); + } +} + +static cs_res_t resolve_form(const cs_ctx_t *c, const cs_ref_t *r) { + if (r->op) { + return resolve_operator(c, r); + } + if (r->docid) { + return resolve_docid(c, r); + } + if (r->nsegs == SKIP_ONE && !c->glob) { + return resolve_simple(c, r); + } + return resolve_qualified(c, r, 0, r->has_params); +} + +static bool reason_is_nothing(const cs_res_t *res) { + return res->st == CS_UNRES && (res->reason == CBM_DOCLINK_REASON_MISSING || + res->reason == CBM_DOCLINK_REASON_EXTERNAL); +} + +/* A reference in this context. For product code a namespace that only test + * code declares does not exist. When the reference came to nothing and such + * a namespace was passed over on the way, what it names may be a test + * declaration: asked once more the way test code sees it, a name that is + * there is test_only_target; one that is not there either keeps the reason + * it has without that namespace. */ +static cs_res_t resolve_ref(const cs_ctx_t *c, const cs_ref_t *r) { + bool passed_over = false; + cs_ctx_t seen = *c; + seen.passed_over = &passed_over; + cs_res_t res = resolve_form(&seen, r); + if (!passed_over || !reason_is_nothing(&res)) { + return res; + } + cs_ctx_t as_test = *c; + as_test.prod = false; + as_test.passed_over = NULL; + cs_res_t there = resolve_form(&as_test, r); + bool nothing = reason_is_nothing(&there) || + (there.st == CS_UNRES && there.reason == CBM_DOCLINK_REASON_UNPARSEABLE); + return nothing ? res : res_unres(CBM_DOCLINK_REASON_TEST_ONLY); +} + +/* ── Context ─────────────────────────────────────────────────────── */ + +/* The innermost type of `f` around `line`, or CS_NONE: the last type that + * starts at or before the line, or the nearest of its outer types that + * reaches the line (the types are in document order). */ +static int type_at(const cs_file_t *f, uint32_t line) { + int lo = 0; + int hi = f->ntypes; + while (lo < hi) { + int mid = lo + ((hi - lo) / PAIR_LEN); + if (f->types[mid].start <= line) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + for (int t = lo - SKIP_ONE; t >= 0; t = f->types[t].outer) { + if (f->types[t].end >= line) { + return t; + } + } + return CS_NONE; +} + +/* The innermost namespace declaration around `line` (0: the file itself). */ +static int region_at(const cs_file_t *f, uint32_t line) { + int lo = SKIP_ONE; + int hi = f->nregions; + while (lo < hi) { + int mid = lo + ((hi - lo) / PAIR_LEN); + if (f->regions[mid].start <= line) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + for (int r = lo - SKIP_ONE; r > 0; r = f->regions[r].parent) { + if (f->regions[r].end >= line) { + return r; + } + } + return 0; +} + +/* The generic method declared at `line`, or CS_NONE: its type parameters are + * in scope in its documentation. Of the members that start at one line the + * generic methods stand first (finish_lookups), so the first one tells -- + * however many members a line holds. */ +static int generic_method_at(const cs_file_t *f, uint32_t line) { + int lo = 0; + int hi = f->nmembers; + while (lo < hi) { + int mid = lo + ((hi - lo) / PAIR_LEN); + if (f->members[f->members_by_start[mid]].start < line) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + if (lo < f->nmembers) { + cs_work(SKIP_ONE); + int mi = f->members_by_start[lo]; + const cs_member_t *m = &f->members[mi]; + if (m->start == line && m->kind == 'c' && m->arity > 0) { + return mi; + } + } + return CS_NONE; +} + +/* A scope of file `file`: the namespace declaration `region`, the type + * `type` and the ones around it. */ +static void ctx_at(cs_ctx_t *c, const cs_index_t *ix, int file, int region, int type) { + memset(c, 0, sizeof(*c)); + c->ix = ix; + c->file = file; + c->f = &ix->files[file]; + c->unit = c->f->unit; + c->group = c->unit >= 0 ? ix->units[c->unit].group : CS_NONE; + c->prod = !c->f->is_test; + c->region = region; + c->type = type; + c->member = CS_NONE; + c->skip_region = CS_NONE; +} + +/* The scope of the definition that starts at `line` (the documented type is + * itself the innermost one: its members and type parameters are in scope in + * its own documentation). A file's own doc stands outside every declaration. */ +static void ctx_init(cs_ctx_t *c, const cs_index_t *ix, int file, uint32_t line, bool file_doc) { + const cs_file_t *f = &ix->files[file]; + int type = file_doc ? CS_NONE : type_at(f, line); + int region = file_doc ? 0 : (type >= 0 ? f->types[type].region : region_at(f, line)); + ctx_at(c, ix, file, region, type); + c->member = (type >= 0 && f->nmembers > 0) ? generic_method_at(f, line) : CS_NONE; +} + +/* True when `line` is in a part of the file whose declarations could not be + * placed (the ranges are sorted and disjoint). */ +static bool line_unplaced(const cs_file_t *f, uint32_t line) { + int lo = 0; + int hi = f->nunplaced; + while (lo < hi) { + int mid = lo + ((hi - lo) / PAIR_LEN); + if (f->unplaced[mid].to < line) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + return lo < f->nunplaced && f->unplaced[lo].from <= line; +} + +/* ── Usings and base lists ───────────────────────────────────────── */ + +static const char CS_GLOBAL_PREFIX[] = "global::"; + +/* A written type or namespace name into `segs`; *glob when it starts at the + * global namespace. false for what is no dotted name (a tuple, an array, a + * pointer, a text this code did not understand). */ +static bool parse_name(const char *s, size_t n, cs_seg_t *segs, int *nsegs, bool *glob) { + size_t gl = sizeof(CS_GLOBAL_PREFIX) - SKIP_ONE; + *glob = n >= gl && strncmp(s, CS_GLOBAL_PREFIX, gl) == 0; + if (*glob) { + s += gl; + n -= gl; + } + return parse_path(s, n, segs, nsegs); +} + +/* What a using directive names: a namespace into u->ns, a type into u->ent + * (CS_AMBIGUOUS for several). Both stay CS_NONE for what the repository does + * not have. `c` is the scope the directive is resolved in. */ +static void resolve_using(cs_using_t *u, const cs_ctx_t *c) { + cs_seg_t segs[CS_MAX_SEGS]; + int n = 0; + cs_ctx_t at = *c; + bool glob = false; + u->ns = CS_NONE; + u->ent = CS_NONE; + if (!parse_name(u->target, strlen(u->target), segs, &n, &glob)) { + return; + } + at.glob = at.glob || glob; + cs_tp_t tp = resolve_path(&at, segs, n, false); + if (tp.st == CS_TP_NS && u->kind != 's') { + u->ns = tp.ns; + } else if (tp.st == CS_TP_TYPE && u->kind != 'n') { + u->ent = tp.ent; + u->joined = tp.joined; + } else if (tp.st == CS_TP_RES && res_is(&tp.res, CBM_DOCLINK_REASON_AMBIGUOUS) && + u->kind != 'n') { + u->ent = CS_AMBIGUOUS; + } +} + +/* True when the directive brings in names the repository does not declare: + * a namespace it has no declaration of (or a standard-library one, which it + * never has all of), a type it does not have. (An alias brings one name, and + * says so itself where it is used.) */ +static bool using_opens(const cs_index_t *ix, const cs_using_t *u) { + if (u->kind == 'n') { + return u->ns < 0 || !ix->nss[u->ns].declared || ix->nss[u->ns].standard; + } + return u->kind == 's' && u->ent == CS_NONE; +} + +/* What every using directive names, and which scopes are open. A directive + * at the top of a file, a `global using` and a project file's are + * resolved from the global namespace: nothing else is in scope there (the + * directives beside it are not). A directive inside a namespace declaration + * is resolved in that namespace, without that declaration's own directives. */ +static bool resolve_usings(cs_index_t *ix) { + for (int ui = 0; ui < ix->nunits; ui++) { + cs_unit_t *unit = &ix->units[ui]; + cs_ctx_t c = {.ix = ix, + .type = CS_NONE, + .member = CS_NONE, + .unit = ui, + .group = unit->group, + .glob = true, + .skip_region = CS_NONE}; + for (int i = 0; i < unit->nusings; i++) { + resolve_using(&unit->usings[i], &c); + unit->open = unit->open || using_opens(ix, &unit->usings[i]); + } + if (!build_using_index(ix, unit->usings, unit->nusings, &unit->using_index)) { + return false; + } + } + for (int fi = 0; fi < ix->nfiles; fi++) { + cs_file_t *f = &ix->files[fi]; + /* Regions are in ancestor order. Publish an outer region's index + * before resolving its children's targets; a region's own imports + * remain excluded by skip_region during their resolution. */ + for (int r = 0; r < f->nregions; r++) { + cs_region_t *reg = &f->regions[r]; + for (int i = reg->u_lo; i < reg->u_hi; i++) { + cs_ctx_t c; + ctx_at(&c, ix, fi, r, CS_NONE); + c.prod = false; /* a directive names whatever the compiler bound */ + c.glob = r == 0; + c.skip_region = r; + resolve_using(&f->usings[i], &c); + } + const cs_using_t *us = reg->u_hi > reg->u_lo ? f->usings + reg->u_lo : NULL; + if (!build_using_index(ix, us, reg->u_hi - reg->u_lo, ®->using_index)) { + return false; + } + } + for (int r = 0; r < f->nregions; r++) { + cs_region_t *reg = &f->regions[r]; + reg->open = r > 0 && f->regions[reg->parent].open; + for (int i = reg->u_lo; !reg->open && i < reg->u_hi; i++) { + reg->open = using_opens(ix, &f->usings[i]); + } + } + } + return true; +} + +/* The entity a written base type names in the scope of its declaration, or + * CS_NONE. */ +static int base_entity(const cs_ctx_t *c, const char *s, size_t n) { + cs_seg_t segs[CS_MAX_SEGS]; + int nsegs = 0; + cs_ctx_t at = *c; + bool glob = false; + if (!parse_name(s, n, segs, &nsegs, &glob)) { + return CS_NONE; + } + at.glob = glob; + cs_tp_t tp = resolve_path(&at, segs, nsegs, false); + return tp.st == CS_TP_TYPE ? tp.ent : CS_NONE; +} + +/* The base types of every entity, from the base lists of all its + * declarations. A base that names nothing of the repository leaves the + * hierarchy open. false when memory ran out. */ +static bool resolve_bases(cs_index_t *ix) { + /* listed[b] == e + 1: b is in e's base list already (a partial type's + * declarations may each write the same base) */ + int *listed = + (int *)cbm_calloc(CBM_MEM_CLASS_OTHER, ((size_t)ix->nents + SKIP_ONE) * sizeof(int)); + bool ok = listed != NULL; + for (int ei = 0; ok && ei < ix->nents; ei++) { + cs_entity_t *e = &ix->ents[ei]; + int written = 0; + for (int d = 0; d < e->ndecls; d++) { + written += count_list(ix->files[e->decls[d].file].types[e->decls[d].type].bases, '|'); + } + if (written == 0) { + continue; + } + e->bases = (int *)ix_alloc(ix, (size_t)written * sizeof(int)); + ok = e->bases != NULL; + for (int d = 0; ok && d < e->ndecls; d++) { + const cs_type_t *t = &ix->files[e->decls[d].file].types[e->decls[d].type]; + cs_ctx_t c; + /* a base list is written outside the type it belongs to */ + ctx_at(&c, ix, e->decls[d].file, t->region, t->outer); + c.prod = false; /* a declared base is whatever the compiler bound */ + for (const char *p = t->bases; p && *p;) { + const char *bar = strchr(p, '|'); + size_t n = bar ? (size_t)(bar - p) : strlen(p); + int base = base_entity(&c, p, n); + if (base < 0) { + e->open = true; /* the hierarchy goes on outside the repository */ + } else if (base != ei && listed[base] != ei + SKIP_ONE) { + listed[base] = ei + SKIP_ONE; + e->bases[e->nbases++] = base; + } + p = bar ? bar + SKIP_ONE : NULL; + } + } + } + cbm_free(CBM_MEM_CLASS_OTHER, listed); + ix->oom = ix->oom || !ok; + return ok; +} + +/* How many entities `e` takes its openness from: its bases, and the shared + * trees' parts that belong to it (what they derive from, it derives from). */ +static int upper_count(const cs_entity_t *e) { + return e->nbases + (e->twin >= 0 ? SKIP_ONE : 0); +} + +static int upper_at(const cs_entity_t *e, int k) { + return k < e->nbases ? e->bases[k] : e->twin; +} + +/* open_any: a type whose own base, or a base of one of its supertypes, is + * outside the repository. Spread from the open types to everything derived + * from them, breadth first over the reversed base lists (a base list that + * goes round in a circle ends at the types already marked). false when + * memory ran out. */ +static bool spread_open(cs_index_t *ix) { + size_t n = (size_t)ix->nents; + int *first = (int *)cbm_calloc(CBM_MEM_CLASS_OTHER, (n + PAIR_LEN) * sizeof(int)); + int *queue = (int *)cbm_alloc(CBM_MEM_CLASS_OTHER, (n + SKIP_ONE) * sizeof(int)); + size_t edges = 0; + for (int i = 0; i < ix->nents; i++) { + edges += (size_t)upper_count(&ix->ents[i]); + } + int *derived = (int *)cbm_alloc(CBM_MEM_CLASS_OTHER, (edges + SKIP_ONE) * sizeof(int)); + bool ok = first && queue && derived; + if (ok) { + int qn = 0; + /* first[b + 2] counts the types derived from b; after the running + * sum first[b + 1] is where b's list starts, and filling the lists + * moves it to where the list ends: first[b] .. first[b + 1] */ + for (int i = 0; i < ix->nents; i++) { + for (int b = 0; b < upper_count(&ix->ents[i]); b++) { + first[upper_at(&ix->ents[i], b) + PAIR_LEN]++; + } + } + for (size_t i = PAIR_LEN; i < n + PAIR_LEN; i++) { + first[i] += first[i - SKIP_ONE]; + } + for (int i = 0; i < ix->nents; i++) { + for (int b = 0; b < upper_count(&ix->ents[i]); b++) { + derived[first[upper_at(&ix->ents[i], b) + SKIP_ONE]++] = i; + } + if (ix->ents[i].open) { + ix->ents[i].open_any = true; + queue[qn++] = i; + } + } + for (int head = 0; head < qn; head++) { + int b = queue[head]; + for (int k = first[b]; k < first[b + SKIP_ONE]; k++) { + if (!ix->ents[derived[k]].open_any) { + ix->ents[derived[k]].open_any = true; + queue[qn++] = derived[k]; + } + } + } + } + cbm_free(CBM_MEM_CLASS_OTHER, first); + cbm_free(CBM_MEM_CLASS_OTHER, queue); + cbm_free(CBM_MEM_CLASS_OTHER, derived); + ix->oom = ix->oom || !ok; + return ok; +} + +/* ── Index lifecycle and the resolver hooks ──────────────────────── */ + +/* What made the run's ambiguous references ambiguous, by kind: the rows all + * say `ambiguous`, and only the scope's rules and the limits are the same in + * every repository. Nothing is logged for a run without one. */ +static void log_ambiguous(const cs_stats_t *st) { + char b[CS_WHY_COUNT][CBM_SZ_32]; + uint64_t total = 0; + for (int i = 0; i < CS_WHY_COUNT; i++) { + uint64_t n = atomic_load(&st->ambiguous[i]); + total += n; + snprintf(b[i], sizeof(b[i]), "%llu", (unsigned long long)n); + } + if (total > 0) { + cbm_log_info("doc_links.cs.ambiguous", "scope_rules", b[CS_WHY_SCOPE], "shared_trees", + b[CS_WHY_SHARED], "assemblies", b[CS_WHY_ASSEMBLIES], "flavours", + b[CS_WHY_FLAVOURS], "unseen_parts", b[CS_WHY_PARTS], "limits", + b[CS_WHY_LIMIT]); + } +} + +static void cs_destroy(void *index) { + cs_index_t *ix = (cs_index_t *)index; + if (!ix) { + return; + } + if (ix->stats) { + log_ambiguous(ix->stats); + } + cbm_free(CBM_MEM_CLASS_OTHER, ix->stats); + cbm_ht_free(ix->project_above); + cbm_ht_free(ix->ns_by_key); + cbm_ht_free(ix->ent_by_key); + cbm_ht_free(ix->type_names); + cbm_ht_free(ix->quarantine); + cbm_ht_free(ix->quarantine_test); + cbm_ht_free(ix->unit_by_dir); + cbm_ht_free(ix->project_dirs); + cbm_ht_free(ix->group_by_stem); + cbm_ht_free(ix->specials); + cbm_msb_free(ix->msb); + cbm_free(CBM_MEM_CLASS_OTHER, ix->ents); + cbm_free(CBM_MEM_CLASS_OTHER, ix->run_to_file); + cbm_arena_destroy(&ix->arena); + cbm_free(CBM_MEM_CLASS_OTHER, ix); +} + +/* Log why the index could not be built, and drop it. */ +static void *build_failed(cs_index_t *ix, const char *step, const char *reason, const char *path) { + cbm_log_error("doc_links.cs.error", "step", step, "reason", reason, "file", path ? path : ""); + cs_destroy(ix); + return NULL; +} + +/* The index's tables and the project files. false when memory ran out. */ +static bool build_tables(cs_index_t *ix, const cbm_doclink_build_in_t *in) { + cbm_arena_init(&ix->arena); + ix->project = in->ctx->project_name; + ix->bind_inherited = CS_BIND_INHERITED; + ix->ns_by_key = cbm_ht_create(CBM_SZ_1K); + ix->ent_by_key = cbm_ht_create(CBM_SZ_4K); + ix->type_names = cbm_ht_create(CBM_SZ_4K); + ix->quarantine = cbm_ht_create(CBM_SZ_64); + ix->quarantine_test = cbm_ht_create(CBM_SZ_64); + ix->unit_by_dir = cbm_ht_create(CBM_SZ_256); + ix->project_dirs = cbm_ht_create(CBM_SZ_256); + ix->project_above = cbm_ht_create(CBM_SZ_256); + ix->group_by_stem = cbm_ht_create(CBM_SZ_256); + ix->specials = cbm_ht_create(CBM_SZ_256); + ix->stats = (cs_stats_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(cs_stats_t)); + ix->msb = cbm_msb_new(); + ix->nfiles = in->file_count; + ix->files = (cs_file_t *)ix_zalloc(ix, (size_t)in->file_count * sizeof(cs_file_t)); + ix->run_count = in->run_file_count; + ix->run_to_file = (int *)cbm_alloc(CBM_MEM_CLASS_OTHER, + ((size_t)in->run_file_count + SKIP_ONE) * sizeof(int)); + if (!ix->ns_by_key || !ix->ent_by_key || !ix->type_names || !ix->quarantine || + !ix->quarantine_test || !ix->unit_by_dir || !ix->project_dirs || !ix->project_above || + !ix->group_by_stem || !ix->specials || !ix->stats || !ix->msb || !ix->files || + !ix->run_to_file) { + return false; + } + for (int i = 0; i < in->run_file_count; i++) { + ix->run_to_file[i] = CS_NONE; + } + /* namespace 0: the global one */ + ix->nss = (cs_ns_t *)ix_zalloc(ix, CBM_SZ_256 * sizeof(cs_ns_t)); + if (!ix->nss) { + return false; + } + ix->nscap = CBM_SZ_256; + ix->nnss = SKIP_ONE; + ix->nss[0] = (cs_ns_t){.parent = CS_NONE, .name = ""}; + return collect_projects(ix, in); +} + +/* Every file of the index with what its scope declares. Returns the path of + * a file whose scope is not one this code wrote (the index is not built over + * such a scope: nothing may be resolved around a declaration that is not + * known), "" when memory ran out, NULL when all is well. */ +static const char *build_files(cs_index_t *ix, const cbm_doclink_build_in_t *in) { + CBMHashTable *dir_unit = cbm_ht_create(CBM_SZ_1K); + const char *bad = dir_unit ? NULL : ""; + for (int i = 0; !bad && i < in->file_count; i++) { + const cbm_doclink_file_t *src = &in->files[i]; + cs_file_t *f = &ix->files[i]; + f->rel_path = src->rel_path; + f->module_qn = + cbm_fqn_module_source_lang(&ix->arena, ix->project, src->rel_path, CBM_LANG_CSHARP); + f->is_test = cs_is_test_path(src->rel_path); + /* a project file declares nothing: its blob went to the evaluator */ + const char *scope = cbm_msb_is_project_scope(src->scope) ? NULL : src->scope; + if (scope && !parse_scope(ix, f, scope)) { + bad = ix->oom ? "" : src->rel_path; + break; + } + if (!scope) { + /* no scope: an empty file (only its own region) */ + f->regions = (cs_region_t *)ix_zalloc(ix, sizeof(cs_region_t)); + f->nregions = SKIP_ONE; + if (f->regions) { + f->regions[0].parent = CS_NONE; + } + } + f->unit = unit_of(ix, dir_unit, src->rel_path); + f->is_ref = f->unit >= 0 && ix->units[f->unit].is_ref; + if (src->run_file >= 0 && src->run_file < in->run_file_count) { + ix->run_to_file[src->run_file] = i; + } + if (!f->module_qn || ix->oom) { + bad = ""; + } + } + cbm_ht_free(dir_unit); + /* which namespaces are the standard library's: a namespace is made after + * the one above it, so one pass in order sees every parent first */ + for (int i = SKIP_ONE; i < ix->nnss; i++) { + cs_ns_t *n = &ix->nss[i]; + n->standard = + n->parent == 0 ? strcmp(n->name, CS_STANDARD_ROOT) == 0 : ix->nss[n->parent].standard; + } + return bad; +} + +/* The graph node of every declaration. false when memory ran out. */ +static bool build_nodes(cs_index_t *ix, const cbm_gbuf_t *g) { + cs_node_pass_t np = {.names = cbm_ht_create(CBM_SZ_256)}; + cbm_arena_init(&np.keys); + bool ok = np.names != NULL; + for (int i = 0; ok && i < ix->nfiles; i++) { + ok = bind_nodes(ix, &ix->files[i], g, &np); + } + cbm_ht_free(np.names); + cbm_arena_destroy(&np.keys); + cbm_free(CBM_MEM_CLASS_OTHER, np.last); + return ok; +} + +static void log_index(const cs_index_t *ix, const cs_msb_totals_t *msb) { + int incomplete = 0; + int open_units = 0; + for (int i = 0; i < ix->nents; i++) { + incomplete += ix->ents[i].incomplete; + } + for (int i = 0; i < ix->nunits; i++) { + open_units += ix->units[i].open; + } + char b[CBM_SZ_7][CBM_SZ_32]; + snprintf(b[0], sizeof(b[0]), "%d", ix->nfiles); + snprintf(b[1], sizeof(b[1]), "%d", ix->nents); + snprintf(b[2], sizeof(b[2]), "%d", ix->nunits - ix->nshared); + snprintf(b[3], sizeof(b[3]), "%d", incomplete); + snprintf(b[4], sizeof(b[4]), "%u", + (unsigned)(cbm_ht_count(ix->quarantine) + cbm_ht_count(ix->quarantine_test))); + snprintf(b[5], sizeof(b[5]), "%d", ix->ngroups); + snprintf(b[6], sizeof(b[6]), "%d", ix->nshared); + cbm_log_info("doc_links.cs.index", "files", b[0], "types", b[1], "projects", b[2], "assemblies", + b[5], "shared_trees", b[6], "incomplete_types", b[3], "quarantined_names", b[4]); + /* what the MSBuild project files could not tell: conditions, values and + * constructs that were not evaluated, and imports of files the index + * does not hold */ + snprintf(b[0], sizeof(b[0]), "%d", ix->nprojects); + snprintf(b[1], sizeof(b[1]), "%d", msb->unevaluable); + snprintf(b[2], sizeof(b[2]), "%d", msb->outside); + snprintf(b[3], sizeof(b[3]), "%d", open_units); + cbm_log_info("doc_links.cs.msbuild", "project_files", b[0], "unevaluable", b[1], + "imports_outside", b[2], "open_projects", b[3]); +} + +static void *cs_build(const cbm_doclink_build_in_t *in) { + cs_index_t *ix = (cs_index_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*ix)); + if (!ix) { + return NULL; + } + if (!build_tables(ix, in)) { + return build_failed(ix, "tables", "alloc", NULL); + } + const char *bad = build_files(ix, in); + if (bad) { + return build_failed(ix, "scopes", bad[0] ? "bad_scope" : "alloc", bad); + } + cs_msb_totals_t msb = {0}; + if (!units_collect_usings(ix, &msb)) { + return build_failed(ix, "projects", "alloc", NULL); + } + if (!build_entities(ix) || !entity_units(ix) || !entity_fulls(ix) || !build_named(ix)) { + return build_failed(ix, "entities", "alloc", NULL); + } + entity_twins(ix); + if (!build_nodes(ix, in->graph) || !build_binds(ix) || !build_mrefs(ix) || !build_extras(ix)) { + return build_failed(ix, "nodes", "alloc", NULL); + } + if (!build_name_scopes(ix) || !resolve_usings(ix)) { + return build_failed(ix, "usings", "alloc", NULL); + } + if (!resolve_bases(ix) || !spread_open(ix) || ix->oom) { + return build_failed(ix, "bases", "alloc", NULL); + } + log_index(ix, &msb); + return ix; +} + +/* The reason for a reference this code does not understand: one that names a + * keyword type in a form that is no name (`int[]`, `string?`) is the BCL's; + * one too long to be read is never judged by its beginning. */ +static int unparsed_reason(const char *raw) { + const char *s = raw ? raw : ""; + while (isspace((unsigned char)*s)) { + s++; + } + char head[CBM_SZ_32]; + size_t n = strcspn(s, "({<.[?* \t"); + if (strlen(s) >= CS_REF_BUF || n == 0 || n >= sizeof(head)) { + return CBM_DOCLINK_REASON_UNPARSEABLE; + } + memcpy(head, s, n); + head[n] = '\0'; + return keyword_type(head) ? CBM_DOCLINK_REASON_EXTERNAL : CBM_DOCLINK_REASON_UNPARSEABLE; +} + +/* A keyword alias (`int`, `string.Empty`) names its System type whatever + * stands around the reference. */ +static void rewrite_keyword(cs_ref_t *r) { + static const char system_ns[] = "System"; + if (r->docid || r->nsegs == 0 || r->nsegs >= CS_MAX_SEGS || r->segs[0].arity > 0) { + return; + } + const char *bcl = keyword_type(r->segs[0].name); + if (!bcl) { + return; + } + memmove(&r->segs[1], &r->segs[0], (size_t)r->nsegs * sizeof(cs_seg_t)); + r->segs[0] = (cs_seg_t){.arity = CS_ARITY_NONE}; + snprintf(r->segs[0].name, sizeof(r->segs[0].name), "%s", system_ns); + snprintf(r->segs[1].name, sizeof(r->segs[1].name), "%s", bcl); + r->nsegs++; + r->keyword = true; + r->glob = true; +} + +/* True when the reference names a type some file declares at a place no + * scope could be established for: it could be that declaration. */ +static bool names_quarantined(const cs_index_t *ix, const cs_file_t *f, const cs_ref_t *r) { + for (int i = 0; i < r->nsegs; i++) { + if (cbm_ht_get(ix->quarantine, r->segs[i].name) || + (f->is_test && cbm_ht_get(ix->quarantine_test, r->segs[i].name))) { + return true; + } + } + return false; +} + +static void cs_resolve(const void *index, int run_file, const CBMDocLink *link, + const cbm_gbuf_t *graph, cbm_doclink_outcome_t *out) { + const cs_index_t *ix = (const cs_index_t *)index; + (void)graph; /* every node was looked up when the index was built */ + out->kind = CBM_DOCLINK_UNRESOLVED; + out->reason = CBM_DOCLINK_REASON_MISSING; + out->target = NULL; + out->exact = false; + if (!ix || run_file < 0 || run_file >= ix->run_count || ix->run_to_file[run_file] < 0) { + return; + } + cs_ref_t r; + if (!parse_cref(link->raw, &r)) { + out->reason = unparsed_reason(link->raw); + return; + } + rewrite_keyword(&r); + /* What could not be placed is not resolved around: a definition in the + * part of its file where the braces stop pairing has no known scope, and + * a name some file declares without a known namespace could be that + * declaration. Both are declared-but-unplaced, i.e. graph gaps. */ + int file = ix->run_to_file[run_file]; + const cs_file_t *f = &ix->files[file]; + bool file_doc = (link->flags & CBM_DOCLINK_FLAG_FILE) != 0; + if ((!file_doc && line_unplaced(f, link->def_line)) || names_quarantined(ix, f, &r)) { + out->reason = CBM_DOCLINK_REASON_GRAPH_GAP; + return; + } + cs_ctx_t c; + ctx_init(&c, ix, file, link->def_line, file_doc); + c.glob = r.glob; + cs_res_t res = resolve_ref(&c, &r); + if (res.st == CS_LOCAL) { + out->kind = CBM_DOCLINK_LOCAL; + } else if (res.st == CS_OK && res.node) { + out->kind = CBM_DOCLINK_EDGE; + out->target = res.node; + out->exact = res.exact; + } else if (r.keyword && res.reason == CBM_DOCLINK_REASON_MISSING) { + out->reason = CBM_DOCLINK_REASON_EXTERNAL; /* the repository does not declare it */ + } else { + out->reason = res.reason; + if (res.reason == CBM_DOCLINK_REASON_AMBIGUOUS && res.why < CS_WHY_COUNT) { + atomic_fetch_add_explicit(&ix->stats->ambiguous[res.why], 1, memory_order_relaxed); + } + } +} + +/* ── Incremental scope rules ─────────────────────────────────────── */ + +/* Scope line fields (0-based, the tag is field 0; internal/cbm/doclink_cs.c): + * `U region kind alias target`, `T region start end kind outer name tparams + * bases` and `M start kind explicit type name tparams sig`. */ +enum { CS_SCOPE_U_KIND = 2, CS_SCOPE_M_NAME = 5, CS_SCOPE_T_BASES = 8 }; + +static const char *delta_field(const char *line, size_t len, int idx, size_t *flen) { + int f = 0; + size_t s = 0; + for (size_t i = 0; i <= len; i++) { + if (i == len || line[i] == '\t') { + if (f == idx) { + *flen = i - s; + return line + s; + } + f++; + s = i + SKIP_ONE; + } + } + *flen = 0; + return NULL; +} + +/* A using directive only its own file sees: one that is not `global`. */ +static bool delta_local_using(const char *line, size_t len) { + if (line[0] != 'U') { + return false; + } + size_t klen = 0; + const char *kind = delta_field(line, len, CS_SCOPE_U_KIND, &klen); + return kind && klen > 0 && !memchr(kind, 'g', klen); +} + +/* The next scope line -- with `local_usings` only the file's own using + * directives, without it every other line; false at the end. */ +static bool delta_next_line(const char **cursor, bool local_usings, const char **line, + size_t *len) { + for (const char *p = *cursor; p && *p;) { + const char *nl = strchr(p, '\n'); + size_t n = nl ? (size_t)(nl - p) : strlen(p); + const char *next = nl ? nl + SKIP_ONE : p + n; + if (n > 0 && delta_local_using(p, n) == local_usings) { + *line = p; + *len = n; + *cursor = next; + return true; + } + p = next; + } + *cursor = NULL; + return false; +} + +/* True when the two scopes have the same own using directives, in order. */ +static bool delta_same_usings(const char *stored, const char *fresh) { + const char *sl = NULL; + const char *fl = NULL; + size_t slen = 0; + size_t flen = 0; + for (;;) { + bool hs = delta_next_line(&stored, true, &sl, &slen); + bool hf = delta_next_line(&fresh, true, &fl, &flen); + if (!hs || !hf) { + return hs == hf; + } + if (slen != flen || memcmp(sl, fl, slen) != 0) { + return false; + } + } +} + +/* True when the scope declares a type that has a base list. */ +static bool delta_has_bases(const char *scope) { + for (const char *p = scope; p && *p;) { + const char *nl = strchr(p, '\n'); + size_t n = nl ? (size_t)(nl - p) : strlen(p); + size_t blen = 0; + if (p[0] == 'T' && delta_field(p, n, CS_SCOPE_T_BASES, &blen) && blen > 0) { + return true; + } + p = nl ? nl + SKIP_ONE : NULL; + } + return false; +} + +/* Compare a changed file's stored and fresh scopes line by line, in order. + * Two differences leave every other file's resolution alone: + * - the file's own using directives (namespace, static, alias), as long as + * the file declares no type with a base list: they scope the file + * itself, and it is re-extracted anyway. A base list is resolved through + * them, and what a type derives from decides how OTHER files' references + * to its members come out; + * - a member the fresh scope no longer has. Every member line is a method, + * constructor, property, field, event, operator, indexer or enum member, + * so a reference that depended on it either bound it (an edge into this + * file) or names it in its unresolved row: the name is reported. + * Everything else is GLOBAL: a new or changed line (a type, a member, a + * signature, a namespace, a global using), a removed type (it may be another + * type's base or an alias target, which changes how references THROUGH those + * classify), a removed namespace, global using, quarantined name or unplaced + * range, and a changed order (same-path declarations own their node by + * order, and a record names its type by its position). + * + * An MSBuild project file's blob sets the global usings of every C# file of + * its project: any difference between two of those is GLOBAL. An edit that + * leaves the blob as it is -- a target, a package reference -- changes + * nobody's scope. */ +static int cs_scope_delta(const char *stored, const char *fresh, cbm_doclink_name_fn removed, + void *ud) { + if (cbm_msb_is_project_scope(stored) || cbm_msb_is_project_scope(fresh)) { + return strcmp(stored, fresh) == 0 ? CBM_DOCLINK_DELTA_LOCAL : CBM_DOCLINK_DELTA_GLOBAL; + } + if (!delta_same_usings(stored, fresh) && (delta_has_bases(stored) || delta_has_bases(fresh))) { + return CBM_DOCLINK_DELTA_GLOBAL; + } + const char *sp = stored; + const char *fp = fresh; + const char *sl = NULL; + const char *fl = NULL; + size_t slen = 0; + size_t flen = 0; + bool hs = delta_next_line(&sp, false, &sl, &slen); + bool hf = delta_next_line(&fp, false, &fl, &flen); + while (hs || hf) { + if (hs && hf && slen == flen && memcmp(sl, fl, slen) == 0) { + hs = delta_next_line(&sp, false, &sl, &slen); + hf = delta_next_line(&fp, false, &fl, &flen); + continue; + } + if (!hs || sl[0] != 'M') { + return CBM_DOCLINK_DELTA_GLOBAL; + } + size_t nlen = 0; + const char *name = delta_field(sl, slen, CS_SCOPE_M_NAME, &nlen); + if (!name || nlen == 0 || !removed || !removed(ud, name, nlen)) { + return CBM_NOT_FOUND; /* not a member line this code wrote, or not recordable */ + } + hs = delta_next_line(&sp, false, &sl, &slen); + } + return CBM_DOCLINK_DELTA_LOCAL; +} + +/* XML is listed for the MSBuild project files: their scope blobs carry the C# + * tag, and the index needs the ones of this run as it needs the stored ones. + * An XML file that is no project file has no blob and no references. */ +const cbm_doclink_resolver_t cbm_doclink_cs_resolver = { + .langs = {CBM_LANG_CSHARP, CBM_LANG_XML}, + .lang_count = 2, + .scope_tag = CBM_DOCLINK_CS_SCOPE_TAG, + .build = cs_build, + .destroy = cs_destroy, + .resolve = cs_resolve, + .scope_delta = cs_scope_delta, +}; diff --git a/src/pipeline/doc_links_msbuild.c b/src/pipeline/doc_links_msbuild.c new file mode 100644 index 0000000000..b665a9f47c --- /dev/null +++ b/src/pipeline/doc_links_msbuild.c @@ -0,0 +1,3863 @@ +/* + * doc_links_msbuild.c — MSBuild global usings of a C# project (R1). + * + * A reader of data, never a build: nothing is executed, fetched or opened. + * The input is the scope blob the extractor writes for every MSBuild project + * file the index holds (internal/cbm/doclink_cs.c, "Project blob"). A file + * discovery did not take -- a symbolic link, a path outside the repository, + * an ignored file -- does not exist here. + * + * Evaluation model (MSBuild's passes, reduced): + * files the nearest Directory.Build.props above the project, the + * project file, the nearest Directory.Build.targets; an + * is evaluated where it stands, every file once + * pass 1 properties in evaluation order, later wins + * pass 2 items with the final properties; + * Static="true" names a type, Alias an alias; a Remove takes a + * target out wherever it was included + * implicit ImplicitUsings enable/true adds the default set of the + * project's SDK (Microsoft.NET.Sdk / .Web / .Worker) + * + * Nothing is guessed. A condition has three values: true, false, and UNKNOWN + * for what this reader cannot decide -- a property no evaluated file sets + * (the SDK, the environment or the command line may set it), a property + * function, an item or metadata reference, a relational operator, a function + * call, a comparison whose outcome depends on how MSBuild converts its + * operands. An element under an unknown condition is not applied and is + * counted. What it could have set is unknown from there on: its properties + * poison the conditions that read them, and a or an ImplicitUsings + * switch that cannot be evaluated leaves the project's usings open + * (cbm_msb_result_t.open). + * + * Paths: $(MSBuildThisFileDirectory) and $(MSBuildProjectDirectory) are + * absolute in MSBuild. Here they start with a mark byte followed by the + * repository-relative path, so an import through them is not joined to the + * importing file's directory a second time, and a path the file writes + * absolute itself (/usr/..., C:\...) stays what it is: outside. + */ +#include "pipeline/doc_links_msbuild.h" + +#include "doclink.h" /* CBM_DOCLINK_CS_SCOPE_TAG */ +#include "foundation/arena.h" +#include "foundation/constants.h" +#include "foundation/hash_table.h" +#include "foundation/mem_core.h" + +#include +#include +#include +#include +#include + +enum { + MSB_FIELDS = 5, + MSB_NAME_MAX = 256, /* a property name; a longer one never has a known value */ + MSB_VALUE_MAX = 4096, /* an expanded value; a longer one is unknown */ + MSB_COND_DEPTH = 32, /* parentheses of a condition; deeper is unknown */ + MSB_SIGNIFICANT = 15, /* the decimal digits a double holds exactly */ + MSB_PATH_MARK = 1, /* first byte of a repository-absolute path */ + MSB_INIT = 16, +}; + +typedef enum { MSB_FALSE = 0, MSB_TRUE = 1, MSB_UNKNOWN = 2 } msb_tri_t; + +/* ── The project files ───────────────────────────────────────────── */ + +typedef struct { + char tag; /* I G V W H N K Y C; W is a V whose value is no plain text */ + const char *f[MSB_FIELDS]; /* NULL: the attribute is not there */ +} msb_rec_t; + +typedef struct { + const char *rel_path; + const char *dir; /* "" for the repository root */ + const char *name; /* the file name */ + const char *stem; /* ... without its extension */ + const char *ext; /* ".csproj"; "" when it has none */ + const char *abs_dir; /* marked, with a trailing '/' */ + const char *abs_path; /* marked */ + const char *sdk; /* the attribute, or NULL */ + bool readable; + msb_rec_t *recs; + int nrecs; +} msb_file_t; + +struct cbm_msb { + CBMArena arena; + msb_file_t *files; + int nfiles; + int cap; + CBMHashTable *by_path; /* rel_path -> index + 1 */ + uint64_t generation; /* includes attempted additions that may partially mutate storage */ + bool oom; +}; + +bool cbm_msb_is_project_scope(const char *scope) { + static const char head[] = CBM_DOCLINK_CS_SCOPE_TAG "\nP\t"; + return scope && strncmp(scope, head, sizeof(head) - SKIP_ONE) == 0; +} + +cbm_msb_t *cbm_msb_new(void) { + cbm_msb_t *m = (cbm_msb_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*m)); + if (!m) { + return NULL; + } + cbm_arena_init(&m->arena); + m->by_path = cbm_ht_create(CBM_SZ_256); + if (!m->by_path) { + cbm_msb_free(m); + return NULL; + } + return m; +} + +void cbm_msb_free(cbm_msb_t *m) { + if (!m) { + return; + } + cbm_ht_free(m->by_path); + cbm_arena_destroy(&m->arena); + cbm_free(CBM_MEM_CLASS_OTHER, m); +} + +static void *msb_alloc(cbm_msb_t *m, size_t n) { + void *p = cbm_arena_alloc(&m->arena, n ? n : SKIP_ONE); + m->oom = m->oom || !p; + return p; +} + +/* a + b + c, in the arena. */ +static char *msb_join3(cbm_msb_t *m, const char *a, const char *b, const char *c) { + size_t al = strlen(a); + size_t bl = strlen(b); + size_t cl = strlen(c); + char *out = (char *)msb_alloc(m, al + bl + cl + SKIP_ONE); + if (out) { + memcpy(out, a, al); + memcpy(out + al, b, bl); + memcpy(out + al + bl, c, cl + SKIP_ONE); + } + return out; +} + +/* A blob field in place: NULL for an absent attribute, else the text after + * the '=' with its escapes resolved. */ +static char *msb_field(char *raw) { + if (!raw || raw[0] != '=') { + return NULL; + } + char *out = raw + SKIP_ONE; + char *w = out; + for (const char *r = out; *r; r++) { + if (*r == '\\' && r[1]) { + r++; + *w++ = *r == 't' ? '\t' : (*r == 'n' ? '\n' : (*r == 'r' ? '\r' : *r)); + } else { + *w++ = *r; + } + } + *w = '\0'; + return out; +} + +/* Cut a line (NUL-terminated, its tag in line[0]) into its fields in place. */ +static void msb_split(char *line, char *raw[MSB_FIELDS]) { + int n = 0; + for (int i = 0; i < MSB_FIELDS; i++) { + raw[i] = NULL; + } + for (char *p = line; *p && n < MSB_FIELDS; p++) { + if (*p == '\t') { + *p = '\0'; + raw[n++] = p + SKIP_ONE; + } + } +} + +/* One line of the blob as a record. */ +static void msb_parse_record(char *line, msb_rec_t *r) { + char *raw[MSB_FIELDS]; + msb_split(line, raw); + memset(r, 0, sizeof(*r)); + r->tag = line[0]; + if (r->tag == 'V') { + r->f[0] = msb_field(raw[0]); + r->f[1] = raw[1] ? raw[1] : ""; + if (raw[2] && raw[2][0] == '=') { + r->f[2] = msb_field(raw[2]); + } else { + r->tag = 'W'; + } + } else if (r->tag == 'K' || r->tag == 'C') { + r->f[0] = raw[0] ? raw[0] : ""; + } else { + for (int i = 0; i < MSB_FIELDS; i++) { + r->f[i] = msb_field(raw[i]); + } + } +} + +/* Fill the names derived from the file's path. */ +static void msb_file_names(cbm_msb_t *m, msb_file_t *f) { + const char *slash = strrchr(f->rel_path, '/'); + f->name = slash ? slash + SKIP_ONE : f->rel_path; + char *dir = + cbm_arena_strndup(&m->arena, f->rel_path, slash ? (size_t)(slash - f->rel_path) : 0); + const char *dot = strrchr(f->name, '.'); + char *stem = + cbm_arena_strndup(&m->arena, f->name, dot ? (size_t)(dot - f->name) : strlen(f->name)); + m->oom = m->oom || !dir || !stem; + f->dir = dir ? dir : ""; + f->stem = stem ? stem : ""; + f->ext = dot ? dot : ""; + static const char mark[] = {MSB_PATH_MARK, '/', '\0'}; + f->abs_dir = msb_join3(m, mark, f->dir, f->dir[0] ? "/" : ""); + f->abs_path = msb_join3(m, mark, f->rel_path, ""); +} + +bool cbm_msb_add(cbm_msb_t *m, const char *rel_path, const char *scope) { + if (!m || m->oom || !rel_path) { + return false; + } + if (!cbm_msb_is_project_scope(scope) || cbm_ht_get(m->by_path, rel_path)) { + return true; /* no project file, or one that is there already */ + } + /* Invalidate derived state before any allocation or partial mutation. */ + m->generation++; + if (m->nfiles >= m->cap) { + int ncap = m->cap ? m->cap * PAIR_LEN : CBM_SZ_64; + msb_file_t *grown = (msb_file_t *)msb_alloc(m, (size_t)ncap * sizeof(*grown)); + if (!grown) { + return false; + } + if (m->nfiles > 0) { + memcpy(grown, m->files, (size_t)m->nfiles * sizeof(*grown)); + } + m->files = grown; + m->cap = ncap; + } + msb_file_t *f = &m->files[m->nfiles]; + memset(f, 0, sizeof(*f)); + f->rel_path = cbm_arena_strdup(&m->arena, rel_path); + char *buf = cbm_arena_strdup(&m->arena, scope); + if (!f->rel_path || !buf) { + m->oom = true; + return false; + } + msb_file_names(m, f); + int lines = 0; + for (const char *p = buf; *p; p++) { + lines += *p == '\n'; + } + f->recs = (msb_rec_t *)msb_alloc(m, (size_t)(lines + SKIP_ONE) * sizeof(msb_rec_t)); + if (m->oom) { + return false; + } + /* line 0 is the tag, line 1 the P record */ + int line_no = 0; + bool import_group = false; + bool bad_group = false; + const char *group_cond = NULL; + for (char *line = buf; line && *line; line_no++) { + char *nl = strchr(line, '\n'); + if (nl) { + *nl = '\0'; + } + if (line_no == SKIP_ONE) { + char *raw[MSB_FIELDS]; + msb_split(line, raw); /* P sdk state */ + f->sdk = msb_field(raw[0]); + f->readable = raw[1] && raw[1][0] == '-'; + } else if (line_no > SKIP_ONE) { + msb_rec_t *r = &f->recs[f->nrecs]; + msb_parse_record(line, r); + if (import_group && r->tag != 'J' && r->tag != 'E') { + bad_group = true; + } + if (r->tag == 'B') { + bad_group = bad_group || line[1] || r->f[1] || r->f[2] || r->f[3] || r->f[4]; + import_group = true; + group_cond = r->f[0]; + } else if (r->tag == 'E') { + bad_group = bad_group || !import_group || line[1] || r->f[0] || r->f[1] || + r->f[2] || r->f[3] || r->f[4]; + import_group = false; + group_cond = NULL; + } else { + if (r->tag == 'J') { + bad_group = bad_group || !import_group || line[1] || r->f[0] || r->f[4]; + /* The decoded text lives in buf, not in the reused record. + * Each import still evaluates it at its original position. */ + r->tag = 'I'; + r->f[0] = group_cond; + } + f->nrecs++; + } + if (bad_group) { + break; + } + } + line = nl ? nl + SKIP_ONE : NULL; + } + if (bad_group || import_group) { + /* A malformed group is unknown, never an empty closed scope. Keep + * this project conservative without rejecting other project files. */ + f->readable = false; + f->recs[0] = (msb_rec_t){.tag = 'Y'}; + f->nrecs = 1; + } + cbm_ht_set(m->by_path, f->rel_path, (void *)(intptr_t)(m->nfiles + SKIP_ONE)); + m->nfiles++; + return !m->oom; +} + +static void msb_work(uint64_t n); + +static int msb_file_index(const cbm_msb_t *m, const char *rel_path) { + msb_work(SKIP_ONE); /* one logical path lookup, not hash-table bucket probes */ + intptr_t v = (intptr_t)cbm_ht_get(m->by_path, rel_path); + return v > 0 ? (int)(v - SKIP_ONE) : CBM_NOT_FOUND; +} + +bool cbm_msb_has(const cbm_msb_t *m, const char *rel_path) { + return m && rel_path && msb_file_index(m, rel_path) >= 0; +} + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +#include +static _Atomic uint64_t msb_record_counter; +static _Atomic uint64_t msb_work_counter; +static _Atomic uint64_t msb_nearest_counter; +static _Atomic uint64_t msb_peak_counter; +static _Atomic int msb_value_alloc_after; +static _Atomic bool msb_value_alloc_failed; +static _Atomic int msb_item_fail_operation; +static _Atomic int msb_item_fail_after; +static _Atomic bool msb_item_operation_failed; +static _Atomic int msb_node_fail_operation; +static _Atomic int msb_node_fail_after; +static _Atomic bool msb_node_alloc_failed; +static _Atomic uint64_t msb_state_allocations; +static _Atomic uint64_t msb_state_slot_reuses; +static _Atomic uint64_t msb_state_witnessed_slot_reuses; +static _Atomic uint64_t msb_state_same_length_value_changes; +static _Atomic uint64_t msb_state_witness_skips; +static _Atomic uint64_t msb_state_revision_errors; + +void cbm_msb_test_fail_node_alloc(cbm_msb_node_fail_operation_t operation, int nth) { + atomic_store(&msb_node_fail_operation, (int)operation); + atomic_store(&msb_node_fail_after, nth); + atomic_store(&msb_node_alloc_failed, false); +} + +bool cbm_msb_test_node_alloc_failed(void) { + return atomic_load(&msb_node_alloc_failed); +} + +void cbm_msb_test_state_stats(cbm_msb_state_test_stats_t *out) { + out->allocations = atomic_load(&msb_state_allocations); + out->slot_reuses = atomic_load(&msb_state_slot_reuses); + out->witnessed_slot_reuses = atomic_load(&msb_state_witnessed_slot_reuses); + out->same_length_value_changes = atomic_load(&msb_state_same_length_value_changes); + out->witness_skips = atomic_load(&msb_state_witness_skips); + out->revision_errors = atomic_load(&msb_state_revision_errors); +} + +void cbm_msb_test_fail_item_operation(cbm_msb_item_fail_operation_t operation, int nth) { + atomic_store(&msb_item_fail_operation, (int)operation); + atomic_store(&msb_item_fail_after, nth); + atomic_store(&msb_item_operation_failed, false); +} + +bool cbm_msb_test_item_operation_failed(void) { + return atomic_load(&msb_item_operation_failed); +} + +static bool msb_fail_item_operation(cbm_msb_item_fail_operation_t operation) { + if (operation == CBM_MSB_ITEM_FAIL_NONE || + atomic_load(&msb_item_fail_operation) != (int)operation) { + return false; + } + int n = atomic_load(&msb_item_fail_after); + while (n > 0) { + if (atomic_compare_exchange_weak(&msb_item_fail_after, &n, n - SKIP_ONE)) { + if (n == SKIP_ONE) { + atomic_store(&msb_item_operation_failed, true); + return true; + } + return false; + } + } + return false; +} + +void cbm_msb_test_fail_value_alloc_after(int nth) { + atomic_store(&msb_value_alloc_after, nth); + atomic_store(&msb_value_alloc_failed, false); +} + +bool cbm_msb_test_value_alloc_failed(void) { + return atomic_load(&msb_value_alloc_failed); +} + +static bool msb_fail_value_alloc(void) { + int n = atomic_load(&msb_value_alloc_after); + while (n > 0) { + if (atomic_compare_exchange_weak(&msb_value_alloc_after, &n, n - SKIP_ONE)) { + if (n == SKIP_ONE) { + atomic_store(&msb_value_alloc_failed, true); + return true; + } + return false; + } + } + return false; +} + +static _Atomic int msb_prop_insert_after; +static _Atomic bool msb_prop_insert_failed; +static _Atomic uint64_t msb_value_live_counter; + +void cbm_msb_test_fail_prop_insert_after(int nth) { + atomic_store(&msb_prop_insert_after, nth); + atomic_store(&msb_prop_insert_failed, false); +} + +bool cbm_msb_test_prop_insert_failed(void) { + return atomic_load(&msb_prop_insert_failed); +} + +uint64_t cbm_msb_test_value_live_bytes(void) { + return atomic_load(&msb_value_live_counter); +} + +static bool msb_fail_prop_insert(void) { + int n = atomic_load(&msb_prop_insert_after); + while (n > 0) { + if (atomic_compare_exchange_weak(&msb_prop_insert_after, &n, n - SKIP_ONE)) { + if (n == SKIP_ONE) { + atomic_store(&msb_prop_insert_failed, true); + return true; + } + return false; + } + } + return false; +} + +void cbm_msb_test_cost_reset(void) { + atomic_store(&msb_record_counter, 0); + atomic_store(&msb_work_counter, 0); + atomic_store(&msb_nearest_counter, 0); + atomic_store(&msb_peak_counter, 0); + atomic_store(&msb_state_allocations, 0); + atomic_store(&msb_state_slot_reuses, 0); + atomic_store(&msb_state_witnessed_slot_reuses, 0); + atomic_store(&msb_state_same_length_value_changes, 0); + atomic_store(&msb_state_witness_skips, 0); + atomic_store(&msb_state_revision_errors, 0); +} + +void cbm_msb_test_cost(uint64_t *records, uint64_t *peak_bytes) { + *records = atomic_load(&msb_record_counter); + *peak_bytes = atomic_load(&msb_peak_counter); +} + +uint64_t cbm_msb_test_work(void) { + return atomic_load(&msb_work_counter); +} + +uint64_t cbm_msb_test_nearest_steps(void) { + return atomic_load(&msb_nearest_counter); +} + +static void msb_nearest_step(void) { + atomic_fetch_add_explicit(&msb_nearest_counter, SKIP_ONE, memory_order_relaxed); +} + +static void msb_work(uint64_t n) { + atomic_fetch_add_explicit(&msb_work_counter, n, memory_order_relaxed); +} + +static void msb_record(void) { + atomic_fetch_add_explicit(&msb_record_counter, SKIP_ONE, memory_order_relaxed); + msb_work(SKIP_ONE); +} + +static void msb_peak(uint64_t bytes) { + uint64_t old = atomic_load(&msb_peak_counter); + while (old < bytes && !atomic_compare_exchange_weak(&msb_peak_counter, &old, bytes)) {} +} +#else +static void msb_nearest_step(void) {} +static void msb_work(uint64_t n) { + (void)n; +} +static void msb_record(void) {} +static bool msb_fail_value_alloc(void) { + return false; +} +static bool msb_fail_prop_insert(void) { + return false; +} +static bool msb_fail_item_operation(cbm_msb_item_fail_operation_t operation) { + (void)operation; + return false; +} +#endif + +static bool msb_fail_node_alloc(cbm_msb_node_fail_operation_t operation) { +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (operation == CBM_MSB_NODE_FAIL_NONE || + atomic_load(&msb_node_fail_operation) != (int)operation) + return false; + int n = atomic_load(&msb_node_fail_after); + while (n > 0) { + if (atomic_compare_exchange_weak(&msb_node_fail_after, &n, n - 1)) { + if (n == 1) { + atomic_store(&msb_node_alloc_failed, true); + return true; + } + return false; + } + } +#else + (void)operation; +#endif + return false; +} + +typedef struct st_node st_node_t; +typedef struct st_rope st_rope_t; +typedef struct st_component st_component_t; +typedef struct msb_state msb_state_t; +typedef struct { + st_node_t *root[3]; +} msb_view_t; + +/* ── One evaluation ──────────────────────────────────────────────── */ + +typedef struct { + char *value; /* owned current value; NULL: set, but to nothing this reader knows */ + size_t capacity; +} msb_prop_t; + +typedef struct { + const msb_rec_t *group; /* the 's H record */ + const msb_rec_t *item; /* the 's N record */ + int file; +} msb_item_t; + +typedef struct { + int file; + int rec; + st_component_t *owner; + msb_tri_t group; /* the condition of the being read */ + const msb_rec_t *hgroup; /* the being read */ +} msb_frame_t; + +enum { MSB_SEEDS = 5, MSB_LIVE_OWNERS = 5 }; +static const char *const MSB_SEED_NAMES[MSB_SEEDS] = { + "msbuildprojectname", "msbuildprojectfile", "msbuildprojectextension", "msbuildprojectfullpath", + "msbuildprojectdirectory"}; + +typedef struct { + bool present; /* absent and present-unknown are different dependencies */ + const char *value; + size_t bytes; + bool exception; /* expected state differs from the immutable input layers */ +} msb_prop_dep_t; + +typedef struct { + bool present; + bool exception; +} msb_set_dep_t; + +typedef struct { + CBMHashTable *props; + CBMHashTable *seen; + CBMHashTable *poisoned; + msb_prop_dep_t *seeds[MSB_SEEDS]; + size_t prop_exceptions; + size_t seen_exceptions; + size_t poisoned_exceptions; +} msb_inputs_t; + +typedef struct msb_eval msb_eval_t; +typedef struct { + const msb_eval_t *owners[MSB_LIVE_OWNERS]; /* prefix, target, project, seeds, items */ + const size_t *metadata_bytes; /* live directory-memo payloads, including pending insertion */ + const size_t *state_bytes; + const CBMArena *state_names; +} msb_live_t; + +struct msb_eval { + const cbm_msb_t *m; + msb_state_t *state; + msb_view_t view, state_inputs; + st_node_t *reuse_nodes; /* operation-local identity reuse, never a lookup layer */ + st_component_t *builder, *completed; + int main_project; + const msb_eval_t *base; /* immutable completed prefix, never written through */ + const msb_eval_t *seeds; /* this project's five initial properties */ + const msb_eval_t *effects; /* completed target writes, highest precedence in pass 2 */ + msb_eval_t *incoming; /* read-only project input while capturing effects or final items */ + const msb_live_t *live; /* simultaneous owners for peak accounting */ + msb_inputs_t inputs; + CBMHashTable *shared_removals; /* borrowed exact item result, only during publication */ + bool capture_inputs; + cbm_msb_item_fail_operation_t item_alloc_operation; + bool track_seed_reads; + bool seed_read; + CBMArena arena; /* keys, property owners, frames, items and output copies */ + CBMArena scratch; /* expressions and paths, released after their consumers finish */ + CBMHashTable *props; /* lower-cased name -> msb_prop_t* */ + CBMHashTable *seen; /* rel_path: files evaluated */ + CBMHashTable *poisoned; /* rel_path: files whose properties were made unknown */ + msb_item_t *items; + int nitems; + int cap_items; + msb_frame_t *frames; + int nframes; + int cap_frames; + bool open; + int unevaluable; + int outside; + size_t result_bytes; + size_t value_bytes; /* live property buffers, including a replacement before publication */ + bool oom; +}; + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +static size_t ev_owned_bytes(const msb_eval_t *ev) { + return ev ? cbm_arena_capacity(&ev->arena) + cbm_arena_capacity(&ev->scratch) + + ev->value_bytes + ev->result_bytes + : 0; +} +#endif + +static void ev_peak(const msb_eval_t *ev) { +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + size_t bytes = ev_owned_bytes(ev); + if (ev->live) { + bytes = ev->live->metadata_bytes ? *ev->live->metadata_bytes : 0; + if (ev->live->state_bytes) + bytes += *ev->live->state_bytes; + if (ev->live->state_names) + bytes += cbm_arena_capacity(ev->live->state_names); + for (int i = 0; i < MSB_LIVE_OWNERS; i++) { + bytes += ev_owned_bytes(ev->live->owners[i]); + } + } + msb_peak(bytes); +#else + (void)ev; +#endif +} + +static void *ev_alloc(msb_eval_t *ev, size_t n) { + void *p = msb_fail_item_operation(ev->item_alloc_operation) + ? NULL + : cbm_arena_alloc(&ev->arena, n ? n : SKIP_ONE); + ev->oom = ev->oom || !p; + ev_peak(ev); + return p; +} + +static char *ev_strndup(msb_eval_t *ev, const char *s, size_t n) { + char *p = msb_fail_item_operation(ev->item_alloc_operation) + ? NULL + : cbm_arena_strndup(&ev->arena, s, n); + if (p) { + msb_work(n + SKIP_ONE); + } + ev->oom = ev->oom || !p; + ev_peak(ev); + return p; +} + +static void *scratch_alloc(msb_eval_t *ev, size_t n) { + void *p = msb_fail_item_operation(ev->item_alloc_operation) + ? NULL + : cbm_arena_alloc(&ev->scratch, n ? n : SKIP_ONE); + ev->oom = ev->oom || !p; + ev_peak(ev); + return p; +} + +static char *scratch_strndup(msb_eval_t *ev, const char *s, size_t n) { + char *p = msb_fail_item_operation(ev->item_alloc_operation) + ? NULL + : cbm_arena_strndup(&ev->scratch, s, n); + if (p) { + msb_work(n + SKIP_ONE); + } + ev->oom = ev->oom || !p; + ev_peak(ev); + return p; +} + +/* Every property buffer is owned by a table entry. Account for the new and + * old buffers simultaneously until copying has finished and the old one is + * released; neither a replacement nor an unknown value retains history. */ +static char *value_alloc(msb_eval_t *ev, size_t n) { + if (n > SIZE_MAX - ev->value_bytes || msb_fail_value_alloc()) { + ev->oom = true; + return NULL; + } + char *p = (char *)cbm_alloc(CBM_MEM_CLASS_OTHER, n); + if (!p) { + ev->oom = true; + return NULL; + } + ev->value_bytes += n; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + atomic_fetch_add_explicit(&msb_value_live_counter, n, memory_order_relaxed); +#endif + ev_peak(ev); + return p; +} + +static void value_clear(msb_eval_t *ev, msb_prop_t *p) { + if (p->value) { + cbm_free(CBM_MEM_CLASS_OTHER, p->value); + ev->value_bytes -= p->capacity; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + atomic_fetch_sub_explicit(&msb_value_live_counter, p->capacity, memory_order_relaxed); +#endif + } + p->value = NULL; + p->capacity = 0; +} + +static void prop_clear(const char *key, void *value, void *userdata) { + (void)key; + msb_work(SKIP_ONE); /* every property owner visited during teardown */ + value_clear((msb_eval_t *)userdata, (msb_prop_t *)value); +} + +/* The table key of a property name; false for a name too long to keep. */ +static bool msb_key(const char *name, size_t len, char key[MSB_NAME_MAX]) { + if (len == 0 || len >= MSB_NAME_MAX) { + return false; + } + for (size_t i = 0; i < len; i++) { + key[i] = (char)tolower((unsigned char)name[i]); + } + key[len] = '\0'; + msb_work(len + SKIP_ONE); /* lower-casing and the terminator */ + return true; +} + +static void st_set_property(msb_eval_t *ev, const char *name, const char *value); +static const msb_prop_t *st_property(msb_eval_t *ev, const char *key); + +/* Set a property; value NULL makes it unknown. */ +static void msb_set(msb_eval_t *ev, const char *name, const char *value) { + if (ev->state) { + st_set_property(ev, name, value); + return; + } + if (ev->oom) { + return; + } + char key[MSB_NAME_MAX]; + if (!msb_key(name, strlen(name), key)) { + return; + } + msb_work(SKIP_ONE); + msb_prop_t *p = (msb_prop_t *)cbm_ht_get(ev->props, key); + if (!p) { + p = (msb_prop_t *)ev_alloc(ev, sizeof(*p)); + if (!p) { + return; + } + memset(p, 0, sizeof(*p)); + char *k = ev_strndup(ev, key, strlen(key)); + if (!k) { + return; + } + if (!msb_fail_prop_insert()) { + msb_work(SKIP_ONE); + cbm_ht_set(ev->props, k, p); + } + /* set returns the old value, so NULL alone cannot prove insertion. + * Acquire no heap buffer until cleanup can reach this owner. */ + msb_work(SKIP_ONE); + if (cbm_ht_get(ev->props, k) != p) { + ev->oom = true; + return; + } + } + if (!value) { + value_clear(ev, p); + return; + } + size_t bytes = strlen(value) + SKIP_ONE; + if (p->capacity == bytes) { + memmove(p->value, value, bytes); + msb_work(bytes); + return; + } + char *replacement = value_alloc(ev, bytes); + if (!replacement) { + return; /* sticky OOM prevents publishing a partial evaluation */ + } + memcpy(replacement, value, bytes); + msb_work(bytes); + value_clear(ev, p); + p->value = replacement; + p->capacity = bytes; +} + +/* Compare exact input state. Known-empty is a one-byte owned value; + * unknown and absent remain distinct even though both expand as unknown. */ +static bool prop_dep_matches(const msb_prop_dep_t *dep, const msb_prop_t *p) { + msb_work(SKIP_ONE); + if (dep->present != (p != NULL)) { + return false; + } + if (!p) { + return true; + } + if ((dep->value != NULL) != (p->value != NULL)) { + return false; + } + if (!p->value) { + return true; + } + if (dep->bytes != p->capacity) { + return false; + } + msb_work(dep->bytes); + return memcmp(dep->value, p->value, dep->bytes) == 0; +} + +/* Context-owned persistent state. Compressed radix branches split on property + * or file IDs; identity belongs to immutable content, never to a pool address. */ +enum { ST_PROPS, ST_SEEN, ST_POISON, ST_DOMAINS, ST_SLOTS = 64 }; +typedef struct st_value { + unsigned refs; + msb_prop_t prop; +} st_value_t; +typedef struct st_slab st_slab_t; +struct st_node { + st_node_t *child[2]; + st_node_t *next; + st_slab_t *slab; + st_value_t *value; + uint64_t revision; + uint64_t witness; + uint64_t epoch; + uint32_t key; + unsigned refs; + int bit; + unsigned char domain; + bool dependency; + bool present; + bool all_present; + bool all_absent; + bool witness_valid; + bool witnessed; +}; +struct st_slab { + st_slab_t *next, *prev; + st_slab_t *available_next, *available_prev; + st_slab_t *empty_next, *empty_prev; + bool on_empty; + st_node_t *free; + unsigned used; + st_node_t slots[ST_SLOTS]; +}; +struct st_rope { + st_rope_t *left, *right, *next; + msb_item_t item; + unsigned refs; + int height; + int count; +}; +struct st_component { + msb_view_t writes, inputs; + st_rope_t *items; + st_component_t *parent; + int file; + unsigned refs; + bool poison, retain; + bool open; + int unevaluable, outside; + bool start_open; + int start_unevaluable, start_outside; +}; +struct msb_state { + CBMHashTable *symbols; + CBMArena names; + uint32_t next_symbol; + st_slab_t *slabs, *available, *empty; + st_component_t **components; + int files; + uint64_t generation, epoch, revision, audited_revision; + size_t bytes; + st_rope_t *item_a, *item_b; + st_rope_t *capture_a, *capture_b; +}; + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +#define ST_STAT(name) atomic_fetch_add_explicit(&msb_state_##name, 1, memory_order_relaxed) +#else +#define ST_STAT(name) ((void)0) +#endif + +static st_node_t *st_hold(st_node_t *n) { + if (n) { + n->refs++; + msb_work(SKIP_ONE); + } + return n; +} +static st_value_t *st_value_hold(st_value_t *v) { + if (v) { + v->refs++; + msb_work(SKIP_ONE); + } + return v; +} +static void st_value_drop(msb_state_t *s, st_value_t *v) { + if (v) { + msb_work(SKIP_ONE); + if (--v->refs == 0) { + s->bytes -= sizeof(*v) + v->prop.capacity; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + atomic_fetch_sub_explicit(&msb_value_live_counter, v->prop.capacity, + memory_order_relaxed); +#endif + cbm_free(CBM_MEM_CLASS_OTHER, v); + } + } +} +static void st_available_remove(msb_state_t *s, st_slab_t *b) { + if (b->available_prev) + b->available_prev->available_next = b->available_next; + else + s->available = b->available_next; + if (b->available_next) + b->available_next->available_prev = b->available_prev; + b->available_next = b->available_prev = NULL; +} +static void st_available_add(msb_state_t *s, st_slab_t *b) { + b->available_next = s->available; + if (s->available) + s->available->available_prev = b; + s->available = b; +} +static void st_empty_remove(msb_state_t *s, st_slab_t *b) { + if (!b->on_empty) + return; + if (b->empty_prev) + b->empty_prev->empty_next = b->empty_next; + else + s->empty = b->empty_next; + if (b->empty_next) + b->empty_next->empty_prev = b->empty_prev; + b->empty_next = b->empty_prev = NULL; + b->on_empty = false; +} +static void st_empty_add(msb_state_t *s, st_slab_t *b) { + b->empty_next = s->empty; + b->empty_prev = NULL; + if (s->empty) + s->empty->empty_prev = b; + s->empty = b; + b->on_empty = true; +} +/* Intrusive zero-ref queue: neither radix nor component/rope depth consumes + * the C call stack, and returning slots cannot allocate. Slabs stay alive. */ +static void st_drop(msb_state_t *s, st_node_t *n) { + st_node_t *pending = NULL; + if (n && --n->refs == 0) { + n->next = pending; + pending = n; + } + while (pending) { + n = pending; + pending = n->next; + msb_work(SKIP_ONE); + for (int i = 0; i < PAIR_LEN; i++) { + st_node_t *c = n->child[i]; + if (c && --c->refs == 0) { + c->next = pending; + pending = c; + } + } + st_value_drop(s, n->value); + st_slab_t *b = n->slab; + if (!b->free) + st_available_add(s, b); + n->next = b->free; + b->free = n; + if (--b->used == 0) + st_empty_add(s, b); + } +} +static void st_view_drop(msb_state_t *s, msb_view_t *v) { + for (int d = 0; d < ST_DOMAINS; d++) { + st_drop(s, v->root[d]); + v->root[d] = NULL; + } +} +static st_node_t *st_new(msb_eval_t *ev, cbm_msb_node_fail_operation_t phase) { + msb_state_t *s = ev->state; + if (ev->oom || s->revision == UINT64_MAX) { + ev->oom = true; + return NULL; + } + /* This is a real slot acquisition. Empty/identity operations never call it. */ + if (msb_fail_node_alloc(phase) || msb_fail_item_operation(ev->item_alloc_operation)) { + ev->oom = true; + return NULL; + } + st_slab_t *b = s->available; + if (!b) { + b = (st_slab_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*b)); + if (!b) { + ev->oom = true; + return NULL; + } + s->bytes += sizeof(*b); + b->next = s->slabs; + if (s->slabs) + s->slabs->prev = b; + s->slabs = b; + for (int i = ST_SLOTS; i > 0;) { + st_node_t *n = &b->slots[--i]; + n->slab = b; + n->next = b->free; + b->free = n; + } + st_available_add(s, b); + msb_work(sizeof(*b)); + ev_peak(ev); + } + st_empty_remove(s, b); + st_node_t *n = b->free; + uint64_t previous = n->revision; + bool witnessed = n->witnessed; + b->free = n->next; + b->used++; + if (!b->free) + st_available_remove(s, b); + memset(n, 0, sizeof(*n)); + n->slab = b; + n->refs = 1; + n->bit = -1; + n->revision = ++s->revision; + n->epoch = s->epoch; + ST_STAT(allocations); + if (previous) { + ST_STAT(slot_reuses); + if (witnessed) + ST_STAT(witnessed_slot_reuses); + } + if (!n->revision || n->revision <= previous || n->revision <= s->audited_revision) { + ST_STAT(revision_errors); + } + s->audited_revision = n->revision; + msb_work(sizeof(*n)); + return n; +} +static int st_high_bit(uint32_t key) { + int bit = -1; + while (key) { + key >>= 1; + bit++; + msb_work(SKIP_ONE); + } + return bit; +} +static bool st_same_range(uint32_t a, uint32_t b, int bit) { + return bit == 31 || (a >> (bit + 1)) == (b >> (bit + 1)); +} +static st_node_t *st_get(st_node_t *n, uint32_t key) { + while (n && n->bit >= 0) { + msb_work(SKIP_ONE); + if (!st_same_range(n->key, key, n->bit)) + return NULL; + n = n->child[(key >> n->bit) & 1]; + } + msb_work(SKIP_ONE); + return n && n->key == key ? n : NULL; +} +/* Restrict a map to the exact dependency prefix, not merely its depth. */ +static st_node_t *st_region(st_node_t *n, uint32_t key, int bit) { + while (n && n->bit > bit) { + msb_work(SKIP_ONE); + if (!st_same_range(n->key, key, n->bit)) + return NULL; + n = n->child[(key >> n->bit) & 1]; + } + msb_work(SKIP_ONE); + return n && st_same_range(n->key, key, bit) ? n : NULL; +} +static bool st_value_equal(st_value_t *a, st_value_t *b) { + msb_work(SKIP_ONE); + if (a == b) + return true; + if (!a || !b || a->prop.capacity != b->prop.capacity) + return false; + msb_work(a->prop.capacity); + return memcmp(a->prop.value, b->prop.value, a->prop.capacity) == 0; +} +static bool st_expected_equal(st_node_t *a, st_node_t *b) { + return a->present == b->present && st_value_equal(a->value, b->value); +} +static void st_certify(msb_eval_t *ev, st_node_t *dep, st_node_t *view, int domain); +static st_node_t *st_branch(msb_eval_t *ev, st_node_t *l, st_node_t *r, int bit, bool dep, + int domain, cbm_msb_node_fail_operation_t phase) { + if (!l) + return st_hold(r); + if (!r) + return st_hold(l); + if (!dep && ev->reuse_nodes) { + /* One borrowed root exists only during this operation. A bounded + * radix lookup may reuse an exact state branch; witnesses do not + * establish equivalence, and no pair table is retained. */ + st_node_t *same = st_region(ev->reuse_nodes, l->key, bit); + if (same && !same->dependency && same->domain == domain && + same->epoch == ev->state->epoch && same->bit == bit && same->key == l->key && + same->child[0] == l && same->child[1] == r) + return st_hold(same); + } + st_node_t *n = st_new(ev, phase); + if (n) { + n->child[0] = st_hold(l); + n->child[1] = st_hold(r); + n->bit = bit; + n->key = l->key; + n->dependency = dep; + n->domain = (unsigned char)domain; + n->all_present = l->all_present && r->all_present; + n->all_absent = dep && l->all_absent && r->all_absent; + /* A fresh join starts without a witness. Certify every constraint + * explicitly before recording an aligned witness for this branch. */ + if (dep) + st_certify(ev, n, ev->view.root[domain], domain); + } + return n; +} +/* Immutable ordered overlay/union. Each recursive call eliminates a radix + * level. Unchanged/empty subtrees are shared; there is no pair-operation cache. */ +static st_node_t *st_union(msb_eval_t *ev, st_node_t *a, st_node_t *b, bool deps, int domain, + cbm_msb_node_fail_operation_t phase) { + msb_work(SKIP_ONE); + if (a == b || !b) + return st_hold(a); + if (!a) + return st_hold(b); + if (ev->oom) + return NULL; + int split = st_high_bit(a->key ^ b->key); + int top = a->bit > b->bit ? a->bit : b->bit; + if (split > top) { + return ((a->key >> split) & 1) ? st_branch(ev, b, a, split, deps, domain, phase) + : st_branch(ev, a, b, split, deps, domain, phase); + } + if (top < 0) { + if (deps && !st_expected_equal(a, b)) { + ev->oom = true; + return NULL; + } + return st_hold(a); + } + st_node_t *ac[2] = {NULL, NULL}, *bc[2] = {NULL, NULL}; + if (a->bit == top) { + ac[0] = a->child[0]; + ac[1] = a->child[1]; + } else + ac[(a->key >> top) & 1] = a; + if (b->bit == top) { + bc[0] = b->child[0]; + bc[1] = b->child[1]; + } else + bc[(b->key >> top) & 1] = b; + st_node_t *l = st_union(ev, ac[0], bc[0], deps, domain, phase); + st_node_t *r = st_union(ev, ac[1], bc[1], deps, domain, phase); + st_node_t *out = NULL; + if (!ev->oom) { + if (a->bit == top && l == a->child[0] && r == a->child[1]) + out = st_hold(a); + else if (b->bit == top && l == b->child[0] && r == b->child[1]) + out = st_hold(b); + else + out = st_branch(ev, l, r, top, deps, domain, phase); + } + st_drop(ev->state, l); + st_drop(ev->state, r); + if (out && out != a && a->bit == top && out->revision == a->revision) + ST_STAT(revision_errors); + if (out && out != b && b->bit == top && out->revision == b->revision) + ST_STAT(revision_errors); + return out; +} +static bool st_validate(msb_eval_t *ev, st_node_t *dep, st_node_t *view, int domain) { + msb_work(SKIP_ONE); + if (!dep) + return true; + st_node_t *at = st_region(view, dep->key, dep->bit); + /* Exact absence constraints match an empty aligned range independently + * of any historical revision witness. Present unknown is not absent. */ + if (!at && dep->dependency && dep->all_absent && dep->domain == domain && + dep->epoch == ev->state->epoch) + return true; + if (dep->witness_valid && dep->epoch == ev->state->epoch && dep->domain == domain && + dep->witness == (at ? at->revision : 0)) { + ST_STAT(witness_skips); + return true; + } + if (dep->bit < 0) { + if (dep->present != (at != NULL)) + return false; + return !at || st_value_equal(dep->value, at->value); + } + return st_validate(ev, dep->child[0], at, domain) && st_validate(ev, dep->child[1], at, domain); +} +/* Only unpublished nodes may gain a witness. All constraints, kind, epoch + * and aligned range are checked; irrelevant historical state is not retained. */ +static void st_certify(msb_eval_t *ev, st_node_t *dep, st_node_t *view, int domain) { + if (!dep || dep->refs != 1 || dep->witness_valid) + return; + if (st_validate(ev, dep, view, domain)) { + st_node_t *at = st_region(view, dep->key, dep->bit); + dep->witness = at ? at->revision : 0; + dep->witness_valid = true; + dep->epoch = ev->state->epoch; + dep->domain = (unsigned char)domain; + if (at) + at->witnessed = true; + } +} +static st_node_t *st_mask(msb_eval_t *ev, st_node_t *dep, st_node_t *writes, int domain) { + msb_work(SKIP_ONE); + if (!dep || !writes) + return st_hold(dep); + if (dep == writes) + return NULL; + st_node_t *at = st_region(writes, dep->key, dep->bit); + if (!at) + return st_hold(dep); + /* An aligned witness plus all-present constraints proves that every + * dependency key is written here, without walking a shared key domain. */ + if (dep->all_present && dep->witness_valid && dep->epoch == ev->state->epoch && + dep->domain == domain && dep->witness == at->revision) + return NULL; + if (dep->bit < 0) + return NULL; + st_node_t *l = st_mask(ev, dep->child[0], at, domain); + st_node_t *r = st_mask(ev, dep->child[1], at, domain); + st_node_t *out = NULL; + if (!ev->oom) { + if (l == dep->child[0] && r == dep->child[1]) + out = st_hold(dep); + else { + out = st_branch(ev, l, r, dep->bit, true, domain, CBM_MSB_NODE_FAIL_DEPENDENCY); + if (out && out->bit == dep->bit) { + out->witness_valid = dep->witness_valid; + out->witness = dep->witness; + out->epoch = dep->epoch; + } + } + } + st_drop(ev->state, l); + st_drop(ev->state, r); + return out; +} +static uint32_t st_symbol(msb_eval_t *ev, const char *key) { + msb_state_t *s = ev->state; + msb_work(SKIP_ONE); + uintptr_t id = (uintptr_t)cbm_ht_get(s->symbols, key); + if (id) + return (uint32_t)id; + if (s->next_symbol == UINT32_MAX) { + ev->oom = true; + return 0; + } + size_t n = strlen(key); + char *owned = cbm_arena_strndup(&s->names, key, n); + if (!owned) { + ev->oom = true; + return 0; + } + id = ++s->next_symbol; + cbm_ht_set(s->symbols, owned, (void *)id); + msb_work(n + 3); + if ((uintptr_t)cbm_ht_get(s->symbols, owned) != id) { + ev->oom = true; + return 0; + } + ev_peak(ev); + return (uint32_t)id; +} +static void st_record_input(msb_eval_t *ev, msb_view_t *inputs, int domain, uint32_t key) { + if (ev->oom || st_get(inputs->root[domain], key)) + return; + st_node_t *current = st_get(ev->view.root[domain], key); + st_node_t *dep = st_new(ev, CBM_MSB_NODE_FAIL_DEPENDENCY); + if (!dep) + return; + dep->dependency = true; + dep->domain = (unsigned char)domain; + dep->key = key; + dep->present = current != NULL; + dep->all_present = dep->present; + dep->all_absent = !dep->present; + dep->value = current ? st_value_hold(current->value) : NULL; + dep->witness_valid = true; + dep->witness = current ? current->revision : 0; + if (current) + current->witnessed = true; + st_node_t *next = + st_union(ev, inputs->root[domain], dep, true, domain, CBM_MSB_NODE_FAIL_DEPENDENCY); + st_drop(ev->state, dep); + if (!ev->oom) { + st_certify(ev, next, ev->view.root[domain], domain); + st_drop(ev->state, inputs->root[domain]); + inputs->root[domain] = next; + } else + st_drop(ev->state, next); +} +static const msb_prop_t *st_property(msb_eval_t *ev, const char *key) { + uint32_t id = st_symbol(ev, key); + if (!id) + return NULL; + st_node_t *n = st_get(ev->view.root[ST_PROPS], id); + if (ev->capture_inputs) + st_record_input(ev, &ev->state_inputs, ST_PROPS, id); + else if (ev->builder && !st_get(ev->builder->writes.root[ST_PROPS], id)) + st_record_input(ev, &ev->builder->inputs, ST_PROPS, id); + static const msb_prop_t unknown = {0}; + return n ? (n->value ? &n->value->prop : &unknown) : NULL; +} +static bool st_membership(msb_eval_t *ev, uint32_t file, int domain) { + bool has = st_get(ev->view.root[domain], file + 1) != NULL; + if (ev->builder && !st_get(ev->builder->writes.root[domain], file + 1)) + st_record_input(ev, &ev->builder->inputs, domain, file + 1); + return has; +} +static st_value_t *st_make_value(msb_eval_t *ev, const char *text) { + if (!text) + return NULL; + size_t bytes = strlen(text) + 1; + st_value_t *v = msb_fail_value_alloc() + ? NULL + : (st_value_t *)cbm_alloc(CBM_MEM_CLASS_OTHER, sizeof(*v) + bytes); + if (!v) { + ev->oom = true; + return NULL; + } + v->refs = 1; + v->prop.value = (char *)(v + 1); + v->prop.capacity = bytes; + memcpy(v->prop.value, text, bytes); + ev->state->bytes += sizeof(*v) + bytes; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + atomic_fetch_add_explicit(&msb_value_live_counter, bytes, memory_order_relaxed); +#endif + msb_work(bytes + sizeof(*v)); + ev_peak(ev); + return v; +} +static void st_write(msb_eval_t *ev, uint32_t key, int domain, const char *text) { + if (ev->oom) + return; + st_node_t *old = st_get(ev->view.root[domain], key); +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + /* Independent before/after audit of every existing ancestor on this + * changed key's path. Snapshot IDs, not addresses that may be recycled. */ + struct { + uint32_t key; + int bit; + uint64_t revision; + } before[33]; + int before_count = 0; + for (st_node_t *n = ev->view.root[domain]; n && before_count < 33;) { + before[before_count].key = n->key; + before[before_count].bit = n->bit; + before[before_count++].revision = n->revision; + if (n->bit < 0 || !st_same_range(n->key, key, n->bit)) + break; + n = n->child[(key >> n->bit) & 1]; + } +#endif + if (domain == ST_PROPS && old && old->value && text && + old->value->prop.capacity == strlen(text) + 1 && strcmp(old->value->prop.value, text)) + ST_STAT(same_length_value_changes); + cbm_msb_node_fail_operation_t phase = + ev->builder && ev->builder->retain ? CBM_MSB_NODE_FAIL_CAPTURE : CBM_MSB_NODE_FAIL_NONE; + st_node_t *leaf = NULL; + if (old && + ((!text && !old->value) || (text && old->value && !strcmp(text, old->value->prop.value)))) + leaf = st_hold(old); + else { + leaf = st_new(ev, phase); + if (!leaf) + return; + leaf->key = key; + leaf->domain = (unsigned char)domain; + leaf->present = true; + leaf->value = domain == ST_PROPS ? st_make_value(ev, text) : NULL; + } + st_node_t *effect = NULL; + st_node_t *saved_reuse = ev->reuse_nodes; + if (!ev->oom && ev->builder) { + ev->reuse_nodes = ev->view.root[domain]; + effect = st_union(ev, leaf, ev->builder->writes.root[domain], false, domain, phase); + ev->reuse_nodes = saved_reuse; + bool is_new = !st_get(ev->builder->writes.root[domain], key); + if (domain == ST_PROPS && is_new && msb_fail_prop_insert()) { + st_drop(ev->state, effect); + effect = st_hold(ev->builder->writes.root[domain]); + } + if (!st_get(effect, key)) + ev->oom = true; + } + ev->reuse_nodes = effect; + st_node_t *view = + !ev->oom ? st_union(ev, leaf, ev->view.root[domain], false, domain, phase) : NULL; + ev->reuse_nodes = saved_reuse; + if (!ev->builder && domain == ST_PROPS && !old && msb_fail_prop_insert()) { + st_drop(ev->state, view); + view = st_hold(ev->view.root[domain]); + } + if (!st_get(view, key)) + ev->oom = true; + if (!ev->oom) { + if (old && leaf != old && leaf->revision == old->revision) + ST_STAT(revision_errors); +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (leaf != old) { + for (int i = 0; i < before_count; i++) { + /* Only old ranges containing the changed key are affected. */ + if (!st_same_range(before[i].key, key, before[i].bit)) + continue; + st_node_t *after = view; + while (after && after->bit > before[i].bit) { + if (!st_same_range(after->key, before[i].key, after->bit)) { + after = NULL; + break; + } + after = after->child[(before[i].key >> after->bit) & 1]; + } + if (after && !st_same_range(after->key, before[i].key, before[i].bit)) + after = NULL; + if (after && after->revision == before[i].revision) + ST_STAT(revision_errors); + } + } +#endif + st_drop(ev->state, ev->view.root[domain]); + ev->view.root[domain] = view; + if (ev->builder) { + st_drop(ev->state, ev->builder->writes.root[domain]); + ev->builder->writes.root[domain] = effect; + } + } else { + st_drop(ev->state, view); + st_drop(ev->state, effect); + } + st_drop(ev->state, leaf); +} +static void st_set_property(msb_eval_t *ev, const char *name, const char *value) { + char key[MSB_NAME_MAX]; + if (!msb_key(name, strlen(name), key)) + return; + uint32_t id = st_symbol(ev, key); + if (id) + st_write(ev, id, ST_PROPS, value); +} + +static int seed_slot(const char *key) { + for (int i = 0; i < MSB_SEEDS; i++) { + msb_work(SKIP_ONE); + if (strcmp(key, MSB_SEED_NAMES[i]) == 0) { + return i; + } + } + return CBM_NOT_FOUND; +} + +static msb_prop_dep_t *copy_prop_dep(msb_eval_t *ev, const msb_prop_t *p) { + msb_prop_dep_t *dep = (msb_prop_dep_t *)ev_alloc(ev, sizeof(*dep)); + if (!dep) { + return NULL; + } + *dep = (msb_prop_dep_t){.present = p != NULL}; + msb_work(sizeof(*dep)); + if (p && p->value) { + dep->bytes = p->capacity; + dep->value = ev_strndup(ev, p->value, dep->bytes - SKIP_ONE); + } + return ev->oom ? NULL : dep; +} + +/* Record reads that escape the capturing owner's local state. Incoming + * state is immutable for this pass. Target effects use the prefix baseline; + * final items also use completed target writes. Dependency values are owned + * copies, so no project buffer survives through a dependency pointer. */ +static void record_prop_input(msb_eval_t *ev, const char *key, const msb_prop_t *p) { + int seed = seed_slot(key); + if (seed >= 0) { + if (!ev->inputs.seeds[seed]) { + ev->inputs.seeds[seed] = copy_prop_dep(ev, p); + } + return; + } + msb_work(SKIP_ONE); + if (cbm_ht_get(ev->inputs.props, key)) { + return; + } + msb_prop_dep_t *dep = copy_prop_dep(ev, p); + char *owned_key = dep ? ev_strndup(ev, key, strlen(key)) : NULL; + if (!owned_key) { + return; + } + const msb_prop_t *baseline = NULL; + if (ev->incoming->effects) { + msb_work(SKIP_ONE); + baseline = (const msb_prop_t *)cbm_ht_get(ev->incoming->effects->props, key); + } + if (!baseline && ev->incoming->base) { + msb_work(SKIP_ONE); + baseline = (const msb_prop_t *)cbm_ht_get(ev->incoming->base->props, key); + } + dep->exception = !prop_dep_matches(dep, baseline); + msb_work(PAIR_LEN); + cbm_ht_set(ev->inputs.props, owned_key, dep); + if (cbm_ht_get(ev->inputs.props, owned_key) != dep) { + ev->oom = true; + return; + } + ev->inputs.prop_exceptions += dep->exception; +} + +/* A present unknown value shadows lower layers just like a known one. + * During target capture, incoming is a separate read-only owner; during + * pass 2, completed target writes are the highest-precedence layer. */ +static const msb_prop_t *prop_lookup(msb_eval_t *ev, const char *key) { + if (ev->state) + return st_property(ev, key); + const msb_prop_t *p = NULL; + if (ev->effects) { + msb_work(SKIP_ONE); + p = (const msb_prop_t *)cbm_ht_get(ev->effects->props, key); + if (p) { + return p; + } + } + if (ev->props) { + msb_work(SKIP_ONE); + p = (const msb_prop_t *)cbm_ht_get(ev->props, key); + } + if (!p && ev->incoming) { + p = prop_lookup(ev->incoming, key); + if (ev->capture_inputs) { + record_prop_input(ev, key, p); + } + return p; + } + if (!p && ev->base) { + msb_work(SKIP_ONE); + p = (const msb_prop_t *)cbm_ht_get(ev->base->props, key); + } + if (!p && ev->seeds) { + msb_work(SKIP_ONE); + p = (const msb_prop_t *)cbm_ht_get(ev->seeds->props, key); + if (p && ev->track_seed_reads) { + ev->seed_read = true; + } + } + return p; +} + +/* The value of $(name) read in `file`; NULL when it is not known. */ +static const char *msb_ref(msb_eval_t *ev, int file, const char *name, size_t len) { + char key[MSB_NAME_MAX]; + if (!msb_key(name, len, key)) { + return NULL; + } + static const char this_file[] = "msbuildthisfile"; + if (strncmp(key, this_file, sizeof(this_file) - SKIP_ONE) == 0) { + const msb_file_t *f = &ev->m->files[file]; + const char *rest = key + sizeof(this_file) - SKIP_ONE; + if (strcmp(rest, "directory") == 0) { + return f->abs_dir; + } + if (rest[0] == '\0') { + return f->name; + } + if (strcmp(rest, "name") == 0) { + return f->stem; + } + if (strcmp(rest, "extension") == 0) { + return f->ext; + } + if (strcmp(rest, "fullpath") == 0) { + return f->abs_path; + } + } + const msb_prop_t *p = prop_lookup(ev, key); + return p ? p->value : NULL; +} + +/* A value being put together; `over` once it is longer than a value may be. */ +typedef struct { + msb_eval_t *ev; + char *buf; + size_t len; + size_t cap; + bool over; +} msb_sb_t; + +static void msb_sb_put(msb_sb_t *b, const char *s, size_t n) { + if (b->over || b->ev->oom) { + return; + } + if (b->len + n >= MSB_VALUE_MAX) { + b->over = true; + return; + } + if (b->len + n + SKIP_ONE > b->cap) { + size_t ncap = b->cap ? b->cap * PAIR_LEN : CBM_SZ_64; + while (ncap < b->len + n + SKIP_ONE) { + ncap *= PAIR_LEN; + } + char *grown = (char *)scratch_alloc(b->ev, ncap); + if (!grown) { + return; + } + if (b->len > 0) { + memcpy(grown, b->buf, b->len); + msb_work(b->len); + } + b->buf = grown; + b->cap = ncap; + } + memcpy(b->buf + b->len, s, n); + b->len += n; + b->buf[b->len] = '\0'; + msb_work(n + SKIP_ONE); /* append plus its terminator */ +} + +/* `text` with its $(Property) references replaced. NULL when it has no value + * this reader knows: a property that is not known, a property function, an + * item or metadata reference, an escape, a value past MSB_VALUE_MAX. */ +static const char *msb_expand(msb_eval_t *ev, int file, const char *text) { + msb_sb_t b = {.ev = ev}; + for (const char *p = text; *p;) { + if (p[0] == '$' && p[1] == '(') { + const char *nm = p + PAIR_LEN; + const char *e = nm; + while (isalnum((unsigned char)*e) || *e == '_') { + e++; + } + if (e == nm || *e != ')' || isdigit((unsigned char)nm[0])) { + return NULL; + } + const char *v = msb_ref(ev, file, nm, (size_t)(e - nm)); + if (!v) { + return NULL; + } + msb_sb_put(&b, v, strlen(v)); + p = e + SKIP_ONE; + continue; + } + if ((p[0] == '@' || p[0] == '%') && p[1] == '(') { + return NULL; + } + if (p[0] == '%' && isxdigit((unsigned char)p[1]) && isxdigit((unsigned char)p[2])) { + return NULL; + } + msb_sb_put(&b, p, SKIP_ONE); + p++; + } + if (b.over || ev->oom) { + return NULL; + } + return b.buf ? b.buf : ""; +} + +/* ── Conditions ──────────────────────────────────────────────────── */ + +static msb_tri_t tri_not(msb_tri_t v) { + return v == MSB_UNKNOWN ? MSB_UNKNOWN : (v == MSB_TRUE ? MSB_FALSE : MSB_TRUE); +} + +static msb_tri_t tri_and(msb_tri_t a, msb_tri_t b) { + if (a == MSB_FALSE || b == MSB_FALSE) { + return MSB_FALSE; + } + return (a == MSB_TRUE && b == MSB_TRUE) ? MSB_TRUE : MSB_UNKNOWN; +} + +static msb_tri_t tri_or(msb_tri_t a, msb_tri_t b) { + if (a == MSB_TRUE || b == MSB_TRUE) { + return MSB_TRUE; + } + return (a == MSB_FALSE && b == MSB_FALSE) ? MSB_FALSE : MSB_UNKNOWN; +} + +static bool ci_eq_n(const char *a, size_t an, const char *b, size_t bn) { + if (an != bn) { + return false; + } + for (size_t i = 0; i < an; i++) { + if (tolower((unsigned char)a[i]) != tolower((unsigned char)b[i])) { + return false; + } + } + return true; +} + +static bool ci_is(const char *a, size_t an, const char *word) { + return ci_eq_n(a, an, word, strlen(word)); +} + +/* A string MSBuild converts to a boolean: 1, 0, or -1 for any other. */ +static int msb_bool(const char *s, size_t n) { + if (ci_is(s, n, "true") || ci_is(s, n, "on") || ci_is(s, n, "yes")) { + return SKIP_ONE; + } + if (ci_is(s, n, "false") || ci_is(s, n, "off") || ci_is(s, n, "no")) { + return 0; + } + return CBM_NOT_FOUND; +} + +/* A decimal number as its parts, leading and trailing zeros dropped. */ +typedef struct { + bool neg; + const char *ip; + size_t il; + const char *fp; + size_t fl; +} msb_num_t; + +/* [+-]digits[.digits] or [+-].digits, nothing else. */ +static bool msb_decimal(const char *s, size_t n, msb_num_t *out) { + size_t i = 0; + memset(out, 0, sizeof(*out)); + if (i < n && (s[i] == '+' || s[i] == '-')) { + out->neg = s[i] == '-'; + i++; + } + size_t is = i; + while (i < n && isdigit((unsigned char)s[i])) { + i++; + } + size_t ie = i; + size_t fs = i; + size_t fe = i; + if (i < n && s[i] == '.') { + i++; + fs = i; + while (i < n && isdigit((unsigned char)s[i])) { + i++; + } + fe = i; + } + if (i != n || (ie == is && fe == fs)) { + return false; + } + bool nonzero = false; + for (size_t k = is; k < fe; k++) { + nonzero = nonzero || (s[k] != '0' && s[k] != '.'); + } + while (is < ie && s[is] == '0') { + is++; + } + while (fe > fs && s[fe - SKIP_ONE] == '0') { + fe--; + } + out->ip = s + is; + out->il = ie - is; + out->fp = s + fs; + out->fl = fe - fs; + if (!nonzero) { + out->neg = false; /* -0 is 0 */ + } + return true; +} + +/* Text some conversion could read as a number although msb_decimal does not + * (hexadecimal, an exponent). */ +static bool msb_numberish(const char *s, size_t n) { + if (n == 0 || !(isdigit((unsigned char)s[0]) || s[0] == '+' || s[0] == '-' || s[0] == '.')) { + return false; + } + for (size_t i = 0; i < n; i++) { + unsigned char c = (unsigned char)s[i]; + if (!(isxdigit(c) || c == 'x' || c == 'X' || c == '.' || c == '+' || c == '-')) { + return false; + } + } + return true; +} + +/* MSBuild's `==` on two expanded operands, as far as it can be known. */ +static msb_tri_t msb_equal_spans(const char *a, size_t an, const char *b, size_t bn) { + if (ci_eq_n(a, an, b, bn)) { + return MSB_TRUE; /* the same text is equal under every conversion */ + } + /* An absolute path: where the repository lies is not known here, so its + * real text is not. Two of them differ as their marked texts do; one is + * never the empty string; any other comparison is open. */ + bool ma = an > 0 && a[0] == MSB_PATH_MARK; + bool mb = bn > 0 && b[0] == MSB_PATH_MARK; + if (memchr(a + ma, MSB_PATH_MARK, an - ma) || memchr(b + mb, MSB_PATH_MARK, bn - mb)) { + return MSB_UNKNOWN; + } + if (ma || mb) { + return ((ma && mb) || an == 0 || bn == 0) ? MSB_FALSE : MSB_UNKNOWN; + } + int ba = msb_bool(a, an); + int bb = msb_bool(b, bn); + if (ba >= 0 && bb >= 0) { + return ba == bb ? MSB_TRUE : MSB_FALSE; + } + msb_num_t na; + msb_num_t nb; + bool da = msb_decimal(a, an, &na); + bool db = msb_decimal(b, bn, &nb); + if (da && db) { + if (na.il + na.fl > MSB_SIGNIFICANT || nb.il + nb.fl > MSB_SIGNIFICANT) { + return MSB_UNKNOWN; /* MSBuild compares doubles: these may round together */ + } + bool same = na.neg == nb.neg && na.il == nb.il && na.fl == nb.fl && + memcmp(na.ip, nb.ip, na.il) == 0 && memcmp(na.fp, nb.fp, na.fl) == 0; + return same ? MSB_TRUE : MSB_FALSE; + } + if ((da || msb_numberish(a, an)) && (db || msb_numberish(b, bn))) { + return MSB_UNKNOWN; + } + return MSB_FALSE; +} + +static void trim_span(const char **s, size_t *n) { + while (*n > 0 && isspace((unsigned char)(*s)[0])) { + (*s)++; + (*n)--; + } + while (*n > 0 && isspace((unsigned char)(*s)[*n - SKIP_ONE])) { + (*n)--; + } +} + +/* A property's value keeps the white space its element was written with. + * Whether a comparison sees it is not something to guess: the two readings + * must agree. */ +static msb_tri_t msb_equal(const char *a, const char *b) { + size_t an = strlen(a); + size_t bn = strlen(b); + msb_tri_t as_written = msb_equal_spans(a, an, b, bn); + trim_span(&a, &an); + trim_span(&b, &bn); + msb_tri_t trimmed = msb_equal_spans(a, an, b, bn); + return as_written == trimmed ? as_written : MSB_UNKNOWN; +} + +/* A value standing alone as a condition. */ +static msb_tri_t msb_truth(const char *v) { + size_t n = strlen(v); + int as_written = msb_bool(v, n); + trim_span(&v, &n); + int trimmed = msb_bool(v, n); + if (as_written != trimmed || trimmed < 0) { + return MSB_UNKNOWN; + } + return trimmed ? MSB_TRUE : MSB_FALSE; +} + +typedef struct { + msb_eval_t *ev; + int file; + const char *s; + int depth; + bool bad; /* not a condition this reader can parse */ +} msb_cond_t; + +static void cond_ws(msb_cond_t *c) { + while (isspace((unsigned char)*c->s)) { + c->s++; + } +} + +/* The keyword `kw` (lower case) stands next: consume it. */ +static bool cond_keyword(msb_cond_t *c, const char *kw) { + cond_ws(c); + size_t n = strlen(kw); + for (size_t i = 0; i < n; i++) { + if (tolower((unsigned char)c->s[i]) != kw[i]) { + return false; + } + } + unsigned char after = (unsigned char)c->s[n]; + if (isalnum(after) || after == '_') { + return false; + } + c->s += n; + return true; +} + +/* Past the group opened by the '(' at c->s (quotes respected); false when it + * does not close. */ +static bool cond_skip_group(msb_cond_t *c) { + int depth = 0; + for (const char *p = c->s; *p; p++) { + if (*p == '\'') { + p = strchr(p + SKIP_ONE, '\''); + if (!p) { + return false; + } + } else if (*p == '(') { + depth++; + } else if (*p == ')' && --depth == 0) { + c->s = p + SKIP_ONE; + return true; + } + } + return false; +} + +/* An operand: 'text', $(Property), a bare word, or a function call. Returns + * its value, NULL when that is not known; *ok false when no operand stands + * here. */ +static const char *cond_value(msb_cond_t *c, bool *ok) { + cond_ws(c); + *ok = true; + const char *s = c->s; + if (*s == '\'') { + const char *e = strchr(s + SKIP_ONE, '\''); + if (!e) { + *ok = false; + return NULL; + } + c->s = e + SKIP_ONE; + const char *lit = scratch_strndup(c->ev, s + SKIP_ONE, (size_t)(e - s - SKIP_ONE)); + return lit ? msb_expand(c->ev, c->file, lit) : NULL; + } + if (s[0] == '$' && s[1] == '(') { + c->s = s + SKIP_ONE; + if (!cond_skip_group(c)) { + *ok = false; + return NULL; + } + const char *ref = scratch_strndup(c->ev, s, (size_t)(c->s - s)); + return ref ? msb_expand(c->ev, c->file, ref) : NULL; + } + const char *e = s; + while (isalnum((unsigned char)*e) || *e == '_' || *e == '.' || *e == '-' || *e == '+') { + e++; + } + if (e == s) { + *ok = false; + return NULL; + } + c->s = e; + cond_ws(c); + if (*c->s == '(') { + *ok = cond_skip_group(c); /* Exists(...), HasTrailingSlash(...): not evaluated */ + return NULL; + } + return scratch_strndup(c->ev, s, (size_t)(e - s)); +} + +static msb_tri_t cond_or(msb_cond_t *c); + +static msb_tri_t cond_comparison_value(msb_cond_t *c) { + bool ok = false; + const char *lhs = cond_value(c, &ok); + if (!ok) { + c->bad = true; + return MSB_UNKNOWN; + } + cond_ws(c); + char op0 = c->s[0]; + char op1 = op0 ? c->s[1] : '\0'; + if ((op0 == '=' || op0 == '!') && op1 == '=') { + c->s += PAIR_LEN; + const char *rhs = cond_value(c, &ok); + if (!ok) { + c->bad = true; + return MSB_UNKNOWN; + } + msb_tri_t v = (lhs && rhs) ? msb_equal(lhs, rhs) : MSB_UNKNOWN; + return op0 == '!' ? tri_not(v) : v; + } + if (op0 == '<' || op0 == '>') { + c->s += op1 == '=' ? PAIR_LEN : SKIP_ONE; /* an order of numbers or versions */ + (void)cond_value(c, &ok); + c->bad = c->bad || !ok; + return MSB_UNKNOWN; + } + return lhs ? msb_truth(lhs) : MSB_UNKNOWN; +} + +/* Both operands must survive through comparison, including every error + * return. Only the primitive result escapes to the surrounding condition. */ +static msb_tri_t cond_comparison(msb_cond_t *c) { + msb_tri_t result = cond_comparison_value(c); + cbm_arena_reset(&c->ev->scratch); + return result; +} + +static msb_tri_t cond_primary(msb_cond_t *c) { + cond_ws(c); + bool negate = false; + while (c->s[0] == '!' && c->s[1] != '=') { + negate = !negate; + c->s++; + cond_ws(c); + } + msb_tri_t v = MSB_UNKNOWN; + if (*c->s == '(') { + if (c->depth >= MSB_COND_DEPTH) { + c->bad = true; + return MSB_UNKNOWN; + } + c->s++; + c->depth++; + v = cond_or(c); + c->depth--; + cond_ws(c); + if (*c->s != ')') { + c->bad = true; + return MSB_UNKNOWN; + } + c->s++; + } else { + v = cond_comparison(c); + } + return negate ? tri_not(v) : v; +} + +static msb_tri_t cond_and(msb_cond_t *c) { + msb_tri_t v = cond_primary(c); + while (!c->bad && cond_keyword(c, "and")) { + v = tri_and(v, cond_primary(c)); + } + return v; +} + +static msb_tri_t cond_or(msb_cond_t *c) { + msb_tri_t v = cond_and(c); + while (!c->bad && cond_keyword(c, "or")) { + v = tri_or(v, cond_and(c)); + } + return v; +} + +/* A Condition attribute, read in `file`. `and` binds tighter than `or`. */ +static msb_tri_t msb_cond(msb_eval_t *ev, int file, const char *cond) { + if (!cond) { + return MSB_TRUE; + } + msb_cond_t c = {.ev = ev, .file = file, .s = cond}; + cond_ws(&c); + if (!*c.s) { + return MSB_TRUE; + } + msb_tri_t v = cond_or(&c); + cond_ws(&c); + return (c.bad || *c.s) ? MSB_UNKNOWN : v; +} + +/* ── Imports ─────────────────────────────────────────────────────── */ + +/* Resolve "." and ".." in a '/'-separated path, in place. false when it + * leaves the repository. */ +static bool msb_normalize(char *path) { + /* The text is read by index up to its original length: what is written + * (every kept segment and a '/' after it) never passes the read position, + * but it does overwrite the terminator. */ + size_t len = strlen(path); + size_t w = 0; + size_t i = 0; + while (i < len) { + size_t s = i; + while (i < len && path[i] != '/') { + i++; + } + size_t n = i - s; + i += i < len; /* past the '/' */ + if (n == PAIR_LEN && path[s] == '.' && path[s + SKIP_ONE] == '.') { + if (w == 0) { + return false; + } + w--; /* the '/' that ends the previous segment */ + while (w > 0 && path[w - SKIP_ONE] != '/') { + w--; + } + } else if (n > 0 && !(n == SKIP_ONE && path[s] == '.')) { + memmove(path + w, path + s, n); + msb_work(n + SKIP_ONE); /* copied segment and slash */ + w += n; + path[w++] = '/'; + } + } + path[w > 0 ? w - SKIP_ONE : 0] = '\0'; + return true; +} + +/* The file an written in `file` names: its index, or + * CBM_NOT_FOUND. What cannot be followed is counted: a path that is not + * known or names several files (unevaluable), a file the index does not hold + * (outside). */ +static int msb_import_target(msb_eval_t *ev, int file, const char *project) { + const char *value = project ? msb_expand(ev, file, project) : NULL; + if (!value) { + ev->unevaluable++; + return CBM_NOT_FOUND; + } + size_t n = strlen(value); + trim_span(&value, &n); + const msb_file_t *f = &ev->m->files[file]; + bool marked = n > 0 && value[0] == MSB_PATH_MARK; + size_t skip = marked ? SKIP_ONE : 0; + if (n == skip || memchr(value + skip, MSB_PATH_MARK, n - skip) || memchr(value, '*', n) || + memchr(value, '?', n)) { + ev->unevaluable++; + return CBM_NOT_FOUND; + } + bool absolute = + !marked && (value[0] == '/' || value[0] == '\\' || (n > SKIP_ONE && value[1] == ':')); + if (absolute) { + ev->outside++; + return CBM_NOT_FOUND; + } + size_t dl = marked ? 0 : strlen(f->dir); + char *path = (char *)scratch_alloc(ev, dl + n + PAIR_LEN); + if (!path) { + return CBM_NOT_FOUND; + } + memcpy(path, f->dir, dl); + path[dl] = '/'; + memcpy(path + dl + SKIP_ONE, value + skip, n - skip); + path[dl + SKIP_ONE + n - skip] = '\0'; + msb_work(dl + n - skip + PAIR_LEN); + for (char *p = path; *p; p++) { + if (*p == '\\') { + *p = '/'; + } + } + int target = msb_normalize(path) ? msb_file_index(ev->m, path) : CBM_NOT_FOUND; + ev->outside += target < 0; + return target; +} + +static bool set_has(const CBMHashTable *set, const char *key) { + msb_work(SKIP_ONE); + return cbm_ht_get(set, key) != NULL; +} + +/* Stable borrowed sentinel: a frozen set must not retain a stack address. */ +static const char MSB_SET_PRESENT = 0; + +static void record_set_input(msb_eval_t *ev, const char *key, bool present, bool poison) { + CBMHashTable *deps = poison ? ev->inputs.poisoned : ev->inputs.seen; + msb_work(SKIP_ONE); + if (cbm_ht_get(deps, key)) { + return; + } + msb_set_dep_t *dep = (msb_set_dep_t *)ev_alloc(ev, sizeof(*dep)); + char *owned_key = dep ? ev_strndup(ev, key, strlen(key)) : NULL; + if (!owned_key) { + return; + } + const msb_eval_t *base = ev->incoming->base; + bool baseline = base && set_has(poison ? base->poisoned : base->seen, key); + *dep = (msb_set_dep_t){.present = present, .exception = present != baseline}; + msb_work(sizeof(*dep) + PAIR_LEN); + cbm_ht_set(deps, owned_key, dep); + if (cbm_ht_get(deps, owned_key) != dep) { + ev->oom = true; + return; + } + if (poison) { + ev->inputs.poisoned_exceptions += dep->exception; + } else { + ev->inputs.seen_exceptions += dep->exception; + } +} + +static bool seen_has(msb_eval_t *ev, const char *key) { + if (ev->state) { + int file = msb_file_index(ev->m, key); + return file >= 0 && st_membership(ev, (uint32_t)file, ST_SEEN); + } + if (set_has(ev->seen, key)) { + return true; + } + bool present = + ev->incoming ? seen_has(ev->incoming, key) : ev->base && set_has(ev->base->seen, key); + if (ev->capture_inputs) { + record_set_input(ev, key, present, false); + } + return present; +} + +static bool poisoned_has(msb_eval_t *ev, const char *key) { + if (ev->state) { + int file = msb_file_index(ev->m, key); + return file >= 0 && st_membership(ev, (uint32_t)file, ST_POISON); + } + if (set_has(ev->poisoned, key)) { + return true; + } + bool present = ev->incoming ? poisoned_has(ev->incoming, key) + : ev->base && set_has(ev->base->poisoned, key); + if (ev->capture_inputs) { + record_set_input(ev, key, present, true); + } + return present; +} + +/* The nearest file called `name` in `dir` or above it; CBM_NOT_FOUND for none. */ +static int msb_nearest(msb_eval_t *ev, const char *dir, const char *name) { + size_t dl = strlen(dir); + size_t nl = strlen(name); + char *path = (char *)scratch_alloc(ev, dl + nl + PAIR_LEN); + if (!path) { + return CBM_NOT_FOUND; + } + for (;;) { + memcpy(path, dir, dl); + path[dl] = '/'; + memcpy(path + dl + (dl ? SKIP_ONE : 0), name, nl + SKIP_ONE); + msb_work(dl + nl + PAIR_LEN); + msb_nearest_step(); + int found = msb_file_index(ev->m, path); + if (found >= 0 || dl == 0) { + return found; + } + while (dl > 0 && dir[dl - SKIP_ONE] != '/') { + msb_work(SKIP_ONE); + dl--; + } + dl = dl > 0 ? dl - SKIP_ONE : 0; + } +} + +static bool msb_push(msb_eval_t *ev, int file) { + if (ev->nframes >= ev->cap_frames) { + int ncap = ev->cap_frames ? ev->cap_frames * PAIR_LEN : MSB_INIT; + msb_frame_t *grown = (msb_frame_t *)ev_alloc(ev, (size_t)ncap * sizeof(*grown)); + if (!grown) { + return false; + } + if (ev->nframes > 0) { + memcpy(grown, ev->frames, (size_t)ev->nframes * sizeof(*grown)); + msb_work((size_t)ev->nframes * sizeof(*grown)); + } + ev->frames = grown; + ev->cap_frames = ncap; + } + ev->frames[ev->nframes++] = (msb_frame_t){.file = file, .group = MSB_TRUE}; + msb_work(sizeof(msb_frame_t)); + return true; +} + +static st_rope_t *st_rope_hold(st_rope_t *r) { + if (r) { + r->refs++; + msb_work(1); + } + return r; +} +static void st_rope_drop(msb_state_t *s, st_rope_t *r) { + st_rope_t *pending = NULL; + if (r && --r->refs == 0) { + r->next = pending; + pending = r; + } + while (pending) { + r = pending; + pending = r->next; + msb_work(1); + st_rope_t *children[2] = {r->left, r->right}; + for (int i = 0; i < 2; i++) + if (children[i] && --children[i]->refs == 0) { + children[i]->next = pending; + pending = children[i]; + } + s->bytes -= sizeof(*r); + cbm_free(CBM_MEM_CLASS_OTHER, r); + } +} +static st_rope_t *st_rope_node(msb_eval_t *ev, st_rope_t *a, st_rope_t *b, const msb_item_t *item) { + if (ev->oom) + return NULL; + st_rope_t *r = (st_rope_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*r)); + if (!r) { + ev->oom = true; + return NULL; + } + ev->state->bytes += sizeof(*r); + r->refs = 1; + r->left = st_rope_hold(a); + r->right = st_rope_hold(b); + if (item) { + r->item = *item; + r->height = 1; + r->count = 1; + } else { + if (a->count > INT_MAX - b->count) { + ev->oom = true; + st_rope_drop(ev->state, r); + return NULL; + } + r->height = 1 + (a->height > b->height ? a->height : b->height); + r->count = a->count + b->count; + } + msb_work(sizeof(*r)); + ev_peak(ev); + return r; +} +static st_rope_t *st_rope_balance(msb_eval_t *ev, st_rope_t *a, st_rope_t *b) { + if (!a || !b) + return st_rope_hold(a ? a : b); + if (a->height > b->height + 1) { + st_rope_t *right = NULL, *left = NULL, *out = NULL; + if (a->left->height >= a->right->height) { + right = st_rope_node(ev, a->right, b, NULL); + if (right) + out = st_rope_node(ev, a->left, right, NULL); + } else { + left = st_rope_node(ev, a->left, a->right->left, NULL); + right = st_rope_node(ev, a->right->right, b, NULL); + if (left && right) + out = st_rope_node(ev, left, right, NULL); + } + st_rope_drop(ev->state, left); + st_rope_drop(ev->state, right); + return out; + } + if (b->height > a->height + 1) { + st_rope_t *right = NULL, *left = NULL, *out = NULL; + if (b->right->height >= b->left->height) { + left = st_rope_node(ev, a, b->left, NULL); + if (left) + out = st_rope_node(ev, left, b->right, NULL); + } else { + left = st_rope_node(ev, a, b->left->left, NULL); + right = st_rope_node(ev, b->left->right, b->right, NULL); + if (left && right) + out = st_rope_node(ev, left, right, NULL); + } + st_rope_drop(ev->state, left); + st_rope_drop(ev->state, right); + return out; + } + return st_rope_node(ev, a, b, NULL); +} +/* Join height-balanced nonempty spans. Empty import chains disappear here. */ +static st_rope_t *st_rope_join(msb_eval_t *ev, st_rope_t *a, st_rope_t *b) { + msb_work(1); + if (!a) + return st_rope_hold(b); + if (!b) + return st_rope_hold(a); + if (ev->oom) + return NULL; + if (a->height > b->height + 1) { + st_rope_t *tail = st_rope_join(ev, a->right, b); + st_rope_t *out = tail ? st_rope_balance(ev, a->left, tail) : NULL; + st_rope_drop(ev->state, tail); + return out; + } + if (b->height > a->height + 1) { + st_rope_t *head = st_rope_join(ev, a, b->left); + st_rope_t *out = head ? st_rope_balance(ev, head, b->right) : NULL; + st_rope_drop(ev->state, head); + return out; + } + return st_rope_node(ev, a, b, NULL); +} +static void st_component_drop(msb_state_t *s, st_component_t *c) { + if (!c) + return; + msb_work(1); + if (--c->refs) + return; + st_view_drop(s, &c->writes); + st_view_drop(s, &c->inputs); + st_rope_drop(s, c->items); + s->bytes -= sizeof(*c); + cbm_free(CBM_MEM_CLASS_OTHER, c); +} +static bool st_inputs_match(msb_eval_t *ev, const msb_view_t *inputs) { + for (int d = 0; d < ST_DOMAINS; d++) + if (!st_validate(ev, inputs->root[d], ev->view.root[d], d)) + return false; + return true; +} +/* B was selected against the actual after-A state before this call. Preserve + * every earlier A read, and mask only B's reads by A writes (unknown included). */ +static bool st_compose(msb_eval_t *ev, st_component_t *parent, st_component_t *child) { + if (!parent) + return true; + for (int d = 0; d < ST_DOMAINS && !ev->oom; d++) { + st_node_t *external = st_mask(ev, child->inputs.root[d], parent->writes.root[d], d); + st_node_t *inputs = + st_union(ev, parent->inputs.root[d], external, true, d, CBM_MSB_NODE_FAIL_DEPENDENCY); + st_drop(ev->state, external); + st_node_t *writes = st_union(ev, child->writes.root[d], parent->writes.root[d], false, d, + CBM_MSB_NODE_FAIL_OVERLAY); + if (!ev->oom) { + st_certify(ev, inputs, ev->view.root[d], d); + st_drop(ev->state, parent->inputs.root[d]); + parent->inputs.root[d] = inputs; + st_drop(ev->state, parent->writes.root[d]); + parent->writes.root[d] = writes; + } else { + st_drop(ev->state, inputs); + st_drop(ev->state, writes); + } + } + if (!ev->oom) { + st_rope_t *items = st_rope_join(ev, parent->items, child->items); + if (!ev->oom) { + st_rope_drop(ev->state, parent->items); + parent->items = items; + } else + st_rope_drop(ev->state, items); + } + return !ev->oom; +} +static bool st_apply(msb_eval_t *ev, st_component_t *c) { + if (!st_compose(ev, ev->builder, c)) + return false; + cbm_msb_node_fail_operation_t phase = + ev->builder ? CBM_MSB_NODE_FAIL_OVERLAY : CBM_MSB_NODE_FAIL_APPLY; + for (int d = 0; d < ST_DOMAINS && !ev->oom; d++) { + st_node_t *saved_reuse = ev->reuse_nodes; + ev->reuse_nodes = ev->builder ? ev->builder->writes.root[d] : NULL; + st_node_t *next = st_union(ev, c->writes.root[d], ev->view.root[d], false, d, phase); + ev->reuse_nodes = saved_reuse; + if (!ev->oom) { + st_drop(ev->state, ev->view.root[d]); + ev->view.root[d] = next; + } else + st_drop(ev->state, next); + } + if (ev->oom) + return false; + ev->open = ev->open || c->open; + ev->unevaluable += c->unevaluable; + ev->outside += c->outside; + if (!ev->builder) { + c->refs++; + ev->completed = c; + } + return true; +} +/* Returns true only when an interpreter frame was pushed. A validated cache + * hit applies the exact effect here, before the caller can mask dependencies. */ +static bool st_enter(msb_eval_t *ev, int file, bool poison) { + int domain = poison ? ST_POISON : ST_SEEN; + if (st_membership(ev, (uint32_t)file, domain) || ev->oom) + return false; + size_t slot = (size_t)file * 2 + (poison ? 1 : 0); + st_component_t *cached = ev->state->components[slot]; + if (cached && st_inputs_match(ev, &cached->inputs)) { + (void)st_apply(ev, cached); + return false; + } + st_component_t *c = (st_component_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*c)); + if (!c) { + ev->oom = true; + return false; + } + c->refs = 1; + c->parent = ev->builder; + c->file = file; + c->poison = poison; + c->retain = !cached && file != ev->main_project; + c->start_open = ev->open; + c->start_unevaluable = ev->unevaluable; + c->start_outside = ev->outside; + ev->state->bytes += sizeof(*c); + msb_work(sizeof(*c)); + ev_peak(ev); + ev->builder = c; + ev->open = false; + ev->unevaluable = 0; + ev->outside = 0; + if (!msb_push(ev, file)) { + ev->oom = true; + return false; + } + ev->frames[ev->nframes - 1].owner = c; + st_write(ev, (uint32_t)file + 1, domain, NULL); + if (!poison) + ev->unevaluable += !ev->m->files[file].readable; + return !ev->oom; +} +static void st_finish(msb_eval_t *ev, st_component_t *c) { + c->open = ev->open; + c->unevaluable = ev->unevaluable; + c->outside = ev->outside; + ev->open = c->start_open || c->open; + ev->unevaluable += c->start_unevaluable; + ev->outside += c->start_outside; + ev->builder = c->parent; + c->parent = NULL; + if (!st_compose(ev, ev->builder, c)) { + st_component_drop(ev->state, c); + return; + } + if (c->retain) { + size_t slot = (size_t)c->file * 2 + (c->poison ? 1 : 0); + if (!ev->state->components[slot]) { + c->refs++; + ev->state->components[slot] = c; + } + } + if (ev->builder) + st_component_drop(ev->state, c); + else + ev->completed = c; +} +static void st_add_item(msb_eval_t *ev, const msb_rec_t *group, const msb_rec_t *item, int file) { + msb_item_t it = {.group = group, .item = item, .file = file}; + st_rope_t *leaf = st_rope_node(ev, NULL, NULL, &it); + st_rope_t *next = leaf ? st_rope_join(ev, ev->builder->items, leaf) : NULL; + st_rope_drop(ev->state, leaf); + if (!ev->oom) { + st_rope_drop(ev->state, ev->builder->items); + ev->builder->items = next; + } else + st_rope_drop(ev->state, next); +} + +/* `file` may or may not be imported: what it (and what it imports, under any + * condition) sets is unknown from here on. */ +static void msb_poison(msb_eval_t *ev, int file) { + int base = ev->nframes; + if (ev->state) { + if (!st_enter(ev, file, true)) + return; + } else { + if (poisoned_has(ev, ev->m->files[file].rel_path) || !msb_push(ev, file)) + return; + msb_work(SKIP_ONE); + cbm_ht_set(ev->poisoned, ev->m->files[file].rel_path, (void *)&MSB_SET_PRESENT); + } + while (ev->nframes > base && !ev->oom) { + msb_frame_t *fr = &ev->frames[ev->nframes - SKIP_ONE]; + const msb_file_t *f = &ev->m->files[fr->file]; + if (fr->rec >= f->nrecs) { + st_component_t *owner = fr->owner; + ev->nframes--; + if (ev->state && owner) + st_finish(ev, owner); + continue; + } + const msb_rec_t *r = &f->recs[fr->rec++]; + msb_record(); + int at = fr->file; + if (r->tag == 'V' || r->tag == 'W') { + msb_set(ev, r->f[1], NULL); + } else if (r->tag == 'K') { + msb_set(ev, r->f[0], NULL); + } else if (r->tag == 'N' || r->tag == 'Y') { + ev->open = true; + } else if (r->tag == 'I' && !r->f[3]) { + int target = msb_import_target(ev, at, r->f[2]); + if (target >= 0) { + if (ev->state) + (void)st_enter(ev, target, true); + else if (!poisoned_has(ev, ev->m->files[target].rel_path) && msb_push(ev, target)) { + msb_work(SKIP_ONE); + cbm_ht_set(ev->poisoned, ev->m->files[target].rel_path, + (void *)&MSB_SET_PRESENT); + } + } + } + cbm_arena_reset(&ev->scratch); + } +} + +static void msb_add_item(msb_eval_t *ev, const msb_rec_t *group, const msb_rec_t *item, int file) { + if (ev->state) { + st_add_item(ev, group, item, file); + return; + } + if (ev->nitems >= ev->cap_items) { + int ncap = ev->cap_items ? ev->cap_items * PAIR_LEN : MSB_INIT; + msb_item_t *grown = (msb_item_t *)ev_alloc(ev, (size_t)ncap * sizeof(*grown)); + if (!grown) { + return; + } + if (ev->nitems > 0) { + memcpy(grown, ev->items, (size_t)ev->nitems * sizeof(*grown)); + msb_work((size_t)ev->nitems * sizeof(*grown)); + } + ev->items = grown; + ev->cap_items = ncap; + } + ev->items[ev->nitems++] = (msb_item_t){.group = group, .item = item, .file = file}; + msb_work(sizeof(msb_item_t)); +} + +/* An standing in the file on top of the frame stack. */ +static void msb_import(msb_eval_t *ev, int file, const msb_rec_t *r) { + msb_tri_t c = tri_and(msb_cond(ev, file, r->f[0]), msb_cond(ev, file, r->f[1])); + if (c == MSB_FALSE || r->f[3]) { + return; /* not taken; or an SDK's file, which is no file of the repository */ + } + int target = msb_import_target(ev, file, r->f[2]); + if (c == MSB_UNKNOWN) { + ev->unevaluable++; + if (target >= 0) { + msb_poison(ev, target); + } + return; + } + if (ev->state) { + if (target >= 0) + (void)st_enter(ev, target, false); + return; + } + if (target >= 0 && !seen_has(ev, ev->m->files[target].rel_path) && msb_push(ev, target)) { + msb_work(SKIP_ONE); + cbm_ht_set(ev->seen, ev->m->files[target].rel_path, (void *)&MSB_SET_PRESENT); + /* a file that was not read (malformed, or too large) says nothing */ + ev->unevaluable += !ev->m->files[target].readable; + } +} + +/* A property record under its group's condition `group`. */ +static void msb_property(msb_eval_t *ev, int file, const msb_rec_t *r, msb_tri_t group) { + msb_tri_t own = msb_cond(ev, file, r->f[0]); + msb_tri_t c = tri_and(group, own); + if (c == MSB_FALSE) { + return; + } + ev->unevaluable += own == MSB_UNKNOWN; + const char *value = + (c == MSB_TRUE && r->tag == 'V') ? msb_expand(ev, file, r->f[2] ? r->f[2] : "") : NULL; + msb_set(ev, r->f[1], value); +} + +/* Pass 1 over `file` and what it imports: properties in order, the + * items collected for pass 2. No recursion: an import chain is as long as + * the repository makes it. */ +static void msb_pass1(msb_eval_t *ev, int file) { + int base = ev->nframes; + const char *rel = ev->m->files[file].rel_path; + if (ev->state) { + if (!st_enter(ev, file, false)) + return; + } else { + if (seen_has(ev, rel) || !msb_push(ev, file)) + return; + msb_work(SKIP_ONE); + cbm_ht_set(ev->seen, rel, (void *)&MSB_SET_PRESENT); + ev->unevaluable += !ev->m->files[file].readable; + } + while (ev->nframes > base && !ev->oom) { + /* an import pushes a frame, which may move the array: the frame is + * read into locals before the record is handled */ + msb_frame_t *fr = &ev->frames[ev->nframes - SKIP_ONE]; + const msb_file_t *f = &ev->m->files[fr->file]; + if (fr->rec >= f->nrecs) { + st_component_t *owner = fr->owner; + ev->nframes--; + if (ev->state && owner) + st_finish(ev, owner); + continue; + } + const msb_rec_t *r = &f->recs[fr->rec++]; + msb_record(); + int at = fr->file; + switch (r->tag) { + case 'G': + fr->group = msb_cond(ev, at, r->f[0]); + ev->unevaluable += fr->group == MSB_UNKNOWN; + break; + case 'V': + case 'W': + msb_property(ev, at, r, fr->group); + break; + case 'K': + msb_set(ev, r->f[0], NULL); + break; + case 'H': + fr->hgroup = r; + break; + case 'N': + msb_add_item(ev, fr->hgroup, r, at); + break; + case 'Y': + ev->open = true; + ev->unevaluable++; + break; + case 'C': + ev->unevaluable++; + break; + case 'I': + msb_import(ev, at, r); + break; + default: + break; + } + cbm_arena_reset(&ev->scratch); + } +} + +/* ── Usings ──────────────────────────────────────────────────────── */ + +typedef struct { + cbm_msb_using_t *v; + int n; + int cap; + CBMHashTable *unique; /* capture only: exact tuples, or removal targets */ + bool target_only; +} msb_ulist_t; + +static void ulist_add(msb_eval_t *ev, msb_ulist_t *l, char kind, const char *alias, + const char *target) { + if (ev->oom) { + return; + } + if (l->n >= l->cap) { + if (l->cap > INT_MAX / PAIR_LEN) { + ev->oom = true; + return; + } + int ncap = l->cap ? l->cap * PAIR_LEN : MSB_INIT; + if ((size_t)ncap > SIZE_MAX / sizeof(cbm_msb_using_t)) { + ev->oom = true; + return; + } + cbm_msb_using_t *grown = (cbm_msb_using_t *)ev_alloc(ev, (size_t)ncap * sizeof(*grown)); + if (!grown) { + return; + } + if (l->n > 0) { + memcpy(grown, l->v, (size_t)l->n * sizeof(*grown)); + msb_work((size_t)l->n * sizeof(*grown)); + } + l->v = grown; + l->cap = ncap; + } + l->v[l->n++] = (cbm_msb_using_t){.kind = kind, .alias = alias, .target = target}; + msb_work(sizeof(cbm_msb_using_t)); +} + +/* The tuple key has a fixed-width hexadecimal target length, followed by + * the target and alias bytes. It cannot confuse embedded separators or a + * different target/alias boundary. Only a new tuple acquires owned strings. */ +static void ulist_add_unique(msb_eval_t *ev, msb_ulist_t *l, char kind, const char *alias, + size_t an, const char *target, size_t tn) { + size_t header = SKIP_ONE + PAIR_LEN * sizeof(size_t); + if (tn > SIZE_MAX - header - SKIP_ONE || an > SIZE_MAX - header - tn - SKIP_ONE) { + ev->oom = true; + return; + } + size_t bytes = l->target_only ? tn : header + tn + an; + char *key = (char *)scratch_alloc(ev, bytes + SKIP_ONE); + if (!key) { + return; + } + if (l->target_only) { + memcpy(key, target, tn); + } else { + static const char digits[] = "0123456789abcdef"; + key[0] = kind; + size_t length = tn; + for (size_t i = header; i > SKIP_ONE;) { + key[--i] = digits[length & 15]; + length >>= 4; + } + memcpy(key + header, target, tn); + memcpy(key + header + tn, alias, an); + } + key[bytes] = '\0'; + msb_work(bytes + PAIR_LEN); /* key construction and unique lookup */ + if (cbm_ht_get(l->unique, key)) { + return; + } + char *owned_key = ev_strndup(ev, key, bytes); + const char *owned_target = l->target_only ? owned_key : ev_strndup(ev, target, tn); + const char *owned_alias = an ? ev_strndup(ev, alias, an) : ""; + if (ev->oom) { + return; + } + msb_work(SKIP_ONE); + if (!msb_fail_item_operation(CBM_MSB_ITEM_FAIL_UNIQUE_INSERT)) { + msb_work(SKIP_ONE); + cbm_ht_set(l->unique, owned_key, owned_key); + } + if (cbm_ht_get(l->unique, owned_key) != owned_key) { + ev->oom = true; + return; + } + if (!l->target_only) { + ulist_add(ev, l, kind, owned_alias, owned_target); + } +} + +/* Add every ';'-separated entry. During capture, neither aliases nor targets + * are copied into persistent storage before exact duplicate detection. */ +static void ulist_add_split(msb_eval_t *ev, msb_ulist_t *l, char kind, const char *alias, size_t an, + const char *list) { + if (!l->unique && an > 0) { + alias = ev_strndup(ev, alias, an); + } + for (const char *p = list; !ev->oom && p;) { + const char *e = strchr(p, ';'); + size_t n = e ? (size_t)(e - p) : strlen(p); + const char *s = p; + trim_span(&s, &n); + if (n > 0) { + if (l->unique) { + ulist_add_unique(ev, l, kind, alias, an, s, n); + } else { + ulist_add(ev, l, kind, an ? alias : "", ev_strndup(ev, s, n)); + } + } + p = e ? e + SKIP_ONE : NULL; + } +} + +static int using_cmp(const void *a, const void *b) { + msb_work(SKIP_ONE); + const cbm_msb_using_t *x = (const cbm_msb_using_t *)a; + const cbm_msb_using_t *y = (const cbm_msb_using_t *)b; + if (x->kind != y->kind) { + return x->kind < y->kind ? -1 : 1; + } + int c = strcmp(x->target, y->target); + return c ? c : strcmp(x->alias, y->alias); +} + +static int target_cmp(const void *a, const void *b) { + msb_work(SKIP_ONE); + return strcmp(((const cbm_msb_using_t *)a)->target, ((const cbm_msb_using_t *)b)->target); +} + +/* One item with the final properties: into `inc` or `rem`. */ +static void msb_using(msb_eval_t *ev, const msb_item_t *it, msb_ulist_t *inc, msb_ulist_t *rem) { + const msb_rec_t *r = it->item; + msb_record(); + msb_tri_t c = tri_and(msb_cond(ev, it->file, it->group ? it->group->f[0] : NULL), + msb_cond(ev, it->file, r->f[0])); + if (c == MSB_FALSE) { + return; + } + const char *include = r->f[1] ? msb_expand(ev, it->file, r->f[1]) : ""; + const char *remove = r->f[2] ? msb_expand(ev, it->file, r->f[2]) : ""; + const char *is_static = r->f[3] ? msb_expand(ev, it->file, r->f[3]) : ""; + const char *alias = r->f[4] ? msb_expand(ev, it->file, r->f[4]) : ""; + if (c == MSB_UNKNOWN || !include || !remove || !is_static || !alias) { + ev->open = true; + ev->unevaluable++; + return; + } + size_t sn = strlen(is_static); + trim_span(&is_static, &sn); + size_t an = strlen(alias); + trim_span(&alias, &an); + char kind = an > 0 ? 'a' : (ci_is(is_static, sn, "true") ? 's' : 'n'); + ulist_add_split(ev, inc, kind, alias, an, include); + ulist_add_split(ev, rem, 'n', "", 0, remove); +} + +static const char *const SDK_DEFAULT[] = { + "System", "System.Collections.Generic", "System.IO", "System.Linq", "System.Net.Http", + "System.Threading", "System.Threading.Tasks", NULL}; +static const char *const SDK_WEB[] = {"System", + "System.Collections.Generic", + "System.IO", + "System.Linq", + "System.Net.Http", + "System.Net.Http.Json", + "System.Threading", + "System.Threading.Tasks", + "Microsoft.AspNetCore.Builder", + "Microsoft.AspNetCore.Hosting", + "Microsoft.AspNetCore.Http", + "Microsoft.AspNetCore.Routing", + "Microsoft.Extensions.Configuration", + "Microsoft.Extensions.DependencyInjection", + "Microsoft.Extensions.Hosting", + "Microsoft.Extensions.Logging", + NULL}; +static const char *const SDK_WORKER[] = {"System", + "System.Collections.Generic", + "System.IO", + "System.Linq", + "System.Net.Http", + "System.Threading", + "System.Threading.Tasks", + "Microsoft.Extensions.Configuration", + "Microsoft.Extensions.DependencyInjection", + "Microsoft.Extensions.Hosting", + "Microsoft.Extensions.Logging", + NULL}; + +/* True when the ';'-separated SDK list names `sdk` (a version after '/' is + * no part of the name). */ +static bool sdk_listed(const char *list, const char *sdk) { + size_t sl = strlen(sdk); + for (const char *p = list; p;) { + const char *e = strchr(p, ';'); + size_t n = e ? (size_t)(e - p) : strlen(p); + const char *s = p; + trim_span(&s, &n); + const char *slash = memchr(s, '/', n); + size_t name = slash ? (size_t)(slash - s) : n; + if (name == sl && memcmp(s, sdk, sl) == 0) { + return true; + } + p = e ? e + SKIP_ONE : NULL; + } + return false; +} + +/* The SDK a project file names: its attribute, else the first + * import that names one. NULL when it names none. */ +static const char *msb_sdk(const msb_file_t *f) { + const char *sdk = f->sdk; + for (int i = 0; !sdk && i < f->nrecs; i++) { + sdk = f->recs[i].tag == 'I' ? f->recs[i].f[3] : NULL; + } + return sdk; +} + +bool cbm_msb_compiles(const cbm_msb_t *m, const char *rel_path) { + /* These two SDKs are defined as producing no assembly (one runs build + * steps, the other builds other projects): that is why they are named. */ + static const char *const idle[] = {"Microsoft.Build.NoTargets", "Microsoft.Build.Traversal"}; + int at = (m && rel_path) ? msb_file_index(m, rel_path) : CBM_NOT_FOUND; + const char *sdk = at >= 0 ? msb_sdk(&m->files[at]) : NULL; + for (size_t i = 0; sdk && i < sizeof(idle) / sizeof(idle[0]); i++) { + if (sdk_listed(sdk, idle[i])) { + return false; + } + } + return true; +} + +/* The usings ImplicitUsings brings in, or NULL. An ImplicitUsings that some + * file sets to a value this reader does not know leaves the usings open. */ +static const char *const *msb_implicit(msb_eval_t *ev, int project) { + static const char name[] = "ImplicitUsings"; + char key[MSB_NAME_MAX]; + (void)msb_key(name, sizeof(name) - SKIP_ONE, key); + const msb_prop_t *p = prop_lookup(ev, key); + if (!p) { + return NULL; + } + msb_tri_t on = + p->value ? tri_or(msb_equal(p->value, "enable"), msb_equal(p->value, "true")) : MSB_UNKNOWN; + if (on == MSB_UNKNOWN) { + ev->open = true; + ev->unevaluable++; + } + if (on != MSB_TRUE) { + return NULL; + } + const char *sdk = msb_sdk(&ev->m->files[project]); + if (sdk && sdk_listed(sdk, "Microsoft.NET.Sdk.Web")) { + return SDK_WEB; + } + return (sdk && sdk_listed(sdk, "Microsoft.NET.Sdk.Worker")) ? SDK_WORKER : SDK_DEFAULT; +} + +/* The evaluation's usings as one heap block: the array, then its strings. */ +static bool msb_result(msb_eval_t *ev, msb_ulist_t *inc, const msb_ulist_t *rem, + cbm_msb_result_t *out) { + if (rem->n > 0) { + qsort(rem->v, (size_t)rem->n, sizeof(rem->v[0]), target_cmp); + } + if (inc->n > 0) { + qsort(inc->v, (size_t)inc->n, sizeof(inc->v[0]), using_cmp); + } + int kept = 0; + size_t bytes = 0; + for (int i = 0; i < inc->n; i++) { + const cbm_msb_using_t *u = &inc->v[i]; + bool removed = + rem->n > 0 && bsearch(u, rem->v, (size_t)rem->n, sizeof(rem->v[0]), target_cmp) != NULL; + if (!removed && ev->shared_removals) { + msb_work(SKIP_ONE); + removed = cbm_ht_get(ev->shared_removals, u->target) != NULL; + } + bool dup = kept > 0 && using_cmp(&inc->v[kept - SKIP_ONE], u) == 0; + if (removed || dup) { + continue; + } + inc->v[kept++] = *u; + msb_work(sizeof(*u)); + bytes += strlen(u->alias) + strlen(u->target) + PAIR_LEN; + } + out->open = ev->open; + out->unevaluable = ev->unevaluable; + out->outside = ev->outside; + if (kept == 0) { + return !ev->oom; + } + size_t head = (size_t)kept * sizeof(cbm_msb_using_t); + char *mem = msb_fail_item_operation(CBM_MSB_ITEM_FAIL_PUBLISH_ALLOC) + ? NULL + : (char *)cbm_alloc(CBM_MEM_CLASS_OTHER, head + bytes); + if (!mem) { + return false; + } + ev->result_bytes = head + bytes; + ev_peak(ev); + cbm_msb_using_t *arr = (cbm_msb_using_t *)(void *)mem; + char *w = mem + head; + for (int i = 0; i < kept; i++) { + size_t al = strlen(inc->v[i].alias) + SKIP_ONE; + size_t tl = strlen(inc->v[i].target) + SKIP_ONE; + memcpy(w, inc->v[i].alias, al); + memcpy(w + al, inc->v[i].target, tl); + arr[i] = (cbm_msb_using_t){.kind = inc->v[i].kind, .alias = w, .target = w + al}; + msb_work(al + tl + sizeof(arr[i])); + w += al + tl; + } + out->usings = arr; + out->count = kept; + out->mem = mem; + return !ev->oom; +} + +/* The properties MSBuild sets before it reads the project file. */ +static void msb_seed(msb_eval_t *ev, int project) { + const msb_file_t *f = &ev->m->files[project]; + msb_set(ev, "MSBuildProjectName", f->stem); + msb_set(ev, "MSBuildProjectFile", f->name); + msb_set(ev, "MSBuildProjectExtension", f->ext); + msb_set(ev, "MSBuildProjectFullPath", f->abs_path); + /* the directory without its trailing '/' */ + size_t dl = strlen(f->abs_dir); + char *dir = scratch_strndup(ev, f->abs_dir, dl > SKIP_ONE ? dl - SKIP_ONE : dl); + if (dir) { + msb_set(ev, "MSBuildProjectDirectory", dir); + } +} + +static bool eval_init(msb_eval_t *ev) { + cbm_arena_init(&ev->arena); + cbm_arena_init(&ev->scratch); + ev_peak(ev); + ev->props = cbm_ht_create(CBM_SZ_64); + ev->seen = cbm_ht_create(MSB_INIT); + ev->poisoned = cbm_ht_create(MSB_INIT); + ev->oom = !ev->props || !ev->seen || !ev->poisoned; + return !ev->oom; +} + +static void eval_clear(msb_eval_t *ev) { + if (ev->state) + st_view_drop(ev->state, &ev->state_inputs); + if (ev->props) { + cbm_ht_foreach(ev->props, prop_clear, ev); + } + cbm_ht_free(ev->props); + cbm_ht_free(ev->seen); + cbm_ht_free(ev->poisoned); + cbm_ht_free(ev->inputs.props); + cbm_ht_free(ev->inputs.seen); + cbm_ht_free(ev->inputs.poisoned); + cbm_arena_destroy(&ev->arena); + cbm_arena_destroy(&ev->scratch); +} + +bool cbm_msb_eval(const cbm_msb_t *m, const char *project_rel, cbm_msb_result_t *out) { + memset(out, 0, sizeof(*out)); + int project = (m && project_rel) ? msb_file_index(m, project_rel) : CBM_NOT_FOUND; + if (project < 0) { + return false; + } + msb_eval_t ev = {.m = m}; + cbm_arena_init(&ev.arena); + cbm_arena_init(&ev.scratch); + ev_peak(&ev); + ev.props = cbm_ht_create(CBM_SZ_64); + ev.seen = cbm_ht_create(MSB_INIT); + ev.poisoned = cbm_ht_create(MSB_INIT); + bool ok = ev.props && ev.seen && ev.poisoned; + if (ok) { + const msb_file_t *f = &m->files[project]; + msb_seed(&ev, project); + cbm_arena_reset(&ev.scratch); + int props = msb_nearest(&ev, f->dir, "Directory.Build.props"); + cbm_arena_reset(&ev.scratch); + if (props >= 0) { + msb_pass1(&ev, props); + } + msb_pass1(&ev, project); + int targets = msb_nearest(&ev, f->dir, "Directory.Build.targets"); + cbm_arena_reset(&ev.scratch); + if (targets >= 0) { + msb_pass1(&ev, targets); + } + msb_ulist_t inc = {0}; + msb_ulist_t rem = {0}; + const char *const *implicit = msb_implicit(&ev, project); + for (int i = 0; implicit && implicit[i]; i++) { + ulist_add(&ev, &inc, 'n', "", implicit[i]); + } + for (int i = 0; i < ev.nitems; i++) { + msb_using(&ev, &ev.items[i], &inc, &rem); + /* All four expansions remain live until aliases and targets + * have been copied into the persistent evaluation arena. */ + cbm_arena_reset(&ev.scratch); + } + ok = !ev.oom && msb_result(&ev, &inc, &rem, out); + } + eval_clear(&ev); + if (!ok) { + cbm_msb_result_free(out); + } + return ok; +} + +/* One immutable prefix and one exact target effect. The target effect's + * dependency keys are indexed, so validation walks CURRENT local overrides, + * not every property or imported file in a large shared closure. */ +typedef struct { + bool known[PAIR_LEN]; + int file[PAIR_LEN]; +} msb_nearest_entry_t; + +struct cbm_msb_eval_context { + const cbm_msb_t *m; + msb_state_t *state; + uint64_t generation; + int root; + bool ready; + msb_eval_t prefix; + int target_root; + const msb_eval_t *target_base; + bool target_live; + bool target_ready; + msb_eval_t target; + bool items_live; + bool items_ready; + msb_eval_t item_owner; /* lazy arena/dependencies; no property or import state */ + msb_ulist_t item_includes; + msb_ulist_t item_removals; + CBMHashTable *nearest; /* immutable model-owned directory -> two nearest-file results */ + size_t nearest_bytes; +}; + +static void nearest_entry_free(const char *key, void *value, void *userdata) { + (void)key; + cbm_msb_eval_context_t *context = (cbm_msb_eval_context_t *)userdata; + cbm_free(CBM_MEM_CLASS_OTHER, value); + context->nearest_bytes -= sizeof(msb_nearest_entry_t); + msb_work(SKIP_ONE); +} + +static void context_drop_items(cbm_msb_eval_context_t *context) { + if (context->items_live) { + msb_work(SKIP_ONE + context->item_owner.arena.nblocks + + context->item_owner.scratch.nblocks); + cbm_ht_free(context->item_includes.unique); + cbm_ht_free(context->item_removals.unique); + eval_clear(&context->item_owner); + memset(&context->item_owner, 0, sizeof(context->item_owner)); + context->item_includes = (msb_ulist_t){0}; + context->item_removals = (msb_ulist_t){0}; + msb_work(sizeof(context->item_owner) + sizeof(msb_ulist_t) * PAIR_LEN); + context->items_live = false; + context->items_ready = false; + } +} + +static void context_drop_target(cbm_msb_eval_context_t *context) { + if (context->target_live) { + context_drop_items(context); + eval_clear(&context->target); + memset(&context->target, 0, sizeof(context->target)); + msb_work(sizeof(context->target)); + context->target_live = false; + context->target_ready = false; + context->target_base = NULL; + } +} + +static void context_drop_prefix(cbm_msb_eval_context_t *context) { + if (context->ready) { + context_drop_items(context); + /* Target dependencies can borrow the immutable prefix identity. */ + context_drop_target(context); + eval_clear(&context->prefix); + memset(&context->prefix, 0, sizeof(context->prefix)); + msb_work(sizeof(context->prefix)); + context->ready = false; + } +} + +static void context_clear_nearest(cbm_msb_eval_context_t *context) { + cbm_ht_foreach(context->nearest, nearest_entry_free, context); + cbm_ht_clear(context->nearest); + msb_work(SKIP_ONE); +} + +static msb_state_t *st_state_new(const cbm_msb_t *m); +static void st_state_free(msb_state_t *s); +static bool st_context_eval(cbm_msb_eval_context_t *context, const char *project_rel, + cbm_msb_result_t *out); + +cbm_msb_eval_context_t *cbm_msb_eval_context_new(const cbm_msb_t *m) { + if (!m) { + return NULL; + } + cbm_msb_eval_context_t *context = + (cbm_msb_eval_context_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*context)); + if (!context) { + return NULL; + } + msb_work(sizeof(*context)); + context->m = m; + context->generation = m->generation; + context->root = CBM_NOT_FOUND; + context->target_root = CBM_NOT_FOUND; + context->nearest = cbm_ht_create(MSB_INIT); + context->state = st_state_new(m); + if (!context->nearest || !context->state) { + st_state_free(context->state); + cbm_ht_free(context->nearest); + cbm_free(CBM_MEM_CLASS_OTHER, context); + return NULL; + } + return context; +} + +void cbm_msb_eval_context_free(cbm_msb_eval_context_t *context) { + if (context) { + context_drop_items(context); + context_drop_target(context); + context_drop_prefix(context); + st_state_free(context->state); + cbm_ht_foreach(context->nearest, nearest_entry_free, context); + cbm_ht_free(context->nearest); + cbm_free(CBM_MEM_CLASS_OTHER, context); + } +} + +/* Cache metadata only. Directory keys live in the model, and an absent + * result is as reusable as a present one until the model generation changes. */ +static int context_nearest(cbm_msb_eval_context_t *context, msb_eval_t *ev, const char *dir, + bool targets) { + msb_work(SKIP_ONE); + msb_nearest_entry_t *entry = (msb_nearest_entry_t *)cbm_ht_get(context->nearest, dir); + if (!entry) { + if (sizeof(*entry) > SIZE_MAX - context->nearest_bytes) { + ev->oom = true; + return CBM_NOT_FOUND; + } + entry = (msb_nearest_entry_t *)cbm_alloc(CBM_MEM_CLASS_OTHER, sizeof(*entry)); + if (!entry) { + ev->oom = true; + return CBM_NOT_FOUND; + } + context->nearest_bytes += sizeof(*entry); + ev_peak(ev); + memset(entry, 0, sizeof(*entry)); + msb_work(sizeof(*entry) + PAIR_LEN); + cbm_ht_set(context->nearest, dir, entry); + if (cbm_ht_get(context->nearest, dir) != entry) { + nearest_entry_free(dir, entry, context); + ev->oom = true; + return CBM_NOT_FOUND; + } + } + int which = targets ? SKIP_ONE : 0; + msb_work(SKIP_ONE); + if (!entry->known[which]) { + int found = + msb_nearest(ev, dir, targets ? "Directory.Build.targets" : "Directory.Build.props"); + cbm_arena_reset(&ev->scratch); + if (ev->oom) { + return CBM_NOT_FOUND; + } + entry->file[which] = found; + entry->known[which] = true; + } + return entry->file[which]; +} + +static void attach_prefix(msb_eval_t *ev, const msb_eval_t *prefix) { + ev->base = prefix; + ev->open = prefix->open; + ev->unevaluable = prefix->unevaluable; + ev->outside = prefix->outside; + msb_work(SKIP_ONE); + ev_peak(ev); +} + +typedef struct { + const msb_inputs_t *inputs; + const msb_eval_t *effects; /* final-item inputs: target writes hide project entries */ + size_t props; + size_t seen; + size_t poisoned; + bool match; +} msb_input_check_t; + +static void check_prop_input(const char *key, void *value, void *userdata) { + msb_input_check_t *check = (msb_input_check_t *)userdata; + msb_work(SKIP_ONE); /* every CURRENT local entry visited, including irrelevant ones */ + if (!check->match) { + return; + } + if (check->effects) { + msb_work(SKIP_ONE); + if (cbm_ht_get(check->effects->props, key)) { + return; + } + } + msb_work(SKIP_ONE); + const msb_prop_dep_t *dep = (const msb_prop_dep_t *)cbm_ht_get(check->inputs->props, key); + if (dep) { + check->match = prop_dep_matches(dep, (const msb_prop_t *)value); + check->props += check->match && dep->exception; + } +} + +static void check_set_input(const char *key, msb_input_check_t *check, bool poison) { + msb_work(SKIP_ONE); + if (!check->match) { + return; + } + msb_work(SKIP_ONE); + const msb_set_dep_t *dep = (const msb_set_dep_t *)cbm_ht_get( + poison ? check->inputs->poisoned : check->inputs->seen, key); + if (dep) { + check->match = dep->present; + if (poison) { + check->poisoned += check->match && dep->exception; + } else { + check->seen += check->match && dep->exception; + } + } +} + +static void check_seen_input(const char *key, void *value, void *userdata) { + (void)value; + check_set_input(key, (msb_input_check_t *)userdata, false); +} + +static void check_poisoned_input(const char *key, void *value, void *userdata) { + (void)value; + check_set_input(key, (msb_input_check_t *)userdata, true); +} + +static bool seed_inputs_match(const msb_inputs_t *inputs, msb_eval_t *incoming) { + for (int i = 0; i < MSB_SEEDS; i++) { + msb_work(SKIP_ONE); + const msb_prop_dep_t *dep = inputs->seeds[i]; + if (dep && !prop_dep_matches(dep, prop_lookup(incoming, MSB_SEED_NAMES[i]))) { + return false; + } + } + return true; +} + +/* A dependency equal to the immutable baseline needs no local override. + * Every other dependency needs a matching CURRENT local entry. Counting + * matched exceptions catches missing inputs without scanning cached maps. */ +static bool target_inputs_match(const msb_eval_t *target, msb_eval_t *incoming) { + msb_input_check_t check = {.inputs = &target->inputs, .match = true}; + cbm_ht_foreach(incoming->props, check_prop_input, &check); + cbm_ht_foreach(incoming->seen, check_seen_input, &check); + cbm_ht_foreach(incoming->poisoned, check_poisoned_input, &check); + if (!check.match || check.props != target->inputs.prop_exceptions || + check.seen != target->inputs.seen_exceptions || + check.poisoned != target->inputs.poisoned_exceptions) { + return false; + } + return seed_inputs_match(&target->inputs, incoming); +} + +static bool apply_targets(cbm_msb_eval_context_t *context, msb_eval_t *ev, int root) { + if (root < 0) { + context_drop_target(context); + return !ev->oom; + } + msb_work(SKIP_ONE); + bool hit = context->target_ready && context->target_root == root && + context->target_base == ev->base && target_inputs_match(&context->target, ev); + if (!hit) { + context_drop_items(context); + context_drop_target(context); + msb_eval_t *target = &context->target; + target->m = ev->m; + target->incoming = ev; + target->live = ev->live; + target->capture_inputs = true; + context->target_live = true; + bool ok = eval_init(target); + target->inputs.props = cbm_ht_create(MSB_INIT); + target->inputs.seen = cbm_ht_create(MSB_INIT); + target->inputs.poisoned = cbm_ht_create(MSB_INIT); + target->oom = target->oom || !target->inputs.props || !target->inputs.seen || + !target->inputs.poisoned; + if (ok && !target->oom) { + msb_pass1(target, root); + } + if (target->oom || target->nframes) { + ev->oom = true; + context_drop_target(context); + return false; + } + target->incoming = NULL; + target->live = NULL; + target->capture_inputs = false; + context->target_root = root; + context->target_base = ev->base; + context->target_ready = true; + } + ev->effects = &context->target; + ev->open = ev->open || context->target.open; + ev->unevaluable += context->target.unevaluable; + ev->outside += context->target.outside; + msb_work(SKIP_ONE); + ev_peak(ev); + return true; +} + +static void eval_items(msb_eval_t *ev, const msb_eval_t *owner, msb_ulist_t *inc, + msb_ulist_t *rem) { + for (int i = 0; !ev->oom && i < owner->nitems; i++) { + msb_using(ev, &owner->items[i], inc, rem); + cbm_arena_reset(&ev->scratch); + } +} + +/* Item dependencies see final properties. A present target value, even + * unknown, hides a project override. Matching exception counts detect inputs + * missing from CURRENT locals without walking the retained dependency map. */ +static bool item_inputs_match(const msb_eval_t *items, msb_eval_t *incoming) { + if (incoming->state) + return st_inputs_match(incoming, &items->state_inputs); + msb_input_check_t check = { + .inputs = &items->inputs, .effects = incoming->effects, .match = true}; + cbm_ht_foreach(incoming->props, check_prop_input, &check); + return check.match && check.props == items->inputs.prop_exceptions && + seed_inputs_match(&items->inputs, incoming); +} + +static void st_eval_rope(msb_eval_t *ev, st_rope_t *rope, msb_ulist_t *inc, msb_ulist_t *rem); + +static bool capture_items(cbm_msb_eval_context_t *context, msb_eval_t *ev) { + msb_eval_t *items = &context->item_owner; + items->m = ev->m; + items->incoming = ev; + items->live = ev->live; + items->capture_inputs = true; + if (ev->state) { + items->state = ev->state; + items->view = ev->view; + } + items->item_alloc_operation = CBM_MSB_ITEM_FAIL_CAPTURE_ALLOC; + context->items_live = true; + cbm_arena_init_lazy(&items->arena, CBM_ARENA_DEFAULT_BLOCK_SIZE); + cbm_arena_init_lazy(&items->scratch, CBM_ARENA_DEFAULT_BLOCK_SIZE); + items->inputs.props = cbm_ht_create(MSB_INIT); + context->item_includes.unique = cbm_ht_create(MSB_INIT); + context->item_removals.unique = cbm_ht_create(MSB_INIT); + context->item_removals.target_only = true; + items->oom = + !items->inputs.props || !context->item_includes.unique || !context->item_removals.unique; + msb_work(sizeof(*items)); + if (ev->state) { + st_eval_rope(items, ev->state->capture_a, &context->item_includes, &context->item_removals); + st_eval_rope(items, ev->state->capture_b, &context->item_includes, &context->item_removals); + } else { + if (!items->oom && ev->base) + eval_items(items, ev->base, &context->item_includes, &context->item_removals); + if (!items->oom && ev->effects) + eval_items(items, ev->effects, &context->item_includes, &context->item_removals); + } + if (items->oom) { + ev->oom = true; + context_drop_items(context); + return false; + } + /* Shared removals apply globally. Prune shared includes once, but retain + * the removal index for project and implicit includes at publication. */ + msb_ulist_t *inc = &context->item_includes; + int kept = 0; + for (int i = 0; i < inc->n; i++) { + msb_work(SKIP_ONE); + if (!cbm_ht_get(context->item_removals.unique, inc->v[i].target)) { + inc->v[kept++] = inc->v[i]; + msb_work(sizeof(inc->v[i])); + } + } + inc->n = kept; + cbm_ht_free(inc->unique); + inc->unique = NULL; + msb_work(SKIP_ONE + items->scratch.nblocks); + cbm_arena_destroy(&items->scratch); + items->incoming = NULL; + items->view = (msb_view_t){0}; + items->live = NULL; + items->capture_inputs = false; + items->item_alloc_operation = CBM_MSB_ITEM_FAIL_NONE; + context->items_ready = true; + return true; +} + +static bool apply_items(cbm_msb_eval_context_t *context, msb_eval_t *ev, msb_ulist_t *inc, + msb_ulist_t *rem) { + if (ev->state) { + if (!ev->state->capture_a && !ev->state->capture_b) + return !ev->oom; + } else if ((!ev->base || !ev->base->nitems) && (!ev->effects || !ev->effects->nitems)) { + return !ev->oom; + } + msb_work(SKIP_ONE); + if (context->items_ready && !item_inputs_match(&context->item_owner, ev)) { + /* Keep the first exact variant for this owner epoch. A different + * project's final inputs do not churn retained item storage. */ + if (ev->state) { + st_eval_rope(ev, ev->state->capture_a, inc, rem); + st_eval_rope(ev, ev->state->capture_b, inc, rem); + } else { + if (ev->base) + eval_items(ev, ev->base, inc, rem); + if (ev->effects) + eval_items(ev, ev->effects, inc, rem); + } + return !ev->oom; + } + if (!context->items_ready && !capture_items(context, ev)) { + return false; + } + const msb_ulist_t *shared = &context->item_includes; + if (shared->n > 0) { + if (inc->n > INT_MAX - shared->n || + (size_t)(inc->n + shared->n) > SIZE_MAX / sizeof(*inc->v)) { + ev->oom = true; + return false; + } + int n = inc->n + shared->n; + cbm_msb_item_fail_operation_t previous = ev->item_alloc_operation; + ev->item_alloc_operation = CBM_MSB_ITEM_FAIL_APPLY_ALLOC; + cbm_msb_using_t *v = (cbm_msb_using_t *)ev_alloc(ev, (size_t)n * sizeof(*v)); + ev->item_alloc_operation = previous; + if (!v) { + return false; + } + if (inc->n) { + memcpy(v, inc->v, (size_t)inc->n * sizeof(*v)); + } + memcpy(v + inc->n, shared->v, (size_t)shared->n * sizeof(*v)); + msb_work((size_t)n * sizeof(*v)); + inc->v = v; + inc->n = n; + inc->cap = n; + } + ev->shared_removals = context->item_removals.unique; + ev->open = ev->open || context->item_owner.open; + ev->unevaluable += context->item_owner.unevaluable; + ev->outside += context->item_owner.outside; + msb_work(SKIP_ONE); + ev_peak(ev); + return !ev->oom; +} + +bool cbm_msb_eval_context_eval(cbm_msb_eval_context_t *context, const char *project_rel, + cbm_msb_result_t *out) { + if (context && context->state) + return st_context_eval(context, project_rel, out); + memset(out, 0, sizeof(*out)); + if (!context) { + return false; + } + const cbm_msb_t *m = context->m; + msb_work(SKIP_ONE); + if (context->generation != m->generation) { + context_drop_items(context); + context_drop_target(context); + context_drop_prefix(context); + context_clear_nearest(context); + context->generation = m->generation; + } + int project = project_rel ? msb_file_index(m, project_rel) : CBM_NOT_FOUND; + if (project < 0) { + return false; + } + msb_eval_t seeds = {.m = m}; + msb_eval_t ev = {.m = m, .seeds = &seeds, .base = context->ready ? &context->prefix : NULL}; + msb_live_t live = { + .owners = {&context->prefix, &context->target, &ev, &seeds, &context->item_owner}, + .metadata_bytes = &context->nearest_bytes}; + ev.live = &live; + seeds.live = &live; + bool ok = eval_init(&ev); + bool seeds_ok = eval_init(&seeds); + if (ok && seeds_ok) { + msb_seed(&seeds, project); + cbm_arena_reset(&seeds.scratch); + ok = !seeds.oom; + } else { + ok = false; + } + if (ok) { + const msb_file_t *f = &m->files[project]; + int props = context_nearest(context, &ev, f->dir, false); + msb_work(SKIP_ONE); + if (context->ready && context->root == props) { + attach_prefix(&ev, &context->prefix); + } else { + ev.base = NULL; + /* No retained prefix means a stable empty baseline. In + * particular, no-props calls must not discard target effects. */ + context_drop_prefix(context); + ev.track_seed_reads = true; + if (props >= 0) { + msb_pass1(&ev, props); + } + ev.track_seed_reads = false; + if (props >= 0 && !ev.oom && !ev.seed_read && ev.nframes == 0) { + context_drop_items(context); + context_drop_target(context); + context->prefix = ev; + msb_work(sizeof(ev)); + context->prefix.seeds = NULL; + context->prefix.base = NULL; + context->prefix.live = NULL; + context->root = props; + context->ready = true; + memset(&ev, 0, sizeof(ev)); + msb_work(sizeof(ev)); + ev.m = m; + ev.seeds = &seeds; + ev.base = &context->prefix; + ev.live = &live; + ok = eval_init(&ev); + attach_prefix(&ev, &context->prefix); + } + } + if (ok && !ev.oom) { + msb_pass1(&ev, project); + int targets = context_nearest(context, &ev, f->dir, true); + ok = !ev.oom && apply_targets(context, &ev, targets); + if (ok) { + msb_ulist_t inc = {0}; + msb_ulist_t rem = {0}; + const char *const *implicit = msb_implicit(&ev, project); + for (int i = 0; !ev.oom && implicit && implicit[i]; i++) { + ulist_add(&ev, &inc, 'n', "", implicit[i]); + } + ok = !ev.oom && apply_items(context, &ev, &inc, &rem); + if (ok) { + eval_items(&ev, &ev, &inc, &rem); + ok = !ev.oom && msb_result(&ev, &inc, &rem, out); + } + } + } else { + ok = false; + } + } + eval_clear(&ev); + eval_clear(&seeds); + if (!ok) { + cbm_msb_result_free(out); + } + return ok; +} + +/* Traverse source spans only. Empty chains collapse at construction; the + * bounded explicit stack follows the balanced rope, not the import graph. */ +static void st_eval_rope(msb_eval_t *ev, st_rope_t *rope, msb_ulist_t *inc, msb_ulist_t *rem) { + st_rope_t *stack[CBM_SZ_64]; + int depth = 0; + while (!ev->oom && (rope || depth)) { + if (!rope) { + rope = stack[--depth]; + continue; + } + msb_work(1); + if (!rope->left) { + msb_using(ev, &rope->item, inc, rem); + cbm_arena_reset(&ev->scratch); + rope = NULL; + } else { + if (depth == CBM_SZ_64) { + ev->oom = true; + break; + } + stack[depth++] = rope->right; + rope = rope->left; + } + } +} +static void st_trim(msb_state_t *s) { + while (s->empty) { + st_slab_t *b = s->empty; + st_empty_remove(s, b); + st_available_remove(s, b); + if (b->prev) + b->prev->next = b->next; + else + s->slabs = b->next; + if (b->next) + b->next->prev = b->prev; + s->bytes -= sizeof(*b); + cbm_free(CBM_MEM_CLASS_OTHER, b); + msb_work(1); + } +} +static void st_state_clear(msb_state_t *s) { + st_rope_drop(s, s->item_a); + st_rope_drop(s, s->item_b); + s->item_a = s->item_b = s->capture_a = s->capture_b = NULL; + for (int i = 0; i < s->files; i++) + for (int mode = 0; mode < 2; mode++) { + st_component_drop(s, s->components[(size_t)i * 2 + mode]); + msb_work(1); + } + s->bytes -= (size_t)s->files * 2 * sizeof(*s->components); + cbm_free(CBM_MEM_CLASS_OTHER, s->components); + s->components = NULL; + s->files = 0; + st_trim(s); + cbm_ht_free(s->symbols); + s->symbols = NULL; + cbm_arena_destroy(&s->names); + s->next_symbol = 0; +} +static bool st_state_epoch(msb_state_t *s, const cbm_msb_t *m) { + if (s->epoch == UINT64_MAX || (size_t)m->nfiles > SIZE_MAX / 2 / sizeof(*s->components)) + return false; + s->epoch++; + s->symbols = cbm_ht_create(CBM_SZ_64); + cbm_arena_init_lazy(&s->names, CBM_ARENA_APPEND_BLOCK); + s->components = (st_component_t **)cbm_calloc(CBM_MEM_CLASS_OTHER, + (size_t)m->nfiles * 2 * sizeof(*s->components)); + if (!s->symbols || (!s->components && m->nfiles)) + return false; + s->generation = m->generation; + s->files = m->nfiles; + s->bytes += (size_t)s->files * 2 * sizeof(*s->components); + msb_work((size_t)s->files * 2 * sizeof(*s->components)); + return true; +} +static msb_state_t *st_state_new(const cbm_msb_t *m) { + msb_state_t *s = (msb_state_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*s)); + if (!s) + return NULL; + s->bytes = sizeof(*s); + msb_work(sizeof(*s)); + if (!st_state_epoch(s, m)) { + st_state_free(s); + return NULL; + } + return s; +} +static void st_state_free(msb_state_t *s) { + if (!s) + return; + st_state_clear(s); + cbm_free(CBM_MEM_CLASS_OTHER, s); +} +static st_component_t *st_run(msb_eval_t *ev, int file) { + ev->completed = NULL; + if (file >= 0 && !ev->oom) + msb_pass1(ev, file); + st_component_t *out = ev->completed; + ev->completed = NULL; + return out; +} +static bool st_context_eval(cbm_msb_eval_context_t *context, const char *project_rel, + cbm_msb_result_t *out) { + memset(out, 0, sizeof(*out)); + const cbm_msb_t *m = context->m; + msb_state_t *s = context->state; + if (s->generation != m->generation) { + context_drop_items(context); + st_state_clear(s); + context_clear_nearest(context); + if (!st_state_epoch(s, m)) { + st_state_clear(s); + return false; + } + } + int project = project_rel ? msb_file_index(m, project_rel) : CBM_NOT_FOUND; + if (project < 0 || !s->symbols || !s->components) + return false; + msb_eval_t ev = {.m = m, .state = s, .main_project = project}; + msb_live_t live = {.owners = {&ev, &context->item_owner}, + .metadata_bytes = &context->nearest_bytes, + .state_bytes = &s->bytes, + .state_names = &s->names}; + ev.live = &live; + /* One call-local arena pair; components never acquire an arena pair. */ + cbm_arena_init(&ev.arena); + cbm_arena_init(&ev.scratch); + ev_peak(&ev); + msb_seed(&ev, project); + cbm_arena_reset(&ev.scratch); + int props = + !ev.oom ? context_nearest(context, &ev, m->files[project].dir, false) : CBM_NOT_FOUND; + st_component_t *prefix = st_run(&ev, props); + st_component_t *local = st_run(&ev, project); + int targets = + !ev.oom ? context_nearest(context, &ev, m->files[project].dir, true) : CBM_NOT_FOUND; + st_component_t *target = st_run(&ev, targets); + bool ok = false; + if (!ev.oom) { + st_rope_t *a = prefix ? prefix->items : NULL, *b = target ? target->items : NULL; + if (s->item_a != a || s->item_b != b) { + context_drop_items(context); + st_rope_drop(s, s->item_a); + st_rope_drop(s, s->item_b); + s->item_a = st_rope_hold(a); + s->item_b = st_rope_hold(b); + } + s->capture_a = a; + s->capture_b = b; + msb_ulist_t inc = {0}, rem = {0}; + const char *const *implicit = msb_implicit(&ev, project); + for (int i = 0; !ev.oom && implicit && implicit[i]; i++) + ulist_add(&ev, &inc, 'n', "", implicit[i]); + ok = !ev.oom && apply_items(context, &ev, &inc, &rem); + if (ok) { + st_eval_rope(&ev, local ? local->items : NULL, &inc, &rem); + ok = !ev.oom && msb_result(&ev, &inc, &rem, out); + } + s->capture_a = s->capture_b = NULL; + } + while (ev.builder) { + st_component_t *c = ev.builder; + ev.builder = c->parent; + c->parent = NULL; + st_component_drop(s, c); + } + st_component_drop(s, prefix); + st_component_drop(s, local); + st_component_drop(s, target); + st_component_drop(s, ev.completed); + st_view_drop(s, &ev.view); + eval_clear(&ev); + st_trim(s); + if (!ok) + cbm_msb_result_free(out); + return ok; +} + +void cbm_msb_result_free(cbm_msb_result_t *r) { + if (!r) { + return; + } + cbm_free(CBM_MEM_CLASS_OTHER, r->mem); + memset(r, 0, sizeof(*r)); +} diff --git a/src/pipeline/doc_links_msbuild.h b/src/pipeline/doc_links_msbuild.h new file mode 100644 index 0000000000..bdb618976a --- /dev/null +++ b/src/pipeline/doc_links_msbuild.h @@ -0,0 +1,146 @@ +/* + * doc_links_msbuild.h — MSBuild global usings of a C# project (R1), read from + * the scope blobs of the repository's project files. Private to the C# leg + * (doc_links_cs.c); no file is opened here. + */ +#ifndef CBM_PIPELINE_DOC_LINKS_MSBUILD_H +#define CBM_PIPELINE_DOC_LINKS_MSBUILD_H + +#include +#include + +/* The project files of one repository. */ +typedef struct cbm_msb cbm_msb_t; + +/* True when `scope` is the scope blob of an MSBuild project file (written by + * cbm_doclink_cs_project_scan_scope), not of a C# source file. */ +bool cbm_msb_is_project_scope(const char *scope); + +cbm_msb_t *cbm_msb_new(void); +void cbm_msb_free(cbm_msb_t *m); + +/* Add the project file `rel_path` (repository-relative, '/'-separated) with + * its blob; both are copied. false when memory ran out. */ +bool cbm_msb_add(cbm_msb_t *m, const char *rel_path, const char *scope); + +/* True when `rel_path` was added. */ +bool cbm_msb_has(const cbm_msb_t *m, const char *rel_path); + +/* False when the project file `rel_path` names an SDK that compiles nothing + * (Microsoft.Build.NoTargets, Microsoft.Build.Traversal: projects that run + * build steps or build other projects). Such a file is the project of no + * source file. True for every other project file, and for one that was not + * added: nothing is known about it. */ +bool cbm_msb_compiles(const cbm_msb_t *m, const char *rel_path); + +typedef struct { + char kind; /* n: a namespace, s: a type (using static), a: an alias */ + const char *alias; /* kind a: the alias; "" otherwise */ + const char *target; +} cbm_msb_using_t; + +typedef struct { + cbm_msb_using_t *usings; /* sorted, without duplicates; NULL when there are none */ + int count; + /* A item or the ImplicitUsings switch could not be evaluated: the + * project may have a global using this list lacks. */ + bool open; + int unevaluable; /* conditions, values and constructs that were not evaluated */ + int outside; /* imports of files the index does not hold */ + void *mem; +} cbm_msb_result_t; + +/* The global usings the project file `project_rel` gives its C# files: the + * nearest Directory.Build.props above it, the file itself, the nearest + * Directory.Build.targets, and what they import, evaluated in that order. + * false when memory ran out or `project_rel` was not added. The result is + * released with cbm_msb_result_free. */ +bool cbm_msb_eval(const cbm_msb_t *m, const char *project_rel, cbm_msb_result_t *out); +void cbm_msb_result_free(cbm_msb_result_t *r); + +/* A single-caller evaluation context; the model must outlive it. The first + * completed normal/poison effect per imported or nearest file is retained. + * Exact dependencies validate reuse; mismatches interpret normally without + * replacing that variant. Ordered persistent state shares unchanged subtrees + * across roots. Project-local evaluation remains call-local, and source item + * spans are evaluated with final properties using one first-variant result + * cache. Results own an independent publication block. Nearest-file metadata + * includes absent files. Model additions invalidate the complete reuse epoch. + * Large overlapping effects or changed inputs can still require more work. */ +typedef struct cbm_msb_eval_context cbm_msb_eval_context_t; +cbm_msb_eval_context_t *cbm_msb_eval_context_new(const cbm_msb_t *m); +void cbm_msb_eval_context_free(cbm_msb_eval_context_t *context); +bool cbm_msb_eval_context_eval(cbm_msb_eval_context_t *context, const char *project_rel, + cbm_msb_result_t *out); + +/* Operation labels stay available to the module in production builds; + * the controls and failure behavior exist only with test seams enabled. */ +typedef enum { + CBM_MSB_ITEM_FAIL_NONE = 0, + CBM_MSB_ITEM_FAIL_CAPTURE_ALLOC = 1, + CBM_MSB_ITEM_FAIL_UNIQUE_INSERT = 2, + CBM_MSB_ITEM_FAIL_APPLY_ALLOC = 3, + CBM_MSB_ITEM_FAIL_PUBLISH_ALLOC = 4, +} cbm_msb_item_fail_operation_t; + +typedef enum { + CBM_MSB_NODE_FAIL_NONE = 0, + CBM_MSB_NODE_FAIL_CAPTURE = 1, + CBM_MSB_NODE_FAIL_OVERLAY = 2, + CBM_MSB_NODE_FAIL_DEPENDENCY = 3, + CBM_MSB_NODE_FAIL_APPLY = 4, +} cbm_msb_node_fail_operation_t; + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +/* Records consumed and peak owned string/array storage: arena capacities + * plus requested heap bytes, including result publication and directory + * memo payloads, packed state-pool capacities (including free slots), symbol + * storage and retained component/value/rope objects. Immutable input, + * hash-table bookkeeping and allocator + * metadata are excluded. */ +void cbm_msb_test_cost_reset(void); +void cbm_msb_test_cost(uint64_t *records, uint64_t *peak_bytes); +/* Deterministic work units since cost_reset: one per record, logical hash + * operation or property cleanup entry; bytes copied/appended by expression, + * property, string, array and publication operations; dependency comparison + * bytes and current-entry validation visits; owner initialization, + * transfer and clearing bytes; unique-item key construction, comparison and + * filtering operations; and bytes processed when forming property keys and + * walking directory paths; persistent-node lookup, overlay, masking, + * validation, source-span composition, allocation and iterative release. + * Test-only revision auditing is observation, not evaluator work. Hash operations + * count API probes, not internal bucket visits. Failed copies add no bytes. + * This measures selected operations, not time or every CPU instruction. */ +uint64_t cbm_msb_test_work(void); +/* Actual directory candidate probes made by msb_nearest since cost_reset. */ +uint64_t cbm_msb_test_nearest_steps(void); +void cbm_msb_test_fail_value_alloc_after(int nth); +bool cbm_msb_test_value_alloc_failed(void); +void cbm_msb_test_fail_prop_insert_after(int nth); +bool cbm_msb_test_prop_insert_failed(void); +uint64_t cbm_msb_test_value_live_bytes(void); +/* One-shot failure of the nth real operation of the selected kind. Zero + * disables it; failed insertions skip the actual set and require verification. */ +void cbm_msb_test_fail_item_operation(cbm_msb_item_fail_operation_t operation, int nth); +bool cbm_msb_test_item_operation_failed(void); + +/* One-shot failure immediately before acquiring a real component node slot. + * Empty/identity fast paths consume no allocation. NONE/nonpositive nth disable. */ +void cbm_msb_test_fail_node_alloc(cbm_msb_node_fail_operation_t operation, int nth); +bool cbm_msb_test_node_alloc_failed(void); + +/* Observations at real state operations; reset clears counters only, never + * revision IDs, pool history, witness metadata or armed fault controls. */ +typedef struct { + uint64_t allocations; + uint64_t slot_reuses; + uint64_t witnessed_slot_reuses; + uint64_t same_length_value_changes; + uint64_t witness_skips; + uint64_t revision_errors; +} cbm_msb_state_test_stats_t; +void cbm_msb_test_state_stats(cbm_msb_state_test_stats_t *out); + +#endif + +#endif /* CBM_PIPELINE_DOC_LINKS_MSBUILD_H */ diff --git a/src/pipeline/lsp_surface.c b/src/pipeline/lsp_surface.c index 4c3cde3363..5acd61b8bd 100644 --- a/src/pipeline/lsp_surface.c +++ b/src/pipeline/lsp_surface.c @@ -23,8 +23,10 @@ #include #include -#include "cbm.h" /* cbm_label_is_relation — reg-only surface membership */ +#include "cbm.h" /* cbm_label_is_relation — reg-only surface membership */ +#include "doclink.h" /* cbm_doclink_portable_scope — the "dl" key */ #include "foundation/log.h" +#include "foundation/mem_core.h" #include "foundation/sha256.h" #include "pipeline/worker_pool.h" #include "yyjson/yyjson.h" @@ -150,6 +152,24 @@ static char *surface_file_to_json(const CBMFileResult *result, const CBMLSPDef * yyjson_mut_obj_add_val(doc, root, "http", http); } + /* The doc-link scope (doclink.h): what other files' doc-comment + * references resolve against -- namespaces, usings, type and member + * declarations -- without line numbers, so a body edit keeps it. A + * changed scope changes the hash, and the closure planner reads it to + * decide whether a repair can stay local. Written only for files that + * have one: every other file's surface bytes stay exactly as they were. + * The defs decoder ignores this key; cbm_doclinks_scopes_from_surfaces + * reads it back for the files an incremental run does not re-extract. */ + if (result && result->doc_scope) { + char *portable = cbm_doclink_portable_scope(result->doc_scope); + if (!portable) { + yyjson_mut_doc_free(doc); + return NULL; /* a missing scope would diverge silently: fail the row */ + } + yyjson_mut_obj_add_strcpy(doc, root, "dl", portable); + cbm_free(CBM_MEM_CLASS_OTHER, portable); + } + char *json = yyjson_mut_write(doc, 0, out_len); yyjson_mut_doc_free(doc); return json; diff --git a/src/pipeline/pass_parallel.c b/src/pipeline/pass_parallel.c index 5c1d690890..8e8ea8fdd6 100644 --- a/src/pipeline/pass_parallel.c +++ b/src/pipeline/pass_parallel.c @@ -70,6 +70,7 @@ enum { PP_CSHARP_M_PREFIX_LEN = 2 }; #define PP_RETAIN_PER_FILE_HARD_MAX_BYTES (32ULL * 1024 * 1024) /* 32 MiB per file */ #include "pipeline/pipeline.h" #include "pipeline/pipeline_internal.h" +#include "pipeline/doc_links.h" /* cbm_doclinks_resolve_file: MENTIONS beside CALLS */ #include "result_spill.h" #include "foundation/platform.h" /* cbm_resolve_cache_dir */ #include "pipeline/pass_lsp_cross.h" /* cbm_pxc_* helpers for fused cross-file LSP */ @@ -2728,14 +2729,20 @@ static void file_node_cache_clear(void) { tl_file_node = NULL; } +const cbm_gbuf_node_t *cbm_pipeline_file_node(const cbm_gbuf_t *gbuf, const char *project, + const char *rel) { + char *file_qn = cbm_pipeline_fqn_compute(project, rel, "__file__"); + const cbm_gbuf_node_t *node = cbm_gbuf_find_by_qn(gbuf, file_qn); + free(file_qn); + return node; +} + static const cbm_gbuf_node_t *file_node_for(const cbm_gbuf_t *gbuf, const char *project, const char *rel) { if (tl_file_node_gbuf == gbuf && tl_file_node_rel == rel) { return tl_file_node; } - char *file_qn = cbm_pipeline_fqn_compute(project, rel, "__file__"); - const cbm_gbuf_node_t *node = cbm_gbuf_find_by_qn(gbuf, file_qn); - free(file_qn); + const cbm_gbuf_node_t *node = cbm_pipeline_file_node(gbuf, project, rel); tl_file_node_gbuf = gbuf; tl_file_node_rel = rel; tl_file_node = node; @@ -3956,6 +3963,12 @@ static void resolve_worker(int worker_id, void *ctx_ptr) { atomic_fetch_add_explicit(&rc->time_ns_semantic, extract_now_ns() - _ph_t0, memory_order_relaxed); + /* ── MENTIONS (doc-comment references) ─────────────────── */ + if (rc->pctx && rc->pctx->doc_links) { + cbm_doclinks_resolve_file(rc->pctx->doc_links, file_idx, result, rc->main_gbuf, + ws->local_edge_buf); + } + cbm_registry_reach_cache_end(); cbm_registry_import_map_cache_end(); cbm_registry_resolve_cache_end(); diff --git a/src/pipeline/pipeline.c b/src/pipeline/pipeline.c index b5073a321a..3e7094daa2 100644 --- a/src/pipeline/pipeline.c +++ b/src/pipeline/pipeline.c @@ -14,11 +14,12 @@ #include "foundation/constants.h" -enum { CBM_DIR_PERMS = 0755, PL_RING = 4, PL_RING_MASK = 3, PL_SEQ_PASSES = 6 }; +enum { CBM_DIR_PERMS = 0755, PL_RING = 4, PL_RING_MASK = 3, PL_SEQ_PASSES = 7 }; #define PL_NSEC_PER_SEC 1000000000LL #include "pipeline/pipeline.h" #include "pipeline/artifact.h" #include "pipeline/pipeline_internal.h" +#include "pipeline/doc_links.h" #include "pipeline/lsp_surface.h" #include "pipeline/pass_lsp_cross.h" #include "pipeline/pass_ensemble_routing.h" @@ -260,6 +261,13 @@ struct cbm_pipeline { cbm_lsp_surface_row_t *surface_rows; int surface_row_count; + /* This run's doc_link_unresolved rows (doc_links.h), handed over by the + * resolve phase; doc_links_ran stays false until a doc-link phase did. */ + cbm_doc_link_row_t *doc_link_rows; + int doc_link_row_count; + bool doc_links_failed; + bool doc_links_ran; + /* Deterministic test-only seam at the final publication boundary. Kept * per pipeline so concurrent test/process activity cannot cross-trigger. */ void (*before_publish_hook)(cbm_pipeline_t *, const char *, void *); @@ -431,6 +439,42 @@ void cbm_pipeline_set_lsp_surfaces(cbm_pipeline_t *p, cbm_lsp_surface_row_t *row p->surface_row_count = count; } +void cbm_pipeline_set_doc_link_rows(cbm_pipeline_t *p, cbm_doc_link_row_t *rows, int count, + bool failed) { + if (!p) { + cbm_doclinks_free_rows(rows, count); + return; + } + cbm_doclinks_free_rows(p->doc_link_rows, p->doc_link_row_count); + p->doc_link_rows = rows; + p->doc_link_row_count = count; + p->doc_links_failed = failed; + p->doc_links_ran = true; +} + +void cbm_pipeline_take_doc_link_rows(cbm_pipeline_t *p, cbm_doc_link_row_t **rows, int *count, + bool *failed, bool *ran) { + *rows = p ? p->doc_link_rows : NULL; + *count = p ? p->doc_link_row_count : 0; + *failed = p ? p->doc_links_failed : true; + *ran = p ? p->doc_links_ran : false; + if (p) { + p->doc_link_rows = NULL; + p->doc_link_row_count = 0; + p->doc_links_failed = false; + p->doc_links_ran = false; + } +} + +/* Forget a previous run's doc-link rows (a pipeline object can run twice). */ +static void pipeline_reset_doc_links(cbm_pipeline_t *p) { + cbm_doclinks_free_rows(p->doc_link_rows, p->doc_link_row_count); + p->doc_link_rows = NULL; + p->doc_link_row_count = 0; + p->doc_links_failed = false; + p->doc_links_ran = false; +} + void cbm_pipeline_free(cbm_pipeline_t *p) { if (!p) { return; @@ -461,6 +505,7 @@ void cbm_pipeline_free(cbm_pipeline_t *p) { cbm_store_free_lsp_surfaces(p->surface_rows, p->surface_row_count); p->surface_rows = NULL; p->surface_row_count = 0; + pipeline_reset_doc_links(p); cbm_git_context_free(&p->git_ctx); /* gbuf, store, registry freed during/after run */ /* Defensively free userconfig in case run() was never called or panicked */ @@ -1445,6 +1490,9 @@ static int run_sequential_pipeline(cbm_pipeline_t *p, cbm_pipeline_ctx_t *ctx, {cbm_pipeline_pass_calls, "calls", false}, {cbm_pipeline_pass_usages, "usages", false}, {cbm_pipeline_pass_semantic, "semantic", false}, + /* doc-comment references: a failure is recorded for publication + * (doc_links.status), never a failed index */ + {cbm_pipeline_pass_doc_links, "doc_links", true}, }; int rc = 0; for (int si = 0; si < PL_SEQ_PASSES && rc == 0; si++) { @@ -1693,9 +1741,16 @@ static int run_parallel_pipeline(cbm_pipeline_t *p, cbm_pipeline_ctx_t *ctx, cbm_log_info("pass.timing", "pass", "lsp_cross_prepare", "elapsed_ms", itoa_buf((int)elapsed_ms(*t))); pipeline_phase_mark("lsp_cross_prepare"); + /* Doc-comment references resolve beside CALLS in the resolve workers; + * their indexes need every file's scope and the complete node set. */ + cbm_clock_gettime(CLOCK_MONOTONIC, t); + cbm_doclinks_begin(ctx, files, file_count, cache); + cbm_log_info("pass.timing", "pass", "doc_links_prepare", "elapsed_ms", + itoa_buf((int)elapsed_ms(*t))); cbm_clock_gettime(CLOCK_MONOTONIC, t); rc = cbm_parallel_resolve(ctx, files, file_count, cache, &shared_ids, worker_count, all_defs, def_count, def_modules, module_def_index, &cross_registries); + cbm_doclinks_end(ctx); cbm_log_info("pass.timing", "pass", "parallel_resolve", "elapsed_ms", itoa_buf((int)elapsed_ms(*t))); pipeline_phase_mark("parallel_resolve"); @@ -2253,6 +2308,31 @@ int cbm_pipeline_publish_generation(const cbm_pipeline_generation_t *generation) return cbm_pipeline_publish_staged(stage_path, generation, true, false); } +/* Write the generation's doc_link_unresolved rows (plus the error marker when + * the doc-link layer failed). */ +static int publish_doc_links(cbm_store_t *store, const cbm_pipeline_generation_t *generation) { + if (!generation->doc_links_failed) { + return cbm_store_doc_links_replace(store, generation->project, generation->doc_link_rows, + generation->doc_link_row_count); + } + cbm_log_error("doc_links.error", "phase", "publish", "reason", "layer_failed", "project", + generation->project); + int n = generation->doc_link_row_count; + cbm_doc_link_row_t *rows = (cbm_doc_link_row_t *)cbm_alloc( + CBM_MEM_CLASS_STORE, (size_t)(n + SKIP_ONE) * sizeof(*rows)); + if (!rows) { + return CBM_STORE_ERR; + } + if (n > 0) { + memcpy(rows, generation->doc_link_rows, (size_t)n * sizeof(*rows)); + } + rows[n] = (cbm_doc_link_row_t){ + .rel_path = "", .line = 0, .syntax = "", .raw = "doc-link layer failed", .reason = "error"}; + int rc = cbm_store_doc_links_replace(store, generation->project, rows, n + SKIP_ONE); + cbm_free(CBM_MEM_CLASS_STORE, rows); /* the array only: the strings are borrowed */ + return rc; +} + /* Complete and publish an already-materialized staging database: metadata * writes, FTS policy, integrity, seal, then the shared finalize leg. Takes * ownership of stage_path (frees it on every path). fts_wholesale selects @@ -2325,6 +2405,15 @@ int cbm_pipeline_publish_staged(char *stage_path, const cbm_pipeline_generation_ cbm_log_info("publish.timing", "block", "coverage_replace", "elapsed_ms", itoa_buf((int)elapsed_ms(t_pub))); cbm_clock_gettime(CLOCK_MONOTONIC, &t_pub); + /* Unresolved doc-comment references belong to the generation, like its + * coverage rows. A failed doc-link layer is recorded as an error marker + * row (rel_path "", reason "error") so index_status can say so. */ + if (ok && publish_doc_links(store, generation) != CBM_STORE_OK) { + ok = false; + } + cbm_log_info("publish.timing", "block", "doc_links", "elapsed_ms", + itoa_buf((int)elapsed_ms(t_pub))); + cbm_clock_gettime(CLOCK_MONOTONIC, &t_pub); /* The column list lives in cbm_store_fts_rebuild() alone — see the delta * merge, which must index the SAME columns or prose goes missing on the * warm path while a full reindex looks perfect. */ @@ -2553,6 +2642,10 @@ static int dump_and_persist_hashes(cbm_pipeline_t *p, const cbm_file_hash_t *bas }, .surface_rows = p->surface_rows, .surface_row_count = p->surface_row_count, + .doc_link_rows = p->doc_link_rows, + .doc_link_row_count = p->doc_link_row_count, + /* every full route runs a doc-link phase; none having run is a fault */ + .doc_links_failed = p->doc_links_failed || !p->doc_links_ran, }; free(db_dir); @@ -2728,6 +2821,7 @@ static int cbm_pipeline_run_staged(cbm_pipeline_t *p) { bool restore_requested_discovery = false; p->mode = p->requested_mode; + pipeline_reset_doc_links(p); bool mode_promoted = promote_mode_to_existing_coverage(p); /* cbm_pipeline_new() may precede the actual run by an arbitrary interval. diff --git a/src/pipeline/pipeline_incremental.c b/src/pipeline/pipeline_incremental.c index 99195cc2c0..ce150f1cc8 100644 --- a/src/pipeline/pipeline_incremental.c +++ b/src/pipeline/pipeline_incremental.c @@ -20,17 +20,20 @@ enum { INCR_RING_BUF = 4, INCR_RING_MASK = 3, INCR_TS_BUF = 24 }; #include "sqlite3.h" #include "yyjson/yyjson.h" #include "pipeline/pipeline_internal.h" +#include "pipeline/doc_links.h" #include "store/store.h" #include "graph_buffer/graph_buffer.h" #include "discover/discover.h" #include "foundation/log.h" #include "foundation/hash_table.h" +#include "foundation/mem_core.h" #include "foundation/compat.h" #include "foundation/compat_fs.h" #include "foundation/compat_thread.h" #include "foundation/platform.h" #include "foundation/sha256.h" +#include #include #include #include @@ -1126,14 +1129,34 @@ typedef struct { int n_dependents; cbm_lsp_surface_row_t *stored_rows; /* whole previous generation */ int stored_count; + /* The previous generation's doc_link_unresolved rows: carried forward for + * the files the repair does not re-extract. */ + cbm_doc_link_row_t *doc_rows; + int doc_row_count; } closure_plan_t; static void closure_plan_free(closure_plan_t *plan) { free(plan->files); cbm_store_free_lsp_surfaces(plan->stored_rows, plan->stored_count); + cbm_store_free_doc_links(plan->doc_rows, plan->doc_row_count); memset(plan, 0, sizeof(*plan)); } +/* Add a heap copy of `name` to a heap-keyed name set (value = key; released + * by surface_name_set_free). No-op when present; false when the copy could + * not be made. */ +static bool surface_name_set_put(CBMHashTable *set, const char *name) { + if (!name[0] || cbm_ht_get(set, name)) { + return true; + } + char *copy = strdup(name); + if (!copy) { + return false; + } + cbm_ht_set(set, copy, copy); + return true; +} + /* Short names a surface JSON defines ("lsp"[].sn plus "reg"[].n), as a * heap-keyed set. Returns NULL on parse failure — callers decline to FULL. */ static CBMHashTable *surface_name_set(const char *defs_json) { @@ -1159,11 +1182,8 @@ static CBMHashTable *surface_name_set(const char *defs_json) { yyjson_val *item; yyjson_arr_foreach(arrs[a], idx, max, item) { const char *name = yyjson_get_str(yyjson_obj_get(item, arr_keys[a])); - if (name && name[0] && !cbm_ht_get(set, name)) { - char *copy = strdup(name); - if (copy) { - cbm_ht_set(set, copy, copy); - } + if (name) { + (void)surface_name_set_put(set, name); } } } @@ -1217,6 +1237,199 @@ static int surface_added_names(const char *stored_json, const char *fresh_json, return rc; } +/* ── Doc-link closure rules ─────────────────────────────────────── + * + * MENTIONS edges are owned by their source file like CALLS: a re-extracted + * file rewrites its own edges and doc_link_unresolved rows, and the files + * with edges INTO a changed file are dependents. What the edge set cannot + * see is decided by the language's resolver hooks (doc_links.h), never here: + * - a scope input (a file without a scope of its own that sets the scope of + * a language's files; no language has one today): a change declines to + * FULL; + * - a scope delta ("dl" surface key) the language calls GLOBAL can re-route + * or re-classify references in files with no edge into the changed file: + * FULL, exactly like an added name. A deleted file with a scope is GLOBAL. + * C#'s MSBuild project files come in here: each has a scope blob, and any + * change to it is GLOBAL (it sets the global usings of a whole project); + * - a REMOVED name can make another file's unresolved reference resolve (an + * overload group shrinks) or change its reason: the files whose rows + * mention it re-resolve with the closure. */ + +/* Add s[0..n) to a heap-keyed name set. false when it cannot be recorded (too + * long for a key, or out of memory): callers fail closed. */ +static bool name_set_add_n(CBMHashTable *set, const char *s, size_t n) { + char key[CBM_SZ_1K]; + if (n >= sizeof(key)) { + return false; + } + memcpy(key, s, n); + key[n] = '\0'; + return surface_name_set_put(set, key); +} + +/* The "dl" scope of a surface JSON (a CBM_MEM_CLASS_OTHER block; NULL when it + * has none). false when the row's scope cannot be read: the caller declines. */ +static bool surface_dl_scope(const char *defs_json, char **out) { + return cbm_doclinks_scope_from_surface_json(defs_json, out) == 0; +} + +/* cbm_doclink_name_fn over a heap-keyed name set. */ +static bool removed_name_put(void *ud, const char *name, size_t len) { + return name_set_add_n((CBMHashTable *)ud, name, len); +} + +typedef struct { + CBMHashTable *fresh; + CBMHashTable *removed; + bool failed; +} removed_walk_t; + +static void removed_name_visitor(const char *key, void *value, void *userdata) { + (void)value; + removed_walk_t *walk = (removed_walk_t *)userdata; + if ((!walk->fresh || !cbm_ht_get(walk->fresh, key)) && + !name_set_add_n(walk->removed, key, strlen(key))) { + walk->failed = true; + } +} + +/* Short names of `stored_json` absent from `fresh_json` (all of them when + * fresh_json is NULL: a deleted file). */ +static int surface_removed_names(const char *stored_json, const char *fresh_json, + CBMHashTable *removed) { + CBMHashTable *stored_set = surface_name_set(stored_json); + CBMHashTable *fresh_set = fresh_json ? surface_name_set(fresh_json) : NULL; + if (!stored_set || (fresh_json && !fresh_set)) { + surface_name_set_free(stored_set); + surface_name_set_free(fresh_set); + return CBM_NOT_FOUND; + } + removed_walk_t walk = {.fresh = fresh_set, .removed = removed, .failed = false}; + cbm_ht_foreach(stored_set, removed_name_visitor, &walk); + surface_name_set_free(stored_set); + surface_name_set_free(fresh_set); + return walk.failed ? CBM_NOT_FOUND : 0; +} + +/* The doc-link view of one changed (fresh_json set) or deleted (fresh_json + * NULL) file: the names it no longer declares go to `removed`. false when the + * change cannot be repaired file by file (the rules above) or the stored data + * cannot be read -- either way the caller declines. */ +static bool doc_scope_repairable(const char *stored_json, const char *fresh_json, + CBMHashTable *removed) { + char *stored_dl = NULL; + char *fresh_dl = NULL; + bool ok = surface_dl_scope(stored_json, &stored_dl) && + (!fresh_json || surface_dl_scope(fresh_json, &fresh_dl)) && + cbm_doclinks_scope_delta(stored_dl, fresh_dl, removed_name_put, removed) == + CBM_DOCLINK_DELTA_LOCAL && + surface_removed_names(stored_json, fresh_json, removed) == 0; + cbm_free(CBM_MEM_CLASS_OTHER, stored_dl); + cbm_free(CBM_MEM_CLASS_OTHER, fresh_dl); + return ok; +} + +typedef struct { + CBMHashTable *closure; + CBMHashTable *files; + int added; +} row_dep_walk_t; + +/* Add a row-named dependent to the closure when discovery still has it (a + * deleted file's rows are dropped by the carry-forward anyway). */ +static void row_dep_visitor(const char *key, void *value, void *userdata) { + (void)value; + row_dep_walk_t *walk = (row_dep_walk_t *)userdata; + if (!cbm_ht_get(walk->files, key) || cbm_ht_get(walk->closure, key)) { + return; + } + cbm_ht_set(walk->closure, key, (void *)key); + walk->added++; +} + +/* Files whose unresolved doc-link rows mention a removed name as an + * identifier token. Keys are borrowed from `rows`. */ +static void doc_row_name_dependents(const cbm_doc_link_row_t *rows, int n, + const CBMHashTable *removed, CBMHashTable *out_paths) { + if (!removed || cbm_ht_count(removed) == 0) { + return; + } + for (int i = 0; i < n; i++) { + const char *raw = rows[i].raw; + const char *rel = rows[i].rel_path; + if (!raw || !rel || !rel[0] || cbm_ht_get(out_paths, rel)) { + continue; + } + for (const char *p = raw; *p;) { + while (*p && !(isalnum((unsigned char)*p) || *p == '_')) { + p++; + } + const char *s = p; + while (*p && (isalnum((unsigned char)*p) || *p == '_')) { + p++; + } + char tok[CBM_SZ_512]; + size_t tl = (size_t)(p - s); + if (tl > 0 && tl < sizeof(tok)) { + memcpy(tok, s, tl); + tok[tl] = '\0'; + if (cbm_ht_get(removed, tok)) { + cbm_ht_set(out_paths, rel, (void *)rel); + break; + } + } + } + } +} + +/* Doc-link carry-forward state of the legacy partial route: the previous + * rows, the stored scopes of the files it does not re-extract, and the set + * of re-extracted or deleted paths whose old rows it replaces. */ +typedef struct { + cbm_doc_link_row_t *old_rows; + int old_count; + cbm_doclink_scope_t *scopes; + int scope_count; + CBMHashTable *replaced; /* heap keys (value = key) */ + bool ok; +} legacy_doc_t; + +static void legacy_doc_load(legacy_doc_t *d, cbm_store_t *store, const char *project, + const cbm_file_info_t *changed, int ci, char *const *deleted, + int deleted_count) { + memset(d, 0, sizeof(*d)); + d->replaced = cbm_ht_create(CBM_SZ_64); + d->ok = d->replaced != NULL; + for (int i = 0; d->ok && i < ci; i++) { + d->ok = name_set_add_n(d->replaced, changed[i].rel_path, strlen(changed[i].rel_path)); + } + for (int i = 0; d->ok && i < deleted_count; i++) { + d->ok = name_set_add_n(d->replaced, deleted[i], strlen(deleted[i])); + } + if (d->ok && cbm_store_doc_links_get(store, project, &d->old_rows, &d->old_count, NULL) != + CBM_STORE_OK) { + d->ok = false; + } + cbm_lsp_surface_row_t *surf = NULL; + int surf_count = 0; + if (d->ok && cbm_store_get_lsp_surfaces(store, project, &surf, &surf_count) == CBM_STORE_OK && + cbm_doclinks_scopes_from_surfaces(surf, surf_count, d->replaced, &d->scopes, + &d->scope_count) != 0) { + d->ok = false; + } + cbm_store_free_lsp_surfaces(surf, surf_count); + if (!d->ok) { + cbm_log_error("doc_links.error", "phase", "legacy_carry_forward", "reason", "read"); + } +} + +static void legacy_doc_free(legacy_doc_t *d) { + cbm_store_free_doc_links(d->old_rows, d->old_count); + cbm_doclinks_free_scopes(d->scopes, d->scope_count); + surface_name_set_free(d->replaced); + memset(d, 0, sizeof(*d)); +} + /* Run parallel or sequential extract+resolve for changed files. Any failure * aborts before persistence: the caller discards this in-memory graph and * preserves the old on-disk database and its retryable hashes. */ @@ -1367,9 +1580,11 @@ static int run_extract_resolve(cbm_pipeline_ctx_t *ctx, cbm_file_info_t *changed "elapsed_ms", itoa_buf((int)elapsed_ms(t))); } cbm_clock_gettime(CLOCK_MONOTONIC, &t); + cbm_doclinks_begin(ctx, changed_files, ci, cache); rc = cbm_parallel_resolve(ctx, changed_files, ci, cache, &shared_ids, worker_count, all_defs, all_def_count, closure ? closure->def_modules : NULL, module_def_index, registries_arg); + cbm_doclinks_end(ctx); if (module_def_index) { cbm_pxc_free_module_def_index(module_def_index); } @@ -1423,6 +1638,9 @@ static int run_extract_resolve(cbm_pipeline_ctx_t *ctx, cbm_file_info_t *changed if (rc == 0) { rc = cbm_pipeline_check_cancel(ctx); } + if (rc == 0) { + (void)cbm_pipeline_pass_doc_links(ctx, changed_files, ci); + } if (owns_cache) { free_incremental_result_cache(cache, ci); ctx->result_cache = prior_cache; @@ -1515,12 +1733,19 @@ static int run_postpasses(cbm_pipeline_ctx_t *ctx, cbm_file_info_t *changed_file } /* Publish the test-only legacy partial result through the same atomic * generation boundary as full indexing. */ +typedef struct { + const cbm_doc_link_row_t *rows; + int count; + bool failed; +} legacy_doc_rows_t; + static int dump_and_persist(cbm_gbuf_t *gbuf, const char *db_path, const char *project, atomic_int *cancelled, const cbm_file_hash_t *manifest, int manifest_count, const char *adr_content, const cbm_coverage_row_t *cov, int cov_count, const cbm_coverage_meta_t *meta_template, - const cbm_lsp_surface_row_t *surface_rows, int surface_row_count) { + const cbm_lsp_surface_row_t *surface_rows, int surface_row_count, + legacy_doc_rows_t doc) { struct timespec t; cbm_clock_gettime(CLOCK_MONOTONIC, &t); cbm_pipeline_generation_t generation = { @@ -1536,6 +1761,9 @@ static int dump_and_persist(cbm_gbuf_t *gbuf, const char *db_path, const char *p .coverage_meta = meta_template ? *meta_template : (cbm_coverage_meta_t){0}, .surface_rows = surface_rows, .surface_row_count = surface_row_count, + .doc_link_rows = doc.rows, + .doc_link_row_count = doc.count, + .doc_links_failed = doc.failed, }; int rc = cbm_pipeline_publish_generation(&generation); cbm_log_info("incremental.dump", "rc", itoa_buf(rc), "elapsed_ms", @@ -1718,7 +1946,12 @@ static int closure_try_plan(cbm_pipeline_t *p, cbm_store_t *store, const char *p cbm_path_alias_collection_t *plan_aliases = NULL; int n_changed = 0; int n_deleted = 0; - if (!fresh_by_path || !files_by_path || !stored_by_path || !closure_set) { + int n_row_dependents = 0; + /* names a changed or deleted file no longer declares (doc-link rows) */ + CBMHashTable *removed_names = cbm_ht_create(CBM_SZ_64); + CBMHashTable *row_deps = cbm_ht_create(CBM_SZ_64); + if (!fresh_by_path || !files_by_path || !stored_by_path || !closure_set || !removed_names || + !row_deps) { decline = "alloc"; goto done; } @@ -1795,6 +2028,14 @@ static int closure_try_plan(cbm_pipeline_t *p, cbm_store_t *store, const char *p decline = "no_file_delta"; goto done; } + /* A doc-link scope input scopes files other than itself and has no scope + * blob the surface comparison below could judge. */ + for (int i = 0; i < n_changed + n_deleted; i++) { + if (cbm_doclinks_is_scope_input(changed_paths[i])) { + decline = "doc_scope_input_changed"; + goto done; + } + } /* Load the previous generation's surfaces and probe the changed files' * fresh ones. Missing rows fail closed. */ @@ -1804,6 +2045,27 @@ static int closure_try_plan(cbm_pipeline_t *p, cbm_store_t *store, const char *p decline = "no_surface_rows"; goto done; } + /* The previous doc_link_unresolved rows: carried forward by the repair. + * A generation without them, or one whose doc-link layer failed, cannot + * be repaired file by file. */ + { + bool doc_present = false; + if (cbm_store_doc_links_get(store, project, &plan->doc_rows, &plan->doc_row_count, + &doc_present) != CBM_STORE_OK) { + decline = "doc_links_read_failed"; + goto done; + } + if (!doc_present) { + decline = "doc_links_missing"; + goto done; + } + for (int i = 0; i < plan->doc_row_count; i++) { + if (!plan->doc_rows[i].rel_path || !plan->doc_rows[i].rel_path[0]) { + decline = "doc_links_error"; + goto done; + } + } + } CBMHashTable *rows_by_path = cbm_ht_create((size_t)plan->stored_count * PAIR_LEN); if (!rows_by_path) { decline = "alloc"; @@ -1866,18 +2128,30 @@ static int closure_try_plan(cbm_pipeline_t *p, cbm_store_t *store, const char *p continue; /* body edit: the file re-resolves, nobody else does */ } bool added = false; + const char *why = NULL; if (surface_added_names(stored_row->defs_json, fresh_row->defs_json, &added) != 0 || added) { + why = "added_definition_names"; + } else if (!doc_scope_repairable(stored_row->defs_json, fresh_row->defs_json, + removed_names)) { + why = "doc_scope_changed"; + } + if (why) { free(dep_targets); cbm_ht_free(rows_by_path); - decline = "added_definition_names"; + decline = why; goto done; } n_surface_changed++; dep_targets[dep_target_count++] = changed_paths[i]; } + bool gone_scope_changed = false; /* a deleted file takes its declarations along */ for (int i = 0; i < n_deleted; i++) { dep_targets[dep_target_count++] = changed_paths[n_changed + i]; + const cbm_lsp_surface_row_t *gone = cbm_ht_get(rows_by_path, changed_paths[n_changed + i]); + if (gone && !doc_scope_repairable(gone->defs_json, NULL, removed_names)) { + gone_scope_changed = true; + } } cbm_ht_free(rows_by_path); @@ -1889,6 +2163,10 @@ static int closure_try_plan(cbm_pipeline_t *p, cbm_store_t *store, const char *p goto done; } free(dep_targets); + if (gone_scope_changed) { + decline = "doc_scope_changed"; + goto done; + } /* Closure = changed ∪ dependents. Every member must be in the current * discovery (a dependent outside it cannot be re-resolved). */ @@ -1906,6 +2184,14 @@ static int closure_try_plan(cbm_pipeline_t *p, cbm_store_t *store, const char *p cbm_ht_set(closure_set, dependents[i], dependents[i]); } } + /* Files whose unresolved doc-link references name something a changed or + * deleted file no longer declares re-resolve too. */ + doc_row_name_dependents(plan->doc_rows, plan->doc_row_count, removed_names, row_deps); + { + row_dep_walk_t walk = {.closure = closure_set, .files = files_by_path, .added = 0}; + cbm_ht_foreach(row_deps, row_dep_visitor, &walk); + n_row_dependents = walk.added; + } /* closure_count == 0 is legitimate: a deleted-only delta with no * dependents has nothing to re-parse, but the purge itself still needs * the executor. */ @@ -1936,7 +2222,8 @@ static int closure_try_plan(cbm_pipeline_t *p, cbm_store_t *store, const char *p cbm_log_info("incremental.closure_plan", "changed", itoa_buf(n_changed), "surface_changed", itoa_buf(n_surface_changed), "deleted", itoa_buf(n_deleted), "dependents", itoa_buf(dependent_count)); - cbm_log_info("incremental.closure_plan_done", "closure", itoa_buf(plan->count), "elapsed_ms", + cbm_log_info("incremental.closure_plan_done", "closure", itoa_buf(plan->count), + "doc_link_dependents", itoa_buf(n_row_dependents), "elapsed_ms", itoa_buf((int)elapsed_ms(t))); done: @@ -1949,10 +2236,13 @@ static int closure_try_plan(cbm_pipeline_t *p, cbm_store_t *store, const char *p cbm_ht_free(files_by_path); cbm_ht_free(stored_by_path); cbm_ht_free(closure_set); + cbm_ht_free(row_deps); /* keys borrowed from plan->doc_rows */ + surface_name_set_free(removed_names); if (decline) { cbm_log_info("incremental.closure_decline", "reason", decline, "elapsed_ms", itoa_buf((int)elapsed_ms(t))); cbm_store_free_lsp_surfaces(plan->stored_rows, plan->stored_count); + cbm_store_free_doc_links(plan->doc_rows, plan->doc_row_count); memset(plan, 0, sizeof(*plan)); return 0; } @@ -1995,6 +2285,13 @@ static int run_closure_delta(cbm_pipeline_t *p, const char *db_path, const char int manifest_count = 0; cbm_coverage_row_t *cov = NULL; int cov_n = 0; + cbm_doclink_scope_t *doc_base = NULL; + int doc_base_count = 0; + cbm_doc_link_row_t *doc_fresh = NULL; + int doc_fresh_count = 0; + cbm_doc_link_row_t *doc_rows = NULL; + int doc_row_count = 0; + bool doc_failed = false; cbm_clock_gettime(CLOCK_MONOTONIC, &t); if (cbm_delta_stage_clone(db_path, &stage) != 0) { @@ -2167,6 +2464,12 @@ static int run_closure_delta(cbm_pipeline_t *p, const char *db_path, const char cbm_pipeline_get_excluded(p, &excluded_dirs, &excluded_count); path_aliases = cbm_load_path_aliases_excluded(cbm_pipeline_repo_path(p), excluded_dirs, excluded_count); + /* Doc-link scopes of every file this repair does not re-extract. */ + if (cbm_doclinks_scopes_from_surfaces(plan->stored_rows, plan->stored_count, stale_surface_set, + &doc_base, &doc_base_count) != 0) { + cbm_log_error("delta.err", "phase", "doc_link_scopes"); + goto out; + } cbm_pipeline_ctx_t ctx = { .project_name = project, .repo_path = cbm_pipeline_repo_path(p), @@ -2178,6 +2481,8 @@ static int run_closure_delta(cbm_pipeline_t *p, const char *db_path, const char .path_aliases = path_aliases, .excluded_dirs = excluded_dirs, .excluded_count = excluded_count, + .doc_link_base = doc_base, + .doc_link_base_count = doc_base_count, }; for (int i = 0; i < ci; i++) { char *file_qn = cbm_pipeline_fqn_compute(project, changed_files[i].rel_path, "__file__"); @@ -2221,6 +2526,19 @@ static int run_closure_delta(cbm_pipeline_t *p, const char *db_path, const char } cbm_log_info("delta.repair", "files", itoa_buf(ci), "elapsed_ms", itoa_buf((int)elapsed_ms(t))); + /* Doc-link rows: the previous rows of every file not re-extracted (or + * deleted), then this repair's. */ + { + bool doc_ran = false; + cbm_pipeline_take_doc_link_rows(p, &doc_fresh, &doc_fresh_count, &doc_failed, &doc_ran); + doc_failed = doc_failed || !doc_ran; + if (cbm_doclinks_merge_rows(plan->doc_rows, plan->doc_row_count, stale_surface_set, + doc_fresh, doc_fresh_count, &doc_rows, &doc_row_count) != 0) { + cbm_log_error("delta.err", "phase", "doc_link_rows"); + goto out; + } + } + cbm_clock_gettime(CLOCK_MONOTONIC, &t); if (cbm_delta_patch(staging, project, gbuf, max_db_id, snapshot, snapshot_count) != 0) { goto out; @@ -2365,6 +2683,9 @@ static int run_closure_delta(cbm_pipeline_t *p, const char *db_path, const char .surface_rows = NULL, .surface_row_count = 0, .surfaces_in_place = true, + .doc_link_rows = doc_rows, + .doc_link_row_count = doc_row_count, + .doc_links_failed = doc_failed, }; cbm_store_close(staging); staging = NULL; @@ -2391,6 +2712,9 @@ static int run_closure_delta(cbm_pipeline_t *p, const char *db_path, const char } cbm_delta_free_snapshot(snapshot, snapshot_count); surface_name_set_free(stale_surface_set); + cbm_doclinks_free_scopes(doc_base, doc_base_count); + cbm_doclinks_free_rows(doc_fresh, doc_fresh_count); + cbm_doclinks_free_rows(doc_rows, doc_row_count); if (cr_arena_live) { cbm_store_free_lsp_surfaces(cr.fresh_rows, cr.fresh_count); for (int i = 0; i < cr.def_module_count; i++) { @@ -2426,6 +2750,25 @@ static int run_closure_delta(cbm_pipeline_t *p, const char *db_path, const char /* ── Incremental pipeline entry point ────────────────────────────── */ +/* The stored generation's doc-link layer is usable: its doc_link_unresolved + * table exists and records no failure of the layer. */ +static bool incr_doc_links_current(cbm_store_t *store, const char *project) { + cbm_doc_link_reason_count_t *reasons = NULL; + int reason_count = 0; + cbm_doc_link_row_t *samples = NULL; + int sample_count = 0; + bool present = false; + bool current = cbm_store_doc_links_summary(store, project, &reasons, &reason_count, &samples, + &sample_count, 0, &present) == CBM_STORE_OK && + present; + for (int i = 0; current && i < reason_count; i++) { + current = strcmp(reasons[i].reason, "error") != 0; + } + cbm_store_free_doc_link_reasons(reasons, reason_count); + cbm_store_free_doc_links(samples, sample_count); + return current; +} + int cbm_pipeline_run_incremental(cbm_pipeline_t *p, const char *db_path, cbm_file_info_t *files, int file_count, const cbm_file_hash_t *baseline_manifest, int baseline_count, bool force_full_on_mismatch) { @@ -2478,7 +2821,11 @@ int cbm_pipeline_run_incremental(cbm_pipeline_t *p, const char *db_path, cbm_fil meta.coverage_version == CBM_SEMANTIC_INDEX_VERSION && meta.hash_records_complete && meta.index_mode && strcmp(meta.index_mode, mode_name) == 0; - bool exact = metadata_current && + /* A generation whose doc-link layer is missing or failed is not + * current even with identical inputs: index_status tells the user to + * re-run, and the re-run has to rebuild. */ + bool doc_links_current = metadata_current && incr_doc_links_current(store, project); + bool exact = doc_links_current && cbm_pipeline_semantic_manifests_equal(stored, stored_count, baseline_manifest, baseline_count); cbm_store_coverage_meta_clear(&meta); @@ -2520,7 +2867,9 @@ int cbm_pipeline_run_incremental(cbm_pipeline_t *p, const char *db_path, cbm_fil #if defined(CBM_INCREMENTAL_TEST_API) && CBM_INCREMENTAL_TEST_API incr_test_set_last_route(CBM_INCREMENTAL_ROUTE_FORCED_FULL); #endif - cbm_log_info("incremental.force_full", "reason", "semantic_manifest_changed"); + cbm_log_info("incremental.force_full", "reason", + metadata_current && !doc_links_current ? "doc_links_not_current" + : "semantic_manifest_changed"); return CBM_PIPELINE_FORCE_FULL_REINDEX; } } @@ -2684,6 +3033,8 @@ int cbm_pipeline_run_incremental(cbm_pipeline_t *p, const char *db_path, cbm_fil return CBM_PIPELINE_ABORT_PRESERVE_DB; } + legacy_doc_t legacy_doc; + legacy_doc_load(&legacy_doc, store, project, changed_files, ci, deleted, deleted_count); cbm_store_close(store); /* Snapshot inbound cross-file edges into changed files BEFORE purging, so @@ -2751,6 +3102,8 @@ int cbm_pipeline_run_incremental(cbm_pipeline_t *p, const char *db_path, cbm_fil .path_aliases = path_aliases, .excluded_dirs = excluded_dirs, .excluded_count = excluded_count, + .doc_link_base = legacy_doc.scopes, + .doc_link_base_count = legacy_doc.scope_count, }; for (int i = 0; i < ci; i++) { @@ -2810,9 +3163,29 @@ int cbm_pipeline_run_incremental(cbm_pipeline_t *p, const char *db_path, cbm_fil free_mode_skipped(mode_skipped, mode_skipped_count); free(saved_adr); cbm_gbuf_free(existing); + legacy_doc_free(&legacy_doc); return CBM_PIPELINE_ABORT_PRESERVE_DB; } + /* Doc-link rows: previous rows of the files not re-extracted, then this + * run's. A failed carry-forward read is a failed doc-link layer. */ + cbm_doc_link_row_t *doc_rows = NULL; + int doc_row_count = 0; + bool doc_failed = false; + { + cbm_doc_link_row_t *fresh = NULL; + int fresh_count = 0; + bool ran = false; + cbm_pipeline_take_doc_link_rows(p, &fresh, &fresh_count, &doc_failed, &ran); + doc_failed = doc_failed || !ran || !legacy_doc.ok; + if (cbm_doclinks_merge_rows(legacy_doc.old_rows, legacy_doc.old_count, legacy_doc.replaced, + fresh, fresh_count, &doc_rows, &doc_row_count) != 0) { + doc_failed = true; + } + cbm_doclinks_free_rows(fresh, fresh_count); + } + legacy_doc_free(&legacy_doc); + /* Coverage rows (#963): merge = previous FAILURE rows for files NOT * re-extracted this run + this run's fresh entries (changed files replace * their old rows — a file that parses cleanly now simply contributes @@ -2902,6 +3275,7 @@ int cbm_pipeline_run_incremental(cbm_pipeline_t *p, const char *db_path, cbm_fil free_mode_skipped(mode_skipped, mode_skipped_count); free(saved_adr); cbm_gbuf_free(existing); + cbm_doclinks_free_rows(doc_rows, doc_row_count); return manifest_rc == CBM_DISCOVER_LIMIT_EXCEEDED ? CBM_PIPELINE_RESOURCE_LIMIT : CBM_PIPELINE_ABORT_PRESERVE_DB; } @@ -2929,9 +3303,11 @@ int cbm_pipeline_run_incremental(cbm_pipeline_t *p, const char *db_path, cbm_fil * re-parsed files have no codec output, and publishing a stale row * would satisfy a future closure plan with yesterday's surface; an * empty table just routes the next incremental to a full rebuild. */ + legacy_doc_rows_t doc_out = {.rows = doc_rows, .count = doc_row_count, .failed = doc_failed}; int persist_rc = dump_and_persist(existing, db_path, project, cbm_pipeline_cancelled_ptr(p), manifest, - manifest_count, saved_adr, cov, cov_n, &coverage_meta, NULL, 0); + manifest_count, saved_adr, cov, cov_n, &coverage_meta, NULL, 0, doc_out); + cbm_doclinks_free_rows(doc_rows, doc_row_count); cbm_pipeline_free_semantic_manifest(manifest, manifest_count); free(saved_adr); free(cov); diff --git a/src/pipeline/pipeline_internal.h b/src/pipeline/pipeline_internal.h index a196dd5c5c..62b9c4c413 100644 --- a/src/pipeline/pipeline_internal.h +++ b/src/pipeline/pipeline_internal.h @@ -156,6 +156,16 @@ typedef struct { * the incremental and probe routes still hand the cache array to passes * that index it directly, so they keep results in memory (follow-up). */ bool spill_allowed; + + /* Doc-comment references -> MENTIONS (doc_links.h). doc_links is the + * run's resolver state while the resolve phase runs (NULL otherwise); + * doc_link_base holds the stored scopes of the files an incremental run + * does not re-extract (borrowed from the route that loaded them); + * doc_links_failed records a failed build so publication can mark it. */ + struct cbm_doclinks *doc_links; + const struct cbm_doclink_scope *doc_link_base; + int doc_link_base_count; + bool doc_links_failed; } cbm_pipeline_ctx_t; /* ── Result-cache access contract (spill mode) ──────────────────────── @@ -174,6 +184,12 @@ void cbm_pipeline_result_release(CBMFileResult *r, bool loaded); /* Log the store counters, close and delete the store, drop the latch. */ void cbm_pipeline_spill_close(cbm_pipeline_ctx_t *ctx); +/* The File node of `rel` in `gbuf` (NULL when it has none): the one lookup + * for "this file as an edge source", by the name cbm_pipeline_fqn_compute + * gives every File node. */ +const cbm_gbuf_node_t *cbm_pipeline_file_node(const cbm_gbuf_t *gbuf, const char *project, + const char *rel); + /* Transcode an ObjectScript Studio Export XML file and compose every generated * UDL class into one cacheable result. The returned result owns all child * extraction arenas and is released with the ordinary cbm_free_result(). */ @@ -829,8 +845,11 @@ int cbm_pipeline_build_fresh_semantic_manifest(cbm_pipeline_t *p, const char *pr * (the enum name is no longer a segment); typedef names, anonymous-enum * constants and macro-prefixed functions are nodes; a bodyless * `struct X` is no node. An index written before this holds the old - * QNs for every unchanged file, so it is rebuilt in full once. */ -enum { CBM_SEMANTIC_INDEX_VERSION = 4 }; + * QNs for every unchanged file, so it is rebuilt in full once. + * 5: doc-comment references became MENTIONS edges and doc_link_unresolved + * rows, and C# LSP surfaces carry the doc-link scope ("dl"); an index + * built before has neither, so it rebuilds once on upgrade. */ +enum { CBM_SEMANTIC_INDEX_VERSION = 5 }; typedef struct { cbm_gbuf_t *gbuf; @@ -853,6 +872,13 @@ typedef struct { * into the staging store (delta patch); publish then skips the * wholesale delete+rewrite. */ bool surfaces_in_place; + /* The generation's doc_link_unresolved rows (complete: an incremental + * route passes the merge of carried-forward and fresh rows), and whether + * the doc-link layer failed for it (publish then adds the error marker + * row that index_status reports as doc_links.status = "error"). */ + const cbm_doc_link_row_t *doc_link_rows; + int doc_link_row_count; + bool doc_links_failed; } cbm_pipeline_generation_t; /* Serialize and fully populate a sibling staging database, then atomically @@ -915,6 +941,15 @@ void cbm_pipeline_discard_stage(const char *stage_path); * Takes ownership; dump_and_persist_hashes writes them into the staging * store and cbm_pipeline_free releases them. Passing NULL/0 clears. */ void cbm_pipeline_set_lsp_surfaces(cbm_pipeline_t *p, cbm_lsp_surface_row_t *rows, int count); +/* The run's doc_link_unresolved rows and failure flag (doc_links.h), taken + * over by the pipeline (set replaces and frees earlier rows; NULL p frees). + * A full run publishes them from dump_and_persist_hashes; an incremental + * route takes them back for its carry-forward merge. `ran` stays false until + * a doc-link phase hands rows over, so a route that never resolved can tell. */ +void cbm_pipeline_set_doc_link_rows(cbm_pipeline_t *p, cbm_doc_link_row_t *rows, int count, + bool failed); +void cbm_pipeline_take_doc_link_rows(cbm_pipeline_t *p, cbm_doc_link_row_t **rows, int *count, + bool *failed, bool *ran); /* Pipeline accessors for incremental use */ const char *cbm_pipeline_repo_path(const cbm_pipeline_t *p); diff --git a/src/store/store.c b/src/store/store.c index a8a2ab21ce..7acb10e87f 100644 --- a/src/store/store.c +++ b/src/store/store.c @@ -77,6 +77,7 @@ enum { #include "foundation/platform.h" #include "foundation/compat.h" #include "foundation/log.h" +#include "foundation/mem_core.h" #include "foundation/compat_regex.h" #include "callable_sig.h" /* cbm_qn_callable_base_len: base-match tier */ #include "foundation/mem_core.h" /* cbm_alloc: pattern buffers */ @@ -2548,6 +2549,8 @@ int cbm_store_list_projects(cbm_store_t *s, cbm_project_t **out, int *count) { return CBM_STORE_OK; } +static int doc_links_table_state(cbm_store_t *s); + int cbm_store_delete_project(cbm_store_t *s, const char *name) { if (!s || !s->db || !name) { return CBM_STORE_ERR; @@ -2561,6 +2564,30 @@ int cbm_store_delete_project(cbm_store_t *s, const char *name) { "DELETE FROM index_coverage_meta WHERE project = ?1;", "DELETE FROM projects WHERE name = ?1 || '::missed';", }; + /* doc_link_unresolved exists only in databases published by a build that + * writes it; an older database has nothing to clean. */ + int doc_links_state = doc_links_table_state(s); + if (doc_links_state < 0) { + (void)exec_sql(s, "ROLLBACK;"); + return CBM_STORE_ERR; + } + if (doc_links_state > 0) { + sqlite3_stmt *dl = NULL; + if (sqlite3_prepare_v2(s->db, "DELETE FROM doc_link_unresolved WHERE project = ?1;", + CBM_NOT_FOUND, &dl, NULL) != SQLITE_OK) { + store_set_error_sqlite(s, "delete project doc_links prepare"); + (void)exec_sql(s, "ROLLBACK;"); + return CBM_STORE_ERR; + } + bind_text(dl, SKIP_ONE, name); + int dl_rc = sqlite3_step(dl); + sqlite3_finalize(dl); + if (dl_rc != SQLITE_DONE) { + store_set_error_sqlite(s, "delete project doc_links"); + (void)exec_sql(s, "ROLLBACK;"); + return CBM_STORE_ERR; + } + } for (size_t i = 0; i < sizeof(cleanup_sql) / sizeof(cleanup_sql[0]); i++) { sqlite3_stmt *cleanup = NULL; if (sqlite3_prepare_v2(s->db, cleanup_sql[i], CBM_NOT_FOUND, &cleanup, NULL) != SQLITE_OK) { @@ -4363,6 +4390,509 @@ void cbm_store_free_coverage(cbm_coverage_row_t *rows, int count) { free(rows); } +/* ── Doc-link unresolved references ─────────────────────────────── */ + +/* Created at publish, not in init_schema: a database written by an older + * build simply has no table, and readers treat that as "no doc-link data" + * instead of failing (no index-format change). */ +static const char DOC_LINKS_DDL[] = "CREATE TABLE IF NOT EXISTS doc_link_unresolved (" + " project TEXT NOT NULL," + " rel_path TEXT NOT NULL," + " line INTEGER NOT NULL DEFAULT 0," + " syntax TEXT NOT NULL DEFAULT ''," + " raw TEXT NOT NULL DEFAULT ''," + " reason TEXT NOT NULL" + ");" + "CREATE INDEX IF NOT EXISTS idx_doc_link_unresolved_path " + "ON doc_link_unresolved(project, rel_path);"; + +/* 1 = the table exists, 0 = it does not, -1 = the probe failed. */ +static int doc_links_table_state(cbm_store_t *s) { + sqlite3_stmt *stmt = NULL; + if (sqlite3_prepare_v2(s->db, + "SELECT 1 FROM sqlite_master WHERE type = 'table' " + "AND name = 'doc_link_unresolved';", + CBM_NOT_FOUND, &stmt, NULL) != SQLITE_OK) { + store_set_error_sqlite(s, "doc_links probe prepare"); + return CBM_NOT_FOUND; + } + int rc = sqlite3_step(stmt); + sqlite3_finalize(stmt); + if (rc == SQLITE_ROW) { + return SKIP_ONE; + } + if (rc == SQLITE_DONE) { + return 0; + } + store_set_error_sqlite(s, "doc_links probe"); + return CBM_NOT_FOUND; +} + +int cbm_store_doc_links_replace(cbm_store_t *s, const char *project, const cbm_doc_link_row_t *rows, + int count) { + if (!s || !s->db || !project || count < 0 || (count > 0 && !rows)) { + return CBM_STORE_ERR; + } + if (exec_sql(s, DOC_LINKS_DDL) != CBM_STORE_OK) { + return CBM_STORE_ERR; + } + if (exec_sql(s, "BEGIN;") != CBM_STORE_OK) { + return CBM_STORE_ERR; + } + sqlite3_stmt *del = NULL; + if (sqlite3_prepare_v2(s->db, "DELETE FROM doc_link_unresolved WHERE project = ?1;", + CBM_NOT_FOUND, &del, NULL) != SQLITE_OK) { + store_set_error_sqlite(s, "doc_links delete prepare"); + (void)exec_sql(s, "ROLLBACK;"); + return CBM_STORE_ERR; + } + bind_text(del, SKIP_ONE, project); + int rc = sqlite3_step(del); + sqlite3_finalize(del); + if (rc != SQLITE_DONE) { + store_set_error_sqlite(s, "doc_links delete"); + (void)exec_sql(s, "ROLLBACK;"); + return CBM_STORE_ERR; + } + sqlite3_stmt *ins = NULL; + if (sqlite3_prepare_v2(s->db, + "INSERT INTO doc_link_unresolved " + "(project, rel_path, line, syntax, raw, reason) " + "VALUES (?1, ?2, ?3, ?4, ?5, ?6);", + CBM_NOT_FOUND, &ins, NULL) != SQLITE_OK) { + store_set_error_sqlite(s, "doc_links insert prepare"); + (void)exec_sql(s, "ROLLBACK;"); + return CBM_STORE_ERR; + } + for (int i = 0; i < count; i++) { + if (!rows[i].rel_path || !rows[i].reason) { + continue; + } + bind_text(ins, SKIP_ONE, project); + bind_text(ins, ST_COL_2, rows[i].rel_path); + sqlite3_bind_int(ins, ST_COL_3, rows[i].line); + bind_text(ins, CBM_SZ_4, rows[i].syntax ? rows[i].syntax : ""); + bind_text(ins, CBM_SZ_5, rows[i].raw ? rows[i].raw : ""); + bind_text(ins, CBM_SZ_6, rows[i].reason); + if (sqlite3_step(ins) != SQLITE_DONE) { + store_set_error_sqlite(s, "doc_links insert"); + sqlite3_finalize(ins); + (void)exec_sql(s, "ROLLBACK;"); + return CBM_STORE_ERR; + } + sqlite3_reset(ins); + } + sqlite3_finalize(ins); + return exec_sql(s, "COMMIT;"); +} + +#ifdef CBM_ENABLE_TEST_SEAMS +static atomic_uint_fast64_t doc_links_sample_field_copies = ATOMIC_VAR_INIT(0); +static atomic_uint_fast64_t doc_links_sample_copied_bytes = ATOMIC_VAR_INIT(0); +static atomic_uint_fast64_t doc_links_sample_requested_bytes = ATOMIC_VAR_INIT(0); +static atomic_uint_fast64_t doc_links_sample_max_request_bytes = ATOMIC_VAR_INIT(0); +static atomic_int doc_links_sample_alloc_countdown = ATOMIC_VAR_INIT(CBM_NOT_FOUND); +static atomic_bool doc_links_sample_alloc_failed = ATOMIC_VAR_INIT(false); + +void cbm_store_doc_links_test_sample_stats_reset(void) { + atomic_store_explicit(&doc_links_sample_field_copies, 0, memory_order_relaxed); + atomic_store_explicit(&doc_links_sample_copied_bytes, 0, memory_order_relaxed); + atomic_store_explicit(&doc_links_sample_requested_bytes, 0, memory_order_relaxed); + atomic_store_explicit(&doc_links_sample_max_request_bytes, 0, memory_order_relaxed); +} + +void cbm_store_doc_links_test_sample_stats(cbm_doc_links_sample_test_stats_t *out) { + if (!out) { + return; + } + out->field_copies = atomic_load_explicit(&doc_links_sample_field_copies, memory_order_relaxed); + out->copied_bytes = atomic_load_explicit(&doc_links_sample_copied_bytes, memory_order_relaxed); + out->requested_bytes = + atomic_load_explicit(&doc_links_sample_requested_bytes, memory_order_relaxed); + out->max_request_bytes = + atomic_load_explicit(&doc_links_sample_max_request_bytes, memory_order_relaxed); +} + +void cbm_store_doc_links_test_fail_sample_alloc_after(int successful_copies) { + atomic_store_explicit(&doc_links_sample_alloc_failed, false, memory_order_relaxed); + atomic_store_explicit(&doc_links_sample_alloc_countdown, + successful_copies < 0 ? CBM_NOT_FOUND : successful_copies, + memory_order_relaxed); +} + +bool cbm_store_doc_links_test_sample_alloc_failed(void) { + return atomic_load_explicit(&doc_links_sample_alloc_failed, memory_order_relaxed); +} +#endif + +/* Copy the exact bytes requested by the sample query, including embedded NUL. + * Counters observe this allocation and copy, never the original column length. */ +static char *doc_links_sample_copy(const void *source, size_t length) { + if (length == SIZE_MAX || (length && !source)) { + return NULL; + } + size_t bytes = length + SKIP_ONE; +#ifdef CBM_ENABLE_TEST_SEAMS + atomic_fetch_add_explicit(&doc_links_sample_requested_bytes, bytes, memory_order_relaxed); + uint_fast64_t maximum = + atomic_load_explicit(&doc_links_sample_max_request_bytes, memory_order_relaxed); + while (maximum < bytes && !atomic_compare_exchange_weak_explicit( + &doc_links_sample_max_request_bytes, &maximum, bytes, + memory_order_relaxed, memory_order_relaxed)) {} + if (graph_compare_test_countdown_fires(&doc_links_sample_alloc_countdown)) { + atomic_store_explicit(&doc_links_sample_alloc_failed, true, memory_order_relaxed); + return NULL; + } +#endif + char *copy = cbm_alloc(CBM_MEM_CLASS_STORE, bytes); + if (copy) { + if (length) { + memcpy(copy, source, length); + } + copy[length] = '\0'; +#ifdef CBM_ENABLE_TEST_SEAMS + atomic_fetch_add_explicit(&doc_links_sample_field_copies, 1, memory_order_relaxed); + atomic_fetch_add_explicit(&doc_links_sample_copied_bytes, bytes, memory_order_relaxed); +#endif + } + return copy; +} + +/* Full getters and the original summary API retain their C-string behavior. */ +static char *doc_links_field_copy(const char *source, bool sample) { + if (!sample || !source) { + return cbm_mem_strdup(CBM_MEM_CLASS_STORE, source); + } + return doc_links_sample_copy(source, strlen(source)); +} + +/* Run a row query (columns rel_path, line, syntax, raw, reason) with the + * project bound to ?1 and an optional integer limit bound to ?2. */ +static int doc_links_query(cbm_store_t *s, const char *sql, const char *project, int limit, + cbm_doc_link_row_t **out, int *count) { + sqlite3_stmt *stmt = NULL; + if (sqlite3_prepare_v2(s->db, sql, CBM_NOT_FOUND, &stmt, NULL) != SQLITE_OK) { + store_set_error_sqlite(s, "doc_links get prepare"); + return CBM_STORE_ERR; + } + bind_text(stmt, SKIP_ONE, project); + if (limit >= 0) { + sqlite3_bind_int(stmt, ST_COL_2, limit); + } + int cap = ST_INIT_CAP_16; + int n = 0; + cbm_doc_link_row_t *arr = cbm_alloc(CBM_MEM_CLASS_STORE, (size_t)cap * sizeof(*arr)); + if (!arr) { + sqlite3_finalize(stmt); + return CBM_STORE_ERR; + } + int scan_rc; + while ((scan_rc = sqlite3_step(stmt)) == SQLITE_ROW) { + if (n >= cap) { + cbm_doc_link_row_t *grown = + cbm_realloc(CBM_MEM_CLASS_STORE, arr, (size_t)cap * ST_GROWTH * sizeof(*arr)); + if (!grown) { + sqlite3_finalize(stmt); + cbm_store_free_doc_links(arr, n); + return CBM_STORE_ERR; + } + arr = grown; + cap *= ST_GROWTH; + } + cbm_doc_link_row_t *r = &arr[n]; + r->rel_path = doc_links_field_copy((const char *)sqlite3_column_text(stmt, 0), limit >= 0); + r->line = sqlite3_column_int(stmt, SKIP_ONE); + r->syntax = + doc_links_field_copy((const char *)sqlite3_column_text(stmt, ST_COL_2), limit >= 0); + r->raw = + doc_links_field_copy((const char *)sqlite3_column_text(stmt, ST_COL_3), limit >= 0); + r->reason = + doc_links_field_copy((const char *)sqlite3_column_text(stmt, CBM_SZ_4), limit >= 0); + n++; + if (!r->rel_path || !r->syntax || !r->raw || !r->reason) { + sqlite3_finalize(stmt); + cbm_store_free_doc_links(arr, n); + return CBM_STORE_ERR; + } + } + sqlite3_finalize(stmt); + if (scan_rc != SQLITE_DONE) { + store_set_error_sqlite(s, "doc_links scan"); + cbm_store_free_doc_links(arr, n); + return CBM_STORE_ERR; + } + *out = arr; + *count = n; + return CBM_STORE_OK; +} + +int cbm_store_doc_links_get(cbm_store_t *s, const char *project, cbm_doc_link_row_t **out, + int *count, bool *table_present) { + if (!out || !count) { + return CBM_STORE_ERR; + } + *out = NULL; + *count = 0; + if (table_present) { + *table_present = false; + } + if (!s || !s->db || !project) { + return CBM_STORE_ERR; + } + int state = doc_links_table_state(s); + if (state < 0) { + return CBM_STORE_ERR; + } + if (state == 0) { + return CBM_STORE_OK; + } + if (table_present) { + *table_present = true; + } + return doc_links_query(s, + "SELECT rel_path, line, syntax, raw, reason FROM doc_link_unresolved " + "WHERE project = ?1 ORDER BY rel_path, line, raw, syntax, reason;", + project, CBM_NOT_FOUND, out, count); +} + +int cbm_store_doc_links_summary(cbm_store_t *s, const char *project, + cbm_doc_link_reason_count_t **reasons, int *reason_count, + cbm_doc_link_row_t **samples, int *sample_count, int sample_limit, + bool *table_present) { + if (!reasons || !reason_count || !samples || !sample_count) { + return CBM_STORE_ERR; + } + *reasons = NULL; + *reason_count = 0; + *samples = NULL; + *sample_count = 0; + if (table_present) { + *table_present = false; + } + if (!s || !s->db || !project) { + return CBM_STORE_ERR; + } + int state = doc_links_table_state(s); + if (state < 0) { + return CBM_STORE_ERR; + } + if (state == 0) { + return CBM_STORE_OK; + } + if (table_present) { + *table_present = true; + } + sqlite3_stmt *stmt = NULL; + if (sqlite3_prepare_v2(s->db, + "SELECT reason, COUNT(*) FROM doc_link_unresolved WHERE project = ?1 " + "GROUP BY reason ORDER BY reason;", + CBM_NOT_FOUND, &stmt, NULL) != SQLITE_OK) { + store_set_error_sqlite(s, "doc_links summary prepare"); + return CBM_STORE_ERR; + } + bind_text(stmt, SKIP_ONE, project); + int cap = ST_INIT_CAP_8; + int n = 0; + cbm_doc_link_reason_count_t *arr = cbm_alloc(CBM_MEM_CLASS_STORE, (size_t)cap * sizeof(*arr)); + if (!arr) { + sqlite3_finalize(stmt); + return CBM_STORE_ERR; + } + int scan_rc; + while ((scan_rc = sqlite3_step(stmt)) == SQLITE_ROW) { + if (n >= cap) { + cbm_doc_link_reason_count_t *grown = + cbm_realloc(CBM_MEM_CLASS_STORE, arr, (size_t)cap * ST_GROWTH * sizeof(*arr)); + if (!grown) { + sqlite3_finalize(stmt); + cbm_store_free_doc_link_reasons(arr, n); + return CBM_STORE_ERR; + } + arr = grown; + cap *= ST_GROWTH; + } + arr[n].reason = + cbm_mem_strdup(CBM_MEM_CLASS_STORE, (const char *)sqlite3_column_text(stmt, 0)); + arr[n].count = sqlite3_column_int(stmt, SKIP_ONE); + n++; + if (!arr[n - SKIP_ONE].reason) { + sqlite3_finalize(stmt); + cbm_store_free_doc_link_reasons(arr, n); + return CBM_STORE_ERR; + } + } + sqlite3_finalize(stmt); + if (scan_rc != SQLITE_DONE) { + store_set_error_sqlite(s, "doc_links summary scan"); + cbm_store_free_doc_link_reasons(arr, n); + return CBM_STORE_ERR; + } + if (sample_limit > 0 && + doc_links_query(s, + "SELECT rel_path, line, syntax, raw, reason FROM doc_link_unresolved " + "WHERE project = ?1 ORDER BY reason, rel_path, line, raw LIMIT ?2;", + project, sample_limit, samples, sample_count) != CBM_STORE_OK) { + cbm_store_free_doc_link_reasons(arr, n); + return CBM_STORE_ERR; + } + *reasons = arr; + *reason_count = n; + return CBM_STORE_OK; +} + +/* Both substr and length operate on bytes; ORDER BY retains the original + * columns, so projection cannot change which fifty source rows are sampled. */ +static bool doc_links_preview_field(sqlite3_stmt *stmt, int column, size_t cap, + cbm_doc_link_preview_text_t *out) { + /* SQLite's BLOB substr returns SQL NULL for a zero-byte BLOB. Its + * separately computed integer length distinguishes that from a NULL + * source or a failed nonempty projection. */ + if (sqlite3_column_type(stmt, column + SKIP_ONE) != SQLITE_INTEGER) { + return false; + } + sqlite3_int64 original = sqlite3_column_int64(stmt, column + SKIP_ONE); + int projection_type = sqlite3_column_type(stmt, column); + if (original < 0 || + (projection_type != SQLITE_BLOB && (projection_type != SQLITE_NULL || original != 0))) { + return false; + } + const void *source = sqlite3_column_blob(stmt, column); + int bytes = sqlite3_column_bytes(stmt, column); + if (bytes < 0 || (size_t)bytes > cap || (bytes && !source)) { + return false; + } + uint64_t expected = (uint64_t)original < cap ? (uint64_t)original : cap; + if ((uint64_t)bytes != expected) { + return false; + } + out->text = doc_links_sample_copy(source, (size_t)bytes); + if (!out->text) { + return false; + } + out->length = (size_t)bytes; + out->original_bytes = (uint64_t)original; + return true; +} + +int cbm_store_doc_links_preview(cbm_store_t *s, const char *project, + cbm_doc_link_preview_row_t **out, int *count, bool *table_present) { + if (!out || !count) { + return CBM_STORE_ERR; + } + *out = NULL; + *count = 0; + if (table_present) { + *table_present = false; + } + if (!s || !s->db || !project) { + return CBM_STORE_ERR; + } + int state = doc_links_table_state(s); + if (state < 0) { + return CBM_STORE_ERR; + } + if (!state) { + return CBM_STORE_OK; + } + if (table_present) { + *table_present = true; + } + sqlite3_stmt *stmt = NULL; + static const char sql[] = + "SELECT substr(CAST(d.rel_path AS BLOB),1,?2),length(CAST(d.rel_path AS BLOB))," + "d.line,substr(CAST(d.syntax AS BLOB),1,?3),length(CAST(d.syntax AS BLOB))," + "substr(CAST(d.raw AS BLOB),1,?2),length(CAST(d.raw AS BLOB))," + "substr(CAST(d.reason AS BLOB),1,?3),length(CAST(d.reason AS BLOB)) " + "FROM doc_link_unresolved AS d WHERE d.project=?1 " + "ORDER BY d.reason,d.rel_path,d.line,d.raw LIMIT ?4;"; + if (sqlite3_prepare_v2(s->db, sql, CBM_NOT_FOUND, &stmt, NULL) != SQLITE_OK) { + store_set_error_sqlite(s, "doc_links preview prepare"); + return CBM_STORE_ERR; + } + const int long_cap = CBM_DOC_LINK_PREVIEW_LONG_BYTES + CBM_DOC_LINK_PREVIEW_LOOKAHEAD; + const int short_cap = CBM_DOC_LINK_PREVIEW_SHORT_BYTES + CBM_DOC_LINK_PREVIEW_LOOKAHEAD; + if (bind_text(stmt, SKIP_ONE, project) != SQLITE_OK || + sqlite3_bind_int(stmt, ST_COL_2, long_cap) != SQLITE_OK || + sqlite3_bind_int(stmt, ST_COL_3, short_cap) != SQLITE_OK || + sqlite3_bind_int(stmt, ST_COL_4, CBM_DOC_LINK_PREVIEW_ROWS) != SQLITE_OK) { + store_set_error_sqlite(s, "doc_links preview bind"); + sqlite3_finalize(stmt); + return CBM_STORE_ERR; + } + cbm_doc_link_preview_row_t *rows = + cbm_calloc(CBM_MEM_CLASS_STORE, CBM_DOC_LINK_PREVIEW_ROWS * sizeof(*rows)); + if (!rows) { + sqlite3_finalize(stmt); + return CBM_STORE_ERR; + } + int n = 0; + int rc; + while ((rc = sqlite3_step(stmt)) == SQLITE_ROW) { + if (n == CBM_DOC_LINK_PREVIEW_ROWS) { + store_set_error(s, "doc_links preview row limit"); + sqlite3_finalize(stmt); + cbm_store_free_doc_link_previews(rows, n); + return CBM_STORE_ERR; + } + cbm_doc_link_preview_row_t *row = &rows[n++]; + row->line = sqlite3_column_int(stmt, ST_COL_2); + if (!doc_links_preview_field(stmt, 0, (size_t)long_cap, &row->rel_path) || + !doc_links_preview_field(stmt, ST_COL_3, (size_t)short_cap, &row->syntax) || + !doc_links_preview_field(stmt, ST_COL_5, (size_t)long_cap, &row->raw) || + !doc_links_preview_field(stmt, ST_COL_7, (size_t)short_cap, &row->reason)) { + store_set_error(s, "doc_links preview copy failed"); + sqlite3_finalize(stmt); + cbm_store_free_doc_link_previews(rows, n); + return CBM_STORE_ERR; + } + } + sqlite3_finalize(stmt); + if (rc != SQLITE_DONE) { + store_set_error_sqlite(s, "doc_links preview scan"); + cbm_store_free_doc_link_previews(rows, n); + return CBM_STORE_ERR; + } + *out = rows; + *count = n; + return CBM_STORE_OK; +} + +void cbm_store_free_doc_link_previews(cbm_doc_link_preview_row_t *rows, int count) { + if (!rows) { + return; + } + for (int i = 0; i < count; i++) { + cbm_free(CBM_MEM_CLASS_STORE, (char *)rows[i].rel_path.text); + cbm_free(CBM_MEM_CLASS_STORE, (char *)rows[i].syntax.text); + cbm_free(CBM_MEM_CLASS_STORE, (char *)rows[i].raw.text); + cbm_free(CBM_MEM_CLASS_STORE, (char *)rows[i].reason.text); + } + cbm_free(CBM_MEM_CLASS_STORE, rows); +} + +void cbm_store_free_doc_links(cbm_doc_link_row_t *rows, int count) { + if (!rows) { + return; + } + for (int i = 0; i < count; i++) { + cbm_free(CBM_MEM_CLASS_STORE, (char *)rows[i].rel_path); + cbm_free(CBM_MEM_CLASS_STORE, (char *)rows[i].syntax); + cbm_free(CBM_MEM_CLASS_STORE, (char *)rows[i].raw); + cbm_free(CBM_MEM_CLASS_STORE, (char *)rows[i].reason); + } + cbm_free(CBM_MEM_CLASS_STORE, rows); +} + +void cbm_store_free_doc_link_reasons(cbm_doc_link_reason_count_t *reasons, int count) { + if (!reasons) { + return; + } + for (int i = 0; i < count; i++) { + cbm_free(CBM_MEM_CLASS_STORE, (char *)reasons[i].reason); + } + cbm_free(CBM_MEM_CLASS_STORE, reasons); +} + /* ── FindNodesByFileOverlap ─────────────────────────────────────── */ int cbm_store_find_nodes_by_file_overlap(cbm_store_t *s, const char *project, const char *file_path, diff --git a/src/store/store.h b/src/store/store.h index 916fa3a655..61b5551775 100644 --- a/src/store/store.h +++ b/src/store/store.h @@ -717,6 +717,99 @@ void cbm_store_coverage_shadow_project(char *dst, size_t dstsz, const char *proj void cbm_store_free_coverage(cbm_coverage_row_t *rows, int count); +/* ── Doc-link unresolved references ─────────────────────────────── */ + +/* One doc-comment reference that did not become a MENTIONS edge, stored in + * doc_link_unresolved (project, rel_path, line, syntax, raw, reason). The + * table is created at publish when missing, so databases written by older + * builds stay readable without an index-format change. `reason` is one of + * missing, ambiguous, external, test_only_target, not_indexed, graph_gap, + * unparseable, below_bar_tier (it resolved, but its link family does not + * ship); a row with rel_path "" and reason "error" records that the + * doc-link layer itself failed for the generation. Rows returned by the + * getters own their strings; rows and strings are memory-core blocks of + * CBM_MEM_CLASS_STORE, released only by cbm_store_free_doc_links. */ +typedef struct { + const char *rel_path; + int line; + const char *syntax; + const char *raw; + const char *reason; +} cbm_doc_link_row_t; + +/* Replace the project's rows in one transaction, creating the table first + * when it does not exist. */ +int cbm_store_doc_links_replace(cbm_store_t *s, const char *project, const cbm_doc_link_row_t *rows, + int count); + +/* All rows of the project ordered by (rel_path, line, raw). *table_present + * (optional) is false — with zero rows and CBM_STORE_OK — for a database that + * has no doc_link_unresolved table yet. */ +int cbm_store_doc_links_get(cbm_store_t *s, const char *project, cbm_doc_link_row_t **out, + int *count, bool *table_present); + +/* Row count per reason (ordered by reason; reasons owned by the result) and + * up to `sample_limit` rows ordered by (reason, rel_path, line). Same + * table_present contract as above. */ +typedef struct { + const char *reason; + int count; +} cbm_doc_link_reason_count_t; +int cbm_store_doc_links_summary(cbm_store_t *s, const char *project, + cbm_doc_link_reason_count_t **reasons, int *reason_count, + cbm_doc_link_row_t **samples, int *sample_count, int sample_limit, + bool *table_present); + +/* Diagnostic display budgets; the query copies three extra bytes per field + * so a formatter can inspect a complete UTF-8 scalar at the display boundary. */ +enum { + CBM_DOC_LINK_PREVIEW_ROWS = 50, + CBM_DOC_LINK_PREVIEW_LONG_BYTES = 1024, + CBM_DOC_LINK_PREVIEW_SHORT_BYTES = 128, + CBM_DOC_LINK_PREVIEW_LOOKAHEAD = 3, +}; +typedef struct { + const char *text; + size_t length; /* copied prefix bytes, excluding the added NUL */ + uint64_t original_bytes; +} cbm_doc_link_preview_text_t; +typedef struct { + cbm_doc_link_preview_text_t rel_path; + int line; + cbm_doc_link_preview_text_t syntax; + cbm_doc_link_preview_text_t raw; + cbm_doc_link_preview_text_t reason; +} cbm_doc_link_preview_row_t; +/* Up to fifty rows ordered by original (reason, rel_path, line, raw), before + * empty-path marker filtering. Text is an explicit-length byte prefix and may + * contain NUL or invalid UTF-8. Missing-table semantics match the full getter. + * On failure, no rows are returned. All allocations belong to STORE. */ +int cbm_store_doc_links_preview(cbm_store_t *s, const char *project, + cbm_doc_link_preview_row_t **out, int *count, bool *table_present); +void cbm_store_free_doc_link_previews(cbm_doc_link_preview_row_t *rows, int count); + +#ifdef CBM_ENABLE_TEST_SEAMS +/* Sample-field copies only: exclude full getters, row arrays and reason keys. + * Byte counts include the trailing NUL. Requests include failed allocations; + * field_copies/copied_bytes count only successful copies. Reset leaves fault + * controls and their consumed flag unchanged. */ +typedef struct { + uint64_t field_copies; + uint64_t copied_bytes; + uint64_t requested_bytes; + uint64_t max_request_bytes; +} cbm_doc_links_sample_test_stats_t; +void cbm_store_doc_links_test_sample_stats_reset(void); +void cbm_store_doc_links_test_sample_stats(cbm_doc_links_sample_test_stats_t *out); +/* One-shot field-copy allocation failure: 0 is next, 4 is fifth, -1 disables. + * Setting the control clears its consumed flag. */ +void cbm_store_doc_links_test_fail_sample_alloc_after(int successful_copies); +bool cbm_store_doc_links_test_sample_alloc_failed(void); +#endif + +void cbm_store_free_doc_links(cbm_doc_link_row_t *rows, int count); +void cbm_store_free_doc_link_reasons(cbm_doc_link_reason_count_t *reasons, int count); + /* ── Search ─────────────────────────────────────────────────────── */ int cbm_store_search(cbm_store_t *s, const cbm_search_params_t *params, cbm_search_output_t *out); diff --git a/tests/test_doc_mentions.c b/tests/test_doc_mentions.c new file mode 100644 index 0000000000..25ddd9783a --- /dev/null +++ b/tests/test_doc_mentions.c @@ -0,0 +1,9206 @@ +/* + * test_doc_mentions.c — doc-comment references -> MENTIONS edges. + * + * Extraction (doclink.c / doclink_cs.c), the C# resolver (doc_links_cs.c), + * MSBuild global usings, publication into doc_link_unresolved, the + * index_status block, delete_project, and incremental == full across edits. + * Every pipeline test indexes a real fixture through cbm_pipeline_run and + * reads the published database (helpers: test_doc_mentions_helpers.h). + */ +#include "../src/foundation/compat.h" +#include "test_framework.h" +#include "test_helpers.h" +#include "test_doc_mentions_helpers.h" + +#include "cbm.h" +#include "doclink.h" +#include "lang_specs.h" +#include "foundation/compat_thread.h" +#include "foundation/mem_core.h" +#include "mcp/mcp.h" +#include "mcp/mcp_internal.h" +#include "pipeline/doc_links.h" +#include "pipeline/doc_links_msbuild.h" +#include "pipeline/lsp_surface.h" +#include "pipeline/pipeline.h" +#include "pipeline/pipeline_internal.h" +#include "store/store.h" +#include "sqlite3.h" + +#include +#include +#include +#ifndef _WIN32 +#include /* mkfifo */ +#include /* symlink */ +#endif + +/* ── extraction ──────────────────────────────────────────────────── */ + +TEST(doc_mentions_extract_cs_tokens) { + const char *src = + "namespace N\n" /* 1 */ + "{\n" /* 2 */ + " /// A and\n" /* 3 */ + " /// , not \n" /* 4 */ + " /// or or .\n" /* 5 */ + " /// \n" /* 6 */ + " /// \n" /* 7 */ + " /// bad\n" /* 8 */ + " /// \n" /* 10 */ + " public class C\n" /* 11 */ + " {\n" + " /// Field .\n" + " public const int K = 1;\n" + " }\n" + "}\n"; + CBMFileResult *r = + cbm_extract_file(src, (int)strlen(src), CBM_LANG_CSHARP, "p", "C.cs", 0, NULL, NULL); + ASSERT_NOT_NULL(r); + const CBMDocLink *foo = dm_find_token(r, "Foo"); + ASSERT_NOT_NULL(foo); + ASSERT_EQ(foo->line, 3); + ASSERT_EQ(foo->syntax, CBM_DOCLINK_CS_SEE); + ASSERT_STR_EQ(foo->source_qn, "p.C.C"); + ASSERT_EQ(foo->def_line, 11); + const CBMDocLink *baz = dm_find_token(r, "Bar.Baz(int)"); + ASSERT_NOT_NULL(baz); + ASSERT_EQ(baz->line, 4); + ASSERT_EQ(baz->syntax, CBM_DOCLINK_CS_SEEALSO); + /* local parameter references and keywords are not references */ + ASSERT_NULL(dm_find_token(r, "x")); + ASSERT_NULL(dm_find_token(r, "T")); + ASSERT_NULL(dm_find_token(r, "null")); + const CBMDocLink *href = dm_find_token(r, "https://x.org"); + ASSERT_NOT_NULL(href); + ASSERT_EQ(href->syntax, CBM_DOCLINK_HREF); + ASSERT_EQ(href->line, 6); + /* XML entities are decoded */ + ASSERT_NOT_NULL(dm_find_token(r, "List")); + const CBMDocLink *ex = dm_find_token(r, "Oops"); + ASSERT_NOT_NULL(ex); + ASSERT_EQ(ex->syntax, CBM_DOCLINK_CS_EXCEPTION); + ASSERT_EQ(ex->line, 8); + const CBMDocLink *inh = dm_find_token(r, "Base.M"); + ASSERT_NOT_NULL(inh); + ASSERT_EQ(inh->syntax, CBM_DOCLINK_CS_INHERITDOC); + /* a tag spanning two comment lines */ + const CBMDocLink *wrapped = dm_find_token(r, "Wrapped.Name"); + ASSERT_NOT_NULL(wrapped); + ASSERT_EQ(wrapped->line, 9); + /* the constant's Field and its Variable twin carry the doc once */ + ASSERT_EQ(dm_count_tokens(r, "Foo"), 2); /* the class's and the constant's */ + cbm_free_result(r); + PASS(); +} + +TEST(doc_mentions_cs_scope_blob) { + const char *src = "global using Acme.G; global using static Acme.GS;\n" /* 1 */ + "using static Acme.S;\n" /* 2 */ + "using A = Acme.Util.Helper;\n" /* 3 */ + "namespace Outer.Inner\n" /* 4 */ + "{\n" /* 5 */ + " using Acme.Local;\n" /* 6 */ + " public partial class W : Base, IThing\n" /* 7 */ + " {\n" /* 8 */ + " void IThing.Do(int x) { }\n" /* 9 */ + " public void Go(ref string s, params int[] rest) { }\n" /* 10 */ + " public event System.EventHandler Fired;\n" /* 11 */ + " public int P { get; set; }\n" /* 12 */ + " public record R(int Width);\n" /* 13 */ + " public delegate void D();\n" /* 14 */ + " public enum E { One }\n" /* 15 */ + " public static W operator +(W a, int b) => a;\n" /* 16 */ + " public static implicit operator int(W w) => 0;\n" /* 17 */ + " public int this[int i] => i;\n" /* 18 */ + " public static int Count(string s) => 0;\n" /* 19 */ + " public static int Ext(this string s) => 0;\n" /* 20 */ + " public const int Max = 1;\n" /* 21 */ + " public int field;\n" /* 22 */ + " static W() { }\n" /* 23 */ + " public W(T first) { }\n" /* 24 */ + " public readonly record struct RS(int X);\n" /* 25 */ + " }\n" /* 26 */ + "}\n"; /* 27 */ + CBMFileResult *r = + cbm_extract_file(src, (int)strlen(src), CBM_LANG_CSHARP, "p", "W.cs", 0, NULL, NULL); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + const char *s = r->doc_scope; + ASSERT(strncmp(s, "cs1\n", 4) == 0); + /* a using says what it brings in (n namespace, s static, a alias), and + * `g` when it is a `global using`: of whatever kind */ + ASSERT_NOT_NULL(strstr(s, "U\t0\tng\t-\tAcme.G\n")); + ASSERT_NOT_NULL(strstr(s, "U\t0\ts\t-\tAcme.S\n")); + ASSERT_NOT_NULL(strstr(s, "U\t0\ta\tA\tAcme.Util.Helper\n")); + ASSERT_NOT_NULL(strstr(s, "U\t0\tsg\t-\tAcme.GS\n")); + /* a region carries its own name and the region it stands in */ + ASSERT_NOT_NULL(strstr(s, "R\t1\t0\t4\t27\tOuter.Inner\n")); + ASSERT_NOT_NULL(strstr(s, "U\t1\tn\t-\tAcme.Local\n")); /* a namespace-block using */ + /* a type: `p` for partial, `-` for no outer type; a nested one names its + * outer type by ordinal, a member its type */ + ASSERT_NOT_NULL(strstr(s, "T\t1\t7\t26\tcp\t-\tW\tT,U\tBase|IThing\n")); + ASSERT_NOT_NULL(strstr(s, "M\t9\tc\t1\t0\tDo\t\tint\n")); /* explicit implementation */ + ASSERT_NOT_NULL(strstr(s, "M\t10\tc\t0\t0\tGo\t\tstring|int[]\n")); + ASSERT_NOT_NULL(strstr(s, "M\t11\te\t0\t0\tFired\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t12\tp\t0\t0\tP\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t13\t13\tr\t0\tR\t\t\n")); + ASSERT_NOT_NULL(strstr(s, "M\t13\tc\t0\t1\tR\t\tint\n")); /* its primary constructor */ + ASSERT_NOT_NULL(strstr(s, "M\t13\tp\t0\t1\tWidth\t\t-\n")); /* positional record property */ + ASSERT_NOT_NULL(strstr(s, "T\t1\t14\t14\td\t0\tD\t\t\n")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t15\t15\te\t0\tE\t\t\n")); + ASSERT_NOT_NULL(strstr(s, "M\t15\tvs\t0\t3\tOne\t\t-\n")); /* an enum's member is static */ + /* operators, conversions and indexers are declared, under their token */ + ASSERT_NOT_NULL(strstr(s, "M\t16\to\t0\t0\t+\t\tW|int\n")); + ASSERT_NOT_NULL(strstr(s, "M\t17\to\t0\t0\timplicit\t\tW\n")); + ASSERT_NOT_NULL(strstr(s, "M\t18\tx\t0\t0\tthis\t\tint\n")); + /* `s`: what a `using static` brings in -- static, and no extension method */ + ASSERT_NOT_NULL(strstr(s, "M\t19\tcs\t0\t0\tCount\t\tstring\n")); + ASSERT_NOT_NULL(strstr(s, "M\t20\tc\t0\t0\tExt\t\tstring\n")); + ASSERT_NOT_NULL(strstr(s, "M\t21\tvs\t0\t0\tMax\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t22\tv\t0\t0\tfield\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t23\tcs\t0\t0\tW\t\t\n")); /* the static constructor */ + ASSERT_NOT_NULL(strstr(s, "M\t24\tc\t0\t0\tW\t\tT\n")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t25\t25\tt\t0\tRS\t\t\n")); /* a record struct */ + /* the persisted scope drops every line number, nothing else */ + char *portable = cbm_doclink_portable_scope(s); + ASSERT_NOT_NULL(portable); + ASSERT_NOT_NULL(strstr(portable, "R\t1\t0\t0\t0\tOuter.Inner\n")); + ASSERT_NOT_NULL(strstr(portable, "T\t1\t0\t0\tcp\t-\tW\tT,U\tBase|IThing\n")); + ASSERT_NOT_NULL(strstr(portable, "M\t0\tc\t0\t0\tGo\t\tstring|int[]\n")); + cbm_free(CBM_MEM_CLASS_OTHER, portable); + cbm_free_result(r); + PASS(); +} + +/* Every name in a field/event declaration shares its static/const modifiers. */ +TEST(doc_mentions_cs_declarator_modifiers) { + const char *src = "class C {\n" + "[System.Obsolete] public int a, b;\n" + "[System.Obsolete] public static int c, d;\n" + "[System.Obsolete] public const int e = 1, f = 2;\n" + "[System.Obsolete] public event System.Action G, H;\n" + "[System.Obsolete] public static event System.Action I, J;\n" + "}\n"; + CBMFileResult *r = dm_extract(src, CBM_LANG_CSHARP, "Fields.cs"); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + const char *s = r->doc_scope; + ASSERT_NOT_NULL(strstr(s, "M\t2\tv\t0\t0\ta\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t2\tv\t0\t0\tb\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t3\tvs\t0\t0\tc\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t3\tvs\t0\t0\td\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t4\tvs\t0\t0\te\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t4\tvs\t0\t0\tf\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t5\te\t0\t0\tG\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t5\te\t0\t0\tH\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t6\tes\t0\t0\tI\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t6\tes\t0\t0\tJ\t\t-\n")); + cbm_free_result(r); + PASS(); +} + +/* The scope blob of one C# source; the result owns it. */ +static CBMFileResult *dm_scope(const char *src) { + return dm_extract(src, CBM_LANG_CSHARP, "S.cs"); +} + +/* Controlled parser/scanner observation inputs. No fixture is executed as C#. + * Each name is exercised in both block and file-scoped declarations. */ +static const struct { + const char *id; + const char *name; + bool valid; + bool recovery; +} dm_namespace_cases[] = { + {"leading_dot", ".Bad", false, false}, + {"interior_empty", "Bad..Inner", false, false}, + {"trailing_dot", "Bad.", false, false}, + {"dot_only", ".", false, false}, + {"spaced_empty", "Bad. .Inner", false, false}, + {"verbatim_empty", "@Bad..@Inner", false, false}, + {"valid_dotted", "Good.Inner", true, false}, + {"valid_verbatim", "@Good.@class", true, false}, + {"valid_unicode", "Gr\xC3\xBCne.Inner", true, false}, + {"valid_recovery", "Good.Recovered", true, true}, + {"valid_prefix_comment", "/* note */ Good.Inner", true, false}, + {"valid_segment_comments", "Good /* note */ . /* note */ Inner", true, false}, + {"comment_empty", "Bad. /* note */ .Inner", false, false}, + {"trailing_recovery", "Bad.", false, true}, + {"verbatim_recovery", "@Good.@class", true, true}, +}; + +static bool dm_namespace_source(char *out, size_t cap, size_t index, bool file_scoped) { + const char *recovery = dm_namespace_cases[index].recovery + ? "public class ParseEdge { unsafe void M(void* p) { " + "_r = ref *(int*)p; } }\n" + : ""; + int n = snprintf(out, cap, + "namespace %s%s" + "/// \n" + "public class FromBad { }\n" + "public class Hidden { }\n" + "%s%s", + dm_namespace_cases[index].name, file_scoped ? ";\n" : " {\n", recovery, + file_scoped ? "" : "}\n"); + return n >= 0 && (size_t)n < cap; +} + +/* Real parser inputs, including both AST and lexical recovery paths. Valid + * controls prevent treating every file with a parse error as unplaceable. */ +TEST(doc_mentions_cs_namespace_scopes) { + static const char *normalized[] = {NULL, + NULL, + NULL, + NULL, + NULL, + NULL, + "Good.Inner", + "Good.class", + "Gr\xC3\xBCne.Inner", + "Good.Recovered", + "Good.Inner", + "Good.Inner", + NULL, + NULL, + "Good.class"}; + bool correct = true; + for (size_t i = 0; i < sizeof(dm_namespace_cases) / sizeof(dm_namespace_cases[0]); i++) { + for (int file_scoped = 0; file_scoped < 2; file_scoped++) { + char source[2048]; + bool made = dm_namespace_source(source, sizeof(source), i, file_scoped != 0); + CBMFileResult *result = made ? dm_extract(source, CBM_LANG_CSHARP, "Bad.cs") : NULL; + const char *scope = result ? result->doc_scope : NULL; + int tokens = result ? dm_count_tokens(result, "global::Good.Target") : -1; + bool valid = dm_namespace_cases[i].valid; + bool shape = false; + if (scope && valid) { + char name[128]; + snprintf(name, sizeof(name), "\t%s\n", normalized[i]); + shape = strstr(scope, "\nR\t1\t0\t") && strstr(scope, name) && + strstr(scope, "\nT\t1\t3\t3\tc\t-\tFromBad\t\t\n") && + strstr(scope, "\nT\t1\t4\t4\tc\t-\tHidden\t\t\n") && + !strstr(scope, "\nX\t") && !strstr(scope, "\nQ\t"); + } else if (scope) { + shape = strstr(scope, "\nX\t") && strstr(scope, "\nQ\tFromBad\n") && + strstr(scope, "\nQ\tHidden\n") && !strstr(scope, "\nR\t") && + !strstr(scope, "\nT\t"); + } + bool one = made && scope && tokens == 1 && shape; + fprintf(stderr, + "doc namespace scope case=%s file_scoped=%d valid=%d tokens=%d shape=%d " + "correct=%d\n", + dm_namespace_cases[i].id, file_scoped, valid, tokens, shape, one); + correct = one && correct; + if (result) { + cbm_free_result(result); + } + } + } + ASSERT(correct); + PASS(); +} + +/* Shared real source fixture: a malformed namespace must not erase the + * independent Target link or permit Hidden to bind around an unknown declaration. */ +static const char *dm_namespace_healthy_source = "namespace Good {\n" + "public class Target { }\n" + "public class Hidden { }\n" + "/// \n" + "public class Uses { }\n" + "/// \n" + "public class TriesHidden { }\n" + "}\n"; + +/* Check the persisted layer and the public status. No assertion here may skip + * the caller's environment restoration or fixture cleanup. */ +static bool dm_namespace_published(const char *db, const char *project, bool valid, + const char *label) { + char props[512], inside_reason[64], conflict_reason[64]; + int healthy = 0, inside = 0, conflict = 0; + dm_edge(db, "Healthy.Uses", "Healthy.Target", props, sizeof(props), &healthy); + dm_edge(db, "Bad.FromBad", "Healthy.Target", props, sizeof(props), &inside); + dm_edge(db, "Healthy.TriesHidden", "Healthy.Hidden", props, sizeof(props), &conflict); + dm_row(db, "Bad.cs", "global::Good.Target", inside_reason, sizeof(inside_reason), NULL, 0); + dm_row(db, "Healthy.cs", "Hidden", conflict_reason, sizeof(conflict_reason), NULL, 0); + int errors = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'"); + char *status = dm_index_status(project, false); + const char *block = status ? strstr(status, "doc_links:\n") : NULL; + bool status_ok = block && strstr(block, "\n status: ok"); + int from_bad = dm_mentions_from(db, "Bad.FromBad"); + bool inside_ok = valid + ? inside == 1 && from_bad == 1 && !inside_reason[0] + : inside == 0 && from_bad == 0 && strcmp(inside_reason, "graph_gap") == 0; + bool conflict_ok = valid ? conflict == 1 && !conflict_reason[0] + : conflict == 0 && strcmp(conflict_reason, "graph_gap") == 0; + bool correct = errors == 0 && healthy == 1 && inside_ok && conflict_ok && status_ok; + fprintf(stderr, + "doc namespace published case=%s valid=%d errors=%d healthy=%d inside=%d " + "inside_reason=%s conflict=%d conflict_reason=%s status_ok=%d correct=%d\n", + label, valid, errors, healthy, inside, inside_reason, conflict, conflict_reason, + status_ok, correct); + free(status); + return correct; +} + +TEST(doc_mentions_cs_namespace_publication) { + char tmp[256] = "/tmp/cbm_dm_ns_XXXXXX"; + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400], cache[400], db[1024], full_db[512]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + snprintf(cache, sizeof(cache), "%s/cache", tmp); + snprintf(full_db, sizeof(full_db), "%s/full.db", tmp); + cbm_mkdir_p(cache, 0700); + cbm_mkdir_p(repo, 0700); /* project identity canonicalizes the existing path */ + char *project = cbm_project_name_from_path(repo); + const char *saved_cache = getenv("CBM_CACHE_DIR"); + char *saved_copy = saved_cache ? strdup(saved_cache) : NULL; + bool correct = project && (!saved_cache || saved_copy); + if (!correct) { + free(saved_copy); + free(project); + th_rmtree(tmp); + ASSERT(correct); + } + snprintf(db, sizeof(db), "%s/%s.db", cache, project); + cbm_setenv("CBM_CACHE_DIR", cache, 1); + for (size_t i = 0; i < sizeof(dm_namespace_cases) / sizeof(dm_namespace_cases[0]); i++) { + for (int file_scoped = 0; file_scoped < 2; file_scoped++) { + char source[2048], label[128]; + bool made = dm_namespace_source(source, sizeof(source), i, file_scoped != 0); + if (!made) { + correct = false; + continue; + } + snprintf(label, sizeof(label), "%s/%s", dm_namespace_cases[i].id, + file_scoped ? "file" : "block"); + dm_unlink_db(db); + dm_unlink_db(full_db); + th_write_file(TH_PATH(repo, "Healthy.cs"), dm_namespace_healthy_source); + th_write_file(TH_PATH(repo, "Bad.cs"), source); + char *indexed_project = NULL; + bool indexed = dm_index(repo, db, &indexed_project) == 0; + bool identity = indexed_project && strcmp(project, indexed_project) == 0; + free(indexed_project); + bool published = + dm_namespace_published(db, project, dm_namespace_cases[i].valid, label); + char *before = dm_doclink_state(db); + cbm_pipeline_incremental_test_reset_faults(); + bool repeated = dm_index(repo, db, NULL) == 0; + cbm_incremental_route_t route = cbm_pipeline_incremental_test_last_route(); + char *after = dm_doclink_state(db); + bool stable = before && after && strcmp(before, after) == 0; + bool noop = route == CBM_INCREMENTAL_ROUTE_NOOP; + fprintf(stderr, + "doc namespace unchanged case=%s indexed=%d identity=%d repeated=%d stable=%d " + "noop=%d " + "route=%d\n", + label, indexed, identity, repeated, stable, noop, (int)route); + correct = indexed && identity && published && repeated && stable && noop && correct; + free(before); + free(after); + + /* These three cases cover clipped names, accepted trailing dots, + * and scope blobs that formerly failed the entire layer. Exercise + * both reuse of stored scopes and edits into/out of the bad region. */ + if (i == 0 || i == 2 || i == 4) { + char edited[2048]; + snprintf(edited, sizeof(edited), "%s\n// body-only edit\n", + dm_namespace_healthy_source); + th_write_file(TH_PATH(repo, "Healthy.cs"), edited); + int step = dm_step(repo, db, full_db, label, CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR); + bool state = dm_namespace_published(db, project, false, label); + correct = step == 0 && state && correct; + + bool repaired = dm_namespace_source(edited, sizeof(edited), 6, file_scoped != 0); + if (repaired) { + th_write_file(TH_PATH(repo, "Bad.cs"), edited); + step = dm_step(repo, db, full_db, "namespace repaired", + CBM_INCREMENTAL_ROUTE_FORCED_FULL); + state = dm_namespace_published(db, project, true, "namespace repaired"); + correct = step == 0 && state && correct; + th_write_file(TH_PATH(repo, "Bad.cs"), source); + step = dm_step(repo, db, full_db, "namespace malformed again", + CBM_INCREMENTAL_ROUTE_FORCED_FULL); + state = dm_namespace_published(db, project, false, label); + correct = step == 0 && state && correct; + } else { + correct = false; + } + } + } + } + if (saved_copy) { + cbm_setenv("CBM_CACHE_DIR", saved_copy, 1); + } else { + cbm_unsetenv("CBM_CACHE_DIR"); + } + free(saved_copy); + free(project); + dm_unlink_db(db); + dm_unlink_db(full_db); + th_rmtree(tmp); + ASSERT(correct); + PASS(); +} + +/* Observe a controlled fixture's recovery path without borrowing CBM's parser + * or retaining Tree-sitter objects across extraction. */ +static bool dm_namespace_parser_state(const char *source, size_t length, bool *has_namespace, + bool *has_error) { + TSParser *parser = ts_parser_new(); + TSTree *tree = NULL; + TSTreeCursor cursor = {0}; + bool cursor_live = false, correct = false; + *has_namespace = false; + *has_error = false; + if (!parser || !ts_parser_set_language(parser, cbm_ts_language(CBM_LANG_CSHARP))) { + goto cleanup; + } + tree = ts_parser_parse_string(parser, NULL, source, (uint32_t)length); + if (!tree) { + goto cleanup; + } + TSNode root = ts_tree_root_node(tree); + *has_error = ts_node_has_error(root); + cursor = ts_tree_cursor_new(root); + cursor_live = true; + size_t visited = 0; + for (;;) { + if (++visited > 4096) { + goto cleanup; + } + const char *kind = ts_node_type(ts_tree_cursor_current_node(&cursor)); + if (strcmp(kind, "namespace_declaration") == 0 || + strcmp(kind, "file_scoped_namespace_declaration") == 0) { + *has_namespace = true; + correct = true; + goto cleanup; + } + if (ts_tree_cursor_goto_first_child(&cursor)) { + continue; + } + while (!ts_tree_cursor_goto_next_sibling(&cursor)) { + if (!ts_tree_cursor_goto_parent(&cursor)) { + correct = true; + goto cleanup; + } + } + } +cleanup: + if (cursor_live) { + ts_tree_cursor_delete(&cursor); + } + if (tree) { + ts_tree_delete(tree); + } + if (parser) { + ts_parser_delete(parser); + } + return correct; +} + +/* Namespace token boundaries must preserve verbatim identifiers without + * accepting bare type keywords as recovered namespace names. */ +TEST(doc_mentions_cs_namespace_boundaries) { + static const struct { + const char *id, *header, *footer, *name; + } cases[] = { + {"bare_keyword", "namespace class{\n", + "public class ParseEdge { unsafe void M(void* p) { _r = ref *(int*)p; } }\n}\n", NULL}, + {"adjacent_file", "namespace@Boundary;\n", "", "Boundary"}, + {"adjacent_block", "namespace@Boundary {\n", "}\n", "Boundary"}, + {"verbatim_keyword_file", "namespace @class;\n", "", "class"}, + {"verbatim_keyword_block", "namespace @class {\n", "}\n", "class"}, + }; + bool correct = true; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + char source[2048], normalized[128]; + int n = snprintf(source, sizeof(source), + "%s/// \n" + "public class FromBad { }\npublic class Hidden { }\n%s", + cases[i].header, cases[i].footer); + bool has_namespace = false, has_error = false; + bool recovery = cases[i].name || + (n > 0 && (size_t)n < sizeof(source) && + dm_namespace_parser_state(source, (size_t)n, &has_namespace, &has_error) && + !has_namespace && has_error); + CBMFileResult *result = n > 0 && (size_t)n < sizeof(source) + ? dm_extract(source, CBM_LANG_CSHARP, "Bad.cs") + : NULL; + const char *scope = result ? result->doc_scope : NULL; + int tokens = result ? dm_count_tokens(result, "global::Good.Target") : -1; + bool placed = scope && strstr(scope, "\nR\t"); + bool quarantined = scope && strstr(scope, "\nX\t") && strstr(scope, "\nQ\tFromBad\n") && + strstr(scope, "\nQ\tHidden\n") && !strstr(scope, "\nT\t"); + bool shape = false; + if (scope && cases[i].name) { + snprintf(normalized, sizeof(normalized), "\t%s\n", cases[i].name); + shape = placed && strstr(scope, normalized) && + strstr(scope, "\nT\t1\t3\t3\tc\t-\tFromBad\t\t\n") && + strstr(scope, "\nT\t1\t4\t4\tc\t-\tHidden\t\t\n") && !strstr(scope, "\nX\t") && + !strstr(scope, "\nQ\t"); + } else if (scope) { + shape = !placed && quarantined; + } + bool one = recovery && tokens == 1 && shape; + fprintf(stderr, + "doc namespace boundary case=%s tokens=%d placed=%d quarantined=%d " + "recovery=%d namespace_node=%d parser_error=%d correct=%d\n", + cases[i].id, tokens, placed, quarantined, recovery, has_namespace, has_error, one); + correct = one && correct; + if (result) { + cbm_free_result(result); + } + } + ASSERT(correct); + PASS(); +} + +static const struct { + const char *id, *text; +} dm_control_cases[] = { + {"using_name", "using Goo~d.Inner;\n"}, + {"using_dot_before", "using Good~.Inner;\n"}, + {"using_dot_after", "using Good.~Inner;\n"}, + {"using_alias", "using Al~ias = Good.Target;\n"}, + {"using_target", "using Alias = Good.Tar~get;\n"}, + {"using_comment", "using Alias = Good./*~*/Target;\n"}, + {"base_name", "public class Before : Goo~d.Base { }\n"}, + {"base_generic", "public class Before : Good.Ba~se { }\n"}, + {"base_comment", "public class Before : Good./*~*/Base { }\n"}, + {"type_parameter", "public class Before { }\n"}, + {"method_parameter", "public class Before { public void M() { } }\n"}, + {"signature", "public class Before { public void M(Good.Tar~get x) { } }\n"}, +}; + +/* The returned length includes the replaced marker even when byte is NUL. */ +static int dm_control_source(char *source, size_t capacity, size_t which, unsigned byte) { + int n = snprintf(source, capacity, + "%s/// \n" + "public class Tail { }\nnamespace . { public class Hidden { } }\n", + dm_control_cases[which].text); + if (n <= 0 || (size_t)n >= capacity) { + return -1; + } + char *mark = strchr(source, '~'); + if (!mark) { + return -1; + } + *mark = (char)byte; + return n; +} + +static bool dm_control_write(const char *path, const char *source, size_t length) { + FILE *file = cbm_fopen(path, "wb"); + if (!file) { + return false; + } + bool complete = fwrite(source, 1, length, file) == length; + return fclose(file) == 0 && complete; +} + +/* Establish the routing contract using complete byte spans, including NUL. + * Own imports are local; changes to a declared base can affect other files. */ +static bool dm_control_delta(const char *before, int before_length, const char *after, + int after_length, int expected, const char *label) { + CBMFileResult *a = + cbm_extract_file(before, before_length, CBM_LANG_CSHARP, "p", "Bad.cs", 0, NULL, NULL); + CBMFileResult *b = + cbm_extract_file(after, after_length, CBM_LANG_CSHARP, "p", "Bad.cs", 0, NULL, NULL); + char *pa = a && a->doc_scope ? cbm_doclink_portable_scope(a->doc_scope) : NULL; + char *pb = b && b->doc_scope ? cbm_doclink_portable_scope(b->doc_scope) : NULL; + dm_names_t names = {{0}}; + int delta = + pa && pb ? cbm_doclinks_scope_delta(pa, pb, dm_name_put, &names) : DM_DELTA_SCAN_FAILED; + bool changed = pa && pb && strcmp(pa, pb) != 0; + bool correct = changed && delta == expected; + fprintf(stderr, "doc control delta case=%s changed=%d delta=%d expected=%d correct=%d\n", label, + changed, delta, expected, correct); + cbm_free(CBM_MEM_CLASS_OTHER, pa); + cbm_free(CBM_MEM_CLASS_OTHER, pb); + if (a) { + cbm_free_result(a); + } + if (b) { + cbm_free_result(b); + } + return correct; +} + +TEST(doc_mentions_cs_control_scopes) { + static const unsigned bytes[] = {0, 1, 127, 32}; + bool correct = true; + for (size_t i = 0; i < sizeof(dm_control_cases) / sizeof(dm_control_cases[0]); i++) { + for (size_t j = 0; j < sizeof(bytes) / sizeof(bytes[0]); j++) { + char source[2048]; + int length = dm_control_source(source, sizeof(source), i, bytes[j]); + cbm_doclink_cs_test_cost_reset(); + CBMFileResult *result = length > 0 ? cbm_extract_file(source, length, CBM_LANG_CSHARP, + "p", "Control.cs", 0, NULL, NULL) + : NULL; + uint64_t built = cbm_doclink_cs_test_scope_bytes(); + const char *scope = result ? result->doc_scope : NULL; + size_t visible = scope ? strlen(scope) : 0; + bool clean = scope != NULL; + for (size_t k = 0; k < visible; k++) { + unsigned char c = (unsigned char)scope[k]; + clean = clean && ((c >= 32 && c != 127) || c == '\t' || c == '\n'); + } + int tokens = result ? dm_count_tokens(result, "global::Good.Target") : -1; + bool tail = scope && strstr(scope, "\tTail\t"); + bool quarantine = scope && strstr(scope, "\nQ\tHidden\n") && strstr(scope, "\nX\t"); + bool marker = true; + if (bytes[j] != 32 && (i == 1 || i == 2)) { + marker = scope && strstr(scope, "\nU\t0\tn\t-\t?\n"); + } else if (bytes[j] != 32 && i == 4) { + marker = scope && strstr(scope, "\nU\t0\ta\tAlias\t?\n"); + } else if (bytes[j] != 32 && (i == 6 || i == 7 || i == 8)) { + marker = scope && strstr(scope, "\tBefore\t\t?\n"); + } + bool one = + scope && built == visible && clean && tokens == 1 && tail && quarantine && marker; + fprintf(stderr, + "doc control scope case=%s byte=%u built=%llu visible=%zu clean=%d " + "tokens=%d tail=%d quarantine=%d marker=%d correct=%d\n", + dm_control_cases[i].id, bytes[j], (unsigned long long)built, visible, clean, + tokens, tail, quarantine, marker, one); + correct = one && correct; + if (result) { + cbm_free_result(result); + } + } + } + ASSERT(correct); + PASS(); +} + +static bool dm_control_published(const char *db, const char *project, const char *label) { + char props[512], tail_reason[64], conflict_reason[64]; + int healthy = 0, tail = 0, conflict = 0; + dm_edge(db, "Healthy.Uses", "Healthy.Target", props, sizeof(props), &healthy); + dm_edge(db, "Bad.Tail", "Healthy.Target", props, sizeof(props), &tail); + dm_edge(db, "Healthy.TriesHidden", "Healthy.Hidden", props, sizeof(props), &conflict); + dm_row(db, "Bad.cs", "global::Good.Target", tail_reason, sizeof(tail_reason), NULL, 0); + dm_row(db, "Healthy.cs", "Hidden", conflict_reason, sizeof(conflict_reason), NULL, 0); + int errors = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'"); + char *status = dm_index_status(project, false); + const char *block = status ? strstr(status, "doc_links:\n") : NULL; + bool status_ok = block && strstr(block, "\n status: ok"); + bool correct = healthy == 1 && tail == 1 && !tail_reason[0] && conflict == 0 && + strcmp(conflict_reason, "graph_gap") == 0 && errors == 0 && status_ok; + fprintf(stderr, + "doc control published case=%s healthy=%d tail=%d tail_reason=%s conflict=%d " + "conflict_reason=%s errors=%d status_ok=%d correct=%d\n", + label, healthy, tail, tail_reason, conflict, conflict_reason, errors, status_ok, + correct); + free(status); + return correct; +} + +TEST(doc_mentions_cs_control_publication) { + static const size_t cases[] = {1, 2, 4, 6, 7, 8}; + static const unsigned bytes[] = {0, 1, 127, 32}; + char tmp[256] = "/tmp/cbm_dm_control_XXXXXX"; + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400], cache[400], db[1024], full_db[512]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + snprintf(cache, sizeof(cache), "%s/cache", tmp); + snprintf(full_db, sizeof(full_db), "%s/full.db", tmp); + cbm_mkdir_p(cache, 0700); + cbm_mkdir_p(repo, 0700); + char *project = cbm_project_name_from_path(repo); + const char *saved_cache = getenv("CBM_CACHE_DIR"); + char *saved_copy = saved_cache ? strdup(saved_cache) : NULL; + bool correct = project && (!saved_cache || saved_copy); + if (!correct) { + free(saved_copy); + free(project); + th_rmtree(tmp); + ASSERT(correct); + } + snprintf(db, sizeof(db), "%s/%s.db", cache, project); + cbm_setenv("CBM_CACHE_DIR", cache, 1); + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + for (size_t j = 0; j < sizeof(bytes) / sizeof(bytes[0]); j++) { + char source[2048], label[128]; + int length = dm_control_source(source, sizeof(source), cases[i], bytes[j]); + snprintf(label, sizeof(label), "%s/%u", dm_control_cases[cases[i]].id, bytes[j]); + dm_unlink_db(db); + dm_unlink_db(full_db); + th_write_file(TH_PATH(repo, "Healthy.cs"), dm_namespace_healthy_source); + bool written = + length > 0 && dm_control_write(TH_PATH(repo, "Bad.cs"), source, (size_t)length); + char *indexed_project = NULL; + bool indexed = written && dm_index(repo, db, &indexed_project) == 0; + bool identity = indexed_project && strcmp(project, indexed_project) == 0; + free(indexed_project); + bool published = dm_control_published(db, project, label); + char *before = dm_doclink_state(db); + cbm_pipeline_incremental_test_reset_faults(); + bool repeated = dm_index(repo, db, NULL) == 0; + cbm_incremental_route_t route = cbm_pipeline_incremental_test_last_route(); + char *after = dm_doclink_state(db); + bool stable = before && after && strcmp(before, after) == 0; + bool noop = route == CBM_INCREMENTAL_ROUTE_NOOP; + fprintf(stderr, + "doc control unchanged case=%s written=%d indexed=%d identity=%d repeated=%d " + "stable=%d noop=%d route=%d\n", + label, written, indexed, identity, repeated, stable, noop, (int)route); + correct = written && indexed && identity && published && repeated && stable && noop && + correct; + free(before); + free(after); + + /* Cover both an import and a base field across stored-scope reuse, + * a repaired field and reintroduction of the embedded NUL. */ + if (bytes[j] == 0 && (cases[i] == 1 || cases[i] == 7)) { + char repaired[2048], edited[2048], step_label[160]; + cbm_incremental_route_t changed_route = cases[i] == 1 + ? CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR + : CBM_INCREMENTAL_ROUTE_FORCED_FULL; + int changed_delta = + cases[i] == 1 ? CBM_DOCLINK_DELTA_LOCAL : CBM_DOCLINK_DELTA_GLOBAL; + snprintf(edited, sizeof(edited), "%s\n// body-only edit\n", + dm_namespace_healthy_source); + th_write_file(TH_PATH(repo, "Healthy.cs"), edited); + snprintf(step_label, sizeof(step_label), "%s/reuse", label); + int step = + dm_step(repo, db, full_db, step_label, CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR); + bool state = dm_control_published(db, project, label); + correct = step == 0 && state && correct; + + int repaired_length = dm_control_source(repaired, sizeof(repaired), cases[i], 32); + bool repaired_written = + repaired_length > 0 && + dm_control_write(TH_PATH(repo, "Bad.cs"), repaired, (size_t)repaired_length); + if (repaired_written) { + snprintf(step_label, sizeof(step_label), "%s/repaired", label); + bool delta = dm_control_delta(source, length, repaired, repaired_length, + changed_delta, step_label); + step = dm_step(repo, db, full_db, step_label, changed_route); + state = dm_control_published(db, project, step_label); + correct = delta && step == 0 && state && correct; + bool restored = + dm_control_write(TH_PATH(repo, "Bad.cs"), source, (size_t)length); + if (restored) { + snprintf(step_label, sizeof(step_label), "%s/reintroduced", label); + delta = dm_control_delta(repaired, repaired_length, source, length, + changed_delta, step_label); + step = dm_step(repo, db, full_db, step_label, changed_route); + state = dm_control_published(db, project, label); + correct = delta && step == 0 && state && correct; + } else { + correct = false; + } + } else { + correct = false; + } + } + } + } + if (saved_copy) { + cbm_setenv("CBM_CACHE_DIR", saved_copy, 1); + } else { + cbm_unsetenv("CBM_CACHE_DIR"); + } + free(saved_copy); + free(project); + dm_unlink_db(db); + dm_unlink_db(full_db); + th_rmtree(tmp); + ASSERT(correct); + PASS(); +} + +/* Temporary observation: extract real source bytes, then exercise the public + * surface codec with complete definitions and with each input isolated. */ +static bool dm_utf8_probe_surface(CBMFileResult *bad, CBMFileResult *healthy, CBMLanguage language, + const char *path, const char *label, const char *bytes) { + CBMFileResult *cache[] = {bad, healthy}; + cbm_file_info_t files[] = {{.rel_path = (char *)path, .language = language}, + {.rel_path = "Healthy.cs", .language = CBM_LANG_CSHARP}}; + CBMArena arena; + cbm_arena_init(&arena); + char *modules[2] = {NULL, NULL}; + int starts[3] = {0}, def_count = 0; + CBMLSPDef *defs = + cbm_pxc_collect_all_defs(NULL, &arena, cache, files, 2, "p", modules, &def_count, starts); + cbm_lsp_surface_row_t *rows = NULL, *again = NULL; + int count = 0, again_count = 0; + int full_rc = + cbm_lsp_surface_build_rows(NULL, "p", cache, files, 2, defs, starts, &rows, &count); + int full_count = count; + int again_rc = + cbm_lsp_surface_build_rows(NULL, "p", cache, files, 2, defs, starts, &again, &again_count); + bool repeat = full_rc == again_rc && count == again_count; + bool dl_equal = false; + char *portable = bad->doc_scope ? cbm_doclink_portable_scope(bad->doc_scope) : NULL; + if (full_rc == 0 && count == 2 && again_rc == 0 && again_count == 2) { + for (int i = 0; i < count; i++) { + repeat = repeat && strcmp(rows[i].defs_json, again[i].defs_json) == 0 && + strcmp(rows[i].surface_sha, again[i].surface_sha) == 0; + } + yyjson_doc *doc = yyjson_read(rows[0].defs_json, strlen(rows[0].defs_json), 0); + yyjson_val *dl = doc ? yyjson_obj_get(yyjson_doc_get_root(doc), "dl") : NULL; + dl_equal = portable && yyjson_is_str(dl) && strcmp(portable, yyjson_get_str(dl)) == 0; + yyjson_doc_free(doc); + } + cbm_store_free_lsp_surfaces(rows, count); + cbm_store_free_lsp_surfaces(again, again_count); + rows = NULL; + count = 0; + const char *saved_scope = bad->doc_scope; + bad->doc_scope = NULL; + int nodl_rc = + cbm_lsp_surface_build_rows(NULL, "p", cache, files, 2, defs, starts, &rows, &count); + int nodl_count = count; + bad->doc_scope = saved_scope; + cbm_store_free_lsp_surfaces(rows, count); + rows = NULL; + count = 0; + CBMFileResult scope_only = {.doc_scope = saved_scope}; + cache[0] = &scope_only; + int scope_rc = + cbm_lsp_surface_build_rows(NULL, "p", cache, files, 2, NULL, NULL, &rows, &count); + bool retained = saved_scope && strstr(saved_scope, bytes) != NULL; + printf("doc_utf8_probe case=%s scope=%d retained=%d defs=%d full_rc=%d full_count=%d " + "nodl_rc=%d nodl_count=%d scope_rc=%d scope_count=%d repeat=%d dl_equal=%d\n", + label, saved_scope != NULL, retained, def_count, full_rc, full_count, nodl_rc, + nodl_count, scope_rc, count, repeat, dl_equal); + cbm_store_free_lsp_surfaces(rows, count); + cbm_free(CBM_MEM_CLASS_OTHER, portable); + free(defs); + free(modules[0]); + free(modules[1]); + cbm_arena_destroy(&arena); + return saved_scope != NULL && def_count > 0 && repeat; +} + +TEST(doc_mentions_cs_utf8_observation) { + static const struct { + const char *label, *bytes; + } sequences[] = { + {"ascii", "A"}, + {"valid2", "\xC3\xA9"}, + {"valid3", "\xE6\xBC\xA2"}, + {"valid4", "\xF0\x90\x90\x80"}, + {"continuation", "\x80"}, + {"overlong", "\xC0\xAF"}, + {"surrogate", "\xED\xA0\x80"}, + {"above_limit", "\xF4\x90\x80\x80"}, + {"truncated", "\xE2\x82"}, + }; + static const struct { + const char *label, *before, *after; + CBMLanguage language; + } positions[] = { + {"using_target", "using Acme.", "Name;\nclass Local {}\n", CBM_LANG_CSHARP}, + {"alias_target", "using Alias = Acme.", "Name;\nclass Local {}\n", CBM_LANG_CSHARP}, + {"namespace_name", "namespace N", "Name { class Local {} }\n", CBM_LANG_CSHARP}, + {"type_name", "class N", "Name {}\n", CBM_LANG_CSHARP}, + {"method_name", "class Local { void N", "Name() {} }\n", CBM_LANG_CSHARP}, + {"parameter_type", "class Local { void M(N", "Name arg) {} }\n", CBM_LANG_CSHARP}, + {"type_parameter", "class Local {}\n", CBM_LANG_CSHARP}, + {"method_parameter", "class Local { void M() {} }\n", CBM_LANG_CSHARP}, + {"field_name", "class Local { int N", "Name; }\n", CBM_LANG_CSHARP}, + {"base_type", "class Local : Acme.N", "Name {}\n", CBM_LANG_CSHARP}, + {"property_value", "

N", "Name

", + CBM_LANG_XML}, + {"property_condition", "

Value

", CBM_LANG_XML}, + {"using_include", "", CBM_LANG_XML}, + {"using_alias", "", CBM_LANG_XML}, + {"import_project", "", + CBM_LANG_XML}, + }; + const char *tail = "/// \nclass Tail {}\n"; + CBMFileResult *healthy = + dm_extract("namespace Good { public class Target {} }\n", CBM_LANG_CSHARP, "Healthy.cs"); + ASSERT_NOT_NULL(healthy); + bool correct = healthy->doc_scope != NULL; + cbm_file_info_t healthy_file = {.rel_path = "Healthy.cs", .language = CBM_LANG_CSHARP}; + cbm_lsp_surface_row_t *healthy_rows = NULL; + int healthy_count = 0; + int healthy_rc = cbm_lsp_surface_build_rows(NULL, "p", &healthy, &healthy_file, 1, NULL, NULL, + &healthy_rows, &healthy_count); + printf("doc_utf8_probe_healthy rc=%d count=%d\n", healthy_rc, healthy_count); + correct = correct && healthy_rc == 0 && healthy_count == 1; + cbm_store_free_lsp_surfaces(healthy_rows, healthy_count); + for (size_t p = 0; p < sizeof(positions) / sizeof(positions[0]); p++) { + for (size_t s = 0; s < sizeof(sequences) / sizeof(sequences[0]); s++) { + char source[2048], label[96]; + const char *path = positions[p].language == CBM_LANG_CSHARP ? "Bad.cs" : "App.csproj"; + size_t a = strlen(positions[p].before), b = strlen(sequences[s].bytes); + size_t c = strlen(positions[p].after); + size_t d = positions[p].language == CBM_LANG_CSHARP ? strlen(tail) : 0; + memcpy(source, positions[p].before, a); + memcpy(source + a, sequences[s].bytes, b); + memcpy(source + a + b, positions[p].after, c); + if (d) { + memcpy(source + a + b + c, tail, d); + } + source[a + b + c + d] = '\0'; + CBMFileResult *bad = cbm_extract_file(source, (int)(a + b + c + d), + positions[p].language, "p", path, 0, NULL, NULL); + snprintf(label, sizeof(label), "%s/%s", positions[p].label, sequences[s].label); + bool setup = bad && dm_utf8_probe_surface(bad, healthy, positions[p].language, path, + label, sequences[s].bytes); + correct = setup && correct; + cbm_free_result(bad); + } + } + static const char *entities[] = {"�", "�", "�", "�", + "é", "𐐀", "😀"}; + for (int position = 0; position < 2; position++) { + for (size_t e = 0; e < sizeof(entities) / sizeof(entities[0]); e++) { + char source[1024], label[96]; + const char *format = + position == 0 ? "

N%sName

" + : ""; + int length = snprintf(source, sizeof(source), format, entities[e]); + CBMFileResult *bad = + cbm_extract_file(source, length, CBM_LANG_XML, "p", "App.csproj", 0, NULL, NULL); + snprintf(label, sizeof(label), "entity_%s/%zu", position == 0 ? "property" : "using", + e); + bool setup = bad && dm_utf8_probe_surface(bad, healthy, CBM_LANG_XML, "App.csproj", + label, entities[e]); + correct = setup && correct; + cbm_free_result(bad); + } + } + cbm_free_result(healthy); + ASSERT(correct); + PASS(); +} + +/* A file whose tree has parse errors takes its nesting from the braces: error + * recovery closes blocks early and late, which would otherwise move the + * declarations after the error into another namespace or outer type. */ +TEST(doc_mentions_cs_scope_parse_errors) { + /* `ref *(int*)p` is beyond the vendored grammar: the method swallows the + * class's closing brace, `P` becomes a local, `Two` a nested class of + * `One`, and the namespace an error node. */ + const char *late = "namespace A\n" /* 1 */ + "{\n" /* 2 */ + " public class One\n" /* 3 */ + " {\n" /* 4 */ + " unsafe void M(void* p) { _r = ref *(int*)p; }\n" /* 5 */ + " public int P;\n" /* 6 */ + " }\n" /* 7 */ + " public class Two { }\n" /* 8 */ + "}\n" /* 9 */ + "namespace B\n" /* 10 */ + "{\n" + " public class Three { }\n" /* 12 */ + "}\n"; /* 13 */ + CBMFileResult *r = dm_scope(late); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + const char *s = r->doc_scope; + ASSERT_NOT_NULL(strstr(s, "R\t1\t0\t1\t9\tA\n")); /* the block, not the tree's node */ + ASSERT_NOT_NULL( + strstr(s, "T\t1\t3\t7\tc!\t-\tOne\t\t\n")); /* ends at its own brace; a member is hidden */ + ASSERT_NOT_NULL(strstr(s, "M\t5\tc\t0\t0\tM\t\tvoid*\n")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t8\t8\tc\t-\tTwo\t\t\n")); /* a sibling in A, not in One */ + ASSERT_NOT_NULL(strstr(s, "R\t2\t0\t10\t13\tB\n")); + ASSERT_NOT_NULL(strstr(s, "T\t2\t12\t12\tc\t-\tThree\t\t\n")); + ASSERT_NULL(strstr(s, "\nQ\t")); + ASSERT_NULL(strstr(s, "\nX\t")); + cbm_free_result(r); + + /* `ref partial struct` is not parsed as a declaration: its header is read + * from the text (kind, name; bases unknown), and the types after it stay + * in the namespace its closing brace seemed to end. */ + const char *early = "namespace Acme\n" /* 1 */ + "{\n" + " public class Before { }\n" /* 3 */ + " public ref partial struct Iter\n" /* 4 */ + " {\n" + " public void End() { }\n" + " }\n" /* 7 */ + " public class After\n" /* 8 */ + " {\n" + " public int Size;\n" /* 10 */ + " }\n" /* 11 */ + "}\n"; /* 12 */ + r = dm_scope(early); + ASSERT_NOT_NULL(r); + s = r->doc_scope; + ASSERT_NOT_NULL(s); + ASSERT_NOT_NULL(strstr(s, "R\t1\t0\t1\t12\tAcme\n")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t3\t3\tc\t-\tBefore\t\t\n")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t4\t7\tsp!\t-\tIter\tT\t?\n")); /* partial, incomplete */ + ASSERT_NOT_NULL(strstr(s, "T\t1\t8\t11\tc\t-\tAfter\t\t\n")); + ASSERT_NOT_NULL(strstr(s, "M\t10\tv\t0\t2\tSize\t\t-\n")); /* of the third type */ + ASSERT_NULL(strstr(s, "T\t0\t")); /* nothing fell into the global namespace */ + cbm_free_result(r); + + /* Both branches of a conditional open the same block: one closing brace + * serves both, and the members belong to the type either way. */ + const char *branches = "namespace Net\n" /* 1 */ + "{\n" + "#if DEBUG\n" + " internal abstract class Cred : DebugHandle {\n" /* 4 */ + "#else\n" + " internal abstract class Cred : PlainHandle {\n" /* 6 */ + "#endif\n" + " public int Size;\n" /* 8 */ + " }\n" /* 9 */ + " public class After { }\n" /* 10 */ + "}\n"; + r = dm_scope(branches); + ASSERT_NOT_NULL(r); + s = r->doc_scope; + ASSERT_NOT_NULL(s); + ASSERT_NOT_NULL(strstr(s, "R\t1\t0\t1\t11\tNet\n")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t4\t9\tc!\t-\tCred\t\t")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t6\t9\tc!\t-\tCred\t\t")); + ASSERT_NOT_NULL(strstr(s, "M\t8\tv\t0\t1\tSize\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t10\t10\tc\t-\tAfter\t\t\n")); + ASSERT_NULL(strstr(s, "\nX\t")); + cbm_free_result(r); + + /* A string the parser's own lexer loses the thread in ($@"..." with a + * doubled quote): it then reads code as string content, so the braces come + * from this scan's own reading of the text. */ + const char *derailed = "namespace Net\n" /* 1 */ + "{\n" + " public class Verb\n" /* 3 */ + " {\n" + " void M() { s += $@\", K = \"\"{K.Name}\"\"\"; }\n" /* 5 */ + " public int Q;\n" + " }\n" /* 7 */ + " public class After { }\n" /* 8 */ + "}\n"; /* 9 */ + r = dm_scope(derailed); + ASSERT_NOT_NULL(r); + s = r->doc_scope; + ASSERT_NOT_NULL(s); + ASSERT_NOT_NULL(strstr(s, "R\t1\t0\t1\t9\tNet\n")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t3\t7\tc!\t-\tVerb\t\t")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t8\t8\tc")); + ASSERT_NOT_NULL(strstr(s, "\tAfter\t\t")); + ASSERT_NULL(strstr(s, "\nX\t")); + cbm_free_result(r); + + /* Braces that do not pair: nothing after the first unpaired one is + * placed; the type names are kept, to be resolved to nothing else. */ + const char *open = "namespace Acme\n" /* 1 */ + "{\n" /* 2 */ + " public class Ok { }\n" + " public class Broken\n" + " {\n" + " public void M() {\n" + " public class Lost { }\n"; + r = dm_scope(open); + ASSERT_NOT_NULL(r); + s = r->doc_scope; + ASSERT_NOT_NULL(s); + ASSERT_NOT_NULL(strstr(s, "\nX\t2\t")); /* from the namespace's brace to the end */ + ASSERT_NOT_NULL(strstr(s, "Q\tOk\n")); + ASSERT_NOT_NULL(strstr(s, "Q\tBroken\n")); + ASSERT_NOT_NULL(strstr(s, "Q\tLost\n")); + ASSERT_NULL(strstr(s, "\nT\t")); + ASSERT_NULL(strstr(s, "\nR\t")); + char *portable = cbm_doclink_portable_scope(s); + ASSERT_NOT_NULL(portable); + ASSERT_NOT_NULL(strstr(portable, "\nX\t0\t0\n")); /* line numbers like the others */ + cbm_free(CBM_MEM_CLASS_OTHER, portable); + cbm_free_result(r); + + /* A namespace only one configuration can name is not named at all. */ + const char *either = "#if GEN\n" + "namespace Gen.Interop\n" + "#else\n" + "namespace Run.Interop\n" + "#endif\n" + "{\n" + " public enum Mode { One }\n" + "}\n"; + r = dm_scope(either); + ASSERT_NOT_NULL(r); + s = r->doc_scope; + ASSERT_NOT_NULL(s); + ASSERT_NULL(strstr(s, "\nR\t")); + ASSERT_NOT_NULL(strstr(s, "Q\tMode\n")); + ASSERT_NOT_NULL(strstr(s, "\nX\t")); /* and what is documented there has no scope */ + cbm_free_result(r); + + /* A member whose header did not parse is hidden, and its type says so: + * `safe extern` comes back as a method named `extern`. */ + const char *hidden = "namespace C\n" + "{\n" + " public class Holey\n" /* 3 */ + " {\n" + " public safe extern int Hidden();\n" /* 5 */ + " public int Seen;\n" /* 6 */ + " }\n" /* 7 */ + "}\n"; + r = dm_scope(hidden); + ASSERT_NOT_NULL(r); + s = r->doc_scope; + ASSERT_NOT_NULL(s); + ASSERT_NOT_NULL(strstr(s, "T\t1\t3\t7\tc!\t-\tHoley\t\t\n")); + ASSERT_NOT_NULL(strstr(s, "M\t6\tv\t0\t0\tSeen\t\t-\n")); + ASSERT_NULL(strstr(s, "\textern\t")); + ASSERT_NULL(strstr(s, "\tHidden\t")); + cbm_free_result(r); + PASS(); +} + +TEST(doc_mentions_cs_norm_type) { + struct { + const char *in; + const char *out; + } cases[] = { + {"ref int x", "int"}, + {"params string[] args", "string[]"}, + {"[NotNull] System.Collections.Generic.List xs", "List"}, + {"Nullable{System.Int32}", "int"}, + {"System.String", "string"}, + {"T?", "T"}, + {"byte*", "byte*"}, + {"int[,]", "int[,]"}, + {"``0", "?"}, + {"", "?"}, + }; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + char out[128]; + cbm_doclink_cs_norm_type(cases[i].in, strlen(cases[i].in), out, sizeof(out)); + if (strcmp(out, cases[i].out) != 0) { + printf(" norm(%s) = %s, want %s\n", cases[i].in, out, cases[i].out); + FAIL("normalized type"); + } + } + /* a name that does not fit the caller's buffer is a type nothing is known + * about, never a cut name: two long names with one beginning would + * compare equal */ + char small[8]; + const char *long_name = "LongTypeNameOne"; + ASSERT_EQ(cbm_doclink_cs_norm_type(long_name, strlen(long_name), small, sizeof(small)), 1); + ASSERT_STR_EQ(small, "?"); + ASSERT_EQ(cbm_doclink_cs_norm_type("Fits", 4, small, sizeof(small)), 4); + ASSERT_STR_EQ(small, "Fits"); + PASS(); +} + +/* ── MSBuild global usings (R1) ──────────────────────────────────── */ + +/* A project file of a test: its path in the repository and its text. */ +typedef struct { + const char *rel_path; + const char *xml; +} dm_project_file_t; + +/* Evaluate `project` over `files`, each through the extractor's scope blob: + * no file is written, none is read. The result is released with + * cbm_msb_result_free. false when a file yields no blob or memory runs out. */ +static bool dm_msb_eval(const dm_project_file_t *files, int n, const char *project, + cbm_msb_result_t *out) { + cbm_msb_t *m = cbm_msb_new(); + bool ok = m != NULL; + for (int i = 0; ok && i < n; i++) { + CBMFileResult *r = dm_extract(files[i].xml, CBM_LANG_XML, files[i].rel_path); + ok = r && r->doc_scope && cbm_msb_is_project_scope(r->doc_scope) && + cbm_msb_add(m, files[i].rel_path, r->doc_scope); + if (r) { + cbm_free_result(r); + } + } + ok = ok && cbm_msb_eval(m, project, out); + cbm_msb_free(m); + return ok; +} + +/* True when the result holds a using of this kind (n namespace, s static, + * a alias) and target. */ +static bool dm_has_using(const cbm_msb_result_t *r, char kind, const char *target) { + for (int i = 0; i < r->count; i++) { + if (r->usings[i].kind == kind && strcmp(r->usings[i].target, target) == 0) { + return true; + } + } + return false; +} + +/* Repeated expansion must not retain every superseded value or condition + * operand. Measure real evaluation arena capacity, not a wall-clock sample. */ +static bool dm_msb_add_xml(cbm_msb_t *m, const char *path, const char *xml) { + CBMFileResult *source = dm_extract(xml, CBM_LANG_XML, path); + bool ok = source && source->doc_scope && cbm_msb_add(m, path, source->doc_scope); + if (source) { + cbm_free_result(source); + } + return ok; +} + +static bool dm_msb_same(const cbm_msb_result_t *a, const cbm_msb_result_t *b) { + if (a->count != b->count || a->open != b->open || a->unevaluable != b->unevaluable || + a->outside != b->outside) { + return false; + } + for (int i = 0; i < a->count; i++) { + if (a->usings[i].kind != b->usings[i].kind || + strcmp(a->usings[i].alias, b->usings[i].alias) || + strcmp(a->usings[i].target, b->usings[i].target)) { + return false; + } + } + return true; +} + +static bool dm_msb_context_matches(cbm_msb_eval_context_t *context, const cbm_msb_t *m, + const char *project) { + cbm_msb_result_t actual = {0}; + cbm_msb_result_t reference = {0}; + bool ok = cbm_msb_eval_context_eval(context, project, &actual) && + cbm_msb_eval(m, project, &reference) && dm_msb_same(&actual, &reference); + if (!ok) { + fprintf(stderr, "MSBuild context differs from uncached evaluation: %s\n", project); + } + cbm_msb_result_free(&actual); + cbm_msb_result_free(&reference); + return ok; +} + +/* Include construction, per-project work and context destruction. A hit must + * not replay/copy all shared properties behind a constant record counter. */ +/* The hit-phase comparison grows shared dependencies while keeping local + * inputs/output fixed. Scanning the entire cached dependency set must fail. */ +/* Dependence on project input is exact, including absence versus unknown. + * Input owners are gone before the next call; cached effects must own any + * captured local values and preserve import/poison history. */ +TEST(doc_mentions_msbuild_targets_isolation) { + const dm_project_file_t files[] = { + {"Directory.Build.props", + "Prefix" + ""}, + {"PrefixOnly.props", "PrefixStable" + ""}, + {"Directory.Build.targets", + "" + "" + "$(Input)$(Empty)Tail" + "$(Custom)First$(Own)" + "Final" + "" + "" + "" + "" + "" + "" + "" + "" + "" + ""}, + {"Seen.targets", "SharedSeen" + ""}, + {"Maybe.targets", "Hidden" + ""}, + {"PoisonInput.props", + "Hidden" + "HiddenHidden"}, + {"Irrelevant.props", "Anything" + ""}, + {"A.csproj", "SameCommon" + "Project"}, + {"B.csproj", "SameCommon" + ""}, + {"Changed.csproj", + "Changed" + "DifferentNonempty"}, + {"Missing.csproj", ""}, + {"Empty.csproj", "" + ""}, + {"Unknown.csproj", ""}, + {"EarlySeen.csproj", "" + "LocalSeen"}, + {"EarlyPoison.csproj", "" + "Restored"}, + {"EarlyRoot.csproj", "" + "AfterAfter" + ""}, + {"Seed/Directory.Build.targets", + "" + "$(MSBuildProjectName)" + ""}, + {"Seed/A.csproj", ""}, + {"Seed/B.csproj", ""}, + {"Seed/left/A.csproj", + "Fixed" + ""}, + {"Seed/right/A.csproj", ""}, + {"Seed/Unknown.csproj", ""}, + {"Seed/SeedPoison.props", + "Hidden" + ""}, + {"Seed/Override.csproj", + "Fixed" + ""}, + {"Later/Directory.Build.targets", ""}, + {"Later/A.csproj", ""}, + {"Later/B.csproj", ""}, + }; + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + for (size_t i = 0; i < sizeof(files) / sizeof(files[0]); i++) { + ASSERT_TRUE(dm_msb_add_xml(m, files[i].rel_path, files[i].xml)); + } + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + const char *projects[] = {"A.csproj", + "B.csproj", + "A.csproj", + "Changed.csproj", + "A.csproj", + "Missing.csproj", + "Empty.csproj", + "Unknown.csproj", + "Missing.csproj", + "A.csproj", + "EarlySeen.csproj", + "Missing.csproj", + "EarlyPoison.csproj", + "Missing.csproj", + "EarlyRoot.csproj", + "A.csproj", + "Seed/A.csproj", + "Seed/B.csproj", + "Seed/Override.csproj", + "Seed/A.csproj", + "Seed/left/A.csproj", + "Seed/right/A.csproj", + "Seed/Unknown.csproj", + "Seed/A.csproj", + "Later/A.csproj", + "Later/B.csproj"}; + for (size_t i = 0; i < sizeof(projects) / sizeof(projects[0]); i++) { + ASSERT_TRUE(dm_msb_context_matches(context, m, projects[i])); + } + ASSERT_TRUE(dm_msb_add_xml(m, "Later/Later.targets", + "" + "")); + ASSERT_TRUE(dm_msb_context_matches(context, m, "Later/A.csproj")); + ASSERT_TRUE(dm_msb_context_matches(context, m, "Later/B.csproj")); + ASSERT_TRUE(dm_msb_context_matches(context, m, "A.csproj")); + cbm_msb_eval_context_free(context); + ASSERT_EQ(cbm_msb_test_value_live_bytes(), 0); + cbm_msb_free(m); + PASS(); +} + +/* A target captures project-owned values before it fails. The same context + * must be usable again, and later failures must leave retained owners sound. */ +TEST(doc_mentions_msbuild_targets_failure) { + char xml[8192]; + size_t w = (size_t)snprintf(xml, sizeof(xml), ""); + for (int i = 0; i < 64; i++) { + w += (size_t)snprintf(xml + w, sizeof(xml) - w, "$(Input)%02d", i, i, i); + } + w += (size_t)snprintf(xml + w, sizeof(xml) - w, + "" + ""); + ASSERT_TRUE(w < sizeof(xml)); + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE( + dm_msb_add_xml(m, "Directory.Build.props", + "Prefix")); + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.targets", xml)); + ASSERT_TRUE(dm_msb_add_xml(m, "A.csproj", + "Shared" + "")); + ASSERT_TRUE(dm_msb_add_xml(m, "B.csproj", + "Changed" + "")); + int failures = 0; + for (int mode = 0; mode < 2; mode++) { + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + cbm_msb_test_fail_value_alloc_after(mode == 0 ? 16 : 0); + cbm_msb_test_fail_prop_insert_after(mode == 1 ? 16 : 0); + cbm_msb_result_t r = {0}; + bool ok = cbm_msb_eval_context_eval(context, "A.csproj", &r); + bool consumed = + mode == 0 ? cbm_msb_test_value_alloc_failed() : cbm_msb_test_prop_insert_failed(); + failures += ok || !consumed || r.mem != NULL || r.usings != NULL || r.count != 0; + cbm_msb_result_free(&r); + cbm_msb_test_fail_value_alloc_after(0); + cbm_msb_test_fail_prop_insert_after(0); + failures += !dm_msb_context_matches(context, m, "A.csproj"); + uint64_t retained = cbm_msb_test_value_live_bytes(); + cbm_msb_test_fail_value_alloc_after(mode == 0 ? 1 : 0); + cbm_msb_test_fail_prop_insert_after(mode == 1 ? 1 : 0); + ok = cbm_msb_eval_context_eval(context, "B.csproj", &r); + consumed = + mode == 0 ? cbm_msb_test_value_alloc_failed() : cbm_msb_test_prop_insert_failed(); + failures += ok || !consumed || r.mem != NULL || r.usings != NULL || r.count != 0 || + cbm_msb_test_value_live_bytes() > retained; + cbm_msb_result_free(&r); + cbm_msb_test_fail_value_alloc_after(0); + cbm_msb_test_fail_prop_insert_after(0); + failures += !dm_msb_context_matches(context, m, "A.csproj"); + failures += !dm_msb_context_matches(context, m, "B.csproj"); + failures += !dm_msb_context_matches(context, m, "A.csproj"); + cbm_msb_eval_context_free(context); + failures += cbm_msb_test_value_live_bytes() != 0; + fprintf(stderr, "msbuild targets fault mode=%d retained=%llu failures=%d\n", mode, + (unsigned long long)retained, failures); + } + cbm_msb_free(m); + ASSERT_EQ(failures, 0); + PASS(); +} + +/* Final shared items read project inputs after target writes. Cached includes + * and removals must compose with local items and retain diagnostic counts. */ +TEST(doc_mentions_msbuild_items_isolation) { + const dm_project_file_t files[] = { + {"Directory.Build.props", + "" + "" + "" + ""}, + {"Parts/One.props", "" + "" + ""}, + {"Directory.Build.targets", + "Targets.Fixed" + "" + "" + "" + ""}, + {"Parts/Two.targets", "" + "" + "" + ""}, + {"Poison.props", "HiddenHidden" + "HiddenHiddenHidden" + "Hidden"}, + {"A.csproj", "Same.InputSameAlias" + "falseShared.Dropon" + "Project.A" + "" + ""}, + {"B.csproj", + "Same.InputSameAlias" + "falseShared.Dropon" + "Project.BDifferent" + "" + ""}, + {"Changed.csproj", + "Changed.InputChangedAlias" + "trueTargets.Fixedoff" + "on"}, + {"Empty.csproj", "" + "" + ""}, + {"Missing.csproj", ""}, + {"Unknown.csproj", ""}, + {"Other/Directory.Build.props", "" + "" + ""}, + {"Other/A.csproj", ""}, + {"Target/Directory.Build.targets", "" + "" + "" + ""}, + {"Target/A.csproj", ""}, + {"Later/A.csproj", ""}, + {"Later/B.csproj", ""}, + }; + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + for (size_t i = 0; i < sizeof(files) / sizeof(files[0]); i++) { + ASSERT_TRUE(dm_msb_add_xml(m, files[i].rel_path, files[i].xml)); + } + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + const char *projects[] = { + "A.csproj", "B.csproj", "A.csproj", "Changed.csproj", "A.csproj", + "Empty.csproj", "Missing.csproj", "Unknown.csproj", "Missing.csproj", "B.csproj", + "Other/A.csproj", "Other/A.csproj", "A.csproj", "Target/A.csproj", "Target/A.csproj", + "A.csproj", "Later/A.csproj", "Later/B.csproj"}; + for (size_t i = 0; i < sizeof(projects) / sizeof(projects[0]); i++) { + ASSERT_TRUE(dm_msb_context_matches(context, m, projects[i])); + } + ASSERT_TRUE(dm_msb_add_xml(m, "Later/Directory.Build.props", + "" + "")); + ASSERT_TRUE(dm_msb_context_matches(context, m, "Later/A.csproj")); + ASSERT_TRUE(dm_msb_context_matches(context, m, "Later/B.csproj")); + ASSERT_TRUE(dm_msb_context_matches(context, m, "A.csproj")); + cbm_msb_eval_context_free(context); + ASSERT_EQ(cbm_msb_test_value_live_bytes(), 0); + cbm_msb_free(m); + PASS(); +} + +/* The seams fail real allocation/insertion operations. Context-owned state + * must remain usable, and all allocator-tracked storage must be released. */ +TEST(doc_mentions_msbuild_items_failure) { + char xml[16384]; + size_t w = (size_t)snprintf(xml, sizeof(xml), ""); + for (int i = 0; i < 64; i++) { + w += (size_t)snprintf(xml + w, sizeof(xml) - w, + "", i); + } + w += (size_t)snprintf(xml + w, sizeof(xml) - w, ""); + ASSERT_TRUE(w < sizeof(xml)); + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.props", xml)); + ASSERT_TRUE(dm_msb_add_xml(m, "A.csproj", + "Shared" + "")); + ASSERT_TRUE(dm_msb_add_xml(m, "B.csproj", + "Shared" + "Different")); + ASSERT_TRUE(dm_msb_add_xml(m, "C.csproj", + "Changed" + "")); + const struct { + cbm_msb_item_fail_operation_t operation; + int nth; + bool warm; + } cases[] = { + {CBM_MSB_ITEM_FAIL_CAPTURE_ALLOC, 1, false}, {CBM_MSB_ITEM_FAIL_CAPTURE_ALLOC, 9, false}, + {CBM_MSB_ITEM_FAIL_CAPTURE_ALLOC, 33, false}, {CBM_MSB_ITEM_FAIL_UNIQUE_INSERT, 1, false}, + {CBM_MSB_ITEM_FAIL_UNIQUE_INSERT, 9, false}, {CBM_MSB_ITEM_FAIL_UNIQUE_INSERT, 33, false}, + {CBM_MSB_ITEM_FAIL_APPLY_ALLOC, 1, true}, {CBM_MSB_ITEM_FAIL_PUBLISH_ALLOC, 1, true}}; + int failures = 0; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + size_t before = cbm_mem_tracked_live_bytes(); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + if (cases[i].warm) { + failures += !dm_msb_context_matches(context, m, "A.csproj"); + } + size_t retained = cbm_mem_tracked_live_bytes(); + cbm_msb_test_fail_item_operation(cases[i].operation, cases[i].nth); + cbm_msb_result_t r = {0}; + bool ok = cbm_msb_eval_context_eval(context, cases[i].warm ? "B.csproj" : "A.csproj", &r); + bool consumed = cbm_msb_test_item_operation_failed(); + failures += ok || !consumed || r.mem != NULL || r.usings != NULL || r.count != 0; + cbm_msb_result_free(&r); + cbm_msb_test_fail_item_operation(CBM_MSB_ITEM_FAIL_NONE, 0); + if (cases[i].warm) { + failures += cbm_mem_tracked_live_bytes() != retained; + } + failures += !dm_msb_context_matches(context, m, "A.csproj"); + failures += !dm_msb_context_matches(context, m, "B.csproj"); + failures += !dm_msb_context_matches(context, m, "C.csproj"); + failures += !dm_msb_context_matches(context, m, "A.csproj"); + cbm_msb_eval_context_free(context); + size_t after = cbm_mem_tracked_live_bytes(); + failures += after != before || cbm_msb_test_value_live_bytes() != 0; + fprintf(stderr, + "msbuild items fault operation=%d nth=%d consumed=%d " + "before=%llu after=%llu failures=%d\n", + (int)cases[i].operation, cases[i].nth, consumed, (unsigned long long)before, + (unsigned long long)after, failures); + } + cbm_msb_free(m); + ASSERT_EQ(failures, 0); + PASS(); +} + +/* A published result owns its strings independently of the retained cache, + * its replacement and the model. A mismatch must keep the first variant. */ +TEST(doc_mentions_msbuild_items_lifetime) { + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.props", + "" + "" + "" + "")); + ASSERT_TRUE(dm_msb_add_xml(m, "A.csproj", + "First.Value" + "")); + ASSERT_TRUE(dm_msb_add_xml(m, "B.csproj", + "Other.Value" + "")); + ASSERT_TRUE(dm_msb_add_xml(m, "Other/Directory.Build.props", + "" + "")); + ASSERT_TRUE(dm_msb_add_xml(m, "Other/A.csproj", "")); + cbm_msb_result_t expected = {0}; + ASSERT_TRUE(cbm_msb_eval(m, "A.csproj", &expected)); + ASSERT_EQ(expected.count, 3); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + cbm_msb_result_t held = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, "A.csproj", &held)); + ASSERT_TRUE(dm_msb_same(&held, &expected)); + cbm_msb_result_t hit = {0}; + cbm_msb_test_cost_reset(); + ASSERT_TRUE(cbm_msb_eval_context_eval(context, "A.csproj", &hit)); + uint64_t before = 0; + uint64_t peak = 0; + cbm_msb_test_cost(&before, &peak); + ASSERT_TRUE(dm_msb_same(&hit, &expected)); + cbm_msb_result_free(&hit); + ASSERT_TRUE(dm_msb_context_matches(context, m, "B.csproj")); + cbm_msb_test_cost_reset(); + ASSERT_TRUE(cbm_msb_eval_context_eval(context, "A.csproj", &hit)); + uint64_t after = 0; + cbm_msb_test_cost(&after, &peak); + bool same_hit_work = before == after; + ASSERT_TRUE(dm_msb_same(&hit, &expected)); + cbm_msb_result_free(&hit); + ASSERT_TRUE(dm_msb_context_matches(context, m, "Other/A.csproj")); + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.targets", + "" + "")); + ASSERT_TRUE(dm_msb_context_matches(context, m, "A.csproj")); + cbm_msb_eval_context_free(context); + cbm_msb_free(m); + bool independent = dm_msb_same(&held, &expected); + cbm_msb_result_free(&held); + cbm_msb_result_free(&expected); + fprintf(stderr, "msbuild items retained variant before=%llu after=%llu independent=%d\n", + (unsigned long long)before, (unsigned long long)after, independent); + ASSERT_TRUE(same_hit_work && independent); + ASSERT_EQ(cbm_msb_test_value_live_bytes(), 0); + PASS(); +} + +/* Shared final-property items can produce no output, or the same unique + * output, despite a large input span. Warm work must not scale with that + * span after an exact reusable result is available. */ +TEST(doc_mentions_msbuild_items_work) { + enum { CAP = 131072, PROJECTS = 24 }; + char *xml = malloc(CAP); + ASSERT_NOT_NULL(xml); + uint64_t hits[9][2] = {{0}}; + bool correct = true; + bool storage_bounded = true; + char long_alias[4096]; + memset(long_alias, 'A', sizeof(long_alias) - 1); + long_alias[sizeof(long_alias) - 1] = '\0'; + for (int mode = 0; mode < 9; mode++) { + for (int size = 0; size < 2; size++) { + int items = 512 << size; + size_t w = (size_t)snprintf(xml, CAP, ""); + if (mode == 6 || mode == 7) { + for (int i = 0; i < items; i++) { + w += (size_t)snprintf(xml + w, CAP - w, "off", i, i); + } + } else if (mode == 8) { + w += (size_t)snprintf(xml + w, CAP - w, "%s", long_alias); + } + w += (size_t)snprintf(xml + w, CAP - w, ""); + for (int i = 0; i < items; i++) { + if (mode == 0) { + w += (size_t)snprintf(xml + w, CAP - w, + "", i); + } else if (mode == 1) { + w += (size_t)snprintf(xml + w, CAP - w, + ""); + } else if (mode == 2) { + w += (size_t)snprintf(xml + w, CAP - w, + "", + i); + } else if (mode == 3) { + w += (size_t)snprintf(xml + w, CAP - w, + "", + i); + } else if (mode == 4) { + w += (size_t)snprintf(xml + w, CAP - w, "", i); + } else if (mode == 5) { + w += (size_t)snprintf(xml + w, CAP - w, + "" + "", + i, i); + } else if (mode == 6 || mode == 7) { + w += (size_t)snprintf(xml + w, CAP - w, + "", + i, i); + } else { + w += + (size_t)snprintf(xml + w, CAP - w, + ""); + } + } + w += (size_t)snprintf(xml + w, CAP - w, ""); + ASSERT_TRUE(w < CAP); + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.props", xml)); + if (mode == 7) { + /* All final K values come from targets; the changing local + * K0000 value must not invalidate their item dependencies. */ + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.targets", xml)); + } + ASSERT_TRUE(dm_msb_add_xml( + m, "src/Warm.csproj", + mode == 4 ? "off" + "" + : "off" + "")); + for (int i = 0; i < PROJECTS; i++) { + char path[64]; + char project[256]; + snprintf(path, sizeof(path), "src/P%02d.csproj", i); + snprintf(project, sizeof(project), + "offP%02d%s" + "%s", + i, mode == 7 ? "on" : "", + mode == 4 ? "" + : ""); + ASSERT_TRUE(dm_msb_add_xml(m, path, project)); + } + cbm_msb_test_cost_reset(); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + cbm_msb_result_t warm = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, "src/Warm.csproj", &warm)); + correct = correct && warm.count == (mode == 1 || mode == 4 || mode == 8 ? 1 : 0) && + warm.open == (mode == 2) && warm.unevaluable == (mode == 2 ? items : 0) && + !warm.outside; + cbm_msb_result_free(&warm); + uint64_t captured = cbm_msb_test_work(); + for (int i = 0; i < PROJECTS; i++) { + char path[64]; + snprintf(path, sizeof(path), "src/P%02d.csproj", i); + cbm_msb_result_t r = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, path, &r)); + correct = correct && r.count == (mode == 1 || mode == 4 || mode == 8 ? 1 : 0) && + r.open == (mode == 2) && r.unevaluable == (mode == 2 ? items : 0) && + !r.outside; + if (mode == 1 || mode == 8) { + correct = + correct && r.count == 1 && r.usings[0].kind == 'a' && + strcmp(r.usings[0].alias, mode == 8 ? long_alias : "SameAlias") == 0 && + strcmp(r.usings[0].target, "Same.Target") == 0; + } else if (mode == 4) { + correct = correct && r.count == 1 && r.usings[0].kind == 'n' && + strcmp(r.usings[0].alias, "") == 0 && + strcmp(r.usings[0].target, "Project.Keep") == 0; + } + cbm_msb_result_free(&r); + } + hits[mode][size] = cbm_msb_test_work() - captured; + cbm_msb_eval_context_free(context); + uint64_t records = 0; + uint64_t peak = 0; + cbm_msb_test_cost(&records, &peak); + if (mode == 8) { + /* One long value and one unique output must not leave a + * persistent alias copy per duplicate item during capture. */ + uint64_t bound = 16 * (uint64_t)w + 1024 * 1024; + storage_bounded = storage_bounded && peak <= bound; + fprintf(stderr, "msbuild items duplicate storage items=%d peak=%llu bound=%llu\n", + items, (unsigned long long)peak, (unsigned long long)bound); + } + fprintf(stderr, + "msbuild items mode=%d items=%d projects=%d interpreted=%llu " + "work=%llu hit_work=%llu peak=%llu live=%llu semantics=%d\n", + mode, items, PROJECTS, (unsigned long long)records, + (unsigned long long)cbm_msb_test_work(), (unsigned long long)hits[mode][size], + (unsigned long long)peak, (unsigned long long)cbm_msb_test_value_live_bytes(), + correct); + correct = correct && cbm_msb_test_value_live_bytes() == 0; + cbm_msb_free(m); + } + } + free(xml); + ASSERT_TRUE(correct); + bool bounded = true; + for (int mode = 0; mode < 9; mode++) { + ASSERT_TRUE(hits[mode][0] > 0); + double ratio = (double)hits[mode][1] / (double)hits[mode][0]; + fprintf(stderr, "msbuild items shared-size hit ratio mode=%d ratio=%.3f\n", mode, ratio); + bounded = bounded && ratio >= 0.9 && ratio <= 1.1; + } + ASSERT_TRUE(bounded && storage_bounded); + PASS(); +} + +TEST(doc_mentions_msbuild_targets_work) { + enum { CAP = 131072 }; + char *props = malloc(CAP); + char *targets = malloc(CAP); + ASSERT_NOT_NULL(props); + ASSERT_NOT_NULL(targets); + uint64_t work[4][2][2] = {{{0}}}; + uint64_t hits[4][2][2] = {{{0}}}; + bool correct = true; + for (int mode = 0; mode < 4; mode++) { + for (int size = 0; size < 2; size++) { + int records = 512 << size; + size_t p = (size_t)snprintf(props, CAP, ""); + size_t t = (size_t)snprintf(targets, CAP, ""); + for (int i = 0; i < records; i++) { + if (mode == 1) { + p += (size_t)snprintf(props + p, CAP - p, "Shared%04d", i, i, i); + t += (size_t)snprintf(targets + t, CAP - t, "$(K%04d)", i, i, i); + } else if (mode == 2) { + t += (size_t)snprintf(targets + t, CAP - t, "$(Custom)%04d", i, + i, i); + } else { + t += (size_t)snprintf(targets + t, CAP - t, "Shared%04d", i, i, + i); + } + } + p += (size_t)snprintf(props + p, CAP - p, + "Prefix"); + t += (size_t)snprintf( + targets + t, CAP - t, + "" + "" + "" + "", + records - 1); + ASSERT_TRUE(p < CAP && t < CAP); + for (int count = 0; count < 2; count++) { + int projects = 16 << count; + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + /* No prefix file in mode 3: absence is a stable baseline, + * not a reason to discard a reusable target on every call. */ + if (mode != 3) { + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.props", props)); + } + ASSERT_TRUE(dm_msb_add_xml(m, "Shared.targets", targets)); + ASSERT_TRUE( + dm_msb_add_xml(m, "Directory.Build.targets", + "")); + for (int i = 0; i < projects; i++) { + char path[64]; + char xml[512]; + snprintf(path, sizeof(path), "src/P%02d.csproj", i); + snprintf( + xml, sizeof(xml), + "" + "Shared0000SharedP%02d" + "", + i); + ASSERT_TRUE(dm_msb_add_xml(m, path, xml)); + } + /* Capture many redundant overrides in mode 1; later projects + * omit them. They must not create required local exceptions. */ + ASSERT_TRUE(dm_msb_add_xml(m, "src/Warm.csproj", + mode == 2 ? "Shared" + : props)); + cbm_msb_test_cost_reset(); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + cbm_msb_result_t warm = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, "src/Warm.csproj", &warm)); + correct = correct && warm.count == 3 && !warm.open && !warm.unevaluable && + !warm.outside && dm_has_using(&warm, 'a', "Shared0000"); + cbm_msb_result_free(&warm); + uint64_t captured = cbm_msb_test_work(); + for (int i = 0; i < projects; i++) { + char path[64]; + char name[32]; + char last[32]; + snprintf(path, sizeof(path), "src/P%02d.csproj", i); + snprintf(name, sizeof(name), "P%02d", i); + snprintf(last, sizeof(last), "Shared%04d", records - 1); + cbm_msb_result_t r = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, path, &r)); + correct = correct && r.count == 3 && !r.open && !r.unevaluable && !r.outside && + dm_has_using(&r, 'a', "Shared0000") && dm_has_using(&r, 'a', last) && + dm_has_using(&r, 'a', name); + cbm_msb_result_free(&r); + } + hits[mode][size][count] = cbm_msb_test_work() - captured; + cbm_msb_eval_context_free(context); + work[mode][size][count] = cbm_msb_test_work(); + uint64_t consumed = 0; + uint64_t peak = 0; + cbm_msb_test_cost(&consumed, &peak); + fprintf( + stderr, + "msbuild targets mode=%d records=%d projects=%d " + "interpreted=%llu work=%llu hit_work=%llu peak=%llu live=%llu semantics=%d\n", + mode, records, projects, (unsigned long long)consumed, + (unsigned long long)work[mode][size][count], + (unsigned long long)hits[mode][size][count], (unsigned long long)peak, + (unsigned long long)cbm_msb_test_value_live_bytes(), correct); + correct = correct && cbm_msb_test_value_live_bytes() == 0; + cbm_msb_free(m); + } + } + } + free(props); + free(targets); + ASSERT_TRUE(correct); + bool bounded = true; + for (int mode = 0; mode < 4; mode++) { + for (int size = 0; size < 2; size++) { + ASSERT_TRUE(work[mode][size][0] > 0); + double ratio = (double)work[mode][size][1] / (double)work[mode][size][0]; + fprintf(stderr, "msbuild targets project ratio mode=%d records=%d ratio=%.3f\n", mode, + 512 << size, ratio); + bounded = bounded && ratio >= 0.9 && ratio <= 1.4; + } + for (int count = 0; count < 2; count++) { + ASSERT_TRUE(hits[mode][0][count] > 0); + double ratio = (double)hits[mode][1][count] / (double)hits[mode][0][count]; + fprintf(stderr, + "msbuild targets shared-size hit ratio mode=%d projects=%d ratio=%.3f\n", mode, + 16 << count, ratio); + bounded = bounded && ratio >= 0.9 && ratio <= 1.1; + } + } + ASSERT_TRUE(bounded); + PASS(); +} + +TEST(doc_mentions_msbuild_nearest_work) { + enum { DEPTH = 48, PROJECTS = 24 }; + char dir[512] = ""; + for (int i = 0; i < DEPTH; i++) { + strcat(dir, "deep/"); + } + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.props", "")); + for (int i = 0; i < PROJECTS; i++) { + char path[600]; + snprintf(path, sizeof(path), "%sP%02d.csproj", dir, i); + ASSERT_TRUE(dm_msb_add_xml(m, path, "")); + } + cbm_msb_test_cost_reset(); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + bool correct = true; + for (int i = 0; i < PROJECTS; i++) { + char path[600]; + snprintf(path, sizeof(path), "%sP%02d.csproj", dir, i); + cbm_msb_result_t r = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, path, &r)); + correct = correct && r.count == 0 && !r.open && !r.unevaluable && !r.outside; + cbm_msb_result_free(&r); + } + uint64_t probes = cbm_msb_test_nearest_steps(); + char target[600]; + snprintf(target, sizeof(target), "%sDirectory.Build.targets", dir); + ASSERT_TRUE(dm_msb_add_xml(m, target, + "" + "")); + char project[600]; + snprintf(project, sizeof(project), "%sP00.csproj", dir); + cbm_msb_result_t r = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, project, &r)); + correct = correct && r.count == 1 && !r.open && !r.unevaluable && !r.outside && + dm_has_using(&r, 'n', "New.Target"); + cbm_msb_result_free(&r); + uint64_t fresh = cbm_msb_test_nearest_steps() - probes; + cbm_msb_eval_context_free(context); + correct = correct && cbm_msb_test_value_live_bytes() == 0; + cbm_msb_free(m); + fprintf(stderr, + "msbuild nearest depth=%d projects=%d probes=%llu after_add=%llu semantics=%d\n", DEPTH, + PROJECTS, (unsigned long long)probes, (unsigned long long)fresh, correct); + ASSERT_TRUE(correct); + ASSERT_TRUE(probes <= 2 * (DEPTH + 1)); + ASSERT_TRUE(fresh > 0 && fresh <= 2 * (DEPTH + 1)); + PASS(); +} + +/* Distinct nearest-file roots must not replay the common imported state. + * All models are built before measurement. The warm root and measured roots + * differ; doubling shared size must not change subsequent per-root work. */ +TEST(doc_mentions_msbuild_shared_closure_work) { + enum { CAP = 131072 }; + char *xml = malloc(CAP); + ASSERT_NOT_NULL(xml); + char *definitions = malloc(CAP); + ASSERT_NOT_NULL(definitions); + uint64_t work[4][2][2] = {{{0}}}; + uint64_t hits[4][2][2] = {{{0}}}; + bool correct = true; + for (int mode = 0; mode < 4; mode++) { + const char *suffix = mode % 2 == 0 ? "props" : "targets"; + for (int size = 0; size < 2; size++) { + int records = 512 << size; + size_t w = (size_t)snprintf(xml, CAP, ""); + size_t d = (size_t)snprintf(definitions, CAP, ""); + for (int i = 0; i < records; i++) { + if (mode >= 2) { + d += (size_t)snprintf(definitions + d, CAP - d, "Shared%04d", i, + i, i); + w += (size_t)snprintf(xml + w, CAP - w, "$(K%04d)", i, i, i); + } else { + w += (size_t)snprintf(xml + w, CAP - w, "Shared%04d", i, i, i); + } + } + d += (size_t)snprintf(definitions + d, CAP - d, ""); + w += (size_t)snprintf( + xml + w, CAP - w, + "" + "" + "" + "", + mode >= 2 ? 'T' : 'K', mode >= 2 ? 'T' : 'K', records - 1); + ASSERT_TRUE(w < CAP && d < CAP); + for (int count = 0; count < 2; count++) { + int projects = 16 << count; + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + char shared[64]; + char wrapper[256]; + snprintf(shared, sizeof(shared), "Shared.%s", suffix); + if (mode >= 2) { + char define_path[64]; + snprintf(define_path, sizeof(define_path), "Definitions.%s", suffix); + ASSERT_TRUE(dm_msb_add_xml(m, define_path, definitions)); + snprintf(wrapper, sizeof(wrapper), + "" + "", + suffix, suffix); + } else { + snprintf(wrapper, sizeof(wrapper), + "", suffix); + } + ASSERT_TRUE(dm_msb_add_xml(m, shared, xml)); + for (int i = 0; i <= projects; i++) { + char root[96]; + char project[96]; + snprintf(root, sizeof(root), "src/R%02d/Directory.Build.%s", i, suffix); + snprintf(project, sizeof(project), "src/R%02d/P%02d.csproj", i, i); + ASSERT_TRUE(dm_msb_add_xml(m, root, wrapper)); + ASSERT_TRUE(dm_msb_add_xml(m, project, "")); + } + cbm_msb_test_cost_reset(); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + uint64_t captured = 0; + for (int i = 0; i <= projects; i++) { + char path[96]; + char name[32]; + char last[32]; + snprintf(path, sizeof(path), "src/R%02d/P%02d.csproj", i, i); + snprintf(name, sizeof(name), "P%02d", i); + snprintf(last, sizeof(last), "Shared%04d", records - 1); + cbm_msb_result_t r = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, path, &r)); + cbm_msb_using_t expected_usings[] = { + {.kind = 'a', .alias = "Project", .target = name}, + {.kind = 'a', .alias = "First", .target = "Shared0000"}, + {.kind = 'a', .alias = "Last", .target = last}, + }; + cbm_msb_result_t expected = {.usings = expected_usings, .count = 3}; + correct = correct && dm_msb_same(&r, &expected); + cbm_msb_result_free(&r); + if (i == 0) { + captured = cbm_msb_test_work(); + } + } + hits[mode][size][count] = cbm_msb_test_work() - captured; + cbm_msb_eval_context_free(context); + work[mode][size][count] = cbm_msb_test_work(); + uint64_t interpreted = 0; + uint64_t peak = 0; + cbm_msb_test_cost(&interpreted, &peak); + correct = correct && cbm_msb_test_value_live_bytes() == 0; + fprintf(stderr, + "msbuild shared closure mode=%d records=%d projects=%d interpreted=%llu " + "work=%llu hit_work=%llu peak=%llu live=%llu semantics=%d\n", + mode, records, projects, (unsigned long long)interpreted, + (unsigned long long)work[mode][size][count], + (unsigned long long)hits[mode][size][count], (unsigned long long)peak, + (unsigned long long)cbm_msb_test_value_live_bytes(), correct); + cbm_msb_free(m); + } + } + } + free(xml); + free(definitions); + ASSERT_TRUE(correct); + bool bounded = true; + for (int mode = 0; mode < 4; mode++) { + for (int count = 0; count < 2; count++) { + ASSERT_TRUE(hits[mode][0][count] > 0); + double ratio = (double)hits[mode][1][count] / (double)hits[mode][0][count]; + fprintf(stderr, + "msbuild shared closure shared-size hit ratio mode=%d projects=%d ratio=%.3f\n", + mode, 16 << count, ratio); + bounded = bounded && ratio >= 0.9 && ratio <= 1.1; + } + /* Include construction and teardown: sharing must not shift the same + * product cost into retained-state copying or cleanup. */ + ASSERT_TRUE(work[mode][1][0] > work[mode][0][0]); + ASSERT_TRUE(work[mode][1][1] > work[mode][0][1]); + uint64_t small = work[mode][1][0] - work[mode][0][0]; + uint64_t large = work[mode][1][1] - work[mode][0][1]; + double ratio = (double)large / (double)small; + fprintf(stderr, "msbuild shared closure extra-size ratio mode=%d ratio=%.3f\n", mode, + ratio); + bounded = bounded && ratio >= 0.9 && ratio <= 1.2; + } + ASSERT_TRUE(bounded); + PASS(); +} + +/* Grow the number of shared files, with fixed tiny outputs. Component lookup, + * item traversal, dependency composition and cleanup must not replay the chain. */ +TEST(doc_mentions_msbuild_shared_file_work) { + uint64_t work[2][2][2] = {{{0}}}; + uint64_t hits[2][2][2] = {{{0}}}; + bool correct = true; + bool storage_bounded = true; + for (int mode = 0; mode < 2; mode++) { + const char *suffix = mode == 0 ? "props" : "targets"; + for (int size = 0; size < 2; size++) { + int files = 128 << size; + for (int count = 0; count < 2; count++) { + int projects = 16 << count; + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + size_t source_bytes = 0; + for (int i = 0; i < files; i++) { + char path[96]; + char next[96] = ""; + char xml[512]; + snprintf(path, sizeof(path), "Shared/Chain%04d.props", i); + if (i + 1 < files) { + snprintf(next, sizeof(next), "", + i + 1); + } + int n = snprintf(xml, sizeof(xml), + "Shared%04d" + "%s", + i, i, i, next); + ASSERT_TRUE(n > 0 && (size_t)n < sizeof(xml)); + source_bytes += (size_t)n; + ASSERT_TRUE(dm_msb_add_xml(m, path, xml)); + } + char root_xml[1024]; + int n = snprintf(root_xml, sizeof(root_xml), + "" + "" + "" + "" + "", + files - 1); + ASSERT_TRUE(n > 0 && (size_t)n < sizeof(root_xml)); + source_bytes += (size_t)n; + ASSERT_TRUE(dm_msb_add_xml(m, "Shared/Root.props", root_xml)); + const char wrapper[] = + ""; + const char project_xml[] = ""; + for (int i = 0; i <= projects; i++) { + char root[96]; + char project[96]; + snprintf(root, sizeof(root), "src/R%02d/Directory.Build.%s", i, suffix); + snprintf(project, sizeof(project), "src/R%02d/P%02d.csproj", i, i); + ASSERT_TRUE(dm_msb_add_xml(m, root, wrapper)); + ASSERT_TRUE(dm_msb_add_xml(m, project, project_xml)); + source_bytes += sizeof(wrapper) + sizeof(project_xml) - 2; + } + cbm_msb_test_cost_reset(); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + uint64_t captured = 0; + for (int i = 0; i <= projects; i++) { + char path[96]; + char name[32]; + char last[32]; + snprintf(path, sizeof(path), "src/R%02d/P%02d.csproj", i, i); + snprintf(name, sizeof(name), "P%02d", i); + snprintf(last, sizeof(last), "Shared%04d", files - 1); + cbm_msb_result_t actual = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, path, &actual)); + cbm_msb_using_t expected_usings[] = { + {.kind = 'a', .alias = "Project", .target = name}, + {.kind = 'a', .alias = "First", .target = "Shared0000"}, + {.kind = 'a', .alias = "Last", .target = last}, + }; + cbm_msb_result_t expected = {.usings = expected_usings, .count = 3}; + correct = dm_msb_same(&actual, &expected) && correct; + cbm_msb_result_free(&actual); + if (i == 0) { + captured = cbm_msb_test_work(); + } + } + hits[mode][size][count] = cbm_msb_test_work() - captured; + cbm_msb_eval_context_free(context); + work[mode][size][count] = cbm_msb_test_work(); + uint64_t interpreted = 0; + uint64_t peak = 0; + cbm_msb_test_cost(&interpreted, &peak); + correct = correct && cbm_msb_test_value_live_bytes() == 0; + /* Allow fixed per-file metadata and bounded-depth state nodes, + * but reject a default64KiB retained arena for every tiny file. */ + uint64_t bound = (uint64_t)source_bytes * 64 + 1024 * 1024; + storage_bounded = storage_bounded && peak <= bound; + fprintf(stderr, + "msbuild shared files mode=%d files=%d projects=%d interpreted=%llu " + "work=%llu hit_work=%llu peak=%llu bound=%llu live=%llu semantics=%d\n", + mode, files, projects, (unsigned long long)interpreted, + (unsigned long long)work[mode][size][count], + (unsigned long long)hits[mode][size][count], (unsigned long long)peak, + (unsigned long long)bound, + (unsigned long long)cbm_msb_test_value_live_bytes(), correct); + cbm_msb_free(m); + } + } + } + ASSERT_TRUE(correct); + bool bounded = true; + for (int mode = 0; mode < 2; mode++) { + for (int count = 0; count < 2; count++) { + ASSERT_TRUE(hits[mode][0][count] > 0); + double ratio = (double)hits[mode][1][count] / (double)hits[mode][0][count]; + fprintf(stderr, "msbuild shared files size hit ratio mode=%d projects=%d ratio=%.3f\n", + mode, 16 << count, ratio); + bounded = bounded && ratio >= 0.9 && ratio <= 1.1; + } + ASSERT_TRUE(work[mode][1][0] > work[mode][0][0]); + ASSERT_TRUE(work[mode][1][1] > work[mode][0][1]); + uint64_t small = work[mode][1][0] - work[mode][0][0]; + uint64_t large = work[mode][1][1] - work[mode][0][1]; + double ratio = (double)large / (double)small; + fprintf(stderr, "msbuild shared files extra-size ratio mode=%d ratio=%.3f\n", mode, ratio); + bounded = bounded && ratio >= 0.9 && ratio <= 1.2; + } + ASSERT_TRUE(bounded && storage_bounded); + PASS(); +} + +/* Shared components must preserve import order and history across distinct + * wrappers; the ordinary evaluator remains the exact semantic oracle. */ +TEST(doc_mentions_msbuild_shared_closure_isolation) { + const dm_project_file_t common[] = { + {"Shared/State.props", + "Shared$(Before)" + "" + "" + "" + "" + "" + "" + "" + ""}, + {"Shared/Leaf.props", "$(Value).Leaf" + "" + "" + "" + ""}, + {"Shared/Diamond.props", + "" + ""}, + {"Shared/Cycle.props", "" + ""}, + }; + bool correct = true; + for (int mode = 0; mode < 2; mode++) { + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + for (size_t i = 0; i < sizeof(common) / sizeof(common[0]); i++) { + ASSERT_TRUE(dm_msb_add_xml(m, common[i].rel_path, common[i].xml)); + } + const char *suffix = mode == 0 ? "props" : "targets"; + char paths[5][96]; + for (int i = 0; i < 5; i++) { + char root[96]; + char wrapper[2048]; + char project[512]; + char name = (char)('A' + i); + snprintf(root, sizeof(root), "Roots/%c/Directory.Build.%s", name, suffix); + snprintf(paths[i], sizeof(paths[i]), "Roots/%c/%c.csproj", name, name); + snprintf(wrapper, sizeof(wrapper), + "%sPre.%c" + "%s" + "" + "Post.%cAfter.%c", + i == 2 ? "Changed" : "Same", name, + i == 3 ? "" + : "", + name, name); + snprintf(project, sizeof(project), + "Project.%c" + "%s", + name, i == 4 ? "" : ""); + ASSERT_TRUE(dm_msb_add_xml(m, root, wrapper)); + ASSERT_TRUE(dm_msb_add_xml(m, paths[i], project)); + } + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + cbm_msb_result_t held = {0}; + cbm_msb_result_t expected = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, paths[0], &held)); + ASSERT_TRUE(cbm_msb_eval(m, paths[0], &expected)); + correct = correct && dm_msb_same(&held, &expected) && held.count == 9 && + dm_has_using(&held, 'a', "Shared.Leaf") && + dm_has_using(&held, 'a', "File.State") && dm_has_using(&held, 'a', "Leaf.Leaf") && + dm_has_using(&held, 'a', mode == 0 ? "Project.A" : "Post.A"); + const int order[] = {0, 1, 2, 0, 3, 1, 4, 0}; + for (size_t i = 0; i < sizeof(order) / sizeof(order[0]); i++) { + correct = dm_msb_context_matches(context, m, paths[order[i]]) && correct; + } + ASSERT_TRUE(dm_msb_add_xml(m, "Shared/Later.props", + "")); + for (int i = 0; i < 5; i++) { + correct = dm_msb_context_matches(context, m, paths[i]) && correct; + } + cbm_msb_eval_context_free(context); + cbm_msb_free(m); + correct = correct && dm_msb_same(&held, &expected) && cbm_msb_test_value_live_bytes() == 0; + cbm_msb_result_free(&held); + cbm_msb_result_free(&expected); + } + /* Preserve the parent's read of entry X while masking the child's X read + * after X is overwritten. Y escapes from the child and must join the input + * requirements. A child-state witness cannot certify the earlier X read. */ + for (int mode = 0; mode < 2; mode++) { + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE( + dm_msb_add_xml(m, "Common/Union.props", + "$(X)Forced" + "" + "" + "" + "")); + ASSERT_TRUE(dm_msb_add_xml(m, "Common/Child.props", + "$(X)$(Y)" + "")); + const char *xs[] = {"Old", "Forced", "Old", "Old"}; + const char *ys[] = {"One", "One", "Two", NULL}; + const char *y_records[] = {"One", "One", "Two", ""}; + char paths[4][96]; + for (int i = 0; i < 4; i++) { + char root[96]; + char xml[512]; + snprintf(root, sizeof(root), "Roots/R%d/Directory.Build.%s", i, + mode == 0 ? "props" : "targets"); + snprintf(paths[i], sizeof(paths[i]), "Roots/R%d/P.csproj", i); + snprintf(xml, sizeof(xml), + "%s%s" + "", + xs[i], y_records[i]); + ASSERT_TRUE(dm_msb_add_xml(m, root, xml)); + ASSERT_TRUE(dm_msb_add_xml(m, paths[i], "")); + } + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + const int order[] = {0, 1, 2, 0, 3, 0}; + for (size_t i = 0; i < sizeof(order) / sizeof(order[0]); i++) { + int which = order[i]; + cbm_msb_result_t actual = {0}; + cbm_msb_result_t reference = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, paths[which], &actual)); + ASSERT_TRUE(cbm_msb_eval(m, paths[which], &reference)); + correct = dm_msb_same(&actual, &reference) && correct; + bool found[3] = {false, false, false}; + for (int j = 0; j < actual.count; j++) { + const cbm_msb_using_t *u = &actual.usings[j]; + found[0] = found[0] || (u->kind == 'a' && strcmp(u->alias, "Snapshot") == 0 && + strcmp(u->target, xs[which]) == 0); + found[1] = found[1] || (u->kind == 'a' && strcmp(u->alias, "FromX") == 0 && + strcmp(u->target, "Forced") == 0); + found[2] = + found[2] || (ys[which] && u->kind == 'a' && strcmp(u->alias, "FromY") == 0 && + strcmp(u->target, ys[which]) == 0); + } + correct = correct && found[0] && found[1] && found[2] == (ys[which] != NULL) && + actual.count == (ys[which] ? 3 : 2) && actual.open == (ys[which] == NULL) && + actual.unevaluable == (ys[which] ? 0 : 1) && actual.outside == 0; + cbm_msb_result_free(&actual); + cbm_msb_result_free(&reference); + } + cbm_msb_eval_context_free(context); + cbm_msb_free(m); + correct = correct && cbm_msb_test_value_live_bytes() == 0; + } + /* Same-length call-local values exercise revision identity independently + * of value length. Repeat callers after their temporary state is freed. */ + for (int mode = 0; mode < 2; mode++) { + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE( + dm_msb_add_xml(m, "Read.props", + "$(Input)" + "" + "")); + const char *values[] = {"Old", "New", "Alt"}; + char paths[3][32]; + for (int i = 0; i < 3; i++) { + char xml[512]; + snprintf(paths[i], sizeof(paths[i]), "P%d.csproj", i); + snprintf(xml, sizeof(xml), + "Tmp%s" + "%s", + values[i], mode == 0 ? "" : ""); + ASSERT_TRUE(dm_msb_add_xml(m, paths[i], xml)); + } + if (mode != 0) { + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.targets", + "")); + } + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + const int order[] = {0, 1, 0, 2, 1, 0}; + for (size_t i = 0; i < sizeof(order) / sizeof(order[0]); i++) { + int which = order[i]; + cbm_msb_result_t actual = {0}; + cbm_msb_result_t reference = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, paths[which], &actual)); + ASSERT_TRUE(cbm_msb_eval(m, paths[which], &reference)); + cbm_msb_using_t expected_using = { + .kind = 'a', .alias = "Value", .target = values[which]}; + cbm_msb_result_t expected = {.usings = &expected_using, .count = 1}; + correct = + dm_msb_same(&actual, &reference) && dm_msb_same(&actual, &expected) && correct; + cbm_msb_result_free(&actual); + cbm_msb_result_free(&reference); + } + cbm_msb_eval_context_free(context); + cbm_msb_free(m); + correct = correct && cbm_msb_test_value_live_bytes() == 0; + } + ASSERT_TRUE(correct); + PASS(); +} + +/* Node acquisition failures must take the ordinary OOM path. The retained + * component remains usable and every transient owner is released. */ +TEST(doc_mentions_msbuild_components_failure) { + char xml[16384]; + size_t w = (size_t)snprintf(xml, sizeof(xml), ""); + for (int i = 0; i < 64; i++) { + w += + (size_t)snprintf(xml + w, sizeof(xml) - w, "$(I%03d).N%03d", i, i, i, i); + } + w += (size_t)snprintf(xml + w, sizeof(xml) - w, + "" + "" + "" + ""); + ASSERT_TRUE(w < sizeof(xml)); + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(dm_msb_add_xml(m, "Shared.props", xml)); + ASSERT_TRUE(dm_msb_add_xml(m, "Child.props", + "$(K063)" + "Forced")); + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.targets", + "Before" + "" + "After")); + for (int project = 0; project < 3; project++) { + char path[32]; + snprintf(path, sizeof(path), "%c.csproj", 'A' + project); + w = (size_t)snprintf(xml, sizeof(xml), "P%d", + project); + for (int i = 0; i < 64; i++) { + w += (size_t)snprintf(xml + w, sizeof(xml) - w, "%s", i, + project == 2 ? "BBBB" : "AAAA", i); + } + w += (size_t)snprintf(xml + w, sizeof(xml) - w, ""); + ASSERT_TRUE(w < sizeof(xml)); + ASSERT_TRUE(dm_msb_add_xml(m, path, xml)); + } + const struct { + cbm_msb_node_fail_operation_t operation; + int nth; + bool warm; + } cases[] = { + {CBM_MSB_NODE_FAIL_CAPTURE, 1, false}, {CBM_MSB_NODE_FAIL_CAPTURE, 5, false}, + {CBM_MSB_NODE_FAIL_OVERLAY, 1, false}, {CBM_MSB_NODE_FAIL_OVERLAY, 5, false}, + {CBM_MSB_NODE_FAIL_DEPENDENCY, 1, false}, {CBM_MSB_NODE_FAIL_DEPENDENCY, 5, false}, + {CBM_MSB_NODE_FAIL_APPLY, 1, true}, {CBM_MSB_NODE_FAIL_APPLY, 5, true}, + }; + int failures = 0; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + size_t before = cbm_mem_tracked_live_bytes(); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + if (cases[i].warm) { + failures += !dm_msb_context_matches(context, m, "A.csproj"); + } + cbm_msb_test_fail_node_alloc(cases[i].operation, cases[i].nth); + cbm_msb_result_t actual = {0}; + bool ok = cbm_msb_eval_context_eval(context, "A.csproj", &actual); + bool consumed = cbm_msb_test_node_alloc_failed(); + failures += + ok || !consumed || actual.mem != NULL || actual.usings != NULL || actual.count != 0; + cbm_msb_result_free(&actual); + cbm_msb_test_fail_node_alloc(CBM_MSB_NODE_FAIL_NONE, 0); + const char *order[] = {"A.csproj", "B.csproj", "C.csproj", "A.csproj"}; + for (size_t j = 0; j < sizeof(order) / sizeof(order[0]); j++) { + failures += !dm_msb_context_matches(context, m, order[j]); + } + cbm_msb_eval_context_free(context); + size_t after = cbm_mem_tracked_live_bytes(); + failures += after != before || cbm_msb_test_value_live_bytes() != 0; + fprintf(stderr, + "msbuild components fault operation=%d nth=%d consumed=%d " + "before=%llu after=%llu failures=%d\n", + (int)cases[i].operation, cases[i].nth, consumed, (unsigned long long)before, + (unsigned long long)after, failures); + } + cbm_msb_free(m); + ASSERT_EQ(failures, 0); + PASS(); +} + +/* Revision witnesses must remain exact when short-lived state reuses slots. + * The observations certify that this fixture actually reaches those paths. */ +TEST(doc_mentions_msbuild_components_revision) { + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.targets", + "" + "")); + ASSERT_TRUE(dm_msb_add_xml(m, "Constants.props", + "Stable.Fixed" + "")); + ASSERT_TRUE( + dm_msb_add_xml(m, "Read.props", + "$(Input)$(Stable)" + "" + "" + "")); + const char *values[] = {"AAAA", "BBBB", "CCCC"}; + char paths[3][32]; + for (int i = 0; i < 3; i++) { + char xml[512]; + snprintf(paths[i], sizeof(paths[i]), "P%d.csproj", i); + snprintf(xml, sizeof(xml), + "Temp%s" + "Unchanged", + values[i]); + ASSERT_TRUE(dm_msb_add_xml(m, paths[i], xml)); + } + size_t before = cbm_mem_tracked_live_bytes(); + cbm_msb_test_cost_reset(); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + const int order[] = {0, 0, 1, 0, 2, 1, 0}; + bool correct = true; + for (size_t i = 0; i < sizeof(order) / sizeof(order[0]); i++) { + int which = order[i]; + cbm_msb_result_t actual = {0}; + cbm_msb_result_t reference = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, paths[which], &actual)); + ASSERT_TRUE(cbm_msb_eval(m, paths[which], &reference)); + cbm_msb_using_t expected_usings[] = { + {.kind = 'a', .alias = "Value", .target = values[which]}, + {.kind = 'a', .alias = "Constant", .target = "Stable.Fixed"}, + }; + cbm_msb_result_t expected = {.usings = expected_usings, .count = 2}; + correct = dm_msb_same(&actual, &reference) && dm_msb_same(&actual, &expected) && correct; + cbm_msb_result_free(&actual); + cbm_msb_result_free(&reference); + } + cbm_msb_state_test_stats_t stats = {0}; + cbm_msb_test_state_stats(&stats); + cbm_msb_eval_context_free(context); + size_t after = cbm_mem_tracked_live_bytes(); + correct = correct && before == after && cbm_msb_test_value_live_bytes() == 0; + cbm_msb_free(m); + fprintf(stderr, + "msbuild components revisions allocations=%llu reused=%llu witnessed_reused=%llu " + "same_length=%llu witness_skips=%llu revision_errors=%llu semantics=%d\n", + (unsigned long long)stats.allocations, (unsigned long long)stats.slot_reuses, + (unsigned long long)stats.witnessed_slot_reuses, + (unsigned long long)stats.same_length_value_changes, + (unsigned long long)stats.witness_skips, (unsigned long long)stats.revision_errors, + correct); + ASSERT_TRUE(correct); + ASSERT_TRUE(stats.allocations > 0 && stats.slot_reuses > 0 && stats.witnessed_slot_reuses > 0 && + stats.same_length_value_changes > 0 && stats.witness_skips > 0 && + stats.revision_errors == 0); + PASS(); +} + +/* Mixed presence constraints must not use the all-absent empty-range shortcut. + * Insertion order places A/B at seen keys 4/5, with all other visited files + * outside that aligned range. Warm Shared sees A present and B absent; Cold + * reaches the same retained component with the entire range empty. */ +TEST(doc_mentions_msbuild_components_mixed_absence) { + const dm_project_file_t files[] = { + {"Warm.csproj", "" + ""}, + {"Cold.csproj", ""}, + {"Padding.props", ""}, + {"A.props", "" + ""}, + {"B.props", "" + ""}, + {"Shared.props", "" + ""}, + }; + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + for (size_t i = 0; i < sizeof(files) / sizeof(files[0]); i++) { + ASSERT_TRUE(dm_msb_add_xml(m, files[i].rel_path, files[i].xml)); + } + cbm_msb_using_t expected_usings[] = { + {.kind = 'n', .alias = "", .target = "A.Effect"}, + {.kind = 'n', .alias = "", .target = "B.Effect"}, + }; + cbm_msb_result_t expected = {.usings = expected_usings, .count = 2}; + size_t before = cbm_mem_tracked_live_bytes(); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + bool correct = true; + const char *order[] = {"Warm.csproj", "Cold.csproj", "Cold.csproj", "Warm.csproj", + "Cold.csproj"}; + for (size_t i = 0; i < sizeof(order) / sizeof(order[0]); i++) { + cbm_msb_result_t actual = {0}; + cbm_msb_result_t reference = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, order[i], &actual)); + ASSERT_TRUE(cbm_msb_eval(m, order[i], &reference)); + correct = dm_msb_same(&actual, &reference) && dm_msb_same(&actual, &expected) && correct; + cbm_msb_result_free(&actual); + cbm_msb_result_free(&reference); + } + cbm_msb_eval_context_free(context); + size_t after = cbm_mem_tracked_live_bytes(); + correct = correct && before == after && cbm_msb_test_value_live_bytes() == 0; + cbm_msb_free(m); + fprintf(stderr, "msbuild components mixed absence semantics=%d before=%llu after=%llu\n", + correct, (unsigned long long)before, (unsigned long long)after); + ASSERT_TRUE(correct); + PASS(); +} + +TEST(doc_mentions_msbuild_prefix_work) { + enum { CAP = 131072 }; + char *xml = malloc(CAP); + ASSERT_NOT_NULL(xml); + uint64_t work[2][2] = {{0}}; + bool correct = true; + for (int size = 0; size < 2; size++) { + int records = 512 << size; + for (int count = 0; count < 2; count++) { + int projects = 16 << count; + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + size_t w = (size_t)snprintf(xml, CAP, ""); + for (int i = 0; i < records; i++) { + w += (size_t)snprintf(xml + w, CAP - w, "Shared%04d", i, i, i); + } + w += (size_t)snprintf( + xml + w, CAP - w, + "" + "" + "" + "", + records - 1); + ASSERT_TRUE(w < CAP); + ASSERT_TRUE(dm_msb_add_xml(m, "Shared.props", xml)); + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.props", + "")); + for (int i = 0; i < projects; i++) { + char path[64]; + snprintf(path, sizeof(path), "src/P%02d.csproj", i); + ASSERT_TRUE(dm_msb_add_xml(m, path, "")); + } + cbm_msb_test_cost_reset(); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + for (int i = 0; i < projects; i++) { + char path[64]; + char project[32]; + char last[32]; + snprintf(path, sizeof(path), "src/P%02d.csproj", i); + snprintf(project, sizeof(project), "P%02d", i); + snprintf(last, sizeof(last), "Shared%04d", records - 1); + cbm_msb_result_t r; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, path, &r)); + correct = correct && r.count == 3 && !r.open && r.unevaluable == 0 && + r.outside == 0 && dm_has_using(&r, 'a', "Shared0000") && + dm_has_using(&r, 'a', last) && dm_has_using(&r, 'a', project); + cbm_msb_result_free(&r); + } + cbm_msb_eval_context_free(context); + work[size][count] = cbm_msb_test_work(); + uint64_t consumed = 0; + uint64_t peak = 0; + cbm_msb_test_cost(&consumed, &peak); + fprintf(stderr, + "msbuild prefix records=%d projects=%d interpreted=%llu work=%llu " + "peak=%llu live=%llu semantics=%d\n", + records, projects, (unsigned long long)consumed, + (unsigned long long)work[size][count], (unsigned long long)peak, + (unsigned long long)cbm_msb_test_value_live_bytes(), correct); + correct = correct && cbm_msb_test_value_live_bytes() == 0; + cbm_msb_free(m); + } + } + free(xml); + ASSERT_TRUE(correct); + bool bounded = true; + for (int size = 0; size < 2; size++) { + ASSERT_TRUE(work[size][0] > 0); + double ratio = (double)work[size][1] / (double)work[size][0]; + fprintf(stderr, "msbuild prefix project ratio records=%d ratio=%.3f\n", 512 << size, ratio); + bounded = bounded && ratio >= 0.9 && ratio <= 1.4; + } + ASSERT_TRUE(bounded); + PASS(); +} + +/* The uncached evaluator is the semantic oracle, including diagnostics and + * ordering. Exercise shared properties and items across project/root changes. */ +TEST(doc_mentions_msbuild_prefix_isolation) { + const dm_project_file_t files[] = { + {"Directory.Build.props", + "Prefix" + "" + "" + "" + "" + ""}, + {"Common.props", "Yes" + ""}, + {"Directory.Build.targets", + "$(Flavor)" + "" + "" + "" + ""}, + {"A.csproj", + "AA" + "Nonempty"}, + {"B.csproj", "B"}, + {"Poison.csproj", "B" + ""}, + {"Poison.props", "HiddenHidden" + ""}, + {"Write/Directory.Build.props", + "" + "Forced$(MSBuildProjectName)" + "" + ""}, + {"Write/A.csproj", ""}, + {"Write/B.csproj", ""}, + {"Read/Directory.Build.props", + "" + "$(MSBuildProjectName)" + "" + ""}, + {"Read/A.csproj", ""}, + {"Read/B.csproj", ""}, + {"Read/A.props", ""}, + {"Read/B.props", ""}, + {"History/Directory.Build.props", + "Seed" + "" + ""}, + {"History/Owned.csproj", "$(V).Again" + ""}, + {"History/Other.csproj", + "Known" + ""}, + {"History/Maybe.props", "Hidden"}, + {"Condition/Directory.Build.props", + "" + "" + "For.A" + "" + "For.B" + ""}, + {"Condition/A.csproj", ""}, + {"Condition/B.csproj", ""}, + {"PoisonSeed/Directory.Build.props", + "Known" + "" + ""}, + {"PoisonSeed/A.csproj", ""}, + {"PoisonSeed/B.csproj", ""}, + {"PoisonSeed/A.props", "Hidden" + ""}, + {"PoisonSeed/B.props", "Hidden" + ""}, + {"Implicit/Directory.Build.props", + "" + "enable"}, + {"Implicit/A.csproj", ""}, + {"Implicit/B.csproj", ""}, + {"Implicit/C.csproj", "disable" + ""}, + }; + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + for (size_t i = 0; i < sizeof(files) / sizeof(files[0]); i++) { + ASSERT_TRUE(dm_msb_add_xml(m, files[i].rel_path, files[i].xml)); + } + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + const char *projects[] = {"A.csproj", + "B.csproj", + "Poison.csproj", + "B.csproj", + "Write/A.csproj", + "Write/B.csproj", + "Read/A.csproj", + "Read/B.csproj", + "History/Owned.csproj", + "History/Other.csproj", + "A.csproj", + "Write/A.csproj", + "B.csproj", + "History/Owned.csproj", + "Write/B.csproj", + "Condition/A.csproj", + "Condition/B.csproj", + "Condition/A.csproj", + "PoisonSeed/A.csproj", + "PoisonSeed/B.csproj", + "PoisonSeed/A.csproj", + "Implicit/A.csproj", + "Implicit/B.csproj", + "Implicit/C.csproj", + "Implicit/A.csproj"}; + for (size_t i = 0; i < sizeof(projects) / sizeof(projects[0]); i++) { + ASSERT_TRUE(dm_msb_context_matches(context, m, projects[i])); + } + ASSERT_TRUE(dm_msb_context_matches(context, m, "A.csproj")); + ASSERT_TRUE(dm_msb_add_xml(m, "Later.props", + "" + "")); + ASSERT_TRUE(dm_msb_context_matches(context, m, "A.csproj")); + ASSERT_TRUE(dm_msb_context_matches(context, m, "B.csproj")); + cbm_msb_eval_context_free(context); + ASSERT_EQ(cbm_msb_test_value_live_bytes(), 0); + cbm_msb_free(m); + PASS(); +} + +/* Construction failure must leave no partial prefix, and a failed project + * must not mutate the already retained prefix. Retry on the same context. */ +TEST(doc_mentions_msbuild_prefix_failure) { + char xml[8192]; + size_t w = (size_t)snprintf(xml, sizeof(xml), ""); + for (int i = 0; i < 64; i++) { + w += (size_t)snprintf(xml + w, sizeof(xml) - w, "Shared%02d", i, i, i); + } + w += (size_t)snprintf(xml + w, sizeof(xml) - w, + "" + ""); + ASSERT_TRUE(w < sizeof(xml)); + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.props", xml)); + ASSERT_TRUE(dm_msb_add_xml(m, "A.csproj", "")); + ASSERT_TRUE(dm_msb_add_xml(m, "B.csproj", + "" + "Local")); + int failures = 0; + for (int mode = 0; mode < 2; mode++) { + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + cbm_msb_test_fail_value_alloc_after(mode == 0 ? 8 : 0); + cbm_msb_test_fail_prop_insert_after(mode == 1 ? 8 : 0); + cbm_msb_result_t r = {0}; + bool ok = cbm_msb_eval_context_eval(context, "A.csproj", &r); + bool consumed = + mode == 0 ? cbm_msb_test_value_alloc_failed() : cbm_msb_test_prop_insert_failed(); + failures += ok || !consumed || r.mem != NULL || r.usings != NULL || r.count != 0 || + cbm_msb_test_value_live_bytes() != 0; + cbm_msb_result_free(&r); + cbm_msb_test_fail_value_alloc_after(0); + cbm_msb_test_fail_prop_insert_after(0); + failures += !dm_msb_context_matches(context, m, "A.csproj"); + uint64_t retained = cbm_msb_test_value_live_bytes(); + cbm_msb_test_fail_value_alloc_after(mode == 0 ? 1 : 0); + cbm_msb_test_fail_prop_insert_after(mode == 1 ? 1 : 0); + ok = cbm_msb_eval_context_eval(context, "B.csproj", &r); + consumed = + mode == 0 ? cbm_msb_test_value_alloc_failed() : cbm_msb_test_prop_insert_failed(); + failures += ok || !consumed || r.mem != NULL || r.usings != NULL || r.count != 0 || + cbm_msb_test_value_live_bytes() != retained; + cbm_msb_result_free(&r); + cbm_msb_test_fail_value_alloc_after(0); + cbm_msb_test_fail_prop_insert_after(0); + failures += !dm_msb_context_matches(context, m, "A.csproj"); + failures += !dm_msb_context_matches(context, m, "B.csproj"); + cbm_msb_eval_context_free(context); + failures += cbm_msb_test_value_live_bytes() != 0; + fprintf(stderr, "msbuild prefix fault mode=%d retained=%llu failures=%d\n", mode, + (unsigned long long)retained, failures); + } + cbm_msb_free(m); + ASSERT_EQ(failures, 0); + PASS(); +} + +TEST(doc_mentions_msbuild_eval_storage) { + enum { CAP = 262144, LONG_LEN = 2048 }; + char *xml = malloc(CAP); + char *long_value = malloc(LONG_LEN + 1); + ASSERT_NOT_NULL(xml); + ASSERT_NOT_NULL(long_value); + memset(long_value, 'A', LONG_LEN); + long_value[LONG_LEN] = '\0'; + int failures = 0; + for (int mode = 0; mode < 3; mode++) { + for (int repeats = 256; repeats <= 512; repeats *= 2) { + size_t w = + (size_t)snprintf(xml, CAP, "%s", long_value); + if (mode == 0) { + for (int i = 0; i < repeats; i++) { + w += (size_t)snprintf( + xml + w, CAP - w, + "$(Long)"); + } + w += (size_t)snprintf( + xml + w, CAP - w, + "$(Last)Short" + "" + ""); + } else if (mode == 1) { + w += (size_t)snprintf(xml + w, CAP - w, "Kept
" + "
"); + } else { + w += (size_t)snprintf(xml + w, CAP - w, "
"); + for (int i = 0; i < repeats; i++) { + w += (size_t)snprintf( + xml + w, CAP - w, + ""); + } + w += (size_t)snprintf(xml + w, CAP - w, + "
"); + } + ASSERT_TRUE(w < CAP); + CBMFileResult *source = dm_extract(xml, CBM_LANG_XML, "App.csproj"); + ASSERT_NOT_NULL(source); + ASSERT_NOT_NULL(source->doc_scope); + size_t blob_bytes = strlen(source->doc_scope); + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(cbm_msb_add(m, "App.csproj", source->doc_scope)); + cbm_free_result(source); + cbm_msb_test_cost_reset(); + cbm_msb_result_t r; + ASSERT_TRUE(cbm_msb_eval(m, "App.csproj", &r)); + uint64_t records = 0; + uint64_t peak = 0; + cbm_msb_test_cost(&records, &peak); + bool semantics = !r.open && r.unevaluable == 0 && r.outside == 0; + if (mode == 0) { + semantics = semantics && r.count == 2 && dm_has_using(&r, 'a', long_value) && + dm_has_using(&r, 'n', "Short"); + for (int i = 0; i < r.count; i++) { + if (r.usings[i].kind == 'a') { + semantics = semantics && strcmp(r.usings[i].alias, "Saved") == 0; + } + } + } else { + semantics = semantics && r.count == 1 && dm_has_using(&r, 'n', "Kept"); + } + uint64_t limit = 131072 + 4 * (uint64_t)blob_bytes; + fprintf(stderr, + "msbuild storage mode=%d repeats=%d blob=%zu records=%llu " + "peak=%llu limit=%llu semantics=%d\n", + mode, repeats, blob_bytes, (unsigned long long)records, + (unsigned long long)peak, (unsigned long long)limit, semantics); + failures += !semantics || !records || !peak || peak > limit; + cbm_msb_result_free(&r); + cbm_msb_free(m); + } + } + free(long_value); + free(xml); + ASSERT_EQ(failures, 0); + PASS(); +} + +/* A growing self-assignment reads the old value before replacing it. A + * failed value allocation returns no partial result and does not poison a + * subsequent independent evaluation on the same input model. */ +TEST(doc_mentions_msbuild_value_allocation) { + enum { LONG_LEN = 2048, CAP = 8192 }; + char long_value[LONG_LEN + 2]; + long_value[0] = 'x'; + memset(long_value + 1, 'A', LONG_LEN); + long_value[LONG_LEN + 1] = '\0'; + char xml[CAP]; + int written = snprintf(xml, sizeof(xml), + "x$(V)" + "$(V)%s$(V)Short" + "" + "" + "" + "", + long_value + 1); + ASSERT_TRUE(written > 0 && written < CAP); + const dm_project_file_t files[] = { + {"App.csproj", xml}, + {"More.props", "$(V).Imported" + ""}, + }; + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + for (int i = 0; i < 2; i++) { + CBMFileResult *source = dm_extract(files[i].xml, CBM_LANG_XML, files[i].rel_path); + ASSERT_NOT_NULL(source); + ASSERT_NOT_NULL(source->doc_scope); + ASSERT_TRUE(cbm_msb_add(m, files[i].rel_path, source->doc_scope)); + cbm_free_result(source); + } + const int fault_after[] = {0, 1, 0, 8, 0, 0, 0}; + const int insert_after[] = {0, 0, 0, 0, 0, 6, 0}; + int failures = 0; + for (size_t i = 0; i < sizeof(fault_after) / sizeof(fault_after[0]); i++) { + cbm_msb_test_fail_value_alloc_after(fault_after[i]); + cbm_msb_test_fail_prop_insert_after(insert_after[i]); + cbm_msb_result_t r; + bool ok = cbm_msb_eval(m, "App.csproj", &r); + bool consumed = cbm_msb_test_value_alloc_failed(); + bool insert_failed = cbm_msb_test_prop_insert_failed(); + bool injected = fault_after[i] || insert_after[i]; + failures += consumed != (fault_after[i] > 0) || insert_failed != (insert_after[i] > 0) || + cbm_msb_test_value_live_bytes() != 0; + if (injected) { + failures += ok || r.mem != NULL || r.usings != NULL || r.count != 0; + } else { + failures += !ok || consumed || r.open || r.count != 3 || !dm_has_using(&r, 'a', "x") || + !dm_has_using(&r, 'a', long_value) || + !dm_has_using(&r, 'n', "Short.Imported"); + for (int k = 0; k < r.count; k++) { + if (r.usings[k].kind == 'a') { + const char *alias = + strcmp(r.usings[k].target, "x") == 0 ? "Saved" : "LongSaved"; + failures += strcmp(r.usings[k].alias, alias) != 0; + } + } + } + fprintf(stderr, + "msbuild value allocation nth=%d consumed=%d insert=%d/%d " + "live=%llu success=%d failures=%d\n", + fault_after[i], consumed, insert_after[i], insert_failed, + (unsigned long long)cbm_msb_test_value_live_bytes(), ok, failures); + cbm_msb_result_free(&r); + } + cbm_msb_test_fail_value_alloc_after(0); + cbm_msb_test_fail_prop_insert_after(0); + cbm_msb_free(m); + ASSERT_EQ(failures, 0); + PASS(); +} + +/* Scratch resets must not change copied properties, output strings, or the + * poisoning caused by an unknown import (including an imported child). */ +TEST(doc_mentions_msbuild_value_lifetimes) { + const dm_project_file_t files[] = { + {"App.csproj", + "" + "One;Two$(Seed)Changed" + "ReadyAliasKept" + "$(Empty)Tail" + "" + "Recovered$(Empty)" + "$(Empty)Again" + "" + "" + "" + "" + "" + "" + "" + ""}, + {"Unknown.props", "Changed" + ""}, + {"Child.props", + "Unknown.ValuePoison" + ""}, + }; + cbm_msb_result_t r; + ASSERT_TRUE(dm_msb_eval(files, 3, "App.csproj", &r)); + ASSERT_EQ(r.count, 6); + ASSERT_TRUE(dm_has_using(&r, 'a', "One")); + ASSERT_TRUE(dm_has_using(&r, 'a', "Two")); + ASSERT_TRUE(dm_has_using(&r, 's', "Static.Kept")); + ASSERT_TRUE(dm_has_using(&r, 'n', "Tail.Kept")); + ASSERT_TRUE(dm_has_using(&r, 'n', "Tail")); + ASSERT_TRUE(dm_has_using(&r, 'n', "Again")); + for (int i = 0; i < r.count; i++) { + if (r.usings[i].kind == 'a') { + ASSERT_STR_EQ(r.usings[i].alias, "AliasKept"); + } + } + ASSERT_TRUE(r.open); + ASSERT_EQ(r.unevaluable, 3); + ASSERT_EQ(r.outside, 0); + cbm_msb_result_free(&r); + PASS(); +} + +TEST(doc_mentions_msbuild_usings) { + const dm_project_file_t files[] = { + {"Directory.Build.props", "\n" + " web\n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + "\n"}, + {"src/App/App.csproj", + "\xEF\xBB\xBF\r\n" + "\r\n" + "\r\n" + " enable\r\n" + " \r\n" + " \r\n" + " \r\n" + " \r\n" + " \r\n" + " \r\n" + " \r\n" + " \r\n" + " \r\n" + "\r\n"}, + }; + cbm_msb_result_t r; + ASSERT_TRUE(dm_msb_eval(files, 2, "src/App/App.csproj", &r)); + ASSERT_TRUE(dm_has_using(&r, 'n', "System")); /* ImplicitUsings: the SDK default set */ + ASSERT_TRUE(dm_has_using(&r, 'n', "Acme.Extra")); /* */ + ASSERT_TRUE(dm_has_using(&r, 'n', "From.Props")); /* Directory.Build.props, condition true */ + ASSERT_FALSE(dm_has_using(&r, 'n', "System.Linq")); /* */ + ASSERT_FALSE(dm_has_using(&r, 'n', "Never.Here")); /* condition false */ + /* Static="true" names a type and Alias an alias: neither is a namespace */ + ASSERT_FALSE(dm_has_using(&r, 'n', "Acme.Static")); + ASSERT_TRUE(dm_has_using(&r, 's', "Acme.Static")); + ASSERT_FALSE(dm_has_using(&r, 'n', "Acme.Aliased")); + ASSERT_TRUE(dm_has_using(&r, 'a', "Acme.Aliased")); + /* an Include is a ';'-separated list */ + ASSERT_TRUE(dm_has_using(&r, 'n', "Acme.One")); + ASSERT_TRUE(dm_has_using(&r, 'n', "Acme.Two")); + ASSERT_FALSE(dm_has_using(&r, 'n', "Acme.One; Acme.Two ;;")); + /* comments, other items and targets are no usings */ + ASSERT_FALSE(dm_has_using(&r, 'n', "In.A.Comment")); + ASSERT_FALSE(dm_has_using(&r, 'n', "Not.A.Using")); + ASSERT_FALSE(dm_has_using(&r, 'n', "In.A.Target")); + /* everything was evaluated: nothing is open */ + ASSERT_FALSE(r.open); + ASSERT_EQ(r.unevaluable, 0); + ASSERT_EQ(r.outside, 0); + cbm_msb_result_free(&r); + PASS(); +} + +/* : a path is relative to the importing file; the two directory + * properties are absolute and are not joined to it again; what the index + * does not hold is not imported, and is counted. */ +TEST(doc_mentions_msbuild_imports) { + const dm_project_file_t files[] = { + {"src/Directory.Build.props", + "\n" + " $(MSBuildThisFileDirectory)eng/\n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + "\n"}, + {"src/eng/ByThisFile.props", + "\n" + " \n" /* itself: taken once */ + "\n"}, + {"src/eng/ByProperty.props", + "\n"}, + {"src/eng/Relative.props", + "\n"}, + {"src/eng/NotTaken.props", + "\n"}, + {"Up.props", "\n"}, + {"src/App/App.csproj", "\n" + " \n" + "\n"}, + {"src/App/Local.props", + "\n"}, + }; + cbm_msb_result_t r; + ASSERT_TRUE(dm_msb_eval(files, 8, "src/App/App.csproj", &r)); + ASSERT_TRUE(dm_has_using(&r, 'n', "By.ThisFile")); + ASSERT_TRUE(dm_has_using(&r, 'n', "By.Property")); + ASSERT_TRUE(dm_has_using(&r, 'n', "By.Relative")); + ASSERT_TRUE(dm_has_using(&r, 'n', "By.Up")); + ASSERT_TRUE(dm_has_using(&r, 'n', "By.ProjectDir")); + ASSERT_FALSE(dm_has_using(&r, 'n', "Not.Taken")); /* its group's condition is false */ + /* the path through an unknown property and the wildcard */ + ASSERT_EQ(r.unevaluable, 2); + /* above the repository, absolute, and not in the index */ + ASSERT_EQ(r.outside, 3); + ASSERT_FALSE(r.open); /* an import that cannot be followed leaves nothing open */ + cbm_msb_result_free(&r); + PASS(); +} + +/* One condition over fixed properties: the usings it lets through. */ +static bool dm_cond_using(const char *cond, cbm_msb_result_t *out) { + char xml[1024]; + snprintf(xml, sizeof(xml), + "\n" + " \n" + " xxytrue\n" + " x 1\n" + " \n" + " \n" + " \n" + " \n" + "\n", + cond); + const dm_project_file_t files[] = {{"P.csproj", xml}}; + return dm_msb_eval(files, 1, "P.csproj", out); +} + +/* A condition is true, false, or not known. What is not known is never + * taken for one of the two: the item is not applied, the project's usings + * are open, and the count says so. */ +TEST(doc_mentions_msbuild_conditions) { + static const struct { + const char *cond; + char want; /* T applied, F not applied, U not known */ + } cases[] = { + {"'$(A)' == 'x'", 'T'}, + {"'$(A)' == 'X'", 'T'}, /* comparison ignores case */ + {"'$(A)' != 'x'", 'F'}, + {"'$(A)' == '$(B)'", 'T'}, /* both sides are expanded */ + {"'$(A)' == '$(C)'", 'F'}, + {"$(A) == x", 'T'}, /* unquoted operands */ + {"'$(A)-$(C)' == 'x-y'", 'T'}, + {"'$(Empty)' == ''", 'T'}, + {"$(Yes)", 'T'}, + {"!$(Yes)", 'F'}, + {"'$(Yes)' == 'on'", 'T'}, /* booleans compare as booleans */ + {"'$(One)' == '1.0'", 'T'}, /* numbers as numbers */ + {"'$(One)' == '2'", 'F'}, + /* `and` binds tighter than `or`, parentheses group */ + {"'e' == 'e' or 'a' == 'b' and 'c' == 'd'", 'T'}, + {"('e' == 'e' or 'a' == 'b') and 'c' == 'd'", 'F'}, + {"'a' == 'a' or ('b' == 'c' and 'd' == 'e')", 'T'}, + {"!('a' == 'b')", 'T'}, + /* not known: a property no file sets ... */ + {"'$(Undefined)' == ''", 'U'}, + {"'$(Undefined)' != 'v'", 'U'}, + /* ... unless the other operand decides it */ + {"'a' == 'b' and '$(Undefined)' == 'v'", 'F'}, + {"'a' == 'a' or '$(Undefined)' == 'v'", 'T'}, + {"'a' == 'a' and '$(Undefined)' == 'v'", 'U'}, + /* property functions, item lists, function calls, an order */ + {"'$(A.StartsWith('x'))' == 'true'", 'U'}, + {"'@(Compile)' == ''", 'U'}, + {"Exists('x.props')", 'U'}, + {"'a' == 'b' and Exists('x.props')", 'F'}, + {"'$(One)' > '0'", 'U'}, + /* white space a property's element was written with */ + {"'$(Spaced)' == 'x'", 'U'}, + {"'$(Spaced)' == 'y'", 'F'}, + /* no condition this reader can parse */ + {"'a' == ", 'U'}, + {"('a' == 'a'", 'U'}, + {"'a' == 'a' 'b'", 'U'}, + }; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + cbm_msb_result_t r; + if (!dm_cond_using(cases[i].cond, &r)) { + printf(" [%s]: no evaluation\n", cases[i].cond); + FAIL("condition"); + } + bool applied = dm_has_using(&r, 'n', "Under.Condition"); + char got = applied ? 'T' : ((r.open && r.unevaluable > 0) ? 'U' : 'F'); + bool consistent = + applied ? (!r.open && r.unevaluable == 0) : (r.open == (r.unevaluable > 0)); + cbm_msb_result_free(&r); + if (got != cases[i].want || !consistent) { + printf(" [%s]: %c, want %c%s\n", cases[i].cond, got, cases[i].want, + consistent ? "" : " (open and the count disagree)"); + FAIL("condition"); + } + } + PASS(); +} + +/* What an element under an unknown condition could have set is not known + * afterwards, and neither is anything that reads it. */ +TEST(doc_mentions_msbuild_unknown_spreads) { + const dm_project_file_t files[] = { + {"Maybe.props", "\n" + " v\n" + " \n" + "\n"}, + {"P.csproj", + "\n" + " \n" + " plain\n" + " legacy\n" + " yes\n" + " enable\n" + " \n" + " \n" + " \n" + " 1\n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + "\n"}, + }; + cbm_msb_result_t r; + ASSERT_TRUE(dm_msb_eval(files, 2, "P.csproj", &r)); + /* Mode is `plain` or `legacy`: nothing that depends on it is applied */ + ASSERT_FALSE(dm_has_using(&r, 'n', "Reads.Poisoned")); + ASSERT_FALSE(dm_has_using(&r, 'n', "plain.Ns")); + ASSERT_FALSE(dm_has_using(&r, 'n', "System")); /* ImplicitUsings: not known either */ + /* a file imported under an unknown condition: its properties and usings */ + ASSERT_FALSE(dm_has_using(&r, 'n', "Reads.Maybe")); + ASSERT_FALSE(dm_has_using(&r, 'n', "From.Maybe")); + /* a property a sets */ + ASSERT_FALSE(dm_has_using(&r, 'n', "Reads.Choose")); + /* what does not depend on any of it stands */ + ASSERT_TRUE(dm_has_using(&r, 'n', "Reads.Sure")); + ASSERT_TRUE(r.open); + ASSERT_GT(r.unevaluable, 5); + cbm_msb_result_free(&r); + PASS(); +} + +/* Growing one group's condition must not copy it once per child import. */ +TEST(doc_mentions_msbuild_import_group_blob_growth) { + enum { IMPORTS = 64, CONDITION_GROWTH = 512 }; + size_t sizes[2] = {0}; + for (int sample = 0; sample < 2; sample++) { + char xml[4096]; + const char *head = "", 2); + used += 2; + for (int i = 0; i < IMPORTS; i++) { + const char *child = ""; + size_t n = strlen(child); + ASSERT_LT(used + n, sizeof(xml)); + memcpy(xml + used, child, n); + used += n; + } + const char *tail = ""; + ASSERT_LT(used + strlen(tail), sizeof(xml)); + memcpy(xml + used, tail, strlen(tail) + 1); + CBMFileResult *r = dm_extract(xml, CBM_LANG_XML, "App.csproj"); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + sizes[sample] = strlen(r->doc_scope); + cbm_free_result(r); + } + ASSERT_GTE(sizes[1], sizes[0]); + ASSERT_LTE(sizes[1] - sizes[0], 2 * CONDITION_GROWTH); + PASS(); +} + +TEST(doc_mentions_msbuild_import_group_roundtrip) { + const dm_project_file_t files[] = { + {"App.csproj", + "" + "" + "" + "" + ""}, + {"A.props", "" + ""}, + {"B.props", ""}, + {"Nested.props", "" + "" + ""}, + {"Skip.props", ""}, + }; + cbm_msb_result_t r; + ASSERT_TRUE(dm_msb_eval(files, 5, "App.csproj", &r)); + ASSERT_TRUE(dm_has_using(&r, 'n', "From.A")); + ASSERT_TRUE(dm_has_using(&r, 'n', "From.B")); + ASSERT_TRUE(dm_has_using(&r, 'n', "From.Nested")); + ASSERT_FALSE(dm_has_using(&r, 'n', "Not.Taken")); + ASSERT_EQ(r.count, 3); + ASSERT_FALSE(r.open); + cbm_msb_result_free(&r); + PASS(); +} + +TEST(doc_mentions_msbuild_legacy_import_blob) { + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(cbm_msb_add(m, "App.csproj", + "cs1\nP\t\t-\n" + "I\t=true\t\t=Take.props\t\n" + "I\t=false\t\t=Skip.props\t\n")); + ASSERT_TRUE(cbm_msb_add(m, "Take.props", "cs1\nP\t\t-\nH\t\nN\t\t=Taken\t\t\t\n")); + ASSERT_TRUE(cbm_msb_add(m, "Skip.props", "cs1\nP\t\t-\nH\t\nN\t\t=Skipped\t\t\t\n")); + cbm_msb_result_t r; + ASSERT_TRUE(cbm_msb_eval(m, "App.csproj", &r)); + ASSERT_TRUE(dm_has_using(&r, 'n', "Taken")); + ASSERT_FALSE(dm_has_using(&r, 'n', "Skipped")); + ASSERT_EQ(r.count, 1); + ASSERT_FALSE(r.open); + cbm_msb_result_free(&r); + cbm_msb_free(m); + PASS(); +} + +TEST(doc_mentions_msbuild_bad_import_group_blob) { + const char *bad[] = { + "J\t\t\t=Take.props\t\n", /* orphan import */ + "B\t=true\nJ\t\t\t=Take.props\t\n", /* missing end */ + "B\t=true\nB\t=false\nE\nE\n", /* nested group */ + "E\n", /* orphan end */ + "B\t=true\nI\t\t\t=Take.props\t\nE\n", /* wrong child */ + "B\t=true\nJ\t=false\t\t=Take.props\t\nE\n", /* inline group condition */ + }; + for (size_t i = 0; i < sizeof(bad) / sizeof(bad[0]); i++) { + char blob[256]; + int n = snprintf(blob, sizeof(blob), "cs1\nP\t\t-\n%s", bad[i]); + ASSERT_GT(n, 0); + ASSERT_LT((size_t)n, sizeof(blob)); + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(cbm_msb_add(m, "App.csproj", blob)); + ASSERT_TRUE(cbm_msb_add(m, "Take.props", "cs1\nP\t\t-\nH\t\nN\t\t=Taken\t\t\t\n")); + cbm_msb_result_t r; + ASSERT_TRUE(cbm_msb_eval(m, "App.csproj", &r)); + ASSERT_TRUE(r.open); + ASSERT_GT(r.unevaluable, 0); + ASSERT_EQ(r.count, 0); + cbm_msb_result_free(&r); + cbm_msb_free(m); + } + PASS(); +} + +/* What the scanner makes of a project file. */ +TEST(doc_mentions_msbuild_blob) { + CBMFileResult *r = dm_extract("\n" + " \n" + " one\ttwo\n" + " & ]]>tail\n" + " \n" + " a\\bAB\n" + " \n" + " \n" + " \n" + " A\n" + " \n" + " \n" + "\n", + CBM_LANG_XML, "src/App.csproj"); + ASSERT_NOT_NULL(r); + const char *s = r->doc_scope; + ASSERT_NOT_NULL(s); + ASSERT_TRUE(cbm_msb_is_project_scope(s)); + ASSERT(strncmp(s, "cs1\nP\t=Microsoft.NET.Sdk/8.0\t-\n", 30) == 0); + ASSERT_NOT_NULL(strstr(s, "\nG\t='$(A)' < 'b'\n")); /* entities are decoded */ + ASSERT_NOT_NULL(strstr(s, "\nV\t\tPlain\t=one\\ttwo\n")); /* a tab is escaped */ + ASSERT_NOT_NULL(strstr(s, "\nV\t\tCdata\t= & tail\n")); + ASSERT_NOT_NULL(strstr(s, "\nV\t\tNested\t?\n")); /* no plain text */ + ASSERT_NOT_NULL(strstr(s, "\nV\t\tBack\t=a\\\\bAB\n")); + ASSERT_NULL(strstr(s, "Reference")); /* an item group without a is not kept */ + ASSERT_NOT_NULL(strstr(s, "\nY\nY\n")); /* metadata elements, Update: not evaluated */ + ASSERT_NULL(strstr(s, "\nN\t")); + /* the persisted form is the blob itself: it has no line numbers */ + char *portable = cbm_doclink_portable_scope(s); + ASSERT_NOT_NULL(portable); + ASSERT_STR_EQ(portable, s); + cbm_free(CBM_MEM_CLASS_OTHER, portable); + cbm_free_result(r); + + /* XML that is no MSBuild project has no scope: by its name ... */ + r = dm_extract("\n", + CBM_LANG_XML, "src/app.config.xml"); + ASSERT_NOT_NULL(r); + ASSERT_NULL(r->doc_scope); + cbm_free_result(r); + /* ... or, under a project file's name, by its root element */ + r = dm_extract("\n", CBM_LANG_XML, + "src/Settings.props"); + ASSERT_NOT_NULL(r); + ASSERT_NULL(r->doc_scope); + cbm_free_result(r); + /* Directory.Build.props and .targets files are project files */ + r = dm_extract("\n", + CBM_LANG_XML, "Directory.Build.targets"); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + ASSERT_NOT_NULL(strstr(r->doc_scope, "\nN\t\t=X\t\t\t\n")); + cbm_free_result(r); + /* a *.csproj that cannot be read is still a project: its blob says so */ + r = dm_extract("1\n", CBM_LANG_XML, "Bad.csproj"); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + ASSERT_STR_EQ(r->doc_scope, "cs1\nP\t\t!\n"); + cbm_free_result(r); + r = dm_extract("not xml at all", CBM_LANG_XML, "Worse.CSPROJ"); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + ASSERT_STR_EQ(r->doc_scope, "cs1\nP\t\t!\n"); + cbm_free_result(r); + PASS(); +} + +/* ── the C# resolver end to end ──────────────────────────────────── */ + +static void dm_write_resolver_fixture(const char *tmp) { + th_write_file(TH_PATH(tmp, "src/Lib/Lib.csproj"), "\n" + " \n" + " \n" + " \n" + " \n" + " \n" + "\n"); + th_write_file(TH_PATH(tmp, "src/Lib/Extra.cs"), "namespace Acme.Extra\n" + "{\n" + " public class Tool { }\n" + "}\n" + "namespace Acme.Removed\n" + "{\n" + " public class Gone { }\n" + "}\n"); + th_write_file(TH_PATH(tmp, "src/Lib/Helper.cs"), + "namespace Acme.Util\n" + "{\n" + " public static class Helper\n" + " {\n" + " public static void Run(int n) { }\n" + " public static void Run(string s) { }\n" + " public static void Once(int n) { }\n" + " }\n" + "}\n"); + th_write_file(TH_PATH(tmp, "src/Lib/Global.cs"), "public class Size { }\n"); + /* the usual polyfill: it makes `System` a namespace of the repository */ + th_write_file(TH_PATH(tmp, "src/Lib/Shim.cs"), "namespace System.Runtime.CompilerServices\n" + "{\n" + " internal static class IsExternalInit { }\n" + "}\n"); + th_write_file(TH_PATH(tmp, "src/Lib/tests/OnlyInTests.cs"), + "namespace Acme.Core\n" + "{\n" + " public class TestOnlyThing { }\n" + "}\n"); + th_write_file( + TH_PATH(tmp, "src/Lib/Widgets.cs"), + "using Acme.Util;\n" + "using H = Acme.Util.Helper;\n" + "\n" + "namespace Acme.Core\n" + "{\n" + " /// \n" + " /// A widget: \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// when bad\n" + " public class Widget\n" + " {\n" + " /// \n" + " public void Spin() { }\n" + "\n" + " /// Group , one\n" + " /// , wrong .\n" + " /// \n" + " public void Go() { }\n" + "\n" + " /// Twice , ; self\n" + " /// ; member .\n" + " public void Self() { }\n" + "\n" + " /// Arity: and .\n" + " public int Size;\n" + " }\n" + "\n" + " /// Gadget.\n" + " public class Gadget\n" + " {\n" + " public void Spin() { }\n" + " }\n" + "\n" + " public class WidgetError { }\n" + "\n" + " public class Box { }\n" + " public class Box { }\n" + "\n" + " public interface IPinger { void Ping(int n); }\n" + "\n" + " public class Ping { }\n" + "\n" + " public class Pinger : IPinger\n" + " {\n" + " void IPinger.Ping(int n) { }\n" + "\n" + " /// Explicit implementations are not addressable, and an\n" + " /// interface's members are not the class's: ,\n" + " /// .\n" + " public void Other() { }\n" + " }\n" + "}\n"); + th_write_file( + TH_PATH(tmp, "src/Other/Plain.cs"), + "namespace Acme.Plain\n" + "{\n" + " /// and .\n" + " public class Plain { }\n" + "}\n"); + /* arity, constructors, generic methods, same-path declarations */ + th_write_file(TH_PATH(tmp, "src/Other/Lists.cs"), + "namespace Coll\n" + "{\n" + " public interface IList { int IndexOf(object o); }\n" + "}\n"); + th_write_file(TH_PATH(tmp, "src/Other/GenericLists.cs"), "namespace Coll.Generic\n" + "{\n" + " public interface IList { }\n" + " public class OnlyGeneric { }\n" + "}\n"); + th_write_file( + TH_PATH(tmp, "src/Other/Arity.cs"), + "using Coll;\n" + "using Coll.Generic;\n" + "\n" + "namespace Acme.Arity\n" + "{\n" + " public class Pair { public int Left; }\n" + " public class Pair { public int Left; public int Right; }\n" + "\n" + " public class Conv\n" + " {\n" + " public Conv() { }\n" + " public Conv(int x) { }\n" + " public void To(int x) { }\n" + " public void To(T x) { }\n" + " public void Many(string first, params int[] rest) { }\n" + " }\n" + "\n" + " public class Derived : Conv { }\n" + "\n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " public class Uses : Conv { }\n" + "}\n"); + /* declarations the parser cannot place */ + th_write_file( + TH_PATH(tmp, "src/Other/Recovered.cs"), + "namespace Acme.Rec\n" + "{\n" + " public class Before { }\n" + " public ref partial struct Iter\n" + " {\n" + " public void End() { }\n" + " }\n" + " public class Holey\n" + " {\n" + " public safe extern int Hidden();\n" + " /// \n" + " public int Seen;\n" + " }\n" + " /// \n" + " /// \n" + " public class Middle { }\n" + "}\n"); + th_write_file(TH_PATH(tmp, "src/Other/Unbalanced.cs"), "namespace Acme.Rec\n" + "{\n" + " public class Lost { }\n" + " public class Open\n" + " {\n" + " public void M() {\n"); + /* a namespace only one build configuration can name: its block is not + * placed, though the declarations in it are extracted */ + th_write_file(TH_PATH(tmp, "src/Other/Either.cs"), + "#if GEN\n" + "namespace Gen.Interop\n" + "#else\n" + "namespace Run.Interop\n" + "#endif\n" + "{\n" + " /// \n" + " public class InEither { }\n" + "}\n"); + /* an incomplete type in a file that imports a namespace the repository + * does not declare */ + th_write_file(TH_PATH(tmp, "src/Other/OpenScope.cs"), + "using Outside.Lib;\n" + "\n" + "namespace Acme.Rec2\n" + "{\n" + " public class Holey2\n" + " {\n" + " public safe extern int Hidden2();\n" + " /// \n" + " public int Seen2;\n" + " }\n" + "}\n"); + /* a field and a method of one name (two declarations of a partial type + * can do that to a reader who sees no build configuration) */ + th_write_file(TH_PATH(tmp, "src/Other/Kinds.cs"), + "namespace Acme.Kinds\n" + "{\n" + " public class Mixed\n" + " {\n" + " public int Both;\n" + " public void Both(int x) { }\n" + " public int Only;\n" + " }\n" + " /// \n" + " /// \n" + " public class UsesKinds { }\n" + "}\n"); + /* two test programs, each with types in the global namespace */ + th_write_file(TH_PATH(tmp, "tests/ProgA/ProgA.csproj"), + "\n\n"); + th_write_file(TH_PATH(tmp, "tests/ProgA/Shared.cs"), + "public class Shared { }\n" + "/// \n" + "public class UsesOwn { }\n"); + th_write_file(TH_PATH(tmp, "tests/ProgB/ProgB.csproj"), + "\n\n"); + th_write_file(TH_PATH(tmp, "tests/ProgB/Uses.cs"), + "/// \n" + "public class UsesOther { }\n"); + /* ... and one only test code fails to place: no concern of product code */ + th_write_file(TH_PATH(tmp, "src/Other/tests/Half.cs"), "namespace Acme.Core\n" + "{\n" + " public class Gadget { }\n" + " public class Cut\n" + " {\n" + " public void M() {\n"); +} + +TEST(doc_mentions_resolver_rules) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_res_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + dm_write_resolver_fixture(tmp); + char db[512]; + snprintf(db, sizeof(db), "%s/res.db", tmp); + char *project = NULL; + ASSERT_EQ(dm_index(tmp, db, &project), 0); + free(project); + char props[512]; + char reason[64]; + char syntax[64]; + int n = 0; + const char *widgets = "src/Lib/Widgets.cs"; + + /* every cref-bearing tag becomes an edge with its syntax */ + dm_edge(db, "Widgets.Widget", "Widgets.Gadget", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + ASSERT_NOT_NULL(strstr(props, "\"via\":\"doc_comment\"")); + ASSERT_NOT_NULL(strstr(props, "\"syntax\":\"see\"")); + ASSERT_NOT_NULL(strstr(props, "\"tier\":\"unique\"")); + ASSERT_NOT_NULL(strstr(props, "\"line\":7")); + dm_edge(db, "Widgets.Widget", "Helper.Helper.Once", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + ASSERT_NOT_NULL(strstr(props, "\"syntax\":\"seealso\"")); + dm_edge(db, "Widgets.Widget", "Widgets.WidgetError", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + ASSERT_NOT_NULL(strstr(props, "\"syntax\":\"exception\"")); + dm_edge(db, "Widgets.Widget.Spin", "Widgets.Gadget.Spin", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + ASSERT_NOT_NULL(strstr(props, "\"syntax\":\"inheritdoc\"")); + /* alias: exact */ + dm_edge(db, "Widgets.Widget", "Helper.Helper", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + ASSERT_NOT_NULL(strstr(props, "\"tier\":\"exact\"")); + /* R1: a namespace imported only by the project's MSBuild */ + dm_edge(db, "Widgets.Widget", "Extra.Tool", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + /* ... and not one the project file removes again */ + dm_edge(db, "Widgets.Widget", "Extra.Gone", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + dm_row(db, widgets, "Gone", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + /* local and keyword references are neither edges nor rows */ + dm_row(db, widgets, "x", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, ""); + /* a URL is external */ + dm_row(db, widgets, "https://example.org/docs", reason, sizeof(reason), syntax, sizeof(syntax)); + ASSERT_STR_EQ(reason, "external"); + ASSERT_STR_EQ(syntax, "href"); + /* R4: keyword aliases and System.* names outside the corpus -- although a + * polyfill (Shim.cs) makes `System` one of the repository's namespaces */ + dm_row(db, widgets, "string", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "external"); + dm_row(db, widgets, "System.Text.StringBuilder", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "external"); + /* a type the repository's own namespace does not declare is doc rot, not + * a gap of the graph */ + dm_row(db, widgets, "Acme.Core.NoSuchType", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + /* a namespace itself is declared code without a node */ + dm_row(db, widgets, "Acme.Util", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "graph_gap"); + /* product code never binds a test-only declaration */ + dm_row(db, widgets, "TestOnlyThing", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "test_only_target"); + /* a test program binds its own global-namespace types, never another + * program's */ + dm_edge(db, "Shared.UsesOwn", "Shared.Shared", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + ASSERT_EQ(dm_mentions_from(db, "Uses.UsesOther"), 0); + dm_row(db, "tests/ProgB/Uses.cs", "Shared", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "test_only_target"); + dm_row(db, widgets, "Bad Syntax!", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "unparseable"); + + /* overloads: a group without a signature is ambiguous, a signature picks */ + dm_row(db, widgets, "Helper.Run", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "ambiguous"); + dm_edge(db, "Widgets.Widget.Go", "Helper.Helper.Run", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_row(db, widgets, "Helper.Run(double)", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + + /* one edge per (source, target): count and the first line; no self edge */ + dm_edge(db, "Widgets.Widget.Self", "Widgets.Gadget", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + ASSERT_NOT_NULL(strstr(props, "\"count\":2")); + ASSERT_NOT_NULL(strstr(props, "\"line\":24")); + dm_edge(db, "Widgets.Widget.Self", "Widgets.Widget.Self", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + dm_row(db, widgets, "Self", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, ""); + /* a single segment never takes a qualified shortcut: the member Size, + * not the global-namespace class Size */ + dm_edge(db, "Widgets.Widget.Self", "Widgets.Widget.Size", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_edge(db, "Widgets.Widget.Self", "Global.Size", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + + /* R3: type arguments select the generic type; R2: `Box` names the + * arity-0 type, whose node the generic twin took -- a graph gap, never a + * fallback to Box */ + dm_edge(db, "Widgets.Widget.Size", "Widgets.Box", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_row(db, widgets, "Box", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "graph_gap"); + + /* An explicit interface implementation is not addressable by name, and a + * class does not find the members of an interface it implements (the + * compiler looks up no inherited member in a cref): `Ping` is the class of + * that name in the namespace, `Ping(int)` a constructor it does not have. */ + dm_edge(db, "Widgets.Pinger.Other", "Widgets.Ping", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_edge(db, "Widgets.Pinger.Other", "Widgets.IPinger.Ping", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + dm_edge(db, "Widgets.Pinger.Other", "Widgets.Pinger.Ping", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + dm_row(db, widgets, "Ping(int)", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + + /* a closed scope: missing */ + dm_row(db, "src/Other/Plain.cs", "Nowhere", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + dm_row(db, "src/Other/Plain.cs", "Plain.Missing", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + + dm_unlink_db(db); + th_rmtree(tmp); + PASS(); +} + +/* Arity (R2/R3), constructors, generic methods, `params`, and declarations + * that share one qualified name. */ +TEST(doc_mentions_resolver_arity_and_members) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_ar_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + dm_write_resolver_fixture(tmp); + char db[512]; + snprintf(db, sizeof(db), "%s/res.db", tmp); + ASSERT_EQ(dm_index(tmp, db, NULL), 0); + char props[512]; + char reason[64]; + int n = 0; + const char *arity = "src/Other/Arity.cs"; + + /* R2: `IList` is the arity-0 interface, wherever one is in scope -- the + * generic IList of another imported namespace is no rival */ + dm_edge(db, "Arity.Uses", "Lists.IList.IndexOf", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_row(db, arity, "IList.IndexOf", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, ""); + /* ... and a generic type never stands in for a name written without type + * arguments, not even when no arity-0 type of that name is in scope: the + * compiler binds `OnlyGeneric` to nothing. Only `OnlyGeneric{T}` names it. */ + dm_row(db, arity, "OnlyGeneric", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + dm_edge(db, "Arity.Uses", "GenericLists.OnlyGeneric", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + ASSERT_NOT_NULL(strstr(props, "\"count\":1")); /* OnlyGeneric{T} only */ + /* R3: type arguments also select a generic METHOD of that arity */ + dm_edge(db, "Arity.Uses", "Arity.Conv.To", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + /* `Pair.Left` and `Pair.Left` share one node, and it is the later + * declaration's: the arity-0 member is a graph gap, like its type */ + dm_row(db, arity, "Pair.Left", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "graph_gap"); + dm_edge(db, "Arity.Uses", "Arity.Pair.Left", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + ASSERT_NOT_NULL(strstr(props, "\"count\":1")); /* Pair{T}.Left only */ + dm_edge(db, "Arity.Uses", "Arity.Pair.Right", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + /* a bare `Conv` is the type: the constructors of a base class are neither + * inherited nor what a name without a parameter list means */ + dm_edge(db, "Arity.Uses", "Arity.Conv", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_row(db, arity, "Conv", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, ""); + /* a parameter list names the constructor */ + dm_edge(db, "Arity.Uses", "Arity.Conv.Conv", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + /* ... of that type only */ + dm_row(db, arity, "Derived.Conv", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + /* a `params` parameter is part of the signature */ + dm_edge(db, "Arity.Uses", "Arity.Conv.Many", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* member kind: a field and a method of one name are ambiguous without a + * parameter list; with one, only the callable is meant */ + const char *kinds = "src/Other/Kinds.cs"; + dm_row(db, kinds, "Mixed.Both", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "ambiguous"); + dm_row(db, kinds, "Mixed.Both(int)", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, ""); + dm_edge(db, "Kinds.UsesKinds", "Kinds.Mixed.Both", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + ASSERT_NOT_NULL(strstr(props, "\"count\":1")); + dm_edge(db, "Kinds.UsesKinds", "Kinds.Mixed.Only", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + dm_unlink_db(db); + th_rmtree(tmp); + PASS(); +} + +/* Declarations the parser cannot place are never resolved around. */ +TEST(doc_mentions_resolver_parse_errors) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_pe_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + dm_write_resolver_fixture(tmp); + char db[512]; + snprintf(db, sizeof(db), "%s/res.db", tmp); + ASSERT_EQ(dm_index(tmp, db, NULL), 0); + char props[512]; + char reason[64]; + int n = 0; + const char *rec = "src/Other/Recovered.cs"; + + /* `Middle` follows a struct the grammar cannot parse; the tree puts it in + * the global namespace, the braces keep it in Acme.Rec, where `Before` is */ + dm_edge(db, "Recovered.Middle", "Recovered.Before", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + /* the unparsed struct is declared (its header is read from the text) but + * has no node */ + dm_row(db, rec, "Iter", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "graph_gap"); + /* a member behind a parse error: a gap, not doc rot */ + dm_row(db, rec, "Holey.Hidden", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "graph_gap"); + dm_edge(db, "Recovered.Middle", "Recovered.Holey.Seen", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + /* inside the incomplete type: a simple name found nowhere may be the + * hidden member -- a gap, where it would otherwise be reported missing */ + dm_row(db, rec, "Hidden", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "graph_gap"); + /* ... but a keyword alias is never a member, */ + dm_row(db, rec, "int", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "external"); + /* ... and a name an imported namespace outside the repository may supply + * stays external */ + dm_row(db, "src/Other/OpenScope.cs", "Thing", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "external"); + /* a type in a file whose braces do not pair: its name resolves to nothing */ + dm_row(db, rec, "Lost", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "graph_gap"); + /* what is documented in a block that could not be placed has no scope to + * resolve in: a gap, not a lookup from the global namespace (which would + * report `Before`, a class of Acme.Rec, as missing) */ + dm_row(db, "src/Other/Either.cs", "Before", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "graph_gap"); + ASSERT_EQ(dm_mentions_from(db, "Either.InEither"), 0); + /* ... unless only TEST code fails to place the name: product code never + * binds test declarations anyway (tests/Half.cs quarantines `Gadget`) */ + dm_edge(db, "Widgets.Widget", "Widgets.Gadget", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_row(db, rec, "Gadget", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); /* Acme.Core.Gadget is not in Acme.Rec's scope */ + + dm_unlink_db(db); + th_rmtree(tmp); + PASS(); +} + +/* The ship gate: a link family that is switched off still resolves, but its + * resolved references are rows (below_bar_tier), not edges. */ +TEST(doc_mentions_ship_gate) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_gate_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + dm_write_resolver_fixture(tmp); + char db[512]; + snprintf(db, sizeof(db), "%s/gate.db", tmp); + const char *widgets = "src/Lib/Widgets.cs"; + char props[512]; + char reason[64]; + char syntax[64]; + int n = 0; + + /* the families and their names */ + ASSERT_STR_EQ(cbm_doclink_syntax_name(CBM_DOCLINK_CS_SEE), "see"); + ASSERT_STR_EQ(cbm_doclink_syntax_name(CBM_DOCLINK_CS_SEEALSO), "seealso"); + ASSERT_STR_EQ(cbm_doclink_syntax_name(CBM_DOCLINK_CS_EXCEPTION), "exception"); + ASSERT_STR_EQ(cbm_doclink_syntax_name(CBM_DOCLINK_CS_INHERITDOC), "inheritdoc"); + for (int s = CBM_DOCLINK_CS_SEE; s <= CBM_DOCLINK_CS_INHERITDOC; s++) { + const CBMDocLinkFamily *f = cbm_doclink_family(s); + ASSERT_NOT_NULL(f); + ASSERT_EQ(f->lang, CBM_LANG_CSHARP); + ASSERT_TRUE(cbm_doclink_syntax_ships(s)); /* every implemented family ships */ + } + ASSERT_FALSE(cbm_doclink_syntax_ships(CBM_DOCLINK_HREF)); /* a URL is never an edge */ + ASSERT_NULL(cbm_doclink_family(CBM_DOCLINK_NONE)); + ASSERT_NULL(cbm_doclink_family(CBM_DOCLINK_SYNTAX_COUNT)); + + cbm_doclink_test_set_ships(CBM_DOCLINK_CS_SEEALSO, false); + int rc = dm_index(tmp, db, NULL); + cbm_doclink_test_reset_ships(); + ASSERT_EQ(rc, 0); + /* the resolved seealso is a row with the gate's reason and no edge */ + dm_edge(db, "Widgets.Widget", "Helper.Helper.Once", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + dm_row(db, widgets, "Helper.Once(int)", reason, sizeof(reason), syntax, sizeof(syntax)); + ASSERT_STR_EQ(reason, "below_bar_tier"); + ASSERT_STR_EQ(syntax, "seealso"); + /* a seealso that does not resolve keeps its own reason */ + dm_row(db, widgets, "NoSuchThing", reason, sizeof(reason), syntax, sizeof(syntax)); + ASSERT_STR_EQ(reason, "missing"); + ASSERT_STR_EQ(syntax, "seealso"); + /* the other families are untouched */ + dm_edge(db, "Widgets.Widget", "Widgets.Gadget", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_edge(db, "Widgets.Widget", "Widgets.WidgetError", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_edge(db, "Widgets.Widget.Spin", "Widgets.Gadget.Spin", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* with the family shipping again, the same reference is an edge */ + dm_unlink_db(db); + ASSERT_EQ(dm_index(tmp, db, NULL), 0); + dm_edge(db, "Widgets.Widget", "Helper.Helper.Once", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_row(db, widgets, "Helper.Once(int)", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, ""); + + dm_unlink_db(db); + th_rmtree(tmp); + PASS(); +} + +/* A reference written in a file's own doc has the file's File node as its + * source. The resolver finds that node with the pipeline's one lookup; this + * holds it against the graph the pipeline published: the lookup must name the + * File node of exactly that file. (No language of this suite has file-level + * docs; the language legs test the references themselves.) */ +TEST(doc_mentions_file_node_lookup) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_fn_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + dm_write_resolver_fixture(tmp); + char db[512]; + snprintf(db, sizeof(db), "%s/res.db", tmp); + char *project = NULL; + ASSERT_EQ(dm_index(tmp, db, &project), 0); + ASSERT_NOT_NULL(project); + cbm_gbuf_t *gb = cbm_gbuf_new(project, tmp); + ASSERT_NOT_NULL(gb); + ASSERT_EQ(cbm_gbuf_load_from_db(gb, db, project), 0); + const char *paths[] = {"src/Lib/Widgets.cs", "src/Other/tests/Half.cs", "src/Lib/Lib.csproj"}; + for (size_t i = 0; i < sizeof(paths) / sizeof(paths[0]); i++) { + const cbm_gbuf_node_t *n = cbm_pipeline_file_node(gb, project, paths[i]); + ASSERT_NOT_NULL(n); + ASSERT_STR_EQ(n->label, "File"); + ASSERT_STR_EQ(n->file_path, paths[i]); + } + ASSERT_NULL(cbm_pipeline_file_node(gb, project, "src/Lib/NoSuchFile.cs")); + cbm_gbuf_free(gb); + free(project); + dm_unlink_db(db); + th_rmtree(tmp); + PASS(); +} + +/* ── publication: index_status, delete_project, older databases ──── */ + +/* The checks of doc_mentions_index_status_and_delete; the caller owns the + * environment and the directories. */ +static int dm_index_status_checks(const char *tmp, const char *repo, const char *cache_dir) { + char *project = cbm_project_name_from_path(repo); + ASSERT_NOT_NULL(project); + char db[1024]; + snprintf(db, sizeof(db), "%s/%s.db", cache_dir, project); + ASSERT_EQ(dm_index(repo, db, NULL), 0); + int edges = dm_count(db, "SELECT COUNT(*) FROM edges WHERE type = 'MENTIONS'"); + ASSERT_GT(edges, 0); + + /* doc_links: the edge count, every reason with its row count, the status */ + char *text = dm_index_status(project, false); + ASSERT_NOT_NULL(text); + const char *block = strstr(text, "doc_links:\n"); + ASSERT_NOT_NULL(block); + char want[64]; + snprintf(want, sizeof(want), "\n mentions: %d\n", edges); + ASSERT_NOT_NULL(strstr(block, want)); + ASSERT_NOT_NULL(strstr(block, "\n unresolved:\n")); + ASSERT_GT(dm_reason_lines(db, block), 4); + ASSERT_NOT_NULL(strstr(block, "\n test_only_target: 2\n")); + ASSERT_NOT_NULL(strstr(block, "\n unparseable: 1\n")); + ASSERT_NOT_NULL(strstr(block, "\n status: ok")); + ASSERT_NULL(strstr(block, "hint")); + ASSERT_NULL(strstr(block, "samples")); /* only under diagnostics=full */ + free(text); + text = dm_index_status(project, true); + ASSERT_NOT_NULL(text); + block = strstr(text, "doc_links:\n"); + ASSERT_NOT_NULL(block); + ASSERT_NOT_NULL(strstr(block, "samples")); + ASSERT_NOT_NULL(strstr(block, "TestOnlyThing")); + ASSERT_NOT_NULL(strstr(block, "src/Lib/Widgets.cs")); + free(text); + + /* A doc-link layer that fails does not fail the index, and is not + * published as "no references" either: the generation carries the error, + * index_status reports it with what to do, and doing it rebuilds. */ + dm_unlink_db(db); + cbm_doclinks_test_fail_build_once(); + ASSERT_EQ(dm_index(repo, db, NULL), 0); + ASSERT_EQ(dm_count(db, "SELECT COUNT(*) FROM edges WHERE type = 'MENTIONS'"), 0); + ASSERT_EQ(dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"), 1); + ASSERT_EQ(dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'"), 1); + text = dm_index_status(project, false); + ASSERT_NOT_NULL(text); + block = strstr(text, "doc_links:\n"); + ASSERT_NOT_NULL(block); + ASSERT_NOT_NULL(strstr(block, "\n mentions: 0\n")); + ASSERT_NOT_NULL(strstr(block, "\n status: error")); + ASSERT_NOT_NULL(strstr(block, "index_repository")); + ASSERT_NULL(strstr(block, "\n error: ")); + free(text); + ASSERT_EQ(dm_index(repo, db, NULL), 0); /* no file changed */ + ASSERT_EQ(cbm_pipeline_incremental_test_last_route(), CBM_INCREMENTAL_ROUTE_FORCED_FULL); + ASSERT_EQ(dm_count(db, "SELECT COUNT(*) FROM edges WHERE type = 'MENTIONS'"), edges); + ASSERT_EQ(dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'"), 0); + + /* the marker row is the status, never a reference: with rows beside it + * the counts and the samples are theirs alone */ + cbm_store_t *s = cbm_store_open_path(db); + ASSERT_NOT_NULL(s); + cbm_doc_link_row_t marker[2] = { + {.rel_path = "src/Lib/Widgets.cs", + .line = 9, + .syntax = "see", + .raw = "Gone", + .reason = "missing"}, + {.rel_path = "", + .line = 0, + .syntax = "", + .raw = "doc-link layer failed", + .reason = "error"}, + }; + ASSERT_EQ(cbm_store_doc_links_replace(s, project, marker, 2), CBM_STORE_OK); + cbm_store_close(s); + text = dm_index_status(project, true); + ASSERT_NOT_NULL(text); + block = strstr(text, "doc_links:\n"); + ASSERT_NOT_NULL(block); + ASSERT_NOT_NULL(strstr(block, "\n status: error")); + ASSERT_NOT_NULL(strstr(block, "hint")); + ASSERT_NOT_NULL(strstr(block, "\n missing: 1\n")); + ASSERT_NULL(strstr(block, "\n error: ")); + /* nor is it a sample */ + ASSERT_NOT_NULL(strstr(block, "\n samples: 1 (cols: rel_path line syntax raw reason)\n" + " src/Lib/Widgets.cs 9 see Gone missing\n")); + free(text); + /* ... and the re-run the hint asks for rebuilds, although no file changed */ + int rows_before = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"); + ASSERT_EQ(rows_before, 2); + ASSERT_EQ(dm_index(repo, db, NULL), 0); + ASSERT_EQ(cbm_pipeline_incremental_test_last_route(), CBM_INCREMENTAL_ROUTE_FORCED_FULL); + ASSERT_GT(dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"), rows_before); + ASSERT_EQ(dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'"), 0); + + /* an index from before the layer has no table: index_status reports an + * error and what to do, not zero unresolved references */ + sqlite3 *h = NULL; + ASSERT_EQ(sqlite3_open_v2(db, &h, SQLITE_OPEN_READWRITE, NULL), SQLITE_OK); + ASSERT_EQ(sqlite3_exec(h, "DROP TABLE doc_link_unresolved", NULL, NULL, NULL), SQLITE_OK); + sqlite3_close(h); + text = dm_index_status(project, false); + ASSERT_NOT_NULL(text); + block = strstr(text, "doc_links:\n"); + ASSERT_NOT_NULL(block); + ASSERT_NOT_NULL(strstr(block, "\n status: error")); + ASSERT_NOT_NULL(strstr(block, "predates")); + free(text); + /* ... and the next index run creates the table again */ + ASSERT_EQ(dm_index(repo, db, NULL), 0); + ASSERT_EQ(cbm_pipeline_incremental_test_last_route(), CBM_INCREMENTAL_ROUTE_FORCED_FULL); + ASSERT_GT(dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"), rows_before); + text = dm_index_status(project, false); + ASSERT_NOT_NULL(text); + block = strstr(text, "doc_links:\n"); + ASSERT_NOT_NULL(block); + ASSERT_NOT_NULL(strstr(block, "\n status: ok")); + free(text); + /* a healthy generation with unchanged inputs is current */ + ASSERT_EQ(dm_index(repo, db, NULL), 0); + ASSERT_EQ(cbm_pipeline_incremental_test_last_route(), CBM_INCREMENTAL_ROUTE_NOOP); + + /* store-level delete_project removes the rows */ + s = cbm_store_open_path(db); + ASSERT_NOT_NULL(s); + cbm_doc_link_row_t *rows = NULL; + int count = 0; + bool present = false; + ASSERT_EQ(cbm_store_doc_links_get(s, project, &rows, &count, &present), CBM_STORE_OK); + ASSERT_TRUE(present); + ASSERT_GT(count, rows_before); + cbm_store_free_doc_links(rows, count); + ASSERT_EQ(cbm_store_delete_project(s, project), CBM_STORE_OK); + ASSERT_EQ(cbm_store_doc_links_get(s, project, &rows, &count, &present), CBM_STORE_OK); + ASSERT_EQ(count, 0); + cbm_store_free_doc_links(rows, count); + cbm_store_close(s); + + /* a database without the table reads as "no data" at the store level, and + * delete_project still works */ + char old_db[600]; + snprintf(old_db, sizeof(old_db), "%s/old.db", tmp); + cbm_store_t *o = cbm_store_open_path(old_db); + ASSERT_NOT_NULL(o); + ASSERT_EQ(cbm_store_doc_links_get(o, "x", &rows, &count, &present), CBM_STORE_OK); + ASSERT_FALSE(present); + ASSERT_EQ(count, 0); + cbm_doc_link_reason_count_t *reasons = NULL; + int nreasons = 0; + cbm_doc_link_row_t *samples = NULL; + int nsamples = 0; + ASSERT_EQ( + cbm_store_doc_links_summary(o, "x", &reasons, &nreasons, &samples, &nsamples, 5, &present), + CBM_STORE_OK); + ASSERT_FALSE(present); + ASSERT_EQ(cbm_store_delete_project(o, "x"), CBM_STORE_OK); + cbm_store_close(o); + + dm_unlink_db(db); + dm_unlink_db(old_db); + free(project); + return 0; +} + +/* Each preview test starts from a real private index. Restore process state and + * remove the fixture even when a deliberately RED check returns failure. */ +static int dm_preview_fixture(int (*check)(const char *db, const char *project)) { + char tmp[256] = "/tmp/cbm_dm_preview_XXXXXX"; + if (!cbm_mkdtemp(tmp)) { + return 1; + } + char repo[400], cache_dir[400], db[1024]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + snprintf(cache_dir, sizeof(cache_dir), "%s/cache", tmp); + dm_write_resolver_fixture(repo); + cbm_mkdir_p(cache_dir, 0700); + char *project = cbm_project_name_from_path(repo); + const char *saved = getenv("CBM_CACHE_DIR"); + char *saved_copy = saved ? strdup(saved) : NULL; + if (!project || (saved && !saved_copy)) { + free(project); + free(saved_copy); + th_rmtree(tmp); + return 1; + } + snprintf(db, sizeof(db), "%s/%s.db", cache_dir, project); + cbm_setenv("CBM_CACHE_DIR", cache_dir, 1); + int rc = dm_index(repo, db, NULL); + if (rc == 0) { + rc = check(db, project); + } + if (saved_copy) { + cbm_setenv("CBM_CACHE_DIR", saved_copy, 1); + } else { + cbm_unsetenv("CBM_CACHE_DIR"); + } + free(saved_copy); + dm_unlink_db(db); + free(project); + th_rmtree(tmp); + return rc; +} + +/* Return the actual MCP envelope, including structuredContent for JSON, so a + * test exercises both transport boundaries instead of a formatter in isolation. */ +static yyjson_doc *dm_preview_status(const char *project, bool json) { + cbm_mcp_server_t *srv = cbm_mcp_server_new(NULL); + if (!srv) { + return NULL; + } + char args[1200]; + snprintf(args, sizeof(args), "{\"project\":\"%s\",\"diagnostics\":\"full\"%s}", project, + json ? ",\"format\":\"json\"" : ""); + char *response = cbm_mcp_handle_tool(srv, "index_status", args); + yyjson_doc *out = response ? yyjson_read(response, strlen(response), 0) : NULL; + free(response); + cbm_mcp_server_free(srv); + return out; +} + +static const char *dm_preview_text(yyjson_doc *envelope) { + yyjson_val *root = envelope ? yyjson_doc_get_root(envelope) : NULL; + yyjson_val *content = yyjson_obj_get(root, "content"); + return yyjson_get_str(yyjson_obj_get(yyjson_arr_get(content, 0), "text")); +} + +static yyjson_val *dm_preview_report(yyjson_doc *envelope) { + yyjson_val *root = envelope ? yyjson_doc_get_root(envelope) : NULL; + return yyjson_obj_get(yyjson_obj_get(root, "structuredContent"), "doc_links"); +} + +static bool dm_preview_status_is(yyjson_doc *envelope, const char *status) { + const char *actual = yyjson_get_str(yyjson_obj_get(dm_preview_report(envelope), "status")); + return actual && strcmp(actual, status) == 0; +} + +static bool dm_preview_json_consistent(yyjson_doc *envelope) { + const char *text = dm_preview_text(envelope); + yyjson_doc *payload = text ? yyjson_read(text, strlen(text), 0) : NULL; + yyjson_val *root = envelope ? yyjson_doc_get_root(envelope) : NULL; + yyjson_val *structured = yyjson_obj_get(root, "structuredContent"); + char *a = structured ? yyjson_val_write(structured, 0, NULL) : NULL; + char *b = payload ? yyjson_val_write(yyjson_doc_get_root(payload), 0, NULL) : NULL; + bool correct = a && b && strcmp(a, b) == 0; + free(a); + free(b); + yyjson_doc_free(payload); + return correct; +} + +static bool dm_preview_compact_status(yyjson_doc *env, const char *status) { + const char *text = dm_preview_text(env); + const char *report = text ? strstr(text, "doc_links:\n") : NULL; + char expected[50]; + snprintf(expected, sizeof(expected), "\n status: %s\n", status); + return report && strstr(report, expected); +} + +static bool dm_preview_compact_metadata(yyjson_doc *env, size_t sample, const char *field, + size_t original, size_t included, bool truncated, + bool escaped) { + const char *text = dm_preview_text(env); + const char *table = text ? strstr(text, "\n samples_preview:") : NULL; + const char *columns = table ? strstr(table, "(cols: sample_index field original_bytes " + "included_source_bytes truncated escaped)\n") + : NULL; + const char *end = table ? strchr(table + 1, '\n') : NULL; + char row[220]; + snprintf(row, sizeof(row), "\n %zu %s %zu %zu %s %s\n", sample, field, original, included, + truncated ? "true" : "false", escaped ? "true" : "false"); + return columns && end && columns < end && strstr(end, row); +} + +static yyjson_val *dm_preview_metadata(yyjson_val *report, size_t sample, const char *field) { + yyjson_val *entries = yyjson_obj_get(report, "samples_preview"); + for (size_t i = 0; i < yyjson_arr_size(entries); i++) { + yyjson_val *entry = yyjson_arr_get(entries, i); + const char *name = yyjson_get_str(yyjson_obj_get(entry, "field")); + yyjson_val *index = yyjson_obj_get(entry, "sample_index"); + if (name && yyjson_is_uint(index) && yyjson_get_uint(index) == sample && + strcmp(name, field) == 0) { + return entry; + } + } + return NULL; +} + +static bool dm_preview_metadata_matches(yyjson_val *entry, size_t original, size_t included, + bool truncated, bool escaped) { + yyjson_val *orig = yyjson_obj_get(entry, "original_bytes"); + yyjson_val *shown = yyjson_obj_get(entry, "included_source_bytes"); + yyjson_val *cut = yyjson_obj_get(entry, "truncated"); + yyjson_val *quoted = yyjson_obj_get(entry, "escaped"); + return yyjson_is_uint(orig) && yyjson_get_uint(orig) == original && yyjson_is_uint(shown) && + yyjson_get_uint(shown) == included && yyjson_is_bool(cut) && + yyjson_get_bool(cut) == truncated && yyjson_is_bool(quoted) && + yyjson_get_bool(quoted) == escaped; +} + +static bool dm_preview_full_row(cbm_store_t *store, const char *project, + const cbm_doc_link_row_t *expected) { + cbm_doc_link_row_t *rows = NULL; + int count = 0; + bool present = false; + bool correct = + cbm_store_doc_links_get(store, project, &rows, &count, &present) == CBM_STORE_OK && + present && count == 1; + if (correct) { + correct = rows[0].line == expected->line && + strcmp(rows[0].rel_path, expected->rel_path) == 0 && + strcmp(rows[0].syntax, expected->syntax) == 0 && + strcmp(rows[0].raw, expected->raw) == 0 && + strcmp(rows[0].reason, expected->reason) == 0; + } + cbm_store_free_doc_links(rows, count); + return correct; +} + +static int dm_preview_bounds_checks(const char *db, const char *project) { + cbm_store_t *store = cbm_store_open_path(db); + if (!store) { + return 1; + } + const char *fields[] = {"rel_path", "syntax", "raw", "reason"}; + const size_t limits[] = {1024, 128, 1024, 128}; + const cbm_doc_link_row_t ordinary = {.rel_path = "src/Normal.cs", + .line = 7, + .syntax = "see", + .raw = "Gone", + .reason = "missing"}; + bool shape = cbm_store_doc_links_replace(store, project, &ordinary, 1) == CBM_STORE_OK; + for (int format = 0; format < 2; format++) { + yyjson_doc *env = dm_preview_status(project, format != 0); + const char *text = dm_preview_text(env); + if (format) { + yyjson_val *report = dm_preview_report(env); + yyjson_val *samples = yyjson_obj_get(report, "samples"); + shape = dm_preview_status_is(env, "ok") && dm_preview_json_consistent(env) && + yyjson_arr_size(samples) == 1 && + yyjson_obj_size(yyjson_arr_get(samples, 0)) == 5 && + !yyjson_obj_get(report, "samples_preview") && + !yyjson_obj_get(report, "samples_preview_note") && shape; + } else { + shape = text && + strstr(text, "samples: 1 (cols: rel_path line syntax raw reason)\n" + " src/Normal.cs 7 see Gone missing\n") && + !strstr(text, "samples_preview") && shape; + } + yyjson_doc_free(env); + } + bool bounded = true, copies = true, preserved = true; + for (int size = 0; size < 3; size++) { + size_t length = size == 1 ? 65536 : 8192; + bool escaped = size == 2; + char *values[4] = {0}; + bool allocated = true; + for (int field = 0; field < 4; field++) { + values[field] = malloc(length + 1); + if (!values[field]) { + allocated = false; + break; + } + memset(values[field], escaped ? '\\' : 'a' + field, length); + values[field][length] = '\0'; + } + if (!allocated) { + for (int field = 0; field < 4; field++) { + free(values[field]); + } + cbm_store_close(store); + return 1; + } + cbm_doc_link_row_t row = {.rel_path = values[0], + .line = 42, + .syntax = values[1], + .raw = values[2], + .reason = values[3]}; + preserved = + cbm_store_doc_links_replace(store, project, &row, 1) == CBM_STORE_OK && preserved; + for (int format = 0; format < 2; format++) { + cbm_store_doc_links_test_sample_stats_reset(); + yyjson_doc *env = dm_preview_status(project, format != 0); + cbm_doc_links_sample_test_stats_t stats = {0}; + cbm_store_doc_links_test_sample_stats(&stats); + copies = stats.field_copies == 4 && stats.copied_bytes <= 2320 && + stats.requested_bytes == stats.copied_bytes && + stats.max_request_bytes <= 1028 && copies; + const char *text = dm_preview_text(env); + if (format) { + yyjson_val *report = dm_preview_report(env); + yyjson_val *samples = yyjson_obj_get(report, "samples"); + yyjson_val *sample = yyjson_arr_get(samples, 0); + yyjson_val *metadata = yyjson_obj_get(report, "samples_preview"); + const char *note = yyjson_get_str(yyjson_obj_get(report, "samples_preview_note")); + bounded = dm_preview_status_is(env, "ok") && dm_preview_json_consistent(env) && + yyjson_arr_size(samples) == 1 && yyjson_obj_size(sample) == 5 && + yyjson_arr_size(metadata) == 4 && note && + strstr(note, "doc_link_unresolved") && bounded; + for (int field = 0; field < 4; field++) { + yyjson_val *value = yyjson_obj_get(sample, fields[field]); + const char *actual = yyjson_get_str(value); + bounded = actual && yyjson_get_len(value) == limits[field] && + memcmp(actual, values[field], limits[field]) == 0 && + dm_preview_metadata_matches( + dm_preview_metadata(report, 0, fields[field]), length, + limits[field] / (escaped ? 2 : 1), true, escaped) && + bounded; + } + char *encoded = samples ? yyjson_val_write(samples, 0, NULL) : NULL; + bounded = encoded && strlen(encoded) <= 5000 && bounded; + free(encoded); + } else { + const char *table = text ? strstr(text, "\n samples:") : NULL; + const char *metadata = table ? strstr(table, "\n samples_preview:") : NULL; + size_t bytes = table ? (metadata ? (size_t)(metadata - table) : strlen(table)) : 0; + bounded = table && metadata && bytes <= 5000 && + dm_preview_compact_status(env, "ok") && + strstr(metadata, "\n samples_preview_note:") && + strstr(table, "(cols: rel_path line syntax raw reason)") && bounded; + for (int field = 0; field < 4; field++) { + bounded = dm_preview_compact_metadata(env, 0, fields[field], length, + limits[field] / (escaped ? 2 : 1), true, + escaped) && + bounded; + } + } + fprintf(stderr, + "doc preview bounds source=%llu escaped=%d json=%d copies=%llu copied=%llu " + "requested=%llu " + "max_request=%llu bounded=%d copy_bound=%d\n", + (unsigned long long)length, escaped, format, + (unsigned long long)stats.field_copies, (unsigned long long)stats.copied_bytes, + (unsigned long long)stats.requested_bytes, + (unsigned long long)stats.max_request_bytes, bounded, copies); + yyjson_doc_free(env); + } + preserved = dm_preview_full_row(store, project, &row) && preserved; + for (int field = 0; field < 4; field++) { + free(values[field]); + } + } + cbm_store_close(store); + fprintf(stderr, "doc preview shape=%d full_rows_preserved=%d\n", shape, preserved); + return !(shape && bounded && copies && preserved); +} + +TEST(doc_mentions_index_status_preview_bounds) { + int result = dm_preview_fixture(dm_preview_bounds_checks); + ASSERT_EQ(result, 0); + PASS(); +} + +/* Check binary storage through SQLite, without asking the legacy C-string + * getter to promise NUL support. All non-NUL cases also exercise that getter. */ +static bool dm_preview_raw_bytes(const char *db, const char *project, const void *bytes, + size_t length, bool write) { + sqlite3 *sql = NULL; + sqlite3_stmt *stmt = NULL; + bool correct = sqlite3_open_v2(db, &sql, write ? SQLITE_OPEN_READWRITE : SQLITE_OPEN_READONLY, + NULL) == SQLITE_OK; + const char *query = write + ? "UPDATE doc_link_unresolved SET raw=?2 WHERE project=?1;" + : "SELECT CAST(raw AS BLOB) FROM doc_link_unresolved WHERE project=?1;"; + correct = correct && sqlite3_prepare_v2(sql, query, -1, &stmt, NULL) == SQLITE_OK && + sqlite3_bind_text(stmt, 1, project, -1, SQLITE_TRANSIENT) == SQLITE_OK; + if (correct && write) { + correct = sqlite3_bind_text(stmt, 2, bytes, (int)length, SQLITE_TRANSIENT) == SQLITE_OK && + sqlite3_step(stmt) == SQLITE_DONE && sqlite3_changes(sql) == 1; + } else if (correct) { + correct = sqlite3_step(stmt) == SQLITE_ROW && + sqlite3_column_bytes(stmt, 0) == (int)length && + (length == 0 || memcmp(sqlite3_column_blob(stmt, 0), bytes, length) == 0) && + sqlite3_step(stmt) == SQLITE_DONE; + } + sqlite3_finalize(stmt); + sqlite3_close(sql); + return correct; +} + +/* Expected preview strings are independent, literal test data. This helper + * only applies the compact table's quote/backslash framing to an expectation. */ +static bool dm_preview_compact_raw(yyjson_doc *env, const char *expected) { + const char *text = dm_preview_text(env); + const char *table = + text ? strstr(text, "samples: 1 (cols: rel_path line syntax raw reason)\n") : NULL; + const char *row = table ? strchr(table, '\n') + 1 : NULL; + const char *end = row ? strchr(row, '\n') : NULL; + if (!row || !end || (size_t)(end - row) > 2200) { + return false; + } + char unquoted[2200], quoted[2200]; + snprintf(unquoted, sizeof(unquoted), " src/Text.cs 9 see %s missing", expected); + size_t used = (size_t)snprintf(quoted, sizeof(quoted), " src/Text.cs 9 see \""); + for (const unsigned char *p = (const unsigned char *)expected; *p; p++) { + if (*p == '\\' || *p == '"') { + quoted[used++] = '\\'; + } + quoted[used++] = (char)*p; + } + snprintf(quoted + used, sizeof(quoted) - used, "\" missing"); + size_t n = (size_t)(end - row); + return (strlen(unquoted) == n && memcmp(row, unquoted, n) == 0) || + (strlen(quoted) == n && memcmp(row, quoted, n) == 0); +} + +static bool dm_preview_text_case(cbm_store_t *store, const char *db, const char *project, + const char *name, const char *source, size_t length, + const char *expected, size_t included, bool escaped) { + cbm_doc_link_row_t row = { + .rel_path = "src/Text.cs", .line = 9, .syntax = "see", .raw = source, .reason = "missing"}; + bool correct = cbm_store_doc_links_replace(store, project, &row, 1) == CBM_STORE_OK && + dm_preview_raw_bytes(db, project, source, length, true); + bool truncated = included < length; + for (int format = 0; format < 2; format++) { + yyjson_doc *env = dm_preview_status(project, format != 0); + bool rendered; + if (format) { + yyjson_val *report = dm_preview_report(env); + yyjson_val *samples = yyjson_obj_get(report, "samples"); + yyjson_val *sample = yyjson_arr_get(samples, 0); + yyjson_val *raw = yyjson_obj_get(sample, "raw"); + const char *actual = yyjson_get_str(raw); + yyjson_val *metadata = yyjson_obj_get(report, "samples_preview"); + rendered = dm_preview_status_is(env, "ok") && dm_preview_json_consistent(env) && + yyjson_arr_size(samples) == 1 && yyjson_obj_size(sample) == 5 && actual && + yyjson_get_len(raw) == strlen(expected) && strcmp(actual, expected) == 0; + if (truncated || escaped) { + const char *note = yyjson_get_str(yyjson_obj_get(report, "samples_preview_note")); + rendered = rendered && yyjson_arr_size(metadata) == 1 && note && + strstr(note, "doc_link_unresolved") && + dm_preview_metadata_matches(dm_preview_metadata(report, 0, "raw"), + length, included, truncated, escaped); + } else { + rendered = rendered && !metadata && !yyjson_obj_get(report, "samples_preview_note"); + } + } else { + rendered = + dm_preview_compact_raw(env, expected) && dm_preview_compact_status(env, "ok"); + const char *text = dm_preview_text(env); + if (truncated || escaped) { + rendered = dm_preview_compact_metadata(env, 0, "raw", length, included, truncated, + escaped) && + text && strstr(text, "\n samples_preview_note:") && rendered; + } else { + rendered = text && !strstr(text, "samples_preview") && rendered; + } + } + correct = rendered && correct; + fprintf(stderr, "doc preview text case=%s json=%d rendered=%d\n", name, format, rendered); + yyjson_doc_free(env); + } + correct = dm_preview_raw_bytes(db, project, source, length, false) && correct; + if (!memchr(source, 0, length)) { + correct = dm_preview_full_row(store, project, &row) && correct; + } + return correct; +} + +static int dm_preview_text_checks(const char *db, const char *project) { + cbm_store_t *store = cbm_store_open_path(db); + if (!store) { + return 1; + } + const struct { + const char *name; + const char *source; + size_t length; + const char *expected; + bool escaped; + } cases[] = { + {"ordinary_utf8", "A\xC3\xA9\xE2\x82\xAC\xF0\x9F\x99\x82Z", 11, + "A\xC3\xA9\xE2\x82\xAC\xF0\x9F\x99\x82Z", false}, + {"c0_del", "A\x01\t\n\r\x1F\x7FZ", 8, + "A\\u{0001}\\u{0009}\\u{000A}\\u{000D}\\u{001F}\\u{007F}Z", true}, + {"nul", "A\0B", 3, "A\\u{0000}B", true}, + {"literal_escapes", "\\u{000A}\\xFF", 12, "\\\\u{000A}\\\\xFF", true}, + {"reserved_bytes", "@bytes:AA", 9, "\\u{0040}bytes:AA", true}, + {"reserved_utf8", "@utf8:AA", 8, "\\u{0040}utf8:AA", true}, + {"middle_prefix", "A@utf8:AA", 9, "A@utf8:AA", false}, + {"malformed", "\xC0\xAF\xED\xA0\x80\xF4\x90\x80\x80\x80\xC2", 11, + "\\xC0\\xAF\\xED\\xA0\\x80\\xF4\\x90\\x80\\x80\\x80\\xC2", true}, + {"format_controls", + "\xE2\x80\x8B\xE2\x80\x8F\xE2\x80\xAA\xE2\x80\xAE" + "\xE2\x81\xA6\xE2\x81\xA9\xF3\xA0\x80\x80\xF3\xA0\x81\xBF", + 26, "\\u{200B}\\u{200F}\\u{202A}\\u{202E}\\u{2066}\\u{2069}\\u{E0000}\\u{E007F}", true}, + {"adjacent_unicode", + "\xE2\x80\x8A\xE2\x80\x90\xE2\x80\xA9\xE2\x80\xAF" + "\xE2\x81\xA5\xE2\x81\xAA\xF3\x9F\xBF\xBF\xF3\xA0\x82\x80", + 26, + "\xE2\x80\x8A\xE2\x80\x90\xE2\x80\xA9\xE2\x80\xAF" + "\xE2\x81\xA5\xE2\x81\xAA\xF3\x9F\xBF\xBF\xF3\xA0\x82\x80", + false}, + }; + bool correct = true; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + correct = dm_preview_text_case(store, db, project, cases[i].name, cases[i].source, + cases[i].length, cases[i].expected, cases[i].length, + cases[i].escaped) && + correct; + } + /* One expected token must either fit entirely or contribute no source + * bytes. The suffix guarantees truncation even for the exactly-fit case. */ + const struct { + const char *name, *source, *expected; + size_t source_bytes, display_bytes; + bool escape; + } tokens[] = { + {"utf8_2", "\xC3\xA9", "\xC3\xA9", 2, 2, false}, + {"utf8_3", "\xE2\x82\xAC", "\xE2\x82\xAC", 3, 3, false}, + {"utf8_4", "\xF0\x9F\x99\x82", "\xF0\x9F\x99\x82", 4, 4, false}, + {"newline", "\n", "\\u{000A}", 1, 8, true}, + {"backslash", "\\", "\\\\", 1, 2, true}, + {"invalid", "\xFF", "\\xFF", 1, 4, true}, + {"tag", "\xF3\xA0\x80\x81", "\\u{E0001}", 4, 9, true}, + }; + for (size_t i = 0; i < sizeof(tokens) / sizeof(tokens[0]); i++) { + for (int fit = 0; fit < 2; fit++) { + char source[1100], expected[1100], name[60]; + size_t prefix = fit ? 1024 - tokens[i].display_bytes : 1023; + memset(source, 'p', prefix); + memcpy(source + prefix, tokens[i].source, tokens[i].source_bytes); + memcpy(source + prefix + tokens[i].source_bytes, "TAIL", 5); + memset(expected, 'p', prefix); + size_t shown = prefix; + if (fit) { + memcpy(expected + shown, tokens[i].expected, tokens[i].display_bytes); + shown += tokens[i].display_bytes; + } + expected[shown] = '\0'; + snprintf(name, sizeof(name), "%s_%s", tokens[i].name, fit ? "fits" : "stops"); + correct = dm_preview_text_case(store, db, project, name, source, + prefix + tokens[i].source_bytes + 4, expected, + prefix + (fit ? tokens[i].source_bytes : 0), + fit && tokens[i].escape) && + correct; + } + } + cbm_store_close(store); + return !correct; +} + +TEST(doc_mentions_index_status_preview_text) { + int result = dm_preview_fixture(dm_preview_text_checks); + ASSERT_EQ(result, 0); + PASS(); +} + +static int dm_preview_order_checks(const char *db, const char *project) { + cbm_store_t *store = cbm_store_open_path(db); + if (!store) { + return 1; + } + cbm_doc_link_row_t rows[52] = {0}; + char paths[51][40], raw[1026]; + memset(raw, 'r', sizeof(raw) - 1); + raw[sizeof(raw) - 1] = '\0'; + /* The marker sorts first and consumes one of the original SQL LIMIT 50. + * Insert references backwards, so a storage-order accident cannot pass. */ + rows[0] = + (cbm_doc_link_row_t){.rel_path = "", .line = 0, .syntax = "", .raw = "", .reason = "error"}; + for (int i = 0; i < 51; i++) { + snprintf(paths[i], sizeof(paths[i]), "src/Case%03d.cs", 50 - i); + rows[i + 1] = (cbm_doc_link_row_t){ + .rel_path = paths[i], .line = 5, .syntax = "see", .raw = raw, .reason = "missing"}; + } + bool correct = cbm_store_doc_links_replace(store, project, rows, 52) == CBM_STORE_OK; + for (int format = 0; format < 2; format++) { + cbm_store_doc_links_test_sample_stats_reset(); + yyjson_doc *env = dm_preview_status(project, format != 0); + cbm_doc_links_sample_test_stats_t stats = {0}; + cbm_store_doc_links_test_sample_stats(&stats); + bool ordered = stats.field_copies >= 196 && stats.field_copies <= 200 && + stats.copied_bytes <= 116000 && + stats.requested_bytes == stats.copied_bytes && + stats.max_request_bytes <= 1028; + if (format) { + yyjson_val *report = dm_preview_report(env); + yyjson_val *samples = yyjson_obj_get(report, "samples"); + ordered = dm_preview_status_is(env, "error") && dm_preview_json_consistent(env) && + yyjson_arr_size(samples) == 49 && + yyjson_arr_size(yyjson_obj_get(report, "samples_preview")) == 49 && ordered; + for (int i = 0; i < 49; i++) { + char path[40]; + snprintf(path, sizeof(path), "src/Case%03d.cs", i); + yyjson_val *sample = yyjson_arr_get(samples, (size_t)i); + const char *actual = yyjson_get_str(yyjson_obj_get(sample, "rel_path")); + ordered = actual && strcmp(actual, path) == 0 && yyjson_obj_size(sample) == 5 && + yyjson_get_len(yyjson_obj_get(sample, "raw")) == 1024 && + dm_preview_metadata_matches(dm_preview_metadata(report, (size_t)i, "raw"), + 1025, 1024, true, false) && + ordered; + } + } else { + const char *text = dm_preview_text(env); + const char *cursor = + text ? strstr(text, "samples: 49 (cols: rel_path line syntax raw reason)\n") + : NULL; + ordered = cursor && ordered; + for (int i = 0; i < 49; i++) { + char prefix[60]; + snprintf(prefix, sizeof(prefix), "\n src/Case%03d.cs 5 see ", i); + const char *next = cursor ? strstr(cursor, prefix) : NULL; + ordered = next && ordered; + cursor = next ? next + strlen(prefix) : NULL; + } + ordered = cursor && !strstr(cursor, "src/Case049.cs") && + !strstr(cursor, "src/Case050.cs") && ordered; + } + fprintf(stderr, "doc preview order json=%d copies=%llu copied=%llu ordered=%d\n", format, + (unsigned long long)stats.field_copies, (unsigned long long)stats.copied_bytes, + ordered); + correct = ordered && correct; + yyjson_doc_free(env); + } + cbm_doc_link_row_t *full = NULL; + int count = 0; + bool present = false; + correct = cbm_store_doc_links_get(store, project, &full, &count, &present) == CBM_STORE_OK && + present && count == 52 && correct; + for (int i = 0; i < count; i++) { + if (full[i].rel_path && full[i].rel_path[0]) { + correct = strcmp(full[i].raw, raw) == 0 && correct; + } + } + cbm_store_free_doc_links(full, count); + /* Sort the complete source before projection. These two rows have the same + * displayed raw prefix, but the longer original sorts first (aa before z). */ + char first[1030], second[1029]; + memset(first, 't', 1027); + memcpy(first + 1027, "aa", 3); + memset(second, 't', 1027); + memcpy(second + 1027, "z", 2); + cbm_doc_link_row_t ties[] = { + {.rel_path = "src/Tie.cs", .line = 2, .syntax = "see", .raw = second, .reason = "missing"}, + {.rel_path = "src/Tie.cs", .line = 2, .syntax = "see", .raw = first, .reason = "missing"}, + {.rel_path = "src/Tie.cs", .line = 1, .syntax = "see", .raw = second, .reason = "missing"}, + {.rel_path = "src/Z.cs", .line = 3, .syntax = "see", .raw = first, .reason = "ambiguous"}, + }; + correct = cbm_store_doc_links_replace(store, project, ties, 4) == CBM_STORE_OK && correct; + yyjson_doc *env = dm_preview_status(project, true); + yyjson_val *report = dm_preview_report(env); + yyjson_val *ordered = yyjson_obj_get(report, "samples"); + const size_t original[] = {1029, 1028, 1029, 1028}; + const int lines[] = {3, 1, 2, 2}; + bool ties_correct = dm_preview_status_is(env, "ok") && dm_preview_json_consistent(env) && + yyjson_arr_size(ordered) == 4; + for (size_t i = 0; i < 4; i++) { + yyjson_val *sample = yyjson_arr_get(ordered, i); + const char *path = yyjson_get_str(yyjson_obj_get(sample, "rel_path")); + ties_correct = path && strcmp(path, i ? "src/Tie.cs" : "src/Z.cs") == 0 && + yyjson_get_int(yyjson_obj_get(sample, "line")) == lines[i] && + dm_preview_metadata_matches(dm_preview_metadata(report, i, "raw"), + original[i], 1024, true, false) && + ties_correct; + } + fprintf(stderr, "doc preview full_source_order=%d\n", ties_correct); + correct = ties_correct && correct; + yyjson_doc_free(env); + cbm_store_close(store); + return !correct; +} + +TEST(doc_mentions_index_status_preview_order) { + int result = dm_preview_fixture(dm_preview_order_checks); + ASSERT_EQ(result, 0); + PASS(); +} + +static bool dm_preview_error_result(yyjson_doc *env, bool json) { + if (json) { + yyjson_val *report = dm_preview_report(env); + return dm_preview_status_is(env, "error") && dm_preview_json_consistent(env) && + yyjson_arr_size(yyjson_obj_get(report, "samples")) == 0 && + yyjson_arr_size(yyjson_obj_get(report, "samples_preview")) == 0 && + !yyjson_obj_get(report, "samples_preview_note"); + } + const char *text = dm_preview_text(env); + const char *report = text ? strstr(text, "doc_links:\n") : NULL; + return report && strstr(report, "\n status: error\n") && !strstr(report, "src/Failure") && + !strstr(report, "samples_preview"); +} + +static bool dm_preview_retry_result(yyjson_doc *env, bool json) { + if (json) { + yyjson_val *samples = yyjson_obj_get(dm_preview_report(env), "samples"); + if (!dm_preview_status_is(env, "ok") || !dm_preview_json_consistent(env) || + yyjson_arr_size(samples) != 2) { + return false; + } + for (size_t i = 0; i < 2; i++) { + yyjson_val *row = yyjson_arr_get(samples, i); + const char *raw = yyjson_get_str(yyjson_obj_get(row, "raw")); + const char *path = yyjson_get_str(yyjson_obj_get(row, "rel_path")); + if (yyjson_obj_size(row) != 5 || !raw || + strcmp(raw, i ? "Second" : "First\\u{000A}") != 0 || !path || + strcmp(path, i ? "src/FailureB.cs" : "src/FailureA.cs") != 0) { + return false; + } + } + yyjson_val *report = dm_preview_report(env); + return yyjson_arr_size(yyjson_obj_get(report, "samples_preview")) == 1 && + dm_preview_metadata_matches(dm_preview_metadata(report, 0, "raw"), 6, 6, false, + true) && + yyjson_get_str(yyjson_obj_get(report, "samples_preview_note")); + } + const char *text = dm_preview_text(env); + return text && dm_preview_compact_status(env, "ok") && + dm_preview_compact_metadata(env, 0, "raw", 6, 6, false, true) && + strstr(text, "\n samples_preview_note:") && + strstr(text, "samples: 2 (cols: rel_path line syntax raw reason)\n" + " src/FailureA.cs 3 see First\\u{000A} missing\n" + " src/FailureB.cs 4 see Second missing\n"); +} + +static int dm_preview_failure_checks(const char *db, const char *project) { + cbm_store_t *store = cbm_store_open_path(db); + if (!store) { + return 1; + } + cbm_doc_link_row_t rows[] = { + {.rel_path = "src/FailureA.cs", + .line = 3, + .syntax = "see", + .raw = "First\n", + .reason = "missing"}, + {.rel_path = "src/FailureB.cs", + .line = 4, + .syntax = "see", + .raw = "Second", + .reason = "missing"}, + }; + bool correct = cbm_store_doc_links_replace(store, project, rows, 2) == CBM_STORE_OK; + for (int format = 0; format < 2; format++) { + yyjson_doc *warm = dm_preview_status(project, format != 0); + correct = dm_preview_retry_result(warm, format != 0) && correct; + yyjson_doc_free(warm); + } + + /* Full getters and summary-only callers must not consume sample faults or + * record projected sample bytes. This also protects incremental callers. */ + cbm_store_doc_links_test_sample_stats_reset(); + cbm_store_doc_links_test_fail_sample_alloc_after(0); + cbm_mcp_doc_links_test_fail_sample_alloc_after(0); + cbm_doc_link_row_t *full = NULL, *samples = NULL; + cbm_doc_link_reason_count_t *reasons = NULL; + int count = 0, nsamples = 0, nreasons = 0; + bool present = false; + bool bypass = + cbm_store_doc_links_get(store, project, &full, &count, &present) == CBM_STORE_OK && + present && count == 2; + cbm_store_free_doc_links(full, count); + bypass = cbm_store_doc_links_summary(store, project, &reasons, &nreasons, &samples, &nsamples, + 0, &present) == CBM_STORE_OK && + present && nsamples == 0 && nreasons == 1 && bypass; + cbm_store_free_doc_links(samples, nsamples); + cbm_store_free_doc_link_reasons(reasons, nreasons); + char *summary = dm_index_status(project, false); + bypass = summary && !strstr(summary, "src/Failure") && bypass; + free(summary); + cbm_doc_links_sample_test_stats_t stats = {0}; + cbm_store_doc_links_test_sample_stats(&stats); + bypass = stats.field_copies == 0 && stats.copied_bytes == 0 && stats.requested_bytes == 0 && + !cbm_store_doc_links_test_sample_alloc_failed() && + !cbm_mcp_doc_links_test_sample_alloc_failed() && bypass; + cbm_store_doc_links_test_fail_sample_alloc_after(-1); + cbm_mcp_doc_links_test_fail_sample_alloc_after(-1); + fprintf(stderr, "doc preview sample_only=%d\n", bypass); + correct = bypass && correct; + + for (int stage = 0; stage < 2; stage++) { + for (int nth = 0; nth < 2; nth++) { + for (int format = 0; format < 2; format++) { + uint64_t before = cbm_mem_tracked_live_bytes(); + if (stage == 0) { + cbm_store_doc_links_test_fail_sample_alloc_after(nth ? 4 : 0); + } else { + cbm_mcp_doc_links_test_fail_sample_alloc_after(nth ? 4 : 0); + } + yyjson_doc *failed = dm_preview_status(project, format != 0); + bool consumed = stage == 0 ? cbm_store_doc_links_test_sample_alloc_failed() + : cbm_mcp_doc_links_test_sample_alloc_failed(); + bool error = dm_preview_error_result(failed, format != 0); + yyjson_doc_free(failed); + cbm_store_doc_links_test_fail_sample_alloc_after(-1); + cbm_mcp_doc_links_test_fail_sample_alloc_after(-1); + uint64_t after_failure = cbm_mem_tracked_live_bytes(); + yyjson_doc *retried = dm_preview_status(project, format != 0); + bool retry = dm_preview_retry_result(retried, format != 0); + yyjson_doc_free(retried); + uint64_t after_retry = cbm_mem_tracked_live_bytes(); + bool clean = before == after_failure && before == after_retry; + correct = consumed && error && retry && clean && correct; + fprintf(stderr, + "doc preview fault stage=%d nth=%d json=%d consumed=%d error=%d retry=%d " + "before=%llu failed=%llu retried=%llu clean=%d\n", + stage, nth ? 5 : 1, format, consumed, error, retry, + (unsigned long long)before, (unsigned long long)after_failure, + (unsigned long long)after_retry, clean); + } + } + } + cbm_store_close(store); + return !correct; +} + +TEST(doc_mentions_index_status_preview_failure) { + int result = dm_preview_fixture(dm_preview_failure_checks); + ASSERT_EQ(result, 0); + PASS(); +} + +TEST(doc_mentions_index_status_and_delete) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_st_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400]; + char cache_dir[400]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + snprintf(cache_dir, sizeof(cache_dir), "%s/cache", tmp); + dm_write_resolver_fixture(repo); + cbm_mkdir_p(cache_dir, 0700); + /* the tool reads the project's database from the cache directory: a + * private one for this test, restored whatever the checks say */ + const char *saved_cache = getenv("CBM_CACHE_DIR"); + char *saved_cache_copy = saved_cache ? strdup(saved_cache) : NULL; + cbm_setenv("CBM_CACHE_DIR", cache_dir, 1); + int rc = dm_index_status_checks(tmp, repo, cache_dir); + if (saved_cache_copy) { + cbm_setenv("CBM_CACHE_DIR", saved_cache_copy, 1); + free(saved_cache_copy); + } else { + cbm_unsetenv("CBM_CACHE_DIR"); + } + th_rmtree(tmp); + return rc; +} + +/* ── incremental == full ─────────────────────────────────────────── */ + +static const char DM_MAIN[] = + "using P;\n" + "using Q;\n" + "namespace N\n" + "{\n" + " /// See , , .\n" + " /// \n" + " /// \n" + " /// \n" + " public class Main { }\n" + "}\n"; +static const char DM_MAIN_EDITED[] = + "using P;\n" + "using Q;\n" + "namespace N\n" + "{\n" + " /// See , , .\n" + " /// \n" + " /// \n" + " /// Also .\n" + " public class Main { }\n" + "}\n"; +static const char DM_MAIN_USING[] = + "using P;\n" + "using Q;\n" + "using S;\n" + "namespace N\n" + "{\n" + " /// See , , .\n" + " /// \n" + " /// \n" + " /// Also .\n" + " public class Main { }\n" + "}\n"; +/* a file with a row and no edge into any file the steps change */ +static const char DM_SIDE[] = "using Q;\n" + "namespace N\n" + "{\n" + " /// \n" + " public class Side { }\n" + "}\n"; +static const char DM_P[] = "namespace P\n" + "{\n" + " public class Target { }\n" + " public class Dup { }\n" + " public class Keep { }\n" + "}\n"; +static const char DM_P_NO_DUP[] = "namespace P\n" + "{\n" + " public class Target { }\n" + " public class Keep { }\n" + "}\n"; +static const char DM_P_RENAMED[] = "namespace P\n" + "{\n" + " public class Target { }\n" + " public class Kept { }\n" + "}\n"; +static const char DM_P_NO_TARGET[] = "namespace P\n" + "{\n" + " public class Kept { }\n" + "}\n"; +static const char DM_Q[] = "namespace Q\n" + "{\n" + " public class Dup { }\n" + " public class Other { }\n" + "}\n"; +static const char DM_Q_TWIN[] = "namespace Q\n" + "{\n" + " public class Dup { }\n" + " public class Other { }\n" + " public class Target { }\n" + "}\n"; +/* the same declarations, every one on another line */ +static const char DM_Q_TWIN_MOVED[] = "// moved\n" + "\n" + "namespace Q\n" + "{\n" + " public class Dup { }\n" + "\n" + " public class Other { }\n" + " public class Target { }\n" + "}\n"; +static const char DM_SIG[] = "namespace Q\n" + "{\n" + " public class Sig\n" + " {\n" + " public void Go(int x) { }\n" + " }\n" + "}\n"; +static const char DM_SIG_BODY[] = "namespace Q\n" + "{\n" + " public class Sig\n" + " {\n" + " public void Go(int x) { x++; }\n" + " }\n" + "}\n"; +static const char DM_SIG_CHANGED[] = "namespace Q\n" + "{\n" + " public class Sig\n" + " {\n" + " public void Go(string x) { }\n" + " }\n" + "}\n"; +static const char DM_OV[] = "namespace Q\n" + "{\n" + " public class Ov\n" + " {\n" + " public void Run(int n) { }\n" + " public void Run(string s) { }\n" + " }\n" + "}\n"; +static const char DM_OV_ONE[] = "namespace Q\n" + "{\n" + " public class Ov\n" + " {\n" + " public void Run(int n) { }\n" + " }\n" + "}\n"; +/* `Tw` and `Tw` share one node; it belongs to the later declaration */ +static const char DM_TW[] = "namespace Q\n" + "{\n" + " public class Tw { }\n" + " public class Tw { }\n" + "}\n"; +static const char DM_TW_SWAPPED[] = "namespace Q\n" + "{\n" + " public class Tw { }\n" + " public class Tw { }\n" + "}\n"; +static const char DM_R[] = "namespace R\n" + "{\n" + " public class ViaProject { }\n" + "}\n"; +static const char DM_S[] = "namespace S\n" + "{\n" + " public class Extra { }\n" + "}\n"; +static const char DM_CSPROJ[] = "\n" + " \n" + " \n" + " \n" + "\n"; +static const char DM_CSPROJ_NO_USING[] = "\n" + "\n"; + +TEST(doc_mentions_incremental_equals_full) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_inc_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + th_write_file(TH_PATH(repo, "src/Main.cs"), DM_MAIN); + th_write_file(TH_PATH(repo, "src/Side.cs"), DM_SIDE); + th_write_file(TH_PATH(repo, "src/P.cs"), DM_P); + th_write_file(TH_PATH(repo, "src/Q.cs"), DM_Q); + th_write_file(TH_PATH(repo, "src/Sig.cs"), DM_SIG); + th_write_file(TH_PATH(repo, "src/Ov.cs"), DM_OV); + th_write_file(TH_PATH(repo, "src/Tw.cs"), DM_TW); + th_write_file(TH_PATH(repo, "src/R.cs"), DM_R); + th_write_file(TH_PATH(repo, "src/S.cs"), DM_S); + th_write_file(TH_PATH(repo, "src/App.csproj"), DM_CSPROJ); + char inc_db[512]; + char full_db[512]; + snprintf(inc_db, sizeof(inc_db), "%s/inc.db", tmp); + snprintf(full_db, sizeof(full_db), "%s/full.db", tmp); + ASSERT_EQ(dm_index(repo, inc_db, NULL), 0); + char props[512]; + char reason[64]; + int n = 0; + const char *main_cs = "src/Main.cs"; + const char *side_cs = "src/Side.cs"; + /* the preconditions the steps below move away from */ + dm_edge(inc_db, "Main.Main", "P.Target", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_edge(inc_db, "Main.Main", "R.ViaProject", props, sizeof(props), &n); + ASSERT_EQ(n, 1); /* through the project file's */ + dm_edge(inc_db, "Main.Main", "Sig.Sig.Go", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_edge(inc_db, "Main.Main", "Tw.Tw", props, sizeof(props), &n); + ASSERT_EQ(n, 1); /* Tw, the later declaration, owns the node */ + dm_row(inc_db, main_cs, "Dup", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "ambiguous"); /* P.Dup and Q.Dup through the usings */ + dm_row(inc_db, main_cs, "Extra", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); /* namespace S is not imported */ + dm_row(inc_db, main_cs, "Ov.Run", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "ambiguous"); /* an overload group */ + dm_row(inc_db, side_cs, "Tw", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "graph_gap"); /* the arity-0 twin has no node */ + + /* a doc-comment edit re-extracts the file alone */ + th_write_file(TH_PATH(repo, "src/Main.cs"), DM_MAIN_EDITED); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "doc edit", CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR), 0); + dm_edge(inc_db, "Main.Main", "Q.Other", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* a body edit in a TARGET file: its scope is unchanged, and the edge into + * it stands */ + th_write_file(TH_PATH(repo, "src/Sig.cs"), DM_SIG_BODY); + ASSERT_EQ( + dm_step(repo, inc_db, full_db, "target body edit", CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR), + 0); + dm_edge(inc_db, "Main.Main", "Sig.Sig.Go", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* the file's own using directives scope nothing but the file */ + th_write_file(TH_PATH(repo, "src/Main.cs"), DM_MAIN_USING); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "using added", CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR), + 0); + dm_edge(inc_db, "Main.Main", "S.Extra", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* a removed member (an overload group shrinks to one) re-resolves the + * files whose rows name it, although they have no edge into the changed + * file */ + th_write_file(TH_PATH(repo, "src/Ov.cs"), DM_OV_ONE); + ASSERT_EQ( + dm_step(repo, inc_db, full_db, "member removed", CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR), 0); + dm_edge(inc_db, "Main.Main", "Ov.Ov.Run", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* a removed TYPE is not repaired file by file (it may be another type's + * base): one of two ambiguous candidates goes, the other binds */ + th_write_file(TH_PATH(repo, "src/P.cs"), DM_P_NO_DUP); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "type removed", CBM_INCREMENTAL_ROUTE_FORCED_FULL), 0); + dm_edge(inc_db, "Main.Main", "Q.Dup", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* renaming a target */ + th_write_file(TH_PATH(repo, "src/P.cs"), DM_P_RENAMED); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "rename", CBM_INCREMENTAL_ROUTE_FORCED_FULL), 0); + dm_row(inc_db, main_cs, "Keep", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + + /* an ambiguous twin */ + th_write_file(TH_PATH(repo, "src/Q.cs"), DM_Q_TWIN); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "ambiguous twin", CBM_INCREMENTAL_ROUTE_FORCED_FULL), + 0); + dm_row(inc_db, main_cs, "Target", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "ambiguous"); + + /* deleting a target */ + th_write_file(TH_PATH(repo, "src/P.cs"), DM_P_NO_TARGET); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "delete target", CBM_INCREMENTAL_ROUTE_FORCED_FULL), + 0); + dm_edge(inc_db, "Main.Main", "Q.Target", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* moving every declaration of a target file to another line changes no + * scope: the stored scopes carry no line numbers */ + th_write_file(TH_PATH(repo, "src/Q.cs"), DM_Q_TWIN_MOVED); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "lines moved", CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR), + 0); + dm_edge(inc_db, "Main.Main", "Q.Target", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* a changed signature can re-route references in files with no edge into + * the changed one */ + th_write_file(TH_PATH(repo, "src/Sig.cs"), DM_SIG_CHANGED); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "signature", CBM_INCREMENTAL_ROUTE_FORCED_FULL), 0); + dm_row(inc_db, main_cs, "Sig.Go(int)", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + + /* declaration order decides which same-path declaration owns the node: + * swapping the twins moves it, for a file with no edge into this one too */ + th_write_file(TH_PATH(repo, "src/Tw.cs"), DM_TW_SWAPPED); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "twins swapped", CBM_INCREMENTAL_ROUTE_FORCED_FULL), + 0); + dm_edge(inc_db, "Side.Side", "Tw.Tw", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_row(inc_db, main_cs, "Tw{T}", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "graph_gap"); + + /* the project file sets the global usings of every file of the project */ + th_write_file(TH_PATH(repo, "src/App.csproj"), DM_CSPROJ_NO_USING); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "project file", CBM_INCREMENTAL_ROUTE_FORCED_FULL), 0); + dm_edge(inc_db, "Main.Main", "R.ViaProject", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + dm_row(inc_db, main_cs, "ViaProject", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + /* Q.Target, Q.Dup, Q.Other, S.Extra, Ov.Run */ + ASSERT_EQ(dm_mentions_from(inc_db, "Main.Main"), 5); + + /* a deleted file takes its namespace and types along */ + ASSERT_EQ(unlink(TH_PATH(repo, "src/Q.cs")), 0); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "delete file", CBM_INCREMENTAL_ROUTE_FORCED_FULL), 0); + ASSERT_EQ(dm_mentions_from(inc_db, "Main.Main"), 2); /* S.Extra, Ov.Run */ + + dm_unlink_db(inc_db); + dm_unlink_db(full_db); + th_rmtree(tmp); + PASS(); +} + +/* A project file is a file of the index like any other: the planner judges a + * change to one by its scope blob. An edit the blob does not hold repairs + * file by file; an edit of a , a new project file and a deleted one + * rebuild. Every step ends in what a full index of the same tree holds. */ +TEST(doc_mentions_incremental_project_files) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_ipf_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + th_write_file(TH_PATH(repo, "src/App/Main.cs"), + "namespace N\n" + "{\n" + " /// \n" + " public class Main { }\n" + "}\n"); + th_write_file(TH_PATH(repo, "src/R.cs"), "namespace R\n" + "{\n" + " public class ViaProject { }\n" + "}\n" + "namespace S\n" + "{\n" + " public class ViaProps { }\n" + "}\n"); + th_write_file(TH_PATH(repo, "src/App/App.csproj"), + "\n" + " \n" + " \n" + "\n"); + char inc_db[512]; + char full_db[512]; + snprintf(inc_db, sizeof(inc_db), "%s/inc.db", tmp); + snprintf(full_db, sizeof(full_db), "%s/full.db", tmp); + ASSERT_EQ(dm_index(repo, inc_db, NULL), 0); + char props[512]; + char reason[64]; + int n = 0; + dm_edge(inc_db, "Main.Main", "R.ViaProject", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_row(inc_db, "src/App/Main.cs", "ViaProps", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + + /* an edit the scope blob does not hold: a package version */ + th_write_file(TH_PATH(repo, "src/App/App.csproj"), + "\n" + " \n" + " \n" + "\n"); + ASSERT_EQ( + dm_step(repo, inc_db, full_db, "package version", CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR), 0); + dm_edge(inc_db, "Main.Main", "R.ViaProject", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* a project file is added: Directory.Build.props brings namespace S in */ + th_write_file(TH_PATH(repo, "Directory.Build.props"), + "\n" + " \n" + "\n"); + ASSERT_EQ( + dm_step(repo, inc_db, full_db, "project file added", CBM_INCREMENTAL_ROUTE_FORCED_FULL), 0); + dm_edge(inc_db, "Main.Main", "R.ViaProps", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* an edit of what the blob holds: the goes */ + th_write_file(TH_PATH(repo, "src/App/App.csproj"), + "\n" + " \n" + "\n"); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "using removed", CBM_INCREMENTAL_ROUTE_FORCED_FULL), + 0); + dm_edge(inc_db, "Main.Main", "R.ViaProject", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + + /* a project file is deleted */ + ASSERT_EQ(unlink(TH_PATH(repo, "Directory.Build.props")), 0); + ASSERT_EQ( + dm_step(repo, inc_db, full_db, "project file deleted", CBM_INCREMENTAL_ROUTE_FORCED_FULL), + 0); + dm_edge(inc_db, "Main.Main", "R.ViaProps", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + ASSERT_EQ(dm_mentions_from(inc_db, "Main.Main"), 0); + + dm_unlink_db(inc_db); + dm_unlink_db(full_db); + th_rmtree(tmp); + PASS(); +} + +/* ── which scope changes are repairable file by file ─────────────── */ + +/* The scope delta between two versions of one C# file (NULL: the file does + * not exist); the removed names joined by ','. */ +static int dm_delta(const char *before, const char *after, char *names, size_t cap) { + return dm_scope_delta(CBM_LANG_CSHARP, "S.cs", before, after, names, cap); +} + +#define DM_DELTA_HEAD "using Acme.Local;\n" +#define DM_DELTA_OPEN "namespace N\n{\n public class W : Base\n {\n" +#define DM_DELTA_RUN_INT " public void Run(int n) { }\n" +#define DM_DELTA_RUN_STR " public void Run(string s) { }\n" +#define DM_DELTA_SIZE " public int Size;\n" +#define DM_DELTA_CLOSE " }\n public class Other { }\n}\n" + +TEST(doc_mentions_scope_delta_rules) { + const char *base = + DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_RUN_STR DM_DELTA_SIZE DM_DELTA_CLOSE; + struct { + const char *what; + const char *after; + int want; + const char *names; + } cases[] = { + {"unchanged", base, CBM_DOCLINK_DELTA_LOCAL, ""}, + /* nothing of the persisted scope carries a line number or a body */ + {"lines moved", + "// moved\n\n" DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_RUN_STR DM_DELTA_SIZE + DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_LOCAL, ""}, + {"body edit", + DM_DELTA_HEAD DM_DELTA_OPEN + " public void Run(int n) { n++; }\n" DM_DELTA_RUN_STR DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_LOCAL, ""}, + /* the file declares a type with a base list (`W : Base`), and a base + * list is resolved through the file's usings: what W derives from + * decides how other files' references to W's members come out */ + {"using added", + DM_DELTA_HEAD "using Acme.More;\n" DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_RUN_STR + DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + {"using removed", + DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_RUN_STR DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + {"static using and alias added", + DM_DELTA_HEAD "using static Acme.S;\nusing A = Acme.B;\n" DM_DELTA_OPEN DM_DELTA_RUN_INT + DM_DELTA_RUN_STR DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + /* ... a global using scopes every file of the project */ + {"global using added", + "global using Acme.G;\n" DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_RUN_STR + DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + /* a removed member is reported by name */ + {"overload removed", + DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_LOCAL, "Run"}, + {"field removed", + DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_RUN_STR DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_LOCAL, "Size"}, + {"two members removed", DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_LOCAL, "Run,Size"}, + /* everything else can re-route references of files with no edge here */ + {"member added", + DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_RUN_STR DM_DELTA_SIZE + " public int More;\n" DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + {"signature changed", + DM_DELTA_HEAD DM_DELTA_OPEN + " public void Run(long n) { }\n" DM_DELTA_RUN_STR DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + {"members reordered", + DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_STR DM_DELTA_RUN_INT DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + {"type removed", + DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_RUN_STR DM_DELTA_SIZE " }\n}\n", + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + {"type added", + DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_RUN_STR DM_DELTA_SIZE + " }\n public class Other { }\n public class New { }\n}\n", + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + {"base changed", + DM_DELTA_HEAD + "namespace N\n{\n public class W : Base2\n {\n" DM_DELTA_RUN_INT DM_DELTA_RUN_STR + DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + {"namespace renamed", + DM_DELTA_HEAD + "namespace M\n{\n public class W : Base\n {\n" DM_DELTA_RUN_INT DM_DELTA_RUN_STR + DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + /* a member the parser can no longer show makes its type incomplete */ + {"member hidden", + DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_RUN_STR + " public safe extern int Size();\n" DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + {"file deleted", NULL, CBM_DOCLINK_DELTA_GLOBAL, NULL}, + }; + char names[256]; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + int got = dm_delta(base, cases[i].after, names, sizeof(names)); + if (got != cases[i].want || (cases[i].names && strcmp(names, cases[i].names) != 0)) { + printf(" %s: delta %d [%s], want %d [%s]\n", cases[i].what, got, names, cases[i].want, + cases[i].names ? cases[i].names : "any"); + FAIL("scope delta"); + } + } + /* A file that declares no type with a base list: its own usings scope + * nothing but the file itself, which is re-extracted anyway. */ +#define DM_DELTA_PLAIN "namespace N\n{\n public class W\n {\n" + const char *plain = DM_DELTA_HEAD DM_DELTA_PLAIN DM_DELTA_RUN_INT DM_DELTA_SIZE DM_DELTA_CLOSE; + struct { + const char *what; + const char *after; + int want; + } usings[] = { + {"plain: using added", + DM_DELTA_HEAD + "using Acme.More;\n" DM_DELTA_PLAIN DM_DELTA_RUN_INT DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_LOCAL}, + {"plain: using removed", DM_DELTA_PLAIN DM_DELTA_RUN_INT DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_LOCAL}, + {"plain: using changed", + "using Acme.Other;\n" DM_DELTA_PLAIN DM_DELTA_RUN_INT DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_LOCAL}, + {"plain: static using and alias added", + DM_DELTA_HEAD "using static Acme.S;\nusing A = Acme.B;\n" DM_DELTA_PLAIN DM_DELTA_RUN_INT + DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_LOCAL}, + /* a using inside a namespace declaration is the file's own too */ + {"plain: block using added", + DM_DELTA_HEAD + "namespace N\n{\n using Acme.In;\n public class W\n {\n" DM_DELTA_RUN_INT + DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_LOCAL}, + /* ... and with the edit that gives W a base list, the usings count */ + {"plain: base list and using added together", + DM_DELTA_HEAD + "using Acme.More;\n" DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL}, + {"plain: global static using added", + "global using static Acme.S;\n" DM_DELTA_HEAD DM_DELTA_PLAIN DM_DELTA_RUN_INT DM_DELTA_SIZE + DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL}, + {"plain: global alias added", + "global using A = Acme.B;\n" DM_DELTA_HEAD DM_DELTA_PLAIN DM_DELTA_RUN_INT DM_DELTA_SIZE + DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL}, + }; + for (size_t i = 0; i < sizeof(usings) / sizeof(usings[0]); i++) { + int got = dm_delta(plain, usings[i].after, names, sizeof(names)); + if (got != usings[i].want || names[0]) { + printf(" %s: delta %d [%s], want %d\n", usings[i].what, got, names, usings[i].want); + FAIL("scope delta of a file's own usings"); + } + } + /* a scope that appears is as global as one that leaves */ + ASSERT_EQ(dm_delta(NULL, base, names, sizeof(names)), CBM_DOCLINK_DELTA_GLOBAL); + /* no scope before and after: a file of a language without one */ + ASSERT_EQ(cbm_doclinks_scope_delta(NULL, NULL, dm_name_put, NULL), CBM_DOCLINK_DELTA_LOCAL); + /* a blob no resolver claims is never repaired file by file */ + dm_names_t none = {{0}}; + ASSERT_EQ( + cbm_doclinks_scope_delta("zz9\nM\t0\tc\t0\tW.Run\t\tint\n", "zz9\n", dm_name_put, &none), + CBM_DOCLINK_DELTA_GLOBAL); + ASSERT_STR_EQ(none.names, ""); + + /* The MSBuild files that set a project's global usings are no scope + * inputs outside the index: each has a scope blob, and the planner judges + * a change to one by that blob like any other file's. */ + ASSERT_FALSE(cbm_doclinks_is_scope_input("src/App/App.csproj")); + ASSERT_FALSE(cbm_doclinks_is_scope_input("Directory.Build.props")); + ASSERT_FALSE(cbm_doclinks_is_scope_input("src/App/App.cs")); + ASSERT_FALSE(cbm_doclinks_is_scope_input(NULL)); +#define DM_PROJ_HEAD "\n" +#define DM_PROJ_PROPS " enable\n" +#define DM_PROJ_USING " \n" +#define DM_PROJ_PKG " \n" +#define DM_PROJ_TAIL "\n" + const char *proj = DM_PROJ_HEAD DM_PROJ_PROPS DM_PROJ_USING DM_PROJ_PKG DM_PROJ_TAIL; + struct { + const char *what; + const char *after; + int want; + } projects[] = { + {"project unchanged", proj, CBM_DOCLINK_DELTA_LOCAL}, + /* what the blob does not hold changes nobody's scope */ + {"package version", + DM_PROJ_HEAD DM_PROJ_PROPS DM_PROJ_USING " \n" DM_PROJ_TAIL, + CBM_DOCLINK_DELTA_LOCAL}, + {"target added", + DM_PROJ_HEAD DM_PROJ_PROPS DM_PROJ_USING DM_PROJ_PKG + " \n" DM_PROJ_TAIL, + CBM_DOCLINK_DELTA_LOCAL}, + /* what it holds changes every file's of the project */ + {"using removed", DM_PROJ_HEAD DM_PROJ_PROPS DM_PROJ_PKG DM_PROJ_TAIL, + CBM_DOCLINK_DELTA_GLOBAL}, + {"property changed", + DM_PROJ_HEAD + " disable\n" DM_PROJ_USING DM_PROJ_PKG + DM_PROJ_TAIL, + CBM_DOCLINK_DELTA_GLOBAL}, + {"condition added", + DM_PROJ_HEAD DM_PROJ_PROPS + " \n" DM_PROJ_PKG DM_PROJ_TAIL, + CBM_DOCLINK_DELTA_GLOBAL}, + {"import added", + DM_PROJ_HEAD + " \n" DM_PROJ_PROPS DM_PROJ_USING DM_PROJ_PKG DM_PROJ_TAIL, + CBM_DOCLINK_DELTA_GLOBAL}, + {"no longer readable", DM_PROJ_HEAD DM_PROJ_PROPS, CBM_DOCLINK_DELTA_GLOBAL}, + {"project file deleted", NULL, CBM_DOCLINK_DELTA_GLOBAL}, + }; + for (size_t i = 0; i < sizeof(projects) / sizeof(projects[0]); i++) { + int got = dm_scope_delta(CBM_LANG_XML, "src/App.csproj", proj, projects[i].after, names, + sizeof(names)); + if (got != projects[i].want || names[0]) { + printf(" %s: delta %d [%s], want %d\n", projects[i].what, got, names, + projects[i].want); + FAIL("project scope delta"); + } + } + PASS(); +} + +/* ── the parallel path == the sequential path ────────────────────── */ + +enum { DM_RING = 64 }; /* above MIN_FILES_FOR_PARALLEL */ + +/* Every class documents its successor, a shared target (by name, by overload + * group and by signature) and a name nothing declares. */ +static void dm_write_ring_fixture(const char *repo) { + for (int i = 0; i < DM_RING; i++) { + char path[512]; + char body[1024]; + snprintf(path, sizeof(path), "%s/src/C%02d.cs", repo, i); + snprintf(body, sizeof(body), + "using Ring.Shared;\n" + "namespace Ring\n" + "{\n" + " /// Next , shared ,\n" + " /// , .\n" + " /// x\n" + " public class C%02d { }\n" + "}\n", + (i + 1) % DM_RING, i, i); + th_write_file(path, body); + } + th_write_file(TH_PATH(repo, "src/Hub.cs"), "namespace Ring.Shared\n" + "{\n" + " public class Hub\n" + " {\n" + " public void Run(int n) { }\n" + " public void Run(string s) { }\n" + " }\n" + "}\n"); +} + +/* The worker pipeline (extract workers, resolve workers, per-file rows merged + * at the end) and the sequential passes publish the same edges and rows. */ +TEST(doc_mentions_parallel_equals_sequential) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_par_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + dm_write_ring_fixture(repo); + char par_db[512]; + char seq_db[512]; + snprintf(par_db, sizeof(par_db), "%s/par.db", tmp); + snprintf(seq_db, sizeof(seq_db), "%s/seq.db", tmp); + ASSERT_EQ(dm_workers_agree(repo, par_db, seq_db), 0); + /* per class: successor (see), Hub (seealso), Hub.Run(int) (exception) */ + ASSERT_EQ(dm_count(par_db, "SELECT COUNT(*) FROM edges WHERE type = 'MENTIONS'"), DM_RING * 3); + /* per class: the overload group (ambiguous) and the undeclared name */ + ASSERT_EQ(dm_count(par_db, "SELECT COUNT(*) FROM doc_link_unresolved"), DM_RING * 2); + ASSERT_EQ( + dm_count(par_db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'ambiguous'"), + DM_RING); + ASSERT_EQ(dm_count(par_db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'missing'"), + DM_RING); + + dm_unlink_db(par_db); + dm_unlink_db(seq_db); + th_rmtree(tmp); + PASS(); +} + +/* ── the scope scanner's limits and cost ─────────────────────────── */ + +/* `prefix`, then `unit` n times, then `suffix`; the caller frees it. */ +static char *dm_repeated(const char *prefix, const char *unit, int n, const char *suffix) { + size_t pl = strlen(prefix); + size_t ul = strlen(unit); + size_t sl = strlen(suffix); + char *out = malloc(pl + (ul * (size_t)n) + sl + 1); + if (!out) { + return NULL; + } + memcpy(out, prefix, pl); + for (int i = 0; i < n; i++) { + memcpy(out + pl + (ul * (size_t)i), unit, ul); + } + memcpy(out + pl + (ul * (size_t)n), suffix, sl + 1); + return out; +} + +typedef struct { + char *src; + const char *rel_path; + CBMFileResult *result; +} dm_extract_job_t; + +/* cbm_thread_create body: extract job->src as C#. */ +static void *dm_extract_thread(void *arg) { + dm_extract_job_t *job = (dm_extract_job_t *)arg; + job->result = dm_extract(job->src, CBM_LANG_CSHARP, job->rel_path); + return NULL; +} + +/* Interpolated strings nest: a hole of code can hold the next string. The + * brace scan follows them to a fixed depth; a file that goes deeper is not + * placed, and scanning it costs no stack. */ +TEST(doc_mentions_cs_scan_nested_holes) { + /* three levels, in a file whose tree has a parse error (the unsafe + * dereference): the braces are read from the text, through the holes */ + const char *nested = "namespace N\n" /* 1 */ + "{\n" + " public class Deep\n" /* 3 */ + " {\n" + " unsafe void E(void* p) { _r = ref *(int*)p; }\n" + " void M() { s = $\"a{$\"b{$\"c{1}\"}\"}\"; }\n" + " public int Q;\n" /* 7 */ + " }\n" /* 8 */ + " public class After { }\n" + "}\n"; + CBMFileResult *r = dm_extract(nested, CBM_LANG_CSHARP, "Nested.cs"); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + ASSERT_NOT_NULL(strstr(r->doc_scope, "T\t1\t3\t8\tc!\t-\tDeep\t")); + ASSERT_NOT_NULL(strstr(r->doc_scope, "\tAfter\t")); + ASSERT_NULL(strstr(r->doc_scope, "\nX\t")); + cbm_free_result(r); + + /* a hundred levels, every one closed again: deeper than the scan follows. + * The file is not placed, and its scope says so */ + enum { DM_DEEP = 100 }; + char *closers = dm_repeated("1", "}\"", DM_DEEP, "; }\n }\n}\n"); + ASSERT_NOT_NULL(closers); + char *deep = dm_repeated("namespace N\n" + "{\n" + " public class Deep\n" + " {\n" + " unsafe void E(void* p) { _r = ref *(int*)p; }\n" + " void M() { s = ", + "$\"{", DM_DEEP, closers); + free(closers); + ASSERT_NOT_NULL(deep); + r = dm_extract(deep, CBM_LANG_CSHARP, "Hundred.cs"); + free(deep); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + ASSERT_NOT_NULL(strstr(r->doc_scope, "\nX\t")); + ASSERT_NOT_NULL(strstr(r->doc_scope, "Q\tDeep\n")); + ASSERT_NULL(strstr(r->doc_scope, "\nT\t")); + cbm_free_result(r); + PASS(); +} + +/* ... and following them costs no stack: 8,000 levels left open, extracted on + * a thread with an eighth of an index worker's stack. A scan that follows + * them all needs one recursion level per three bytes of source. */ +TEST(doc_mentions_cs_scan_holes_stack) { + enum { DM_HOLES = 8000, DM_SMALL_STACK = 1024 * 1024 }; + dm_extract_job_t job = {.src = dm_repeated("namespace N\n" + "{\n" + " public class Ok { }\n" + " public class C\n" + " {\n" + " string s = ", + "$\"{", DM_HOLES, "\n"), + .rel_path = "Holes.cs"}; + ASSERT_NOT_NULL(job.src); + cbm_thread_t thread; + ASSERT_EQ(cbm_thread_create(&thread, DM_SMALL_STACK, dm_extract_thread, &job), 0); + ASSERT_EQ(cbm_thread_join(&thread), 0); + free(job.src); + CBMFileResult *r = job.result; + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + /* nothing is placed: the namespace's brace never closes */ + ASSERT_NOT_NULL(strstr(r->doc_scope, "\nX\t")); + ASSERT_NULL(strstr(r->doc_scope, "\nT\t")); + ASSERT_NULL(strstr(r->doc_scope, "\nR\t")); + ASSERT_NOT_NULL(strstr(r->doc_scope, "Q\tOk\n")); + cbm_free_result(r); + PASS(); +} + +/* The cost of scanning `src` as C#: text positions, brace-stack entries and + * modifier children visited, bytes taken from the scratch arena. false when the file has no scope. + */ +static bool dm_scan_cost(const char *src, uint64_t *steps, uint64_t *bytes) { + cbm_doclink_cs_test_cost_reset(); + CBMFileResult *r = src ? dm_extract(src, CBM_LANG_CSHARP, "Cost.cs") : NULL; + bool ok = r && r->doc_scope; + cbm_doclink_cs_test_cost(steps, bytes); + if (r) { + cbm_free_result(r); + } + return ok; +} + +/* The scan's cost of an input twice as large, as a multiple of the smaller + * one's: about 2 for a scan that is linear, 4 for a quadratic one. -1 when a + * scan fails. `prefix` + `unit` x n + `mid` + `unit2` x n + `suffix`. */ +static double dm_cost_growth(const char *prefix, const char *unit, const char *mid, + const char *unit2, int n, bool bytes) { + uint64_t cost[2] = {0, 0}; + for (int k = 0; k < 2; k++) { + int reps = n * (k + 1); + char *tail = dm_repeated(mid, unit2, reps, "\n"); + char *src = tail ? dm_repeated(prefix, unit, reps, tail) : NULL; + uint64_t steps = 0; + uint64_t scratch = 0; + bool ok = dm_scan_cost(src, &steps, &scratch); + free(tail); + free(src); + if (!ok) { + return -1.0; + } + cost[k] = bytes ? scratch : steps; + } + return cost[0] ? (double)cost[1] / (double)cost[0] : -1.0; +} + +/* The scan's work grows with its input, not faster: no clock decides these, + * the scanner's own counters do. */ + +/* A run of `$` that starts no string is passed once, not once per `$`. */ +TEST(doc_mentions_cs_scan_dollar_run) { + double growth = dm_cost_growth("class C { int x = ", "$", "; }", "", 20000, false); + if (!(growth > 0 && growth < 3.0)) { + printf(" `$` run: twice the input costs %.2f times the steps\n", growth); + FAIL("the scan of a `$` run is not linear"); + } + PASS(); +} + +/* A conditional remembers where the open braces stood, not a copy of them. */ +TEST(doc_mentions_cs_scan_branch_memory) { + double growth = + dm_cost_growth("class C { void M() {\n", "{", "\n", "#if X\n#endif\n", 3000, true); + if (!(growth > 0 && growth < 3.0)) { + printf(" #if under open braces: twice the input takes %.2f times the memory\n", growth); + FAIL("the memory of the brace scan is not linear"); + } + PASS(); +} + +/* Empty sibling branches cannot revisit a deep stack that predates them. */ +TEST(doc_mentions_cs_scan_branch_merge_work) { + enum { DEPTH = 512, SIBLINGS = 512 }; + char *opened = dm_repeated("class C { void M() {\n", "{", DEPTH, "\n#if A\n"); + ASSERT_NOT_NULL(opened); + char *closed = dm_repeated(opened, "}", DEPTH, "\n"); + free(opened); + ASSERT_NOT_NULL(closed); + char *reopened = dm_repeated(closed, "{", DEPTH, "\n"); + free(closed); + ASSERT_NOT_NULL(reopened); + char *branches = dm_repeated(reopened, "#elif B\n", SIBLINGS, "#endif\n"); + free(reopened); + ASSERT_NOT_NULL(branches); + char *src = dm_repeated(branches, "}", DEPTH, "\n} }\n"); + free(branches); + ASSERT_NOT_NULL(src); + uint64_t steps = 0; + uint64_t scratch = 0; + bool ok = dm_scan_cost(src, &steps, &scratch); + size_t bytes = strlen(src); + free(src); + ASSERT_TRUE(ok); + ASSERT_LTE(steps, (uint64_t)bytes * 16); + PASS(); +} + +/* Attributes belong to the declaration, not to each variable it declares. */ +TEST(doc_mentions_cs_scan_declarator_modifier_work) { + enum { ATTRIBUTES = 512, DECLARATORS = 512 }; + char *head = dm_repeated("class C {\n", "[A]\n", ATTRIBUTES, "public int "); + ASSERT_NOT_NULL(head); + size_t cap = strlen(head) + DECLARATORS * 16 + 16; + char *src = malloc(cap); + ASSERT_NOT_NULL(src); + size_t pos = (size_t)snprintf(src, cap, "%s", head); + free(head); + for (int i = 0; i < DECLARATORS; i++) { + pos += (size_t)snprintf(src + pos, cap - pos, "%sv%d", i ? "," : "", i); + } + pos += (size_t)snprintf(src + pos, cap - pos, ";\n}\n"); + cbm_doclink_cs_test_cost_reset(); + CBMFileResult *r = dm_extract(src, CBM_LANG_CSHARP, "Fields.cs"); + free(src); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + /* Require the grammar to expose every declarator before assessing work. */ + int members = 0; + const char *p = r->doc_scope; + while ((p = strstr(p, "\nM\t")) != NULL) { + members++; + p++; + } + uint64_t steps = 0; + uint64_t scratch = 0; + cbm_doclink_cs_test_cost(&steps, &scratch); + cbm_free_result(r); + ASSERT_EQ(members, DECLARATORS); + ASSERT_LTE(steps, (uint64_t)pos * 16); + PASS(); +} + +/* Sharing a large comment must not copy or parse its prose per declarator. + * Tokens remain per source; the measured work excludes that required output. */ +TEST(doc_mentions_cs_shared_doc_work) { + bool bounded = true; + for (int n = 16; n <= 32; n *= 2) { + for (int prose = 2048; prose <= 4096; prose *= 2) { + char *head = + dm_repeated("class C {\n/// ", "x", prose, " \npublic int "); + ASSERT_NOT_NULL(head); + size_t cap = strlen(head) + (size_t)n * 16 + 16; + char *src = malloc(cap); + ASSERT_NOT_NULL(src); + size_t len = (size_t)snprintf(src, cap, "%s", head); + free(head); + for (int i = 0; i < n; i++) { + len += (size_t)snprintf(src + len, cap - len, "%sv%d", i ? "," : "", i); + } + len += (size_t)snprintf(src + len, cap - len, ";\n}\n"); + cbm_doclink_test_doc_work_reset(); + CBMFileResult *r = dm_extract(src, CBM_LANG_CSHARP, "Shared.cs"); + free(src); + ASSERT_NOT_NULL(r); + int tokens = r->doc_links.count; + uint64_t copied = 0, parse_input = 0, cleaned = 0; + cbm_doclink_test_doc_work(&copied, &parse_input, &cleaned); + cbm_free_result(r); + printf(" shared doc: n=%d prose=%d input=%zu copied=%llu parse_input=%llu " + "cleaned=%llu tokens=%d\n", + n, prose, len, (unsigned long long)copied, (unsigned long long)parse_input, + (unsigned long long)cleaned, tokens); + ASSERT_EQ(tokens, n); + bounded = bounded && copied <= (uint64_t)len * 2 && parse_input <= (uint64_t)len * 2 && + cleaned <= 12; + } + } + ASSERT_TRUE(bounded); + PASS(); +} + +/* A one-shot resource failure must not suppress later sources via sharing. */ +TEST(doc_mentions_cs_shared_doc_allocation_failure) { + static const struct { + int kind, nth, refs, expected[3]; + } cases[] = { + {CBM_DOCLINK_ALLOC_VALUE, 3, 2, {2, 1, 2}}, + {CBM_DOCLINK_ALLOC_TOKENS, 2, 10, {10, 9, 10}}, + {CBM_DOCLINK_ALLOC_TEXT, 2, 2, {2, 2, 2}}, + /* The second comment collection loses a span while growing past8. + * It must not become the shared doc for the later variables. */ + {CBM_DOCLINK_ALLOC_SPAN, 4, 12, {12, 12, 12}}, + }; + bool ok = true; + for (size_t k = 0; k < sizeof(cases) / sizeof(cases[0]); k++) { + char *src = dm_repeated("class C {\n", "/// \n", cases[k].refs, + "public int a,b,c;\n}\n"); + ASSERT_NOT_NULL(src); + cbm_doclink_test_fail_alloc_after(cases[k].kind, cases[k].nth); + CBMFileResult *r = dm_extract(src, CBM_LANG_CSHARP, "Failure.cs"); + cbm_doclink_test_reset_alloc(); + free(src); + ASSERT_NOT_NULL(r); + int counts[3] = {0}; + for (int i = 0; i < r->doc_links.count; i++) { + const char *name = strrchr(r->doc_links.items[i].source_qn, '.'); + ASSERT_NOT_NULL(name); + name++; + ASSERT_TRUE(name[0] >= 'a' && name[0] <= 'c' && name[1] == '\0'); + counts[name[0] - 'a']++; + } + cbm_free_result(r); + printf(" shared doc allocation: stage=%d counts=%d,%d,%d expected=%d,%d,%d\n", + cases[k].kind, counts[0], counts[1], counts[2], cases[k].expected[0], + cases[k].expected[1], cases[k].expected[2]); + for (int i = 0; i < 3; i++) { + ok = ok && counts[i] == cases[k].expected[i]; + } + } + ASSERT_TRUE(ok); + PASS(); +} + +/* Shared lexical tokens survive output growth, distinct comments and files. */ +TEST(doc_mentions_cs_shared_doc_replay) { + enum { REFS = 80 }; + char *first = dm_repeated("class C {\n/// ", " ", REFS, + "\npublic int a,\nb,\nc;\n/// "); + ASSERT_NOT_NULL(first); + char *src = dm_repeated(first, " ", REFS, "\npublic int d,\ne,\nf;\n}\n"); + free(first); + ASSERT_NOT_NULL(src); + CBMFileResult *r = dm_extract(src, CBM_LANG_CSHARP, "Shared.cs"); + free(src); + ASSERT_NOT_NULL(r); + ASSERT_EQ(r->doc_links.count, 6 * REFS); + int counts[6] = {0}; + for (int i = 0; i < r->doc_links.count; i++) { + const CBMDocLink *link = &r->doc_links.items[i]; + const char *name = strrchr(link->source_qn, '.'); + ASSERT_NOT_NULL(name); + name++; + ASSERT_TRUE(name[0] >= 'a' && name[0] <= 'f' && name[1] == '\0'); + int n = name[0] - 'a'; + ASSERT_STR_EQ(link->raw, "First"); + ASSERT_EQ(link->line, n < 3 ? 2 : 6); + ASSERT_EQ(link->def_line, (uint32_t)(n < 3 ? n + 3 : n + 4)); + ASSERT_EQ(link->syntax, CBM_DOCLINK_CS_SEE); + ASSERT_EQ(link->flags, 0); + counts[n]++; + } + cbm_free_result(r); + for (int i = 0; i < 6; i++) { + ASSERT_EQ(counts[i], REFS); + } + r = dm_extract("class C {\n/// \npublic int a,b,c;\n}\n", CBM_LANG_CSHARP, + "Shared.cs"); + ASSERT_NOT_NULL(r); + ASSERT_EQ(r->doc_links.count, 3); + for (int i = 0; i < r->doc_links.count; i++) { + ASSERT_STR_EQ(r->doc_links.items[i].raw, "Other"); + ASSERT_EQ(r->doc_links.items[i].line, 2); + } + cbm_free_result(r); + PASS(); +} + +/* A declaration keyword's header is read up to the next keyword, so every + * byte of the file is read a bounded number of times. */ +TEST(doc_mentions_cs_scan_header_reads) { + const char *heads[] = {"class a ", "class a<[ "}; + for (size_t i = 0; i < sizeof(heads) / sizeof(heads[0]); i++) { + char *src = dm_repeated("namespace N {\n", heads[i], 20000, "\n"); + ASSERT_NOT_NULL(src); + uint64_t steps = 0; + uint64_t scratch = 0; + bool ok = dm_scan_cost(src, &steps, &scratch); + size_t len = strlen(src); + free(src); + ASSERT_TRUE(ok); + if (steps > (uint64_t)len * 16) { + printf(" `%s` x 20000: %llu steps for %zu bytes\n", heads[i], + (unsigned long long)steps, len); + FAIL("declaration headers are read over and over"); + } + } + PASS(); +} + +/* The gate of the project scan is the file's name: an XML file that cannot be + * an MSBuild project file is not looked at, whatever its size, and a project + * file larger than a project file is not read and says so. No clock decides + * this: the scan counts the bytes it passes. */ +TEST(doc_mentions_msbuild_gate) { + enum { DM_BIG = 2 * 1024 * 1024 }; + char *filler = malloc(DM_BIG + 1); + ASSERT_NOT_NULL(filler); + memset(filler, 'x', DM_BIG); + filler[DM_BIG] = '\0'; + char *big = + dm_repeated("

", filler, 1, "

\n"); + free(filler); + ASSERT_NOT_NULL(big); + uint64_t steps = 0; + uint64_t bytes = 0; + /* not a project file's name: no scope, and not one byte passed */ + cbm_doclink_cs_test_cost_reset(); + CBMFileResult *r = dm_extract(big, CBM_LANG_XML, "data/huge.xml"); + ASSERT_NOT_NULL(r); + ASSERT_NULL(r->doc_scope); + cbm_free_result(r); + cbm_doclink_cs_test_cost(&steps, &bytes); + ASSERT_EQ(steps, 0); + /* a project file's name on a file larger than one: not read either, and + * the blob says that it was not */ + r = dm_extract(big, CBM_LANG_XML, "eng/Huge.props"); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + ASSERT_STR_EQ(r->doc_scope, "cs1\nP\t\t>\n"); + cbm_free_result(r); + cbm_doclink_cs_test_cost(&steps, &bytes); + ASSERT_EQ(steps, 0); + /* a project file: every byte is passed once */ + const char *small = "\n"; + r = dm_extract(small, CBM_LANG_XML, "eng/Small.targets"); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + cbm_free_result(r); + cbm_doclink_cs_test_cost(&steps, &bytes); + ASSERT_EQ(steps, strlen(small)); + /* XML of another kind under a project file's name: the scan stops at its + * root element */ + cbm_doclink_cs_test_cost_reset(); + char *other = dm_repeated("", "", 4000, ""); + ASSERT_NOT_NULL(other); + r = dm_extract(other, CBM_LANG_XML, "eng/Other.props"); + free(other); + ASSERT_NOT_NULL(r); + ASSERT_NULL(r->doc_scope); + cbm_free_result(r); + cbm_doclink_cs_test_cost(&steps, &bytes); + ASSERT_EQ(steps, strlen("")); + + /* a file that was not read is counted where it is evaluated, as the + * project file itself or as an import; it opens nobody's scope */ + const dm_project_file_t files[] = { + {"eng/Huge.props", big}, + {"eng/Broken.props", ""}, + {"App.csproj", "\n" + " \n" + " \n" + " \n" + "\n"}, + }; + cbm_msb_result_t res; + ASSERT_TRUE(dm_msb_eval(files, 3, "App.csproj", &res)); + ASSERT_TRUE(dm_has_using(&res, 'n', "Still.Here")); + ASSERT_EQ(res.unevaluable, 2); + ASSERT_FALSE(res.open); + cbm_msb_result_free(&res); + ASSERT_TRUE(dm_msb_eval(files, 2, "eng/Broken.props", &res)); + ASSERT_EQ(res.count, 0); + ASSERT_EQ(res.unevaluable, 1); + cbm_msb_result_free(&res); + free(big); + PASS(); +} + +/* ── MSBuild inputs are files of the index, never of the disk ────── */ + +#ifndef _WIN32 +static const char DM_USING_EXTRA[] = "\n" + " \n" + " \n" + " \n" + "\n"; +static const char DM_USING_MORE[] = "\n" + " \n" + " \n" + " \n" + "\n"; + +/* A repository whose project files point out of it: `src/App/Evil.csproj` is a + * symbolic link to a project file outside, and `src/Lib/Lib.csproj` imports + * through `src/Lib/ext`, a link to a directory outside. Both outside files + * hold a that would bring a type of the repository into scope. */ +static void dm_write_linked_fixture(const char *repo, const char *outside) { + th_write_file(TH_PATH(outside, "Real.csproj"), DM_USING_EXTRA); + th_write_file(TH_PATH(outside, "dir/Linked.props"), DM_USING_MORE); + th_write_file(TH_PATH(repo, "src/Decl.cs"), "namespace Acme.Extra\n" + "{\n" + " public class Tool { }\n" + "}\n" + "namespace Acme.More\n" + "{\n" + " public class Gear { }\n" + "}\n"); + th_write_file(TH_PATH(repo, "src/App/Uses.cs"), + "namespace Acme.App\n" + "{\n" + " /// \n" + " public class Uses { }\n" + "}\n"); + th_write_file(TH_PATH(repo, "src/Lib/Lib.csproj"), "\n" + " \n" + "\n"); + th_write_file(TH_PATH(repo, "src/Lib/UsesLib.cs"), + "namespace Acme.Lib\n" + "{\n" + " /// \n" + " public class UsesLib { }\n" + "}\n"); +} + +/* Index the linked fixture under `tmp` into `db` (a buffer of 512 bytes). */ +static int dm_index_linked_fixture(const char *tmp, char *db) { + char repo[400]; + char outside[400]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + snprintf(outside, sizeof(outside), "%s/outside", tmp); + dm_write_linked_fixture(repo, outside); + char target[512]; + snprintf(target, sizeof(target), "%s/Real.csproj", outside); + if (symlink(target, TH_PATH(repo, "src/App/Evil.csproj")) != 0) { + return -1; + } + snprintf(target, sizeof(target), "%s/dir", outside); + if (symlink(target, TH_PATH(repo, "src/Lib/ext")) != 0) { + return -1; + } + /* a named pipe with a project file's name: opening it would block */ + if (mkfifo(TH_PATH(repo, "src/Lib/Pipe.csproj"), 0600) != 0) { + return -1; + } + snprintf(db, 512, "%s/lnk.db", tmp); + return dm_index(repo, db, NULL); +} + +/* What is not a file of the repository is not read: discovery skips symbolic + * links, and the resolver takes project files from the index alone. A project + * file that is a link to a file outside has no in effect. */ +TEST(doc_mentions_msbuild_linked_project_file) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_lnk_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char db[512]; + ASSERT_EQ(dm_index_linked_fixture(tmp, db), 0); + char props[512]; + char reason[64]; + int n = 0; + dm_edge(db, "Uses.Uses", "Decl.Tool", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + dm_row(db, "src/App/Uses.cs", "Tool", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + dm_unlink_db(db); + th_rmtree(tmp); + PASS(); +} + +/* ... and neither has a file imported through a linked directory. */ +TEST(doc_mentions_msbuild_linked_import_directory) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_lnd_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char db[512]; + ASSERT_EQ(dm_index_linked_fixture(tmp, db), 0); + char props[512]; + char reason[64]; + int n = 0; + dm_edge(db, "UsesLib.UsesLib", "Decl.Gear", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + dm_row(db, "src/Lib/UsesLib.cs", "Gear", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + dm_unlink_db(db); + th_rmtree(tmp); + PASS(); +} +#endif /* !_WIN32 */ + +/* ── the C# lookup rules, one small repository per rule ──────────── */ + +typedef struct { + const char *path; + const char *text; +} dm_source_t; + +/* What one written reference must come to. `rel` and `raw` name the + * reference (a raw text is written once per file, or comes to the same row + * every time), `src` the documented definition. + * target an edge from `src` to the node whose qualified name ends so, and + * no row; `tier` (when set) is in the edge's properties + * reason a row with that reason + * neither no row: the reference is local, or only `never` is asked + * never a node `src` must not mention */ +typedef struct { + const char *rel; + const char *raw; + const char *src; + const char *target; + const char *reason; + const char *never; + const char *tier; +} dm_want_t; + +#define DM_EXACT "\"tier\":\"exact\"" +#define DM_UNIQUE "\"tier\":\"unique\"" + +/* 1 when the database does not hold what `w` asks for (and says what it + * holds instead), else 0. */ +static int dm_want_failed(const char *db, const dm_want_t *w) { + char props[512]; + char reason[64]; + int n = 0; + int bad = 0; + dm_row(db, w->rel, w->raw, reason, sizeof(reason), NULL, 0); + if (strcmp(reason, w->reason ? w->reason : "") != 0) { + printf(" `%s` in %s: row [%s], want [%s]\n", w->raw, w->rel, reason, + w->reason ? w->reason : ""); + bad = 1; + } + if (w->target) { + dm_edge(db, w->src, w->target, props, sizeof(props), &n); + if (n != 1) { + printf(" `%s` in %s: %d edges %s -> %s, want 1\n", w->raw, w->rel, n, w->src, + w->target); + bad = 1; + } else if (w->tier && !strstr(props, w->tier)) { + printf(" `%s` in %s: edge %s, want %s\n", w->raw, w->rel, props, w->tier); + bad = 1; + } + } + if (w->never) { + dm_edge(db, w->src, w->never, props, sizeof(props), &n); + if (n != 0) { + printf(" `%s` in %s: %s -> %s is bound, want no such edge\n", w->raw, w->rel, w->src, + w->never); + bad = 1; + } + } + return bad; +} + +/* Index `files` as a repository of their own and hold the result against + * `wants`. Returns how many of them failed (each is printed), -1 when the + * repository could not be indexed. */ +static int dm_check_repo(const char *tag, const dm_source_t *files, int nfiles, + const dm_want_t *wants, int nwants) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_%s_XXXXXX", tag); + if (!cbm_mkdtemp(tmp)) { + return -1; + } + for (int i = 0; i < nfiles; i++) { + th_write_file(TH_PATH(tmp, files[i].path), files[i].text); + } + char db[512]; + snprintf(db, sizeof(db), "%s/lookup.db", tmp); + int bad = dm_index(tmp, db, NULL) == 0 ? 0 : -1; + for (int i = 0; bad >= 0 && i < nwants; i++) { + bad += dm_want_failed(db, &wants[i]); + } + dm_unlink_db(db); + th_rmtree(tmp); + return bad; +} + +#define DM_COUNT(a) ((int)(sizeof(a) / sizeof((a)[0]))) + +/* Distinct variables documented together remain distinct MENTIONS sources. */ +TEST(doc_mentions_cs_shared_doc_sources) { + static const char source[] = "namespace N {\n" + "public class First { } public class Second { }\n" + "public class C {\n" + "/// \n" + "public int a,\n" + "b,\n" + "c;\n" + "}\n}\n"; + CBMFileResult *r = dm_extract(source, CBM_LANG_CSHARP, "Shared.cs"); + ASSERT_NOT_NULL(r); + ASSERT_EQ(r->doc_links.count, 6); + int pairs[3][2] = {{0}}; + for (int i = 0; i < r->doc_links.count; i++) { + const CBMDocLink *link = &r->doc_links.items[i]; + const char *name = strrchr(link->source_qn, '.'); + ASSERT_NOT_NULL(name); + name++; + int src = strcmp(name, "a") == 0 ? 0 + : strcmp(name, "b") == 0 ? 1 + : strcmp(name, "c") == 0 ? 2 + : -1; + int dst = strcmp(link->raw, "First") == 0 ? 0 : strcmp(link->raw, "Second") == 0 ? 1 : -1; + ASSERT_TRUE(src >= 0 && dst >= 0); + ASSERT_EQ(link->line, 4); + ASSERT_EQ(link->def_line, (uint32_t)(5 + src)); + pairs[src][dst]++; + } + cbm_free_result(r); + for (int i = 0; i < 3; i++) { + for (int j = 0; j < 2; j++) { + ASSERT_EQ(pairs[i][j], 1); + } + } + const dm_source_t files[] = {{"Shared.cs", source}}; + static const dm_want_t wants[] = { + {"Shared.cs", "First", "a", "First", NULL, NULL, NULL}, + {"Shared.cs", "Second", "a", "Second", NULL, NULL, NULL}, + {"Shared.cs", "First", "b", "First", NULL, NULL, NULL}, + {"Shared.cs", "Second", "b", "Second", NULL, NULL, NULL}, + {"Shared.cs", "First", "c", "First", NULL, NULL, NULL}, + {"Shared.cs", "Second", "c", "Second", NULL, NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("shared_doc", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* The order of the scope levels: an inner namespace declaration's own usings + * and aliases are asked before an outer namespace's types; a qualified name + * and a using's target are relative to the namespaces around them before + * they are absolute; and a nearer namespace that has the first segment is + * the end of the search. */ +TEST(doc_mentions_cs_lookup_order) { + static const dm_source_t files[] = { + {"src/AcmeLogger.cs", "namespace Acme\n{\n public class Logger { }\n}\n"}, + {"src/WidgetsLogger.cs", "namespace Acme.Widgets\n" + "{\n" + " public class Logger { }\n" + " public class OnlyWidgets { }\n" + "}\n"}, + {"src/UtilGlobal.cs", "namespace Util\n" + "{\n" + " public class Helper { }\n" + " public class OnlyGlobal { }\n" + "}\n"}, + {"src/UtilAcme.cs", "namespace Acme.Util\n{\n public class Helper { }\n}\n"}, + {"src/BlockUsing.cs", "namespace Acme.App\n" + "{\n" + " using Acme.Widgets;\n" + "\n" + " /// \n" + " public class BlockUsing { }\n" + "}\n"}, + {"src/BlockAlias.cs", "namespace Acme.App\n" + "{\n" + " using Logger = Acme.Widgets.OnlyWidgets;\n" + "\n" + " /// \n" + " public class BlockAlias { }\n" + "}\n"}, + {"src/FileUsing.cs", "using Acme.Widgets;\n" + "\n" + "namespace Acme.App\n" + "{\n" + " /// \n" + " public class FileUsing { }\n" + "}\n"}, + {"src/Relative.cs", + "namespace Acme.App\n" + "{\n" + " /// \n" + " public class Relative { }\n" + "\n" + " /// \n" + " public class Absolute { }\n" + "}\n"}, + {"src/RelativeUsing.cs", "namespace Acme.App\n" + "{\n" + " using Widgets;\n" + " using W = Widgets.OnlyWidgets;\n" + "\n" + " /// \n" + " public class RelativeUsing { }\n" + "\n" + " /// \n" + " public class RelativeAlias { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + /* the block's using comes before the outer namespace Acme */ + {"src/BlockUsing.cs", "Logger", "BlockUsing.BlockUsing", "WidgetsLogger.Logger", NULL, + "AcmeLogger.Logger", DM_UNIQUE}, + /* ... and so does its alias */ + {"src/BlockAlias.cs", "Logger", "BlockAlias.BlockAlias", "WidgetsLogger.OnlyWidgets", NULL, + "AcmeLogger.Logger", DM_EXACT}, + /* the same using at the top of the file comes after every namespace */ + {"src/FileUsing.cs", "Logger", "FileUsing.FileUsing", "AcmeLogger.Logger", NULL, + "WidgetsLogger.Logger", NULL}, + /* `Util` is Acme.Util from inside Acme.App, not the global Util */ + {"src/Relative.cs", "Util.Helper", "Relative.Relative", "UtilAcme.Helper", NULL, + "UtilGlobal.Helper", DM_EXACT}, + /* ... and Acme.Util is where the search ends: no second try further out */ + {"src/Relative.cs", "Util.OnlyGlobal", "Relative.Relative", NULL, "missing", + "UtilGlobal.OnlyGlobal", NULL}, + {"src/Relative.cs", "global::Util.Helper", "Relative.Absolute", "UtilGlobal.Helper", NULL, + "UtilAcme.Helper", DM_EXACT}, + /* a using's target and an alias's are relative to their block */ + {"src/RelativeUsing.cs", "OnlyWidgets", "RelativeUsing.RelativeUsing", + "WidgetsLogger.OnlyWidgets", NULL, NULL, DM_UNIQUE}, + {"src/RelativeUsing.cs", "W", "RelativeUsing.RelativeAlias", "WidgetsLogger.OnlyWidgets", + NULL, NULL, DM_EXACT}, + }; + ASSERT_EQ(dm_check_repo("order", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* A type parameter in scope shadows a type of its name: a reference to it + * names the definition's own parameter, and is neither an edge nor a row. */ +TEST(doc_mentions_cs_type_parameters) { + static const dm_source_t files[] = { + {"src/Generic.cs", + "namespace Acme.Gen\n" + "{\n" + " public class TItem { }\n" + " public class TKey { }\n" + "\n" + " /// \n" + " public class Bag\n" + " {\n" + " /// \n" + " public void Put(TKey key) { }\n" + "\n" + " /// \n" + " public void Other() { }\n" + " }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + /* on the type: its own parameter is local, the class TKey is a class */ + {"src/Generic.cs", "TItem", "Generic.Bag", NULL, NULL, "Generic.TItem", NULL}, + {"src/Generic.cs", "TKey", "Generic.Bag", "Generic.TKey", NULL, NULL, NULL}, + /* on the generic method: both names are parameters in scope */ + {"src/Generic.cs", "TItem", "Generic.Bag.Put", NULL, NULL, "Generic.TItem", NULL}, + {"src/Generic.cs", "TKey", "Generic.Bag.Put", NULL, NULL, "Generic.TKey", NULL}, + /* on a method without the parameter: the class again */ + {"src/Generic.cs", "TKey", "Generic.Bag.Other", "Generic.TKey", NULL, NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("tparam", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +#define DM_EMPTY_PROJECT "\n\n" + +/* `using static` brings in a type's nested types and its static members + * (no instance member, no extension method). A `global using` -- of a + * namespace, of a type's static members, of an alias alike -- and a project + * file's serve every file of their project, and no other project. */ +TEST(doc_mentions_cs_usings) { + static const dm_source_t files[] = { + {"src/Lib/Maths.cs", "namespace Acme.Calc\n" + "{\n" + " public static class Maths\n" + " {\n" + " public static int Max(int a, int b) { return a; }\n" + " public static int Extension(this string s) { return 0; }\n" + " public const int Limit = 1;\n" + " public class Nested { }\n" + " }\n" + " public class Shape\n" + " {\n" + " public int Instance() { return 0; }\n" + " public static int Area() { return 0; }\n" + " }\n" + " public enum Color { Red }\n" + "}\n"}, + {"src/App/UseStatic.cs", + "using static Acme.Calc.Maths;\n" + "using static Acme.Calc.Shape;\n" + "using static Acme.Calc.Color;\n" + "\n" + "namespace Acme.App\n" + "{\n" + " /// \n" + " /// \n" + " /// \n" + " public class UseStatic { }\n" + "}\n"}, + {"src/Proj/Proj.csproj", DM_EMPTY_PROJECT}, + {"src/Proj/Globals.cs", "global using Acme.Calc;\n" + "global using static Acme.Calc.Maths;\n" + "global using GC = Acme.Calc.Color;\n"}, + {"src/Proj/UseGlobal.cs", + "namespace Proj.App\n" + "{\n" + " /// \n" + " public class UseGlobal { }\n" + "}\n"}, + {"src/Other/Other.csproj", DM_EMPTY_PROJECT}, + {"src/Other/NoGlobal.cs", + "namespace Other.App\n" + "{\n" + " /// \n" + " public class NoGlobal { }\n" + "}\n"}, + {"src/Msb/Msb.csproj", "\n" + " \n" + " \n" + " \n" + " \n" + "\n"}, + {"src/Msb/UseMsb.cs", + "namespace Msb.App\n" + "{\n" + " /// \n" + " public class UseMsb { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"src/App/UseStatic.cs", "Max", "UseStatic.UseStatic", "Maths.Maths.Max", NULL, NULL, + DM_UNIQUE}, + {"src/App/UseStatic.cs", "Limit", "UseStatic.UseStatic", "Maths.Maths.Limit", NULL, NULL, + NULL}, + {"src/App/UseStatic.cs", "Nested", "UseStatic.UseStatic", "Maths.Maths.Nested", NULL, NULL, + NULL}, + {"src/App/UseStatic.cs", "Area", "UseStatic.UseStatic", "Maths.Shape.Area", NULL, NULL, + NULL}, + {"src/App/UseStatic.cs", "Red", "UseStatic.UseStatic", "Maths.Color.Red", NULL, NULL, NULL}, + /* an instance member and an extension method do not come with it */ + {"src/App/UseStatic.cs", "Instance", "UseStatic.UseStatic", NULL, "missing", + "Maths.Shape.Instance", NULL}, + {"src/App/UseStatic.cs", "Extension", "UseStatic.UseStatic", NULL, "missing", + "Maths.Maths.Extension", NULL}, + /* the three kinds of `global using`, from another file of the project */ + {"src/Proj/UseGlobal.cs", "Shape", "UseGlobal.UseGlobal", "Maths.Shape", NULL, NULL, NULL}, + {"src/Proj/UseGlobal.cs", "Max", "UseGlobal.UseGlobal", "Maths.Maths.Max", NULL, NULL, + NULL}, + {"src/Proj/UseGlobal.cs", "GC", "UseGlobal.UseGlobal", "Maths.Color", NULL, NULL, DM_EXACT}, + /* ... and not from another project */ + {"src/Other/NoGlobal.cs", "Shape", "NoGlobal.NoGlobal", NULL, "missing", "Maths.Shape", + NULL}, + {"src/Other/NoGlobal.cs", "Max", "NoGlobal.NoGlobal", NULL, "missing", "Maths.Maths.Max", + NULL}, + {"src/Other/NoGlobal.cs", "GC", "NoGlobal.NoGlobal", NULL, "missing", "Maths.Color", NULL}, + /* a project file's and */ + {"src/Msb/UseMsb.cs", "Max", "UseMsb.UseMsb", "Maths.Maths.Max", NULL, NULL, NULL}, + {"src/Msb/UseMsb.cs", "MC", "UseMsb.UseMsb", "Maths.Color", NULL, NULL, DM_EXACT}, + {"src/Msb/UseMsb.cs", "Shape", "UseMsb.UseMsb", NULL, "missing", "Maths.Shape", NULL}, + }; + ASSERT_EQ(dm_check_repo("usings", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* What a type is: its name, its arity, the arity of every type around it, and + * its assembly. `Outer.Inner` and `Outer.Inner` are two types; two + * projects that each declare `Acme.Shared.Config` declare two types; the + * parts of a partial type in one place are one type. */ +TEST(doc_mentions_cs_entities) { + static const dm_source_t files[] = { + {"src/Twins.cs", + "namespace Acme.Ent\n" + "{\n" + " public class Outer { public class Inner { public void A() { } } }\n" + " public class Outer { public class Inner { public void B() { } } }\n" + "\n" + " /// \n" + " /// \n" + " public class UsesPlain { }\n" + "\n" + " /// \n" + " /// \n" + " public class UsesGeneric { }\n" + "}\n"}, + {"src/Part1.cs", "namespace Acme.Parts\n" + "{\n" + " public partial class Split { public void First() { } }\n" + "}\n"}, + {"src/Part2.cs", "namespace Acme.Parts\n" + "{\n" + " public partial class Split\n" + " {\n" + " public void Second() { }\n" + " public class In { public class Deep { } }\n" + " }\n" + "\n" + " /// \n" + " /// \n" + " public class UsesSplit { }\n" + "}\n"}, + {"src/P1/P1.csproj", DM_EMPTY_PROJECT}, + {"src/P1/Config.cs", + "namespace Acme.Shared\n" + "{\n" + " public class Config { public void OnlyOne() { } }\n" + "\n" + " /// \n" + " /// \n" + " public class UsesOne { }\n" + "}\n"}, + {"src/P2/P2.csproj", DM_EMPTY_PROJECT}, + {"src/P2/Config.cs", + "namespace Acme.Shared\n" + "{\n" + " public class Config { public void OnlyTwo() { } }\n" + "\n" + " /// \n" + " /// \n" + " public class UsesTwo { }\n" + "}\n"}, + {"src/P3/P3.csproj", DM_EMPTY_PROJECT}, + {"src/P3/Uses.cs", "namespace Acme.Shared\n" + "{\n" + " /// \n" + " public class UsesThree { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + /* `Outer.Inner` is the type nested in the arity-0 Outer. The node of + * that path belongs to the later declaration, the one in Outer: a + * graph gap, never the twin's node. Its members are its own. */ + {"src/Twins.cs", "Outer.Inner", "Twins.UsesPlain", NULL, "graph_gap", "Twins.Outer.Inner", + NULL}, + {"src/Twins.cs", "Outer.Inner.A", "Twins.UsesPlain", "Twins.Outer.Inner.A", NULL, NULL, + NULL}, + {"src/Twins.cs", "Outer.Inner.B", "Twins.UsesPlain", NULL, "missing", "Twins.Outer.Inner.B", + NULL}, + {"src/Twins.cs", "Outer{T}.Inner", "Twins.UsesGeneric", "Twins.Outer.Inner", NULL, NULL, + NULL}, + {"src/Twins.cs", "Outer{T}.Inner.B", "Twins.UsesGeneric", "Twins.Outer.Inner.B", NULL, NULL, + NULL}, + {"src/Twins.cs", "Outer{T}.Inner.A", "Twins.UsesGeneric", NULL, "missing", + "Twins.Outer.Inner.A", NULL}, + /* a partial type over two files: one type, the members of both, and + * the node of the declaration nearest to the reference */ + {"src/Part2.cs", "Split", "Part2.UsesSplit", "Part2.Split", NULL, "Part1.Split", NULL}, + {"src/Part2.cs", "Split.First", "Part2.UsesSplit", "Part1.Split.First", NULL, NULL, NULL}, + {"src/Part2.cs", "Split.Second", "Part2.UsesSplit", "Part2.Split.Second", NULL, NULL, NULL}, + /* nested types: by their path, never by their simple name from outside */ + {"src/Part2.cs", "Split.In", "Part2.UsesSplit", "Part2.Split.In", NULL, NULL, NULL}, + {"src/Part2.cs", "Split.In.Deep", "Part2.UsesSplit", "Part2.Split.In.Deep", NULL, NULL, + NULL}, + {"src/Part2.cs", "In", "Part2.UsesSplit", NULL, "missing", NULL, NULL}, + /* one project's Config is not the other's: each sees its own type + * and its own type's members */ + {"src/P1/Config.cs", "Config", "P1.Config.UsesOne", "P1.Config.Config", NULL, + "P2.Config.Config", NULL}, + {"src/P1/Config.cs", "Config.OnlyOne", "P1.Config.UsesOne", "P1.Config.Config.OnlyOne", + NULL, NULL, NULL}, + {"src/P1/Config.cs", "Config.OnlyTwo", "P1.Config.UsesOne", NULL, "missing", + "P2.Config.Config.OnlyTwo", NULL}, + {"src/P2/Config.cs", "Config", "P2.Config.UsesTwo", "P2.Config.Config", NULL, + "P1.Config.Config", NULL}, + {"src/P2/Config.cs", "Config.OnlyOne", "P2.Config.UsesTwo", NULL, "missing", + "P1.Config.Config.OnlyOne", NULL}, + /* a third project sees two types of that name: neither is chosen */ + {"src/P3/Uses.cs", "Config", "Uses.UsesThree", NULL, "ambiguous", "P1.Config.Config", NULL}, + }; + ASSERT_EQ(dm_check_repo("ent", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* An assembly is named by its project file: projects of one name are one + * assembly. The parts they declare are one type; complete declarations in + * two of them are flavours -- the referencing file's own project's binds, and + * from anywhere else none is chosen. Decoys: a part in a project of another + * name is no part of it, and a directory that holds two project files is an + * assembly of its own, whatever they are called. */ +TEST(doc_mentions_cs_assemblies_by_name) { + static const dm_source_t files[] = { + {"kit/win/Acme.Kit.csproj", DM_EMPTY_PROJECT}, + {"kit/win/EngineWin.cs", + "namespace Acme.Kit\n" + "{\n" + " public partial class Engine { public void Start() { } }\n" + " public class Clock { public void Tick() { } }\n" + "\n" + " /// \n" + " /// \n" + " /// \n" + " public class UsesWin { }\n" + "}\n"}, + {"kit/unix/Acme.Kit.csproj", DM_EMPTY_PROJECT}, + {"kit/unix/EngineUnix.cs", "namespace Acme.Kit\n" + "{\n" + " public partial class Engine { public void Stop() { } }\n" + " public class Clock { public void Tick() { } }\n" + "}\n"}, + {"kit/tools/Acme.Kit.csproj", DM_EMPTY_PROJECT}, + {"kit/tools/Tools.cs", + "namespace Acme.Kit\n" + "{\n" + " /// \n" + " /// \n" + " public class UsesTools { }\n" + "}\n"}, + {"other/Acme.Other.csproj", DM_EMPTY_PROJECT}, + {"other/EngineOther.cs", "namespace Acme.Kit\n" + "{\n" + " public partial class Engine { public void Halt() { } }\n" + " public class Solo { }\n" + "}\n"}, + {"pair/Acme.Kit.csproj", DM_EMPTY_PROJECT}, + {"pair/Second.csproj", DM_EMPTY_PROJECT}, + {"pair/EnginePair.cs", "namespace Acme.Kit\n" + "{\n" + " public partial class Engine { public void Pair() { } }\n" + "}\n"}, + {"app/App.csproj", DM_EMPTY_PROJECT}, + {"app/UsesApp.cs", + "namespace Acme.App\n" + "{\n" + " /// \n" + " public class UsesApp { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + /* the parts of two projects of one name are one type */ + {"kit/win/EngineWin.cs", "Engine.Stop", "EngineWin.UsesWin", "unix.EngineUnix.Engine.Stop", + NULL, NULL, NULL}, + {"kit/tools/Tools.cs", "Engine.Start", "Tools.UsesTools", "win.EngineWin.Engine.Start", + NULL, NULL, NULL}, + /* decoys: another name, and a directory with two project files */ + {"kit/win/EngineWin.cs", "Engine.Halt", "EngineWin.UsesWin", NULL, "missing", + "EngineOther.Engine.Halt", NULL}, + {"kit/win/EngineWin.cs", "Engine.Pair", "EngineWin.UsesWin", NULL, "missing", + "EnginePair.Engine.Pair", NULL}, + /* flavours: the own project's declaration and its member */ + {"kit/win/EngineWin.cs", "Clock", "EngineWin.UsesWin", "win.EngineWin.Clock", NULL, + "unix.EngineUnix.Clock", NULL}, + {"kit/win/EngineWin.cs", "Clock.Tick", "EngineWin.UsesWin", "win.EngineWin.Clock.Tick", + NULL, "unix.EngineUnix.Clock.Tick", NULL}, + /* ... and from a project that has none of them, of the same assembly + * or of another, none is chosen */ + {"kit/tools/Tools.cs", "Clock", "Tools.UsesTools", NULL, "ambiguous", "win.EngineWin.Clock", + NULL}, + {"kit/tools/Tools.cs", "Clock.Tick", "Tools.UsesTools", NULL, "ambiguous", + "win.EngineWin.Clock.Tick", NULL}, + {"app/UsesApp.cs", "Acme.Kit.Clock", "UsesApp.UsesApp", NULL, "ambiguous", + "unix.EngineUnix.Clock", NULL}, + /* what one other assembly declares binds */ + {"app/UsesApp.cs", "Acme.Kit.Solo", "UsesApp.UsesApp", "EngineOther.Solo", NULL, NULL, + DM_EXACT}, + }; + ASSERT_EQ(dm_check_repo("asm", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* Inside one assembly a declaration in a project directory named `ref` -- a + * reference assembly's source -- stands behind the implementation: the + * implementation's node binds, the stub's only where the implementation has + * none (here: `Hidden` and `Hidden` share one node in the implementation's + * file, and it is `Hidden`'s). Decoy: a directory named `ref` inside a + * project is no reference assembly, and what stands in it is no stub. */ +TEST(doc_mentions_cs_reference_source) { + static const dm_source_t files[] = { + {"lib/ref/Acme.Lib.csproj", DM_EMPTY_PROJECT}, + {"lib/ref/Stubs.cs", + "namespace Acme.Lib\n" + "{\n" + " public partial class Widget { public void Spin() { } public void OnlyStub() { } }\n" + " public partial class Hidden { public void Peek() { } }\n" + "}\n"}, + {"lib/src/Acme.Lib.csproj", DM_EMPTY_PROJECT}, + {"lib/src/Widget.cs", "namespace Acme.Lib\n" + "{\n" + " public class Widget { public void Spin() { } }\n" + "}\n"}, + {"lib/src/Hidden.cs", "namespace Acme.Lib\n" + "{\n" + " public class Hidden { public void Peek() { } }\n" + " public class Hidden { }\n" + "}\n"}, + {"lib/src/ref/Helper.cs", "namespace Acme.Lib\n" + "{\n" + " public partial class Helper { public void Near() { } }\n" + "\n" + " /// \n" + " public class UsesHelper { }\n" + "}\n"}, + {"lib/src/HelperMore.cs", "namespace Acme.Lib\n" + "{\n" + " public partial class Helper { public void Far() { } }\n" + "}\n"}, + {"app/App.csproj", DM_EMPTY_PROJECT}, + {"app/UsesApp.cs", + "namespace Acme.App\n" + "{\n" + " /// \n" + " /// \n" + " /// \n" + " public class UsesApp { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + /* stub and implementation are one type: the implementation's nodes */ + {"app/UsesApp.cs", "Acme.Lib.Widget", "UsesApp.UsesApp", "src.Widget.Widget", NULL, + "ref.Stubs.Widget", DM_EXACT}, + {"app/UsesApp.cs", "Acme.Lib.Widget.Spin", "UsesApp.UsesApp", "src.Widget.Widget.Spin", + NULL, "ref.Stubs.Widget.Spin", NULL}, + /* the stub's, where the implementation has none */ + {"app/UsesApp.cs", "Acme.Lib.Widget.OnlyStub", "UsesApp.UsesApp", + "ref.Stubs.Widget.OnlyStub", NULL, NULL, NULL}, + {"app/UsesApp.cs", "Acme.Lib.Hidden", "UsesApp.UsesApp", "ref.Stubs.Hidden", NULL, + "src.Hidden.Hidden", NULL}, + {"app/UsesApp.cs", "Acme.Lib.Hidden.Peek", "UsesApp.UsesApp", "src.Hidden.Hidden.Peek", + NULL, "ref.Stubs.Hidden.Peek", NULL}, + /* decoy: both parts are implementation, the nearest is the one beside + * the reference */ + {"lib/src/ref/Helper.cs", "Helper", "ref.Helper.UsesHelper", "ref.Helper.Helper", NULL, + "HelperMore.Helper", NULL}, + }; + ASSERT_EQ(dm_check_repo("refsrc", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* Declarations of one full name in projects of different names are different + * types, and a project that declares none of them is not told which one it + * references: nothing chooses -- not the one that is a stub, not the one that + * is not, not the nearer directory. Decoys: the declaring assembly sees its + * own type, and a name only one other assembly declares binds. */ +TEST(doc_mentions_cs_assemblies_differ) { + static const dm_source_t files[] = { + {"impl/Acme.Impl.csproj", DM_EMPTY_PROJECT}, + {"impl/Gadget.cs", "namespace Acme.Things\n" + "{\n" + " public class Gadget { public void Run() { } }\n" + " public class OnlyImpl { }\n" + "\n" + " /// \n" + " public class UsesImpl { }\n" + "}\n"}, + {"facade/ref/Acme.Facade.csproj", DM_EMPTY_PROJECT}, + {"facade/ref/Stubs.cs", "namespace Acme.Things\n" + "{\n" + " public partial class Gadget { public void Run() { } }\n" + "}\n"}, + {"impl/near/Near.csproj", DM_EMPTY_PROJECT}, + {"impl/near/UsesNear.cs", "namespace Acme.Near\n" + "{\n" + " /// \n" + " /// \n" + " public class UsesNear { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"impl/near/UsesNear.cs", "Acme.Things.Gadget", "UsesNear.UsesNear", NULL, "ambiguous", + "impl.Gadget.Gadget", NULL}, + {"impl/near/UsesNear.cs", "Acme.Things.Gadget", "UsesNear.UsesNear", NULL, "ambiguous", + "ref.Stubs.Gadget", NULL}, + {"impl/near/UsesNear.cs", "Acme.Things.Gadget.Run", "UsesNear.UsesNear", NULL, "ambiguous", + "impl.Gadget.Gadget.Run", NULL}, + {"impl/near/UsesNear.cs", "Acme.Things.OnlyImpl", "UsesNear.UsesNear", + "impl.Gadget.OnlyImpl", NULL, NULL, DM_EXACT}, + {"impl/Gadget.cs", "Gadget", "Gadget.UsesImpl", "impl.Gadget.Gadget", NULL, + "ref.Stubs.Gadget", NULL}, + }; + ASSERT_EQ(dm_check_repo("differ", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* A file no project file stands above is of a shared tree, and the parts of + * a partial type there are parts of every assembly's partial type of that + * name: what a reference sees is its own assembly's parts and the shared + * trees'. A part in a project of another name is no part of it, and a name + * such a part declares is, seen without it, neither bound nor missing. + * Decoy: a complete type takes no shared parts. */ +TEST(doc_mentions_cs_shared_parts) { + static const dm_source_t files[] = { + {"shared/kit/ToolShared.cs", + "namespace Acme.Kit\n" + "{\n" + " public partial class Tool { public void Common() { } }\n" + " public partial class Gear\n" + " {\n" + " /// \n" + " public void Turn() { }\n" + " public void Mesh() { }\n" + " }\n" + " public class Spare { }\n" + "\n" + " /// \n" + " public class UsesShared { }\n" + "}\n"}, + {"one/One.csproj", DM_EMPTY_PROJECT}, + {"one/ToolOne.cs", "namespace Acme.Kit\n" + "{\n" + " public partial class Tool { public void OnlyOne() { } }\n" + " public partial class Gear\n" + " {\n" + " public void OnlyOne() { }\n" + " public void Mesh(int teeth) { }\n" + " public int Spare;\n" + " public class Inner { public void Deep() { } }\n" + " }\n" + "\n" + " /// \n" + " /// \n" + " public class UsesOne { }\n" + "}\n"}, + {"two/Two.csproj", DM_EMPTY_PROJECT}, + {"two/ToolTwo.cs", "namespace Acme.Kit\n" + "{\n" + " public partial class Tool { public void OnlyTwo() { } }\n" + "}\n"}, + {"three/Three.csproj", DM_EMPTY_PROJECT}, + {"three/UsesThree.cs", + "namespace Acme.Kit\n" + "{\n" + " /// \n" + " /// \n" + " /// \n" + " public class UsesThree { }\n" + "}\n"}, + {"four/Four.csproj", DM_EMPTY_PROJECT}, + {"four/ToolFour.cs", + "namespace Acme.Kit\n" + "{\n" + " public class Tool { public void OnlyFour() { } }\n" + "\n" + " /// \n" + " public class UsesFour { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + /* an assembly's partial type: its own parts and the shared trees' */ + {"one/ToolOne.cs", "Tool", "ToolOne.UsesOne", "one.ToolOne.Tool", NULL, "two.ToolTwo.Tool", + NULL}, + {"one/ToolOne.cs", "Tool.Common", "ToolOne.UsesOne", "kit.ToolShared.Tool.Common", NULL, + NULL, NULL}, + {"one/ToolOne.cs", "Tool.OnlyOne", "ToolOne.UsesOne", "one.ToolOne.Tool.OnlyOne", NULL, + NULL, NULL}, + /* ... and not another assembly's */ + {"one/ToolOne.cs", "Tool.OnlyTwo", "ToolOne.UsesOne", NULL, "missing", + "two.ToolTwo.Tool.OnlyTwo", NULL}, + /* an assembly without parts of its own sees the shared parts */ + {"three/UsesThree.cs", "Gear", "UsesThree.UsesThree", "kit.ToolShared.Gear", NULL, NULL, + NULL}, + {"three/UsesThree.cs", "Gear.Turn", "UsesThree.UsesThree", "kit.ToolShared.Gear.Turn", NULL, + NULL, NULL}, + /* a name another assembly's part declares: not bound, not missing */ + {"three/UsesThree.cs", "Gear.OnlyOne", "UsesThree.UsesThree", NULL, "ambiguous", + "one.ToolOne.Gear.OnlyOne", NULL}, + {"three/UsesThree.cs", "Gear.Gone", "UsesThree.UsesThree", NULL, "missing", NULL, NULL}, + /* ... also where the shared parts have a member of that name, and + * where something further out has: the name is the part's first */ + {"three/UsesThree.cs", "Gear.Mesh", "UsesThree.UsesThree", NULL, "ambiguous", + "kit.ToolShared.Gear.Mesh", NULL}, + {"shared/kit/ToolShared.cs", "Spare", "ToolShared.Gear.Turn", NULL, "ambiguous", + "kit.ToolShared.Spare", NULL}, + /* ... and a type nested in such a part, on the way to its member */ + {"three/UsesThree.cs", "Gear.Inner.Deep", "UsesThree.UsesThree", NULL, "ambiguous", + "one.ToolOne.Gear.Inner.Deep", NULL}, + /* ... and so from the shared tree itself */ + {"shared/kit/ToolShared.cs", "Gear.Turn", "ToolShared.UsesShared", + "kit.ToolShared.Gear.Turn", NULL, NULL, NULL}, + {"shared/kit/ToolShared.cs", "Gear.OnlyOne", "ToolShared.UsesShared", NULL, "ambiguous", + "one.ToolOne.Gear.OnlyOne", NULL}, + /* decoy: a complete type has no further parts */ + {"four/ToolFour.cs", "Tool.Common", "ToolFour.UsesFour", NULL, "missing", + "kit.ToolShared.Tool.Common", NULL}, + {"four/ToolFour.cs", "Tool.OnlyFour", "ToolFour.UsesFour", "four.ToolFour.Tool.OnlyFour", + NULL, NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("parts", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* A shared tree is the unit of the files no project file stands above: the + * largest directory around them with no project file in or below it. Its + * files see its declarations as their own, whatever directory of it they + * stand in; two trees' declarations of one name are not chosen between; one + * tree's partial and complete declarations of a name are one type. Decoy: a + * project is one unit whatever its directories. */ +TEST(doc_mentions_cs_shared_trees) { + static const dm_source_t files[] = { + {"core/sys/Thing.cs", "namespace Acme.Sys\n" + "{\n" + " public class Thing { public void InCore() { } }\n" + " public class OnlyCore { }\n" + " public partial class Part { public void Same() { } }\n" + "}\n"}, + {"core/sys/Wrap.cs", "namespace Acme.Sys\n" + "{\n" + " public partial class Wrap\n" + " {\n" + " /// \n" + " public void Real() { }\n" + " }\n" + "}\n"}, + {"core/sys/WrapOther.cs", "namespace Acme.Sys\n" + "{\n" + " public class Wrap { public void Real() { } }\n" + "}\n"}, + {"core/text/UsesCore.cs", + "namespace Acme.Sys.Text\n" + "{\n" + " /// \n" + " /// \n" + " public class UsesCore { }\n" + "}\n"}, + {"base/sys/Thing.cs", "namespace Acme.Sys\n" + "{\n" + " public class Thing { public void InBase() { } }\n" + " public partial class Part { public void Same() { } }\n" + "}\n"}, + {"base/UsesBase.cs", "namespace Acme.Sys\n" + "{\n" + " /// \n" + " public class UsesBase { }\n" + "}\n"}, + {"app/App.csproj", DM_EMPTY_PROJECT}, + {"app/UsesApp.cs", + "namespace Acme.App\n" + "{\n" + " /// \n" + " /// \n" + " public class UsesApp { }\n" + "}\n"}, + {"app/inner/Deep.cs", "namespace Acme.App\n{\n public class Deep { }\n}\n"}, + {"app/other/UsesDeep.cs", "namespace Acme.App\n" + "{\n" + " /// \n" + " public class UsesDeep { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + /* its own tree's declaration, from another directory of the tree */ + {"core/text/UsesCore.cs", "Thing", "UsesCore.UsesCore", "core.sys.Thing.Thing", NULL, + "base.sys.Thing.Thing", NULL}, + {"core/text/UsesCore.cs", "Thing.InCore", "UsesCore.UsesCore", + "core.sys.Thing.Thing.InCore", NULL, NULL, NULL}, + {"core/text/UsesCore.cs", "Thing.InBase", "UsesCore.UsesCore", NULL, "missing", + "base.sys.Thing.Thing.InBase", NULL}, + {"base/UsesBase.cs", "Thing", "UsesBase.UsesBase", "base.sys.Thing.Thing", NULL, + "core.sys.Thing.Thing", NULL}, + /* parts in two trees: the own tree's part and member */ + {"core/text/UsesCore.cs", "Part", "UsesCore.UsesCore", "core.sys.Thing.Part", NULL, + "base.sys.Thing.Part", NULL}, + {"core/text/UsesCore.cs", "Part.Same", "UsesCore.UsesCore", "core.sys.Thing.Part.Same", + NULL, "base.sys.Thing.Part.Same", NULL}, + /* one tree's partial and complete declarations of a name: one type */ + {"core/sys/Wrap.cs", "Wrap", "Wrap.Wrap.Real", "core.sys.Wrap.Wrap", NULL, "WrapOther.Wrap", + NULL}, + /* from outside the trees nothing is chosen between two of them */ + {"app/UsesApp.cs", "Acme.Sys.Thing", "UsesApp.UsesApp", NULL, "ambiguous", + "core.sys.Thing.Thing", NULL}, + {"app/UsesApp.cs", "Acme.Sys.Part", "UsesApp.UsesApp", NULL, "ambiguous", + "core.sys.Thing.Part", NULL}, + {"app/UsesApp.cs", "Acme.Sys.Part.Same", "UsesApp.UsesApp", NULL, "ambiguous", + "core.sys.Thing.Part.Same", NULL}, + /* ... and what one tree declares binds */ + {"app/UsesApp.cs", "Acme.Sys.OnlyCore", "UsesApp.UsesApp", "core.sys.Thing.OnlyCore", NULL, + NULL, DM_EXACT}, + /* decoy: the directories of a project are one unit */ + {"app/other/UsesDeep.cs", "Deep", "UsesDeep.UsesDeep", "inner.Deep.Deep", NULL, NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("trees", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* A repository without any project file is one unit: a `global using` in one + * directory serves the files of every other one. */ +TEST(doc_mentions_cs_no_project_files) { + static const dm_source_t files[] = { + {"far/Far.cs", "namespace Acme.Far\n{\n public class Remote { }\n}\n"}, + {"conf/Globals.cs", "global using Acme.Far;\n"}, + {"use/Uses.cs", "namespace Acme.Use\n" + "{\n" + " /// \n" + " public class Uses { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"use/Uses.cs", "Remote", "Uses.Uses", "far.Far.Remote", NULL, NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("noproj", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* A contract and its one implementation are one type: an assembly whose + * every declaration of a type stands in a project directory named `ref`, and + * the one declaration of that name and arity the shared trees hold. The + * implementation's node binds, the stub's only where the implementation has + * none, and no binding that exists only through the join is exact. Decoys: + * two shared trees declare the name; the assembly has a declaration that is + * no stub; the arity differs. */ +TEST(doc_mentions_cs_contract_join) { + static const dm_source_t files[] = { + {"shared/impl/Bag.cs", "namespace Acme.Coll\n" + "{\n" + " public class Bag { public void Add() { } }\n" + " public class Twice { }\n" + " public class Mix { public void Shared() { } }\n" + " public class Pair { }\n" + " public partial class Spread { public void A() { } }\n" + " public class Rivaled { }\n" + "}\n"}, + {"shared/impl/Ghost.cs", "namespace Acme.Coll\n" + "{\n" + " public class Ghost { public void Boo() { } }\n" + " public class Ghost { }\n" + "}\n"}, + {"second/Twice.cs", "namespace Acme.Coll\n" + "{\n" + " public class Twice { }\n" + " public partial class Spread { public void B() { } }\n" + "}\n"}, + {"facade/ref/Acme.Facade.csproj", DM_EMPTY_PROJECT}, + {"facade/ref/Stubs.cs", + "namespace Acme.Coll\n" + "{\n" + " public partial class Bag { public void Add() { } public void OnlyStub() { } }\n" + " public partial class Twice { }\n" + " public partial class Pair { }\n" + " public partial class Ghost { public void Boo() { } }\n" + " public partial class Spread { }\n" + " public partial class Rivaled { }\n" + "\n" + " /// \n" + " /// \n" + " public partial class UsesFacade { }\n" + "\n" + " /// \n" + " /// \n" + " public partial class UsesFacadeFull { }\n" + "}\n"}, + {"mixed/ref/Acme.Mixed.csproj", DM_EMPTY_PROJECT}, + {"mixed/ref/Stubs.cs", "namespace Acme.Coll\n" + "{\n" + " public partial class Mix { public void Own() { } }\n" + "}\n"}, + {"mixed/src/Acme.Mixed.csproj", DM_EMPTY_PROJECT}, + {"mixed/src/Mix.cs", + "namespace Acme.Coll\n" + "{\n" + " public class Mix { public void Own() { } }\n" + "\n" + " /// \n" + " public class UsesMixed { }\n" + "}\n"}, + {"rival/Acme.Rival.csproj", DM_EMPTY_PROJECT}, + {"rival/Rivaled.cs", "namespace Acme.Coll\n{\n public class Rivaled { }\n}\n"}, + {"app/App.csproj", DM_EMPTY_PROJECT}, + {"app/UsesApp.cs", + "namespace Acme.App\n" + "{\n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " public class UsesApp { }\n" + "}\n"}, + {"app/UsesAlias.cs", "using B = Acme.Coll.Bag;\n" + "\n" + "namespace Acme.App\n" + "{\n" + " /// \n" + " public class UsesAlias { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + /* the stub is no second type: the implementation binds, and never as + * an exact binding -- by its full name, by an alias */ + {"app/UsesApp.cs", "Acme.Coll.Bag", "UsesApp.UsesApp", "impl.Bag.Bag", NULL, + "ref.Stubs.Bag", DM_UNIQUE}, + {"app/UsesApp.cs", "Acme.Coll.Bag.Add", "UsesApp.UsesApp", "impl.Bag.Bag.Add", NULL, + "ref.Stubs.Bag.Add", DM_UNIQUE}, + {"app/UsesAlias.cs", "B", "UsesAlias.UsesAlias", "impl.Bag.Bag", NULL, NULL, DM_UNIQUE}, + /* from the contract's own assembly: the implementation's node, and + * the stub's own member where the implementation has none */ + {"facade/ref/Stubs.cs", "Bag", "Stubs.UsesFacade", "impl.Bag.Bag", NULL, "ref.Stubs.Bag", + DM_UNIQUE}, + {"facade/ref/Stubs.cs", "Bag.OnlyStub", "Stubs.UsesFacade", "ref.Stubs.Bag.OnlyStub", NULL, + NULL, NULL}, + {"facade/ref/Stubs.cs", "Acme.Coll.Bag", "Stubs.UsesFacadeFull", "impl.Bag.Bag", NULL, + "ref.Stubs.Bag", DM_UNIQUE}, + {"facade/ref/Stubs.cs", "Acme.Coll.Bag.Add", "Stubs.UsesFacadeFull", "impl.Bag.Bag.Add", + NULL, "ref.Stubs.Bag.Add", DM_UNIQUE}, + /* the implementation without a node: the stub's node stands in */ + {"app/UsesApp.cs", "Acme.Coll.Ghost", "UsesApp.UsesApp", "ref.Stubs.Ghost", NULL, + "impl.Ghost.Ghost", DM_UNIQUE}, + /* decoy: two shared trees declare the name -- complete in each, or a + * part in each: no join, and the contract's assembly sees its stub */ + {"app/UsesApp.cs", "Acme.Coll.Twice", "UsesApp.UsesApp", NULL, "ambiguous", + "impl.Bag.Twice", NULL}, + {"facade/ref/Stubs.cs", "Twice", "Stubs.UsesFacade", "ref.Stubs.Twice", NULL, + "impl.Bag.Twice", NULL}, + {"facade/ref/Stubs.cs", "Spread", "Stubs.UsesFacade", "ref.Stubs.Spread", NULL, + "impl.Bag.Spread", NULL}, + /* decoy: another assembly holds an implementation of the name too -- + * the contract is not joined to one of two implementations */ + {"facade/ref/Stubs.cs", "Rivaled", "Stubs.UsesFacade", "ref.Stubs.Rivaled", NULL, + "impl.Bag.Rivaled", NULL}, + /* decoy: the assembly has a declaration that is no stub */ + {"app/UsesApp.cs", "Acme.Coll.Mix", "UsesApp.UsesApp", NULL, "ambiguous", "impl.Bag.Mix", + NULL}, + {"mixed/src/Mix.cs", "Mix.Own", "Mix.UsesMixed", "src.Mix.Mix.Own", NULL, NULL, NULL}, + {"mixed/src/Mix.cs", "Mix.Shared", "Mix.UsesMixed", NULL, "missing", "impl.Bag.Mix.Shared", + NULL}, + /* decoy: another arity is another type -- each is the only one of its + * arity, and bound without the join */ + {"app/UsesApp.cs", "Acme.Coll.Pair", "UsesApp.UsesApp", "impl.Bag.Pair", NULL, NULL, + DM_EXACT}, + {"app/UsesApp.cs", "Acme.Coll.Pair{T}", "UsesApp.UsesApp", "ref.Stubs.Pair", NULL, NULL, + DM_EXACT}, + }; + ASSERT_EQ(dm_check_repo("join", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* A project file of an SDK that compiles nothing (NoTargets: it only runs + * build steps) is the project of no source file: the sources below it that + * have no nearer project are a shared tree, and a contract finds its one + * implementation there. Decoy: a project file that compiles keeps the files + * below it, and their types are that assembly's. */ +TEST(doc_mentions_cs_idle_project_file) { + static const dm_source_t files[] = { + {"libs/build.csproj", "\n\n"}, + {"libs/core/src/Impl.cs", "namespace Acme.Core\n" + "{\n" + " public class Thing { public void Go() { } }\n" + "}\n"}, + {"libs/core/ref/Acme.Core.csproj", DM_EMPTY_PROJECT}, + {"libs/core/ref/Stubs.cs", "namespace Acme.Core\n" + "{\n" + " public partial class Thing { public void Go() { } }\n" + "}\n"}, + {"libs/user/User.csproj", DM_EMPTY_PROJECT}, + {"libs/user/Uses.cs", + "namespace Acme.User\n" + "{\n" + " /// \n" + " /// \n" + " public class Uses { }\n" + "}\n"}, + {"apps/Apps.csproj", DM_EMPTY_PROJECT}, + {"apps/tool/src/Impl.cs", "namespace Acme.Apps\n{\n public class Gizmo { }\n}\n"}, + {"apps/tool/ref/Tool.csproj", DM_EMPTY_PROJECT}, + {"apps/tool/ref/Stubs.cs", "namespace Acme.Apps\n" + "{\n" + " public partial class Gizmo { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"libs/user/Uses.cs", "Acme.Core.Thing", "Uses.Uses", "core.src.Impl.Thing", NULL, + "core.ref.Stubs.Thing", DM_UNIQUE}, + {"libs/user/Uses.cs", "Acme.Apps.Gizmo", "Uses.Uses", NULL, "ambiguous", + "tool.src.Impl.Gizmo", NULL}, + }; + ASSERT_EQ(dm_check_repo("idle", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* For product code a namespace that only test code declares does not exist: + * it does not stand in the way of a type a using brings in, and a name that + * is not under it is judged without it -- while a name that is under it is + * test code's (test_only_target). Decoys: a namespace product code declares + * does stand there (the global namespace's own names come before the file's + * usings), and test code sees the test namespace. */ +TEST(doc_mentions_cs_test_namespaces) { + static const dm_source_t files[] = { + {"src/Lib/Widget.cs", "namespace Acme.Lib\n" + "{\n" + " public class Widget { }\n" + " public class Gadget { }\n" + "}\n"}, + {"src/App/Uses.cs", + "using Acme.Lib;\n" + "\n" + "namespace Acme.App\n" + "{\n" + " /// \n" + " public class Uses { }\n" + "\n" + " /// \n" + " /// \n" + " /// \n" + " public class UsesFull { }\n" + "}\n"}, + {"tests/Widget/WidgetTests.cs", "namespace Widget\n{\n public class Fixture { }\n}\n"}, + {"tests/Probe/Rig.cs", "namespace Probe\n{\n public class Rig { }\n}\n"}, + {"tests/Lib/Extra.cs", "namespace Acme.Lib.Extra\n{\n public class Bonus { }\n}\n"}, + {"src/Gadget/Kinds.cs", "namespace Gadget\n{\n public class Kind { }\n}\n"}, + {"tests/App/UsesTests.cs", "using Acme.Lib;\n" + "\n" + "namespace Acme.App.Tests\n" + "{\n" + " /// \n" + " public class UsesTests { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"src/App/Uses.cs", "Widget", "Uses.Uses", "Lib.Widget.Widget", NULL, NULL, NULL}, + {"src/App/Uses.cs", "Gadget", "Uses.Uses", NULL, "graph_gap", "Lib.Widget.Gadget", NULL}, + {"tests/App/UsesTests.cs", "Widget", "UsesTests.UsesTests", NULL, "graph_gap", + "Lib.Widget.Widget", NULL}, + /* through a test-only namespace: what is there is test code's ... */ + {"src/App/Uses.cs", "Probe.Rig", "Uses.UsesFull", NULL, "test_only_target", "Probe.Rig.Rig", + NULL}, + {"src/App/Uses.cs", "Acme.Lib.Extra.Bonus", "Uses.UsesFull", NULL, "test_only_target", + "Lib.Extra.Bonus", NULL}, + {"src/App/Uses.cs", "T:Probe.Rig", "Uses.UsesFull", NULL, "test_only_target", + "Probe.Rig.Rig", NULL}, + /* ... and what is not there is judged as if the namespace were not: + * no namespace of the repository, and a namespace without the name */ + {"src/App/Uses.cs", "Probe.Nothing", "Uses.UsesFull", NULL, "external", NULL, NULL}, + {"src/App/Uses.cs", "Acme.Lib.Extra.Nothing", "Uses.UsesFull", NULL, "missing", NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("testns", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* Members: the type that has the name decides, and the written parameter + * list is matched against ITS overloads -- exactly first, a type variable by + * its position, a type nothing is known about only when nothing matches + * exactly. Operators and indexers are declared or they are not. */ +TEST(doc_mentions_cs_members) { + static const dm_source_t files[] = { + {"src/Members.cs", + "namespace Acme.M\n" + "{\n" + " public class Value { public Value(int x) { } }\n" + " public class Two { public Two(long x) { } }\n" + "\n" + " public class Box\n" + " {\n" + " public void Put(T x) { }\n" + " public void Two(int a) { }\n" + " public void Two(string a) { }\n" + " public void Gen() { }\n" + " public void Gen() { }\n" + " public void Dup(int a) { }\n" + " public void Dup(int a) { }\n" + " public void OnlyGen(U u) { }\n" + " public void Map(K key, V value) { }\n" + " public void Loose(Unknown.Thing x) { }\n" + " public int Value { get; set; }\n" + " public int field;\n" + " public event System.Action Changed;\n" + " public int this[int i] { get { return i; } }\n" + " public static Box operator +(Box a, Box b) { return a; }\n" + " public void Item(string key) { }\n" + "\n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " public void Doc() { }\n" + " }\n" + "}\n"}, + {"src/Plain.cs", + "namespace Acme.M\n" + "{\n" + " public class Plain\n" + " {\n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " public void Doc() { }\n" + " }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"src/Members.cs", "Put(T)", "Members.Box.Doc", "Members.Box.Put", NULL, NULL, NULL}, + /* the type has `Put`, and no overload of it takes an int */ + {"src/Members.cs", "Put(int)", "Members.Box.Doc", NULL, "missing", NULL, NULL}, + {"src/Members.cs", "Two(int)", "Members.Box.Doc", "Members.Box.Two", NULL, NULL, NULL}, + /* ... and nothing further out is asked: not the class Two, whose + * constructor takes a long, and not the class Value for a property */ + {"src/Members.cs", "Two(long)", "Members.Box.Doc", NULL, "missing", "Members.Two.Two", + NULL}, + {"src/Members.cs", "Value(int)", "Members.Box.Doc", NULL, "missing", "Members.Value.Value", + NULL}, + /* without type arguments a non-generic method stands before a generic one */ + {"src/Members.cs", "Gen", "Members.Box.Doc", "Members.Box.Gen", NULL, NULL, NULL}, + {"src/Members.cs", "Gen{U}", "Members.Box.Doc", "Members.Box.Gen", NULL, NULL, NULL}, + {"src/Members.cs", "OnlyGen", "Members.Box.Doc", "Members.Box.OnlyGen", NULL, NULL, NULL}, + /* ... with a parameter list too: both take an int, and that is no ambiguity */ + {"src/Members.cs", "Dup(int)", "Members.Box.Doc", "Members.Box.Dup", NULL, NULL, NULL}, + {"src/Members.cs", "Dup{U}(int)", "Members.Box.Doc", "Members.Box.Dup", NULL, NULL, NULL}, + {"src/Members.cs", "Dup", "Members.Box.Doc", "Members.Box.Dup", NULL, NULL, NULL}, + /* a type variable is its position, whatever it is called */ + {"src/Members.cs", "Map{A, B}(A, B)", "Members.Box.Doc", "Members.Box.Map", NULL, NULL, + NULL}, + {"src/Members.cs", "Map{A, B}(B, A)", "Members.Box.Doc", NULL, "missing", NULL, NULL}, + /* the declared parameter type is the same written text */ + {"src/Members.cs", "Loose(Thing{int})", "Members.Box.Doc", "Members.Box.Loose", NULL, NULL, + NULL}, + {"src/Members.cs", "Loose(Other)", "Members.Box.Doc", NULL, "missing", NULL, NULL}, + /* an accessor names its property or its event, nothing else */ + {"src/Members.cs", "get_Value", "Members.Box.Doc", "Members.Box.Value", NULL, NULL, NULL}, + {"src/Members.cs", "get_field", "Members.Box.Doc", NULL, "missing", "Members.Box.field", + NULL}, + {"src/Members.cs", "add_Changed", "Members.Box.Doc", NULL, "graph_gap", NULL, NULL}, + /* a method named Item is a method; without one, `Item(int)` is how + * a doc ID spells the indexer */ + {"src/Members.cs", "Item(string)", "Members.Box.Doc", "Members.Box.Item", NULL, NULL, NULL}, + {"src/Members.cs", "Item(int)", "Members.Box.Doc", NULL, "missing", NULL, NULL}, + /* declared operators and indexers have no node: a gap */ + {"src/Members.cs", "this[int]", "Members.Box.Doc", NULL, "graph_gap", NULL, NULL}, + {"src/Members.cs", "operator +", "Members.Box.Doc", NULL, "graph_gap", NULL, NULL}, + /* ... and one the type does not declare is not there */ + {"src/Members.cs", "operator -", "Members.Box.Doc", NULL, "missing", NULL, NULL}, + {"src/Plain.cs", "this[int]", "Plain.Plain.Doc", NULL, "missing", NULL, NULL}, + {"src/Plain.cs", "operator +", "Plain.Plain.Doc", NULL, "missing", NULL, NULL}, + {"src/Plain.cs", "Plain.this[int]", "Plain.Plain.Doc", NULL, "missing", NULL, NULL}, + {"src/Plain.cs", "Box{T}.this[int]", "Plain.Plain.Doc", NULL, "graph_gap", NULL, NULL}, + {"src/Plain.cs", "Box{T}.operator +", "Plain.Plain.Doc", NULL, "graph_gap", NULL, NULL}, + /* through the type's name too: a property has accessors, a field has none */ + {"src/Plain.cs", "Box{T}.get_Value", "Plain.Plain.Doc", "Members.Box.Value", NULL, NULL, + NULL}, + {"src/Plain.cs", "Box{T}.get_field", "Plain.Plain.Doc", NULL, "missing", + "Members.Box.field", NULL}, + }; + ASSERT_EQ(dm_check_repo("members", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + /* `Item(int)` is written twice in Plain.cs with two outcomes, so it is + * asked per definition: nothing declared (missing) and an indexer (gap) */ + static const dm_source_t item_files[] = { + {"src/NoIndexer.cs", "namespace Acme.M\n" + "{\n" + " public class NoIndexer\n" + " {\n" + " /// \n" + " public void Doc() { }\n" + " }\n" + "}\n"}, + {"src/Indexer.cs", "namespace Acme.M\n" + "{\n" + " public class Indexer\n" + " {\n" + " public int this[int i] { get { return i; } }\n" + "\n" + " /// \n" + " public void Doc() { }\n" + " }\n" + "}\n"}, + }; + static const dm_want_t item_wants[] = { + {"src/NoIndexer.cs", "Item(int)", "NoIndexer.NoIndexer.Doc", NULL, "missing", NULL, NULL}, + {"src/Indexer.cs", "Item(int)", "Indexer.Indexer.Doc", NULL, "graph_gap", NULL, NULL}, + }; + ASSERT_EQ( + dm_check_repo("item", item_files, DM_COUNT(item_files), item_wants, DM_COUNT(item_wants)), + 0); + PASS(); +} + +/* Constructors: a parameter list on a type's name, however the type is + * named; `Foo.Foo`; and `Foo(int)` written inside a generic `Foo`. The + * constructors no source writes are declared and have no node. */ +TEST(doc_mentions_cs_constructors) { + static const dm_source_t files[] = { + {"src/Ctors.cs", + "namespace Acme.C\n" + "{\n" + " public class Widget { }\n" + " public class Gadget\n" + " {\n" + " public Gadget(int size) { }\n" + " public class Part { public Part(string s) { } }\n" + " }\n" + " public record Rec(int Width);\n" + " public class Prim(int seed) { }\n" + " public struct Point { public Point(int x) { } }\n" + " public interface IThing { }\n" + "\n" + " public class Gen\n" + " {\n" + " public Gen(T first) { }\n" + "\n" + " /// \n" + " /// \n" + " public void Doc() { }\n" + " }\n" + "\n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " public class Rows { }\n" + "\n" + " /// \n" + " public class Simple { }\n" + "\n" + " /// \n" + " public class Qualified { }\n" + "\n" + " /// \n" + " public class Nested { }\n" + "\n" + " /// \n" + " public class ByName { }\n" + "\n" + " /// \n" + " public class OfStruct { }\n" + "}\n"}, + {"src/Outside.cs", "namespace Acme.C\n" + "{\n" + " /// \n" + " public class Outside { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + /* a class that writes no constructor has one: declared, no node */ + {"src/Ctors.cs", "Widget()", "Ctors.Rows", NULL, "graph_gap", NULL, NULL}, + {"src/Ctors.cs", "Widget(int)", "Ctors.Rows", NULL, "missing", NULL, NULL}, + /* ... and a class that writes one has no other */ + {"src/Ctors.cs", "Gadget()", "Ctors.Rows", NULL, "missing", NULL, NULL}, + /* primary constructors and a record's copy constructor: declared, no node */ + {"src/Ctors.cs", "Rec(int)", "Ctors.Rows", NULL, "graph_gap", NULL, NULL}, + {"src/Ctors.cs", "Rec(Rec)", "Ctors.Rows", NULL, "graph_gap", NULL, NULL}, + {"src/Ctors.cs", "Prim(int)", "Ctors.Rows", NULL, "graph_gap", NULL, NULL}, + /* a struct keeps its parameterless constructor beside the one it writes */ + {"src/Ctors.cs", "Point()", "Ctors.Rows", NULL, "graph_gap", NULL, NULL}, + {"src/Ctors.cs", "Point.Point", "Ctors.Rows", NULL, "ambiguous", NULL, NULL}, + {"src/Ctors.cs", "IThing()", "Ctors.Rows", NULL, "missing", NULL, NULL}, + {"src/Ctors.cs", "Gadget(int)", "Ctors.Simple", "Ctors.Gadget.Gadget", NULL, NULL, NULL}, + /* the type may be named by any path */ + {"src/Ctors.cs", "Acme.C.Gadget(int)", "Ctors.Qualified", "Ctors.Gadget.Gadget", NULL, NULL, + DM_EXACT}, + {"src/Ctors.cs", "Gadget.Part(string)", "Ctors.Nested", "Ctors.Gadget.Part.Part", NULL, + NULL, NULL}, + {"src/Ctors.cs", "Gadget.Gadget", "Ctors.ByName", "Ctors.Gadget.Gadget", NULL, NULL, NULL}, + {"src/Ctors.cs", "Point(int)", "Ctors.OfStruct", "Ctors.Point.Point", NULL, NULL, NULL}, + /* inside Gen: `Gen(T)` is its constructor although no type `Gen` + * without type arguments is in scope; a bare `Gen` names nothing */ + {"src/Ctors.cs", "Gen(T)", "Ctors.Gen.Doc", "Ctors.Gen.Gen", NULL, NULL, NULL}, + {"src/Ctors.cs", "Gen", "Ctors.Gen.Doc", NULL, "missing", "Ctors.Gen", NULL}, + /* `Gen{T}.Gen` is the constructor from outside the type; written on the + * generic type itself, without parentheses, the compiler binds nothing */ + {"src/Outside.cs", "Gen{T}.Gen", "Outside.Outside", "Ctors.Gen.Gen", NULL, NULL, NULL}, + {"src/Ctors.cs", "Gen{T}.Gen", "Ctors.Gen.Doc", NULL, "missing", NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("ctors", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* The compiler looks up no inherited member in a cref: not through a class's + * base class, not through an interface's base interface, not through the + * roots every type has. A member is named through the type that declares it. */ +TEST(doc_mentions_cs_inherited_members) { + static const dm_source_t files[] = { + {"src/Sys.cs", + "namespace System\n" + "{\n" + " public class Object { public virtual string ToString() { return null; } }\n" + " public class ValueType { public string Kind() { return null; } }\n" + "}\n"}, + {"src/Inherit.cs", + "namespace Acme.I\n" + "{\n" + " public class Base\n" + " {\n" + " public void Run() { }\n" + " public void Put(int x) { }\n" + " public class Nested { }\n" + " }\n" + " public interface IBase { void Ping(); }\n" + " public interface IDerived : IBase { void Own(); }\n" + " public class Ping { }\n" + "\n" + " /// \n" + " public class Derived : Base, IBase\n" + " {\n" + " void IBase.Ping() { }\n" + " public void Put(string s) { }\n" + "\n" + " /// \n" + " /// \n" + " public void Doc() { }\n" + " }\n" + "\n" + " /// \n" + " public class Declared { }\n" + "\n" + " /// \n" + " public class ThroughInterface { }\n" + "\n" + " public record class RecC(int A);\n" + "\n" + " /// \n" + " /// \n" + " public class OfRecord { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + /* through the declaring type: bound */ + {"src/Inherit.cs", "Base.Run", "Inherit.Derived", "Inherit.Base.Run", NULL, NULL, NULL}, + /* a class finds no member of an interface it implements: `Ping` is + * the class of that name */ + {"src/Inherit.cs", "Ping", "Inherit.Derived", "Inherit.Ping", NULL, "Inherit.IBase.Ping", + NULL}, + /* by a simple name, and through the derived type's name: not bound */ + {"src/Inherit.cs", "Run", "Inherit.Derived.Doc", NULL, "missing", "Inherit.Base.Run", NULL}, + {"src/Inherit.cs", "Derived.Run", "Inherit.Derived.Doc", NULL, "missing", NULL, NULL}, + {"src/Inherit.cs", "Nested", "Inherit.Derived.Doc", NULL, "missing", "Inherit.Base.Nested", + NULL}, + /* the type has `Put`; the overload asked for is its base class's */ + {"src/Inherit.cs", "Put(int)", "Inherit.Derived.Doc", NULL, "missing", "Inherit.Base.Put", + NULL}, + {"src/Inherit.cs", "ToString", "Inherit.Derived.Doc", NULL, "missing", + "Sys.Object.ToString", NULL}, + /* an interface's own members, and not its base interface's */ + {"src/Inherit.cs", "IDerived.Own", "Inherit.Declared", "Inherit.IDerived.Own", NULL, NULL, + NULL}, + {"src/Inherit.cs", "IBase.Ping", "Inherit.Declared", "Inherit.IBase.Ping", NULL, NULL, + NULL}, + {"src/Inherit.cs", "IDerived.Ping", "Inherit.ThroughInterface", NULL, "missing", + "Inherit.IBase.Ping", NULL}, + /* a record class is no value type, and its roots are not searched */ + {"src/Inherit.cs", "RecC.ToString", "Inherit.OfRecord", NULL, "missing", + "Sys.Object.ToString", NULL}, + {"src/Inherit.cs", "RecC.Kind", "Inherit.OfRecord", NULL, "missing", "Sys.ValueType.Kind", + NULL}, + {"src/Inherit.cs", "RecC.A", "Inherit.OfRecord", "Inherit.RecC.A", NULL, NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("inherit", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* Doc IDs: a full name from the global namespace, the letter saying what it + * names. `M:` without parentheses is the overload without parameters; a type + * parameter is written as its position. */ +TEST(doc_mentions_cs_doc_ids) { + static const dm_source_t files[] = { + {"src/Ids.cs", + "namespace Acme.D\n" + "{\n" + " public class Outer\n" + " {\n" + " public class Inner { }\n" + " public void Go() { }\n" + " public void Go(int n) { }\n" + " public void Take(T item, int n) { }\n" + " public void Take(int n, T item) { }\n" + " public int Prop { get; set; }\n" + " public int Fld;\n" + " public event System.Action Evt;\n" + " public Outer() { }\n" + " public Outer(string s) { }\n" + " static Outer() { }\n" + " public int this[int i] { get { return i; } }\n" + " public static Outer operator +(Outer a, Outer b) { return a; }\n" + " }\n" + " public class Box\n" + " {\n" + " public void Put(T item) { }\n" + " public void Put(int n) { }\n" + " public class In { public void Both(T a, U b) { } }\n" + " }\n" + "\n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " public class Rows { }\n" + "\n" + " /// \n" + " public class OfType { }\n" + " /// \n" + " public class OfNested { }\n" + " /// \n" + " public class OfGeneric { }\n" + " /// \n" + " public class OfMethod { }\n" + " /// \n" + " public class OfOverload { }\n" + " /// \n" + " public class OfGenericMethod { }\n" + " /// \n" + " public class OfCtor { }\n" + " /// \n" + " public class OfStaticCtor { }\n" + " /// \n" + " public class OfProperty { }\n" + " /// \n" + " public class OfField { }\n" + " /// \n" + " public class OfSlot { }\n" + " /// \n" + " public class OfNestedSlots { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"src/Ids.cs", "T:Acme.D.Outer", "Ids.OfType", "Ids.Outer", NULL, NULL, DM_EXACT}, + {"src/Ids.cs", "T:Acme.D.Outer.Inner", "Ids.OfNested", "Ids.Outer.Inner", NULL, NULL, + DM_EXACT}, + {"src/Ids.cs", "T:Acme.D.Box`1", "Ids.OfGeneric", "Ids.Box", NULL, NULL, DM_EXACT}, + /* the namespace is the repository's: the type is not there */ + {"src/Ids.cs", "T:Acme.D.Nope", "Ids.Rows", NULL, "missing", NULL, NULL}, + {"src/Ids.cs", "T:Acme.D.Box", "Ids.Rows", NULL, "missing", "Ids.Box", NULL}, + {"src/Ids.cs", "T:Acme.D", "Ids.Rows", NULL, "missing", NULL, NULL}, + /* ... a namespace the repository does not declare is outside */ + {"src/Ids.cs", "T:Other.Thing", "Ids.Rows", NULL, "external", NULL, NULL}, + {"src/Ids.cs", "N:Acme.D", "Ids.Rows", NULL, "graph_gap", NULL, NULL}, + {"src/Ids.cs", "N:Acme.Nope", "Ids.Rows", NULL, "external", NULL, NULL}, + /* `M:` without parentheses: the overload without parameters, not the group */ + {"src/Ids.cs", "M:Acme.D.Outer.Go", "Ids.OfMethod", "Ids.Outer.Go", NULL, NULL, DM_EXACT}, + {"src/Ids.cs", "M:Acme.D.Outer.Go(System.Int32)", "Ids.OfOverload", "Ids.Outer.Go", NULL, + NULL, NULL}, + {"src/Ids.cs", "M:Acme.D.Outer.Go(System.String)", "Ids.Rows", NULL, "missing", NULL, NULL}, + /* a method's type parameter by its position: the overload it is in */ + {"src/Ids.cs", "M:Acme.D.Outer.Take``1(``0,System.Int32)", "Ids.OfGenericMethod", + "Ids.Outer.Take", NULL, NULL, NULL}, + {"src/Ids.cs", "M:Acme.D.Outer.#ctor(System.String)", "Ids.OfCtor", "Ids.Outer.Outer", NULL, + NULL, NULL}, + {"src/Ids.cs", "M:Acme.D.Outer.#cctor", "Ids.OfStaticCtor", "Ids.Outer.Outer", NULL, NULL, + NULL}, + /* the letter names the kind */ + {"src/Ids.cs", "P:Acme.D.Outer.Prop", "Ids.OfProperty", "Ids.Outer.Prop", NULL, NULL, NULL}, + {"src/Ids.cs", "F:Acme.D.Outer.Fld", "Ids.OfField", "Ids.Outer.Fld", NULL, NULL, NULL}, + {"src/Ids.cs", "F:Acme.D.Outer.Prop", "Ids.Rows", NULL, "missing", "Ids.Outer.Prop", NULL}, + {"src/Ids.cs", "E:Acme.D.Outer.Evt", "Ids.Rows", NULL, "graph_gap", NULL, NULL}, + {"src/Ids.cs", "P:Acme.D.Outer.Item(System.Int32)", "Ids.Rows", NULL, "graph_gap", NULL, + NULL}, + /* a type's parameter by its position, counted on from its outer types */ + {"src/Ids.cs", "M:Acme.D.Box`1.Put(`0)", "Ids.OfSlot", "Ids.Box.Put", NULL, NULL, NULL}, + {"src/Ids.cs", "M:Acme.D.Box`1.In`1.Both(`0,`1)", "Ids.OfNestedSlots", "Ids.Box.In.Both", + NULL, NULL, NULL}, + {"src/Ids.cs", "M:Acme.D.Box`1.In`1.Both(`1,`0)", "Ids.Rows", NULL, "missing", NULL, NULL}, + {"src/Ids.cs", "M:Acme.D.Outer.op_Addition(Acme.D.Outer,Acme.D.Outer)", "Ids.Rows", NULL, + "graph_gap", NULL, NULL}, + {"src/Ids.cs", "M:Acme.D.Outer.op_Subtraction(Acme.D.Outer,Acme.D.Outer)", "Ids.Rows", NULL, + "missing", NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("ids", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* The reasons. `external` is what the repository's own structure says is + * outside: a namespace it does not declare. No name is outside because it is + * on a list -- a repository that IS Xunit declares Xunit. A reference too + * long to be read is not judged by its beginning; a product reference never + * binds what only a test declares, and does not count it either. */ +TEST(doc_mentions_cs_reasons) { + /* `Known.M(TTT...T, int)`: cut at a buffer's end it would read as a call + * with ONE parameter of a type nothing is known about, which M(int) fits */ + char longref[1300]; + char long_cs[1600]; + memset(longref, 'T', sizeof(longref)); + memcpy(longref, "Known.M(", 8); + snprintf(longref + 1100, sizeof(longref) - 1100, ", int)"); + snprintf(long_cs, sizeof(long_cs), + "namespace Acme.R\n" + "{\n" + " public class Known { public void M(int a) { } }\n" + "\n" + " /// \n" + " public class TooLong { }\n" + "}\n", + longref); + const dm_source_t files[] = { + {"src/Declared.cs", "namespace Xunit.Sdk\n" + "{\n" + " public class Runner { }\n" + "}\n" + "namespace Acme.R\n" + "{\n" + " public class Here { }\n" + "}\n"}, + {"src/Structure.cs", + "using Xunit.Sdk;\n" + "\n" + "namespace Acme.R\n" + "{\n" + " /// \n" + " /// \n" + " /// \n" + " public class Structure { }\n" + "}\n"}, + {"src/OpenScope.cs", "using Moq;\n" + "\n" + "namespace Acme.R\n" + "{\n" + " /// \n" + " public class OpenScope { }\n" + "}\n"}, + /* the repository adds to System and to a namespace under it: polyfills */ + {"src/Polyfill.cs", "namespace System\n" + "{\n" + " public class Polyfill { }\n" + "}\n" + "namespace System.Text.Json\n" + "{\n" + " public class Extra { }\n" + "}\n"}, + {"src/Standard.cs", + "using System.Text.Json;\n" + "\n" + "namespace Acme.R\n" + "{\n" + " /// \n" + " /// \n" + " /// \n" + " public class Standard { }\n" + "}\n"}, + {"src/Unicode.cs", "namespace Acme.R\n" + "{\n" + " public class Gr\xC3\xB6\xC3\x9F" + "e { public void L\xC3\xA4nge() { } }\n" + "\n" + " /// \n" + " /// \n" + " public class Unicode { }\n" + "}\n"}, + {"src/Long.cs", long_cs}, + {"src/Mix.cs", + "namespace Acme.R\n" + "{\n" + " public partial class Mix\n" + " {\n" + " public void Prod() { }\n" + " public void Over(string s) { }\n" + " }\n" + "\n" + " /// \n" + " /// \n" + " public class UsesMix { }\n" + "}\n"}, + {"src/tests/MixTests.cs", "namespace Acme.R\n" + "{\n" + " public partial class Mix\n" + " {\n" + " public void OnlyTest() { }\n" + " public void Over(int a) { }\n" + " }\n" + "\n" + " /// \n" + " public class FromTest { }\n" + "}\n"}, + }; + const dm_want_t wants[] = { + /* a namespace the repository declares: its missing names are missing */ + {"src/Structure.cs", "Xunit.Sdk.Nope", "Structure.Structure", NULL, "missing", NULL, NULL}, + {"src/Structure.cs", "Acme.R.Nope", "Structure.Structure", NULL, "missing", NULL, NULL}, + {"src/Structure.cs", "Xunit.Sdk.Runner", "Structure.Structure", "Declared.Runner", NULL, + NULL, DM_EXACT}, + /* ... one it only has namespaces under, and one it does not have: outside */ + {"src/Structure.cs", "Xunit.Assert", "Structure.Structure", NULL, "external", NULL, NULL}, + {"src/Structure.cs", "Outside.Lib.Thing", "Structure.Structure", NULL, "external", NULL, + NULL}, + /* every using of the file names a declared namespace: the scope is closed */ + {"src/Structure.cs", "NotAnywhere", "Structure.Structure", NULL, "missing", NULL, NULL}, + /* ... a using of a namespace the repository does not declare opens it */ + {"src/OpenScope.cs", "NotAnywhere", "OpenScope.OpenScope", NULL, "external", NULL, NULL}, + /* System is the standard library's whoever adds to it: what the + * repository does not have there is outside, though it declares the + * namespace -- by a full name, by a doc ID, through a using */ + {"src/Standard.cs", "System.Nope", "Standard.Standard", NULL, "external", NULL, NULL}, + {"src/Standard.cs", "System.Text.Json.Nope", "Standard.Standard", NULL, "external", NULL, + NULL}, + {"src/Standard.cs", "System.Text.Nope", "Standard.Standard", NULL, "external", NULL, NULL}, + {"src/Standard.cs", "T:System.Nope2", "Standard.Standard", NULL, "external", NULL, NULL}, + {"src/Standard.cs", "NotInJson", "Standard.Standard", NULL, "external", NULL, NULL}, + /* ... and what it has there is its own */ + {"src/Standard.cs", "System.Polyfill", "Standard.Standard", "Polyfill.Polyfill", NULL, NULL, + DM_EXACT}, + /* identifiers are not ASCII only */ + {"src/Unicode.cs", + "Gr\xC3\xB6\xC3\x9F" + "e", + "Unicode.Unicode", + "Unicode.Gr\xC3\xB6\xC3\x9F" + "e", + NULL, NULL, NULL}, + {"src/Unicode.cs", + "Gr\xC3\xB6\xC3\x9F" + "e.L\xC3\xA4nge", + "Unicode.Unicode", + "Unicode.Gr\xC3\xB6\xC3\x9F" + "e.L\xC3\xA4nge", + NULL, NULL, NULL}, + {"src/Unicode.cs", + "Gr\xC3\xB6\xC3\x9F" + "e.Nicht", + "Unicode.Unicode", NULL, "missing", NULL, NULL}, + /* too long to be read: never resolved by what is left after a cut */ + {"src/Long.cs", longref, "Long.TooLong", NULL, "unparseable", "Long.Known.M", NULL}, + /* only the test part of the type declares OnlyTest */ + {"src/Mix.cs", "Mix.OnlyTest", "Mix.UsesMix", NULL, "test_only_target", + "MixTests.Mix.OnlyTest", NULL}, + /* ... and its Over(int) is no overload for product code: one Over */ + {"src/Mix.cs", "Mix.Over", "Mix.UsesMix", "Mix.Mix.Over", NULL, "MixTests.Mix.Over", NULL}, + {"src/Mix.cs", "Mix.Prod", "Mix.UsesMix", "Mix.Mix.Prod", NULL, NULL, NULL}, + /* test code sees both parts: an overload group */ + {"src/tests/MixTests.cs", "Mix.Over", "MixTests.FromTest", NULL, "ambiguous", NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("reasons", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* The text around a reference: CRLF line ends and a byte-order mark change + * nothing; a reference inside an XML comment or a CDATA section of the doc + * text is no reference. */ +TEST(doc_mentions_cs_text_forms) { + const char *crlf = "\xEF\xBB\xBFnamespace Acme.T\r\n" /* 1 */ + "{\r\n" /* 2 */ + " public class Target { }\r\n" /* 3 */ + "\r\n" /* 4 */ + " /// First \r\n" /* 5 */ + " /// and .\r\n" /* 7 */ + " public class Uses\r\n" /* 8 */ + " {\r\n" /* 9 */ + " public int Field;\r\n" /* 10 */ + " }\r\n" /* 11 */ + "}\r\n"; /* 12 */ + CBMFileResult *r = dm_extract(crlf, CBM_LANG_CSHARP, "Crlf.cs"); + ASSERT_NOT_NULL(r); + const CBMDocLink *first = dm_find_token(r, "Target"); + ASSERT_NOT_NULL(first); + ASSERT_EQ(first->line, 5); + ASSERT_EQ(first->def_line, 8); + const CBMDocLink *second = dm_find_token(r, "Target.Nope"); + ASSERT_NOT_NULL(second); + ASSERT_EQ(second->line, 6); + ASSERT_NOT_NULL(r->doc_scope); + ASSERT_NOT_NULL(strstr(r->doc_scope, "R\t1\t0\t1\t12\tAcme.T\n")); + ASSERT_NOT_NULL(strstr(r->doc_scope, "T\t1\t3\t3\tc\t-\tTarget\t\t\n")); + ASSERT_NOT_NULL(strstr(r->doc_scope, "T\t1\t8\t11\tc\t-\tUses\t\t\n")); + ASSERT_NOT_NULL(strstr(r->doc_scope, "M\t10\tv\t0\t1\tField\t\t-\n")); + ASSERT_NULL(strchr(r->doc_scope, '\r')); + cbm_free_result(r); + + const char *hidden = "namespace Acme.T\n" + "{\n" + " /// \n" + " /// \n" + " /// ]]>\n" + " /// \n" + " /// \n" + " /// \n" + " public class Hidden { }\n" + "}\n"; + r = dm_extract(hidden, CBM_LANG_CSHARP, "Hidden.cs"); + ASSERT_NOT_NULL(r); + ASSERT_NULL(dm_find_token(r, "InComment")); + ASSERT_NULL(dm_find_token(r, "InCdata")); + ASSERT_NULL(dm_find_token(r, "InLongComment")); + ASSERT_NOT_NULL(dm_find_token(r, "Shown")); + /* markup inside is markup: the compiler binds a cref there too */ + ASSERT_NOT_NULL(dm_find_token(r, "InCode")); + cbm_free_result(r); + + /* ... and end to end, through the pipeline */ + const dm_source_t files[] = {{"src/Crlf.cs", crlf}, {"src/Hidden.cs", hidden}}; + static const dm_want_t wants[] = { + {"src/Crlf.cs", "Target", "Crlf.Uses", "Crlf.Target", NULL, NULL, NULL}, + {"src/Crlf.cs", "Target.Nope", "Crlf.Uses", NULL, "missing", NULL, NULL}, + /* what a comment or a CDATA section holds is no reference: no row */ + {"src/Hidden.cs", "InComment", "Hidden.Hidden", NULL, NULL, NULL, NULL}, + {"src/Hidden.cs", "InCdata", "Hidden.Hidden", NULL, NULL, NULL, NULL}, + {"src/Hidden.cs", "InLongComment", "Hidden.Hidden", NULL, NULL, NULL, NULL}, + {"src/Hidden.cs", "Shown", "Hidden.Hidden", NULL, "missing", NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("crlf", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* ── cost and limits ─────────────────────────────────────────────── */ + +enum { DM_COST_REFS = 40, DM_COST_TEXT = 16384 }; + +/* What the C# resolver looks at for DM_COST_REFS simple names that are not in + * scope, documented `types` types deep in a namespace of `depth` segments, in + * a file with `usings` using directives. `declared`: the names are types of + * a namespace the file does not import (else no file declares them). 0 when + * the repository cannot be indexed. */ +static uint64_t dm_lookup_work(int depth, int usings, int types, bool declared) { + char *src = malloc(DM_COST_TEXT); + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_cost_XXXXXX"); + if (!src || !cbm_mkdtemp(tmp)) { + free(src); + return 0; + } + /* the namespaces the usings name, and (when declared) the names asked + * for, in a namespace nothing imports */ + size_t w = 0; + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "namespace Far\n{\n"); + for (int i = 0; declared && i < DM_COST_REFS; i++) { + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "public class Nowhere%d { }\n", i); + } + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "}\n"); + for (int i = 0; i < usings; i++) { + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, + "namespace In.U%d { public class Other%d { } }\n", i, i); + } + th_write_file(TH_PATH(tmp, "src/Far.cs"), src); + w = 0; + for (int i = 0; i < usings; i++) { + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "using In.U%d;\n", i); + } + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "namespace "); + for (int i = 0; i < depth; i++) { + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "%sN%d", i ? "." : "", i); + } + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "\n{\n"); + for (int i = 0; i + 1 < types; i++) { + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "public class T%d {\n", i); + } + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "/// "); + for (int i = 0; i < DM_COST_REFS; i++) { + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, " ", i); + } + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "\npublic class Leaf { }\n"); + for (int i = 0; i < types; i++) { + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "}\n"); + } + th_write_file(TH_PATH(tmp, "src/Deep.cs"), src); + free(src); + char db[512]; + snprintf(db, sizeof(db), "%s/cost.db", tmp); + cbm_doclink_cs_test_work_reset(); + uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_work() : 0; + int rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"); + dm_unlink_db(db); + th_rmtree(tmp); + return rows == DM_COST_REFS ? work : 0; /* every name was looked up, and found nowhere */ +} + +/* The same for DM_COST_REFS references, written in product code, to an + * operator that only test code declares -- `overloads` times over. */ +static uint64_t dm_operator_work(int overloads) { + char *src = malloc(DM_COST_TEXT * 4); + size_t cap = DM_COST_TEXT * 4; + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_opcost_XXXXXX"); + if (!src || !cbm_mkdtemp(tmp)) { + free(src); + return 0; + } + th_write_file(TH_PATH(tmp, "src/Vec.cs"), + "namespace Geo\n{\n public partial struct Vec { }\n}\n"); + size_t w = + (size_t)snprintf(src, cap, "namespace Geo\n{\n public partial struct Vec\n {\n"); + for (int i = 0; i < overloads; i++) { + w += (size_t)snprintf( + src + w, cap - w, + " public static Vec operator +(Vec a, P%03d b) { return a; }\n", i); + } + snprintf(src + w, cap - w, " }\n}\n"); + th_write_file(TH_PATH(tmp, "tests/VecOps.cs"), src); + w = (size_t)snprintf(src, cap, "namespace Geo\n{\n /// "); + for (int i = 0; i < DM_COST_REFS; i++) { + w += (size_t)snprintf(src + w, cap - w, " ", i); + } + snprintf(src + w, cap - w, "\n public class Uses { }\n}\n"); + th_write_file(TH_PATH(tmp, "src/Uses.cs"), src); + free(src); + char db[512]; + snprintf(db, sizeof(db), "%s/cost.db", tmp); + cbm_doclink_cs_test_work_reset(); + uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_work() : 0; + int rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'missing'"); + dm_unlink_db(db); + th_rmtree(tmp); + return rows == DM_COST_REFS ? work : 0; /* product code sees none of them */ +} + +/* The same for DM_COST_REFS references `Big.M(int)` to a method with 100 + * overloads of `params` parameters each: none takes one parameter. */ +static uint64_t dm_signature_work(int params) { + char *src = malloc(DM_COST_TEXT * 4); + size_t cap = DM_COST_TEXT * 4; + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_sigcost_XXXXXX"); + if (!src || !cbm_mkdtemp(tmp)) { + free(src); + return 0; + } + size_t w = (size_t)snprintf(src, cap, "namespace Sig\n{\n public class Big\n {\n"); + for (int i = 0; i < 100; i++) { + w += (size_t)snprintf(src + w, cap - w, " public void M(P%03d a0", i); + for (int p = 1; p < params; p++) { + w += (size_t)snprintf(src + w, cap - w, ", int a%d", p); + } + w += (size_t)snprintf(src + w, cap - w, ") { }\n"); + } + w += (size_t)snprintf(src + w, cap - w, " }\n\n /// "); + for (int i = 0; i < DM_COST_REFS; i++) { + w += (size_t)snprintf(src + w, cap - w, " ", i); + } + snprintf(src + w, cap - w, "\n public class Uses { }\n}\n"); + th_write_file(TH_PATH(tmp, "src/Big.cs"), src); + free(src); + char db[512]; + snprintf(db, sizeof(db), "%s/cost.db", tmp); + cbm_doclink_cs_test_work_reset(); + uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_work() : 0; + int rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'missing'"); + dm_unlink_db(db); + th_rmtree(tmp); + return rows == DM_COST_REFS ? work : 0; +} + +/* The same for DM_COST_REFS references in the documentation of a method that + * shares its line with `members` - 1 other methods. */ +static uint64_t dm_line_work(int members) { + char *src = malloc(DM_COST_TEXT * 2); + size_t cap = DM_COST_TEXT * 2; + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_linecost_XXXXXX"); + if (!src || !cbm_mkdtemp(tmp)) { + free(src); + return 0; + } + size_t w = (size_t)snprintf( + src, cap, "namespace Line\n{\n public class Wide\n {\n /// "); + for (int i = 0; i < DM_COST_REFS; i++) { + w += (size_t)snprintf(src + w, cap - w, " ", i); + } + w += (size_t)snprintf(src + w, cap - w, "\n "); + for (int i = 0; i < members; i++) { + w += (size_t)snprintf(src + w, cap - w, " public void M%d() { }", i); + } + snprintf(src + w, cap - w, "\n }\n}\n"); + th_write_file(TH_PATH(tmp, "src/Wide.cs"), src); + free(src); + char db[512]; + snprintf(db, sizeof(db), "%s/cost.db", tmp); + cbm_doclink_cs_test_work_reset(); + uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_work() : 0; + int rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"); + dm_unlink_db(db); + th_rmtree(tmp); + return rows == DM_COST_REFS ? work : 0; +} + +enum { DM_TREE_FILES = 120 }; + +/* What the resolver looks at to find the project of DM_TREE_FILES files that + * stand `depth` directories deep, in a repository without a project file. */ +static uint64_t dm_tree_work(int depth) { + char tmp[256]; + char dir[256]; + char rel[320]; + char text[192]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_tree_XXXXXX"); + if (depth * 2 >= (int)sizeof(dir) || !cbm_mkdtemp(tmp)) { + return 0; + } + size_t d = 0; + dir[0] = '\0'; + for (int i = 0; i < depth; i++) { + d += (size_t)snprintf(dir + d, sizeof(dir) - d, "d/"); + } + for (int i = 0; i < DM_TREE_FILES; i++) { + snprintf(rel, sizeof(rel), "%sF%03d.cs", dir, i); + snprintf(text, sizeof(text), "namespace Deep\n{\n%s public class F%03d { }\n}\n", + i == 0 ? " /// \n" : "", i); + th_write_file(TH_PATH(tmp, rel), text); + } + char db[512]; + snprintf(db, sizeof(db), "%s/cost.db", tmp); + cbm_doclink_cs_test_work_reset(); + uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_work() : 0; + int rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"); + dm_unlink_db(db); + th_rmtree(tmp); + return rows == 1 ? work : 0; +} + +/* Every import kind is queried with names absent everywhere or declared only + * outside the imported scopes. Count visits/probes, never elapsed time. */ +typedef struct { + uint64_t lookup, build; +} dm_import_work_t; + +static dm_import_work_t dm_import_lookup_work(int imports, char kind, bool global, bool declared) { + enum { CAP = DM_COST_TEXT * 4 }; + char *defs = malloc(CAP); + char *directives = malloc(CAP); + char *use = malloc(CAP); + char tmp[256] = "/tmp/cbm_dm_importcost_XXXXXX"; + if (!defs || !directives || !use || !cbm_mkdtemp(tmp)) { + free(defs); + free(directives); + free(use); + return (dm_import_work_t){0}; + } + size_t d = 0, u = 0; + for (int i = 0; i < imports; i++) { + d += (size_t)snprintf(defs + d, CAP - d, + "namespace In.U%d { public static class Other%d { " + "public static int Member%d; } }\n", + i, i, i); + if (kind == 'n') { + u += (size_t)snprintf(directives + u, CAP - u, "%susing In.U%d;\n", + global ? "global " : "", i); + } else if (kind == 's') { + u += (size_t)snprintf(directives + u, CAP - u, "%susing static In.U%d.Other%d;\n", + global ? "global " : "", i, i); + } else { + u += (size_t)snprintf(directives + u, CAP - u, "%susing Alias%d = In.U%d.Other%d;\n", + global ? "global " : "", i, i, i); + } + } + d += (size_t)snprintf(defs + d, CAP - d, "namespace Far {\n"); + for (int i = 0; declared && i < DM_COST_REFS; i++) { + d += (size_t)snprintf(defs + d, CAP - d, "public class Nowhere%d { }\n", i); + } + snprintf(defs + d, CAP - d, "}\n"); + size_t w = (size_t)snprintf(use, CAP, "%snamespace App {\n/// ", global ? "" : directives); + for (int i = 0; i < DM_COST_REFS; i++) { + w += (size_t)snprintf(use + w, CAP - w, " ", i); + } + snprintf(use + w, CAP - w, "\npublic class Use { }\n}\n"); + th_write_file(TH_PATH(tmp, "Defs.cs"), defs); + th_write_file(TH_PATH(tmp, "Use.cs"), use); + th_write_file(TH_PATH(tmp, "App.csproj"), DM_EMPTY_PROJECT); + if (global) { + th_write_file(TH_PATH(tmp, "Globals.cs"), directives); + } + free(defs); + free(directives); + free(use); + char db[512]; + snprintf(db, sizeof(db), "%s/cost.db", tmp); + cbm_doclink_cs_test_work_reset(); + dm_import_work_t work = {0}; + if (dm_index(tmp, db, NULL) == 0) { + work.lookup = cbm_doclink_cs_test_work(); + work.build = cbm_doclink_cs_test_index_work(); + } + int rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"); + dm_unlink_db(db); + th_rmtree(tmp); + return rows == DM_COST_REFS ? work : (dm_import_work_t){0}; +} + +/* Many repository declarations of a name must not burden a scope importing + * only one of them. Static members have distinct graph owners in this file. */ +static uint64_t dm_common_import_work(int declarations) { + enum { CAP = DM_COST_TEXT * 4 }; + char *defs = malloc(CAP); + char *use = malloc(CAP); + char tmp[256] = "/tmp/cbm_dm_commonimport_XXXXXX"; + if (!defs || !use || !cbm_mkdtemp(tmp)) { + free(defs); + free(use); + return 0; + } + size_t d = (size_t)snprintf(defs, CAP, "namespace In {\n"); + for (int i = 0; i < declarations; i++) { + d += (size_t)snprintf(defs + d, CAP - d, + "public static class C%d { public static int Common; }\n", i); + } + snprintf(defs + d, CAP - d, "}\n"); + size_t u = (size_t)snprintf(use, CAP, "using static In.C0;\nnamespace App {\n/// "); + for (int i = 0; i < DM_COST_REFS; i++) { + u += (size_t)snprintf(use + u, CAP - u, " "); + } + snprintf(use + u, CAP - u, "\npublic class Use { }\n}\n"); + th_write_file(TH_PATH(tmp, "Defs.cs"), defs); + th_write_file(TH_PATH(tmp, "Use.cs"), use); + free(defs); + free(use); + char db[512], props[512]; + snprintf(db, sizeof(db), "%s/cost.db", tmp); + cbm_doclink_cs_test_work_reset(); + uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_work() : 0; + int edges = 0; + dm_edge(db, "Use.Use", "Defs.C0.Common", props, sizeof(props), &edges); + int rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"); + dm_unlink_db(db); + th_rmtree(tmp); + return edges == 1 && rows == 0 ? work : 0; +} + +TEST(doc_mentions_cs_import_lookup_cost) { + bool bounded = true; + static const char kinds[] = {'n', 'a', 's'}; + for (int k = 0; k < 3; k++) { + for (int global = 0; global < 2; global++) { + for (int declared = 0; declared < 2; declared++) { + dm_import_work_t small = + dm_import_lookup_work(60, kinds[k], global != 0, declared != 0); + dm_import_work_t large = + dm_import_lookup_work(120, kinds[k], global != 0, declared != 0); + double ratio = small.lookup ? (double)large.lookup / (double)small.lookup : 0.0; + double build_ratio = small.build ? (double)large.build / (double)small.build : 0.0; + printf(" import lookup kind=%c global=%d declared=%d: %llu -> %llu, %.2f; " + "index build %llu -> %llu, %.2f\n", + kinds[k], global, declared, (unsigned long long)small.lookup, + (unsigned long long)large.lookup, ratio, (unsigned long long)small.build, + (unsigned long long)large.build, build_ratio); + bounded = bounded && small.lookup >= DM_COST_REFS && ratio >= 0.8 && ratio <= 1.4 && + small.build > 0 && build_ratio >= 0.8 && build_ratio <= 3.0; + } + } + } + uint64_t small = dm_common_import_work(64); + uint64_t large = dm_common_import_work(128); + double ratio = small ? (double)large / (double)small : 0.0; + printf(" common name, one import: %llu -> %llu, %.2f\n", (unsigned long long)small, + (unsigned long long)large, ratio); + bounded = bounded && small >= DM_COST_REFS && ratio >= 0.8 && ratio <= 1.4; + ASSERT_TRUE(bounded); + PASS(); +} + +/* Parent/global imports are ready for a child's directives. The child's own + * directives stay excluded while those targets are resolved. */ +TEST(doc_mentions_cs_import_stage_aliases) { + static const dm_source_t files[] = { + {"App.csproj", DM_EMPTY_PROJECT}, + {"Defs.cs", "namespace Lib { public class Target { } public class Other { } " + "public static class Statics { public static int Value; } }\n"}, + {"Globals.cs", "global using Base = Lib;\nglobal using static Lib.Statics;\n"}, + {"Use.cs", "namespace App {\nusing Parent = Lib;\nnamespace Child {\n" + "using Pick = Parent.Target;\nusing Again = Base.Target;\n" + "/// \n" + "public class Use { }\n}\nnamespace Sibling {\n" + "using Parent = Absent;\nusing Decoy = Parent.Target;\n" + "/// \n" + "public class Mask { }\n}\n}\n"}, + {"Aliases.cs", "using Clash = Lib.Target;\nusing Z = Absent;\n" + "using Same = Lib.Target;\nusing Noise = Lib;\nusing Same = Lib.Other;\n" + "public class Clash { }\n" + "/// \n" + "public class Aliases { }\n"}, + }; + static const dm_want_t wants[] = { + {"Use.cs", "Pick", "Use.Use", "Defs.Target", NULL, NULL, DM_EXACT}, + {"Use.cs", "Again", "Use.Use", "Defs.Target", NULL, NULL, DM_EXACT}, + {"Use.cs", "Value", "Use.Use", "Defs.Statics.Value", NULL, NULL, NULL}, + {"Use.cs", "Decoy", "Use.Mask", "Defs.Target", NULL, NULL, DM_EXACT}, + {"Use.cs", "Parent", "Use.Mask", NULL, "external", NULL, NULL}, + {"Aliases.cs", "Clash", "Aliases.Aliases", NULL, "ambiguous", "Defs.Target", NULL}, + {"Aliases.cs", "Same", "Aliases.Aliases", NULL, "ambiguous", "Defs.Target", NULL}, + {"Aliases.cs", "Z", "Aliases.Aliases", NULL, "external", NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("importstage", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* Static imports see both parts of a type; unseen named parts still make an + * outsider's reference ambiguous. No unrelated directive may hide that fact. */ +TEST(doc_mentions_cs_import_static_parts) { + enum { CAP = DM_COST_TEXT * 4 }; + char *noise = malloc(CAP); + char *own = malloc(CAP); + char *outside = malloc(CAP); + ASSERT_NOT_NULL(noise); + ASSERT_NOT_NULL(own); + ASSERT_NOT_NULL(outside); + size_t n = 0, u = 0; + for (int i = 0; i < 40; i++) { + n += (size_t)snprintf(noise + n, CAP - n, + "namespace Noise.N%d { public static class K%d { " + "public static int Other%d; } }\n", + i, i, i); + u += (size_t)snprintf(own + u, CAP - u, "using static Noise.N%d.K%d;\n", i, i); + } + u += (size_t)snprintf(own + u, CAP - u, "using static Lib.Tool;\nusing static Lib.TestOnly;\n"); + memcpy(outside, own, u); + snprintf(own + u, CAP - u, + "/// " + "\npublic class Use { }\n"); + snprintf(outside + u, CAP - u, + "/// " + "\npublic class Outside { }\n"); + const dm_source_t files[] = { + {"shared/Core.cs", "namespace Lib { public partial class Tool { " + "public static int Common; public class Shared { } } }\n"}, + {"shared/Noise.cs", noise}, + {"tests/TestOnly.cs", "namespace Lib { public static class TestOnly { " + "public static int Hidden; } }\n"}, + {"one/One.csproj", DM_EMPTY_PROJECT}, + {"one/Part.cs", "namespace Lib { public partial class Tool { " + "public static int Own; public class OnlyOne { } } }\n"}, + {"one/Use.cs", own}, + {"two/Two.csproj", DM_EMPTY_PROJECT}, + {"two/Outside.cs", outside}, + }; + static const dm_want_t wants[] = { + {"one/Use.cs", "Common", "Use.Use", "Core.Tool.Common", NULL, NULL, NULL}, + {"one/Use.cs", "Shared", "Use.Use", "Core.Tool.Shared", NULL, NULL, NULL}, + {"one/Use.cs", "Own", "Use.Use", "Part.Tool.Own", NULL, NULL, NULL}, + {"one/Use.cs", "Hidden", "Use.Use", NULL, "test_only_target", "TestOnly.TestOnly.Hidden", + NULL}, + {"two/Outside.cs", "Common", "Outside.Outside", "Core.Tool.Common", NULL, NULL, NULL}, + {"two/Outside.cs", "OnlyOne", "Outside.Outside", NULL, "ambiguous", "Part.Tool.OnlyOne", + NULL}, + {"two/Outside.cs", "Own", "Outside.Outside", NULL, "ambiguous", "Part.Tool.Own", NULL}, + {"two/Outside.cs", "Gone", "Outside.Outside", NULL, "missing", NULL, NULL}, + }; + int bad = dm_check_repo("staticparts", files, DM_COUNT(files), wants, DM_COUNT(wants)); + free(noise); + free(own); + free(outside); + ASSERT_EQ(bad, 0); + PASS(); +} + +/* A failed candidate buffer must use the complete existing scan. */ +TEST(doc_mentions_cs_import_candidate_allocation) { + enum { CAP = DM_COST_TEXT * 4, IMPORTS = 128, MATCHES = 40 }; + char *defs = malloc(CAP); + char *use = malloc(CAP); + ASSERT_NOT_NULL(defs); + ASSERT_NOT_NULL(use); + size_t d = 0, u = 0; + for (int i = 0; i < IMPORTS; i++) { + d += (size_t)snprintf(defs + d, CAP - d, "namespace N%d { public class %s { } }\n", i, + i < MATCHES ? "Common" : "Other"); + u += (size_t)snprintf(use + u, CAP - u, "using N%d;\n", i); + } + snprintf(use + u, CAP - u, "/// \npublic class Use { }\n"); + const dm_source_t files[] = {{"Defs.cs", defs}, {"Use.cs", use}}; + static const dm_want_t wants[] = { + {"Use.cs", "Common", "Use.Use", NULL, "ambiguous", NULL, NULL}, + }; + bool ok = true; + for (int fail = 0; fail < 2; fail++) { + cbm_doclink_cs_test_fail_candidate_alloc(fail != 0); + int bad = dm_check_repo("importalloc", files, DM_COUNT(files), wants, DM_COUNT(wants)); + bool hit = cbm_doclink_cs_test_candidate_alloc_failed(); + cbm_doclink_cs_test_fail_candidate_alloc(false); + printf(" import candidate allocation: injected=%d consumed=%d failures=%d\n", fail, hit, + bad); + ok = ok && bad == 0 && hit == (fail != 0); + } + free(defs); + free(use); + ASSERT_TRUE(ok); + PASS(); +} + +/* Enclosing scopes cost one step each. Import lookup uses the queried name; + * unrelated directives add only index probes. Other fixtures retain their + * original overload, source-line, signature and project-directory contracts. */ +TEST(doc_mentions_cs_lookup_cost) { + struct { + const char *what; + uint64_t small; + uint64_t large; + double low; + double high; + } runs[] = { + {"namespace depth 16 -> 32", dm_lookup_work(16, 0, 1, true), dm_lookup_work(32, 0, 1, true), + 1.5, 2.5}, + {"type nesting 16 -> 32", dm_lookup_work(1, 0, 16, true), dm_lookup_work(1, 0, 32, true), + 1.5, 2.5}, + {"usings 60 -> 120", dm_lookup_work(1, 60, 1, true), dm_lookup_work(1, 120, 1, true), 0.9, + 1.4}, + {"usings 60 -> 120, names no file declares", dm_lookup_work(1, 60, 1, false), + dm_lookup_work(1, 120, 1, false), 0.9, 1.1}, + {"test-only operator overloads 100 -> 200", dm_operator_work(100), dm_operator_work(200), + 0.9, 1.1}, + {"members on the documented line 100 -> 200", dm_line_work(100), dm_line_work(200), 0.9, + 1.1}, + {"parameters of 100 overloads none of which fits 4 -> 8", dm_signature_work(4), + dm_signature_work(8), 0.9, 1.1}, + {"directory depth 6 -> 48", dm_tree_work(6), dm_tree_work(48), 0.9, 2.5}, + }; + for (size_t i = 0; i < sizeof(runs) / sizeof(runs[0]); i++) { + double ratio = runs[i].small ? (double)runs[i].large / (double)runs[i].small : 0.0; + printf(" %s: %llu -> %llu steps, %.2f times the work, want %.1f to %.1f\n", runs[i].what, + (unsigned long long)runs[i].small, (unsigned long long)runs[i].large, ratio, + runs[i].low, runs[i].high); + if (runs[i].small < DM_COST_REFS || ratio < runs[i].low || ratio > runs[i].high) { + FAIL("lookup cost"); + } + } + PASS(); +} + +enum { DM_MANY = 300, DM_MANY_TEXT = 65536 }; + +/* No limit decides silently. More usings than any fixed list holds are all + * asked. A type with more overloads of one name than a lookup compares still + * binds the overload a reference writes out; what would need the comparison + * is `ambiguous`, never a guess and never `missing`. The same for a project + * that holds more complete declarations of one type than a lookup compares + * (beside another project of the assembly that holds one): none of them is + * picked. */ +TEST(doc_mentions_cs_no_silent_limits) { + char *decls = malloc(DM_MANY_TEXT); + char *uses = malloc(DM_MANY_TEXT); + char *big = malloc(DM_MANY_TEXT); + char *flav = malloc(DM_MANY_TEXT); + ASSERT_NOT_NULL(decls); + ASSERT_NOT_NULL(uses); + ASSERT_NOT_NULL(big); + ASSERT_NOT_NULL(flav); + size_t d = 0; + size_t u = 0; + size_t b = 0; + size_t f = (size_t)snprintf(flav, DM_MANY_TEXT, "namespace Lib\n{\n"); + for (int i = 0; i < DM_MANY; i++) { + f += (size_t)snprintf(flav + f, DM_MANY_TEXT - f, " public class Flav { }\n"); + } + snprintf(flav + f, DM_MANY_TEXT - f, "}\n"); + for (int i = 0; i < DM_MANY; i++) { + d += (size_t)snprintf(decls + d, DM_MANY_TEXT - d, + "namespace Many.N%03d { public class C%03d { } }\n", i, i); + u += (size_t)snprintf(uses + u, DM_MANY_TEXT - u, "using Many.N%03d;\n", i); + } + snprintf(uses + u, DM_MANY_TEXT - u, + "namespace App\n" + "{\n" + " /// \n" + " public class Uses { }\n" + "}\n"); + b += (size_t)snprintf(big + b, DM_MANY_TEXT - b, + "namespace App\n{\n public class Big\n {\n"); + for (int i = 0; i < DM_MANY; i++) { + b += (size_t)snprintf(big + b, DM_MANY_TEXT - b, " public void M(P%03d a) { }\n", i); + } + snprintf(big + b, DM_MANY_TEXT - b, + " }\n" + "\n" + " /// \n" + " public class Written { }\n" + "\n" + " /// \n" + " public class Unwritten { }\n" + "}\n"); + const dm_source_t files[] = { + {"src/Decls.cs", decls}, + {"src/Uses.cs", uses}, + {"src/Big.cs", big}, + {"a/Lib.csproj", DM_EMPTY_PROJECT}, + {"a/Many.cs", flav}, + {"a/UsesA.cs", "namespace Lib\n" + "{\n" + " /// \n" + " public class UsesA { }\n" + "}\n"}, + {"b/Lib.csproj", DM_EMPTY_PROJECT}, + {"b/Flav.cs", "namespace Lib\n{\n public class Flav { }\n}\n"}, + {"b/UsesB.cs", "namespace Lib\n" + "{\n" + " /// \n" + " public class UsesB { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"src/Uses.cs", "C000", "Uses.Uses", "Decls.C000", NULL, NULL, NULL}, + {"src/Uses.cs", "C299", "Uses.Uses", "Decls.C299", NULL, NULL, NULL}, + {"src/Big.cs", "Big.M(P299)", "Big.Written", "Big.Big.M", NULL, NULL, NULL}, + {"src/Big.cs", "Big.M", "Big.Unwritten", NULL, "ambiguous", "Big.Big.M", NULL}, + {"src/Big.cs", "Big.M(Other)", "Big.Unwritten", NULL, "ambiguous", "Big.Big.M", NULL}, + /* more declarations in the own project than a lookup compares: none + * is picked -- and the other project of the assembly binds its own */ + {"a/UsesA.cs", "Flav", "UsesA.UsesA", NULL, "ambiguous", "Many.Flav", NULL}, + {"b/UsesB.cs", "Flav", "UsesB.UsesB", "b.Flav.Flav", NULL, "Many.Flav", NULL}, + }; + int bad = dm_check_repo("limits", files, DM_COUNT(files), wants, DM_COUNT(wants)); + free(decls); + free(uses); + free(big); + free(flav); + ASSERT_EQ(bad, 0); + PASS(); +} + +enum { DM_WIDE = 66 }; /* more assemblies than one lookup compares */ + +/* The number of assemblies that hold a part or a stub of a type decides + * nothing. A contract is joined to its one implementation however many + * assemblies hold that contract; where the implementation has no node, the + * one stub-only assembly's stub stands in however many assemblies have parts + * of the type; and a name is told from what an unseen part declares by + * looking the name up, not by counting the assemblies: one that no part + * declares is `missing`, and an unrelated name written inside such a type is + * bound as anywhere else. */ +TEST(doc_mentions_cs_many_assemblies) { + static const char stub[] = "namespace Acme.Coll\n" + "{\n" + " public partial class Thing { public void Go() { } }\n" + "\n" + " /// \n" + " public partial class UsesStub { }\n" + "}\n"; + static const char part[] = "namespace Acme.Coll\n" + "{\n" + " public partial class Phantom { public void Own() { } }\n" + " public partial class Wide { public void Extra() { } }\n" + "}\n"; + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_wide_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + th_write_file(TH_PATH(tmp, "shared/impl/Things.cs"), + "namespace Acme.Coll\n" + "{\n" + " public class Thing { public void Go() { } }\n" + " public partial class Phantom { public void Boo() { } }\n" + " public partial class Phantom { }\n" + "\n" + " public partial class Wide\n" + " {\n" + " /// \n" + " public void Common() { }\n" + " }\n" + "\n" + " /// \n" + " /// \n" + " public class UsesShared { }\n" + "}\n"); + th_write_file(TH_PATH(tmp, "facade/ref/Facade.csproj"), DM_EMPTY_PROJECT); + th_write_file(TH_PATH(tmp, "facade/ref/Stubs.cs"), + "namespace Acme.Coll\n" + "{\n" + " public partial class Phantom { public void Boo() { } }\n" + "}\n"); + for (int i = 0; i < DM_WIDE; i++) { + char rel[64]; + snprintf(rel, sizeof(rel), "s%02d/ref/S%02d.csproj", i, i); + th_write_file(TH_PATH(tmp, rel), DM_EMPTY_PROJECT); + snprintf(rel, sizeof(rel), "s%02d/ref/Stub.cs", i); + th_write_file(TH_PATH(tmp, rel), stub); + snprintf(rel, sizeof(rel), "p%02d/P%02d.csproj", i, i); + th_write_file(TH_PATH(tmp, rel), DM_EMPTY_PROJECT); + snprintf(rel, sizeof(rel), "p%02d/Part.cs", i); + th_write_file(TH_PATH(tmp, rel), part); + } + static const dm_want_t wants[] = { + /* DM_WIDE assemblies hold the contract: each is joined */ + {"s00/ref/Stub.cs", "Thing", "s00.ref.Stub.UsesStub", "impl.Things.Thing", NULL, + "s00.ref.Stub.Thing", DM_UNIQUE}, + {"s65/ref/Stub.cs", "Thing", "s65.ref.Stub.UsesStub", "impl.Things.Thing", NULL, + "s65.ref.Stub.Thing", DM_UNIQUE}, + /* DM_WIDE assemblies have parts of the type, one has only its stub: + * that stub stands in for the implementation without a node */ + {"shared/impl/Things.cs", "Phantom", "Things.UsesShared", "facade.ref.Stubs.Phantom", NULL, + NULL, DM_UNIQUE}, + /* what the unseen parts declare, and what they do not */ + {"shared/impl/Things.cs", "Wide.Extra", "Things.UsesShared", NULL, "ambiguous", NULL, NULL}, + {"shared/impl/Things.cs", "Wide.Nothing", "Things.UsesShared", NULL, "missing", NULL, NULL}, + {"shared/impl/Things.cs", "Thing", "Things.Wide.Common", "impl.Things.Thing", NULL, NULL, + NULL}, + }; + char db[512]; + snprintf(db, sizeof(db), "%s/wide.db", tmp); + int bad = dm_index(tmp, db, NULL) == 0 ? 0 : -1; + for (int i = 0; bad >= 0 && i < DM_COUNT(wants); i++) { + bad += dm_want_failed(db, &wants[i]); + } + dm_unlink_db(db); + th_rmtree(tmp); + ASSERT_EQ(bad, 0); + PASS(); +} + +/* A using directive of a file whose type has a base list is no private + * matter of that file: the base list is resolved through it, and what a type + * derives from decides how another file's reference to its members comes + * out. Changing it must not be repaired as if only the file had changed. */ +TEST(doc_mentions_incremental_using_of_based_type) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_incu_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + th_write_file(TH_PATH(repo, "src/V1.cs"), + "namespace Lib.V1\n{\n public class Base { public void Run() { } }\n}\n"); + th_write_file(TH_PATH(repo, "src/F.cs"), "using Lib.V1;\n" + "\n" + "namespace App\n" + "{\n" + " public class Job : Base { }\n" + "}\n"); + th_write_file(TH_PATH(repo, "src/G.cs"), "namespace App\n" + "{\n" + " /// \n" + " public class Uses { }\n" + "}\n"); + char inc_db[512]; + char full_db[512]; + snprintf(inc_db, sizeof(inc_db), "%s/inc.db", tmp); + snprintf(full_db, sizeof(full_db), "%s/full.db", tmp); + ASSERT_EQ(dm_index(repo, inc_db, NULL), 0); + char reason[64]; + /* Job derives from a class of the repository: what it lacks is missing */ + dm_row(inc_db, "src/G.cs", "Job.Nope", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + /* the using now names a namespace the repository does not declare: Job's + * base is outside, and so may be the member G.cs asks for */ + th_write_file(TH_PATH(repo, "src/F.cs"), "using Lib.V9;\n" + "\n" + "namespace App\n" + "{\n" + " public class Job : Base { }\n" + "}\n"); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "using of a type with a base list changed", + CBM_INCREMENTAL_ROUTE_FORCED_FULL), + 0); + dm_row(inc_db, "src/G.cs", "Job.Nope", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "external"); + dm_unlink_db(inc_db); + dm_unlink_db(full_db); + th_rmtree(tmp); + PASS(); +} + +enum { DM_PROJECTS = 40 }; + +/* A directory may hold any number of project files, and every one of them + * sets global usings of the files there: none is left out, whatever order a + * directory listing would have had. */ +TEST(doc_mentions_msbuild_many_project_files) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_many_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + enum { SHARED_RECORDS = 512 }; + char *props = dm_repeated("", "G", SHARED_RECORDS, + ""); + ASSERT_NOT_NULL(props); + th_write_file(TH_PATH(tmp, "Directory.Build.props"), props); + free(props); + char *targets = + dm_repeated("", "$(Shared)", + SHARED_RECORDS, ""); + ASSERT_NOT_NULL(targets); + th_write_file(TH_PATH(tmp, "Directory.Build.targets"), targets); + free(targets); + char *decls = malloc(DM_MANY_TEXT); + char *uses = malloc(DM_MANY_TEXT); + ASSERT_NOT_NULL(decls); + ASSERT_NOT_NULL(uses); + size_t d = 0; + size_t u = (size_t)snprintf(uses, DM_MANY_TEXT, "namespace App\n{\n /// "); + for (int i = 0; i < DM_PROJECTS; i++) { + char rel[64]; + char xml[256]; + snprintf(rel, sizeof(rel), "src/Many/P%02d.csproj", i); + snprintf(xml, sizeof(xml), + "\n" + " \n" + "\n", + i); + th_write_file(TH_PATH(tmp, rel), xml); + d += (size_t)snprintf(decls + d, DM_MANY_TEXT - d, + "namespace G.N%02d { public class K%02d { } }\n", i, i); + u += (size_t)snprintf(uses + u, DM_MANY_TEXT - u, " ", i); + } + snprintf(uses + u, DM_MANY_TEXT - u, "\n public class Uses { }\n}\n"); + th_write_file(TH_PATH(tmp, "src/Decl.cs"), decls); + th_write_file(TH_PATH(tmp, "src/Many/Uses.cs"), uses); + free(decls); + free(uses); + char db[512]; + snprintf(db, sizeof(db), "%s/many.db", tmp); + cbm_msb_test_cost_reset(); + ASSERT_EQ(dm_index(tmp, db, NULL), 0); + uint64_t records = 0; + uint64_t peak = 0; + cbm_msb_test_cost(&records, &peak); + fprintf(stderr, + "msbuild pipeline projects=%d shared=%d interpreted=%llu work=%llu " + "peak=%llu live=%llu\n", + DM_PROJECTS, SHARED_RECORDS, (unsigned long long)records, + (unsigned long long)cbm_msb_test_work(), (unsigned long long)peak, + (unsigned long long)cbm_msb_test_value_live_bytes()); + ASSERT_EQ(dm_mentions_from(db, "Uses.Uses"), DM_PROJECTS); + ASSERT_EQ(dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"), 0); + bool shared_once = records <= 2 * SHARED_RECORDS + 16 * DM_PROJECTS; + bool released = cbm_msb_test_value_live_bytes() == 0; + dm_unlink_db(db); + th_rmtree(tmp); + ASSERT_TRUE(shared_once); + ASSERT_TRUE(released); + PASS(); +} + +enum { DM_NEST = 70, DM_NEST_KEPT = 64, DM_NEST_TEXT = 8192 }; + +static int dm_count_lines(const char *blob, const char *prefix) { + int n = 0; + size_t pl = strlen(prefix); + for (const char *p = blob; p && *p;) { + n += strncmp(p, prefix, pl) == 0; + p = strchr(p, '\n'); + p = p ? p + 1 : NULL; + } + return n; +} + +/* The scope records nesting to a depth no program has. What nests deeper is + * not placed, and says so: the type names are kept (to be resolved to + * nothing else), the lines have no scope. Nothing is cut silently. */ +TEST(doc_mentions_cs_scan_nesting_limits) { + char *src = malloc(DM_NEST_TEXT); + ASSERT_NOT_NULL(src); + size_t w = (size_t)snprintf(src, DM_NEST_TEXT, "namespace N\n{\n"); + for (int i = 0; i < DM_NEST; i++) { + w += (size_t)snprintf(src + w, DM_NEST_TEXT - w, "public class T%d\n{\n", i); + } + for (int i = 0; i <= DM_NEST; i++) { + w += (size_t)snprintf(src + w, DM_NEST_TEXT - w, "}\n"); + } + CBMFileResult *r = dm_scope(src); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + ASSERT_EQ(dm_count_lines(r->doc_scope, "T\t"), DM_NEST_KEPT); + ASSERT_NOT_NULL(strstr(r->doc_scope, "\tT63\t")); + ASSERT_NULL(strstr(r->doc_scope, "\tT64\t")); + ASSERT_EQ(dm_count_lines(r->doc_scope, "Q\t"), DM_NEST - DM_NEST_KEPT); + ASSERT_NOT_NULL(strstr(r->doc_scope, "\nQ\tT64\n")); + ASSERT_NOT_NULL(strstr(r->doc_scope, "\nQ\tT69\n")); + ASSERT_NOT_NULL(strstr(r->doc_scope, "\nX\t")); /* the lines of the 65th type */ + cbm_free_result(r); + + /* ... and a namespace of more segments than that */ + w = (size_t)snprintf(src, DM_NEST_TEXT, "namespace "); + for (int i = 0; i < DM_NEST; i++) { + w += (size_t)snprintf(src + w, DM_NEST_TEXT - w, "%sA%d", i ? "." : "", i); + } + snprintf(src + w, DM_NEST_TEXT - w, "\n{\n public class Deep { }\n}\n"); + r = dm_scope(src); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + ASSERT_NULL(strstr(r->doc_scope, "\nR\t")); + ASSERT_NULL(strstr(r->doc_scope, "\nT\t")); + ASSERT_NOT_NULL(strstr(r->doc_scope, "\nQ\tDeep\n")); + ASSERT_NOT_NULL(strstr(r->doc_scope, "\nX\t")); + cbm_free_result(r); + free(src); + PASS(); +} + +/* Run one statement against the database; returns the rows it changed, -1 + * when it failed. */ +static int dm_exec(const char *db, const char *sql) { + sqlite3 *h = NULL; + int changed = -1; + if (sqlite3_open_v2(db, &h, SQLITE_OPEN_READWRITE, NULL) == SQLITE_OK && + sqlite3_exec(h, sql, NULL, NULL, NULL) == SQLITE_OK) { + changed = sqlite3_changes(h); + } + sqlite3_close(h); + return changed; +} + +/* A stored scope this code did not write -- a damaged row -- is not read + * around: the file's declarations would silently be missing from every + * lookup, and references to them would be reported `missing`. The layer + * fails for that run, visibly, and the next index rebuilds from the files. */ +TEST(doc_mentions_cs_damaged_stored_scope) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_dmg_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + th_write_file(TH_PATH(repo, "src/A.cs"), "namespace N\n{\n public class Target { }\n}\n"); + th_write_file(TH_PATH(repo, "src/B.cs"), "namespace N\n" + "{\n" + " /// \n" + " public class Uses { }\n" + "}\n"); + char db[512]; + snprintf(db, sizeof(db), "%s/dmg.db", tmp); + ASSERT_EQ(dm_index(repo, db, NULL), 0); + char props[512]; + int n = 0; + dm_edge(db, "B.Uses", "A.Target", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + /* one field too many in A's type record: no record this code writes */ + ASSERT_EQ(dm_exec(db, "UPDATE lsp_surface SET defs_json = " + "replace(defs_json, '\\nT\\t', '\\nT\\tX\\t') " + "WHERE rel_path = 'src/A.cs' AND instr(defs_json, '\\nT\\t') > 0"), + 1); + th_write_file(TH_PATH(repo, "src/B.cs"), + "namespace N\n" + "{\n" + " /// Again \n" + " public class Uses { }\n" + "}\n"); + cbm_pipeline_incremental_test_reset_faults(); + ASSERT_EQ(dm_index(repo, db, NULL), 0); + ASSERT_EQ(dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'"), 1); + /* never the row an index without A's declarations would write */ + ASSERT_EQ(dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE raw = 'Target'"), 0); + /* the next run rebuilds everything, and the edge is back */ + ASSERT_EQ(dm_index(repo, db, NULL), 0); + ASSERT_EQ(cbm_pipeline_incremental_test_last_route(), CBM_INCREMENTAL_ROUTE_FORCED_FULL); + ASSERT_EQ(dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'"), 0); + dm_edge(db, "B.Uses", "A.Target", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_unlink_db(db); + th_rmtree(tmp); + PASS(); +} + +SUITE(doc_mentions) { + RUN_TEST(doc_mentions_extract_cs_tokens); + RUN_TEST(doc_mentions_cs_scope_blob); + RUN_TEST(doc_mentions_cs_declarator_modifiers); + RUN_TEST(doc_mentions_cs_namespace_scopes); + RUN_TEST(doc_mentions_cs_namespace_boundaries); + RUN_TEST(doc_mentions_cs_control_scopes); + RUN_TEST(doc_mentions_cs_control_publication); + RUN_TEST(doc_mentions_cs_utf8_observation); + RUN_TEST(doc_mentions_cs_namespace_publication); + RUN_TEST(doc_mentions_cs_scope_parse_errors); + RUN_TEST(doc_mentions_cs_norm_type); + RUN_TEST(doc_mentions_msbuild_usings); + RUN_TEST(doc_mentions_msbuild_targets_isolation); + RUN_TEST(doc_mentions_msbuild_targets_failure); + RUN_TEST(doc_mentions_msbuild_items_isolation); + RUN_TEST(doc_mentions_msbuild_items_failure); + RUN_TEST(doc_mentions_msbuild_items_lifetime); + RUN_TEST(doc_mentions_msbuild_items_work); + RUN_TEST(doc_mentions_msbuild_targets_work); + RUN_TEST(doc_mentions_msbuild_nearest_work); + RUN_TEST(doc_mentions_msbuild_shared_closure_work); + RUN_TEST(doc_mentions_msbuild_shared_file_work); + RUN_TEST(doc_mentions_msbuild_shared_closure_isolation); + RUN_TEST(doc_mentions_msbuild_components_failure); + RUN_TEST(doc_mentions_msbuild_components_revision); + RUN_TEST(doc_mentions_msbuild_components_mixed_absence); + RUN_TEST(doc_mentions_msbuild_prefix_work); + RUN_TEST(doc_mentions_msbuild_prefix_isolation); + RUN_TEST(doc_mentions_msbuild_prefix_failure); + RUN_TEST(doc_mentions_msbuild_eval_storage); + RUN_TEST(doc_mentions_msbuild_value_lifetimes); + RUN_TEST(doc_mentions_msbuild_value_allocation); + RUN_TEST(doc_mentions_resolver_rules); + RUN_TEST(doc_mentions_resolver_arity_and_members); + RUN_TEST(doc_mentions_resolver_parse_errors); + RUN_TEST(doc_mentions_ship_gate); + RUN_TEST(doc_mentions_file_node_lookup); + RUN_TEST(doc_mentions_index_status_and_delete); + RUN_TEST(doc_mentions_index_status_preview_bounds); + RUN_TEST(doc_mentions_index_status_preview_text); + RUN_TEST(doc_mentions_index_status_preview_order); + RUN_TEST(doc_mentions_index_status_preview_failure); + RUN_TEST(doc_mentions_scope_delta_rules); + RUN_TEST(doc_mentions_incremental_equals_full); + RUN_TEST(doc_mentions_parallel_equals_sequential); + RUN_TEST(doc_mentions_cs_scan_dollar_run); + RUN_TEST(doc_mentions_cs_scan_branch_memory); + RUN_TEST(doc_mentions_cs_scan_branch_merge_work); + RUN_TEST(doc_mentions_cs_scan_declarator_modifier_work); + RUN_TEST(doc_mentions_cs_shared_doc_sources); + RUN_TEST(doc_mentions_cs_shared_doc_work); + RUN_TEST(doc_mentions_cs_shared_doc_replay); + RUN_TEST(doc_mentions_cs_shared_doc_allocation_failure); + RUN_TEST(doc_mentions_cs_scan_header_reads); + RUN_TEST(doc_mentions_cs_scan_nested_holes); + RUN_TEST(doc_mentions_cs_scan_holes_stack); + RUN_TEST(doc_mentions_msbuild_blob); + RUN_TEST(doc_mentions_msbuild_import_group_blob_growth); + RUN_TEST(doc_mentions_msbuild_import_group_roundtrip); + RUN_TEST(doc_mentions_msbuild_legacy_import_blob); + RUN_TEST(doc_mentions_msbuild_bad_import_group_blob); + RUN_TEST(doc_mentions_msbuild_imports); + RUN_TEST(doc_mentions_msbuild_conditions); + RUN_TEST(doc_mentions_msbuild_unknown_spreads); + RUN_TEST(doc_mentions_msbuild_gate); + RUN_TEST(doc_mentions_incremental_project_files); +#ifndef _WIN32 + RUN_TEST(doc_mentions_msbuild_linked_project_file); + RUN_TEST(doc_mentions_msbuild_linked_import_directory); +#endif + RUN_TEST(doc_mentions_cs_lookup_order); + RUN_TEST(doc_mentions_cs_type_parameters); + RUN_TEST(doc_mentions_cs_usings); + RUN_TEST(doc_mentions_cs_entities); + RUN_TEST(doc_mentions_cs_assemblies_by_name); + RUN_TEST(doc_mentions_cs_reference_source); + RUN_TEST(doc_mentions_cs_assemblies_differ); + RUN_TEST(doc_mentions_cs_shared_parts); + RUN_TEST(doc_mentions_cs_shared_trees); + RUN_TEST(doc_mentions_cs_no_project_files); + RUN_TEST(doc_mentions_cs_contract_join); + RUN_TEST(doc_mentions_cs_idle_project_file); + RUN_TEST(doc_mentions_cs_test_namespaces); + RUN_TEST(doc_mentions_cs_members); + RUN_TEST(doc_mentions_cs_constructors); + RUN_TEST(doc_mentions_cs_inherited_members); + RUN_TEST(doc_mentions_cs_doc_ids); + RUN_TEST(doc_mentions_cs_reasons); + RUN_TEST(doc_mentions_cs_text_forms); + RUN_TEST(doc_mentions_cs_lookup_cost); + RUN_TEST(doc_mentions_cs_import_lookup_cost); + RUN_TEST(doc_mentions_cs_import_stage_aliases); + RUN_TEST(doc_mentions_cs_import_static_parts); + RUN_TEST(doc_mentions_cs_import_candidate_allocation); + RUN_TEST(doc_mentions_cs_no_silent_limits); + RUN_TEST(doc_mentions_cs_many_assemblies); + RUN_TEST(doc_mentions_incremental_using_of_based_type); + RUN_TEST(doc_mentions_msbuild_many_project_files); + RUN_TEST(doc_mentions_cs_scan_nesting_limits); + RUN_TEST(doc_mentions_cs_damaged_stored_scope); +} diff --git a/tests/test_doc_mentions_helpers.h b/tests/test_doc_mentions_helpers.h new file mode 100644 index 0000000000..66fbcd52ce --- /dev/null +++ b/tests/test_doc_mentions_helpers.h @@ -0,0 +1,414 @@ +/* + * test_doc_mentions_helpers.h — helpers shared by the doc-mentions suites + * (tests/test_doc_mentions.c and one test_doc_mentions_.c per language + * leg). + * + * Every pipeline helper indexes a real fixture through cbm_pipeline_run and + * reads the published database, so a test sees what a user's index holds. + * All functions are static inline, like test_helpers.h: including the header + * from several test files causes no linker issue, and a file that uses only + * some of them gets no unused-function warning. + */ +#ifndef TEST_DOC_MENTIONS_HELPERS_H +#define TEST_DOC_MENTIONS_HELPERS_H + +#include "../src/foundation/compat.h" +#include "test_helpers.h" + +#include "cbm.h" +#include "doclink.h" +#include "foundation/mem_core.h" +#include "mcp/mcp.h" +#include "pipeline/doc_links.h" +#include "pipeline/pipeline.h" +#include "pipeline/pipeline_internal.h" +#include "sqlite3.h" +#include + +#include +#include +#include +#include + +/* ── extraction ──────────────────────────────────────────────────── */ + +/* Extract one source text as language `lang` under project "p"; the result + * (tokens in doc_links, scope blob in doc_scope) is freed with + * cbm_free_result. */ +static inline CBMFileResult *dm_extract(const char *src, CBMLanguage lang, const char *rel_path) { + return cbm_extract_file(src, (int)strlen(src), lang, "p", rel_path, 0, NULL, NULL); +} + +/* The first token written as `raw`; NULL when there is none. */ +static inline const CBMDocLink *dm_find_token(const CBMFileResult *r, const char *raw) { + for (int i = 0; i < r->doc_links.count; i++) { + if (strcmp(r->doc_links.items[i].raw, raw) == 0) { + return &r->doc_links.items[i]; + } + } + return NULL; +} + +static inline int dm_count_tokens(const CBMFileResult *r, const char *raw) { + int n = 0; + for (int i = 0; i < r->doc_links.count; i++) { + n += strcmp(r->doc_links.items[i].raw, raw) == 0; + } + return n; +} + +/* ── the published database ──────────────────────────────────────── */ + +/* Index `repo` into `db` (full mode; a second call on the same database takes + * the incremental route). The project name is strdup'ed into *project_out + * when that is not NULL. Returns the pipeline's result. */ +static inline int dm_index(const char *repo, const char *db, char **project_out) { + cbm_pipeline_t *p = cbm_pipeline_new(repo, db, CBM_MODE_FULL); + if (!p) { + return -1; + } + int rc = cbm_pipeline_run(p); + if (project_out) { + *project_out = strdup(cbm_pipeline_project_name(p)); + } + cbm_pipeline_free(p); + return rc; +} + +/* Properties of the MENTIONS edge whose endpoints' qualified names END with + * the given local paths ("Widget", "Helper.Once"); "" when absent; count via + * *n. */ +static inline void dm_edge(const char *db, const char *src_suffix, const char *tgt_suffix, + char *props, size_t cap, int *n) { + props[0] = '\0'; + *n = 0; + sqlite3 *h = NULL; + if (sqlite3_open_v2(db, &h, SQLITE_OPEN_READONLY, NULL) != SQLITE_OK) { + sqlite3_close(h); + *n = -1; + return; + } + sqlite3_stmt *st = NULL; + const char *sql = + "SELECT e.properties FROM edges e JOIN nodes s ON s.id = e.source_id " + "JOIN nodes t ON t.id = e.target_id WHERE e.type = 'MENTIONS' " + "AND (s.qualified_name LIKE '%.' || ?1) AND (t.qualified_name LIKE '%.' || ?2)"; + if (sqlite3_prepare_v2(h, sql, -1, &st, NULL) == SQLITE_OK) { + sqlite3_bind_text(st, 1, src_suffix, -1, SQLITE_TRANSIENT); + sqlite3_bind_text(st, 2, tgt_suffix, -1, SQLITE_TRANSIENT); + while (sqlite3_step(st) == SQLITE_ROW) { + (*n)++; + snprintf(props, cap, "%s", (const char *)sqlite3_column_text(st, 0)); + } + } + sqlite3_finalize(st); + sqlite3_close(h); +} + +/* Number of MENTIONS edges leaving the definition whose qualified name ends + * with `src_suffix`; -1 when the database cannot be read. */ +static inline int dm_mentions_from(const char *db, const char *src_suffix) { + sqlite3 *h = NULL; + int n = -1; + if (sqlite3_open_v2(db, &h, SQLITE_OPEN_READONLY, NULL) == SQLITE_OK) { + sqlite3_stmt *st = NULL; + if (sqlite3_prepare_v2(h, + "SELECT COUNT(*) FROM edges e JOIN nodes s ON s.id = e.source_id " + "WHERE e.type = 'MENTIONS' AND s.qualified_name LIKE '%.' || ?1", + -1, &st, NULL) == SQLITE_OK) { + sqlite3_bind_text(st, 1, src_suffix, -1, SQLITE_TRANSIENT); + if (sqlite3_step(st) == SQLITE_ROW) { + n = sqlite3_column_int(st, 0); + } + } + sqlite3_finalize(st); + } + sqlite3_close(h); + return n; +} + +/* The single integer a query returns; -1 when it cannot be read. */ +static inline int dm_count(const char *db, const char *sql) { + sqlite3 *h = NULL; + int n = -1; + if (sqlite3_open_v2(db, &h, SQLITE_OPEN_READONLY, NULL) == SQLITE_OK) { + sqlite3_stmt *st = NULL; + if (sqlite3_prepare_v2(h, sql, -1, &st, NULL) == SQLITE_OK && + sqlite3_step(st) == SQLITE_ROW) { + n = sqlite3_column_int(st, 0); + } + sqlite3_finalize(st); + } + sqlite3_close(h); + return n; +} + +/* Reason (and, when `syntax` is not NULL, the family name) of the unresolved + * row with this raw text in this file; "" when there is none. */ +static inline void dm_row(const char *db, const char *rel, const char *raw, char *reason, + size_t cap, char *syntax, size_t scap) { + reason[0] = '\0'; + if (syntax) { + syntax[0] = '\0'; + } + sqlite3 *h = NULL; + if (sqlite3_open_v2(db, &h, SQLITE_OPEN_READONLY, NULL) == SQLITE_OK) { + sqlite3_stmt *st = NULL; + if (sqlite3_prepare_v2(h, + "SELECT reason, syntax FROM doc_link_unresolved WHERE rel_path = ?1 " + "AND raw = ?2", + -1, &st, NULL) == SQLITE_OK) { + sqlite3_bind_text(st, 1, rel, -1, SQLITE_TRANSIENT); + sqlite3_bind_text(st, 2, raw, -1, SQLITE_TRANSIENT); + if (sqlite3_step(st) == SQLITE_ROW) { + snprintf(reason, cap, "%s", (const char *)sqlite3_column_text(st, 0)); + if (syntax) { + snprintf(syntax, scap, "%s", (const char *)sqlite3_column_text(st, 1)); + } + } + } + sqlite3_finalize(st); + } + sqlite3_close(h); +} + +/* Canonical text of every MENTIONS edge and unresolved row: what two indexes + * of the same tree (full and incremental, one worker and several) must agree + * on byte for byte. The caller frees it; NULL when the database cannot be + * read. */ +static inline char *dm_doclink_state(const char *db) { + sqlite3 *h = NULL; + if (sqlite3_open_v2(db, &h, SQLITE_OPEN_READONLY, NULL) != SQLITE_OK) { + sqlite3_close(h); + return NULL; + } + size_t cap = 4096; + size_t len = 0; + char *buf = malloc(cap); + buf[0] = '\0'; + const char *queries[] = { + "SELECT 'E ' || s.qualified_name || ' -> ' || t.qualified_name || ' ' || e.properties " + "FROM edges e JOIN nodes s ON s.id = e.source_id JOIN nodes t ON t.id = e.target_id " + "WHERE e.type = 'MENTIONS' ORDER BY 1", + "SELECT 'R ' || rel_path || ':' || line || ' ' || syntax || ' [' || raw || '] ' || reason " + "FROM doc_link_unresolved ORDER BY 1", + }; + for (size_t q = 0; q < sizeof(queries) / sizeof(queries[0]); q++) { + sqlite3_stmt *st = NULL; + if (sqlite3_prepare_v2(h, queries[q], -1, &st, NULL) != SQLITE_OK) { + continue; + } + while (sqlite3_step(st) == SQLITE_ROW) { + const char *line = (const char *)sqlite3_column_text(st, 0); + size_t l = strlen(line); + if (len + l + 2 > cap) { + cap = (len + l + 2) * 2; + buf = realloc(buf, cap); + } + memcpy(buf + len, line, l); + len += l; + buf[len++] = '\n'; + buf[len] = '\0'; + } + sqlite3_finalize(st); + } + sqlite3_close(h); + return buf; +} + +/* Remove a database and its sidecars. */ +static inline void dm_unlink_db(const char *db) { + char side[600]; + unlink(db); + snprintf(side, sizeof(side), "%s-wal", db); + unlink(side); + snprintf(side, sizeof(side), "%s-shm", db); + unlink(side); +} + +/* ── incremental == full ─────────────────────────────────────────── */ + +/* One step of an incremental test, after the caller edited the tree: index + * `repo` incrementally into `inc_db`, then fully into a fresh `full_db`, and + * require identical MENTIONS edges and unresolved rows AND the expected + * route. Returns 0, or -1 after printing what differs (`what` names the + * step). */ +static inline int dm_step(const char *repo, const char *inc_db, const char *full_db, + const char *what, cbm_incremental_route_t want_route) { + cbm_pipeline_incremental_test_reset_faults(); + if (dm_index(repo, inc_db, NULL) != 0) { + printf(" %s: incremental index failed\n", what); + return -1; + } + cbm_incremental_route_t route = cbm_pipeline_incremental_test_last_route(); + dm_unlink_db(full_db); + if (dm_index(repo, full_db, NULL) != 0) { + printf(" %s: full index failed\n", what); + return -1; + } + char *inc = dm_doclink_state(inc_db); + char *full = dm_doclink_state(full_db); + int rc = 0; + if (!inc || !full || strcmp(inc, full) != 0) { + printf(" %s: incremental != full\n--- incremental (route %d)\n%s--- full\n%s", what, + (int)route, inc ? inc : "(null)", full ? full : "(null)"); + rc = -1; + } else if (route != want_route) { + printf(" %s: route %d, expected %d\n", what, (int)route, (int)want_route); + rc = -1; + } + free(inc); + free(full); + return rc; +} + +/* ── one worker == several workers ───────────────────────────────── */ + +/* Index `repo` with four workers into `par_db` and with one worker into + * `seq_db` (the fixture needs more than 50 files, or both take the sequential + * passes), and require identical MENTIONS edges and unresolved rows. + * CBM_WORKERS is restored. Returns 0, or -1 after printing what differs. */ +static inline int dm_workers_agree(const char *repo, const char *par_db, const char *seq_db) { + const char *saved_workers = getenv("CBM_WORKERS"); + char *saved_workers_copy = saved_workers ? strdup(saved_workers) : NULL; + cbm_setenv("CBM_WORKERS", "4", 1); + int par_rc = dm_index(repo, par_db, NULL); + cbm_setenv("CBM_WORKERS", "1", 1); /* one worker: the sequential passes */ + int seq_rc = dm_index(repo, seq_db, NULL); + if (saved_workers_copy) { + cbm_setenv("CBM_WORKERS", saved_workers_copy, 1); + free(saved_workers_copy); + } else { + cbm_unsetenv("CBM_WORKERS"); + } + if (par_rc != 0 || seq_rc != 0) { + printf(" index failed: parallel rc %d, sequential rc %d\n", par_rc, seq_rc); + return -1; + } + char *par = dm_doclink_state(par_db); + char *seq = dm_doclink_state(seq_db); + bool same = par && seq && strcmp(par, seq) == 0; + if (!same) { + printf(" parallel != sequential\n--- parallel\n%s--- sequential\n%s", par ? par : "(null)", + seq ? seq : "(null)"); + } + free(par); + free(seq); + return same ? 0 : -1; +} + +/* ── scope deltas ────────────────────────────────────────────────── */ + +/* The names a scope delta reported, joined by ','. */ +typedef struct { + char names[256]; +} dm_names_t; + +/* cbm_doclink_name_fn collecting into a dm_names_t. */ +static inline bool dm_name_put(void *ud, const char *name, size_t len) { + dm_names_t *n = (dm_names_t *)ud; + size_t used = strlen(n->names); + if (used + len + 2 > sizeof(n->names)) { + return false; + } + if (used > 0) { + n->names[used++] = ','; + } + memcpy(n->names + used, name, len); + n->names[used + len] = '\0'; + return true; +} + +enum { DM_DELTA_SCAN_FAILED = -2 }; + +/* The scope delta between two versions of one file of language `lang` (NULL: + * the file does not exist), through the language's real scanner and the + * persisted form of its scope; the removed names joined by ',' in `names`. + * Returns the cbm_doclink_delta_t, -1 as the hook does, or + * DM_DELTA_SCAN_FAILED when a version yields no scope. */ +static inline int dm_scope_delta(CBMLanguage lang, const char *rel_path, const char *before, + const char *after, char *names, size_t cap) { + names[0] = '\0'; + CBMFileResult *a = before ? dm_extract(before, lang, rel_path) : NULL; + CBMFileResult *b = after ? dm_extract(after, lang, rel_path) : NULL; + char *pa = (a && a->doc_scope) ? cbm_doclink_portable_scope(a->doc_scope) : NULL; + char *pb = (b && b->doc_scope) ? cbm_doclink_portable_scope(b->doc_scope) : NULL; + dm_names_t n = {{0}}; + int rc = ((before && !pa) || (after && !pb)) + ? DM_DELTA_SCAN_FAILED + : cbm_doclinks_scope_delta(pa, pb, dm_name_put, &n); + snprintf(names, cap, "%s", n.names); + cbm_free(CBM_MEM_CLASS_OTHER, pa); + cbm_free(CBM_MEM_CLASS_OTHER, pb); + if (a) { + cbm_free_result(a); + } + if (b) { + cbm_free_result(b); + } + return rc; +} + +/* ── index_status ────────────────────────────────────────────────── */ + +/* The text content of an MCP tool result (the report itself); the caller + * frees it. */ +static inline char *dm_tool_text(const char *mcp_result) { + yyjson_doc *doc = mcp_result ? yyjson_read(mcp_result, strlen(mcp_result), 0) : NULL; + yyjson_val *root = doc ? yyjson_doc_get_root(doc) : NULL; + yyjson_val *content = root ? yyjson_obj_get(root, "content") : NULL; + yyjson_val *item = content ? yyjson_arr_get(content, 0) : NULL; + const char *text = item ? yyjson_get_str(yyjson_obj_get(item, "text")) : NULL; + char *out = text ? strdup(text) : NULL; + yyjson_doc_free(doc); + return out; +} + +/* index_status of the project as text, from a server of its own (nothing + * cached from an earlier call). The project's database is looked up in + * CBM_CACHE_DIR: point that at a private directory first. */ +static inline char *dm_index_status(const char *project, bool full) { + cbm_mcp_server_t *srv = cbm_mcp_server_new(NULL); + if (!srv) { + return NULL; + } + char args[1200]; + snprintf(args, sizeof(args), "{\"project\":\"%s\"%s}", project, + full ? ",\"diagnostics\":\"full\"" : ""); + char *resp = cbm_mcp_handle_tool(srv, "index_status", args); + char *text = dm_tool_text(resp); + free(resp); + cbm_mcp_server_free(srv); + return text; +} + +/* Every reason the table holds is listed under doc_links.unresolved with its + * row count. The number of reasons checked; -1 when a line is missing. */ +static inline int dm_reason_lines(const char *db, const char *block) { + sqlite3 *h = NULL; + int n = -1; + if (sqlite3_open_v2(db, &h, SQLITE_OPEN_READONLY, NULL) == SQLITE_OK) { + sqlite3_stmt *st = NULL; + if (sqlite3_prepare_v2(h, + "SELECT reason, COUNT(*) FROM doc_link_unresolved GROUP BY reason", + -1, &st, NULL) == SQLITE_OK) { + n = 0; + while (n >= 0 && sqlite3_step(st) == SQLITE_ROW) { + char line[128]; + snprintf(line, sizeof(line), "\n %s: %d\n", + (const char *)sqlite3_column_text(st, 0), sqlite3_column_int(st, 1)); + if (strstr(block, line)) { + n++; + } else { + printf(" no line%s in\n%s\n", line, block); + n = -1; + } + } + } + sqlite3_finalize(st); + } + sqlite3_close(h); + return n; +} + +#endif /* TEST_DOC_MENTIONS_HELPERS_H */ diff --git a/tests/test_edge_structural.c b/tests/test_edge_structural.c index 798fef3d08..4b91ae05f2 100644 --- a/tests/test_edge_structural.c +++ b/tests/test_edge_structural.c @@ -257,6 +257,7 @@ static const char *ES_ALL_EDGE_TYPES[] = {"CALLS", "IMPORTS", "INHERITS", "INFRA_MAPS", + "MENTIONS", "OVERRIDE", "REFERENCES_FILE", "SEMANTICALLY_RELATED", diff --git a/tests/test_lang_contract.c b/tests/test_lang_contract.c index 6eb993564e..8d0192feb9 100644 --- a/tests/test_lang_contract.c +++ b/tests/test_lang_contract.c @@ -1069,6 +1069,7 @@ static const char *ALL_EDGE_TYPES[] = {"CALLS", "IMPORTS", "INHERITS", "INFRA_MAPS", + "MENTIONS", "OVERRIDE", "REFERENCES_FILE", "SEMANTICALLY_RELATED", diff --git a/tests/test_main.c b/tests/test_main.c index b63f07035d..98e0cdc47e 100644 --- a/tests/test_main.c +++ b/tests/test_main.c @@ -969,6 +969,7 @@ extern void suite_store_checkpoint(void); extern void suite_traces(void); extern void suite_configlink(void); extern void suite_doclinks(void); +extern void suite_doc_mentions(void); extern void suite_infrascan(void); extern void suite_cli(void); extern void suite_agent_clients(void); @@ -1336,6 +1337,9 @@ int main(int argc, char **argv) { /* Markdown file reference link */ RUN_SELECTED_SUITE(doclinks); + /* Doc-comment references -> MENTIONS */ + RUN_SELECTED_SUITE(doc_mentions); + /* Infrastructure scanning */ RUN_SELECTED_SUITE(infrascan); From 35855569c178cbc74560f7fc215806422919d98a Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Sun, 4 Oct 2026 17:15:47 +0200 Subject: [PATCH 02/19] doc-links(cs): bounds and checks from security review 2 (S11 to S16, S18) - S11: a namespace read back from a stored scope is held to the nesting the scanner writes (64 segments of its full name); past it the scope is refused like any other stored scope this code did not write. - S12: index_status lists only the reasons the layer writes; any other reason text in the table is counted under one fixed key (unrecognized_reason) and the status is error. - S13: the per-file scratch table of the index build is made anew when a far larger file grew it, so emptying it costs about the file's own size. - S14: the portable copy of a scope keeps an empty line field empty and is never longer than the scope it is made from. - S15: a reference whose brackets do not pair, or that goes on after its parameter list or its type arguments, is unparseable instead of being resolved by the part that can be read. - S16: a project file the scan did not read (malformed, or larger than a project file is) opens the scope of the projects that evaluate it: what it holds is unknown, not absent. It is counted once. - S18: an incremental repair matches the names a changed file no longer declares by the resolver's own identifier rule (bytes of multi-byte characters included) and without a length cut. Tests: cs_portable_scope_bound, cs_unbalanced_brackets, cs_stored_scope_nesting, index_status_unknown_reason, msbuild_unread_project_opens, cs_scratch_table_work, incremental_non_ascii_name; msbuild_gate and index_status_preview_bounds follow the new S16 and S12 outcomes. Signed-off-by: Martin Vogel --- internal/cbm/doclink_cs.c | 8 +- src/mcp/mcp.c | 26 ++- src/pipeline/doc_links.h | 3 + src/pipeline/doc_links_cs.c | 100 ++++++-- src/pipeline/doc_links_msbuild.c | 16 +- src/pipeline/pipeline_incremental.c | 39 +++- tests/test_doc_mentions.c | 343 +++++++++++++++++++++++++++- 7 files changed, 497 insertions(+), 38 deletions(-) diff --git a/internal/cbm/doclink_cs.c b/internal/cbm/doclink_cs.c index 354debc294..2e744069ae 100644 --- a/internal/cbm/doclink_cs.c +++ b/internal/cbm/doclink_cs.c @@ -727,8 +727,8 @@ typedef struct { int end1; /* ... and where its first branch ended; valid once has_end1 */ int n_end1; int first_open; /* first open-brace entry allocated in the current branch */ - bool has_end1; /* an #else was seen */ - int mark; /* its entry in the brace list */ + bool has_end1; /* an #else was seen */ + int mark; /* its entry in the brace list */ } cs_pp_t; enum { @@ -2933,7 +2933,9 @@ char *cbm_doclink_cs_portable_scope(const char *scope) { while (fend < len && p[fend] != '\t') { fend++; } - if (cs_line_field(tag, field)) { + /* a line number becomes 0; an empty field stays empty, so the + * copy is never longer than the scope it is made from */ + if (cs_line_field(tag, field) && fend > i) { out[w++] = '0'; } else { memcpy(out + w, p + i, fend - i); diff --git a/src/mcp/mcp.c b/src/mcp/mcp.c index 1dc548fbf4..ba4a2c41ff 100644 --- a/src/mcp/mcp.c +++ b/src/mcp/mcp.c @@ -57,7 +57,8 @@ enum { #include "cypher/cypher.h" #include "discover/discover.h" #include "pipeline/pipeline.h" -#include "callable_sig.h" /* cbm_qn_callable_base_len */ +#include "callable_sig.h" /* cbm_qn_callable_base_len */ +#include "pipeline/doc_links.h" /* cbm_doclink_reason_name: the reasons the layer writes */ #include "pipeline/pass_cross_repo.h" #include "git/git_context.h" #include "cli/cli.h" @@ -6395,6 +6396,18 @@ static bool doc_link_preview_samples(yyjson_mut_doc *doc, const cbm_doc_link_pre return true; } +/* True for a reason the doc-link layer writes. The table is read back from + * the database: any other text in it was not written by this layer, and + * never becomes a key of the report. */ +static bool doc_link_reason_known(const char *reason) { + for (int r = 0; r < CBM_DOCLINK_REASON_COUNT; r++) { + if (strcmp(reason, cbm_doclink_reason_name(r)) == 0) { + return true; + } + } + return false; +} + static bool add_doc_links_report(yyjson_mut_doc *doc, yyjson_mut_val *root, cbm_store_t *store, const char *project, bool with_samples) { int mentions = cbm_store_count_edges_by_type(store, project, "MENTIONS"); @@ -6421,6 +6434,7 @@ static bool add_doc_links_report(yyjson_mut_doc *doc, yyjson_mut_val *root, cbm_ } bool error = rc != CBM_STORE_OK || mentions < 0 || !present || sample_failed; bool built = false; + int64_t unrecognized = 0; yyjson_mut_val *dl = yyjson_mut_obj(doc); yyjson_mut_val *unresolved = yyjson_mut_obj(doc); if (!dl || !unresolved) { @@ -6431,12 +6445,22 @@ static bool add_doc_links_report(yyjson_mut_doc *doc, yyjson_mut_val *root, cbm_ error = true; continue; } + if (!doc_link_reason_known(reasons[i].reason)) { + /* rows this layer did not write: counted under one fixed key */ + unrecognized += reasons[i].count; + error = true; + continue; + } yyjson_mut_val *key = yyjson_mut_strcpy(doc, reasons[i].reason); yyjson_mut_val *number = yyjson_mut_int(doc, reasons[i].count); if (!key || !number || !yyjson_mut_obj_add(unresolved, key, number)) { goto cleanup; } } + if (unrecognized > 0 && + !yyjson_mut_obj_add_int(doc, unresolved, "unrecognized_reason", unrecognized)) { + goto cleanup; + } if (!yyjson_mut_obj_add_int(doc, dl, "mentions", mentions < 0 ? 0 : mentions) || !yyjson_mut_obj_add_val(doc, dl, "unresolved", unresolved) || !yyjson_mut_obj_add_str(doc, dl, "status", error ? "error" : "ok")) { diff --git a/src/pipeline/doc_links.h b/src/pipeline/doc_links.h index f44241678b..40291d86a7 100644 --- a/src/pipeline/doc_links.h +++ b/src/pipeline/doc_links.h @@ -216,6 +216,9 @@ extern const cbm_doclink_resolver_t cbm_doclink_cs_resolver; * without a clock. Test builds only. */ void cbm_doclink_cs_test_work_reset(void); uint64_t cbm_doclink_cs_test_work(void); +/* The buckets the index build walked to empty its per-file scratch table + * (reset with the work above). */ +uint64_t cbm_doclink_cs_test_scratch_work(void); #endif /* True when `rel_path` is a scope input of some language (scope_input). */ diff --git a/src/pipeline/doc_links_cs.c b/src/pipeline/doc_links_cs.c index 16216d9ce1..c957841e7f 100644 --- a/src/pipeline/doc_links_cs.c +++ b/src/pipeline/doc_links_cs.c @@ -200,6 +200,7 @@ enum { * and every namespace above a declared one. */ typedef struct { int parent; + int depth; /* segments of its full name (the global namespace: 0) */ const char *name; /* its own segment */ bool declared; /* a file declares it; else it is only above a declared one */ bool prod; /* product code declares it, or a namespace under it */ @@ -525,6 +526,8 @@ typedef struct { static _Atomic uint64_t cs_work_counter; /* New import index construction is measured separately from lookup work. */ static _Atomic uint64_t cs_index_work_counter; +/* Buckets of the node pass's scratch table walked when it is emptied. */ +static _Atomic uint64_t cs_scratch_work_counter; static _Atomic bool cs_fail_candidate_alloc; static _Atomic bool cs_candidate_alloc_failed; @@ -540,6 +543,7 @@ bool cbm_doclink_cs_test_candidate_alloc_failed(void) { void cbm_doclink_cs_test_work_reset(void) { atomic_store(&cs_work_counter, 0); atomic_store(&cs_index_work_counter, 0); + atomic_store(&cs_scratch_work_counter, 0); } uint64_t cbm_doclink_cs_test_work(void) { @@ -550,6 +554,10 @@ uint64_t cbm_doclink_cs_test_index_work(void) { return atomic_load(&cs_index_work_counter); } +uint64_t cbm_doclink_cs_test_scratch_work(void) { + return atomic_load(&cs_scratch_work_counter); +} + static void cs_work(uint64_t n) { atomic_fetch_add_explicit(&cs_work_counter, n, memory_order_relaxed); } @@ -557,6 +565,10 @@ static void cs_work(uint64_t n) { static void cs_index_work(uint64_t n) { atomic_fetch_add_explicit(&cs_index_work_counter, n, memory_order_relaxed); } + +static void cs_scratch_work(uint64_t n) { + atomic_fetch_add_explicit(&cs_scratch_work_counter, n, memory_order_relaxed); +} #else static void cs_work(uint64_t n) { (void)n; @@ -565,6 +577,10 @@ static void cs_work(uint64_t n) { static void cs_index_work(uint64_t n) { (void)n; } + +static void cs_scratch_work(uint64_t n) { + (void)n; +} #endif /* ── Small helpers ───────────────────────────────────────────────── */ @@ -750,16 +766,22 @@ static int ns_make(cs_index_t *ix, int parent, const char *seg, size_t len) { return CS_NONE; } int id = ix->nnss++; - ix->nss[id] = (cs_ns_t){.parent = parent, .name = name}; + ix->nss[id] = + (cs_ns_t){.parent = parent, .depth = ix->nss[parent].depth + SKIP_ONE, .name = name}; cbm_ht_set(ix->ns_by_key, k, (void *)(intptr_t)(id + SKIP_ONE)); return id; } /* The namespace the dotted `path` names under `from`, every segment created - * on the way; CS_NONE for a bad name or when memory ran out. */ + * on the way; CS_NONE for a bad name, when memory ran out, and past the + * scanner's nesting limit: a scope read back from the store is held to the + * limit its writer keeps (every enclosing namespace is a lookup step). */ static int ns_make_path(cs_index_t *ix, int from, const char *path) { int ns = from; for (const char *p = path; ns >= 0 && *p;) { + if (ix->nss[ns].depth >= CS_MAX_NEST) { + return CS_NONE; + } const char *dot = strchr(p, '.'); size_t n = dot ? (size_t)(dot - p) : strlen(p); ns = ns_make(ix, ns, p, n); @@ -2029,14 +2051,46 @@ static const cbm_gbuf_node_t *member_node_at(const cbm_gbuf_t *g, const cs_membe return (m->kind == 'c' ? label_is_callable(n->label) : label_is_value(n->label)) ? n : NULL; } -/* Scratch tables of the node pass: cleared for every file. */ +/* Scratch tables of the node pass: emptied for every file. */ typedef struct { CBMHashTable *names; /* "\x1f" -> index + 1 */ + uint32_t sized; /* what the table is sized for: the most keys it held, or + * its initial capacity -- what emptying it walks */ CBMArena keys; int *last; /* per gid: the last type that has it */ int cap_last; } cs_node_pass_t; +enum { CS_SCRATCH_MIN = 64, CS_SCRATCH_SLACK = 4 }; + +/* Empty the scratch table for a file that puts about `need` keys into it. + * Emptying a table walks every bucket it has, and a table never shrinks: one + * that a far larger file grew is made anew, so that one large file does not + * make every later file pay for its size. false when memory ran out. */ +static bool node_pass_reset(cs_index_t *ix, cs_node_pass_t *np, int need) { + uint32_t want = need > CS_SCRATCH_MIN ? (uint32_t)need : CS_SCRATCH_MIN; + if (np->sized > want * CS_SCRATCH_SLACK) { + cbm_ht_free(np->names); + np->names = cbm_ht_create(want); + np->sized = want; + if (!np->names) { + ix->oom = true; + return false; + } + } else { + cs_scratch_work(np->sized); + cbm_ht_clear(np->names); + } + cbm_arena_rewind(&np->keys); /* the emptied table held the only pointers into it */ + return true; +} + +/* Note how many keys the scratch table holds now. */ +static void node_pass_held(cs_node_pass_t *np) { + uint32_t held = cbm_ht_count(np->names); + np->sized = held > np->sized ? held : np->sized; +} + /* The path group of every type of the file (types with one path -- `Foo` and * `Foo`, and what is nested in them under one name -- share the node * .), and which declaration owns that node: the last one. */ @@ -2050,8 +2104,9 @@ static bool assign_gids(cs_index_t *ix, cs_file_t *f, cs_node_pass_t *np) { return false; } } - cbm_ht_clear(np->names); - cbm_arena_rewind(&np->keys); /* the cleared table held the only pointers into it */ + if (!node_pass_reset(ix, np, f->ntypes)) { + return false; + } int gids = 0; for (int t = 0; t < f->ntypes; t++) { cs_type_t *ty = &f->types[t]; @@ -2098,8 +2153,10 @@ static bool bind_nodes(cs_index_t *ix, cs_file_t *f, const cbm_gbuf_t *g, cs_nod } } /* which member is the last of its (path, name) */ - cbm_ht_clear(np->names); - cbm_arena_rewind(&np->keys); + node_pass_held(np); + if (!node_pass_reset(ix, np, f->nmembers)) { + return false; + } for (int m = 0; m < f->nmembers; m++) { char key[(CS_NAME_BUF) + CBM_SZ_16]; snprintf(key, sizeof(key), "%d\x1f%s", f->types[f->members[m].type].gid, @@ -2114,6 +2171,7 @@ static bool bind_nodes(cs_index_t *ix, cs_file_t *f, const cbm_gbuf_t *g, cs_nod } cbm_ht_set(np->names, k, (void *)(intptr_t)(m + SKIP_ONE)); } + node_pass_held(np); int cached = CS_NONE; bool fits = false; for (int m = 0; m < f->nmembers; m++) { @@ -3274,6 +3332,9 @@ static bool parse_seg(const char *s, size_t n, cs_seg_t *out) { if (s[i] == '{' || s[i] == '<') { name_end = i; size_t e = group_end(s, n, i); + if (e < n) { + return false; /* text after the type arguments: no name */ + } size_t inner = e > i + PAIR_LEN ? e - i - PAIR_LEN : 0; out->arity = count_top(s + i + SKIP_ONE, inner); out->targs = s + i + SKIP_ONE; @@ -3295,6 +3356,8 @@ static bool parse_seg(const char *s, size_t n, cs_seg_t *out) { } /* Split a dotted path (dots inside type-argument groups do not split). */ +/* A path whose brackets do not pair is not read at all: its last segment + * would be dropped and the reference resolved by the segments before it. */ static bool parse_path(const char *s, size_t n, cs_seg_t *segs, int *nsegs) { *nsegs = 0; size_t start = 0; @@ -3304,7 +3367,9 @@ static bool parse_path(const char *s, size_t n, cs_seg_t *segs, int *nsegs) { if (is_open_bracket(c)) { depth++; } else if (is_close_bracket(c)) { - depth--; + if (--depth < 0) { + return false; + } } else if (c == '.' && depth == 0) { if (*nsegs >= CS_MAX_SEGS || !parse_seg(s + start, i - start, &segs[*nsegs])) { return false; @@ -3313,7 +3378,7 @@ static bool parse_path(const char *s, size_t n, cs_seg_t *segs, int *nsegs) { start = i + SKIP_ONE; } } - return *nsegs > 0; + return depth == 0 && *nsegs > 0; } static const char *const CS_KEYWORD_TYPES[][2] = { @@ -3513,7 +3578,9 @@ static bool parse_params(cs_ref_t *r, const char *s, size_t from, size_t to) { if (is_open_bracket(c)) { depth++; } else if (is_close_bracket(c)) { - depth--; + if (--depth < 0) { + return false; /* a bracket that closes nothing */ + } } else if (c == ',' && depth == 0) { if (r->nparams >= CS_MAX_PARAMS) { return false; @@ -3533,7 +3600,9 @@ static bool parse_params(cs_ref_t *r, const char *s, size_t from, size_t to) { } } r->sig[w] = '\0'; - return true; + /* a bracket left open would have swallowed the last parameter: the + * reference would be matched without it */ + return depth == 0; } /* A written reference into its parts. false for one this code does not @@ -3611,7 +3680,12 @@ static bool parse_cref(const char *raw, cs_ref_t *r) { r->has_params = true; size_t close = group_end(s, n, paren); bool closed = close > paren && s[close - SKIP_ONE] == ')'; - if (!parse_params(r, s, paren + SKIP_ONE, closed ? close - SKIP_ONE : n)) { + /* a parameter list that is not closed, or text after it, is not + * matched by what can be read of it */ + for (size_t k = close; closed && k < n; k++) { + closed = isspace((unsigned char)s[k]) != 0; + } + if (!closed || !parse_params(r, s, paren + SKIP_ONE, close - SKIP_ONE)) { return false; } } @@ -5664,7 +5738,7 @@ static const char *build_files(cs_index_t *ix, const cbm_doclink_build_in_t *in) /* The graph node of every declaration. false when memory ran out. */ static bool build_nodes(cs_index_t *ix, const cbm_gbuf_t *g) { - cs_node_pass_t np = {.names = cbm_ht_create(CBM_SZ_256)}; + cs_node_pass_t np = {.names = cbm_ht_create(CBM_SZ_256), .sized = CBM_SZ_256}; cbm_arena_init(&np.keys); bool ok = np.names != NULL; for (int i = 0; ok && i < ix->nfiles; i++) { diff --git a/src/pipeline/doc_links_msbuild.c b/src/pipeline/doc_links_msbuild.c index b665a9f47c..ec3375709f 100644 --- a/src/pipeline/doc_links_msbuild.c +++ b/src/pipeline/doc_links_msbuild.c @@ -299,9 +299,12 @@ bool cbm_msb_add(cbm_msb_t *m, const char *rel_path, const char *scope) { } line = nl ? nl + SKIP_ONE : NULL; } - if (bad_group || import_group) { + if (bad_group || import_group || !f->readable) { /* A malformed group is unknown, never an empty closed scope. Keep - * this project conservative without rejecting other project files. */ + * this project conservative without rejecting other project files. + * So is a file the scan did not read (malformed, or larger than a + * project file): what it holds is unknown, not absent -- the + * projects that evaluate it have an open scope. */ f->readable = false; f->recs[0] = (msb_rec_t){.tag = 'Y'}; f->nrecs = 1; @@ -2495,8 +2498,8 @@ static bool st_enter(msb_eval_t *ev, int file, bool poison) { } ev->frames[ev->nframes - 1].owner = c; st_write(ev, (uint32_t)file + 1, domain, NULL); - if (!poison) - ev->unevaluable += !ev->m->files[file].readable; + /* a file that was not read is counted (and opens the scope) by its 'Y' + * record, as the walk reaches it */ return !ev->oom; } static void st_finish(msb_eval_t *ev, st_component_t *c) { @@ -2628,8 +2631,8 @@ static void msb_import(msb_eval_t *ev, int file, const msb_rec_t *r) { if (target >= 0 && !seen_has(ev, ev->m->files[target].rel_path) && msb_push(ev, target)) { msb_work(SKIP_ONE); cbm_ht_set(ev->seen, ev->m->files[target].rel_path, (void *)&MSB_SET_PRESENT); - /* a file that was not read (malformed, or too large) says nothing */ - ev->unevaluable += !ev->m->files[target].readable; + /* a file that was not read (malformed, or too large) is counted and + * opens the scope by its 'Y' record (cbm_msb_add) */ } } @@ -2660,7 +2663,6 @@ static void msb_pass1(msb_eval_t *ev, int file) { return; msb_work(SKIP_ONE); cbm_ht_set(ev->seen, rel, (void *)&MSB_SET_PRESENT); - ev->unevaluable += !ev->m->files[file].readable; } while (ev->nframes > base && !ev->oom) { /* an import pushes a frame, which may move the array: the frame is diff --git a/src/pipeline/pipeline_incremental.c b/src/pipeline/pipeline_incremental.c index ce150f1cc8..8ac3db44ba 100644 --- a/src/pipeline/pipeline_incremental.c +++ b/src/pipeline/pipeline_incremental.c @@ -1349,6 +1349,30 @@ static void row_dep_visitor(const char *key, void *value, void *userdata) { /* Files whose unresolved doc-link rows mention a removed name as an * identifier token. Keys are borrowed from `rows`. */ +/* A byte of an identifier as the resolvers read one: a letter, a digit, '_', + * and every byte of a multi-byte character (doc_links_cs.c ident_ok). */ +static bool doc_row_ident_byte(unsigned char c) { + return isalnum(c) || c == '_' || c >= CBM_SZ_128; +} + +/* True when the identifier raw[0, n) is one of the removed names -- of any + * length. When memory runs out the answer is yes: re-resolving a file that + * did not need it costs time, keeping a stale row costs correctness. */ +static bool doc_row_token_removed(const CBMHashTable *removed, const char *s, size_t n) { + char small[CBM_SZ_512]; + char *tok = n < sizeof(small) ? small : (char *)cbm_alloc(CBM_MEM_CLASS_OTHER, n + SKIP_ONE); + if (!tok) { + return true; + } + memcpy(tok, s, n); + tok[n] = '\0'; + bool hit = cbm_ht_get(removed, tok) != NULL; + if (tok != small) { + cbm_free(CBM_MEM_CLASS_OTHER, tok); + } + return hit; +} + static void doc_row_name_dependents(const cbm_doc_link_row_t *rows, int n, const CBMHashTable *removed, CBMHashTable *out_paths) { if (!removed || cbm_ht_count(removed) == 0) { @@ -1361,22 +1385,17 @@ static void doc_row_name_dependents(const cbm_doc_link_row_t *rows, int n, continue; } for (const char *p = raw; *p;) { - while (*p && !(isalnum((unsigned char)*p) || *p == '_')) { + while (*p && !doc_row_ident_byte((unsigned char)*p)) { p++; } const char *s = p; - while (*p && (isalnum((unsigned char)*p) || *p == '_')) { + while (*p && doc_row_ident_byte((unsigned char)*p)) { p++; } - char tok[CBM_SZ_512]; size_t tl = (size_t)(p - s); - if (tl > 0 && tl < sizeof(tok)) { - memcpy(tok, s, tl); - tok[tl] = '\0'; - if (cbm_ht_get(removed, tok)) { - cbm_ht_set(out_paths, rel, (void *)rel); - break; - } + if (tl > 0 && doc_row_token_removed(removed, s, tl)) { + cbm_ht_set(out_paths, rel, (void *)rel); + break; } } } diff --git a/tests/test_doc_mentions.c b/tests/test_doc_mentions.c index 25ddd9783a..e8f8272d04 100644 --- a/tests/test_doc_mentions.c +++ b/tests/test_doc_mentions.c @@ -4584,7 +4584,9 @@ static int dm_preview_bounds_checks(const char *db, const char *project) { yyjson_val *sample = yyjson_arr_get(samples, 0); yyjson_val *metadata = yyjson_obj_get(report, "samples_preview"); const char *note = yyjson_get_str(yyjson_obj_get(report, "samples_preview_note")); - bounded = dm_preview_status_is(env, "ok") && dm_preview_json_consistent(env) && + /* the row's reason is no reason this layer writes: the + * status is error (S12); its sample is still shown bounded */ + bounded = dm_preview_status_is(env, "error") && dm_preview_json_consistent(env) && yyjson_arr_size(samples) == 1 && yyjson_obj_size(sample) == 5 && yyjson_arr_size(metadata) == 4 && note && strstr(note, "doc_link_unresolved") && bounded; @@ -4606,7 +4608,7 @@ static int dm_preview_bounds_checks(const char *db, const char *project) { const char *metadata = table ? strstr(table, "\n samples_preview:") : NULL; size_t bytes = table ? (metadata ? (size_t)(metadata - table) : strlen(table)) : 0; bounded = table && metadata && bytes <= 5000 && - dm_preview_compact_status(env, "ok") && + dm_preview_compact_status(env, "error") && strstr(metadata, "\n samples_preview_note:") && strstr(table, "(cols: rel_path line syntax raw reason)") && bounded; for (int field = 0; field < 4; field++) { @@ -6236,7 +6238,8 @@ TEST(doc_mentions_msbuild_gate) { ASSERT_EQ(steps, strlen("")); /* a file that was not read is counted where it is evaluated, as the - * project file itself or as an import; it opens nobody's scope */ + * project file itself or as an import; what it holds is unknown, so the + * scope of the project that evaluates it is open (S16) */ const dm_project_file_t files[] = { {"eng/Huge.props", big}, {"eng/Broken.props", ""}, @@ -6250,7 +6253,7 @@ TEST(doc_mentions_msbuild_gate) { ASSERT_TRUE(dm_msb_eval(files, 3, "App.csproj", &res)); ASSERT_TRUE(dm_has_using(&res, 'n', "Still.Here")); ASSERT_EQ(res.unevaluable, 2); - ASSERT_FALSE(res.open); + ASSERT_TRUE(res.open); cbm_msb_result_free(&res); ASSERT_TRUE(dm_msb_eval(files, 2, "eng/Broken.props", &res)); ASSERT_EQ(res.count, 0); @@ -9102,6 +9105,331 @@ TEST(doc_mentions_cs_damaged_stored_scope) { PASS(); } +/* ── security review 2 ───────────────────────────────────────────── */ + +/* The portable copy of a scope writes 0 for every line number and is never + * longer than the scope it is made from: an empty line field stays empty + * (writing a 0 for it made the copy one byte longer per such field, past the + * end of its buffer). */ +TEST(doc_mentions_cs_portable_scope_bound) { + static const char blob[] = "cs1\n" + "R\t1\t0\t\t\tN\n" + "T\t1\t\t\tc\t-\tC\t\t\n" + "M\t\tc\t0\t0\tGo\t\t\n" + "X\t\t\n" + "X\t\t\n" + "X\t\t\n"; + char *portable = cbm_doclink_cs_portable_scope(blob); + ASSERT_NOT_NULL(portable); + size_t n = strlen(portable); + bool same = strcmp(portable, blob) == 0; + cbm_free(CBM_MEM_CLASS_OTHER, portable); + ASSERT_LTE(n, strlen(blob)); + ASSERT_TRUE(same); + PASS(); +} + +/* A reference whose brackets do not pair, or that goes on after its + * parameter list, is unparseable: it is never resolved by the part that + * can be read (the segments before the open bracket, the parameters before + * it). Decoys: the same references written whole bind. */ +TEST(doc_mentions_cs_unbalanced_brackets) { + static const dm_source_t files[] = { + {"src/Outer.cs", + "namespace Acme\n" + "{\n" + " public class Outer\n" + " {\n" + " public class Inner { }\n" + " public void M(int a) { }\n" + " public void M(int a, System.Collections.Generic.List b) { }\n" + " }\n" + "\n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " public class Broken { }\n" + "\n" + " /// \n" + " public class Whole { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"src/Outer.cs", "Outer.Inner{T", "Outer.Broken", NULL, "unparseable", "Outer.Outer", NULL}, + {"src/Outer.cs", "Outer.Inner}", "Outer.Broken", NULL, "unparseable", "Outer.Outer", NULL}, + {"src/Outer.cs", "Outer.M(int", "Outer.Broken", NULL, "unparseable", "Outer.Outer.M", NULL}, + {"src/Outer.cs", "Outer.M(int, List{string)", "Outer.Broken", NULL, "unparseable", + "Outer.Outer.M", NULL}, + {"src/Outer.cs", "Outer.M(int))", "Outer.Broken", NULL, "unparseable", "Outer.Outer.M", + NULL}, + {"src/Outer.cs", "Outer.M(int)x", "Outer.Broken", NULL, "unparseable", "Outer.Outer.M", + NULL}, + {"src/Outer.cs", "Outer.Inner{T}x", "Outer.Broken", NULL, "unparseable", + "Outer.Outer.Inner", NULL}, + {"src/Outer.cs", "Outer.Inner{T}", "Outer.Whole", "Outer.Outer.Inner", NULL, NULL, NULL}, + {"src/Outer.cs", "Outer.M(int)", "Outer.Whole", "Outer.Outer.M", NULL, NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("brackets", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +enum { DM_DEEP_SEGMENTS = 70 }; /* past the scanner's 64 namespace segments */ + +/* A scope read back from the store is held to the nesting its writer keeps: + * a region whose namespace has more segments than the scanner ever writes + * fails the build of the run (bad_scope, the error row), as any other stored + * scope this code did not write. The next run rebuilds everything. */ +TEST(doc_mentions_cs_stored_scope_nesting) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_deep_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + th_write_file(TH_PATH(repo, "src/A.cs"), "namespace N\n{\n public class Target { }\n}\n"); + th_write_file(TH_PATH(repo, "src/B.cs"), "namespace N\n" + "{\n" + " /// \n" + " public class Uses { }\n" + "}\n"); + char db[512]; + snprintf(db, sizeof(db), "%s/deep.db", tmp); + ASSERT_EQ(dm_index(repo, db, NULL), 0); + char props[512]; + int n = 0; + dm_edge(db, "B.Uses", "A.Target", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + /* A's region record names a namespace DM_DEEP_SEGMENTS segments deep */ + char sql[1024]; + size_t w = (size_t)snprintf(sql, sizeof(sql), + "UPDATE lsp_surface SET defs_json = replace(defs_json, " + "'\\tN\\n', '\\tN"); + for (int i = 1; i < DM_DEEP_SEGMENTS; i++) { + w += (size_t)snprintf(sql + w, sizeof(sql) - w, ".a"); + } + snprintf(sql + w, sizeof(sql) - w, + "\\n') WHERE rel_path = 'src/A.cs' AND instr(defs_json, '\\tN\\n') > 0"); + ASSERT_EQ(dm_exec(db, sql), 1); + th_write_file(TH_PATH(repo, "src/B.cs"), + "namespace N\n" + "{\n" + " /// Again \n" + " public class Uses { }\n" + "}\n"); + cbm_pipeline_incremental_test_reset_faults(); + ASSERT_EQ(dm_index(repo, db, NULL), 0); + int errors = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'"); + /* the next run rebuilds everything, and the edge is back */ + ASSERT_EQ(dm_index(repo, db, NULL), 0); + cbm_incremental_route_t route = cbm_pipeline_incremental_test_last_route(); + dm_edge(db, "B.Uses", "A.Target", props, sizeof(props), &n); + dm_unlink_db(db); + th_rmtree(tmp); + ASSERT_EQ(errors, 1); + ASSERT_EQ(route, CBM_INCREMENTAL_ROUTE_FORCED_FULL); + ASSERT_EQ(n, 1); + PASS(); +} + +/* One row whose reason this layer never writes, then index_status in both + * forms: no key carries that text, the row is counted under one fixed key, + * and the status is error. 0 when all of that holds. */ +static int dm_unknown_reason_checks(const char *db, const char *project) { + if (dm_exec(db, "INSERT INTO doc_link_unresolved(project, rel_path, line, syntax, raw, reason) " + "SELECT project, 'src/Injected.cs', 1, 'see', 'X', 'injected_reason_text' " + "FROM doc_link_unresolved LIMIT 1") != 1) { + return 1; + } + int bad = 0; + char *text = dm_index_status(project, false); + const char *block = text ? strstr(text, "doc_links:\n") : NULL; + if (!block || strstr(text, "injected_reason_text") || + !strstr(block, "\n unrecognized_reason: 1\n") || !strstr(block, "\n status: error\n")) { + printf(" index_status text: %s\n", block ? block : "(no doc_links block)"); + bad = 1; + } + free(text); + yyjson_doc *env = dm_preview_status(project, true); + yyjson_val *unresolved = yyjson_obj_get(dm_preview_report(env), "unresolved"); + if (!unresolved || yyjson_obj_get(unresolved, "injected_reason_text") || + yyjson_get_int(yyjson_obj_get(unresolved, "unrecognized_reason")) != 1 || + !dm_preview_status_is(env, "error")) { + printf(" index_status json: unknown reason shown, or status not error\n"); + bad = 1; + } + yyjson_doc_free(env); + return bad; +} + +TEST(doc_mentions_index_status_unknown_reason) { + ASSERT_EQ(dm_preview_fixture(dm_unknown_reason_checks), 0); + PASS(); +} + +enum { DM_HUGE_PROJECT = 1100000 }; /* past CSX_MAX_PROJECT_BYTES */ + +/* A project file the scan does not read -- larger than a project file is, + * as the project itself or as a file it imports -- holds what nobody knows, + * not nothing: a simple name found nowhere in such a project is external. + * Decoy: a project whose files were read keeps a closed scope (missing). */ +TEST(doc_mentions_msbuild_unread_project_opens) { + char *filler = malloc(DM_HUGE_PROJECT + 1); + ASSERT_NOT_NULL(filler); + memset(filler, 'x', DM_HUGE_PROJECT); + filler[DM_HUGE_PROJECT] = '\0'; + char *huge_project = + dm_repeated("\n"); + char *huge_props = dm_repeated("\n"); + free(filler); + ASSERT_NOT_NULL(huge_project); + ASSERT_NOT_NULL(huge_props); + const dm_source_t files[] = { + {"a/A.csproj", huge_project}, + {"a/UsesA.cs", "namespace App\n" + "{\n" + " /// \n" + " public class UsesA { }\n" + "}\n"}, + {"b/Directory.Build.props", huge_props}, + {"b/B.csproj", DM_EMPTY_PROJECT}, + {"b/UsesB.cs", "namespace App\n" + "{\n" + " /// \n" + " public class UsesB { }\n" + "}\n"}, + {"c/C.csproj", DM_EMPTY_PROJECT}, + {"c/UsesC.cs", "namespace App\n" + "{\n" + " /// \n" + " public class UsesC { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"a/UsesA.cs", "NowhereA", "UsesA.UsesA", NULL, "external", NULL, NULL}, + {"b/UsesB.cs", "NowhereB", "UsesB.UsesB", NULL, "external", NULL, NULL}, + {"c/UsesC.cs", "NowhereC", "UsesC.UsesC", NULL, "missing", NULL, NULL}, + }; + int bad = dm_check_repo("unread", files, DM_COUNT(files), wants, DM_COUNT(wants)); + free(huge_project); + free(huge_props); + ASSERT_EQ(bad, 0); + PASS(); +} + +enum { DM_BIG_MEMBERS = 20000, DM_SMALL_FILES = 200 }; + +/* What the resolver looks at to build its index over one file of + * DM_BIG_MEMBERS fields followed (in path order) by DM_SMALL_FILES files of + * one type and one field each. 0 when the repository cannot be indexed. */ +static uint64_t dm_scratch_work(void) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_scratch_XXXXXX"); + if (!cbm_mkdtemp(tmp)) { + return 0; + } + size_t cap = (size_t)DM_BIG_MEMBERS * 32 + 64; + char *src = malloc(cap); + if (!src) { + th_rmtree(tmp); + return 0; + } + size_t w = (size_t)snprintf(src, cap, "namespace Big\n{\n public class Wide\n {\n"); + for (int i = 0; i < DM_BIG_MEMBERS; i++) { + w += (size_t)snprintf(src + w, cap - w, " public int f%d;\n", i); + } + snprintf(src + w, cap - w, " }\n}\n"); + th_write_file(TH_PATH(tmp, "a/Big.cs"), src); + free(src); + for (int i = 0; i < DM_SMALL_FILES; i++) { + char rel[64]; + char text[256]; + snprintf(rel, sizeof(rel), "b/S%03d.cs", i); + snprintf(text, sizeof(text), + "namespace Small\n{\n%s public class C%03d { public int x; }\n}\n", + i == 0 ? " /// \n" : "", i); + th_write_file(TH_PATH(tmp, rel), text); + } + char db[512]; + snprintf(db, sizeof(db), "%s/scratch.db", tmp); + cbm_doclink_cs_test_work_reset(); + uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_scratch_work() : 0; + int rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"); + dm_unlink_db(db); + th_rmtree(tmp); + return rows == 1 ? work : 0; +} + +/* The per-file scratch table of the index build is not emptied at the size + * one large file grew it to: after a file of DM_BIG_MEMBERS names, every one + * of DM_SMALL_FILES small files costs about its own size. The work stays + * within a small multiple of all the names the files declare (emptying the + * large table for every small file cost DM_SMALL_FILES times its size). */ +TEST(doc_mentions_cs_scratch_table_work) { + uint64_t names = (uint64_t)DM_BIG_MEMBERS + (uint64_t)DM_SMALL_FILES * 2; + uint64_t work = dm_scratch_work(); + printf(" scratch table: %llu steps for %llu names\n", (unsigned long long)work, + (unsigned long long)names); + ASSERT_TRUE(work > 0); + ASSERT_LTE(work, names * 4); + PASS(); +} + +/* An incremental repair finds the files whose unresolved rows name what a + * changed file no longer declares by the identifiers the resolver reads -- + * a name with a non-ASCII letter included. The parts of `Hub` in two shared + * trees both declare the member `Grüße` (the reference is ambiguous); one of + * them stops declaring it, and the repaired index equals a full one (the + * reference binds). */ +TEST(doc_mentions_incremental_non_ascii_name) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_utf8name_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400]; + char db[512]; + char full_db[512]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + snprintf(db, sizeof(db), "%s/inc.db", tmp); + snprintf(full_db, sizeof(full_db), "%s/full.db", tmp); + th_write_file(TH_PATH(repo, "t1/X.cs"), + "namespace N\n{\n public partial class Hub { public void Gr\xC3\xBC\xC3\x9F" + "e() { } }\n}\n"); + th_write_file(TH_PATH(repo, "t2/Y.cs"), + "namespace N\n{\n public partial class Hub { public void Gr\xC3\xBC\xC3\x9F" + "e() { } public void Keep() { } }\n}\n"); + th_write_file(TH_PATH(repo, "p/P.csproj"), DM_EMPTY_PROJECT); + th_write_file(TH_PATH(repo, "p/Uses.cs"), + "namespace App\n" + "{\n" + " /// \n" + " public class Uses { }\n" + "}\n"); + ASSERT_EQ(dm_index(repo, db, NULL), 0); + char reason[64]; + dm_row(db, "p/Uses.cs", + "N.Hub.Gr\xC3\xBC\xC3\x9F" + "e", + reason, sizeof(reason), NULL, 0); + bool ambiguous = strcmp(reason, "ambiguous") == 0; + th_write_file(TH_PATH(repo, "t2/Y.cs"), + "namespace N\n{\n public partial class Hub { public void Keep() { } }\n}\n"); + int step = dm_step(repo, db, full_db, "a name with a non-ASCII letter removed", + CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR); + char props[512]; + int n = 0; + dm_edge(db, "Uses.Uses", + "X.Hub.Gr\xC3\xBC\xC3\x9F" + "e", + props, sizeof(props), &n); + dm_unlink_db(db); + dm_unlink_db(full_db); + th_rmtree(tmp); + ASSERT_TRUE(ambiguous); + ASSERT_EQ(step, 0); + ASSERT_EQ(n, 1); + PASS(); +} + SUITE(doc_mentions) { RUN_TEST(doc_mentions_extract_cs_tokens); RUN_TEST(doc_mentions_cs_scope_blob); @@ -9203,4 +9531,11 @@ SUITE(doc_mentions) { RUN_TEST(doc_mentions_msbuild_many_project_files); RUN_TEST(doc_mentions_cs_scan_nesting_limits); RUN_TEST(doc_mentions_cs_damaged_stored_scope); + RUN_TEST(doc_mentions_cs_portable_scope_bound); + RUN_TEST(doc_mentions_cs_unbalanced_brackets); + RUN_TEST(doc_mentions_cs_stored_scope_nesting); + RUN_TEST(doc_mentions_index_status_unknown_reason); + RUN_TEST(doc_mentions_msbuild_unread_project_opens); + RUN_TEST(doc_mentions_cs_scratch_table_work); + RUN_TEST(doc_mentions_incremental_non_ascii_name); } From b65ea09fa5bc10e7030d81c8e4a0d649cdccfe70 Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Sun, 4 Oct 2026 17:27:58 +0200 Subject: [PATCH 03/19] doc-links(cs): a shared doc comment counts once; lost doc-link data fails the layer (S4, S17) S4: the declarators of one C# field declaration share one doc comment. Its references are taken once, from the first declarator (the Field definition that carries them, or the first variable where there is none); the other declarators take nothing from it, so tokens, resolutions and rows grow with the references, not with references x declarators. A shared doc that could not be collected whole is no doc of any declarator and is not looked up again per declarator. S17: memory that runs out while a file's doc-link data is extracted -- a doc span or doc text, a reference value, a token, the scope scan -- is recorded on the file's doc links (CBMDocLinkArray.failed), and the resolving half fails the layer for the run (status error). A MENTIONS edge that could not be stored fails the layer too and is not counted. Seams: the scope scan (CBM_DOCLINK_ALLOC_SCOPE) and the edge insert. Tests: cs_shared_doc_once, alloc_failure_status; the shared-doc tests of the checkpoint follow the once rule. Signed-off-by: Martin Vogel --- internal/cbm/cbm.h | 5 + internal/cbm/doclink.c | 53 +++++----- internal/cbm/doclink.h | 1 + internal/cbm/doclink_cs.c | 19 +++- internal/cbm/extract_defs.c | 32 ++++-- src/pipeline/doc_links.c | 49 +++++++-- src/pipeline/doc_links.h | 3 + tests/test_doc_mentions.c | 191 +++++++++++++++++++++++++++++------- 8 files changed, 273 insertions(+), 80 deletions(-) diff --git a/internal/cbm/cbm.h b/internal/cbm/cbm.h index 4ed02f1ddd..65f3554176 100644 --- a/internal/cbm/cbm.h +++ b/internal/cbm/cbm.h @@ -592,6 +592,11 @@ typedef struct { CBMDocLink *items; int count; int cap; + /* Memory ran out while this file's doc-link data was extracted (a token, + * a reference value, a doc text or the scope blob was lost): the + * resolving half fails the layer instead of publishing a silently + * incomplete graph. */ + bool failed; } CBMDocLinkArray; // Full extraction result for one file. diff --git a/internal/cbm/doclink.c b/internal/cbm/doclink.c index 1b55ab792e..81d5a620f4 100644 --- a/internal/cbm/doclink.c +++ b/internal/cbm/doclink.c @@ -179,9 +179,7 @@ bool cbm_doclink_lang_supported(CBMLanguage lang) { typedef struct { const char *doc; uint32_t line; - int token_first; - int token_count; - bool parsed; + bool parsed; /* its references were taken, for the first definition that has it */ } doc_line_ent_t; typedef struct { @@ -303,6 +301,7 @@ void cbm_doclinks_push(CBMDocLinkArray *arr, CBMArena *a, CBMDocLink link) { grown = (CBMDocLink *)cbm_arena_alloc(a, (size_t)ncap * sizeof(*grown)); } if (!grown) { + arr->failed = true; /* the token is lost: the layer must not say ok */ return; } if (arr->count > 0) { @@ -381,36 +380,36 @@ void cbm_doclinks_extract(CBMExtractCtx *ctx) { if (twins && d->label && strcmp(d->label, L->twin_label) == 0) { doclink_twin_key_t key = {d->start_line, d->name}; if (bsearch(&key, twins, (size_t)twin_count, sizeof(twins[0]), twin_key_cmp)) { - continue; /* the Field twin carries these references */ + /* the Field twin carries these references -- and so those of + * the declarators after it, which share this doc text */ + doc_line_ent_t *taken = doc_info_of(ctx, d->docstring); + if (taken) { + taken->parsed = true; + } + continue; } } doc_line_ent_t *info = doc_info_of(ctx, d->docstring); uint32_t doc_line = info ? info->line : 0; - bool reusable = ctx->language == CBM_LANG_CSHARP && info && info->line > 0; - CBMDocLinkArray *links = &ctx->result->doc_links; - if (reusable && info->parsed) { - /* C# lexical tokens depend only on the shared text and its line. - * Bind each replay to this definition. Indices survive array - * growth; no token pointer is kept across a push. */ - for (int k = 0; k < info->token_count; k++) { - CBMDocLink link = links->items[info->token_first + k]; - link.source_qn = d->qualified_name; - link.def_line = d->start_line; - cbm_doclinks_push(links, ctx->arena, link); + bool csharp = ctx->language == CBM_LANG_CSHARP; + if (csharp && info && info->parsed) { + /* One C# doc comment documents every declarator of a field + * declaration (`int a, b, c;` share its text, extract_defs.c): its + * references are taken once, from the first declarator. Taken + * again per declarator, the tokens, their resolutions and the + * rows grow with references x declarators. */ + continue; + } + if (csharp) { + if (!cbm_doclink_cs_parse_doc_checked(ctx, d, d->docstring, + doc_line ? doc_line : d->start_line)) { + ctx->result->doc_links.failed = true; /* a reference value was lost */ } } else { - int first = links->count; - bool complete = false; - if (reusable) { - complete = cbm_doclink_cs_parse_doc_checked(ctx, d, d->docstring, doc_line); - } else { - L->parse_doc(ctx, d, d->docstring, doc_line ? doc_line : d->start_line); - } - if (reusable && complete) { - info->token_first = first; - info->token_count = links->count - first; - info->parsed = true; - } + L->parse_doc(ctx, d, d->docstring, doc_line ? doc_line : d->start_line); + } + if (info) { + info->parsed = true; } } /* The file's own doc: its references belong to the file. The parser gets diff --git a/internal/cbm/doclink.h b/internal/cbm/doclink.h index 960fdc329f..fe6b461f11 100644 --- a/internal/cbm/doclink.h +++ b/internal/cbm/doclink.h @@ -149,6 +149,7 @@ enum { CBM_DOCLINK_ALLOC_TEXT, CBM_DOCLINK_ALLOC_VALUE, CBM_DOCLINK_ALLOC_TOKENS, + CBM_DOCLINK_ALLOC_SCOPE, /* the C# scope scan, as if its builder ran out */ CBM_DOCLINK_ALLOC_KINDS, }; void cbm_doclink_test_fail_alloc_after(int kind, int nth); diff --git a/internal/cbm/doclink_cs.c b/internal/cbm/doclink_cs.c index 2e744069ae..14c8a7b039 100644 --- a/internal/cbm/doclink_cs.c +++ b/internal/cbm/doclink_cs.c @@ -2984,7 +2984,17 @@ const char *cbm_doclink_cs_scan_scope(CBMExtractCtx *ctx) { cs_emit_unplaced(&s, s.untrusted_row + TS_LINE_OFFSET, s.root_end_line); } cs_cost_publish(&s); +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (cbm_doclink_test_fail_alloc(CBM_DOCLINK_ALLOC_SCOPE)) { + s.failed = true; + } +#endif if (s.failed || s.sb.failed || !s.sb.buf) { + /* memory ran out: no scope would read as a file that declares + * nothing, so the layer is told */ + if (ctx->result) { + ctx->result->doc_links.failed = true; + } return NULL; } return s.sb.buf; @@ -3584,7 +3594,11 @@ const char *cbm_doclink_cs_project_scan_scope(CBMExtractCtx *ctx) { /* A project file past the size a project file has is not read either, * and its blob says so: what it holds is unknown, not absent. */ if (ctx->source_len > CSX_MAX_PROJECT_BYTES) { - return cbm_arena_strdup(ctx->arena, CBM_DOCLINK_CS_SCOPE_TAG "\nP\t\t>\n"); + const char *blob = cbm_arena_strdup(ctx->arena, CBM_DOCLINK_CS_SCOPE_TAG "\nP\t\t>\n"); + if (!blob && ctx->result) { + ctx->result->doc_links.failed = true; /* memory ran out */ + } + return blob; } /* a *.csproj marks its directory as a C# project even when it cannot be * read; a *.props or *.targets file has a blob only as an MSBuild @@ -3649,6 +3663,9 @@ const char *cbm_doclink_cs_project_scan_scope(CBMExtractCtx *ctx) { return NULL; } if (p.out.failed || p.value.failed || !p.out.buf) { + if (ctx->result) { + ctx->result->doc_links.failed = true; /* memory ran out, see above */ + } return NULL; } return p.out.buf; diff --git a/internal/cbm/extract_defs.c b/internal/cbm/extract_defs.c index 67a2e925f1..0621382d7e 100644 --- a/internal/cbm/extract_defs.c +++ b/internal/cbm/extract_defs.c @@ -1923,6 +1923,15 @@ static void doc_collect_kotlin(CBMExtractCtx *ctx, TSNode parent, TSNode anchor, } } +/* A doc comment was lost because memory ran out. For a language whose doc + * comments are read for links, the doc-link layer must not then report a + * complete graph (CBMDocLinkArray.failed). */ +static void doc_lost(CBMExtractCtx *ctx) { + if (ctx->result && cbm_doclink_lang_supported(ctx->language)) { + ctx->result->doc_links.failed = true; + } +} + /* Leading trivia of `anchor`, in source order. */ static void doc_collect_trivia(CBMExtractCtx *ctx, TSNode anchor, doc_trivia_t *t) { memset(t, 0, sizeof(*t)); @@ -1939,12 +1948,15 @@ static void doc_collect_trivia(CBMExtractCtx *ctx, TSNode anchor, doc_trivia_t * found = doc_collect_cursor(ctx, parent, anchor, t); } if (!found) { + bool failed = t->failed; memset(t, 0, sizeof(*t)); - return; - } - if (t->count == 0 && ctx->language == CBM_LANG_KOTLIN) { + t->failed = failed; + } else if (t->count == 0 && ctx->language == CBM_LANG_KOTLIN) { doc_collect_kotlin(ctx, parent, anchor, t); } + if (t->failed) { + doc_lost(ctx); + } } /* go/ast CommentGroup.Text: a line comment with no space after the slashes is @@ -2066,6 +2078,7 @@ static const char *doc_run_text(CBMExtractCtx *ctx, const doc_trivia_t *t, int f buf = (char *)cbm_arena_alloc(ctx->arena, total + SKIP_ONE); } if (!buf) { + doc_lost(ctx); return NULL; } size_t w = 0; @@ -7612,8 +7625,11 @@ static void extract_csharp_vars(CBMExtractCtx *ctx, TSNode node, CBMArena *a) { return; } /* All declarators have this field as their documentation anchor. Keep one - * immutable arena string, while each variable retains its own identity. - * Failed or absent collection keeps the original per-variable lookup. */ + * immutable arena string, while each variable retains its own identity: + * the doc-link driver takes the references of that one text once, from + * the first declarator. A text that could not be collected whole is no + * doc of any of them (doc_lost has marked the file's doc links failed); + * it is not looked up again per declarator. */ bool complete = false; const char *doc = extract_member_docstring_status(ctx, node, &complete); if (!complete) { @@ -7635,11 +7651,7 @@ static void extract_csharp_vars(CBMExtractCtx *ctx, TSNode node, CBMArena *a) { } if (!ts_node_is_null(id)) { const char *name = cbm_node_text(a, id, ctx->source); - if (doc) { - push_var_def_qn_doc(ctx, name, NULL, decl, doc); - } else { - push_var_def(ctx, name, decl); - } + push_var_def_qn_doc(ctx, name, NULL, decl, doc); } } } diff --git a/src/pipeline/doc_links.c b/src/pipeline/doc_links.c index 5de452e337..7c1af2bb67 100644 --- a/src/pipeline/doc_links.c +++ b/src/pipeline/doc_links.c @@ -342,9 +342,35 @@ static void mark_failed(cbm_doclinks_t *dl, const char *rel, const char *why) { } } +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +static _Atomic int doclinks_test_edge_fail_after; + +void cbm_doclinks_test_fail_edge_insert_after(int nth) { + atomic_store(&doclinks_test_edge_fail_after, nth > 0 ? nth : 0); +} +#endif + +/* Insert one MENTIONS edge; 0 when it could not be stored. */ +static int64_t doclinks_insert_edge(cbm_gbuf_t *gb, int64_t src, int64_t tgt, const char *props) { +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + int n = atomic_load(&doclinks_test_edge_fail_after); + while (n > 0) { + if (atomic_compare_exchange_weak(&doclinks_test_edge_fail_after, &n, n - SKIP_ONE)) { + if (n == SKIP_ONE) { + return 0; + } + break; + } + } +#endif + return cbm_gbuf_insert_edge(gb, src, tgt, "MENTIONS", props); +} + /* Emit one MENTIONS edge per (source, target): first line, its syntax, the - * mention count, and tier exact when any mention bound exactly. */ -static void emit_mentions(cbm_doclinks_t *dl, doclink_mention_t *m, int n, cbm_gbuf_t *edge_out) { + * mention count, and tier exact when any mention bound exactly. A failed + * insert fails the layer: the edge is not there. */ +static void emit_mentions(cbm_doclinks_t *dl, const char *rel, doclink_mention_t *m, int n, + cbm_gbuf_t *edge_out) { qsort(m, (size_t)n, sizeof(*m), mention_cmp); int i = 0; while (i < n) { @@ -360,16 +386,25 @@ static void emit_mentions(cbm_doclinks_t *dl, doclink_mention_t *m, int n, cbm_g "\"count\":%d}", cbm_doclink_syntax_name(m[i].syntax), exact ? "exact" : "unique", m[i].line, j - i); - cbm_gbuf_insert_edge(edge_out, m[i].src, m[i].tgt, "MENTIONS", props); - atomic_fetch_add_explicit(&dl->edges, 1, memory_order_relaxed); + if (doclinks_insert_edge(edge_out, m[i].src, m[i].tgt, props) == 0) { + mark_failed(dl, rel, "alloc"); /* an edge that is not there is not counted */ + } else { + atomic_fetch_add_explicit(&dl->edges, 1, memory_order_relaxed); + } i = j; } } void cbm_doclinks_resolve_file(cbm_doclinks_t *dl, int file_idx, const CBMFileResult *result, const cbm_gbuf_t *graph, cbm_gbuf_t *edge_out) { - if (!dl || !result || file_idx < 0 || file_idx >= dl->file_count || - result->doc_links.count == 0 || !result->doc_links.items) { + if (!dl || !result || file_idx < 0 || file_idx >= dl->file_count) { + return; + } + if (result->doc_links.failed) { + /* the extraction lost doc-link data of this file to memory */ + mark_failed(dl, dl->files[file_idx].rel_path, "extract"); + } + if (result->doc_links.count == 0 || !result->doc_links.items) { return; } int64_t started = doclinks_now_ns(); @@ -461,7 +496,7 @@ void cbm_doclinks_resolve_file(cbm_doclinks_t *dl, int file_idx, const CBMFileRe atomic_fetch_add_explicit(&dl->reasons[reason], 1, memory_order_relaxed); } if (nm > 0) { - emit_mentions(dl, mentions, nm, edge_out); + emit_mentions(dl, fi->rel_path, mentions, nm, edge_out); } cbm_free(CBM_MEM_CLASS_OTHER, mentions); atomic_fetch_add_explicit(&dl->resolve_ns, doclinks_now_ns() - started, memory_order_relaxed); diff --git a/src/pipeline/doc_links.h b/src/pipeline/doc_links.h index 40291d86a7..69fed4bd13 100644 --- a/src/pipeline/doc_links.h +++ b/src/pipeline/doc_links.h @@ -219,6 +219,9 @@ uint64_t cbm_doclink_cs_test_work(void); /* The buckets the index build walked to empty its per-file scratch table * (reset with the work above). */ uint64_t cbm_doclink_cs_test_scratch_work(void); +/* Test seam (doc_links.c): the nth MENTIONS edge insert from now on fails as + * if memory ran out (0: none). */ +void cbm_doclinks_test_fail_edge_insert_after(int nth); #endif /* True when `rel_path` is a scope input of some language (scope_input). */ diff --git a/tests/test_doc_mentions.c b/tests/test_doc_mentions.c index e8f8272d04..0969fdc6bc 100644 --- a/tests/test_doc_mentions.c +++ b/tests/test_doc_mentions.c @@ -6036,8 +6036,8 @@ TEST(doc_mentions_cs_scan_declarator_modifier_work) { PASS(); } -/* Sharing a large comment must not copy or parse its prose per declarator. - * Tokens remain per source; the measured work excludes that required output. */ +/* Sharing a large comment must not copy or parse its prose per declarator, + * and its references are taken once, from the first declarator (S4). */ TEST(doc_mentions_cs_shared_doc_work) { bool bounded = true; for (int n = 16; n <= 32; n *= 2) { @@ -6066,7 +6066,7 @@ TEST(doc_mentions_cs_shared_doc_work) { "cleaned=%llu tokens=%d\n", n, prose, len, (unsigned long long)copied, (unsigned long long)parse_input, (unsigned long long)cleaned, tokens); - ASSERT_EQ(tokens, n); + ASSERT_EQ(tokens, 1); bounded = bounded && copied <= (uint64_t)len * 2 && parse_input <= (uint64_t)len * 2 && cleaned <= 12; } @@ -6075,24 +6075,27 @@ TEST(doc_mentions_cs_shared_doc_work) { PASS(); } -/* A one-shot resource failure must not suppress later sources via sharing. */ +/* A failed allocation while a shared doc comment is read loses references: + * the file's doc links say so (S17), and the later declarators do not take + * the comment again -- neither from the shared text nor by a lookup of their + * own (S4). Without a failure the first declarator has every reference. */ TEST(doc_mentions_cs_shared_doc_allocation_failure) { static const struct { - int kind, nth, refs, expected[3]; + int kind, nth, refs; } cases[] = { - {CBM_DOCLINK_ALLOC_VALUE, 3, 2, {2, 1, 2}}, - {CBM_DOCLINK_ALLOC_TOKENS, 2, 10, {10, 9, 10}}, - {CBM_DOCLINK_ALLOC_TEXT, 2, 2, {2, 2, 2}}, - /* The second comment collection loses a span while growing past8. - * It must not become the shared doc for the later variables. */ - {CBM_DOCLINK_ALLOC_SPAN, 4, 12, {12, 12, 12}}, + {CBM_DOCLINK_ALLOC_KINDS, 0, 2}, /* no failure */ + {CBM_DOCLINK_ALLOC_VALUE, 2, 2}, {CBM_DOCLINK_ALLOC_TOKENS, 2, 20}, + {CBM_DOCLINK_ALLOC_TEXT, 1, 2}, {CBM_DOCLINK_ALLOC_SPAN, 2, 12}, }; bool ok = true; for (size_t k = 0; k < sizeof(cases) / sizeof(cases[0]); k++) { char *src = dm_repeated("class C {\n", "/// \n", cases[k].refs, "public int a,b,c;\n}\n"); ASSERT_NOT_NULL(src); - cbm_doclink_test_fail_alloc_after(cases[k].kind, cases[k].nth); + bool fail = cases[k].kind < CBM_DOCLINK_ALLOC_KINDS; + if (fail) { + cbm_doclink_test_fail_alloc_after(cases[k].kind, cases[k].nth); + } CBMFileResult *r = dm_extract(src, CBM_LANG_CSHARP, "Failure.cs"); cbm_doclink_test_reset_alloc(); free(src); @@ -6105,19 +6108,19 @@ TEST(doc_mentions_cs_shared_doc_allocation_failure) { ASSERT_TRUE(name[0] >= 'a' && name[0] <= 'c' && name[1] == '\0'); counts[name[0] - 'a']++; } + bool failed = r->doc_links.failed; cbm_free_result(r); - printf(" shared doc allocation: stage=%d counts=%d,%d,%d expected=%d,%d,%d\n", - cases[k].kind, counts[0], counts[1], counts[2], cases[k].expected[0], - cases[k].expected[1], cases[k].expected[2]); - for (int i = 0; i < 3; i++) { - ok = ok && counts[i] == cases[k].expected[i]; - } + printf(" shared doc allocation: stage=%d counts=%d,%d,%d failed=%d\n", cases[k].kind, + counts[0], counts[1], counts[2], failed); + bool first = fail ? counts[0] < cases[k].refs : counts[0] == cases[k].refs; + ok = ok && failed == fail && first && counts[1] == 0 && counts[2] == 0; } ASSERT_TRUE(ok); PASS(); } -/* Shared lexical tokens survive output growth, distinct comments and files. */ +/* Each shared comment yields its references once, from the first declarator + * of its own declaration -- distinct comments and files stay distinct. */ TEST(doc_mentions_cs_shared_doc_replay) { enum { REFS = 80 }; char *first = dm_repeated("class C {\n/// ", " ", REFS, @@ -6129,7 +6132,7 @@ TEST(doc_mentions_cs_shared_doc_replay) { CBMFileResult *r = dm_extract(src, CBM_LANG_CSHARP, "Shared.cs"); free(src); ASSERT_NOT_NULL(r); - ASSERT_EQ(r->doc_links.count, 6 * REFS); + ASSERT_EQ(r->doc_links.count, 2 * REFS); int counts[6] = {0}; for (int i = 0; i < r->doc_links.count; i++) { const CBMDocLink *link = &r->doc_links.items[i]; @@ -6147,16 +6150,14 @@ TEST(doc_mentions_cs_shared_doc_replay) { } cbm_free_result(r); for (int i = 0; i < 6; i++) { - ASSERT_EQ(counts[i], REFS); + ASSERT_EQ(counts[i], (i == 0 || i == 3) ? REFS : 0); } r = dm_extract("class C {\n/// \npublic int a,b,c;\n}\n", CBM_LANG_CSHARP, "Shared.cs"); ASSERT_NOT_NULL(r); - ASSERT_EQ(r->doc_links.count, 3); - for (int i = 0; i < r->doc_links.count; i++) { - ASSERT_STR_EQ(r->doc_links.items[i].raw, "Other"); - ASSERT_EQ(r->doc_links.items[i].line, 2); - } + ASSERT_EQ(r->doc_links.count, 1); + ASSERT_STR_EQ(r->doc_links.items[0].raw, "Other"); + ASSERT_EQ(r->doc_links.items[0].line, 2); cbm_free_result(r); PASS(); } @@ -6463,7 +6464,8 @@ static int dm_check_repo(const char *tag, const dm_source_t *files, int nfiles, #define DM_COUNT(a) ((int)(sizeof(a) / sizeof((a)[0]))) -/* Distinct variables documented together remain distinct MENTIONS sources. */ +/* Variables documented together by one comment: the comment's references + * are the first declarator's (S4); the other declarators mention nothing. */ TEST(doc_mentions_cs_shared_doc_sources) { static const char source[] = "namespace N {\n" "public class First { } public class Second { }\n" @@ -6475,7 +6477,7 @@ TEST(doc_mentions_cs_shared_doc_sources) { "}\n}\n"; CBMFileResult *r = dm_extract(source, CBM_LANG_CSHARP, "Shared.cs"); ASSERT_NOT_NULL(r); - ASSERT_EQ(r->doc_links.count, 6); + ASSERT_EQ(r->doc_links.count, 2); int pairs[3][2] = {{0}}; for (int i = 0; i < r->doc_links.count; i++) { const CBMDocLink *link = &r->doc_links.items[i]; @@ -6495,22 +6497,76 @@ TEST(doc_mentions_cs_shared_doc_sources) { cbm_free_result(r); for (int i = 0; i < 3; i++) { for (int j = 0; j < 2; j++) { - ASSERT_EQ(pairs[i][j], 1); + ASSERT_EQ(pairs[i][j], i == 0 ? 1 : 0); } } const dm_source_t files[] = {{"Shared.cs", source}}; static const dm_want_t wants[] = { - {"Shared.cs", "First", "a", "First", NULL, NULL, NULL}, - {"Shared.cs", "Second", "a", "Second", NULL, NULL, NULL}, - {"Shared.cs", "First", "b", "First", NULL, NULL, NULL}, - {"Shared.cs", "Second", "b", "Second", NULL, NULL, NULL}, - {"Shared.cs", "First", "c", "First", NULL, NULL, NULL}, - {"Shared.cs", "Second", "c", "Second", NULL, NULL, NULL}, + {"Shared.cs", "First", "C.a", "First", NULL, NULL, NULL}, + {"Shared.cs", "Second", "C.a", "Second", NULL, NULL, NULL}, + {"Shared.cs", "First", "C.b", NULL, NULL, "First", NULL}, + {"Shared.cs", "Second", "C.b", NULL, NULL, "Second", NULL}, + {"Shared.cs", "First", "C.c", NULL, NULL, "First", NULL}, + {"Shared.cs", "Second", "C.c", NULL, NULL, "Second", NULL}, }; ASSERT_EQ(dm_check_repo("shared_doc", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); PASS(); } +/* S4: tokens, resolutions and rows of a doc comment shared by the + * declarators of one field declaration grow with its references, not with + * references x declarators. Doubling both the declarators and the + * references doubles the tokens and the rows, no more. */ +static int dm_shared_rows(int declarators, int refs, int *tokens) { + size_t cap = (size_t)(declarators + refs) * 40 + 256; + char *src = malloc(cap); + if (!src) { + return -1; + } + size_t w = (size_t)snprintf(src, cap, "namespace N\n{\n public class C\n {\n ///"); + for (int i = 0; i < refs; i++) { + w += (size_t)snprintf(src + w, cap - w, " ", i); + } + w += (size_t)snprintf(src + w, cap - w, "\n public int v0"); + for (int i = 1; i < declarators; i++) { + w += (size_t)snprintf(src + w, cap - w, ", v%d", i); + } + snprintf(src + w, cap - w, ";\n }\n}\n"); + CBMFileResult *r = dm_extract(src, CBM_LANG_CSHARP, "Fields.cs"); + *tokens = r ? r->doc_links.count : -1; + cbm_free_result(r); + const dm_source_t files[] = {{"Fields.cs", src}}; + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_s4rows_XXXXXX"); + int rows = -1; + if (cbm_mkdtemp(tmp)) { + th_write_file(TH_PATH(tmp, files[0].path), files[0].text); + char db[512]; + snprintf(db, sizeof(db), "%s/rows.db", tmp); + if (dm_index(tmp, db, NULL) == 0) { + rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"); + } + dm_unlink_db(db); + th_rmtree(tmp); + } + free(src); + return rows; +} + +TEST(doc_mentions_cs_shared_doc_once) { + int tokens_small = 0; + int tokens_large = 0; + int rows_small = dm_shared_rows(8, 8, &tokens_small); + int rows_large = dm_shared_rows(16, 16, &tokens_large); + printf(" shared doc: 8 x 8 -> %d tokens, %d rows; 16 x 16 -> %d tokens, %d rows\n", + tokens_small, rows_small, tokens_large, rows_large); + ASSERT_EQ(tokens_small, 8); + ASSERT_EQ(rows_small, 8); + ASSERT_EQ(tokens_large, 16); + ASSERT_EQ(rows_large, 16); + PASS(); +} + /* The order of the scope levels: an inner namespace declaration's own usings * and aliases are asked before an outer namespace's types; a qualified name * and a using's target are relative to the namespaces around them before @@ -9430,6 +9486,69 @@ TEST(doc_mentions_incremental_non_ascii_name) { PASS(); } +/* S17: memory that runs out while doc-link data is extracted or published + * loses references, a scope or an edge -- the layer then says error, never + * ok over a silently thinner graph. One allocation fails at each point in + * turn (a doc span, a doc text, a reference value, a token, the scope scan, + * a MENTIONS edge); the run without a failure has no error row. */ +enum { DM_FAIL_EDGE = CBM_DOCLINK_ALLOC_KINDS }; + +static int dm_alloc_failure_errors(int point) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_s17_XXXXXX"); + if (!cbm_mkdtemp(tmp)) { + return -1; + } + th_write_file(TH_PATH(tmp, "src/A.cs"), "namespace N\n{\n public class Target { }\n}\n"); + th_write_file(TH_PATH(tmp, "src/B.cs"), "namespace N\n" + "{\n" + " /// \n" + " public class Uses\n" + " {\n" + " /// \n" + " public int a, b;\n" + " }\n" + "}\n"); + if (point == DM_FAIL_EDGE) { + cbm_doclinks_test_fail_edge_insert_after(1); + } else if (point >= 0) { + cbm_doclink_test_fail_alloc_after(point, 1); + } + char db[512]; + snprintf(db, sizeof(db), "%s/s17.db", tmp); + int errors = + dm_index(tmp, db, NULL) == 0 + ? dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'") + : -1; + cbm_doclinks_test_fail_edge_insert_after(0); + cbm_doclink_test_reset_alloc(); + dm_unlink_db(db); + th_rmtree(tmp); + return errors; +} + +TEST(doc_mentions_alloc_failure_status) { + static const struct { + int point; + const char *what; + } points[] = { + {CBM_DOCLINK_ALLOC_SPAN, "doc span"}, {CBM_DOCLINK_ALLOC_TEXT, "doc text"}, + {CBM_DOCLINK_ALLOC_VALUE, "value"}, {CBM_DOCLINK_ALLOC_TOKENS, "token"}, + {CBM_DOCLINK_ALLOC_SCOPE, "scope scan"}, {DM_FAIL_EDGE, "MENTIONS edge"}, + }; + int baseline = dm_alloc_failure_errors(-1); + bool ok = baseline == 0; + for (size_t i = 0; i < sizeof(points) / sizeof(points[0]); i++) { + int errors = dm_alloc_failure_errors(points[i].point); + if (errors != 1) { + printf(" a failed %s allocation: %d error rows, want 1\n", points[i].what, errors); + ok = false; + } + } + ASSERT_TRUE(ok); + PASS(); +} + SUITE(doc_mentions) { RUN_TEST(doc_mentions_extract_cs_tokens); RUN_TEST(doc_mentions_cs_scope_blob); @@ -9481,6 +9600,7 @@ SUITE(doc_mentions) { RUN_TEST(doc_mentions_cs_scan_branch_merge_work); RUN_TEST(doc_mentions_cs_scan_declarator_modifier_work); RUN_TEST(doc_mentions_cs_shared_doc_sources); + RUN_TEST(doc_mentions_cs_shared_doc_once); RUN_TEST(doc_mentions_cs_shared_doc_work); RUN_TEST(doc_mentions_cs_shared_doc_replay); RUN_TEST(doc_mentions_cs_shared_doc_allocation_failure); @@ -9538,4 +9658,5 @@ SUITE(doc_mentions) { RUN_TEST(doc_mentions_msbuild_unread_project_opens); RUN_TEST(doc_mentions_cs_scratch_table_work); RUN_TEST(doc_mentions_incremental_non_ascii_name); + RUN_TEST(doc_mentions_alloc_failure_status); } From 2358608976a83107cc05cd40d0e23b247e4176fb Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Sun, 4 Oct 2026 17:55:05 +0200 Subject: [PATCH 04/19] doc-links(cs): a scope this run wrote that does not parse costs only its file; blobs are UTF-8 (S8, S10) S8: a C# scope blob written in this run that the resolver cannot parse no longer fails the layer for the whole repository. What the file had put into the shared index is taken back (the namespaces it created and their keys, the declared and production marks it set, the quarantine keys it added), the file is kept as an empty scope marked rejected, and each of its references becomes a graph_gap row. The rejected files and references are counted (stats, and the log line doc_links.cs.rejected_scopes); the layer status stays ok. A STORED scope that does not parse still fails the run. The scanner's output passes the reader's record checks for the shapes the review names: a namespace name with an empty dotted segment is a namespace the scan cannot name. Seams: spoil the scope of one file in the run, and parse a blob with the resolver's reader. S10: every byte a scan writes into a scope blob is well-formed UTF-8, where it is written. In a C# file a name that is not UTF-8 is not placed (an identifier, a namespace or type name) and a text that is not is written as "?" (a kept text, a signature); in a project file every byte that starts no well-formed sequence is written as U+FFFD, in element names, attribute values and text, and a numeric reference to a surrogate code point is U+FFFD too. The stored bytes stay a function of the file alone, and the surface writer no longer fails the run on such a file. Tests: cs_rejected_scope_contained, cs_scanner_output_reads_back, cs_utf8_surfaces (the former observation test now requires the writer to succeed and the blob to be UTF-8). Signed-off-by: Martin Vogel --- internal/cbm/doclink_cs.c | 120 +++++++++++++++-- src/pipeline/doc_links.h | 5 + src/pipeline/doc_links_cs.c | 212 ++++++++++++++++++++++++++++-- tests/test_doc_mentions.c | 255 ++++++++++++++++++++++++++++++++++-- 4 files changed, 558 insertions(+), 34 deletions(-) diff --git a/internal/cbm/doclink_cs.c b/internal/cbm/doclink_cs.c index 14c8a7b039..5996b00033 100644 --- a/internal/cbm/doclink_cs.c +++ b/internal/cbm/doclink_cs.c @@ -122,6 +122,69 @@ static void sb_putu(cs_sb_t *sb, uint32_t v) { } } +/* ── Well-formed text ───────────────────────────────────────────── + * + * A scope blob is stored as a JSON string, and a JSON writer refuses a + * string that is not UTF-8: every byte the scans write is well-formed UTF-8. + * A name or a text of a C# source that is not is not kept (a name: the + * declaration is not placed; a text: "?"); project-file text gets U+FFFD for + * every byte that is no part of a well-formed sequence. Both are a function + * of the file's bytes alone. */ + +static const char CS_REPLACEMENT[] = "\xEF\xBF\xBD"; /* U+FFFD */ + +enum { + CS_UTF8_TAIL_MASK = 0xC0, + CS_UTF8_TAIL = 0x80, +}; + +/* Length of the well-formed UTF-8 sequence at s[0, n), 1 to 4; 0 when there + * is none: a stray continuation byte, an overlong form, a surrogate, a code + * point past U+10FFFF, a sequence cut short. */ +static size_t cs_utf8_len(const unsigned char *s, size_t n) { + if (n == 0) { + return 0; + } + unsigned char c = s[0]; + if (c < 0x80) { + return SKIP_ONE; + } + size_t len = 0; + if (c >= 0xC2 && c <= 0xDF) { + len = PAIR_LEN; + } else if (c >= 0xE0 && c <= 0xEF) { + len = 3; + } else if (c >= 0xF0 && c <= 0xF4) { + len = 4; + } + if (len == 0 || n < len) { + return 0; + } + unsigned char c1 = s[1]; + if ((c == 0xE0 && c1 < 0xA0) || (c == 0xED && c1 > 0x9F) || (c == 0xF0 && c1 < 0x90) || + (c == 0xF4 && c1 > 0x8F)) { + return 0; /* overlong, surrogate, or past U+10FFFF */ + } + for (size_t k = SKIP_ONE; k < len; k++) { + if ((s[k] & CS_UTF8_TAIL_MASK) != CS_UTF8_TAIL) { + return 0; + } + } + return len; +} + +/* True when s[0, n) is well-formed UTF-8. */ +static bool cs_utf8_ok(const char *s, size_t n) { + for (size_t i = 0; i < n;) { + size_t l = cs_utf8_len((const unsigned char *)s + i, n - i); + if (l == 0) { + return false; + } + i += l; + } + return true; +} + /* ── Doc-comment references ──────────────────────────────────────── */ static int cs_tag_syntax(const char *name, size_t len) { @@ -939,9 +1002,10 @@ static TSNode cs_type_params(TSNode decl) { /* Node text without whitespace and without verbatim '@' markers, appended to * sb. `global::` prefixes are dropped. Text that is too long to be a type - * name, or that holds a scope separator or non-whitespace control byte, is - * written as "?" (an unresolvable name: the declaring type then counts as - * having an open hierarchy). The whole field is replaced before any copy. */ + * name, that holds a scope separator or non-whitespace control byte, or that + * is not well-formed UTF-8, is written as "?" (an unresolvable name: the + * declaring type then counts as having an open hierarchy). The whole field is + * replaced before any copy. */ static void cs_put_text_nows(cs_scan_t *s, TSNode n) { const char *src = s->ctx->source; uint32_t a = ts_node_start_byte(n); @@ -955,7 +1019,7 @@ static void cs_put_text_nows(cs_scan_t *s, TSNode n) { ok = c != '|' && c != ';' && c != '{' && c != '}' && !((c < 0x20 && !isspace(c)) || c == 0x7f); } - if (!ok) { + if (!ok || !cs_utf8_ok(src + a, b - a)) { sb_putc(&s->sb, '?'); return; } @@ -978,8 +1042,9 @@ static void *cs_tmp_alloc(cs_scan_t *s, size_t n) { } /* Copy of source bytes [a, b) without whitespace and '@'. NULL unless it is a - * (dotted) identifier of sane length: an error-recovered parse can hand back a - * "name" spanning arbitrary code, which must not become a declaration. */ + * (dotted) identifier of sane length and well-formed UTF-8: an error-recovered + * parse can hand back a "name" spanning arbitrary code, which must not become + * a declaration, and a name that is not UTF-8 cannot be written. */ static char *cs_ident_dup(cs_scan_t *s, uint32_t a, uint32_t b) { const char *src = s->ctx->source; if (b <= a || b - a > CS_NAME_MAX) { @@ -1001,7 +1066,7 @@ static char *cs_ident_dup(cs_scan_t *s, uint32_t a, uint32_t b) { out[w++] = (char)c; } out[w] = '\0'; - return w > 0 ? out : NULL; + return w > 0 && cs_utf8_ok(out, w) ? out : NULL; } static char *cs_name_dup(cs_scan_t *s, TSNode n) { @@ -1073,6 +1138,9 @@ static void cs_put_sig(cs_scan_t *s, TSNode params) { uint32_t a = ts_node_start_byte(ty); uint32_t b = ts_node_end_byte(ty); cbm_doclink_cs_norm_type(src + a, (size_t)(b - a), norm, sizeof(norm)); + if (!cs_utf8_ok(norm, strlen(norm))) { + snprintf(norm, sizeof(norm), "?"); /* a type nothing is known about */ + } } if (!first) { sb_putc(&s->sb, '|'); @@ -1829,7 +1897,7 @@ static char *cs_namespace_name_dup(cs_scan_t *s, uint32_t a, uint32_t b) { i = cs_skip_space(src, i, b); if (i == b) { out[w] = '\0'; - return out; + return cs_utf8_ok(out, w) ? out : NULL; /* a name that cannot be written */ } if (src[i] != '.') { return NULL; @@ -1937,7 +2005,7 @@ static const char *cs_text_tparams(cs_scan_t *s, uint32_t *pos, uint32_t stop) { if (c == '>') { out[w] = '\0'; *pos = k + SKIP_ONE; - return out; + return cs_utf8_ok(out, w) ? out : NULL; /* names that cannot be written */ } } else if (cs_word_char((unsigned char)c)) { uint32_t e = cs_ident_end(src, k, limit, false); @@ -3301,9 +3369,12 @@ static void csx_put_char(cs_sb_t *sb, unsigned char c) { } } -/* A code point as UTF-8. */ +/* A code point as UTF-8. A surrogate (a numeric reference may name one) is + * no character: it is written as U+FFFD. */ static void csx_put_codepoint(cs_sb_t *sb, uint32_t cp) { - if (cp < 0x80) { + if (cp >= 0xD800 && cp <= 0xDFFF) { + sb_puts(sb, CS_REPLACEMENT); + } else if (cp < 0x80) { csx_put_char(sb, (unsigned char)cp); } else if (cp < 0x800) { sb_putc(sb, (char)(0xC0 | (cp >> 6))); @@ -3356,6 +3427,19 @@ static uint32_t csx_entity(const csx_t *x, uint32_t i, uint32_t end, uint32_t *c return 0; } +/* The bytes src[i, end) of a project file that start at i into sb: a + * well-formed UTF-8 sequence as it stands, a byte that starts none as + * U+FFFD. Returns the index past what was taken. */ +static uint32_t csx_put_utf8(cs_sb_t *sb, const char *src, uint32_t i, uint32_t end) { + size_t len = cs_utf8_len((const unsigned char *)src + i, (size_t)(end - i)); + if (len == 0) { + sb_puts(sb, CS_REPLACEMENT); + return i + SKIP_ONE; + } + sb_putn(sb, src + i, len); + return i + (uint32_t)len; +} + /* The text of `v`, entities decoded (unless `raw`), escaped. */ static void csx_put_text(cs_sb_t *sb, const csx_t *x, csx_span_t v, bool raw) { for (uint32_t i = v.s; i < v.e;) { @@ -3364,6 +3448,8 @@ static void csx_put_text(cs_sb_t *sb, const csx_t *x, csx_span_t v, bool raw) { if (past) { csx_put_codepoint(sb, cp); i = past; + } else if ((unsigned char)x->src[i] >= 0x80) { + i = csx_put_utf8(sb, x->src, i, v.e); } else { csx_put_char(sb, (unsigned char)x->src[i]); i++; @@ -3371,6 +3457,14 @@ static void csx_put_text(cs_sb_t *sb, const csx_t *x, csx_span_t v, bool raw) { } } +/* An element's name into sb, each byte that starts no well-formed UTF-8 + * sequence as U+FFFD (the tokenizer's names hold no separator). */ +static void csx_put_name(cs_sb_t *sb, const csx_t *x, csx_span_t name) { + for (uint32_t i = name.s; i < name.e;) { + i = csx_put_utf8(sb, x->src, i, name.e); + } +} + /* A field: a tab, then nothing for an absent attribute, else '=' and its text. */ static void csx_put_field(cs_sb_t *sb, const csx_t *x, csx_span_t v) { sb_putc(sb, '\t'); @@ -3435,7 +3529,7 @@ static void csx_put_property(csx_scan_t *p) { sb_putc(&p->out, 'V'); csx_put_field(&p->out, &p->x, t->attr[CSX_A_CONDITION]); sb_putc(&p->out, '\t'); - sb_putn(&p->out, p->x.src + t->name.s, t->name.e - t->name.s); + csx_put_name(&p->out, &p->x, t->name); sb_putc(&p->out, '\t'); if (p->prop_complex) { sb_putc(&p->out, '?'); @@ -3452,7 +3546,7 @@ static void csx_put_property(csx_scan_t *p) { static void csx_choose_child(csx_scan_t *p, const csx_tok_t *t) { if (p->choose_props >= 0 && p->depth == p->choose_props) { sb_puts(&p->out, "K\t"); - sb_putn(&p->out, p->x.src + t->name.s, t->name.e - t->name.s); + csx_put_name(&p->out, &p->x, t->name); sb_putc(&p->out, '\n'); } else if (csx_is(&p->x, t->name, "PropertyGroup") && p->choose_props < 0 && t->kind == CSX_OPEN) { diff --git a/src/pipeline/doc_links.h b/src/pipeline/doc_links.h index 69fed4bd13..9a6452a77a 100644 --- a/src/pipeline/doc_links.h +++ b/src/pipeline/doc_links.h @@ -222,6 +222,11 @@ uint64_t cbm_doclink_cs_test_scratch_work(void); /* Test seam (doc_links.c): the nth MENTIONS edge insert from now on fails as * if memory ran out (0: none). */ void cbm_doclinks_test_fail_edge_insert_after(int nth); +/* Test seams (doc_links_cs.c): spoil the scope of the file at `rel_path` + * (every record, then one the reader refuses; NULL or "": none), and ask the + * reader whether it takes a scope blob. */ +void cbm_doclink_cs_test_spoil_scope(const char *rel_path); +bool cbm_doclink_cs_test_scope_parses(const char *scope); #endif /* True when `rel_path` is a scope input of some language (scope_input). */ diff --git a/src/pipeline/doc_links_cs.c b/src/pipeline/doc_links_cs.c index c957841e7f..92ab111e6b 100644 --- a/src/pipeline/doc_links_cs.c +++ b/src/pipeline/doc_links_cs.c @@ -335,6 +335,9 @@ typedef struct { int *members_by_start; cs_tparam_t *tparams; /* by (owner, name) */ int ntparams; + /* Its scope was written in this run and this reader refuses it: the file + * declares nothing here, and its own references are graph gaps. */ + bool rejected; } cs_file_t; typedef struct { @@ -468,8 +471,33 @@ typedef enum { /* Counted while references are resolved (by every worker at once). */ typedef struct { _Atomic uint64_t ambiguous[CS_WHY_COUNT]; + _Atomic uint64_t rejected; /* references of files whose fresh scope was refused */ } cs_stats_t; +/* What parsing one scope set in the tables all files share: so that a scope + * written in this run that this reader refuses can be taken back whole, and + * costs only its own file. Reset for every file. */ +typedef struct { + int ns; + bool declared; + bool prod; +} cs_ns_was_t; + +typedef struct { + CBMHashTable *ht; + const char *key; +} cs_mark_was_t; + +typedef struct { + int nnss; /* namespaces before the file */ + cs_ns_was_t *flags; + int nflags; + int cap_flags; + cs_mark_was_t *marks; + int nmarks; + int cap_marks; +} cs_undo_t; + typedef struct { CBMArena arena; bool oom; /* an index allocation failed */ @@ -517,6 +545,8 @@ typedef struct { int nshared; /* shared trees */ cs_stats_t *stats; bool bind_inherited; /* CS_BIND_INHERITED */ + cs_undo_t undo; /* of the file being parsed */ + int rejected; /* files whose scope, written in this run, was refused */ } cs_index_t; #if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS @@ -627,11 +657,56 @@ static bool ht_mark(cs_index_t *ix, CBMHashTable *ht, const char *key) { return true; } +/* Grow an undo array to hold one more entry. false when memory ran out. */ +static bool undo_room(cs_index_t *ix, void **arr, int n, int *cap, size_t size) { + if (n < *cap) { + return true; + } + int ncap = *cap ? *cap * PAIR_LEN : CBM_SZ_16; + void *grown = cbm_alloc(CBM_MEM_CLASS_OTHER, (size_t)ncap * size); + if (!grown) { + ix->oom = true; + return false; + } + if (n > 0) { + memcpy(grown, *arr, (size_t)n * size); + } + cbm_free(CBM_MEM_CLASS_OTHER, *arr); + *arr = grown; + *cap = ncap; + return true; +} + +/* Remember the flags of namespace `ns` before the file's scope sets them. */ +static bool undo_note_ns(cs_index_t *ix, int ns) { + cs_undo_t *u = &ix->undo; + if (ns >= u->nnss) { + return true; /* made by this file: taken back as a whole */ + } + if (!undo_room(ix, (void **)&u->flags, u->nflags, &u->cap_flags, sizeof(cs_ns_was_t))) { + return false; + } + u->flags[u->nflags++] = + (cs_ns_was_t){.ns = ns, .declared = ix->nss[ns].declared, .prod = ix->nss[ns].prod}; + return true; +} + /* A type name declared in `f` at a place no namespace or outer type could be * established for. Product code never binds test-only declarations, so a name * only test files quarantine blocks references from test files only. */ static bool quarantine_name(cs_index_t *ix, const cs_file_t *f, const char *name) { - return ht_mark(ix, f->is_test ? ix->quarantine_test : ix->quarantine, name); + CBMHashTable *ht = f->is_test ? ix->quarantine_test : ix->quarantine; + if (cbm_ht_get(ht, name)) { + return true; + } + char *k = ix_strdup(ix, name); + cs_undo_t *u = &ix->undo; + if (!k || !undo_room(ix, (void **)&u->marks, u->nmarks, &u->cap_marks, sizeof(cs_mark_was_t))) { + return false; + } + cbm_ht_set(ht, k, (void *)k); + u->marks[u->nmarks++] = (cs_mark_was_t){.ht = ht, .key = k}; + return true; } static bool cs_is_test_path(const char *rel) { @@ -937,13 +1012,16 @@ static bool parse_region(cs_index_t *ix, cs_file_t *f, char **fld, int n, int *n } (*next_region)++; int ns = ns_make_path(ix, f->regions[parent].ns, fld[5]); - if (ns < 0) { + if (ns < 0 || !undo_note_ns(ix, ns)) { return false; } ix->nss[ns].declared = true; /* product code declares it, and with it every namespace above: one that * is marked has its upper ones marked already */ for (int up = ns; !f->is_test && up > 0 && !ix->nss[up].prod; up = ix->nss[up].parent) { + if (!undo_note_ns(ix, up)) { + return false; + } ix->nss[up].prod = true; } f->regions[id] = (cs_region_t){.parent = parent, @@ -5621,6 +5699,18 @@ static void cs_destroy(void *index) { if (ix->stats) { log_ambiguous(ix->stats); } + if (ix->rejected > 0) { + /* scopes written in this run that this reader refused: each cost + * only its own file, whose references are graph gaps */ + char files[CBM_SZ_32]; + char refs[CBM_SZ_32]; + snprintf(files, sizeof(files), "%d", ix->rejected); + snprintf(refs, sizeof(refs), "%llu", + (unsigned long long)(ix->stats ? atomic_load(&ix->stats->rejected) : 0)); + cbm_log_warn("doc_links.cs.rejected_scopes", "files", files, "references", refs); + } + cbm_free(CBM_MEM_CLASS_OTHER, ix->undo.flags); + cbm_free(CBM_MEM_CLASS_OTHER, ix->undo.marks); cbm_free(CBM_MEM_CLASS_OTHER, ix->stats); cbm_ht_free(ix->project_above); cbm_ht_free(ix->ns_by_key); @@ -5688,10 +5778,96 @@ static bool build_tables(cs_index_t *ix, const cbm_doclink_build_in_t *in) { return collect_projects(ix, in); } -/* Every file of the index with what its scope declares. Returns the path of - * a file whose scope is not one this code wrote (the index is not built over - * such a scope: nothing may be resolved around a declaration that is not - * known), "" when memory ran out, NULL when all is well. */ +/* Start the undo record of one file's scope. */ +static void undo_begin(cs_index_t *ix) { + ix->undo.nnss = ix->nnss; + ix->undo.nflags = 0; + ix->undo.nmarks = 0; +} + +/* Take back what the file's scope set in the shared tables: the names it + * quarantined, the flags it set on namespaces that were there before it, and + * the namespaces it made. */ +static void undo_scope(cs_index_t *ix) { + cs_undo_t *u = &ix->undo; + for (int i = u->nmarks - SKIP_ONE; i >= 0; i--) { + cbm_ht_delete(u->marks[i].ht, u->marks[i].key); + } + for (int i = u->nflags - SKIP_ONE; i >= 0; i--) { + ix->nss[u->flags[i].ns].declared = u->flags[i].declared; + ix->nss[u->flags[i].ns].prod = u->flags[i].prod; + } + for (int id = ix->nnss - SKIP_ONE; id >= u->nnss; id--) { + char key[CS_NAME_BUF + CBM_SZ_16]; + const cs_ns_t *n = &ix->nss[id]; + if (ns_key(key, sizeof(key), n->parent, n->name, strlen(n->name))) { + cbm_ht_delete(ix->ns_by_key, key); + } + } + ix->nnss = u->nnss; + u->nflags = 0; + u->nmarks = 0; +} + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +/* Test seam: the scope of the file at this path is spoiled -- every record + * it has, then one this reader refuses -- as if its writer and this reader + * disagreed after the file's declarations were read. "" or NULL: none. */ +static char cs_test_spoiled_path[CBM_SZ_512]; + +void cbm_doclink_cs_test_spoil_scope(const char *rel_path) { + snprintf(cs_test_spoiled_path, sizeof(cs_test_spoiled_path), "%s", rel_path ? rel_path : ""); +} + +static const char *cs_test_spoiled(cs_index_t *ix, const char *rel_path, const char *scope) { + if (!scope || !cs_test_spoiled_path[0] || strcmp(rel_path, cs_test_spoiled_path) != 0) { + return scope; + } + const char *spoiled = cbm_arena_sprintf(&ix->arena, "%sZ\tspoiled\n", scope); + return spoiled ? spoiled : scope; +} + +/* Test seam: true when this reader takes the scope blob `scope` (its record + * checks, on an index of its own). A test holds every blob the scanner + * writes against it. */ +bool cbm_doclink_cs_test_scope_parses(const char *scope) { + cs_index_t *ix = (cs_index_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*ix)); + if (!ix) { + return false; + } + cbm_arena_init(&ix->arena); + ix->ns_by_key = cbm_ht_create(CBM_SZ_64); + ix->quarantine = cbm_ht_create(CBM_SZ_64); + ix->quarantine_test = cbm_ht_create(CBM_SZ_64); + ix->nss = (cs_ns_t *)ix_zalloc(ix, CBM_SZ_256 * sizeof(cs_ns_t)); + bool ok = ix->ns_by_key && ix->quarantine && ix->quarantine_test && ix->nss; + if (ok) { + ix->nscap = CBM_SZ_256; + ix->nnss = SKIP_ONE; + ix->nss[0] = (cs_ns_t){.parent = CS_NONE, .name = ""}; + cs_file_t f = {.rel_path = "Scope.cs"}; + undo_begin(ix); + ok = parse_scope(ix, &f, scope) && !ix->oom; + } + cbm_ht_free(ix->ns_by_key); + cbm_ht_free(ix->quarantine); + cbm_ht_free(ix->quarantine_test); + cbm_free(CBM_MEM_CLASS_OTHER, ix->undo.flags); + cbm_free(CBM_MEM_CLASS_OTHER, ix->undo.marks); + cbm_arena_destroy(&ix->arena); + cbm_free(CBM_MEM_CLASS_OTHER, ix); + return ok; +} +#endif + +/* Every file of the index with what its scope declares. A scope read back + * from the store that this reader refuses is not one this code wrote: the + * index is not built over it (nothing may be resolved around a declaration + * that is not known), and its path is returned. A scope written in this run + * that the reader refuses is the scanner's and this reader's disagreement, + * which costs only its own file: what its parse set is taken back, the file + * declares nothing, and its references are graph gaps (counted, logged). + * Returns "" when memory ran out, NULL when all is well. */ static const char *build_files(cs_index_t *ix, const cbm_doclink_build_in_t *in) { CBMHashTable *dir_unit = cbm_ht_create(CBM_SZ_1K); const char *bad = dir_unit ? NULL : ""; @@ -5704,9 +5880,23 @@ static const char *build_files(cs_index_t *ix, const cbm_doclink_build_in_t *in) f->is_test = cs_is_test_path(src->rel_path); /* a project file declares nothing: its blob went to the evaluator */ const char *scope = cbm_msb_is_project_scope(src->scope) ? NULL : src->scope; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + scope = cs_test_spoiled(ix, src->rel_path, scope); +#endif + undo_begin(ix); if (scope && !parse_scope(ix, f, scope)) { - bad = ix->oom ? "" : src->rel_path; - break; + bool fresh = src->run_file >= 0 && src->run_file < in->run_file_count; + if (ix->oom || !fresh) { + bad = ix->oom ? "" : src->rel_path; + break; + } + undo_scope(ix); + *f = (cs_file_t){.rel_path = f->rel_path, + .module_qn = f->module_qn, + .is_test = f->is_test, + .rejected = true}; + ix->rejected++; + scope = NULL; } if (!scope) { /* no scope: an empty file (only its own region) */ @@ -5887,6 +6077,12 @@ static void cs_resolve(const void *index, int run_file, const CBMDocLink *link, * declaration. Both are declared-but-unplaced, i.e. graph gaps. */ int file = ix->run_to_file[run_file]; const cs_file_t *f = &ix->files[file]; + if (f->rejected) { + /* its scope was refused: nothing is known around the reference */ + out->reason = CBM_DOCLINK_REASON_GRAPH_GAP; + atomic_fetch_add_explicit(&ix->stats->rejected, 1, memory_order_relaxed); + return; + } bool file_doc = (link->flags & CBM_DOCLINK_FLAG_FILE) != 0; if ((!file_doc && line_unplaced(f, link->def_line)) || names_quarantined(ix, f, &r)) { out->reason = CBM_DOCLINK_REASON_GRAPH_GAP; diff --git a/tests/test_doc_mentions.c b/tests/test_doc_mentions.c index 0969fdc6bc..bcad1ed4f2 100644 --- a/tests/test_doc_mentions.c +++ b/tests/test_doc_mentions.c @@ -814,10 +814,16 @@ TEST(doc_mentions_cs_control_publication) { PASS(); } -/* Temporary observation: extract real source bytes, then exercise the public - * surface codec with complete definitions and with each input isolated. */ +/* S10: extract real source bytes, then run the surface writer of the run + * with the file's complete definitions, without its scope, and with only + * its scope. Every run must succeed, twice alike, and the stored `dl` must + * be the portable scope. `expect`: 1 the inserted bytes stay in the scope + * (well-formed UTF-8), 0 they do not (malformed), -1 not asked (an entity). */ +static bool dm_scope_utf8_ok(const char *scope); + static bool dm_utf8_probe_surface(CBMFileResult *bad, CBMFileResult *healthy, CBMLanguage language, - const char *path, const char *label, const char *bytes) { + const char *path, const char *label, const char *bytes, + int expect) { CBMFileResult *cache[] = {bad, healthy}; cbm_file_info_t files[] = {{.rel_path = (char *)path, .language = language}, {.rel_path = "Healthy.cs", .language = CBM_LANG_CSHARP}}; @@ -865,20 +871,33 @@ static bool dm_utf8_probe_surface(CBMFileResult *bad, CBMFileResult *healthy, CB int scope_rc = cbm_lsp_surface_build_rows(NULL, "p", cache, files, 2, NULL, NULL, &rows, &count); bool retained = saved_scope && strstr(saved_scope, bytes) != NULL; - printf("doc_utf8_probe case=%s scope=%d retained=%d defs=%d full_rc=%d full_count=%d " - "nodl_rc=%d nodl_count=%d scope_rc=%d scope_count=%d repeat=%d dl_equal=%d\n", - label, saved_scope != NULL, retained, def_count, full_rc, full_count, nodl_rc, - nodl_count, scope_rc, count, repeat, dl_equal); + bool written = full_rc == 0 && full_count == 2 && again_rc == 0 && dl_equal && nodl_rc == 0 && + nodl_count == 2 && scope_rc == 0 && count == 2; + bool kept = expect < 0 || retained == (expect == 1); + bool utf8 = saved_scope && dm_scope_utf8_ok(saved_scope); + bool correct = saved_scope != NULL && def_count > 0 && repeat && written && kept && utf8; + if (!correct) { + printf("doc_utf8_probe case=%s scope=%d retained=%d utf8=%d defs=%d full_rc=%d " + "full_count=%d nodl_rc=%d nodl_count=%d scope_rc=%d scope_count=%d repeat=%d " + "dl_equal=%d\n", + label, saved_scope != NULL, retained, utf8, def_count, full_rc, full_count, nodl_rc, + nodl_count, scope_rc, count, repeat, dl_equal); + } cbm_store_free_lsp_surfaces(rows, count); cbm_free(CBM_MEM_CLASS_OTHER, portable); free(defs); free(modules[0]); free(modules[1]); cbm_arena_destroy(&arena); - return saved_scope != NULL && def_count > 0 && repeat; + return correct; } -TEST(doc_mentions_cs_utf8_observation) { +/* S10: a malformed byte sequence in any place a scope keeps text -- a name + * or a text of a C# source, an element, a value or an attribute of a project + * file, a numeric reference to a surrogate -- leaves a well-formed scope, and + * the surface writer of the run succeeds. Well-formed sequences stay as they + * are; malformed ones do not reach the scope. */ +TEST(doc_mentions_cs_utf8_surfaces) { static const struct { const char *label, *bytes; } sequences[] = { @@ -927,7 +946,6 @@ TEST(doc_mentions_cs_utf8_observation) { int healthy_count = 0; int healthy_rc = cbm_lsp_surface_build_rows(NULL, "p", &healthy, &healthy_file, 1, NULL, NULL, &healthy_rows, &healthy_count); - printf("doc_utf8_probe_healthy rc=%d count=%d\n", healthy_rc, healthy_count); correct = correct && healthy_rc == 0 && healthy_count == 1; cbm_store_free_lsp_surfaces(healthy_rows, healthy_count); for (size_t p = 0; p < sizeof(positions) / sizeof(positions[0]); p++) { @@ -947,8 +965,9 @@ TEST(doc_mentions_cs_utf8_observation) { CBMFileResult *bad = cbm_extract_file(source, (int)(a + b + c + d), positions[p].language, "p", path, 0, NULL, NULL); snprintf(label, sizeof(label), "%s/%s", positions[p].label, sequences[s].label); + bool well_formed = s < 4; /* ascii, valid2, valid3, valid4 */ bool setup = bad && dm_utf8_probe_surface(bad, healthy, positions[p].language, path, - label, sequences[s].bytes); + label, sequences[s].bytes, well_formed); correct = setup && correct; cbm_free_result(bad); } @@ -968,7 +987,7 @@ TEST(doc_mentions_cs_utf8_observation) { snprintf(label, sizeof(label), "entity_%s/%zu", position == 0 ? "property" : "using", e); bool setup = bad && dm_utf8_probe_surface(bad, healthy, CBM_LANG_XML, "App.csproj", - label, entities[e]); + label, entities[e], -1); correct = setup && correct; cbm_free_result(bad); } @@ -9527,6 +9546,214 @@ static int dm_alloc_failure_errors(int point) { return errors; } +/* True when s[0, n) is well-formed UTF-8 (no surrogate, nothing past + * U+10FFFF, no overlong form, nothing cut short). */ +static bool dm_utf8_ok(const char *s, size_t n) { + const unsigned char *u = (const unsigned char *)s; + for (size_t i = 0; i < n;) { + unsigned char c = u[i]; + size_t len = c < 0x80 ? 1 + : (c >= 0xC2 && c <= 0xDF) ? 2 + : (c >= 0xE0 && c <= 0xEF) ? 3 + : (c >= 0xF0 && c <= 0xF4) ? 4 + : 0; + if (len == 0 || i + len > n) { + return false; + } + if (len > 1) { + unsigned char c1 = u[i + 1]; + if ((c == 0xE0 && c1 < 0xA0) || (c == 0xED && c1 > 0x9F) || (c == 0xF0 && c1 < 0x90) || + (c == 0xF4 && c1 > 0x8F)) { + return false; + } + for (size_t k = 1; k < len; k++) { + if ((u[i + k] & 0xC0) != 0x80) { + return false; + } + } + } + i += len; + } + return true; +} + +static bool dm_scope_utf8_ok(const char *scope) { + return dm_utf8_ok(scope, strlen(scope)); +} + +/* S8, S9, S10: every scope blob the scanners write passes the reader's own + * record checks, is a C string of the length that was built, and is + * well-formed UTF-8 -- for namespace names with empty segments, and for + * control bytes and malformed UTF-8 in every place a C# scope keeps text. + * A project file's blob is well-formed UTF-8 for malformed bytes and for + * numeric references to surrogates. */ +TEST(doc_mentions_cs_scanner_output_reads_back) { + static const struct { + const char *label, *bytes; + size_t n; + } bytes[] = { + {"nul", "\x00", 1}, + {"soh", "\x01", 1}, + {"us", "\x1F", 1}, + {"del", "\x7F", 1}, + {"continuation", "\x80", 1}, + {"overlong", "\xC0\xAF", 2}, + {"surrogate", "\xED\xA0\x80", 3}, + {"above", "\xF4\x90\x80\x80", 4}, + {"truncated", "\xE2\x82", 2}, + {"valid", "\xC3\xA9", 2}, + }; + static const struct { + const char *label, *before, *after; + } places[] = { + {"using_target", "using Acme.", "Name;\nclass Local {}\n"}, + {"alias_target", "using Alias = Acme.", "Name;\nclass Local {}\n"}, + {"namespace_name", "namespace N", "Name { class Local {} }\n"}, + {"type_name", "class N", "Name {}\n"}, + {"method_name", "class Local { void N", "Name() {} }\n"}, + {"parameter_type", "class Local { void M(N", "Name arg) {} }\n"}, + {"type_parameter", "class Local {}\n"}, + {"method_type_parameter", "class Local { void M() {} }\n"}, + {"field_name", "class Local { int N", "Name; }\n"}, + {"base_type", "class Local : Acme.N", "Name {}\n"}, + {"broken_file", "class Local { void M( { N", "Name } }\n"}, + }; + bool ok = true; + for (size_t p = 0; p < sizeof(places) / sizeof(places[0]); p++) { + for (size_t b = 0; b < sizeof(bytes) / sizeof(bytes[0]); b++) { + char source[512]; + size_t a = strlen(places[p].before); + size_t c = strlen(places[p].after); + memcpy(source, places[p].before, a); + memcpy(source + a, bytes[b].bytes, bytes[b].n); + memcpy(source + a + bytes[b].n, places[p].after, c); + size_t len = a + bytes[b].n + c; + source[len] = '\0'; + cbm_doclink_cs_test_cost_reset(); + CBMFileResult *r = + cbm_extract_file(source, (int)len, CBM_LANG_CSHARP, "p", "Bad.cs", 0, NULL, NULL); + const char *scope = r ? r->doc_scope : NULL; + size_t built = (size_t)cbm_doclink_cs_test_scope_bytes(); + bool good = scope && strlen(scope) == built && dm_utf8_ok(scope, strlen(scope)) && + cbm_doclink_cs_test_scope_parses(scope); + if (!good) { + printf(" %s/%s: scope=%d length=%zu built=%zu utf8=%d reads=%d\n", places[p].label, + bytes[b].label, scope != NULL, scope ? strlen(scope) : (size_t)0, built, + scope ? dm_utf8_ok(scope, strlen(scope)) : 0, + scope ? cbm_doclink_cs_test_scope_parses(scope) : 0); + ok = false; + } + cbm_free_result(r); + } + } + for (size_t i = 0; i < sizeof(dm_namespace_cases) / sizeof(dm_namespace_cases[0]); i++) { + for (int file_scoped = 0; file_scoped < 2; file_scoped++) { + char source[2048]; + if (!dm_namespace_source(source, sizeof(source), i, file_scoped != 0)) { + ok = false; + continue; + } + CBMFileResult *r = dm_extract(source, CBM_LANG_CSHARP, "Bad.cs"); + const char *scope = r ? r->doc_scope : NULL; + if (!scope || !cbm_doclink_cs_test_scope_parses(scope)) { + printf(" namespace %s/%s: the reader refuses the scope\n", + dm_namespace_cases[i].id, file_scoped ? "file" : "block"); + ok = false; + } + cbm_free_result(r); + } + } + static const char *xml[] = { + "

N\x80Name

", + "Value", + "

N�Name�

", + "", + "

V

" + "
", + "V" + "", + }; + for (size_t i = 0; i < sizeof(xml) / sizeof(xml[0]); i++) { + CBMFileResult *r = cbm_extract_file(xml[i], (int)strlen(xml[i]), CBM_LANG_XML, "p", + "App.csproj", 0, NULL, NULL); + const char *scope = r ? r->doc_scope : NULL; + if (!scope || !dm_utf8_ok(scope, strlen(scope))) { + printf(" project file %zu: blob=%d utf8=%d\n", i, scope != NULL, + scope ? dm_utf8_ok(scope, strlen(scope)) : 0); + ok = false; + } + cbm_free_result(r); + } + ASSERT_TRUE(ok); + PASS(); +} + +/* S8: a scope written in this run that the reader refuses costs only its own + * file. The status stays ok and the other file's edges are there; the file's + * own reference is a graph gap; and nothing its parse set before the refusal + * stays -- a namespace it declares does not stand beside a type of that name + * (Shadow, Good.Lurker), and a name it quarantined does not block another + * file's declaration (Hidden2). */ +TEST(doc_mentions_cs_rejected_scope_contained) { + static const dm_source_t files[] = { + {"Bad.cs", "namespace Shadow\n" + "{\n" + " public class Inner { }\n" + "}\n" + "namespace Good\n" + "{\n" + " namespace Lurker { public class Deep { } }\n" + "\n" + " /// \n" + " public class FromBad { }\n" + "}\n" + "namespace .Broken\n" + "{\n" + " public class Hidden2 { }\n" + "}\n"}, + {"Healthy.cs", "public class Shadow { }\n" + "namespace Good\n" + "{\n" + " public class Target { }\n" + " public class Lurker { }\n" + " public class Hidden2 { }\n" + "\n" + " /// \n" + " /// \n" + " public class Uses { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"Healthy.cs", "Target", "Healthy.Uses", "Healthy.Target", NULL, NULL, NULL}, + {"Healthy.cs", "Shadow", "Healthy.Uses", "Healthy.Shadow", NULL, NULL, NULL}, + {"Healthy.cs", "Lurker", "Healthy.Uses", "Healthy.Lurker", NULL, NULL, NULL}, + {"Healthy.cs", "Hidden2", "Healthy.Uses", "Healthy.Hidden2", NULL, NULL, NULL}, + {"Bad.cs", "Target", "Bad.FromBad", NULL, "graph_gap", "Healthy.Target", NULL}, + }; + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_rejected_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + for (int i = 0; i < DM_COUNT(files); i++) { + th_write_file(TH_PATH(tmp, files[i].path), files[i].text); + } + char db[512]; + snprintf(db, sizeof(db), "%s/rejected.db", tmp); + cbm_doclink_cs_test_spoil_scope("Bad.cs"); + int bad = dm_index(tmp, db, NULL) == 0 ? 0 : -1; + cbm_doclink_cs_test_spoil_scope(NULL); + int errors = bad == 0 ? dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved " + "WHERE reason = 'error'") + : -1; + for (int i = 0; bad >= 0 && i < DM_COUNT(wants); i++) { + bad += dm_want_failed(db, &wants[i]); + } + dm_unlink_db(db); + th_rmtree(tmp); + ASSERT_EQ(errors, 0); + ASSERT_EQ(bad, 0); + PASS(); +} + TEST(doc_mentions_alloc_failure_status) { static const struct { int point; @@ -9557,7 +9784,7 @@ SUITE(doc_mentions) { RUN_TEST(doc_mentions_cs_namespace_boundaries); RUN_TEST(doc_mentions_cs_control_scopes); RUN_TEST(doc_mentions_cs_control_publication); - RUN_TEST(doc_mentions_cs_utf8_observation); + RUN_TEST(doc_mentions_cs_utf8_surfaces); RUN_TEST(doc_mentions_cs_namespace_publication); RUN_TEST(doc_mentions_cs_scope_parse_errors); RUN_TEST(doc_mentions_cs_norm_type); @@ -9659,4 +9886,6 @@ SUITE(doc_mentions) { RUN_TEST(doc_mentions_cs_scratch_table_work); RUN_TEST(doc_mentions_incremental_non_ascii_name); RUN_TEST(doc_mentions_alloc_failure_status); + RUN_TEST(doc_mentions_cs_scanner_output_reads_back); + RUN_TEST(doc_mentions_cs_rejected_scope_contained); } From 2848ce77879f538ce767b661e19c6c1e3e1cb289 Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Sun, 4 Oct 2026 17:59:12 +0200 Subject: [PATCH 05/19] doc-links(msbuild): a group's condition is evaluated once, not per child (S1, evaluator side) The checkpoint writes an 's condition once into the project blob (S1), but the evaluator still evaluated it again for every import of the group, and an 's condition again for every of the group: the product imports (or items) x condition length that S1 removed from the blob came back as evaluator work. An 's condition is now evaluated once, at the group's first import, which is where MSBuild evaluates it: a property the group's first import sets no longer takes the group's later imports away. A legacy blob (the condition copied into every import) is still evaluated per import. An 's condition is evaluated once per group for the items of one pass; the final properties the items see do not change while they are read, so the result is the same. Tests: msbuild_import_group_condition_work and msbuild_item_group_condition_work (doubling both the children and the condition length at most doubles the evaluator's work counter; it quadrupled), msbuild_import_group_condition_once. Signed-off-by: Martin Vogel --- src/pipeline/doc_links_msbuild.c | 63 +++++++++++--- tests/test_doc_mentions.c | 137 +++++++++++++++++++++++++++++++ 2 files changed, 189 insertions(+), 11 deletions(-) diff --git a/src/pipeline/doc_links_msbuild.c b/src/pipeline/doc_links_msbuild.c index ec3375709f..cf17503dcd 100644 --- a/src/pipeline/doc_links_msbuild.c +++ b/src/pipeline/doc_links_msbuild.c @@ -557,6 +557,12 @@ typedef struct { st_component_t *owner; msb_tri_t group; /* the condition of the being read */ const msb_rec_t *hgroup; /* the being read */ + /* The being read: its condition's text (one string the + * group's imports share) and what it came to. MSBuild evaluates a + * group's condition once, where the group stands; evaluating it again + * for every import cost imports x condition length. */ + const char *igroup_text; + msb_tri_t igroup; } msb_frame_t; enum { MSB_SEEDS = 5, MSB_LIVE_OWNERS = 5 }; @@ -2609,9 +2615,10 @@ static void msb_add_item(msb_eval_t *ev, const msb_rec_t *group, const msb_rec_t msb_work(sizeof(msb_item_t)); } -/* An standing in the file on top of the frame stack. */ -static void msb_import(msb_eval_t *ev, int file, const msb_rec_t *r) { - msb_tri_t c = tri_and(msb_cond(ev, file, r->f[0]), msb_cond(ev, file, r->f[1])); +/* An standing in the file on top of the frame stack, under its + * 's condition `group` (MSB_TRUE outside a group). */ +static void msb_import(msb_eval_t *ev, int file, const msb_rec_t *r, msb_tri_t group) { + msb_tri_t c = tri_and(group, msb_cond(ev, file, r->f[1])); if (c == MSB_FALSE || r->f[3]) { return; /* not taken; or an SDK's file, which is no file of the repository */ } @@ -2704,9 +2711,20 @@ static void msb_pass1(msb_eval_t *ev, int file) { case 'C': ev->unevaluable++; break; - case 'I': - msb_import(ev, at, r); + case 'I': { + /* the group's condition once per group: its imports share the + * text (cbm_msb_add); a legacy blob's copies are evaluated each */ + msb_tri_t group = MSB_TRUE; + if (r->f[0]) { + if (fr->igroup_text != r->f[0]) { + fr->igroup_text = r->f[0]; + fr->igroup = msb_cond(ev, at, r->f[0]); + } + group = fr->igroup; + } + msb_import(ev, at, r, group); /* may push a frame: fr is not used after it */ break; + } default: break; } @@ -2846,12 +2864,32 @@ static int target_cmp(const void *a, const void *b) { return strcmp(((const cbm_msb_using_t *)a)->target, ((const cbm_msb_using_t *)b)->target); } +/* An 's condition for the items of one pass over them. MSBuild + * evaluates it once, where the group stands, and the final properties the + * items see do not change while they are read: evaluating it again for + * every item cost items x condition length. */ +typedef struct { + const msb_rec_t *group; + msb_tri_t value; +} msb_group_memo_t; + +static msb_tri_t item_group_cond(msb_eval_t *ev, const msb_item_t *it, msb_group_memo_t *memo) { + if (!it->group) { + return MSB_TRUE; + } + if (memo->group != it->group) { + memo->group = it->group; + memo->value = msb_cond(ev, it->file, it->group->f[0]); + } + return memo->value; +} + /* One item with the final properties: into `inc` or `rem`. */ -static void msb_using(msb_eval_t *ev, const msb_item_t *it, msb_ulist_t *inc, msb_ulist_t *rem) { +static void msb_using(msb_eval_t *ev, const msb_item_t *it, msb_ulist_t *inc, msb_ulist_t *rem, + msb_group_memo_t *memo) { const msb_rec_t *r = it->item; msb_record(); - msb_tri_t c = tri_and(msb_cond(ev, it->file, it->group ? it->group->f[0] : NULL), - msb_cond(ev, it->file, r->f[0])); + msb_tri_t c = tri_and(item_group_cond(ev, it, memo), msb_cond(ev, it->file, r->f[0])); if (c == MSB_FALSE) { return; } @@ -3111,8 +3149,9 @@ bool cbm_msb_eval(const cbm_msb_t *m, const char *project_rel, cbm_msb_result_t for (int i = 0; implicit && implicit[i]; i++) { ulist_add(&ev, &inc, 'n', "", implicit[i]); } + msb_group_memo_t memo = {0}; for (int i = 0; i < ev.nitems; i++) { - msb_using(&ev, &ev.items[i], &inc, &rem); + msb_using(&ev, &ev.items[i], &inc, &rem, &memo); /* All four expansions remain live until aliases and targets * have been copied into the persistent evaluation arena. */ cbm_arena_reset(&ev.scratch); @@ -3435,8 +3474,9 @@ static bool apply_targets(cbm_msb_eval_context_t *context, msb_eval_t *ev, int r static void eval_items(msb_eval_t *ev, const msb_eval_t *owner, msb_ulist_t *inc, msb_ulist_t *rem) { + msb_group_memo_t memo = {0}; for (int i = 0; !ev->oom && i < owner->nitems; i++) { - msb_using(ev, &owner->items[i], inc, rem); + msb_using(ev, &owner->items[i], inc, rem, &memo); cbm_arena_reset(&ev->scratch); } } @@ -3682,6 +3722,7 @@ bool cbm_msb_eval_context_eval(cbm_msb_eval_context_t *context, const char *proj static void st_eval_rope(msb_eval_t *ev, st_rope_t *rope, msb_ulist_t *inc, msb_ulist_t *rem) { st_rope_t *stack[CBM_SZ_64]; int depth = 0; + msb_group_memo_t memo = {0}; while (!ev->oom && (rope || depth)) { if (!rope) { rope = stack[--depth]; @@ -3689,7 +3730,7 @@ static void st_eval_rope(msb_eval_t *ev, st_rope_t *rope, msb_ulist_t *inc, msb_ } msb_work(1); if (!rope->left) { - msb_using(ev, &rope->item, inc, rem); + msb_using(ev, &rope->item, inc, rem, &memo); cbm_arena_reset(&ev->scratch); rope = NULL; } else { diff --git a/tests/test_doc_mentions.c b/tests/test_doc_mentions.c index bcad1ed4f2..e5fb2024ae 100644 --- a/tests/test_doc_mentions.c +++ b/tests/test_doc_mentions.c @@ -3483,6 +3483,140 @@ TEST(doc_mentions_msbuild_import_group_roundtrip) { PASS(); } +/* The evaluator side of the same product: an 's condition is + * evaluated once, where the group stands, not again for every import in it. + * Doubling both the imports and the condition's length at most doubles the + * evaluator's work; evaluated per import, it would quadruple. */ +TEST(doc_mentions_msbuild_import_group_condition_work) { + enum { IMPORTS = 100, TERMS = 100, CAP = 64 * 1024 }; + /* Flag is defined (an undefined property is unknown); no term holds */ + static const char term[] = "'$(Flag)' == 'aaaaaaaaaaaaaaaa'"; + uint64_t work[2] = {0}; + char *xml = malloc(CAP); + ASSERT_NOT_NULL(xml); + for (int sample = 0; sample < 2; sample++) { + int imports = IMPORTS << sample; + int terms = TERMS << sample; + size_t used = (size_t)snprintf(xml, CAP, + "b" + ""); + for (int i = 0; i < imports; i++) { + used += (size_t)snprintf(xml + used, CAP - used, ""); + } + used += (size_t)snprintf(xml + used, CAP - used, + "" + ""); + ASSERT_LT(used, (size_t)CAP); + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(dm_msb_add_xml(m, "App.csproj", xml)); + ASSERT_TRUE(dm_msb_add_xml( + m, "X.props", + "")); + cbm_msb_test_cost_reset(); + cbm_msb_result_t r; + ASSERT_TRUE(cbm_msb_eval(m, "App.csproj", &r)); + work[sample] = cbm_msb_test_work(); + ASSERT_TRUE(dm_has_using(&r, 'n', "Kept")); + ASSERT_FALSE(dm_has_using(&r, 'n', "Not.Taken")); + ASSERT_EQ(r.count, 1); + ASSERT_FALSE(r.open); + ASSERT_EQ(r.unevaluable, 0); + cbm_msb_result_free(&r); + cbm_msb_free(m); + } + free(xml); + fprintf(stderr, "msbuild import group condition work: %llu -> %llu\n", + (unsigned long long)work[0], (unsigned long long)work[1]); + ASSERT_GT(work[0], 0); + ASSERT_LTE(work[1], 2 * work[0]); + PASS(); +} + +/* The same for an : its condition is evaluated once for the + * items of a pass, not again for every in it. */ +TEST(doc_mentions_msbuild_item_group_condition_work) { + enum { USINGS = 100, TERMS = 100, CAP = 64 * 1024 }; + static const char term[] = "'$(Flag)' == 'aaaaaaaaaaaaaaaa'"; + uint64_t work[2] = {0}; + char *xml = malloc(CAP); + ASSERT_NOT_NULL(xml); + for (int sample = 0; sample < 2; sample++) { + int usings = USINGS << sample; + int terms = TERMS << sample; + size_t used = (size_t)snprintf(xml, CAP, + "b" + ""); + for (int i = 0; i < usings; i++) { + used += + (size_t)snprintf(xml + used, CAP - used, "", i); + } + used += (size_t)snprintf(xml + used, CAP - used, + "" + ""); + ASSERT_LT(used, (size_t)CAP); + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(dm_msb_add_xml(m, "App.csproj", xml)); + cbm_msb_test_cost_reset(); + cbm_msb_result_t r; + ASSERT_TRUE(cbm_msb_eval(m, "App.csproj", &r)); + work[sample] = cbm_msb_test_work(); + ASSERT_TRUE(dm_has_using(&r, 'n', "Kept")); + ASSERT_EQ(r.count, 1); + ASSERT_FALSE(r.open); + ASSERT_EQ(r.unevaluable, 0); + cbm_msb_result_free(&r); + cbm_msb_free(m); + } + free(xml); + fprintf(stderr, "msbuild item group condition work: %llu -> %llu\n", + (unsigned long long)work[0], (unsigned long long)work[1]); + ASSERT_GT(work[0], 0); + ASSERT_LTE(work[1], 2 * work[0]); + PASS(); +} + +/* MSBuild evaluates an 's condition once, before its imports: + * a property the group's first import sets does not take the second import + * away. */ +TEST(doc_mentions_msbuild_import_group_condition_once) { + const dm_project_file_t files[] = { + /* a property no file defines is unknown (the SDK or the + * environment may set it): Stop is defined first */ + {"App.csproj", "no" + "" + "" + "" + "" + ""}, + {"First.props", "yes" + ""}, + {"Second.props", + ""}, + {"Third.props", + ""}, + }; + cbm_msb_result_t r; + ASSERT_TRUE(dm_msb_eval(files, 4, "App.csproj", &r)); + ASSERT_TRUE(dm_has_using(&r, 'n', "From.First")); + ASSERT_TRUE(dm_has_using(&r, 'n', "From.Second")); + /* a later group with the same text is evaluated where it stands */ + ASSERT_FALSE(dm_has_using(&r, 'n', "From.Third")); + ASSERT_EQ(r.count, 2); + ASSERT_FALSE(r.open); + cbm_msb_result_free(&r); + PASS(); +} + TEST(doc_mentions_msbuild_legacy_import_blob) { cbm_msb_t *m = cbm_msb_new(); ASSERT_NOT_NULL(m); @@ -9837,6 +9971,9 @@ SUITE(doc_mentions) { RUN_TEST(doc_mentions_msbuild_blob); RUN_TEST(doc_mentions_msbuild_import_group_blob_growth); RUN_TEST(doc_mentions_msbuild_import_group_roundtrip); + RUN_TEST(doc_mentions_msbuild_import_group_condition_work); + RUN_TEST(doc_mentions_msbuild_item_group_condition_work); + RUN_TEST(doc_mentions_msbuild_import_group_condition_once); RUN_TEST(doc_mentions_msbuild_legacy_import_blob); RUN_TEST(doc_mentions_msbuild_bad_import_group_blob); RUN_TEST(doc_mentions_msbuild_imports); From 7f3f80c379dc165bf5d51f46e3eb6891b2b6e248 Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Sun, 4 Oct 2026 18:10:47 +0200 Subject: [PATCH 06/19] doc-links(cs): test the project-file scan's allocation failure too (S17) The review names two scans whose failure returned no scope: the C# scope scan and the project-file scan. Both record the failure on the file's doc links since e775dcd9, but only the C# scan had a failure seam. The project-file scan gets one (CBM_DOCLINK_ALLOC_PROJECT), and the status test gets a seventh point: a project file whose scan ran out of memory fails the layer (one error row), like the other six. Test: alloc_failure_status (its repository now has a project file). Signed-off-by: Martin Vogel --- internal/cbm/doclink.h | 3 ++- internal/cbm/doclink_cs.c | 5 +++++ tests/test_doc_mentions.c | 5 ++++- 3 files changed, 11 insertions(+), 2 deletions(-) diff --git a/internal/cbm/doclink.h b/internal/cbm/doclink.h index fe6b461f11..bb3ffa46f4 100644 --- a/internal/cbm/doclink.h +++ b/internal/cbm/doclink.h @@ -149,7 +149,8 @@ enum { CBM_DOCLINK_ALLOC_TEXT, CBM_DOCLINK_ALLOC_VALUE, CBM_DOCLINK_ALLOC_TOKENS, - CBM_DOCLINK_ALLOC_SCOPE, /* the C# scope scan, as if its builder ran out */ + CBM_DOCLINK_ALLOC_SCOPE, /* the C# scope scan, as if its builder ran out */ + CBM_DOCLINK_ALLOC_PROJECT, /* the project-file scan, likewise */ CBM_DOCLINK_ALLOC_KINDS, }; void cbm_doclink_test_fail_alloc_after(int kind, int nth); diff --git a/internal/cbm/doclink_cs.c b/internal/cbm/doclink_cs.c index 5996b00033..23db08f5a2 100644 --- a/internal/cbm/doclink_cs.c +++ b/internal/cbm/doclink_cs.c @@ -3756,6 +3756,11 @@ const char *cbm_doclink_cs_project_scan_scope(CBMExtractCtx *ctx) { } else if (!is_project) { return NULL; } +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (cbm_doclink_test_fail_alloc(CBM_DOCLINK_ALLOC_PROJECT)) { + p.out.failed = true; + } +#endif if (p.out.failed || p.value.failed || !p.out.buf) { if (ctx->result) { ctx->result->doc_links.failed = true; /* memory ran out, see above */ diff --git a/tests/test_doc_mentions.c b/tests/test_doc_mentions.c index e5fb2024ae..aec90228f1 100644 --- a/tests/test_doc_mentions.c +++ b/tests/test_doc_mentions.c @@ -9652,6 +9652,8 @@ static int dm_alloc_failure_errors(int point) { if (!cbm_mkdtemp(tmp)) { return -1; } + th_write_file(TH_PATH(tmp, "src/App.csproj"), + "\n"); th_write_file(TH_PATH(tmp, "src/A.cs"), "namespace N\n{\n public class Target { }\n}\n"); th_write_file(TH_PATH(tmp, "src/B.cs"), "namespace N\n" "{\n" @@ -9895,7 +9897,8 @@ TEST(doc_mentions_alloc_failure_status) { } points[] = { {CBM_DOCLINK_ALLOC_SPAN, "doc span"}, {CBM_DOCLINK_ALLOC_TEXT, "doc text"}, {CBM_DOCLINK_ALLOC_VALUE, "value"}, {CBM_DOCLINK_ALLOC_TOKENS, "token"}, - {CBM_DOCLINK_ALLOC_SCOPE, "scope scan"}, {DM_FAIL_EDGE, "MENTIONS edge"}, + {CBM_DOCLINK_ALLOC_SCOPE, "scope scan"}, {CBM_DOCLINK_ALLOC_PROJECT, "project scan"}, + {DM_FAIL_EDGE, "MENTIONS edge"}, }; int baseline = dm_alloc_failure_errors(-1); bool ok = baseline == 0; From 015c326f8cc9b7503425277213a14682ef4c739a Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Sun, 4 Oct 2026 18:30:17 +0200 Subject: [PATCH 07/19] doc-links: lint-ci clean again (seam calls only in seam builds; escape check type) make -f Makefile.cbm lint-ci failed on code of the checkpoint: - doc_links_msbuild.c: outside a seam build, msb_fail_value_alloc and msb_fail_prop_insert were stubs that return false, so cppcheck reported every condition that calls them as always false or always true (knownConditionTrueFalse, five sites). The calls are now compiled only with CBM_ENABLE_TEST_SEAMS, as the layer's other seams are, and the stubs are gone. - mcp.c: in one configuration of vendored yyjson.h's fallback typedef chain cppcheck takes uint32_t for unsigned short and reports the bounds of the U+E0000 tag block as out of range (compareValueOutOfTypeRangeError). The preview's escape check takes unsigned long, which the standard makes at least 32 bits wide. The product binary is byte-identical before and after. Signed-off-by: Martin Vogel --- src/mcp/mcp.c | 6 +++++- src/pipeline/doc_links_msbuild.c | 31 ++++++++++++++++++++----------- 2 files changed, 25 insertions(+), 12 deletions(-) diff --git a/src/mcp/mcp.c b/src/mcp/mcp.c index ba4a2c41ff..892c56ece1 100644 --- a/src/mcp/mcp.c +++ b/src/mcp/mcp.c @@ -6275,7 +6275,11 @@ static size_t doc_link_scalar(const unsigned char *text, size_t length, uint32_t return width; } -static bool doc_link_visible_escape(uint32_t scalar) { +/* `unsigned long` (at least 32 bits by the standard), not uint32_t: the + * vendored yyjson.h has a fallback typedef chain for uint32_t, and an + * analysis that cannot evaluate it takes uint32_t for a 16-bit type and the + * tag block's bounds for out of range. */ +static bool doc_link_visible_escape(unsigned long scalar) { return scalar < 0x20 || scalar == 0x7F || (scalar >= 0x200B && scalar <= 0x200F) || (scalar >= 0x202A && scalar <= 0x202E) || (scalar >= 0x2066 && scalar <= 0x2069) || (scalar >= 0xE0000 && scalar <= 0xE007F); diff --git a/src/pipeline/doc_links_msbuild.c b/src/pipeline/doc_links_msbuild.c index cf17503dcd..dfb5bd6990 100644 --- a/src/pipeline/doc_links_msbuild.c +++ b/src/pipeline/doc_links_msbuild.c @@ -497,12 +497,6 @@ static void msb_work(uint64_t n) { (void)n; } static void msb_record(void) {} -static bool msb_fail_value_alloc(void) { - return false; -} -static bool msb_fail_prop_insert(void) { - return false; -} static bool msb_fail_item_operation(cbm_msb_item_fail_operation_t operation) { (void)operation; return false; @@ -710,7 +704,11 @@ static char *scratch_strndup(msb_eval_t *ev, const char *s, size_t n) { * old buffers simultaneously until copying has finished and the old one is * released; neither a replacement nor an unknown value retains history. */ static char *value_alloc(msb_eval_t *ev, size_t n) { - if (n > SIZE_MAX - ev->value_bytes || msb_fail_value_alloc()) { + bool fail = n > SIZE_MAX - ev->value_bytes; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + fail = fail || msb_fail_value_alloc(); +#endif + if (fail) { ev->oom = true; return NULL; } @@ -786,7 +784,10 @@ static void msb_set(msb_eval_t *ev, const char *name, const char *value) { if (!k) { return; } - if (!msb_fail_prop_insert()) { +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (!msb_fail_prop_insert()) +#endif + { msb_work(SKIP_ONE); cbm_ht_set(ev->props, k, p); } @@ -1354,9 +1355,13 @@ static st_value_t *st_make_value(msb_eval_t *ev, const char *text) { if (!text) return NULL; size_t bytes = strlen(text) + 1; - st_value_t *v = msb_fail_value_alloc() - ? NULL - : (st_value_t *)cbm_alloc(CBM_MEM_CLASS_OTHER, sizeof(*v) + bytes); + st_value_t *v = NULL; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (!msb_fail_value_alloc()) +#endif + { + v = (st_value_t *)cbm_alloc(CBM_MEM_CLASS_OTHER, sizeof(*v) + bytes); + } if (!v) { ev->oom = true; return NULL; @@ -1419,11 +1424,13 @@ static void st_write(msb_eval_t *ev, uint32_t key, int domain, const char *text) ev->reuse_nodes = ev->view.root[domain]; effect = st_union(ev, leaf, ev->builder->writes.root[domain], false, domain, phase); ev->reuse_nodes = saved_reuse; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS bool is_new = !st_get(ev->builder->writes.root[domain], key); if (domain == ST_PROPS && is_new && msb_fail_prop_insert()) { st_drop(ev->state, effect); effect = st_hold(ev->builder->writes.root[domain]); } +#endif if (!st_get(effect, key)) ev->oom = true; } @@ -1431,10 +1438,12 @@ static void st_write(msb_eval_t *ev, uint32_t key, int domain, const char *text) st_node_t *view = !ev->oom ? st_union(ev, leaf, ev->view.root[domain], false, domain, phase) : NULL; ev->reuse_nodes = saved_reuse; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS if (!ev->builder && domain == ST_PROPS && !old && msb_fail_prop_insert()) { st_drop(ev->state, view); view = st_hold(ev->view.root[domain]); } +#endif if (!st_get(view, key)) ev->oom = true; if (!ev->oom) { From 9d985f0f6450910eb19c6705de4e13f91fd733c4 Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Sun, 4 Oct 2026 18:50:32 +0200 Subject: [PATCH 08/19] doc-links(cs): a name is looked up once per region; a second candidate decides (S5) A name that no enclosing type or namespace has is looked up in the using directives of each scope level, and a name some namespaces declare walked the min(postings, directives) of the level for every reference. A file with r references to such a name under k directives cost r x k. The using step of a lookup -- what one region's directives, and at the global namespace the unit's, give one query -- is now remembered for the pass: the references of one file (a per-file working state the doc-link layer hands to the resolver: file_begin / file_end), and the build's passes over using directives and base lists. The key holds everything the step reads: the file, the region and unit asked, the context's visibility (product code, unit, assembly) and the query. A memo that runs out of memory stops remembering; the lookups stay exact. Inside a lookup, the alias and using walks stop at a level's second candidate: the result is ambiguous whatever the rest would bring. Which candidate came first, and whether there was one, never depends on the directives skipped; only the ambiguity tally of the log line (why) may count such a reference under another kind. There is no cap: every outcome is what it was. Tests: cs_lookup_repeated_name_work (k 60 -> 240 and r 40 -> 160: 236 -> 900 steps, 3.8 x; without the memo 3004 -> 41444, 13.8 x), cs_lookup_ambiguous_stops (one ambiguous reference under 50 -> 200 imports that declare it: 17 -> 21 steps; without the stop 65 -> 219). Signed-off-by: Martin Vogel --- src/pipeline/doc_links.c | 6 +- src/pipeline/doc_links.h | 11 ++- src/pipeline/doc_links_cs.c | 149 +++++++++++++++++++++++++++++++++--- tests/test_doc_mentions.c | 134 ++++++++++++++++++++++++++++++++ 4 files changed, 288 insertions(+), 12 deletions(-) diff --git a/src/pipeline/doc_links.c b/src/pipeline/doc_links.c index 7c1af2bb67..fd2b32a326 100644 --- a/src/pipeline/doc_links.c +++ b/src/pipeline/doc_links.c @@ -429,6 +429,7 @@ void cbm_doclinks_resolve_file(cbm_doclinks_t *dl, int file_idx, const CBMFileRe * file-level doc resolves */ const cbm_gbuf_node_t *file_node = NULL; bool file_node_looked_up = false; + void *state = (R && index && R->file_begin) ? R->file_begin(index, file_idx) : NULL; for (int i = 0; i < n; i++) { const CBMDocLink *link = &result->doc_links.items[i]; cbm_doclink_outcome_t out = {.kind = CBM_DOCLINK_UNRESOLVED, @@ -438,7 +439,7 @@ void cbm_doclinks_resolve_file(cbm_doclinks_t *dl, int file_idx, const CBMFileRe } else if (!R || !index) { continue; /* no resolver for this language: nothing to say */ } else { - R->resolve(index, file_idx, link, graph, &out); + R->resolve(index, state, file_idx, link, graph, &out); } if (out.kind == CBM_DOCLINK_LOCAL) { atomic_fetch_add_explicit(&dl->local_refs, 1, memory_order_relaxed); @@ -495,6 +496,9 @@ void cbm_doclinks_resolve_file(cbm_doclinks_t *dl, int file_idx, const CBMFileRe } atomic_fetch_add_explicit(&dl->reasons[reason], 1, memory_order_relaxed); } + if (state && R->file_end) { + R->file_end(state); + } if (nm > 0) { emit_mentions(dl, fi->rel_path, mentions, nm, edge_out); } diff --git a/src/pipeline/doc_links.h b/src/pipeline/doc_links.h index 9a6452a77a..2f08f5c228 100644 --- a/src/pipeline/doc_links.h +++ b/src/pipeline/doc_links.h @@ -189,9 +189,16 @@ typedef struct { /* The language's project-wide index; NULL on allocation failure. */ void *(*build)(const cbm_doclink_build_in_t *in); void (*destroy)(void *index); + /* Optional working state for the references of one file, which one thread + * resolves: made before its first reference, handed to every resolve call + * of the file, released after its last. A NULL state (no hook, or memory + * ran out) changes the cost of resolving, never an outcome. */ + void *(*file_begin)(const void *index, int run_file); + void (*file_end)(void *state); /* Resolve one reference of run file `run_file` (its own scope is in the - * index). Thread-safe: the index is read-only after build. */ - void (*resolve)(const void *index, int run_file, const CBMDocLink *link, + * index). Thread-safe: the index is read-only after build; `state` is the + * file's own (file_begin) or NULL. */ + void (*resolve)(const void *index, void *state, int run_file, const CBMDocLink *link, const cbm_gbuf_t *graph, cbm_doclink_outcome_t *out); /* Incremental runs; both optional. * scope_input true for a file that is no source of the language and has diff --git a/src/pipeline/doc_links_cs.c b/src/pipeline/doc_links_cs.c index 92ab111e6b..a32b13451f 100644 --- a/src/pipeline/doc_links_cs.c +++ b/src/pipeline/doc_links_cs.c @@ -2867,6 +2867,8 @@ typedef struct { cs_why_t why; /* of an ambiguous one */ } cs_res_t; +typedef struct cs_memo cs_memo_t; + typedef struct { const cs_index_t *ix; int file; @@ -2885,6 +2887,9 @@ typedef struct { /* Set when a namespace this context does not see was passed over (NULL: * nobody asks). */ bool *passed_over; + /* What the using directives of a scope level gave a query, remembered + * for the pass the caller runs (lookup); NULL: nothing is remembered. */ + cs_memo_t *memo; } cs_ctx_t; static cs_res_t res_edge(const cbm_gbuf_node_t *n, bool exact) { @@ -4437,6 +4442,13 @@ static void add_alias_target(const cs_ctx_t *c, const cs_using_t *u, cs_found_t } } +/* True once a level has two candidates: the lookup is ambiguous whatever the + * directives after them bring, so they are not asked (S5). Which candidate + * came first, and whether there was one, never depends on the ones skipped. */ +static bool found_decided(const cs_found_t *fd) { + return fd->n > SKIP_ONE; +} + static void level_aliases(const cs_ctx_t *c, const cs_using_t *us, int n, const cs_using_index_t *index, const cs_query_t *q, cs_found_t *fd) { /* an alias names a type or a namespace: no candidate for `Name{T}` */ @@ -4444,7 +4456,7 @@ static void level_aliases(const cs_ctx_t *c, const cs_using_t *us, int n, return; } if (!index || !index->ready) { - for (int i = 0; i < n; i++) { + for (int i = 0; i < n && !found_decided(fd); i++) { cs_work(SKIP_ONE); if (us[i].kind == 'a' && strcmp(us[i].alias, q->name) == 0) { add_alias_target(c, &us[i], fd); @@ -4463,7 +4475,7 @@ static void level_aliases(const cs_ctx_t *c, const cs_using_t *us, int n, hi = mid; } } - for (size_t i = lo; i < index->naliases; i++) { + for (size_t i = lo; i < index->naliases && !found_decided(fd); i++) { cs_work(SKIP_ONE); if (strcmp(index->aliases[i].name, q->name)) { break; @@ -4599,7 +4611,7 @@ static void level_usings(const cs_ctx_t *c, const cs_using_t *us, int n, /* A common name in a large repository must not make a small scope walk * more postings than it has directives. The baseline scan stays cheap. */ if (!indexed || hi - lo >= (size_t)n) { - for (int i = 0; i < n; i++) { + for (int i = 0; i < n && !found_decided(fd); i++) { level_using(c, &us[i], q, a_type_name, fd); } return; @@ -4618,7 +4630,7 @@ static void level_usings(const cs_ctx_t *c, const cs_using_t *us, int n, if (ids.n > 1) { qsort(ids.ids, ids.n, sizeof(int), candidate_id_cmp); } - for (size_t i = 0; i < ids.n; i++) { + for (size_t i = 0; i < ids.n && !found_decided(fd); i++) { cs_work(SKIP_ONE); /* Own and twin postings may select the SAME directive twice. * Distinct directive IDs must still be resolved separately. */ @@ -4631,12 +4643,91 @@ static void level_usings(const cs_ctx_t *c, const cs_using_t *us, int n, cbm_free(CBM_MEM_CLASS_OTHER, ids.ids); } if (!complete) { - for (int i = 0; i < n; i++) { + for (int i = 0; i < n && !found_decided(fd); i++) { level_using(c, &us[i], q, a_type_name, fd); } } } +/* The using step of a lookup -- what the directives of one region, and at + * the global namespace the unit's, give one query -- remembered for one pass + * over a file's references (or one pass of the build): a name looked up + * again in the same region costs one probe, not the directives (S5). The key + * holds everything the step reads: the file, the region and unit asked, the + * context's visibility (product code, unit, assembly) and the query. A memo + * that ran out of memory stops remembering; the lookups stay exact. */ +struct cs_memo { + CBMHashTable *steps; /* key -> cs_found_t in `arena` */ + CBMArena arena; + bool off; +}; + +enum { CS_MEMO_KEY = CS_NAME_BUF + CBM_SZ_128 }; + +static void memo_init(cs_memo_t *m) { + memset(m, 0, sizeof(*m)); + cbm_arena_init(&m->arena); + m->steps = cbm_ht_create(CBM_SZ_256); + m->off = m->steps == NULL; +} + +static void memo_destroy(cs_memo_t *m) { + cbm_ht_free(m->steps); + cbm_arena_destroy(&m->arena); + memset(m, 0, sizeof(*m)); +} + +/* The key of one using step; false when it does not fit (not remembered). */ +static bool memo_key(char *buf, size_t cap, const cs_ctx_t *c, int region, int unit, + const cs_query_t *q) { + int n = snprintf(buf, cap, "%d|%d|%d|%d|%d|%d|%d|%d%d%d%c|%s", c->file, region, unit, c->unit, + c->group, (int)c->prod, q->arity, (int)q->types_only, (int)q->statics, + (int)q->ctors, q->kind ? q->kind : '-', q->name); + return n > 0 && (size_t)n < cap; +} + +static bool memo_get(const cs_memo_t *m, const char *key, cs_found_t *out) { + cs_work(SKIP_ONE); + const cs_found_t *hit = (const cs_found_t *)cbm_ht_get(m->steps, key); + if (hit) { + *out = *hit; + } + return hit != NULL; +} + +static void memo_put(cs_memo_t *m, const char *key, const cs_found_t *step) { + char *k = cbm_arena_strdup(&m->arena, key); + cs_found_t *v = (cs_found_t *)cbm_arena_alloc(&m->arena, sizeof(*v)); + if (!k || !v) { + m->off = true; + return; + } + *v = *step; + cbm_ht_set(m->steps, k, v); + m->off = cbm_ht_get(m->steps, k) != v; /* an insert that did not take */ +} + +/* What the directives of this scope level give the query: the region's own + * (`us`, when asked) and, at the global namespace, the unit's. Asked once per + * key when the context has a memo. */ +static void using_step(const cs_ctx_t *c, int region, const cs_using_t *us, int nus, + const cs_using_index_t *usi, const cs_unit_t *unit, const cs_query_t *q, + cs_found_t *step) { + char key[CS_MEMO_KEY]; + bool keyed = c->memo && !c->memo->off && + memo_key(key, sizeof(key), c, region, unit ? c->unit : CS_NONE, q); + if (keyed && memo_get(c->memo, key, step)) { + return; + } + level_usings(c, us, nus, usi, q, step); + if (unit && !found_decided(step)) { + level_usings(c, unit->usings, unit->nusings, &unit->using_index, q, step); + } + if (keyed) { + memo_put(c->memo, key, step); + } +} + /* True when the lookup is settled by what the last level added. A level that * had only invisible candidates does not bind: the lookup goes on, and * remembers. */ @@ -4714,9 +4805,16 @@ static void lookup(const cs_ctx_t *c, const cs_query_t *q, cs_found_t *fd, bool fd->exact = true; return; } - level_usings(c, us, nus, usi, q, fd); - if (unit) { - level_usings(c, unit->usings, unit->nusings, &unit->using_index, q, fd); + if (nus > 0 || unit) { + /* Nothing is found yet (settled), and a step only adds: the step + * is asked on its own and its candidates are the level's. */ + cs_found_t step = {0}; + using_step(c, asked ? reg : CS_NONE, us, nus, usi, unit, q, &step); + fd->first = step.first; + fd->n = step.n; + fd->invisible = step.invisible; + fd->why = step.why; + fd->joined = fd->joined || step.joined; } if (settled(fd, &invisible)) { *statics = fd->first.kind == 'M'; @@ -5515,6 +5613,10 @@ static bool resolve_usings(cs_index_t *ix) { return false; } } + /* A lookup asks only regions above the directive's own, whose directives + * are resolved already: what it remembers of them stays true. */ + cs_memo_t memo; + memo_init(&memo); for (int fi = 0; fi < ix->nfiles; fi++) { cs_file_t *f = &ix->files[fi]; /* Regions are in ancestor order. Publish an outer region's index @@ -5528,10 +5630,12 @@ static bool resolve_usings(cs_index_t *ix) { c.prod = false; /* a directive names whatever the compiler bound */ c.glob = r == 0; c.skip_region = r; + c.memo = &memo; resolve_using(&f->usings[i], &c); } const cs_using_t *us = reg->u_hi > reg->u_lo ? f->usings + reg->u_lo : NULL; if (!build_using_index(ix, us, reg->u_hi - reg->u_lo, ®->using_index)) { + memo_destroy(&memo); return false; } } @@ -5543,6 +5647,7 @@ static bool resolve_usings(cs_index_t *ix) { } } } + memo_destroy(&memo); return true; } @@ -5570,6 +5675,8 @@ static bool resolve_bases(cs_index_t *ix) { int *listed = (int *)cbm_calloc(CBM_MEM_CLASS_OTHER, ((size_t)ix->nents + SKIP_ONE) * sizeof(int)); bool ok = listed != NULL; + cs_memo_t memo; /* every directive is resolved: what a lookup remembers stays true */ + memo_init(&memo); for (int ei = 0; ok && ei < ix->nents; ei++) { cs_entity_t *e = &ix->ents[ei]; int written = 0; @@ -5587,6 +5694,7 @@ static bool resolve_bases(cs_index_t *ix) { /* a base list is written outside the type it belongs to */ ctx_at(&c, ix, e->decls[d].file, t->region, t->outer); c.prod = false; /* a declared base is whatever the compiler bound */ + c.memo = &memo; for (const char *p = t->bases; p && *p;) { const char *bar = strchr(p, '|'); size_t n = bar ? (size_t)(bar - p) : strlen(p); @@ -5601,6 +5709,7 @@ static bool resolve_bases(cs_index_t *ix) { } } } + memo_destroy(&memo); cbm_free(CBM_MEM_CLASS_OTHER, listed); ix->oom = ix->oom || !ok; return ok; @@ -6054,7 +6163,26 @@ static bool names_quarantined(const cs_index_t *ix, const cs_file_t *f, const cs return false; } -static void cs_resolve(const void *index, int run_file, const CBMDocLink *link, +/* The working state of one file's references: the memo of their lookups. */ +static void *cs_file_begin(const void *index, int run_file) { + (void)index; + (void)run_file; + cs_memo_t *m = (cs_memo_t *)cbm_alloc(CBM_MEM_CLASS_OTHER, sizeof(*m)); + if (m) { + memo_init(m); + } + return m; +} + +static void cs_file_end(void *state) { + cs_memo_t *m = (cs_memo_t *)state; + if (m) { + memo_destroy(m); + cbm_free(CBM_MEM_CLASS_OTHER, m); + } +} + +static void cs_resolve(const void *index, void *state, int run_file, const CBMDocLink *link, const cbm_gbuf_t *graph, cbm_doclink_outcome_t *out) { const cs_index_t *ix = (const cs_index_t *)index; (void)graph; /* every node was looked up when the index was built */ @@ -6091,6 +6219,7 @@ static void cs_resolve(const void *index, int run_file, const CBMDocLink *link, cs_ctx_t c; ctx_init(&c, ix, file, link->def_line, file_doc); c.glob = r.glob; + c.memo = (cs_memo_t *)state; cs_res_t res = resolve_ref(&c, &r); if (res.st == CS_LOCAL) { out->kind = CBM_DOCLINK_LOCAL; @@ -6260,6 +6389,8 @@ const cbm_doclink_resolver_t cbm_doclink_cs_resolver = { .scope_tag = CBM_DOCLINK_CS_SCOPE_TAG, .build = cs_build, .destroy = cs_destroy, + .file_begin = cs_file_begin, + .file_end = cs_file_end, .resolve = cs_resolve, .scope_delta = cs_scope_delta, }; diff --git a/tests/test_doc_mentions.c b/tests/test_doc_mentions.c index aec90228f1..99c4090eb9 100644 --- a/tests/test_doc_mentions.c +++ b/tests/test_doc_mentions.c @@ -8729,6 +8729,138 @@ TEST(doc_mentions_cs_import_lookup_cost) { PASS(); } +/* S5: `Shared` is declared in k namespaces nobody imports; one file has k + * usings of other namespaces and r documented classes that each name it, so + * every one of its r lookups would walk the k directives. The resolver's + * lookup work, and the number of `missing` rows for `Shared` (r when all are + * resolved; -1 when the repository cannot be indexed). */ +static uint64_t dm_repeated_name_work(int k, int r, int *rows) { + enum { CAP = 256 * 1024 }; + char *src = malloc(CAP); + char tmp[256] = "/tmp/cbm_dm_repeated_XXXXXX"; + *rows = -1; + if (!src || !cbm_mkdtemp(tmp)) { + free(src); + return 0; + } + size_t w = 0; + for (int i = 0; i < k; i++) { + w += (size_t)snprintf(src + w, CAP - w, + "namespace P%d { public class Shared { } }\n" + "namespace In.U%d { public class Other%d { } }\n", + i, i, i); + } + th_write_file(TH_PATH(tmp, "src/Far.cs"), src); + w = 0; + for (int i = 0; i < k; i++) { + w += (size_t)snprintf(src + w, CAP - w, "using In.U%d;\n", i); + } + w += (size_t)snprintf(src + w, CAP - w, "namespace N0\n{\n"); + for (int j = 0; j < r; j++) { + w += (size_t)snprintf(src + w, CAP - w, + "/// \n" + "public class L%d { }\n", + j); + } + snprintf(src + w, CAP - w, "}\n"); + th_write_file(TH_PATH(tmp, "src/Uses.cs"), src); + free(src); + char db[512]; + snprintf(db, sizeof(db), "%s/repeated.db", tmp); + cbm_doclink_cs_test_work_reset(); + uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_work() : 0; + *rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved " + "WHERE raw = 'Shared' AND reason = 'missing'"); + dm_unlink_db(db); + th_rmtree(tmp); + return work; +} + +/* A name looked up again in the same region costs once: with k and r both + * four times as large the work grows with the input (about 4 x), not with + * r x k (16 x). */ +TEST(doc_mentions_cs_lookup_repeated_name_work) { + int rows[2] = {0}; + uint64_t small = dm_repeated_name_work(60, 40, &rows[0]); + uint64_t large = dm_repeated_name_work(240, 160, &rows[1]); + double ratio = small ? (double)large / (double)small : 0.0; + printf(" repeated name, k 60 -> 240 and r 40 -> 160: %llu -> %llu steps, %.2f times the " + "work, want at most 6\n", + (unsigned long long)small, (unsigned long long)large, ratio); + ASSERT_EQ(rows[0], 40); + ASSERT_EQ(rows[1], 160); + ASSERT_GT(small, 0); + ASSERT_LTE(large, 6 * small); + PASS(); +} + +/* S5: k namespaces all declare `Shared` and one file imports all of them; + * `refs` documented classes name it (0 or 1). The resolver's lookup work and + * the reason of the row (empty when there is none). */ +static uint64_t dm_ambiguous_name_work(int k, int refs, char *reason, size_t cap) { + enum { SRC_CAP = 64 * 1024 }; + char *src = malloc(SRC_CAP); + char tmp[256] = "/tmp/cbm_dm_ambiguous_XXXXXX"; + reason[0] = '\0'; + if (!src || !cbm_mkdtemp(tmp)) { + free(src); + return 0; + } + size_t w = 0; + for (int i = 0; i < k; i++) { + w += (size_t)snprintf(src + w, SRC_CAP - w, "namespace P%d { public class Shared { } }\n", + i); + } + th_write_file(TH_PATH(tmp, "src/Far.cs"), src); + w = 0; + for (int i = 0; i < k; i++) { + w += (size_t)snprintf(src + w, SRC_CAP - w, "using P%d;\n", i); + } + w += (size_t)snprintf(src + w, SRC_CAP - w, "namespace N0\n{\n"); + for (int j = 0; j < refs; j++) { + w += (size_t)snprintf(src + w, SRC_CAP - w, + "/// \n" + "public class L%d { }\n", + j); + } + snprintf(src + w, SRC_CAP - w, "public class Last { }\n}\n"); + th_write_file(TH_PATH(tmp, "src/Uses.cs"), src); + free(src); + char db[512]; + snprintf(db, sizeof(db), "%s/ambiguous.db", tmp); + cbm_doclink_cs_test_work_reset(); + uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_work() : 0; + dm_row(db, "src/Uses.cs", "Shared", reason, cap, NULL, 0); + dm_unlink_db(db); + th_rmtree(tmp); + return work; +} + +/* One lookup stops at its second candidate: the name is ambiguous whatever + * the directives after it bring. The work of one ambiguous reference (the + * run with it less the run without it) does not grow with the k imports that + * all declare the name. */ +TEST(doc_mentions_cs_lookup_ambiguous_stops) { + static const int ks[2] = {50, 200}; + uint64_t one[2] = {0}; + bool ambiguous = true; + for (int i = 0; i < 2; i++) { + char reason[64]; + char none[64]; + uint64_t with = dm_ambiguous_name_work(ks[i], 1, reason, sizeof(reason)); + uint64_t without = dm_ambiguous_name_work(ks[i], 0, none, sizeof(none)); + one[i] = with > without ? with - without : 0; + ambiguous = ambiguous && strcmp(reason, "ambiguous") == 0 && !none[0]; + } + printf(" one ambiguous reference, k 50 -> 200 imports that declare it: %llu -> %llu steps, " + "want at most 1.5 times\n", + (unsigned long long)one[0], (unsigned long long)one[1]); + ASSERT_TRUE(ambiguous); + ASSERT_GT(one[0], 0); + ASSERT_LTE(2 * one[1], 3 * one[0]); + PASS(); +} + /* Parent/global imports are ready for a child's directives. The child's own * directives stay excluded while those targets are resolved. */ TEST(doc_mentions_cs_import_stage_aliases) { @@ -10009,6 +10141,8 @@ SUITE(doc_mentions) { RUN_TEST(doc_mentions_cs_text_forms); RUN_TEST(doc_mentions_cs_lookup_cost); RUN_TEST(doc_mentions_cs_import_lookup_cost); + RUN_TEST(doc_mentions_cs_lookup_repeated_name_work); + RUN_TEST(doc_mentions_cs_lookup_ambiguous_stops); RUN_TEST(doc_mentions_cs_import_stage_aliases); RUN_TEST(doc_mentions_cs_import_static_parts); RUN_TEST(doc_mentions_cs_import_candidate_allocation); From e99755c7200d8ce28ba0c7b5eb8d1a1c1085b014 Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Sun, 4 Oct 2026 18:54:09 +0200 Subject: [PATCH 09/19] doc-links(cs): the lookup memo holds no memory until a step is remembered A memo is made for every file that has references, and most files ask few using steps: its table and arena are now opened by the first step it keeps (8 KB arena block, 16-entry table), not when the file starts. Outcomes and work counts are unchanged. Signed-off-by: Martin Vogel --- src/pipeline/doc_links_cs.c | 15 +++++++++++---- 1 file changed, 11 insertions(+), 4 deletions(-) diff --git a/src/pipeline/doc_links_cs.c b/src/pipeline/doc_links_cs.c index a32b13451f..70057b3760 100644 --- a/src/pipeline/doc_links_cs.c +++ b/src/pipeline/doc_links_cs.c @@ -4664,11 +4664,11 @@ struct cs_memo { enum { CS_MEMO_KEY = CS_NAME_BUF + CBM_SZ_128 }; +/* Holds no memory until the first step is remembered: most files ask few + * using steps, and a memo is made for every file that has references. */ static void memo_init(cs_memo_t *m) { memset(m, 0, sizeof(*m)); - cbm_arena_init(&m->arena); - m->steps = cbm_ht_create(CBM_SZ_256); - m->off = m->steps == NULL; + cbm_arena_init_lazy(&m->arena, CBM_ARENA_APPEND_BLOCK); } static void memo_destroy(cs_memo_t *m) { @@ -4688,7 +4688,7 @@ static bool memo_key(char *buf, size_t cap, const cs_ctx_t *c, int region, int u static bool memo_get(const cs_memo_t *m, const char *key, cs_found_t *out) { cs_work(SKIP_ONE); - const cs_found_t *hit = (const cs_found_t *)cbm_ht_get(m->steps, key); + const cs_found_t *hit = m->steps ? (const cs_found_t *)cbm_ht_get(m->steps, key) : NULL; if (hit) { *out = *hit; } @@ -4696,6 +4696,13 @@ static bool memo_get(const cs_memo_t *m, const char *key, cs_found_t *out) { } static void memo_put(cs_memo_t *m, const char *key, const cs_found_t *step) { + if (!m->steps) { + m->steps = cbm_ht_create(CBM_SZ_16); + if (!m->steps) { + m->off = true; + return; + } + } char *k = cbm_arena_strdup(&m->arena, key); cs_found_t *v = (cs_found_t *)cbm_arena_alloc(&m->arena, sizeof(*v)); if (!k || !v) { From c10146e43f312618bbe6af79a374c5c8621355c7 Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Sun, 4 Oct 2026 19:00:41 +0200 Subject: [PATCH 10/19] doc-links(cs): a scope its run refused is stored as a rejected marker (S8 across runs) A C# scope blob written in a run that the reader refuses costs only its own file in that run (a8c5708d). The blob itself was stored, though, and the next incremental run read it back as a stored scope the reader refuses, which fails the layer for the whole run. The surface row now stores, in place of such a blob, the C# rejected marker (the tag and one `!` record). Whether the reader takes a blob is decided where the row is made, by the reader's own record checks on an index of the blob's own (what makes a blob refused depends on the blob alone); the doc-link layer asks the language through two new resolver fields, scope_accepted and rejected_scope (cbm_doclinks_storable_scope). A later run that reads the marker back treats the file as the run that wrote it: it declares nothing, its references are graph gaps, it is counted with the rejected files. A stored scope that does not parse for any other reason still fails the run. The scope delta of a file whose blob is the marker on either side is LOCAL only when both are. The test seam that spoils a file's scope now does it where the scope is written (internal/cbm/doclink_cs.c), so what is stored carries it too. Test: cs_rejected_scope_across_runs (a full run with the spoiled file, then a body-only edit elsewhere and one in the spoiled file: each incremental step equals a full run, and the stored scope is the marker). Signed-off-by: Martin Vogel --- internal/cbm/doclink.h | 6 ++ internal/cbm/doclink_cs.c | 12 +++ src/pipeline/doc_links.c | 12 +++ src/pipeline/doc_links.h | 21 ++++- src/pipeline/doc_links_cs.c | 87 +++++++++++---------- src/pipeline/lsp_surface.c | 7 +- tests/test_doc_mentions.c | 149 +++++++++++++++++++++++++++--------- 7 files changed, 214 insertions(+), 80 deletions(-) diff --git a/internal/cbm/doclink.h b/internal/cbm/doclink.h index bb3ffa46f4..3588db1b09 100644 --- a/internal/cbm/doclink.h +++ b/internal/cbm/doclink.h @@ -157,6 +157,12 @@ void cbm_doclink_test_fail_alloc_after(int kind, int nth); bool cbm_doclink_test_fail_alloc(int kind); void cbm_doclink_test_reset_alloc(void); +/* The C# scope scan of the file at `rel_path` writes, after its records, one + * the resolver refuses -- as if writer and reader disagreed (NULL or "": + * none). What is stored for the file and what this run resolves both carry + * it. */ +void cbm_doclink_cs_test_spoil_scope(const char *rel_path); + /* Actual comment memcpy bytes, input bytes submitted to the C# lexical parser, * and bytes allocated for cleaned reference values; separate from token output. */ void cbm_doclink_test_doc_work_reset(void); diff --git a/internal/cbm/doclink_cs.c b/internal/cbm/doclink_cs.c index 23db08f5a2..db09f61824 100644 --- a/internal/cbm/doclink_cs.c +++ b/internal/cbm/doclink_cs.c @@ -3027,6 +3027,14 @@ char *cbm_doclink_cs_portable_scope(const char *scope) { return out; } +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +static char cs_test_spoiled_path[CBM_SZ_512]; + +void cbm_doclink_cs_test_spoil_scope(const char *rel_path) { + snprintf(cs_test_spoiled_path, sizeof(cs_test_spoiled_path), "%s", rel_path ? rel_path : ""); +} +#endif + const char *cbm_doclink_cs_scan_scope(CBMExtractCtx *ctx) { if (ts_node_is_null(ctx->root)) { return NULL; @@ -3056,6 +3064,10 @@ const char *cbm_doclink_cs_scan_scope(CBMExtractCtx *ctx) { if (cbm_doclink_test_fail_alloc(CBM_DOCLINK_ALLOC_SCOPE)) { s.failed = true; } + if (cs_test_spoiled_path[0] && ctx->rel_path && + strcmp(ctx->rel_path, cs_test_spoiled_path) == 0) { + sb_puts(&s.sb, "Z\tspoiled\n"); + } #endif if (s.failed || s.sb.failed || !s.sb.buf) { /* memory ran out: no scope would read as a file that declares diff --git a/src/pipeline/doc_links.c b/src/pipeline/doc_links.c index fd2b32a326..fa4b0096ab 100644 --- a/src/pipeline/doc_links.c +++ b/src/pipeline/doc_links.c @@ -113,6 +113,18 @@ int cbm_doclinks_scope_delta(const char *stored, const char *fresh, cbm_doclink_ return R->scope_delta(stored, fresh, removed, ud); } +const char *cbm_doclinks_storable_scope(const char *scope) { + const cbm_doclink_resolver_t *R = resolver_of_scope(scope); + if (!R || !R->scope_accepted || !R->rejected_scope) { + return scope; + } + int accepted = R->scope_accepted(scope); + if (accepted < 0) { + return NULL; + } + return accepted ? scope : R->rejected_scope; +} + /* ── Run state ───────────────────────────────────────────────────── */ typedef struct { diff --git a/src/pipeline/doc_links.h b/src/pipeline/doc_links.h index 2f08f5c228..b175ebd76a 100644 --- a/src/pipeline/doc_links.h +++ b/src/pipeline/doc_links.h @@ -212,6 +212,15 @@ typedef struct { bool (*scope_input)(const char *rel_path); int (*scope_delta)(const char *stored, const char *fresh, cbm_doclink_name_fn removed, void *ud); + /* Optional, together: whether the reader takes a scope blob written in + * this run (1), refuses it (0), or could not tell for memory (-1); and + * what is stored instead of a refused blob. A run refuses the blob of its + * own file and goes on without the file's declarations; the marker makes + * a later run that reads the file's scope back do the same, where the + * blob itself would be a stored scope the reader refuses, which fails + * the run. */ + int (*scope_accepted)(const char *scope); + const char *rejected_scope; } cbm_doclink_resolver_t; extern const cbm_doclink_resolver_t cbm_doclink_cs_resolver; @@ -229,10 +238,8 @@ uint64_t cbm_doclink_cs_test_scratch_work(void); /* Test seam (doc_links.c): the nth MENTIONS edge insert from now on fails as * if memory ran out (0: none). */ void cbm_doclinks_test_fail_edge_insert_after(int nth); -/* Test seams (doc_links_cs.c): spoil the scope of the file at `rel_path` - * (every record, then one the reader refuses; NULL or "": none), and ask the - * reader whether it takes a scope blob. */ -void cbm_doclink_cs_test_spoil_scope(const char *rel_path); +/* Test seam (doc_links_cs.c): ask the reader whether it takes a scope blob. + * (The scope of one file is spoiled where it is written: doclink.h.) */ bool cbm_doclink_cs_test_scope_parses(const char *scope); #endif @@ -246,4 +253,10 @@ bool cbm_doclinks_is_scope_input(const char *rel_path); int cbm_doclinks_scope_delta(const char *stored, const char *fresh, cbm_doclink_name_fn removed, void *ud); +/* What is stored for a scope blob written in this run (the surface row's + * `dl`, lsp_surface.c): the blob itself, or its language's rejected-scope + * marker when that language's reader refuses it (scope_accepted). NULL when + * memory ran out deciding. */ +const char *cbm_doclinks_storable_scope(const char *scope); + #endif /* CBM_PIPELINE_DOC_LINKS_H */ diff --git a/src/pipeline/doc_links_cs.c b/src/pipeline/doc_links_cs.c index 70057b3760..0995c134c1 100644 --- a/src/pipeline/doc_links_cs.c +++ b/src/pipeline/doc_links_cs.c @@ -5925,45 +5925,44 @@ static void undo_scope(cs_index_t *ix) { u->nmarks = 0; } -#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS -/* Test seam: the scope of the file at this path is spoiled -- every record - * it has, then one this reader refuses -- as if its writer and this reader - * disagreed after the file's declarations were read. "" or NULL: none. */ -static char cs_test_spoiled_path[CBM_SZ_512]; - -void cbm_doclink_cs_test_spoil_scope(const char *rel_path) { - snprintf(cs_test_spoiled_path, sizeof(cs_test_spoiled_path), "%s", rel_path ? rel_path : ""); -} - -static const char *cs_test_spoiled(cs_index_t *ix, const char *rel_path, const char *scope) { - if (!scope || !cs_test_spoiled_path[0] || strcmp(rel_path, cs_test_spoiled_path) != 0) { - return scope; +/* What is stored, in place of its scope blob, for a file whose scope the + * reader refused in the run that wrote it (cs_scope_accepted, lsp_surface.c): + * the tag and one `!` record. A later run that reads it back treats the file + * as that run did -- it declares nothing, its references are graph gaps -- + * where the refused blob itself would fail that run as a stored scope this + * reader does not take. */ +static const char CS_REJECTED_SCOPE[] = CBM_DOCLINK_CS_SCOPE_TAG "\n!\trejected\n"; + +static bool cs_scope_marked_rejected(const char *scope) { + return scope && strcmp(scope, CS_REJECTED_SCOPE) == 0; +} + +/* Whether this reader takes a scope blob written in this run: its record + * checks, on an index of the blob's own (what makes a blob refused depends on + * the blob alone). 1 taken, 0 refused, -1 memory ran out. A project file's + * blob goes to the MSBuild evaluator, which takes every blob. */ +static int cs_scope_accepted(const char *scope) { + if (!scope || cbm_msb_is_project_scope(scope) || cs_scope_marked_rejected(scope)) { + return SKIP_ONE; } - const char *spoiled = cbm_arena_sprintf(&ix->arena, "%sZ\tspoiled\n", scope); - return spoiled ? spoiled : scope; -} - -/* Test seam: true when this reader takes the scope blob `scope` (its record - * checks, on an index of its own). A test holds every blob the scanner - * writes against it. */ -bool cbm_doclink_cs_test_scope_parses(const char *scope) { cs_index_t *ix = (cs_index_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*ix)); if (!ix) { - return false; - } - cbm_arena_init(&ix->arena); - ix->ns_by_key = cbm_ht_create(CBM_SZ_64); - ix->quarantine = cbm_ht_create(CBM_SZ_64); - ix->quarantine_test = cbm_ht_create(CBM_SZ_64); - ix->nss = (cs_ns_t *)ix_zalloc(ix, CBM_SZ_256 * sizeof(cs_ns_t)); - bool ok = ix->ns_by_key && ix->quarantine && ix->quarantine_test && ix->nss; - if (ok) { - ix->nscap = CBM_SZ_256; + return CBM_NOT_FOUND; + } + cbm_arena_init_lazy(&ix->arena, CBM_ARENA_APPEND_BLOCK); + ix->ns_by_key = cbm_ht_create(CBM_SZ_16); + ix->quarantine = cbm_ht_create(CBM_SZ_16); + ix->quarantine_test = cbm_ht_create(CBM_SZ_16); + ix->nss = (cs_ns_t *)ix_zalloc(ix, CBM_SZ_16 * sizeof(cs_ns_t)); + int accepted = CBM_NOT_FOUND; + if (ix->ns_by_key && ix->quarantine && ix->quarantine_test && ix->nss) { + ix->nscap = CBM_SZ_16; ix->nnss = SKIP_ONE; ix->nss[0] = (cs_ns_t){.parent = CS_NONE, .name = ""}; cs_file_t f = {.rel_path = "Scope.cs"}; undo_begin(ix); - ok = parse_scope(ix, &f, scope) && !ix->oom; + bool parsed = parse_scope(ix, &f, scope); + accepted = ix->oom ? CBM_NOT_FOUND : (parsed ? SKIP_ONE : 0); } cbm_ht_free(ix->ns_by_key); cbm_ht_free(ix->quarantine); @@ -5972,7 +5971,14 @@ bool cbm_doclink_cs_test_scope_parses(const char *scope) { cbm_free(CBM_MEM_CLASS_OTHER, ix->undo.marks); cbm_arena_destroy(&ix->arena); cbm_free(CBM_MEM_CLASS_OTHER, ix); - return ok; + return accepted; +} + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +/* Test seam: true when this reader takes the scope blob `scope`. A test holds + * every blob the scanner writes against it. */ +bool cbm_doclink_cs_test_scope_parses(const char *scope) { + return cs_scope_accepted(scope) == SKIP_ONE; } #endif @@ -5996,13 +6002,12 @@ static const char *build_files(cs_index_t *ix, const cbm_doclink_build_in_t *in) f->is_test = cs_is_test_path(src->rel_path); /* a project file declares nothing: its blob went to the evaluator */ const char *scope = cbm_msb_is_project_scope(src->scope) ? NULL : src->scope; -#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS - scope = cs_test_spoiled(ix, src->rel_path, scope); -#endif + /* a scope its own run refused, stored as the marker: as in that run */ + bool marked = cs_scope_marked_rejected(scope); undo_begin(ix); - if (scope && !parse_scope(ix, f, scope)) { + if (marked || (scope && !parse_scope(ix, f, scope))) { bool fresh = src->run_file >= 0 && src->run_file < in->run_file_count; - if (ix->oom || !fresh) { + if (!marked && (ix->oom || !fresh)) { bad = ix->oom ? "" : src->rel_path; break; } @@ -6354,7 +6359,9 @@ static bool delta_has_bases(const char *scope) { * nobody's scope. */ static int cs_scope_delta(const char *stored, const char *fresh, cbm_doclink_name_fn removed, void *ud) { - if (cbm_msb_is_project_scope(stored) || cbm_msb_is_project_scope(fresh)) { + /* a rejected file declares nothing: unchanged while it stays rejected */ + if (cbm_msb_is_project_scope(stored) || cbm_msb_is_project_scope(fresh) || + cs_scope_marked_rejected(stored) || cs_scope_marked_rejected(fresh)) { return strcmp(stored, fresh) == 0 ? CBM_DOCLINK_DELTA_LOCAL : CBM_DOCLINK_DELTA_GLOBAL; } if (!delta_same_usings(stored, fresh) && (delta_has_bases(stored) || delta_has_bases(fresh))) { @@ -6400,4 +6407,6 @@ const cbm_doclink_resolver_t cbm_doclink_cs_resolver = { .file_end = cs_file_end, .resolve = cs_resolve, .scope_delta = cs_scope_delta, + .scope_accepted = cs_scope_accepted, + .rejected_scope = CS_REJECTED_SCOPE, }; diff --git a/src/pipeline/lsp_surface.c b/src/pipeline/lsp_surface.c index 5acd61b8bd..0d281f05d2 100644 --- a/src/pipeline/lsp_surface.c +++ b/src/pipeline/lsp_surface.c @@ -28,6 +28,7 @@ #include "foundation/log.h" #include "foundation/mem_core.h" #include "foundation/sha256.h" +#include "pipeline/doc_links.h" /* cbm_doclinks_storable_scope */ #include "pipeline/worker_pool.h" #include "yyjson/yyjson.h" @@ -161,7 +162,11 @@ static char *surface_file_to_json(const CBMFileResult *result, const CBMLSPDef * * The defs decoder ignores this key; cbm_doclinks_scopes_from_surfaces * reads it back for the files an incremental run does not re-extract. */ if (result && result->doc_scope) { - char *portable = cbm_doclink_portable_scope(result->doc_scope); + /* a scope this run's reader refuses is stored as its language's + * rejected marker: a later run then treats the file as this one does + * (doc_links.h, scope_accepted) */ + const char *kept = cbm_doclinks_storable_scope(result->doc_scope); + char *portable = kept ? cbm_doclink_portable_scope(kept) : NULL; if (!portable) { yyjson_mut_doc_free(doc); return NULL; /* a missing scope would diverge silently: fail the row */ diff --git a/tests/test_doc_mentions.c b/tests/test_doc_mentions.c index 99c4090eb9..6e9fe9bf5e 100644 --- a/tests/test_doc_mentions.c +++ b/tests/test_doc_mentions.c @@ -9962,47 +9962,51 @@ TEST(doc_mentions_cs_scanner_output_reads_back) { * stays -- a namespace it declares does not stand beside a type of that name * (Shadow, Good.Lurker), and a name it quarantined does not block another * file's declaration (Hidden2). */ -TEST(doc_mentions_cs_rejected_scope_contained) { - static const dm_source_t files[] = { - {"Bad.cs", "namespace Shadow\n" - "{\n" - " public class Inner { }\n" - "}\n" +/* S8: Bad.cs declares what would shadow and hide Healthy.cs's names; its + * scope is spoiled where it is written (cbm_doclink_cs_test_spoil_scope). */ +static const dm_source_t dm_rejected_files[] = { + {"Bad.cs", "namespace Shadow\n" + "{\n" + " public class Inner { }\n" + "}\n" + "namespace Good\n" + "{\n" + " namespace Lurker { public class Deep { } }\n" + "\n" + " /// \n" + " public class FromBad { }\n" + "}\n" + "namespace .Broken\n" + "{\n" + " public class Hidden2 { }\n" + "}\n"}, + {"Healthy.cs", "public class Shadow { }\n" "namespace Good\n" "{\n" - " namespace Lurker { public class Deep { } }\n" - "\n" - " /// \n" - " public class FromBad { }\n" - "}\n" - "namespace .Broken\n" - "{\n" + " public class Target { }\n" + " public class Lurker { }\n" " public class Hidden2 { }\n" + "\n" + " /// \n" + " /// \n" + " public class Uses { }\n" "}\n"}, - {"Healthy.cs", "public class Shadow { }\n" - "namespace Good\n" - "{\n" - " public class Target { }\n" - " public class Lurker { }\n" - " public class Hidden2 { }\n" - "\n" - " /// \n" - " /// \n" - " public class Uses { }\n" - "}\n"}, - }; - static const dm_want_t wants[] = { - {"Healthy.cs", "Target", "Healthy.Uses", "Healthy.Target", NULL, NULL, NULL}, - {"Healthy.cs", "Shadow", "Healthy.Uses", "Healthy.Shadow", NULL, NULL, NULL}, - {"Healthy.cs", "Lurker", "Healthy.Uses", "Healthy.Lurker", NULL, NULL, NULL}, - {"Healthy.cs", "Hidden2", "Healthy.Uses", "Healthy.Hidden2", NULL, NULL, NULL}, - {"Bad.cs", "Target", "Bad.FromBad", NULL, "graph_gap", "Healthy.Target", NULL}, - }; +}; + +static const dm_want_t dm_rejected_wants[] = { + {"Healthy.cs", "Target", "Healthy.Uses", "Healthy.Target", NULL, NULL, NULL}, + {"Healthy.cs", "Shadow", "Healthy.Uses", "Healthy.Shadow", NULL, NULL, NULL}, + {"Healthy.cs", "Lurker", "Healthy.Uses", "Healthy.Lurker", NULL, NULL, NULL}, + {"Healthy.cs", "Hidden2", "Healthy.Uses", "Healthy.Hidden2", NULL, NULL, NULL}, + {"Bad.cs", "Target", "Bad.FromBad", NULL, "graph_gap", "Healthy.Target", NULL}, +}; + +TEST(doc_mentions_cs_rejected_scope_contained) { char tmp[256]; snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_rejected_XXXXXX"); ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); - for (int i = 0; i < DM_COUNT(files); i++) { - th_write_file(TH_PATH(tmp, files[i].path), files[i].text); + for (int i = 0; i < DM_COUNT(dm_rejected_files); i++) { + th_write_file(TH_PATH(tmp, dm_rejected_files[i].path), dm_rejected_files[i].text); } char db[512]; snprintf(db, sizeof(db), "%s/rejected.db", tmp); @@ -10012,11 +10016,83 @@ TEST(doc_mentions_cs_rejected_scope_contained) { int errors = bad == 0 ? dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved " "WHERE reason = 'error'") : -1; - for (int i = 0; bad >= 0 && i < DM_COUNT(wants); i++) { - bad += dm_want_failed(db, &wants[i]); + for (int i = 0; bad >= 0 && i < DM_COUNT(dm_rejected_wants); i++) { + bad += dm_want_failed(db, &dm_rejected_wants[i]); + } + dm_unlink_db(db); + th_rmtree(tmp); + ASSERT_EQ(errors, 0); + ASSERT_EQ(bad, 0); + PASS(); +} + +/* The scope blob stored for `rel_path` (the surface row's `dl`), into out; + * "" when there is none. */ +static void dm_stored_scope(const char *db, const char *rel_path, char *out, size_t cap) { + out[0] = '\0'; + sqlite3 *h = NULL; + if (sqlite3_open_v2(db, &h, SQLITE_OPEN_READONLY, NULL) == SQLITE_OK) { + sqlite3_stmt *st = NULL; + if (sqlite3_prepare_v2(h, + "SELECT json_extract(defs_json, '$.dl') FROM lsp_surface " + "WHERE rel_path = ?1", + -1, &st, NULL) == SQLITE_OK) { + sqlite3_bind_text(st, 1, rel_path, -1, SQLITE_TRANSIENT); + if (sqlite3_step(st) == SQLITE_ROW && sqlite3_column_text(st, 0)) { + snprintf(out, cap, "%s", (const char *)sqlite3_column_text(st, 0)); + } + } + sqlite3_finalize(st); + } + sqlite3_close(h); +} + +/* S8 across runs: the scope its own run refused is stored as the rejected + * marker, so a later incremental run that reads it back treats the file as + * the full run does -- it declares nothing, its references are graph gaps -- + * instead of failing the layer over a stored scope the reader refuses. A + * body-only edit elsewhere reads the marker back; one in the rejected file + * compares its marker with the fresh one. Each step equals a full run. */ +TEST(doc_mentions_cs_rejected_scope_across_runs) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_rejected_runs_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400]; + char db[512]; + char full_db[512]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + snprintf(db, sizeof(db), "%s/inc.db", tmp); + snprintf(full_db, sizeof(full_db), "%s/full.db", tmp); + for (int i = 0; i < DM_COUNT(dm_rejected_files); i++) { + th_write_file(TH_PATH(repo, dm_rejected_files[i].path), dm_rejected_files[i].text); + } + cbm_doclink_cs_test_spoil_scope("Bad.cs"); + int indexed = dm_index(repo, db, NULL); + char stored[256]; + dm_stored_scope(db, "Bad.cs", stored, sizeof(stored)); + bool marked = strcmp(stored, CBM_DOCLINK_CS_SCOPE_TAG "\n!\trejected\n") == 0; + char edited[2048]; + snprintf(edited, sizeof(edited), "%s\n// body-only edit\n", dm_rejected_files[1].text); + th_write_file(TH_PATH(repo, "Healthy.cs"), edited); + int read_back = dm_step(repo, db, full_db, "the rejected scope read back", + CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR); + snprintf(edited, sizeof(edited), "%s\n// body-only edit\n", dm_rejected_files[0].text); + th_write_file(TH_PATH(repo, "Bad.cs"), edited); + int again = dm_step(repo, db, full_db, "the rejected file edited", + CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR); + cbm_doclink_cs_test_spoil_scope(NULL); + int errors = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'"); + int bad = 0; + for (int i = 0; i < DM_COUNT(dm_rejected_wants); i++) { + bad += dm_want_failed(db, &dm_rejected_wants[i]); } dm_unlink_db(db); + dm_unlink_db(full_db); th_rmtree(tmp); + ASSERT_EQ(indexed, 0); + ASSERT_TRUE(marked); + ASSERT_EQ(read_back, 0); + ASSERT_EQ(again, 0); ASSERT_EQ(errors, 0); ASSERT_EQ(bad, 0); PASS(); @@ -10162,4 +10238,5 @@ SUITE(doc_mentions) { RUN_TEST(doc_mentions_alloc_failure_status); RUN_TEST(doc_mentions_cs_scanner_output_reads_back); RUN_TEST(doc_mentions_cs_rejected_scope_contained); + RUN_TEST(doc_mentions_cs_rejected_scope_across_runs); } From 13e3036e7278cc4f5ffa7686494482aa0501f9fd Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Mon, 5 Oct 2026 02:32:34 +0200 Subject: [PATCH 11/19] doc-links(cs): a doc-line map that cannot grow fails the layer (R5) cbm_doclink_note_doc_line returned silently when the map (or its first block) could not be allocated. The doc it was noting then keeps its definition's line, and a doc shared by the declarators of one field declaration is no longer marked as taken, so its references are parsed once per declarator -- while the layer reports ok. Both allocation failures now set the file's doc_links.failed, as every other lost piece of doc-link data does; the layer then says error. A test seam, CBM_DOCLINK_ALLOC_DOC_LINE, fails the map's growth. Test: alloc_failure_status gets the doc-line map as one more point (one error row). Signed-off-by: Martin Vogel --- internal/cbm/doclink.c | 23 +++++++++++++++++++++-- internal/cbm/doclink.h | 5 +++-- tests/test_doc_mentions.c | 11 ++++++----- 3 files changed, 30 insertions(+), 9 deletions(-) diff --git a/internal/cbm/doclink.c b/internal/cbm/doclink.c index 81d5a620f4..d7d9b8b419 100644 --- a/internal/cbm/doclink.c +++ b/internal/cbm/doclink.c @@ -195,6 +195,15 @@ static CBMArena *doclink_scratch(CBMExtractCtx *ctx) { return ctx->scratch ? ctx->scratch : ctx->arena; } +/* The map could not hold a doc: its references then take their definition's + * line, and a doc shared by several declarators is taken once per declarator. + * The layer must not say ok over that (CBMDocLinkArray.failed). */ +static void doc_line_lost(CBMExtractCtx *ctx) { + if (ctx->result) { + ctx->result->doc_links.failed = true; + } +} + void cbm_doclink_note_doc_line(CBMExtractCtx *ctx, const char *doc, uint32_t line) { if (!ctx || !doc || !cbm_doclink_lang_supported(ctx->language)) { return; @@ -204,6 +213,7 @@ void cbm_doclink_note_doc_line(CBMExtractCtx *ctx, const char *doc, uint32_t lin if (!m) { m = (doc_line_map_t *)cbm_arena_alloc(a, sizeof(*m)); if (!m) { + doc_line_lost(ctx); return; } memset(m, 0, sizeof(*m)); @@ -211,9 +221,18 @@ void cbm_doclink_note_doc_line(CBMExtractCtx *ctx, const char *doc, uint32_t lin } if (m->count >= m->cap) { int ncap = m->cap ? m->cap * PAIR_LEN : DOC_LINE_MAP_INIT; - doc_line_ent_t *grown = (doc_line_ent_t *)cbm_arena_alloc(a, (size_t)ncap * sizeof(*grown)); + doc_line_ent_t *grown; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (cbm_doclink_test_fail_alloc(CBM_DOCLINK_ALLOC_DOC_LINE)) { + grown = NULL; + } else +#endif + { + grown = (doc_line_ent_t *)cbm_arena_alloc(a, (size_t)ncap * sizeof(*grown)); + } if (!grown) { - return; /* the reference keeps its definition's line instead */ + doc_line_lost(ctx); + return; } if (m->count > 0) { memcpy(grown, m->items, (size_t)m->count * sizeof(*grown)); diff --git a/internal/cbm/doclink.h b/internal/cbm/doclink.h index 3588db1b09..7b85ad45a3 100644 --- a/internal/cbm/doclink.h +++ b/internal/cbm/doclink.h @@ -149,8 +149,9 @@ enum { CBM_DOCLINK_ALLOC_TEXT, CBM_DOCLINK_ALLOC_VALUE, CBM_DOCLINK_ALLOC_TOKENS, - CBM_DOCLINK_ALLOC_SCOPE, /* the C# scope scan, as if its builder ran out */ - CBM_DOCLINK_ALLOC_PROJECT, /* the project-file scan, likewise */ + CBM_DOCLINK_ALLOC_SCOPE, /* the C# scope scan, as if its builder ran out */ + CBM_DOCLINK_ALLOC_PROJECT, /* the project-file scan, likewise */ + CBM_DOCLINK_ALLOC_DOC_LINE, /* the doc-line map's growth */ CBM_DOCLINK_ALLOC_KINDS, }; void cbm_doclink_test_fail_alloc_after(int kind, int nth); diff --git a/tests/test_doc_mentions.c b/tests/test_doc_mentions.c index 6e9fe9bf5e..dabc7c2ced 100644 --- a/tests/test_doc_mentions.c +++ b/tests/test_doc_mentions.c @@ -9775,7 +9775,8 @@ TEST(doc_mentions_incremental_non_ascii_name) { * loses references, a scope or an edge -- the layer then says error, never * ok over a silently thinner graph. One allocation fails at each point in * turn (a doc span, a doc text, a reference value, a token, the scope scan, - * a MENTIONS edge); the run without a failure has no error row. */ + * the project scan, the doc-line map, a MENTIONS edge); the run without a + * failure has no error row. */ enum { DM_FAIL_EDGE = CBM_DOCLINK_ALLOC_KINDS }; static int dm_alloc_failure_errors(int point) { @@ -10103,10 +10104,10 @@ TEST(doc_mentions_alloc_failure_status) { int point; const char *what; } points[] = { - {CBM_DOCLINK_ALLOC_SPAN, "doc span"}, {CBM_DOCLINK_ALLOC_TEXT, "doc text"}, - {CBM_DOCLINK_ALLOC_VALUE, "value"}, {CBM_DOCLINK_ALLOC_TOKENS, "token"}, - {CBM_DOCLINK_ALLOC_SCOPE, "scope scan"}, {CBM_DOCLINK_ALLOC_PROJECT, "project scan"}, - {DM_FAIL_EDGE, "MENTIONS edge"}, + {CBM_DOCLINK_ALLOC_SPAN, "doc span"}, {CBM_DOCLINK_ALLOC_TEXT, "doc text"}, + {CBM_DOCLINK_ALLOC_VALUE, "value"}, {CBM_DOCLINK_ALLOC_TOKENS, "token"}, + {CBM_DOCLINK_ALLOC_SCOPE, "scope scan"}, {CBM_DOCLINK_ALLOC_PROJECT, "project scan"}, + {CBM_DOCLINK_ALLOC_DOC_LINE, "doc-line map"}, {DM_FAIL_EDGE, "MENTIONS edge"}, }; int baseline = dm_alloc_failure_errors(-1); bool ok = baseline == 0; From 9f2518d0f361eee79dfcc43a486c59b664c901f1 Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Mon, 5 Oct 2026 02:37:40 +0200 Subject: [PATCH 12/19] doc-links(cs): a stored region record with an empty name is refused (R8) The scanner names every namespace region it writes (each segment holds at least one identifier character). The reader took an empty name from the store all the same: ns_make_path then returns the parent namespace, so the region nests without deepening it, its types are declared one level up, and lookup asks the directives of fewer regions than the file has. parse_region now refuses an empty name field, so such a stored scope fails the run with bad_scope like any other record this code does not write, and the next index rebuilds from the files. Test: cs_damaged_stored_scope runs each damage in turn; the new case blanks the name of A's region record (one error row, no `missing` row for Target, then a forced full run that restores the edge). Signed-off-by: Martin Vogel --- src/pipeline/doc_links_cs.c | 5 +- tests/test_doc_mentions.c | 103 +++++++++++++++++++++++++----------- 2 files changed, 75 insertions(+), 33 deletions(-) diff --git a/src/pipeline/doc_links_cs.c b/src/pipeline/doc_links_cs.c index 0995c134c1..62b429b8b2 100644 --- a/src/pipeline/doc_links_cs.c +++ b/src/pipeline/doc_links_cs.c @@ -1002,8 +1002,9 @@ static void add_tparams(cs_file_t *f, int owner, char *list) { static bool parse_region(cs_index_t *ix, cs_file_t *f, char **fld, int n, int *next_region) { /* R id parent start end name */ - if (n != 6) { - return false; + if (n != 6 || !fld[5][0]) { + return false; /* the scanner names every region: an empty name would + * nest one without deepening the namespace */ } int id = field_index(fld[1], f->nregions); int parent = field_index(fld[2], f->nregions); diff --git a/tests/test_doc_mentions.c b/tests/test_doc_mentions.c index dabc7c2ced..9c9946d4d2 100644 --- a/tests/test_doc_mentions.c +++ b/tests/test_doc_mentions.c @@ -9396,14 +9396,17 @@ static int dm_exec(const char *db, const char *sql) { return changed; } -/* A stored scope this code did not write -- a damaged row -- is not read - * around: the file's declarations would silently be missing from every - * lookup, and references to them would be reported `missing`. The layer - * fails for that run, visibly, and the next index rebuilds from the files. */ -TEST(doc_mentions_cs_damaged_stored_scope) { +/* One damaged stored scope (A's row, `find` replaced by `put`): 0 when the + * run that reads it fails visibly and the next one rebuilds; else the step + * that went wrong (1 setup, 2 the damaged run, 3 the rebuild), with the + * damaged run's error rows in *errors. */ +static int dm_damaged_scope_run(const char *find, const char *put, int *errors) { + *errors = -1; char tmp[256]; snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_dmg_XXXXXX"); - ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + if (!cbm_mkdtemp(tmp)) { + return 1; + } char repo[400]; snprintf(repo, sizeof(repo), "%s/repo", tmp); th_write_file(TH_PATH(repo, "src/A.cs"), "namespace N\n{\n public class Target { }\n}\n"); @@ -9414,35 +9417,73 @@ TEST(doc_mentions_cs_damaged_stored_scope) { "}\n"); char db[512]; snprintf(db, sizeof(db), "%s/dmg.db", tmp); - ASSERT_EQ(dm_index(repo, db, NULL), 0); + char sql[512]; + snprintf(sql, sizeof(sql), + "UPDATE lsp_surface SET defs_json = replace(defs_json, '%s', '%s') " + "WHERE rel_path = 'src/A.cs' AND instr(defs_json, '%s') > 0", + find, put, find); char props[512]; int n = 0; - dm_edge(db, "B.Uses", "A.Target", props, sizeof(props), &n); - ASSERT_EQ(n, 1); - /* one field too many in A's type record: no record this code writes */ - ASSERT_EQ(dm_exec(db, "UPDATE lsp_surface SET defs_json = " - "replace(defs_json, '\\nT\\t', '\\nT\\tX\\t') " - "WHERE rel_path = 'src/A.cs' AND instr(defs_json, '\\nT\\t') > 0"), - 1); - th_write_file(TH_PATH(repo, "src/B.cs"), - "namespace N\n" - "{\n" - " /// Again \n" - " public class Uses { }\n" - "}\n"); - cbm_pipeline_incremental_test_reset_faults(); - ASSERT_EQ(dm_index(repo, db, NULL), 0); - ASSERT_EQ(dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'"), 1); - /* never the row an index without A's declarations would write */ - ASSERT_EQ(dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE raw = 'Target'"), 0); - /* the next run rebuilds everything, and the edge is back */ - ASSERT_EQ(dm_index(repo, db, NULL), 0); - ASSERT_EQ(cbm_pipeline_incremental_test_last_route(), CBM_INCREMENTAL_ROUTE_FORCED_FULL); - ASSERT_EQ(dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'"), 0); - dm_edge(db, "B.Uses", "A.Target", props, sizeof(props), &n); - ASSERT_EQ(n, 1); + int step = 1; + bool ok = dm_index(repo, db, NULL) == 0; + if (ok) { + dm_edge(db, "B.Uses", "A.Target", props, sizeof(props), &n); + ok = n == 1 && dm_exec(db, sql) == 1; + } + if (ok) { + step = 2; + th_write_file(TH_PATH(repo, "src/B.cs"), + "namespace N\n" + "{\n" + " /// Again \n" + " public class Uses { }\n" + "}\n"); + cbm_pipeline_incremental_test_reset_faults(); + ok = dm_index(repo, db, NULL) == 0; + *errors = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'"); + /* never the row an index without A's declarations would write */ + ok = ok && *errors == 1 && + dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE raw = 'Target'") == 0; + } + if (ok) { + /* the next run rebuilds everything, and the edge is back */ + step = 3; + ok = dm_index(repo, db, NULL) == 0 && + cbm_pipeline_incremental_test_last_route() == CBM_INCREMENTAL_ROUTE_FORCED_FULL && + dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'") == 0; + if (ok) { + dm_edge(db, "B.Uses", "A.Target", props, sizeof(props), &n); + ok = n == 1; + } + } dm_unlink_db(db); th_rmtree(tmp); + return ok ? 0 : step; +} + +/* A stored scope this code did not write -- a damaged row -- is not read + * around: the file's declarations would silently be missing from every + * lookup, and references to them would be reported `missing`. The layer + * fails for that run, visibly, and the next index rebuilds from the files. + * R8: a region with an empty name is such a record (the scanner never writes + * one); read, it would nest a region without deepening the namespace. */ +TEST(doc_mentions_cs_damaged_stored_scope) { + static const struct { + const char *what, *find, *put; + } damages[] = { + {"one field too many in A's type record", "\\nT\\t", "\\nT\\tX\\t"}, + {"A's region without a name", "\\nR\\t1\\t0\\t0\\t0\\tN\\n", "\\nR\\t1\\t0\\t0\\t0\\t\\n"}, + }; + bool ok = true; + for (size_t i = 0; i < sizeof(damages) / sizeof(damages[0]); i++) { + int errors = -1; + int step = dm_damaged_scope_run(damages[i].find, damages[i].put, &errors); + if (step != 0) { + printf(" %s: step %d went wrong (%d error rows)\n", damages[i].what, step, errors); + ok = false; + } + } + ASSERT_TRUE(ok); PASS(); } From 6573606ab6aac152ab324d998c1c299398eb915f Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Mon, 5 Oct 2026 02:43:10 +0200 Subject: [PATCH 13/19] doc-links(cs): a props or targets file that does not parse up to its root is unknown (R6) The project-file scan wrote the "could not be read" blob for a *.csproj that is no readable MSBuild , and for a *.props or *.targets file only once its root was seen. A *.props or *.targets file that is malformed before its root, or ends before one, got no blob at all: an import of it counted as outside the repository with a closed scope, and as the nearest Directory.Build.* it was passed over for the one above, whose usings were then applied although MSBuild does not read that file. Such a file now gets the "could not be read" blob too, so the projects that evaluate it have an open scope. A *.props or *.targets file whose root element is not is still no MSBuild file and has no blob. Test: msbuild_malformed_props_opens (a malformed nearest Directory.Build.props under a usable one: a name found nowhere is external, and a type only the upper file's using reaches is not bound; decoy: a project without a nearer file binds through the upper file and keeps a closed scope). Signed-off-by: Martin Vogel --- internal/cbm/doclink_cs.c | 13 ++++++++++--- tests/test_doc_mentions.c | 40 +++++++++++++++++++++++++++++++++++++++ 2 files changed, 50 insertions(+), 3 deletions(-) diff --git a/internal/cbm/doclink_cs.c b/internal/cbm/doclink_cs.c index db09f61824..57780da9ce 100644 --- a/internal/cbm/doclink_cs.c +++ b/internal/cbm/doclink_cs.c @@ -3707,8 +3707,8 @@ const char *cbm_doclink_cs_project_scan_scope(CBMExtractCtx *ctx) { return blob; } /* a *.csproj marks its directory as a C# project even when it cannot be - * read; a *.props or *.targets file has a blob only as an MSBuild - * */ + * read; a *.props or *.targets file has no blob only when it parses up to + * a root element that is not an MSBuild */ csx_scan_t p = { .x = {.src = ctx->source, .n = ctx->source_len > 0 ? (uint32_t)ctx->source_len : 0}, .out = {.a = ctx->arena}, @@ -3719,6 +3719,7 @@ const char *cbm_doclink_cs_project_scan_scope(CBMExtractCtx *ctx) { p.x.i = 3; } bool is_project = false; + bool other_root = false; /* XML whose root element is not */ bool bad = false; csx_tok_t t; for (;;) { @@ -3751,6 +3752,7 @@ const char *cbm_doclink_cs_project_scan_scope(CBMExtractCtx *ctx) { break; } if (!csx_is(&p.x, t.name, "Project")) { + other_root = true; break; /* XML, but no MSBuild file */ } is_project = true; @@ -3762,7 +3764,12 @@ const char *cbm_doclink_cs_project_scan_scope(CBMExtractCtx *ctx) { } } cs_cost_add(p.x.i, 0); - if ((bad && (is_project || named_project)) || (!is_project && named_project)) { + /* What the scan could not read is unknown, not absent: a file malformed + * anywhere or ending before its root, and a *.csproj that is no MSBuild + * . With no blob, an import of such a file would count as one + * outside the repository, and as the nearest Directory.Build.* it would + * be passed over for the one above, which MSBuild does not read. */ + if (bad || (!is_project && (named_project || !other_root))) { p.out.len = 0; sb_puts(&p.out, CBM_DOCLINK_CS_SCOPE_TAG "\nP\t\t!\n"); } else if (!is_project) { diff --git a/tests/test_doc_mentions.c b/tests/test_doc_mentions.c index 9c9946d4d2..9e1bee8f05 100644 --- a/tests/test_doc_mentions.c +++ b/tests/test_doc_mentions.c @@ -9698,6 +9698,45 @@ TEST(doc_mentions_msbuild_unread_project_opens) { PASS(); } +/* R6: a *.props or *.targets file that does not parse up to its root is not + * read either: what it holds is unknown. As the nearest Directory.Build.props + * it is the one MSBuild reads, so the scope of its project is open (a simple + * name found nowhere is external) and the usable file further up is not + * applied (its `Using Include="Lib"` imports nothing here). Decoy: a project + * with no nearer file reads the upper one (Gadget binds through it, a name + * found nowhere is missing). */ +TEST(doc_mentions_msbuild_malformed_props_opens) { + const dm_source_t files[] = { + {"Directory.Build.props", "" + "\n"}, + {"a/Directory.Build.props", "<\n" + "\n"}, + {"a/A.csproj", DM_EMPTY_PROJECT}, + {"a/Thing.cs", "namespace Lib\n{\n public class Thing { }\n}\n"}, + {"a/UsesA.cs", "namespace App\n" + "{\n" + " /// \n" + " public class UsesA { }\n" + "}\n"}, + {"c/C.csproj", DM_EMPTY_PROJECT}, + {"c/Gadget.cs", "namespace Lib\n{\n public class Gadget { }\n}\n"}, + {"c/UsesC.cs", + "namespace App\n" + "{\n" + " /// \n" + " public class UsesC { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"a/UsesA.cs", "NowhereA", "UsesA.UsesA", NULL, "external", NULL, NULL}, + {"a/UsesA.cs", "Thing", "UsesA.UsesA", NULL, "external", "Thing.Thing", NULL}, + {"c/UsesC.cs", "Gadget", "UsesC.UsesC", "Gadget.Gadget", NULL, NULL, NULL}, + {"c/UsesC.cs", "NowhereC", "UsesC.UsesC", NULL, "missing", NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("malformed_props", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + enum { DM_BIG_MEMBERS = 20000, DM_SMALL_FILES = 200 }; /* What the resolver looks at to build its index over one file of @@ -10275,6 +10314,7 @@ SUITE(doc_mentions) { RUN_TEST(doc_mentions_cs_stored_scope_nesting); RUN_TEST(doc_mentions_index_status_unknown_reason); RUN_TEST(doc_mentions_msbuild_unread_project_opens); + RUN_TEST(doc_mentions_msbuild_malformed_props_opens); RUN_TEST(doc_mentions_cs_scratch_table_work); RUN_TEST(doc_mentions_incremental_non_ascii_name); RUN_TEST(doc_mentions_alloc_failure_status); From 19deeff285968a68ff916839e761356d4cf8ebfa Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Mon, 5 Oct 2026 02:50:19 +0200 Subject: [PATCH 14/19] doc-links(cs): a directives-only file that loses its scope fails the layer in both routes (R7) The re-check suspected that the worker pipeline never reads the doc-link failure flag of a C# file holding only directives, because its resolve loop skips a file with no definitions, calls, usages, throws, reads or writes, or impl traits. That case does not occur: every extraction pushes the file's Module definition first (cbm_extract_definitions, before the doc-link driver runs, with no return in between), so such a file has one definition and reaches cbm_doclinks_resolve_file, which reads the flag. No production code changes. Test: directives_only_failure (a repository of 62 files whose only C# file holds two global usings, its scope scan failing: one error row with four workers and with one; none without the failure; the file has nothing but its Module). Making the skip pass over a Module-only file turns the four-worker case red (0 error rows), so the test binds that path. Signed-off-by: Martin Vogel --- tests/test_doc_mentions.c | 78 +++++++++++++++++++++++++++++++++++++++ 1 file changed, 78 insertions(+) diff --git a/tests/test_doc_mentions.c b/tests/test_doc_mentions.c index 9e1bee8f05..55f252cc8f 100644 --- a/tests/test_doc_mentions.c +++ b/tests/test_doc_mentions.c @@ -10179,6 +10179,83 @@ TEST(doc_mentions_cs_rejected_scope_across_runs) { PASS(); } +enum { DM_DIRECTIVE_FILLERS = 60 }; /* with the C# file: above MIN_FILES_FOR_PARALLEL */ + +static const char DM_DIRECTIVES_ONLY[] = "global using System;\nglobal using System.IO;\n"; + +/* Error rows of one index of a repository whose only C# file holds nothing + * but global usings, its scope scan failing when `fail`. `workers` "4" runs + * the worker pipeline, "1" the sequential passes (CBM_WORKERS is restored). + * -1 when the repository cannot be indexed. */ +static int dm_directives_only_errors(const char *workers, bool fail) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_r7_XXXXXX"); + if (!cbm_mkdtemp(tmp)) { + return -1; + } + char repo[400]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + th_write_file(TH_PATH(repo, "src/App.csproj"), DM_EMPTY_PROJECT); + th_write_file(TH_PATH(repo, "src/Usings.cs"), DM_DIRECTIVES_ONLY); + for (int i = 0; i < DM_DIRECTIVE_FILLERS; i++) { + char path[512]; + snprintf(path, sizeof(path), "%s/py/m%02d.py", repo, i); + th_write_file(path, "def f():\n return 1\n"); + } + const char *saved = getenv("CBM_WORKERS"); + char *saved_copy = saved ? strdup(saved) : NULL; + cbm_setenv("CBM_WORKERS", workers, 1); + if (fail) { + cbm_doclink_test_fail_alloc_after(CBM_DOCLINK_ALLOC_SCOPE, 1); /* the one C# file */ + } + char db[512]; + snprintf(db, sizeof(db), "%s/r7.db", tmp); + int errors = + dm_index(repo, db, NULL) == 0 + ? dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'") + : -1; + cbm_doclink_test_reset_alloc(); + if (saved_copy) { + cbm_setenv("CBM_WORKERS", saved_copy, 1); + free(saved_copy); + } else { + cbm_unsetenv("CBM_WORKERS"); + } + dm_unlink_db(db); + th_rmtree(tmp); + return errors; +} + +/* R7: a C# file of nothing but global usings has no call, usage, throw, + * read/write or impl trait; its one definition is the Module every + * extraction pushes first. That keeps it out of the worker pipeline's skip + * of files with nothing to resolve, so its doc-link failure flag is read + * there: its scope scan running out of memory fails the layer in both + * routes. The fixture holds only while the file has nothing but the Module. */ +TEST(doc_mentions_directives_only_failure) { + CBMFileResult *r = dm_extract(DM_DIRECTIVES_ONLY, CBM_LANG_CSHARP, "src/Usings.cs"); + ASSERT_NOT_NULL(r); + bool module_only = r->calls.count == 0 && r->usages.count == 0 && r->throws.count == 0 && + r->rw.count == 0 && r->impl_traits.count == 0 && r->defs.count == 1 && + strcmp(r->defs.items[0].label, "Module") == 0; + cbm_free_result(r); + ASSERT_TRUE(module_only); + static const char *const routes[] = {"4", "1"}; + bool ok = true; + for (size_t k = 0; k < sizeof(routes) / sizeof(routes[0]); k++) { + int clean = dm_directives_only_errors(routes[k], false); + int failed = dm_directives_only_errors(routes[k], true); + if (clean != 0 || failed != 1) { + printf(" %s worker(s): %d error rows without a failure (want 0), %d with one (want " + "1)\n", + routes[k], clean, failed); + ok = false; + } + } + ASSERT_TRUE(ok); + PASS(); +} + TEST(doc_mentions_alloc_failure_status) { static const struct { int point; @@ -10318,6 +10395,7 @@ SUITE(doc_mentions) { RUN_TEST(doc_mentions_cs_scratch_table_work); RUN_TEST(doc_mentions_incremental_non_ascii_name); RUN_TEST(doc_mentions_alloc_failure_status); + RUN_TEST(doc_mentions_directives_only_failure); RUN_TEST(doc_mentions_cs_scanner_output_reads_back); RUN_TEST(doc_mentions_cs_rejected_scope_contained); RUN_TEST(doc_mentions_cs_rejected_scope_across_runs); From 177259a1cb4a8b7361458467645b169d0f059ee8 Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Mon, 5 Oct 2026 03:03:40 +0200 Subject: [PATCH 15/19] doc-links(cs): the type names of a refused scope are quarantined, in every run (R4) A scope written in this run that the reader refuses costs its own file: what its parse set is taken back and the file declares nothing. Other files then found none of its declarations either: a reference to one of its types was `missing`, or bound uniquely to another type of that name which the full picture would not choose (a type of the enclosing namespace comes before an imported one). Unplaced declarations are quarantined for exactly this reason; a refused file's were not. Now the type names the refused blob's T and Q records carry, read leniently (a line counts when its name field is there and is a name, however the rest of it reads), are quarantined, so a reference to one of them from any file is a graph gap. The stored marker carries the same names as Q records after its `!` record, so a later run that reads the marker back quarantines them too and an incremental run equals a full one; cs_scope_marked_rejected takes the marker followed only by Q records of names. The resolver's rejected_scope is therefore a function of the refused blob, and cbm_doclinks_storable_scope hands the marker to its caller to free (lsp_surface.c). One S8 expectation changes with this: Healthy.cs's reference to Hidden2, a name Bad.cs declares where no namespace can be established (a Q record), is now a graph gap rather than an edge -- as it is when Bad.cs's scope is taken. The namespace checks of S8 (Shadow, Good.Lurker) are unchanged. Test: cs_rejected_scope_contained and cs_rejected_scope_across_runs, whose shared fixture gains Good.Widget in the refused file and Lib.Widget in an imported namespace: Healthy.cs's Widget is a graph gap with no edge, and the stored marker holds the five names. Signed-off-by: Martin Vogel --- src/pipeline/doc_links.c | 9 +- src/pipeline/doc_links.h | 19 +++-- src/pipeline/doc_links_cs.c | 158 +++++++++++++++++++++++++++++------- src/pipeline/lsp_surface.c | 4 +- tests/test_doc_mentions.c | 39 +++++++-- 5 files changed, 183 insertions(+), 46 deletions(-) diff --git a/src/pipeline/doc_links.c b/src/pipeline/doc_links.c index fa4b0096ab..8ac85fcb57 100644 --- a/src/pipeline/doc_links.c +++ b/src/pipeline/doc_links.c @@ -113,7 +113,8 @@ int cbm_doclinks_scope_delta(const char *stored, const char *fresh, cbm_doclink_ return R->scope_delta(stored, fresh, removed, ud); } -const char *cbm_doclinks_storable_scope(const char *scope) { +const char *cbm_doclinks_storable_scope(const char *scope, char **owned) { + *owned = NULL; const cbm_doclink_resolver_t *R = resolver_of_scope(scope); if (!R || !R->scope_accepted || !R->rejected_scope) { return scope; @@ -122,7 +123,11 @@ const char *cbm_doclinks_storable_scope(const char *scope) { if (accepted < 0) { return NULL; } - return accepted ? scope : R->rejected_scope; + if (accepted) { + return scope; + } + *owned = R->rejected_scope(scope); + return *owned; } /* ── Run state ───────────────────────────────────────────────────── */ diff --git a/src/pipeline/doc_links.h b/src/pipeline/doc_links.h index b175ebd76a..be99cd5097 100644 --- a/src/pipeline/doc_links.h +++ b/src/pipeline/doc_links.h @@ -214,13 +214,14 @@ typedef struct { void *ud); /* Optional, together: whether the reader takes a scope blob written in * this run (1), refuses it (0), or could not tell for memory (-1); and - * what is stored instead of a refused blob. A run refuses the blob of its - * own file and goes on without the file's declarations; the marker makes - * a later run that reads the file's scope back do the same, where the - * blob itself would be a stored scope the reader refuses, which fails + * the marker stored instead of a refused blob (a memory-core block of + * CBM_MEM_CLASS_OTHER, NULL when memory ran out). A run refuses the blob + * of its own file and goes on without the file's declarations; the marker + * makes a later run that reads the file's scope back do the same, where + * the blob itself would be a stored scope the reader refuses, which fails * the run. */ int (*scope_accepted)(const char *scope); - const char *rejected_scope; + char *(*rejected_scope)(const char *scope); } cbm_doclink_resolver_t; extern const cbm_doclink_resolver_t cbm_doclink_cs_resolver; @@ -255,8 +256,10 @@ int cbm_doclinks_scope_delta(const char *stored, const char *fresh, cbm_doclink_ /* What is stored for a scope blob written in this run (the surface row's * `dl`, lsp_surface.c): the blob itself, or its language's rejected-scope - * marker when that language's reader refuses it (scope_accepted). NULL when - * memory ran out deciding. */ -const char *cbm_doclinks_storable_scope(const char *scope); + * marker when that language's reader refuses it (scope_accepted), made into + * *owned (release with cbm_free(CBM_MEM_CLASS_OTHER, *owned); NULL when the + * blob itself is returned). NULL when memory ran out deciding or making the + * marker. */ +const char *cbm_doclinks_storable_scope(const char *scope, char **owned); #endif /* CBM_PIPELINE_DOC_LINKS_H */ diff --git a/src/pipeline/doc_links_cs.c b/src/pipeline/doc_links_cs.c index 62b429b8b2..e397266460 100644 --- a/src/pipeline/doc_links_cs.c +++ b/src/pipeline/doc_links_cs.c @@ -5926,16 +5926,131 @@ static void undo_scope(cs_index_t *ix) { u->nmarks = 0; } +/* Scope line fields (0-based, the tag is field 0; internal/cbm/doclink_cs.c): + * `U region kind alias target`, `T region start end kind outer name tparams + * bases`, `M start kind explicit type name tparams sig` and `Q name`. */ +enum { + CS_SCOPE_Q_NAME = 1, + CS_SCOPE_U_KIND = 2, + CS_SCOPE_M_NAME = 5, + CS_SCOPE_T_NAME = 6, + CS_SCOPE_T_BASES = 8 +}; + +static const char *delta_field(const char *line, size_t len, int idx, size_t *flen) { + int f = 0; + size_t s = 0; + for (size_t i = 0; i <= len; i++) { + if (i == len || line[i] == '\t') { + if (f == idx) { + *flen = i - s; + return line + s; + } + f++; + s = i + SKIP_ONE; + } + } + *flen = 0; + return NULL; +} + +/* True when s[0, n) is a name a reference can carry (ident_ok). */ +static bool cs_name_ok(const char *s, size_t n) { + char buf[CS_NAME_BUF]; + if (n == 0 || n >= sizeof(buf)) { + return false; + } + memcpy(buf, s, n); + buf[n] = '\0'; + return ident_ok(buf); +} + +/* The next type name a T or Q record of a scope blob carries, read leniently + * (the blob is one the reader refused): a line is taken when its name field + * is there and is a name, however the rest of it reads. Moves *cursor past + * the lines it read; false at the end of the blob. */ +static bool cs_next_type_name(const char **cursor, const char **name, size_t *len) { + while (**cursor) { + const char *line = *cursor; + const char *nl = strchr(line, '\n'); + size_t n = nl ? (size_t)(nl - line) : strlen(line); + *cursor = line + n + (nl ? SKIP_ONE : 0); + int idx = line[0] == 'T' ? CS_SCOPE_T_NAME : line[0] == 'Q' ? CS_SCOPE_Q_NAME : CS_NONE; + *name = idx > 0 && n > SKIP_ONE && line[SKIP_ONE] == '\t' ? delta_field(line, n, idx, len) + : NULL; + if (*name && cs_name_ok(*name, *len)) { + return true; + } + } + return false; +} + /* What is stored, in place of its scope blob, for a file whose scope the * reader refused in the run that wrote it (cs_scope_accepted, lsp_surface.c): - * the tag and one `!` record. A later run that reads it back treats the file - * as that run did -- it declares nothing, its references are graph gaps -- - * where the refused blob itself would fail that run as a stored scope this - * reader does not take. */ + * the tag, one `!` record, and one Q record per type name the T and Q records + * of the refused blob carry (cs_next_type_name). A later run that reads it + * back treats the file as that run did -- it declares nothing, those names + * are quarantined, its references are graph gaps -- where the refused blob + * itself would fail that run as a stored scope this reader does not take. */ static const char CS_REJECTED_SCOPE[] = CBM_DOCLINK_CS_SCOPE_TAG "\n!\trejected\n"; +/* The marker, followed by nothing but Q records of names. */ static bool cs_scope_marked_rejected(const char *scope) { - return scope && strcmp(scope, CS_REJECTED_SCOPE) == 0; + size_t n = sizeof(CS_REJECTED_SCOPE) - SKIP_ONE; + if (!scope || strncmp(scope, CS_REJECTED_SCOPE, n) != 0) { + return false; + } + for (const char *line = scope + n; *line;) { + const char *nl = strchr(line, '\n'); + if (!nl || line[0] != 'Q' || line[SKIP_ONE] != '\t' || + !cs_name_ok(line + PAIR_LEN, (size_t)(nl - line) - PAIR_LEN)) { + return false; + } + line = nl + SKIP_ONE; + } + return true; +} + +/* The marker stored for the refused blob `scope` (the resolver's + * rejected_scope): a memory-core block, NULL when memory ran out. A Q record + * is never longer than the line its name is read from plus a newline. */ +static char *cs_rejected_scope(const char *scope) { + size_t w = sizeof(CS_REJECTED_SCOPE) - SKIP_ONE; + char *out = (char *)cbm_alloc(CBM_MEM_CLASS_OTHER, w + strlen(scope) + PAIR_LEN); + if (!out) { + return NULL; + } + memcpy(out, CS_REJECTED_SCOPE, w); + const char *cursor = scope; + const char *name = NULL; + size_t len = 0; + while (cs_next_type_name(&cursor, &name, &len)) { + out[w++] = 'Q'; + out[w++] = '\t'; + memcpy(out + w, name, len); + w += len; + out[w++] = '\n'; + } + out[w] = '\0'; + return out; +} + +/* Quarantine the type names a refused scope blob, or its stored marker, + * carries: what the file declares is unknown, so a reference to one of them + * from any file is a graph gap. false when memory ran out. */ +static bool quarantine_refused(cs_index_t *ix, const cs_file_t *f, const char *scope) { + const char *cursor = scope; + const char *name = NULL; + size_t len = 0; + while (cs_next_type_name(&cursor, &name, &len)) { + char buf[CS_NAME_BUF]; + memcpy(buf, name, len); /* shorter than the buffer: cs_name_ok */ + buf[len] = '\0'; + if (!quarantine_name(ix, f, buf)) { + return false; + } + } + return true; } /* Whether this reader takes a scope blob written in this run: its record @@ -5989,7 +6104,10 @@ bool cbm_doclink_cs_test_scope_parses(const char *scope) { * that is not known), and its path is returned. A scope written in this run * that the reader refuses is the scanner's and this reader's disagreement, * which costs only its own file: what its parse set is taken back, the file - * declares nothing, and its references are graph gaps (counted, logged). + * declares nothing, and its references are graph gaps (counted, logged). The + * names of its types are quarantined: what it declares is unknown, so a + * reference to one of them from any file is a graph gap too. A scope stored + * as the rejected marker is read as in the run that refused it. * Returns "" when memory ran out, NULL when all is well. */ static const char *build_files(cs_index_t *ix, const cbm_doclink_build_in_t *in) { CBMHashTable *dir_unit = cbm_ht_create(CBM_SZ_1K); @@ -6018,6 +6136,10 @@ static const char *build_files(cs_index_t *ix, const cbm_doclink_build_in_t *in) .is_test = f->is_test, .rejected = true}; ix->rejected++; + if (!quarantine_refused(ix, f, scope)) { + bad = ""; /* memory ran out */ + break; + } scope = NULL; } if (!scope) { @@ -6252,28 +6374,6 @@ static void cs_resolve(const void *index, void *state, int run_file, const CBMDo /* ── Incremental scope rules ─────────────────────────────────────── */ -/* Scope line fields (0-based, the tag is field 0; internal/cbm/doclink_cs.c): - * `U region kind alias target`, `T region start end kind outer name tparams - * bases` and `M start kind explicit type name tparams sig`. */ -enum { CS_SCOPE_U_KIND = 2, CS_SCOPE_M_NAME = 5, CS_SCOPE_T_BASES = 8 }; - -static const char *delta_field(const char *line, size_t len, int idx, size_t *flen) { - int f = 0; - size_t s = 0; - for (size_t i = 0; i <= len; i++) { - if (i == len || line[i] == '\t') { - if (f == idx) { - *flen = i - s; - return line + s; - } - f++; - s = i + SKIP_ONE; - } - } - *flen = 0; - return NULL; -} - /* A using directive only its own file sees: one that is not `global`. */ static bool delta_local_using(const char *line, size_t len) { if (line[0] != 'U') { @@ -6409,5 +6509,5 @@ const cbm_doclink_resolver_t cbm_doclink_cs_resolver = { .resolve = cs_resolve, .scope_delta = cs_scope_delta, .scope_accepted = cs_scope_accepted, - .rejected_scope = CS_REJECTED_SCOPE, + .rejected_scope = cs_rejected_scope, }; diff --git a/src/pipeline/lsp_surface.c b/src/pipeline/lsp_surface.c index 0d281f05d2..c96d48ad98 100644 --- a/src/pipeline/lsp_surface.c +++ b/src/pipeline/lsp_surface.c @@ -165,8 +165,10 @@ static char *surface_file_to_json(const CBMFileResult *result, const CBMLSPDef * /* a scope this run's reader refuses is stored as its language's * rejected marker: a later run then treats the file as this one does * (doc_links.h, scope_accepted) */ - const char *kept = cbm_doclinks_storable_scope(result->doc_scope); + char *marker = NULL; + const char *kept = cbm_doclinks_storable_scope(result->doc_scope, &marker); char *portable = kept ? cbm_doclink_portable_scope(kept) : NULL; + cbm_free(CBM_MEM_CLASS_OTHER, marker); if (!portable) { yyjson_mut_doc_free(doc); return NULL; /* a missing scope would diverge silently: fail the row */ diff --git a/tests/test_doc_mentions.c b/tests/test_doc_mentions.c index 55f252cc8f..43f253e538 100644 --- a/tests/test_doc_mentions.c +++ b/tests/test_doc_mentions.c @@ -10041,8 +10041,13 @@ TEST(doc_mentions_cs_scanner_output_reads_back) { * file. The status stays ok and the other file's edges are there; the file's * own reference is a graph gap; and nothing its parse set before the refusal * stays -- a namespace it declares does not stand beside a type of that name - * (Shadow, Good.Lurker), and a name it quarantined does not block another - * file's declaration (Hidden2). */ + * (Shadow, Good.Lurker). + * R4: the type names its T and Q records carry are quarantined, as the names + * of declarations no scope is known for are: a reference to one, from any + * file, is a graph gap -- never a binding to another type of that name the + * full picture would not choose (Widget: Good.Widget of Bad.cs comes before + * the imported Lib.Widget), nor to one the unplaced declaration would make + * uncertain (Hidden2, as when the scope is taken). */ /* S8: Bad.cs declares what would shadow and hide Healthy.cs's names; its * scope is spoiled where it is written (cbm_doclink_cs_test_spoil_scope). */ static const dm_source_t dm_rejected_files[] = { @@ -10056,12 +10061,14 @@ static const dm_source_t dm_rejected_files[] = { "\n" " /// \n" " public class FromBad { }\n" + " public class Widget { }\n" "}\n" "namespace .Broken\n" "{\n" " public class Hidden2 { }\n" "}\n"}, - {"Healthy.cs", "public class Shadow { }\n" + {"Healthy.cs", "using Lib;\n" + "public class Shadow { }\n" "namespace Good\n" "{\n" " public class Target { }\n" @@ -10069,16 +10076,22 @@ static const dm_source_t dm_rejected_files[] = { " public class Hidden2 { }\n" "\n" " /// \n" - " /// \n" + " /// \n" + " /// \n" " public class Uses { }\n" "}\n"}, + {"Lib.cs", "namespace Lib\n" + "{\n" + " public class Widget { }\n" + "}\n"}, }; static const dm_want_t dm_rejected_wants[] = { {"Healthy.cs", "Target", "Healthy.Uses", "Healthy.Target", NULL, NULL, NULL}, {"Healthy.cs", "Shadow", "Healthy.Uses", "Healthy.Shadow", NULL, NULL, NULL}, {"Healthy.cs", "Lurker", "Healthy.Uses", "Healthy.Lurker", NULL, NULL, NULL}, - {"Healthy.cs", "Hidden2", "Healthy.Uses", "Healthy.Hidden2", NULL, NULL, NULL}, + {"Healthy.cs", "Hidden2", "Healthy.Uses", NULL, "graph_gap", "Healthy.Hidden2", NULL}, + {"Healthy.cs", "Widget", "Healthy.Uses", NULL, "graph_gap", "Lib.Widget", NULL}, {"Bad.cs", "Target", "Bad.FromBad", NULL, "graph_gap", "Healthy.Target", NULL}, }; @@ -10151,7 +10164,21 @@ TEST(doc_mentions_cs_rejected_scope_across_runs) { int indexed = dm_index(repo, db, NULL); char stored[256]; dm_stored_scope(db, "Bad.cs", stored, sizeof(stored)); - bool marked = strcmp(stored, CBM_DOCLINK_CS_SCOPE_TAG "\n!\trejected\n") == 0; + /* the marker, then one Q record per type name of the refused blob (R4) */ + static const char *const names[] = {"Inner", "Deep", "FromBad", "Widget", "Hidden2"}; + static const char marker[] = CBM_DOCLINK_CS_SCOPE_TAG "\n!\trejected\n"; + bool marked = strncmp(stored, marker, sizeof(marker) - 1) == 0; + size_t want_len = sizeof(marker) - 1; + for (size_t k = 0; k < sizeof(names) / sizeof(names[0]); k++) { + char q[64]; + snprintf(q, sizeof(q), "\nQ\t%s\n", names[k]); + marked = marked && strstr(stored, q) != NULL; + want_len += strlen(q) - 1; + } + if (!marked || strlen(stored) != want_len) { + printf(" stored for Bad.cs: %s\n", stored); + marked = false; + } char edited[2048]; snprintf(edited, sizeof(edited), "%s\n// body-only edit\n", dm_rejected_files[1].text); th_write_file(TH_PATH(repo, "Healthy.cs"), edited); From d3ec77491942bab61ec96cf8855c1e22c03e86dd Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Mon, 5 Oct 2026 03:26:38 +0200 Subject: [PATCH 16/19] doc-links(cs): a unit's using directives are asked once per query, not once per file (R1) The using step of a lookup is remembered per file (S5), so the unit's directives -- its global usings and items, asked at the global namespace -- were walked again by every file that asks a name: files x min(directives, scopes declaring the name), a hang in resolution for a project with many global usings and a common name. What a unit directive gives a query reads only the context's unit, assembly and product flag (level_using and what it calls), never the file. The run now keeps, per (unit, assembly, product/test, query), the unit directives that give the query anything at all, in order, up to where they alone decide it (cs_unit_memo). Each file asks those after its own directives and stops where the walk of all of them stops. The edges and rows are unchanged: a directive that leaves an empty step empty makes no change in any step, and a step that starts with the file's candidates is decided no later than the unit's directives alone are (a candidate counts unless it is the step's first). The memo is made once the index is complete (the build's own lookups see directives that are not all resolved yet) and is shared by the resolve workers: each list is found exactly once, by the first worker that asks; the others that ask for that key take its lock until the list is there, so the work of a run does not depend on the scheduling. Out of memory, it stops remembering and the walk is the full one. A test seam builds the index without the memo. Tests: cs_unit_usings_across_files_work (k global usings, k scopes declaring the name, f files naming it once: k 150 -> 300 and f 100 -> 200 give 768 -> 1520 steps, 1.98 x; without the memo 17003 -> 64403, 3.79 x); cs_unit_usings_unchanged (the same edges and rows with and without the memo, over the cases of file and unit candidates: none, other, same, same then other, test-only, another unit). Signed-off-by: Martin Vogel --- src/pipeline/doc_links.h | 4 + src/pipeline/doc_links_cs.c | 210 ++++++++++++++++++++++++++++++++++-- tests/test_doc_mentions.c | 157 +++++++++++++++++++++++++++ 3 files changed, 365 insertions(+), 6 deletions(-) diff --git a/src/pipeline/doc_links.h b/src/pipeline/doc_links.h index be99cd5097..20f9fbfaa5 100644 --- a/src/pipeline/doc_links.h +++ b/src/pipeline/doc_links.h @@ -242,6 +242,10 @@ void cbm_doclinks_test_fail_edge_insert_after(int nth); /* Test seam (doc_links_cs.c): ask the reader whether it takes a scope blob. * (The scope of one file is spoiled where it is written: doclink.h.) */ bool cbm_doclink_cs_test_scope_parses(const char *scope); +/* Test seam (doc_links_cs.c): false builds the C# indexes from now on without + * the run's memo of the units' using steps (every unit step walks its + * directives), so a run can be held against one with it; true restores it. */ +void cbm_doclink_cs_test_unit_memo(bool on); #endif /* True when `rel_path` is a scope input of some language (scope_input). */ diff --git a/src/pipeline/doc_links_cs.c b/src/pipeline/doc_links_cs.c index e397266460..223e998116 100644 --- a/src/pipeline/doc_links_cs.c +++ b/src/pipeline/doc_links_cs.c @@ -155,6 +155,7 @@ #include "helpers.h" /* cbm_fqn_module_source_lang */ #include "pipeline/doc_links_msbuild.h" #include "foundation/arena.h" +#include "foundation/compat_thread.h" #include "foundation/constants.h" #include "foundation/hash_table.h" #include "foundation/log.h" @@ -547,6 +548,9 @@ typedef struct { bool bind_inherited; /* CS_BIND_INHERITED */ cs_undo_t undo; /* of the file being parsed */ int rejected; /* files whose scope, written in this run, was refused */ + /* the unit part of the using steps, shared by the resolve workers; made + * once the index is complete (NULL before, and when memory ran out) */ + struct cs_unit_memo *unit_memo; } cs_index_t; #if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS @@ -4597,8 +4601,53 @@ static int candidate_id_cmp(const void *a, const void *b) { return int_cmp(a, b); } +/* The directives of a using step that give a query anything, in order (R1): + * noted while the step is walked, for the unit memo. */ +typedef struct { + int *ids; + int n; + int cap; + bool failed; /* memory ran out: the list is not whole */ +} cs_active_rec_t; + +static void active_add(cs_active_rec_t *rec, int id) { + if (rec->n == rec->cap) { + int cap = rec->cap ? rec->cap * PAIR_LEN : CBM_SZ_16; + int *grown = (int *)cbm_alloc(CBM_MEM_CLASS_OTHER, (size_t)cap * sizeof(int)); + if (!grown) { + rec->failed = true; + return; + } + if (rec->n > 0) { + memcpy(grown, rec->ids, (size_t)rec->n * sizeof(int)); + } + cbm_free(CBM_MEM_CLASS_OTHER, rec->ids); + rec->ids = grown; + rec->cap = cap; + } + rec->ids[rec->n++] = id; +} + +/* Directive `id` of a step, for the query. With `rec`, it is noted when it + * gives the query anything at all: one that leaves an empty step empty adds + * no candidate and sets no flag whatever the step holds, so leaving it out + * changes no lookup. */ +static void using_visit(const cs_ctx_t *c, const cs_using_t *us, int id, const cs_query_t *q, + bool a_type_name, cs_found_t *fd, cs_active_rec_t *rec) { + if (rec) { + cs_found_t probe = {0}; + level_using(c, &us[id], q, a_type_name, &probe); + if (probe.n == 0 && !probe.invisible && !probe.joined && probe.why == CS_WHY_SCOPE) { + return; + } + active_add(rec, id); + } + level_using(c, &us[id], q, a_type_name, fd); +} + static void level_usings(const cs_ctx_t *c, const cs_using_t *us, int n, - const cs_using_index_t *index, const cs_query_t *q, cs_found_t *fd) { + const cs_using_index_t *index, const cs_query_t *q, cs_found_t *fd, + cs_active_rec_t *rec) { if (!n) { return; } @@ -4613,7 +4662,7 @@ static void level_usings(const cs_ctx_t *c, const cs_using_t *us, int n, * more postings than it has directives. The baseline scan stays cheap. */ if (!indexed || hi - lo >= (size_t)n) { for (int i = 0; i < n && !found_decided(fd); i++) { - level_using(c, &us[i], q, a_type_name, fd); + using_visit(c, us, i, q, a_type_name, fd, rec); } return; } @@ -4636,7 +4685,7 @@ static void level_usings(const cs_ctx_t *c, const cs_using_t *us, int n, /* Own and twin postings may select the SAME directive twice. * Distinct directive IDs must still be resolved separately. */ if (!i || ids.ids[i] != ids.ids[i - SKIP_ONE]) { - level_using(c, &us[ids.ids[i]], q, a_type_name, fd); + using_visit(c, us, ids.ids[i], q, a_type_name, fd, rec); } } } @@ -4645,7 +4694,7 @@ static void level_usings(const cs_ctx_t *c, const cs_using_t *us, int n, } if (!complete) { for (int i = 0; i < n && !found_decided(fd); i++) { - level_using(c, &us[i], q, a_type_name, fd); + using_visit(c, us, i, q, a_type_name, fd, rec); } } } @@ -4715,6 +4764,139 @@ static void memo_put(cs_memo_t *m, const char *key, const cs_found_t *step) { m->off = cbm_ht_get(m->steps, k) != v; /* an insert that did not take */ } +/* The unit part of a using step (R1): which of a unit's directives give a + * query anything, in order, up to where they alone decide it. What a + * directive gives reads only the context's unit, assembly and product flag + * (level_using), so the list is the same for every file of the unit that + * asks the query: it is found once per run, shared by the resolve workers + * (guarded), and each file then asks only those directives after its own -- + * stopping where the walk of all of them would, which is never later than + * where they alone are decided. Each list is found once: by the first + * worker that asks for it, while the others that ask for the same key hold + * on until it is there (a step's cost never depends on the scheduling). A + * memo that ran out of memory stops remembering; the lookups stay exact. */ +typedef struct cs_unit_step { + cbm_mutex_t mu; /* held by the worker that finds the list, until it is there */ + int *ids; /* the directives, in order (memory-core block, or NULL) */ + int n; + bool failed; /* memory ran out finding it: askers walk all directives */ + struct cs_unit_step *next; /* every step, for the memo's release */ +} cs_unit_step_t; + +typedef struct cs_unit_memo { + cbm_mutex_t mu; /* guards steps, arena, all and off */ + CBMHashTable *steps; /* key -> cs_unit_step_t in `arena` */ + CBMArena arena; + cs_unit_step_t *all; + bool off; +} cs_unit_memo_t; + +static cs_unit_memo_t *unit_memo_new(void) { + cs_unit_memo_t *m = (cs_unit_memo_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*m)); + if (!m) { + return NULL; + } + cbm_mutex_init(&m->mu); + cbm_arena_init_lazy(&m->arena, CBM_ARENA_APPEND_BLOCK); + return m; +} + +static void unit_memo_free(cs_unit_memo_t *m) { + if (!m) { + return; + } + for (cs_unit_step_t *e = m->all; e; e = e->next) { + cbm_mutex_destroy(&e->mu); + cbm_free(CBM_MEM_CLASS_OTHER, e->ids); + } + cbm_ht_free(m->steps); + cbm_arena_destroy(&m->arena); + cbm_mutex_destroy(&m->mu); + cbm_free(CBM_MEM_CLASS_OTHER, m); +} + +/* The key of one unit step: everything it reads. false when it does not fit. */ +static bool unit_key(char *buf, size_t cap, const cs_ctx_t *c, const cs_query_t *q) { + int n = snprintf(buf, cap, "%d|%d|%d|%d|%d%d%d%c|%s", c->unit, c->group, (int)c->prod, q->arity, + (int)q->types_only, (int)q->statics, (int)q->ctors, q->kind ? q->kind : '-', + q->name); + return n > 0 && (size_t)n < cap; +} + +/* The step under `key` (the caller holds the memo's lock); a new one is + * added with its lock held, and *mine set: the caller finds its list. NULL + * when memory ran out. */ +static cs_unit_step_t *unit_step_at(cs_unit_memo_t *m, const char *key, bool *mine) { + *mine = false; + if (!m->steps) { + m->steps = cbm_ht_create(CBM_SZ_64); + if (!m->steps) { + return NULL; + } + } + cs_unit_step_t *had = (cs_unit_step_t *)cbm_ht_get(m->steps, key); + if (had) { + return had; + } + char *k = cbm_arena_strdup(&m->arena, key); + cs_unit_step_t *e = (cs_unit_step_t *)cbm_arena_alloc(&m->arena, sizeof(*e)); + if (!k || !e) { + return NULL; + } + memset(e, 0, sizeof(*e)); + cbm_mutex_init(&e->mu); + cbm_mutex_lock(&e->mu); + e->next = m->all; + m->all = e; + cbm_ht_set(m->steps, k, e); + if (cbm_ht_get(m->steps, k) != e) { + e->failed = true; /* an insert that did not take: kept for release only */ + cbm_mutex_unlock(&e->mu); + return NULL; + } + *mine = true; + return e; +} + +/* What the unit's directives give the query after the region's (`fd`). */ +static void unit_usings(const cs_ctx_t *c, const cs_unit_t *unit, const cs_query_t *q, + cs_found_t *fd) { + cs_unit_memo_t *m = c->ix->unit_memo; + char key[CS_MEMO_KEY]; + cs_unit_step_t *e = NULL; + bool mine = false; + if (m && unit_key(key, sizeof(key), c, q)) { + cs_work(SKIP_ONE); + cbm_mutex_lock(&m->mu); + e = m->off ? NULL : unit_step_at(m, key, &mine); + m->off = m->off || !e; + cbm_mutex_unlock(&m->mu); + } + if (e && mine) { + cs_active_rec_t rec = {0}; + cs_found_t alone = {0}; + level_usings(c, unit->usings, unit->nusings, &unit->using_index, q, &alone, &rec); + e->failed = rec.failed; + e->ids = rec.failed ? NULL : rec.ids; + e->n = rec.failed ? 0 : rec.n; + if (rec.failed) { + cbm_free(CBM_MEM_CLASS_OTHER, rec.ids); + } + cbm_mutex_unlock(&e->mu); + } else if (e) { + cbm_mutex_lock(&e->mu); /* the list is there once its finder lets go */ + cbm_mutex_unlock(&e->mu); + } + if (!e || e->failed) { + level_usings(c, unit->usings, unit->nusings, &unit->using_index, q, fd, NULL); + return; + } + bool a_type_name = cbm_ht_get(c->ix->type_names, q->name) != NULL; + for (int i = 0; i < e->n && !found_decided(fd); i++) { + level_using(c, &unit->usings[e->ids[i]], q, a_type_name, fd); + } +} + /* What the directives of this scope level give the query: the region's own * (`us`, when asked) and, at the global namespace, the unit's. Asked once per * key when the context has a memo. */ @@ -4727,9 +4909,9 @@ static void using_step(const cs_ctx_t *c, int region, const cs_using_t *us, int if (keyed && memo_get(c->memo, key, step)) { return; } - level_usings(c, us, nus, usi, q, step); + level_usings(c, us, nus, usi, q, step, NULL); if (unit && !found_decided(step)) { - level_usings(c, unit->usings, unit->nusings, &unit->using_index, q, step); + unit_usings(c, unit, q, step); } if (keyed) { memo_put(c->memo, key, step); @@ -5826,6 +6008,7 @@ static void cs_destroy(void *index) { (unsigned long long)(ix->stats ? atomic_load(&ix->stats->rejected) : 0)); cbm_log_warn("doc_links.cs.rejected_scopes", "files", files, "references", refs); } + unit_memo_free(ix->unit_memo); cbm_free(CBM_MEM_CLASS_OTHER, ix->undo.flags); cbm_free(CBM_MEM_CLASS_OTHER, ix->undo.marks); cbm_free(CBM_MEM_CLASS_OTHER, ix->stats); @@ -6096,6 +6279,13 @@ static int cs_scope_accepted(const char *scope) { bool cbm_doclink_cs_test_scope_parses(const char *scope) { return cs_scope_accepted(scope) == SKIP_ONE; } + +/* Test seam: set, the indexes are built without the unit memo. */ +static atomic_bool cs_test_no_unit_memo; + +void cbm_doclink_cs_test_unit_memo(bool on) { + atomic_store(&cs_test_no_unit_memo, !on); +} #endif /* Every file of the index with what its scope declares. A scope read back @@ -6244,6 +6434,14 @@ static void *cs_build(const cbm_doclink_build_in_t *in) { if (!resolve_bases(ix) || !spread_open(ix) || ix->oom) { return build_failed(ix, "bases", "alloc", NULL); } + /* only now: what the build's own lookups saw of the units' directives + * was not yet the whole index (NULL when memory ran out: no memo) */ +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (!atomic_load(&cs_test_no_unit_memo)) +#endif + { + ix->unit_memo = unit_memo_new(); + } log_index(ix, &msb); return ix; } diff --git a/tests/test_doc_mentions.c b/tests/test_doc_mentions.c index 43f253e538..b2cfaaa2ef 100644 --- a/tests/test_doc_mentions.c +++ b/tests/test_doc_mentions.c @@ -8794,6 +8794,161 @@ TEST(doc_mentions_cs_lookup_repeated_name_work) { PASS(); } +/* R1: `Shared` is declared in k namespaces nobody imports; the project has k + * global usings of other namespaces, and f files each name `Shared` once, so + * each of them would walk the k directives of the unit. The resolver's work + * and the number of `missing` rows for `Shared` (f when all are resolved; -1 + * when the repository cannot be indexed). */ +static uint64_t dm_unit_usings_work(int k, int f, int *rows) { + enum { CAP = 256 * 1024 }; + char *src = malloc(CAP); + char tmp[256] = "/tmp/cbm_dm_unit_usings_XXXXXX"; + *rows = -1; + if (!src || !cbm_mkdtemp(tmp)) { + free(src); + return 0; + } + th_write_file(TH_PATH(tmp, "src/App.csproj"), DM_EMPTY_PROJECT); + size_t w = 0; + for (int i = 0; i < k; i++) { + w += (size_t)snprintf(src + w, CAP - w, + "namespace P%d { public class Shared { } }\n" + "namespace In.U%d { public class Other%d { } }\n", + i, i, i); + } + th_write_file(TH_PATH(tmp, "src/Far.cs"), src); + w = 0; + for (int i = 0; i < k; i++) { + w += (size_t)snprintf(src + w, CAP - w, "global using In.U%d;\n", i); + } + th_write_file(TH_PATH(tmp, "src/GlobalUsings.cs"), src); + for (int j = 0; j < f; j++) { + char path[512]; + snprintf(path, sizeof(path), "%s/src/Uses%d.cs", tmp, j); + snprintf(src, CAP, + "namespace N0\n{\n" + " /// \n" + " public class L%d { }\n" + "}\n", + j); + th_write_file(path, src); + } + free(src); + char db[512]; + snprintf(db, sizeof(db), "%s/unit_usings.db", tmp); + cbm_doclink_cs_test_work_reset(); + uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_work() : 0; + *rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved " + "WHERE raw = 'Shared' AND reason = 'missing'"); + dm_unlink_db(db); + th_rmtree(tmp); + return work; +} + +/* R1: what the unit's directives give a query is the same for every file of + * the unit that asks it: with k and f both twice as large the work about + * doubles (the input does), not f x k (4 x). */ +TEST(doc_mentions_cs_unit_usings_across_files_work) { + int rows[2] = {0}; + uint64_t small = dm_unit_usings_work(150, 100, &rows[0]); + uint64_t large = dm_unit_usings_work(300, 200, &rows[1]); + double ratio = small ? (double)large / (double)small : 0.0; + printf(" unit usings, k 150 -> 300 and f 100 -> 200: %llu -> %llu steps, %.2f times the " + "work, want at most 2.6\n", + (unsigned long long)small, (unsigned long long)large, ratio); + ASSERT_EQ(rows[0], 100); + ASSERT_EQ(rows[1], 200); + ASSERT_GT(small, 0); + ASSERT_LTE(large * 5, small * 13); + PASS(); +} + +/* R1: a file asks only the unit directives that give its query anything, and + * stops where the walk of all of them would: the edges and rows are those of + * a run without the unit memo. The unit imports A (X, Y), B (X), D (Z) and the + * test-only T (W). Cases: nothing from the file and two unit candidates (U1); + * one from the file, another from the unit (U2); the same one from both (U3); + * the same one, then another (U4); one from the unit only (U5); test code + * only (U6 product, UT test code); a unit without global usings (UO). */ +TEST(doc_mentions_cs_unit_usings_unchanged) { + static const dm_source_t files[] = { + {"src/App.csproj", DM_EMPTY_PROJECT}, + {"src/A.cs", "namespace A\n{\n public class X { }\n public class Y { }\n}\n"}, + {"src/B.cs", "namespace B\n{\n public class X { }\n}\n"}, + {"src/C.cs", "namespace C\n{\n public class X { }\n}\n"}, + {"src/D.cs", "namespace D\n{\n public class Z { }\n}\n"}, + {"src/tests/T.cs", "namespace T\n{\n public class W { }\n}\n"}, + {"src/GlobalUsings.cs", "global using A;\nglobal using B;\nglobal using D;\n" + "global using T;\n"}, + {"src/U1.cs", "namespace N\n{\n /// \n" + " public class U1 { }\n}\n"}, + {"src/U2.cs", "using C;\nnamespace N\n{\n /// \n" + " public class U2 { }\n}\n"}, + {"src/U3.cs", "using A;\nnamespace N\n{\n /// \n" + " public class U3 { }\n}\n"}, + {"src/U4.cs", "using A;\nnamespace N\n{\n /// \n" + " public class U4 { }\n}\n"}, + {"src/U5.cs", "namespace N\n{\n /// \n" + " public class U5 { }\n}\n"}, + {"src/U6.cs", "namespace N\n{\n /// \n" + " public class U6 { }\n}\n"}, + {"src/tests/UT.cs", "namespace N\n{\n /// " + "\n public class UT { }\n}\n"}, + {"other/Other.csproj", DM_EMPTY_PROJECT}, + {"other/UO.cs", "namespace N\n{\n /// " + "\n public class UO { }\n}\n"}, + }; + static const dm_want_t wants[] = { + {"src/U1.cs", "X", "U1.U1", NULL, "ambiguous", NULL, NULL}, + {"src/U2.cs", "X", "U2.U2", NULL, "ambiguous", NULL, NULL}, + {"src/U3.cs", "Y", "U3.U3", "A.Y", NULL, NULL, NULL}, + {"src/U4.cs", "X", "U4.U4", NULL, "ambiguous", NULL, NULL}, + {"src/U5.cs", "Z", "U5.U5", "D.Z", NULL, NULL, NULL}, + {"src/U6.cs", "W", "U6.U6", NULL, "test_only_target", NULL, NULL}, + {"src/tests/UT.cs", "W", "UT.UT", "T.W", NULL, NULL, NULL}, + {"src/tests/UT.cs", "X", "UT.UT", NULL, "ambiguous", NULL, NULL}, + {"other/UO.cs", "X", "UO.UO", NULL, "missing", NULL, NULL}, + {"other/UO.cs", "Z", "UO.UO", NULL, "missing", NULL, NULL}, + }; + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_unit_same_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + for (int i = 0; i < DM_COUNT(files); i++) { + th_write_file(TH_PATH(repo, files[i].path), files[i].text); + } + char with_db[512]; + char without_db[512]; + snprintf(with_db, sizeof(with_db), "%s/with.db", tmp); + snprintf(without_db, sizeof(without_db), "%s/without.db", tmp); + int with_rc = dm_index(repo, with_db, NULL); + cbm_doclink_cs_test_unit_memo(false); + int without_rc = dm_index(repo, without_db, NULL); + cbm_doclink_cs_test_unit_memo(true); + char *with = dm_doclink_state(with_db); + char *without = dm_doclink_state(without_db); + bool same = with && without && strcmp(with, without) == 0; + if (!same) { + printf(" with the unit memo\n%s without it\n%s", with ? with : "(null)", + without ? without : "(null)"); + } + int bad = 0; + for (int i = 0; i < DM_COUNT(wants); i++) { + bad += dm_want_failed(with_db, &wants[i]); + } + free(with); + free(without); + dm_unlink_db(with_db); + dm_unlink_db(without_db); + th_rmtree(tmp); + ASSERT_EQ(with_rc, 0); + ASSERT_EQ(without_rc, 0); + ASSERT_TRUE(same); + ASSERT_EQ(bad, 0); + PASS(); +} + /* S5: k namespaces all declare `Shared` and one file imports all of them; * `refs` documented classes name it (0 or 1). The resolver's lookup work and * the reason of the row (empty when there is none). */ @@ -10403,6 +10558,8 @@ SUITE(doc_mentions) { RUN_TEST(doc_mentions_cs_lookup_cost); RUN_TEST(doc_mentions_cs_import_lookup_cost); RUN_TEST(doc_mentions_cs_lookup_repeated_name_work); + RUN_TEST(doc_mentions_cs_unit_usings_across_files_work); + RUN_TEST(doc_mentions_cs_unit_usings_unchanged); RUN_TEST(doc_mentions_cs_lookup_ambiguous_stops); RUN_TEST(doc_mentions_cs_import_stage_aliases); RUN_TEST(doc_mentions_cs_import_static_parts); From 1a9dbac843f74fd62ef61db6204373652a110fad Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Mon, 5 Oct 2026 03:37:29 +0200 Subject: [PATCH 17/19] doc-links(cs): a unit step no directive can give anything asks no memo (R1) The unit memo of the previous commit was asked -- the run's lock taken, an entry made -- for every unit step, also when the unit has no directive or its index rules the name out, where the walk returned at once before. That is lock traffic and memory for most projects, which have no global usings. The test that walk starts with is now one function (usings_may_give), asked before the memo too. The work counts of the doc_mentions cost tests are those from before the memo again, except where unit directives can give the name: one memo probe per lookup there. Test: cs_unit_usings_across_files_work and cs_unit_usings_unchanged as before (768 -> 1520 steps; without the memo 17003 -> 64403). Signed-off-by: Martin Vogel --- src/pipeline/doc_links_cs.c | 25 +++++++++++++++++++------ 1 file changed, 19 insertions(+), 6 deletions(-) diff --git a/src/pipeline/doc_links_cs.c b/src/pipeline/doc_links_cs.c index 223e998116..2c459a88b6 100644 --- a/src/pipeline/doc_links_cs.c +++ b/src/pipeline/doc_links_cs.c @@ -4645,17 +4645,27 @@ static void using_visit(const cs_ctx_t *c, const cs_using_t *us, int id, const c level_using(c, &us[id], q, a_type_name, fd); } +/* False when none of a step's n directives can give the query anything: + * there are none, or the index shows no static candidates and no namespace + * that can contribute this name. *a_type_name: the name is some type's. */ +static bool usings_may_give(const cs_ctx_t *c, int n, const cs_using_index_t *index, + const cs_query_t *q, bool *a_type_name) { + if (!n) { + return false; + } + *a_type_name = cbm_ht_get(c->ix->type_names, q->name) != NULL; + bool indexed = index && index->ready && c->ix->name_scopes_ready; + return !(indexed && !index->nentities && (!index->nnamespaces || !*a_type_name)); +} + static void level_usings(const cs_ctx_t *c, const cs_using_t *us, int n, const cs_using_index_t *index, const cs_query_t *q, cs_found_t *fd, cs_active_rec_t *rec) { - if (!n) { + bool a_type_name = false; + if (!usings_may_give(c, n, index, q, &a_type_name)) { return; } - bool a_type_name = cbm_ht_get(c->ix->type_names, q->name) != NULL; bool indexed = index && index->ready && c->ix->name_scopes_ready; - if (indexed && !index->nentities && (!index->nnamespaces || !a_type_name)) { - return; /* no static candidates, and no namespace can contribute this name */ - } size_t hi = 0; size_t lo = indexed ? name_scope_range(c->ix, q->name, &hi) : 0; /* A common name in a large repository must not make a small scope walk @@ -4861,6 +4871,10 @@ static cs_unit_step_t *unit_step_at(cs_unit_memo_t *m, const char *key, bool *mi /* What the unit's directives give the query after the region's (`fd`). */ static void unit_usings(const cs_ctx_t *c, const cs_unit_t *unit, const cs_query_t *q, cs_found_t *fd) { + bool a_type_name = false; + if (!usings_may_give(c, unit->nusings, &unit->using_index, q, &a_type_name)) { + return; /* nothing to ask, nothing to remember */ + } cs_unit_memo_t *m = c->ix->unit_memo; char key[CS_MEMO_KEY]; cs_unit_step_t *e = NULL; @@ -4891,7 +4905,6 @@ static void unit_usings(const cs_ctx_t *c, const cs_unit_t *unit, const cs_query level_usings(c, unit->usings, unit->nusings, &unit->using_index, q, fd, NULL); return; } - bool a_type_name = cbm_ht_get(c->ix->type_names, q->name) != NULL; for (int i = 0; i < e->n && !found_decided(fd); i++) { level_using(c, &unit->usings[e->ids[i]], q, a_type_name, fd); } From 5283fe03427f6fed4aa309aaa0362d22048775d5 Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Mon, 5 Oct 2026 05:17:11 +0200 Subject: [PATCH 18/19] doc-links(cs): what the Linux leg showed: path-ordered binding, a test thread's caches, split suites The first run of the doc-link suites on the Linux arm64 leg (gcc, ASan with LeakSanitizer) found three things macOS does not show: - The node pass bound files in the order the file system listed them, and the scratch table's cost depends on that order (the table is emptied per file and made anew only after a far larger file). doc_mentions_cs_scratch_table_work counted 102,912 steps on Linux against 26,048 on macOS for the same files. Files are now bound in path order: a file's bindings do not depend on the others, so only the cost changes, and it is the same on every system. The test itself had a second dependence: its 20,000-field file exceeded the extraction budget on that leg (5 s, defs=0), so the test ran without it; 4,000 fields were still cut there beside five other suites. It now uses 1,000 fields and 20 small files (3,008 steps against a bound of 4,160; 40,512 with the S13 fix reverted) and requires the large file to reach the index whole. - A test thread that extracted C# on its own small stack ended without freeing its parser and node-kind caches, as worker threads do: 40 KB reported leaked at exit. It frees them now. - The suite ran 1,014 s against the harness's 900 s per-suite wall clock (gcc on aarch64 makes ASan's stack-use-after-return check about 100x slower per call). Its 119 tests now run as doc_mentions (72), doc_mentions_cost (9, the work-counter tests) and doc_mentions_msbuild (38): about 230 s, 210 s and 480 s there. Signed-off-by: Martin Vogel --- src/pipeline/doc_links_cs.c | 33 ++++++++- tests/test_doc_mentions.c | 140 +++++++++++++++++++++--------------- tests/test_main.c | 4 ++ 3 files changed, 118 insertions(+), 59 deletions(-) diff --git a/src/pipeline/doc_links_cs.c b/src/pipeline/doc_links_cs.c index 2c459a88b6..db5d8aeb70 100644 --- a/src/pipeline/doc_links_cs.c +++ b/src/pipeline/doc_links_cs.c @@ -6373,14 +6373,41 @@ static const char *build_files(cs_index_t *ix, const cbm_doclink_build_in_t *in) return bad; } -/* The graph node of every declaration. false when memory ran out. */ +typedef struct { + const char *path; + int idx; +} cs_file_order_t; + +static int file_order_cmp(const void *a, const void *b) { + const cs_file_order_t *x = (const cs_file_order_t *)a; + const cs_file_order_t *y = (const cs_file_order_t *)b; + int c = strcmp(x->path ? x->path : "", y->path ? y->path : ""); + return c ? c : (x->idx > y->idx) - (x->idx < y->idx); +} + +/* The graph node of every declaration. false when memory ran out. Files are + * bound in path order: a file's bindings do not depend on the others, but the + * scratch table's cost does on the order (it is emptied per file), and the + * order the file system listed the files in differs between systems. */ static bool build_nodes(cs_index_t *ix, const cbm_gbuf_t *g) { cs_node_pass_t np = {.names = cbm_ht_create(CBM_SZ_256), .sized = CBM_SZ_256}; cbm_arena_init(&np.keys); - bool ok = np.names != NULL; + cs_file_order_t *order = (cs_file_order_t *)cbm_alloc( + CBM_MEM_CLASS_OTHER, (size_t)(ix->nfiles ? ix->nfiles : 1) * sizeof(*order)); + bool ok = np.names != NULL && order != NULL; + if (!order) { + ix->oom = true; + } + for (int i = 0; ok && i < ix->nfiles; i++) { + order[i] = (cs_file_order_t){ix->files[i].rel_path, i}; + } + if (ok && ix->nfiles > 1) { + qsort(order, (size_t)ix->nfiles, sizeof(*order), file_order_cmp); + } for (int i = 0; ok && i < ix->nfiles; i++) { - ok = bind_nodes(ix, &ix->files[i], g, &np); + ok = bind_nodes(ix, &ix->files[order[i].idx], g, &np); } + cbm_free(CBM_MEM_CLASS_OTHER, order); cbm_ht_free(np.names); cbm_arena_destroy(&np.keys); cbm_free(CBM_MEM_CLASS_OTHER, np.last); diff --git a/tests/test_doc_mentions.c b/tests/test_doc_mentions.c index b2cfaaa2ef..9f4697a96e 100644 --- a/tests/test_doc_mentions.c +++ b/tests/test_doc_mentions.c @@ -13,6 +13,7 @@ #include "test_doc_mentions_helpers.h" #include "cbm.h" +#include "helpers.h" /* cbm_kind_in_set_free_cache */ #include "doclink.h" #include "lang_specs.h" #include "foundation/compat_thread.h" @@ -5980,10 +5981,14 @@ typedef struct { CBMFileResult *result; } dm_extract_job_t; -/* cbm_thread_create body: extract job->src as C#. */ +/* cbm_thread_create body: extract job->src as C#, then free this thread's + * parser and caches as a worker thread does when it ends (LeakSanitizer + * reports them otherwise). */ static void *dm_extract_thread(void *arg) { dm_extract_job_t *job = (dm_extract_job_t *)arg; job->result = dm_extract(job->src, CBM_LANG_CSHARP, job->rel_path); + cbm_destroy_thread_parser(); + cbm_kind_in_set_free_cache(); return NULL; } @@ -9892,11 +9897,18 @@ TEST(doc_mentions_msbuild_malformed_props_opens) { PASS(); } -enum { DM_BIG_MEMBERS = 20000, DM_SMALL_FILES = 200 }; +/* Sized so that the large file parses far within the extraction budget on + * the slowest sanitizer leg, under load too (20,000 fields were cut there, + * and 4,000 beside five other suites, leaving the test without its large + * file). Every small file costs at least the scratch table's minimum size: + * these 20 and the large file come to 3,008 steps, within the bound, while + * the regression, DM_SMALL_FILES x 2 x DM_BIG_MEMBERS, is ten times it. */ +enum { DM_BIG_MEMBERS = 1000, DM_SMALL_FILES = 20 }; /* What the resolver looks at to build its index over one file of * DM_BIG_MEMBERS fields followed (in path order) by DM_SMALL_FILES files of - * one type and one field each. 0 when the repository cannot be indexed. */ + * one type and one field each. 0 when the repository cannot be indexed or the + * large file did not reach the index whole. */ static uint64_t dm_scratch_work(void) { char tmp[256]; snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_scratch_XXXXXX"); @@ -9930,16 +9942,23 @@ static uint64_t dm_scratch_work(void) { cbm_doclink_cs_test_work_reset(); uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_scratch_work() : 0; int rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"); + int fields = dm_count(db, "SELECT COUNT(*) FROM nodes WHERE label = 'Field' AND " + "file_path = 'a/Big.cs'"); + if (fields != DM_BIG_MEMBERS) { + printf(" scratch table: a/Big.cs reached the index with %d of %d fields\n", fields, + (int)DM_BIG_MEMBERS); + } dm_unlink_db(db); th_rmtree(tmp); - return rows == 1 ? work : 0; + return rows == 1 && fields == DM_BIG_MEMBERS ? work : 0; } /* The per-file scratch table of the index build is not emptied at the size * one large file grew it to: after a file of DM_BIG_MEMBERS names, every one - * of DM_SMALL_FILES small files costs about its own size. The work stays - * within a small multiple of all the names the files declare (emptying the - * large table for every small file cost DM_SMALL_FILES times its size). */ + * of DM_SMALL_FILES small files costs about the table's minimum size. The + * work stays within a small multiple of all the names the files declare + * (emptying the large table for every small file cost DM_SMALL_FILES times + * its size). */ TEST(doc_mentions_cs_scratch_table_work) { uint64_t names = (uint64_t)DM_BIG_MEMBERS + (uint64_t)DM_SMALL_FILES * 2; uint64_t work = dm_scratch_work(); @@ -10473,27 +10492,6 @@ SUITE(doc_mentions) { RUN_TEST(doc_mentions_cs_namespace_publication); RUN_TEST(doc_mentions_cs_scope_parse_errors); RUN_TEST(doc_mentions_cs_norm_type); - RUN_TEST(doc_mentions_msbuild_usings); - RUN_TEST(doc_mentions_msbuild_targets_isolation); - RUN_TEST(doc_mentions_msbuild_targets_failure); - RUN_TEST(doc_mentions_msbuild_items_isolation); - RUN_TEST(doc_mentions_msbuild_items_failure); - RUN_TEST(doc_mentions_msbuild_items_lifetime); - RUN_TEST(doc_mentions_msbuild_items_work); - RUN_TEST(doc_mentions_msbuild_targets_work); - RUN_TEST(doc_mentions_msbuild_nearest_work); - RUN_TEST(doc_mentions_msbuild_shared_closure_work); - RUN_TEST(doc_mentions_msbuild_shared_file_work); - RUN_TEST(doc_mentions_msbuild_shared_closure_isolation); - RUN_TEST(doc_mentions_msbuild_components_failure); - RUN_TEST(doc_mentions_msbuild_components_revision); - RUN_TEST(doc_mentions_msbuild_components_mixed_absence); - RUN_TEST(doc_mentions_msbuild_prefix_work); - RUN_TEST(doc_mentions_msbuild_prefix_isolation); - RUN_TEST(doc_mentions_msbuild_prefix_failure); - RUN_TEST(doc_mentions_msbuild_eval_storage); - RUN_TEST(doc_mentions_msbuild_value_lifetimes); - RUN_TEST(doc_mentions_msbuild_value_allocation); RUN_TEST(doc_mentions_resolver_rules); RUN_TEST(doc_mentions_resolver_arity_and_members); RUN_TEST(doc_mentions_resolver_parse_errors); @@ -10508,34 +10506,14 @@ SUITE(doc_mentions) { RUN_TEST(doc_mentions_incremental_equals_full); RUN_TEST(doc_mentions_parallel_equals_sequential); RUN_TEST(doc_mentions_cs_scan_dollar_run); - RUN_TEST(doc_mentions_cs_scan_branch_memory); - RUN_TEST(doc_mentions_cs_scan_branch_merge_work); - RUN_TEST(doc_mentions_cs_scan_declarator_modifier_work); RUN_TEST(doc_mentions_cs_shared_doc_sources); RUN_TEST(doc_mentions_cs_shared_doc_once); - RUN_TEST(doc_mentions_cs_shared_doc_work); RUN_TEST(doc_mentions_cs_shared_doc_replay); RUN_TEST(doc_mentions_cs_shared_doc_allocation_failure); RUN_TEST(doc_mentions_cs_scan_header_reads); RUN_TEST(doc_mentions_cs_scan_nested_holes); RUN_TEST(doc_mentions_cs_scan_holes_stack); - RUN_TEST(doc_mentions_msbuild_blob); - RUN_TEST(doc_mentions_msbuild_import_group_blob_growth); - RUN_TEST(doc_mentions_msbuild_import_group_roundtrip); - RUN_TEST(doc_mentions_msbuild_import_group_condition_work); - RUN_TEST(doc_mentions_msbuild_item_group_condition_work); - RUN_TEST(doc_mentions_msbuild_import_group_condition_once); - RUN_TEST(doc_mentions_msbuild_legacy_import_blob); - RUN_TEST(doc_mentions_msbuild_bad_import_group_blob); - RUN_TEST(doc_mentions_msbuild_imports); - RUN_TEST(doc_mentions_msbuild_conditions); - RUN_TEST(doc_mentions_msbuild_unknown_spreads); - RUN_TEST(doc_mentions_msbuild_gate); RUN_TEST(doc_mentions_incremental_project_files); -#ifndef _WIN32 - RUN_TEST(doc_mentions_msbuild_linked_project_file); - RUN_TEST(doc_mentions_msbuild_linked_import_directory); -#endif RUN_TEST(doc_mentions_cs_lookup_order); RUN_TEST(doc_mentions_cs_type_parameters); RUN_TEST(doc_mentions_cs_usings); @@ -10555,10 +10533,6 @@ SUITE(doc_mentions) { RUN_TEST(doc_mentions_cs_doc_ids); RUN_TEST(doc_mentions_cs_reasons); RUN_TEST(doc_mentions_cs_text_forms); - RUN_TEST(doc_mentions_cs_lookup_cost); - RUN_TEST(doc_mentions_cs_import_lookup_cost); - RUN_TEST(doc_mentions_cs_lookup_repeated_name_work); - RUN_TEST(doc_mentions_cs_unit_usings_across_files_work); RUN_TEST(doc_mentions_cs_unit_usings_unchanged); RUN_TEST(doc_mentions_cs_lookup_ambiguous_stops); RUN_TEST(doc_mentions_cs_import_stage_aliases); @@ -10567,16 +10541,12 @@ SUITE(doc_mentions) { RUN_TEST(doc_mentions_cs_no_silent_limits); RUN_TEST(doc_mentions_cs_many_assemblies); RUN_TEST(doc_mentions_incremental_using_of_based_type); - RUN_TEST(doc_mentions_msbuild_many_project_files); RUN_TEST(doc_mentions_cs_scan_nesting_limits); RUN_TEST(doc_mentions_cs_damaged_stored_scope); RUN_TEST(doc_mentions_cs_portable_scope_bound); RUN_TEST(doc_mentions_cs_unbalanced_brackets); RUN_TEST(doc_mentions_cs_stored_scope_nesting); RUN_TEST(doc_mentions_index_status_unknown_reason); - RUN_TEST(doc_mentions_msbuild_unread_project_opens); - RUN_TEST(doc_mentions_msbuild_malformed_props_opens); - RUN_TEST(doc_mentions_cs_scratch_table_work); RUN_TEST(doc_mentions_incremental_non_ascii_name); RUN_TEST(doc_mentions_alloc_failure_status); RUN_TEST(doc_mentions_directives_only_failure); @@ -10584,3 +10554,61 @@ SUITE(doc_mentions) { RUN_TEST(doc_mentions_cs_rejected_scope_contained); RUN_TEST(doc_mentions_cs_rejected_scope_across_runs); } + +/* The cost tests (work counters held to a multiple of the input) and the MSBuild + * evaluator run as suites of their own, so that each stays within the per-suite + * wall clock of the harness on the slowest sanitizer legs. */ +SUITE(doc_mentions_cost) { + RUN_TEST(doc_mentions_cs_scan_branch_memory); + RUN_TEST(doc_mentions_cs_scan_branch_merge_work); + RUN_TEST(doc_mentions_cs_scan_declarator_modifier_work); + RUN_TEST(doc_mentions_cs_shared_doc_work); + RUN_TEST(doc_mentions_cs_lookup_cost); + RUN_TEST(doc_mentions_cs_import_lookup_cost); + RUN_TEST(doc_mentions_cs_lookup_repeated_name_work); + RUN_TEST(doc_mentions_cs_unit_usings_across_files_work); + RUN_TEST(doc_mentions_cs_scratch_table_work); +} + +SUITE(doc_mentions_msbuild) { + RUN_TEST(doc_mentions_msbuild_usings); + RUN_TEST(doc_mentions_msbuild_targets_isolation); + RUN_TEST(doc_mentions_msbuild_targets_failure); + RUN_TEST(doc_mentions_msbuild_items_isolation); + RUN_TEST(doc_mentions_msbuild_items_failure); + RUN_TEST(doc_mentions_msbuild_items_lifetime); + RUN_TEST(doc_mentions_msbuild_items_work); + RUN_TEST(doc_mentions_msbuild_targets_work); + RUN_TEST(doc_mentions_msbuild_nearest_work); + RUN_TEST(doc_mentions_msbuild_shared_closure_work); + RUN_TEST(doc_mentions_msbuild_shared_file_work); + RUN_TEST(doc_mentions_msbuild_shared_closure_isolation); + RUN_TEST(doc_mentions_msbuild_components_failure); + RUN_TEST(doc_mentions_msbuild_components_revision); + RUN_TEST(doc_mentions_msbuild_components_mixed_absence); + RUN_TEST(doc_mentions_msbuild_prefix_work); + RUN_TEST(doc_mentions_msbuild_prefix_isolation); + RUN_TEST(doc_mentions_msbuild_prefix_failure); + RUN_TEST(doc_mentions_msbuild_eval_storage); + RUN_TEST(doc_mentions_msbuild_value_lifetimes); + RUN_TEST(doc_mentions_msbuild_value_allocation); + RUN_TEST(doc_mentions_msbuild_blob); + RUN_TEST(doc_mentions_msbuild_import_group_blob_growth); + RUN_TEST(doc_mentions_msbuild_import_group_roundtrip); + RUN_TEST(doc_mentions_msbuild_import_group_condition_work); + RUN_TEST(doc_mentions_msbuild_item_group_condition_work); + RUN_TEST(doc_mentions_msbuild_import_group_condition_once); + RUN_TEST(doc_mentions_msbuild_legacy_import_blob); + RUN_TEST(doc_mentions_msbuild_bad_import_group_blob); + RUN_TEST(doc_mentions_msbuild_imports); + RUN_TEST(doc_mentions_msbuild_conditions); + RUN_TEST(doc_mentions_msbuild_unknown_spreads); + RUN_TEST(doc_mentions_msbuild_gate); +#ifndef _WIN32 + RUN_TEST(doc_mentions_msbuild_linked_project_file); + RUN_TEST(doc_mentions_msbuild_linked_import_directory); +#endif + RUN_TEST(doc_mentions_msbuild_many_project_files); + RUN_TEST(doc_mentions_msbuild_unread_project_opens); + RUN_TEST(doc_mentions_msbuild_malformed_props_opens); +} diff --git a/tests/test_main.c b/tests/test_main.c index 98e0cdc47e..8bd1a81d5a 100644 --- a/tests/test_main.c +++ b/tests/test_main.c @@ -970,6 +970,8 @@ extern void suite_traces(void); extern void suite_configlink(void); extern void suite_doclinks(void); extern void suite_doc_mentions(void); +extern void suite_doc_mentions_cost(void); +extern void suite_doc_mentions_msbuild(void); extern void suite_infrascan(void); extern void suite_cli(void); extern void suite_agent_clients(void); @@ -1339,6 +1341,8 @@ int main(int argc, char **argv) { /* Doc-comment references -> MENTIONS */ RUN_SELECTED_SUITE(doc_mentions); + RUN_SELECTED_SUITE(doc_mentions_cost); + RUN_SELECTED_SUITE(doc_mentions_msbuild); /* Infrastructure scanning */ RUN_SELECTED_SUITE(infrascan); From d2ea04a84a2f5f40327e0b5269aaa05cdbf71a64 Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Mon, 5 Oct 2026 05:44:48 +0200 Subject: [PATCH 19/19] doc-links(core): index_status failure paths leave through its one exit The doc_links report added two early returns to handle_index_status, each with its own free of the project name: raw allocator sites the memory-core ratchet does not allow (mcp.c went from 778 to 779 on this base). A failed allocation or report now sets a flag and leaves through the existing exit, which frees the name and answers with the MCP allocation-error response as before. The baseline of mcp.c is lowered to the 777 the ratchet now measures. Tests: doc_mentions (incl. the doc-link report fault injection), mcp pass; lint-memory-core passes. Signed-off-by: Martin Vogel --- scripts/memory-core-baseline.txt | 2 +- src/mcp/mcp.c | 23 +++++++++++------------ 2 files changed, 12 insertions(+), 13 deletions(-) diff --git a/scripts/memory-core-baseline.txt b/scripts/memory-core-baseline.txt index 1453388034..a44ce0d2b8 100644 --- a/scripts/memory-core-baseline.txt +++ b/scripts/memory-core-baseline.txt @@ -52,7 +52,7 @@ src/git/git_context.c 18 src/main.c 32 src/mcp/compact_out.c 39 src/mcp/index_supervisor.c 8 -src/mcp/mcp.c 778 +src/mcp/mcp.c 777 src/pipeline/artifact.c 38 src/pipeline/fqn.c 23 src/pipeline/lsp_resolve.h 4 diff --git a/src/mcp/mcp.c b/src/mcp/mcp.c index 892c56ece1..c799282f1e 100644 --- a/src/mcp/mcp.c +++ b/src/mcp/mcp.c @@ -7204,14 +7204,16 @@ static char *handle_index_status(cbm_mcp_server_t *srv, const char *args) { yyjson_mut_doc *doc = yyjson_mut_doc_new(NULL); yyjson_mut_val *root = doc ? yyjson_mut_obj(doc) : NULL; - if (!root) { - yyjson_mut_doc_free(doc); - free(project); - return mcp_result_from_json(args, NULL); + /* A failed allocation or report answers with the MCP allocation-error + * response, through the one exit below. */ + bool failed = !root; + if (root) { + yyjson_mut_doc_set_root(doc, root); } - yyjson_mut_doc_set_root(doc, root); - if (project) { + if (failed) { + /* nothing to report into */ + } else if (project) { int nodes = cbm_store_count_nodes(store, project); int edges = cbm_store_count_edges(store, project); /* A negative count is a failed read (CBM_STORE_ERR), not a small @@ -7244,11 +7246,8 @@ static char *handle_index_status(cbm_mcp_server_t *srv, const char *args) { safe_str_free(&proj_info.indexed_at); safe_str_free(&proj_info.root_path); if (!doc_links_built) { - yyjson_mut_doc_free(doc); - free(project); - return mcp_result_from_json(args, NULL); - } - if (counts_unreadable) { + failed = true; + } else if (counts_unreadable) { const char *hint; if (nodes < 0 && edges < 0) { hint = "The nodes and edges tables could not be read; the database may be " @@ -7271,7 +7270,7 @@ static char *handle_index_status(cbm_mcp_server_t *srv, const char *args) { yyjson_mut_obj_add_str(doc, root, "status", "no_project"); } - char *json = yy_doc_to_str(doc); + char *json = failed ? NULL : yy_doc_to_str(doc); yyjson_mut_doc_free(doc); free(project);