diff --git a/Makefile.cbm b/Makefile.cbm index d76269f66..3c3455914 100644 --- a/Makefile.cbm +++ b/Makefile.cbm @@ -350,6 +350,8 @@ EXTRACTION_SRCS = \ $(CBM_DIR)/extract_channels.c \ $(CBM_DIR)/extract_k8s.c \ $(CBM_DIR)/extract_dbt.c \ + $(CBM_DIR)/doclink.c \ + $(CBM_DIR)/doclink_cs.c \ $(CBM_DIR)/sql_values.c \ $(CBM_DIR)/helpers.c \ $(CBM_DIR)/result_compact.c \ @@ -454,6 +456,9 @@ PIPELINE_SRCS = \ src/pipeline/pass_configures.c \ src/pipeline/pass_configlink.c \ src/pipeline/pass_doclinks.c \ + src/pipeline/doc_links.c \ + src/pipeline/doc_links_cs.c \ + src/pipeline/doc_links_msbuild.c \ src/pipeline/pass_route_nodes.c \ src/pipeline/pass_enrichment.c \ src/pipeline/pass_envscan.c \ @@ -681,7 +686,7 @@ TEST_DISCOVER_SRCS = \ TEST_GRAPH_BUFFER_SRCS = tests/test_graph_buffer.c -TEST_PIPELINE_SRCS = tests/test_registry.c tests/test_pipeline.c tests/test_importance.c tests/test_cross_repo.c tests/test_fqn.c tests/test_route_canon.c tests/test_path_alias.c tests/test_configlink.c tests/test_doclinks.c tests/test_infrascan.c tests/test_worker_pool.c tests/test_parallel.c tests/test_index_resilience.c tests/test_index_format.c tests/test_call_reference_contract.c tests/repro/repro_call_scope_usages.c tests/repro/repro_call_argument_usages.c tests/repro/repro_reference_precision.c tests/repro/repro_lexical_binding_precision.c tests/repro/repro_call_argument_matrix_a.c tests/repro/repro_call_argument_matrix_b.c tests/repro/repro_call_node_behaviors.c tests/repro/repro_language_registry.c tests/repro/repro_call_node_manifest.c tests/repro/repro_lsp_ordered_signatures.c tests/repro/repro_lsp_ordered_local.c tests/repro/repro_ts_overload_return_chains.c tests/repro/repro_harness_cleanup.c tests/repro/repro_runner_filter.c +TEST_PIPELINE_SRCS = tests/test_registry.c tests/test_pipeline.c tests/test_importance.c tests/test_cross_repo.c tests/test_fqn.c tests/test_route_canon.c tests/test_path_alias.c tests/test_configlink.c tests/test_doclinks.c tests/test_doc_mentions.c tests/test_infrascan.c tests/test_worker_pool.c tests/test_parallel.c tests/test_index_resilience.c tests/test_index_format.c tests/test_call_reference_contract.c tests/repro/repro_call_scope_usages.c tests/repro/repro_call_argument_usages.c tests/repro/repro_reference_precision.c tests/repro/repro_lexical_binding_precision.c tests/repro/repro_call_argument_matrix_a.c tests/repro/repro_call_argument_matrix_b.c tests/repro/repro_call_node_behaviors.c tests/repro/repro_language_registry.c tests/repro/repro_call_node_manifest.c tests/repro/repro_lsp_ordered_signatures.c tests/repro/repro_lsp_ordered_local.c tests/repro/repro_ts_overload_return_chains.c tests/repro/repro_harness_cleanup.c tests/repro/repro_runner_filter.c TEST_WATCHER_SRCS = tests/test_watcher.c diff --git a/README.md b/README.md index f644775a7..0de469ac4 100644 --- a/README.md +++ b/README.md @@ -740,7 +740,9 @@ mode. `index_status` keeps describing the published graph and its freshness. ### Edge Types -`CONTAINS_PACKAGE`, `CONTAINS_FOLDER`, `CONTAINS_FILE`, `DEFINES`, `DEFINES_METHOD`, `IMPORTS`, `CALLS`, `CALL_REFERENCE`, `HTTP_CALLS`, `ASYNC_CALLS`, `IMPLEMENTS`, `HANDLES`, `USAGE`, `CONFIGURES`, `REFERENCES_FILE`, `WRITES`, `MEMBER_OF`, `TESTS`, `USES_TYPE`, `FILE_CHANGES_WITH` +`CONTAINS_PACKAGE`, `CONTAINS_FOLDER`, `CONTAINS_FILE`, `DEFINES`, `DEFINES_METHOD`, `IMPORTS`, `CALLS`, `CALL_REFERENCE`, `HTTP_CALLS`, `ASYNC_CALLS`, `IMPLEMENTS`, `HANDLES`, `USAGE`, `CONFIGURES`, `REFERENCES_FILE`, `MENTIONS`, `WRITES`, `MEMBER_OF`, `TESTS`, `USES_TYPE`, `FILE_CHANGES_WITH` + +`MENTIONS` links a documented definition to the code its doc comment references (C# ``, ``, ``, ``), one edge per pair with `via`, `syntax`, `tier` (`exact` or `unique`), first `line` and `count`. A reference is bound only when it names exactly one definition under the language's own lookup rules; the rest are counted by reason in `index_status` under `doc_links` (samples with `diagnostics='full'`). ### Qualified Names diff --git a/docs/EVALUATION_PLAN.md b/docs/EVALUATION_PLAN.md index 5f8bb532c..efd56207d 100644 --- a/docs/EVALUATION_PLAN.md +++ b/docs/EVALUATION_PLAN.md @@ -311,15 +311,16 @@ SHA, and — **not just totals** — a **per-type breakdown**: - `node-types.json` — a histogram of **node count by label** (`Function`, `Method`, `Class`, `Interface`, `Type`/`Enum`/`Struct`, `Field`, `Variable`, `Route`, `Module`, `Section`, `Macro`, `File`, `Folder`) + total. -- `edge-types.json` — a histogram of **edge count by type, with one entry for every one of the 32 +- `edge-types.json` — a histogram of **edge count by type, with one entry for every one of the 33 edge types, including those that came back `0`**. The canonical set is the indexer's own - `ALL_EDGE_TYPES[]` (26 intra-repo types — `tests/test_lang_contract.c`): `CALLS`, `ASYNC_CALLS`, + `ALL_EDGE_TYPES[]` (27 intra-repo types — `tests/test_lang_contract.c`): `CALLS`, `ASYNC_CALLS`, `HTTP_CALLS`, `GRPC_CALLS`, `GRAPHQL_CALLS`, `TRPC_CALLS`, `DEFINES`, `DEFINES_METHOD`, `IMPLEMENTS`, `INHERITS`, `OVERRIDE`, `DECORATES`, `IMPORTS`, `HANDLES`, `CONFIGURES`, `DEPENDS_ON`, `USAGE`, `DATA_FLOWS`, `SEMANTICALLY_RELATED`, `SIMILAR_TO`, `TESTS`, `TESTS_FILE`, `INFRA_MAPS`, - `FILE_CHANGES_WITH`, `CONTAINS_FILE`, `CONTAINS_FOLDER` — **plus the 6 cross-repo types** from the + `FILE_CHANGES_WITH`, `CONTAINS_FILE`, `CONTAINS_FOLDER`, `MENTIONS` (doc comment -> referenced + code) — **plus the 6 cross-repo types** from the cross-repo pass (`CROSS_HTTP_CALLS`, `CROSS_ASYNC_CALLS`, `CROSS_GRPC_CALLS`, `CROSS_GRAPHQL_CALLS`, - `CROSS_TRPC_CALLS`, `CROSS_CHANNEL`) and a total. The writer **emits the full 32-type list and + `CROSS_TRPC_CALLS`, `CROSS_CHANNEL`) and a total. The writer **emits the full 33-type list and back-fills missing types with `0`** rather than recording only the types that appeared. These come straight from `query_graph` (`MATCH (n) RETURN labels(n), count(*)` and diff --git a/internal/cbm/cbm.c b/internal/cbm/cbm.c index a39a93ae4..b351e5f2e 100644 --- a/internal/cbm/cbm.c +++ b/internal/cbm/cbm.c @@ -6,7 +6,8 @@ #include "foundation/mem_events.h" // waste sanitizer: the bound allocators bypass every observer #include "foundation/log.h" // cbm_log_warn -- extract.lsp.skipped #include "cbm.h" -#include "arena.h" // CBMArena, cbm_arena_init/alloc/strdup/destroy +#include "arena.h" // CBMArena, cbm_arena_init/alloc/strdup/destroy +#include "doclink.h" // cbm_doclinks_extract: doc-comment references + doc-link scope #include "helpers.h" #include "lang_specs.h" #include "extract_unified.h" @@ -3405,6 +3406,10 @@ static CBMFileResult *extract_file_ex_body(const char *source, int source_len, C result->imports_count = result->imports.count; + /* Doc-comment references of the documented definitions, and the file's + * doc-link scope: both read the complete docstrings and the tree. */ + cbm_doclinks_extract(&ctx); + // Accumulate profiling counters atomic_fetch_add(&total_parse_ns, t1 - t0); atomic_fetch_add(&total_extract_ns, t2 - t1); diff --git a/internal/cbm/cbm.h b/internal/cbm/cbm.h index 0058f0a51..65f355417 100644 --- a/internal/cbm/cbm.h +++ b/internal/cbm/cbm.h @@ -574,6 +574,31 @@ typedef struct { int cap; } CBMFieldTypeArray; +/* One reference found in a definition's complete doc comment, or in the file's + * own doc (doclink.h has the syntax table and the per-language parsers). + * Resolution happens later, per file, in the pipeline + * (src/pipeline/doc_links.c). */ +typedef struct { + const char *source_qn; // QN of the documented definition (the edge source); + // for a file-level reference the file's module QN + const char *raw; // the reference as written, markup entities decoded + uint32_t line; // 1-based source line of the reference + uint32_t def_line; // 1-based start line of the documented definition + uint16_t syntax; // CBMDocLinkSyntax (doclink.h) + uint16_t flags; // CBM_DOCLINK_FLAG_* (doclink.h); set by the driver +} CBMDocLink; + +typedef struct { + CBMDocLink *items; + int count; + int cap; + /* Memory ran out while this file's doc-link data was extracted (a token, + * a reference value, a doc text or the scope blob was lost): the + * resolving half fails the layer instead of publishing a silently + * incomplete graph. */ + bool failed; +} CBMDocLinkArray; + // Full extraction result for one file. typedef struct CBMFileResult { CBMArena arena; // owns local memory; composites may also retain child arenas below @@ -668,6 +693,14 @@ typedef struct CBMFileResult { * the Rust inner docs (//!). NULL for other languages and undocumented * files. */ const char *module_doc; + + /* Doc-comment references of this file's definitions (doclink.h), and the + * file's doc-link scope: a language-tagged text blob with what OTHER + * files' doc-link resolution needs from this file (C#: namespaces, + * usings, type and member declarations). A pure function of the file's + * bytes; NULL for languages without a scope scanner. */ + CBMDocLinkArray doc_links; + const char *doc_scope; } CBMFileResult; // --- Enclosing function cache --- @@ -759,6 +792,18 @@ typedef struct { * POD section index. */ void *doc_memo; void *doc_pod_index; + /* Start line of every doc text doc_run_text built, keyed by the returned + * pointer (doclink.c), so a definition's doc references get exact source + * lines. NULL until the first doc; allocated in `scratch`. */ + void *doc_lines; + /* A per-file slot for the language's doc-link hooks (parse_doc and + * scan_scope, internal/cbm/doclink_.c): state that has to survive + * between the hook calls of ONE file, e.g. an index of the file's doc + * sections. NULL at the start of every file. What a hook stores here must + * be allocated in `scratch` (or `arena`), so that it ends with the file. + * The core never reads, interprets or frees it. It is the only place for + * such state: a static or thread-local cache is not. */ + void *doclink_state; } CBMExtractCtx; // --- Public API --- diff --git a/internal/cbm/doclink.c b/internal/cbm/doclink.c new file mode 100644 index 000000000..d7d9b8b41 --- /dev/null +++ b/internal/cbm/doclink.c @@ -0,0 +1,460 @@ +/* + * doclink.c — doc-comment references, extraction half: the language table, + * the doc-line side map and the per-file driver. See doclink.h. + */ +#include "doclink.h" + +#include "arena.h" +#include "foundation/constants.h" +#include "foundation/mem_core.h" + +#include +#include +#include + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +static _Atomic int doc_alloc_fail_after[CBM_DOCLINK_ALLOC_KINDS]; + +void cbm_doclink_test_fail_alloc_after(int kind, int nth) { + if (kind >= 0 && kind < CBM_DOCLINK_ALLOC_KINDS) { + atomic_store(&doc_alloc_fail_after[kind], nth); + } +} + +bool cbm_doclink_test_fail_alloc(int kind) { + if (kind < 0 || kind >= CBM_DOCLINK_ALLOC_KINDS) { + return false; + } + int n = atomic_load(&doc_alloc_fail_after[kind]); + while (n > 0) { + if (atomic_compare_exchange_weak(&doc_alloc_fail_after[kind], &n, n - 1)) { + return n == 1; + } + } + return false; +} + +void cbm_doclink_test_reset_alloc(void) { + for (int i = 0; i < CBM_DOCLINK_ALLOC_KINDS; i++) { + atomic_store(&doc_alloc_fail_after[i], 0); + } +} + +static _Atomic uint64_t doc_work_copied; +static _Atomic uint64_t doc_work_parse_input; +static _Atomic uint64_t doc_work_cleaned; + +void cbm_doclink_test_doc_work_reset(void) { + atomic_store(&doc_work_copied, 0); + atomic_store(&doc_work_parse_input, 0); + atomic_store(&doc_work_cleaned, 0); +} + +void cbm_doclink_test_doc_work(uint64_t *copied, uint64_t *parse_input, uint64_t *cleaned) { + *copied = atomic_load(&doc_work_copied); + *parse_input = atomic_load(&doc_work_parse_input); + *cleaned = atomic_load(&doc_work_cleaned); +} + +void cbm_doclink_test_note_doc_work(uint64_t copied, uint64_t parse_input, uint64_t cleaned) { + atomic_fetch_add(&doc_work_copied, copied); + atomic_fetch_add(&doc_work_parse_input, parse_input); + atomic_fetch_add(&doc_work_cleaned, cleaned); +} +#endif + +/* ── Link families ─────────────────────────────────────────────────── + * + * One row per CBMDocLinkSyntax value, generated from the family list in + * doclink.h (language, name, external, and the ship gate). */ + +static const CBMDocLinkFamily DOCLINK_FAMILIES[CBM_DOCLINK_SYNTAX_COUNT] = { +#define DOCLINK_FAMILY_ROW(id, lang, name, external, ships) [id] = {lang, name, external, ships}, + CBM_DOCLINK_FAMILY_LIST(DOCLINK_FAMILY_ROW) +#undef DOCLINK_FAMILY_ROW +}; + +const CBMDocLinkFamily *cbm_doclink_family(int syntax) { + if (syntax <= CBM_DOCLINK_NONE || syntax >= CBM_DOCLINK_SYNTAX_COUNT || + !DOCLINK_FAMILIES[syntax].name) { + return NULL; + } + return &DOCLINK_FAMILIES[syntax]; +} + +const char *cbm_doclink_syntax_name(int syntax) { + const CBMDocLinkFamily *f = cbm_doclink_family(syntax); + return f ? f->name : ""; +} + +bool cbm_doclink_syntax_is_external(int syntax) { + const CBMDocLinkFamily *f = cbm_doclink_family(syntax); + return f && f->external; +} + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +enum { DOCLINK_SHIP_AS_TABLE = 0, DOCLINK_SHIP_ON, DOCLINK_SHIP_OFF }; +static _Atomic unsigned char doclink_ship_override[CBM_DOCLINK_SYNTAX_COUNT]; + +void cbm_doclink_test_set_ships(int syntax, bool ships) { + if (cbm_doclink_family(syntax)) { + atomic_store(&doclink_ship_override[syntax], ships ? DOCLINK_SHIP_ON : DOCLINK_SHIP_OFF); + } +} + +void cbm_doclink_test_reset_ships(void) { + for (int i = 0; i < CBM_DOCLINK_SYNTAX_COUNT; i++) { + atomic_store(&doclink_ship_override[i], DOCLINK_SHIP_AS_TABLE); + } +} +#endif + +bool cbm_doclink_syntax_ships(int syntax) { + const CBMDocLinkFamily *f = cbm_doclink_family(syntax); + if (!f) { + return false; + } +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + unsigned char forced = atomic_load(&doclink_ship_override[syntax]); + if (forced != DOCLINK_SHIP_AS_TABLE) { + return forced == DOCLINK_SHIP_ON; + } +#endif + return f->ships; +} + +/* ── Language table ──────────────────────────────────────────────── */ + +/* One row per language with a doc-reference parser: everything the extraction + * half needs from a language leg. */ +typedef struct { + CBMLanguage lang; + /* One documented definition's doc text -> tokens (cbm_doclinks_push). */ + void (*parse_doc)(CBMExtractCtx *ctx, const CBMDefinition *def, const char *doc, + uint32_t doc_line); + /* The file's scope blob, or NULL: the language's resolver needs none. */ + const char *(*scan_scope)(CBMExtractCtx *ctx); + /* The tag line that blob starts with, and the blob as it is persisted + * (line numbers dropped). NULL: the blob is persisted as it is. */ + const char *scope_tag; + char *(*portable_scope)(const char *scope); + /* Labels that twin another definition of the same name and line (C# + * fields and constants are also emitted as module-level Variables with + * the same doc): only the first label carries the doc's references. */ + const char *twin_label; + const char *twin_of; +} doclink_lang_t; + +static const doclink_lang_t DOCLINK_LANGS[] = { + {.lang = CBM_LANG_CSHARP, + .parse_doc = cbm_doclink_cs_parse_doc, + .scan_scope = cbm_doclink_cs_scan_scope, + .scope_tag = CBM_DOCLINK_CS_SCOPE_TAG, + .portable_scope = cbm_doclink_cs_portable_scope, + .twin_label = "Variable", + .twin_of = "Field"}, + /* MSBuild project files: no references, but a scope the C# resolver reads */ + {.lang = CBM_LANG_XML, + .parse_doc = cbm_doclink_cs_project_parse_doc, + .scan_scope = cbm_doclink_cs_project_scan_scope, + .scope_tag = CBM_DOCLINK_CS_SCOPE_TAG, + .portable_scope = cbm_doclink_cs_portable_scope}, +}; + +static const doclink_lang_t *doclink_lang(CBMLanguage lang) { + for (size_t i = 0; i < sizeof(DOCLINK_LANGS) / sizeof(DOCLINK_LANGS[0]); i++) { + if (DOCLINK_LANGS[i].lang == lang) { + return &DOCLINK_LANGS[i]; + } + } + return NULL; +} + +bool cbm_doclink_lang_supported(CBMLanguage lang) { + return doclink_lang(lang) != NULL; +} + +/* ── Doc-line side map ───────────────────────────────────────────── */ + +typedef struct { + const char *doc; + uint32_t line; + bool parsed; /* its references were taken, for the first definition that has it */ +} doc_line_ent_t; + +typedef struct { + doc_line_ent_t *items; + int count; + int cap; + bool sorted; +} doc_line_map_t; + +enum { DOC_LINE_MAP_INIT = 64 }; + +static CBMArena *doclink_scratch(CBMExtractCtx *ctx) { + return ctx->scratch ? ctx->scratch : ctx->arena; +} + +/* The map could not hold a doc: its references then take their definition's + * line, and a doc shared by several declarators is taken once per declarator. + * The layer must not say ok over that (CBMDocLinkArray.failed). */ +static void doc_line_lost(CBMExtractCtx *ctx) { + if (ctx->result) { + ctx->result->doc_links.failed = true; + } +} + +void cbm_doclink_note_doc_line(CBMExtractCtx *ctx, const char *doc, uint32_t line) { + if (!ctx || !doc || !cbm_doclink_lang_supported(ctx->language)) { + return; + } + CBMArena *a = doclink_scratch(ctx); + doc_line_map_t *m = (doc_line_map_t *)ctx->doc_lines; + if (!m) { + m = (doc_line_map_t *)cbm_arena_alloc(a, sizeof(*m)); + if (!m) { + doc_line_lost(ctx); + return; + } + memset(m, 0, sizeof(*m)); + ctx->doc_lines = m; + } + if (m->count >= m->cap) { + int ncap = m->cap ? m->cap * PAIR_LEN : DOC_LINE_MAP_INIT; + doc_line_ent_t *grown; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (cbm_doclink_test_fail_alloc(CBM_DOCLINK_ALLOC_DOC_LINE)) { + grown = NULL; + } else +#endif + { + grown = (doc_line_ent_t *)cbm_arena_alloc(a, (size_t)ncap * sizeof(*grown)); + } + if (!grown) { + doc_line_lost(ctx); + return; + } + if (m->count > 0) { + memcpy(grown, m->items, (size_t)m->count * sizeof(*grown)); + } + m->items = grown; + m->cap = ncap; + } + m->items[m->count++] = (doc_line_ent_t){.doc = doc, .line = line}; + m->sorted = false; +} + +static int doc_line_cmp(const void *a, const void *b) { + uintptr_t x = (uintptr_t)((const doc_line_ent_t *)a)->doc; + uintptr_t y = (uintptr_t)((const doc_line_ent_t *)b)->doc; + return (x > y) - (x < y); +} + +/* Per-file metadata for this exact immutable doc pointer, never for a line. */ +static doc_line_ent_t *doc_info_of(CBMExtractCtx *ctx, const char *doc) { + doc_line_map_t *m = (doc_line_map_t *)ctx->doc_lines; + if (!m || m->count == 0) { + return NULL; + } + if (!m->sorted) { + qsort(m->items, (size_t)m->count, sizeof(m->items[0]), doc_line_cmp); + m->sorted = true; + } + int lo = 0; + int hi = m->count - SKIP_ONE; + uintptr_t key = (uintptr_t)doc; + while (lo <= hi) { + int mid = lo + ((hi - lo) / PAIR_LEN); + uintptr_t v = (uintptr_t)m->items[mid].doc; + if (v == key) { + return &m->items[mid]; + } + if (v < key) { + lo = mid + SKIP_ONE; + } else { + hi = mid - SKIP_ONE; + } + } + return NULL; +} + +/* A doc not built by doc_run_text keeps the definition's own line. */ +static uint32_t doc_line_of(CBMExtractCtx *ctx, const char *doc) { + const doc_line_ent_t *info = doc_info_of(ctx, doc); + return info ? info->line : 0; +} + +/* ── Persisted scope ─────────────────────────────────────────────── */ + +char *cbm_doclink_portable_scope(const char *scope) { + if (!scope) { + return NULL; + } + for (size_t i = 0; i < sizeof(DOCLINK_LANGS) / sizeof(DOCLINK_LANGS[0]); i++) { + const doclink_lang_t *L = &DOCLINK_LANGS[i]; + if (!L->scope_tag || !L->portable_scope) { + continue; + } + size_t tl = strlen(L->scope_tag); + if (strncmp(scope, L->scope_tag, tl) == 0 && (scope[tl] == '\n' || scope[tl] == '\0')) { + return L->portable_scope(scope); + } + } + return cbm_mem_strdup(CBM_MEM_CLASS_OTHER, scope); +} + +/* ── Driver ──────────────────────────────────────────────────────── */ + +enum { DOCLINK_INIT_CAP = 16 }; + +void cbm_doclinks_push(CBMDocLinkArray *arr, CBMArena *a, CBMDocLink link) { + if (arr->count >= arr->cap) { + int ncap = arr->cap ? arr->cap * PAIR_LEN : DOCLINK_INIT_CAP; + CBMDocLink *grown; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (cbm_doclink_test_fail_alloc(CBM_DOCLINK_ALLOC_TOKENS)) { + grown = NULL; + } else +#endif + { + grown = (CBMDocLink *)cbm_arena_alloc(a, (size_t)ncap * sizeof(*grown)); + } + if (!grown) { + arr->failed = true; /* the token is lost: the layer must not say ok */ + return; + } + if (arr->count > 0) { + memcpy(grown, arr->items, (size_t)arr->count * sizeof(*grown)); + } + arr->items = grown; + arr->cap = ncap; + } + arr->items[arr->count++] = link; +} + +typedef struct { + uint32_t line; + const char *name; +} doclink_twin_key_t; + +static int twin_key_cmp(const void *a, const void *b) { + const doclink_twin_key_t *x = (const doclink_twin_key_t *)a; + const doclink_twin_key_t *y = (const doclink_twin_key_t *)b; + if (x->line != y->line) { + return x->line < y->line ? -1 : 1; + } + return strcmp(x->name, y->name); +} + +/* The (line, name) keys of the definitions a twin label duplicates, sorted; + * NULL when there are none. */ +static doclink_twin_key_t *twin_keys(CBMExtractCtx *ctx, const char *label, int *out_count) { + *out_count = 0; + const CBMDefArray *defs = &ctx->result->defs; + int n = 0; + for (int i = 0; i < defs->count; i++) { + const CBMDefinition *d = &defs->items[i]; + if (d->label && d->name && strcmp(d->label, label) == 0) { + n++; + } + } + if (n == 0) { + return NULL; + } + doclink_twin_key_t *keys = (doclink_twin_key_t *)cbm_arena_alloc( + doclink_scratch(ctx), (size_t)n * sizeof(doclink_twin_key_t)); + if (!keys) { + return NULL; + } + int k = 0; + for (int i = 0; i < defs->count; i++) { + const CBMDefinition *d = &defs->items[i]; + if (d->label && d->name && strcmp(d->label, label) == 0) { + keys[k].line = d->start_line; + keys[k].name = d->name; + k++; + } + } + qsort(keys, (size_t)k, sizeof(keys[0]), twin_key_cmp); + *out_count = k; + return keys; +} + +void cbm_doclinks_extract(CBMExtractCtx *ctx) { + if (!ctx || !ctx->result) { + return; + } + const doclink_lang_t *L = doclink_lang(ctx->language); + if (!L) { + return; + } + int twin_count = 0; + doclink_twin_key_t *twins = L->twin_label ? twin_keys(ctx, L->twin_of, &twin_count) : NULL; + CBMDefArray *defs = &ctx->result->defs; + for (int i = 0; i < defs->count; i++) { + const CBMDefinition *d = &defs->items[i]; + if (!d->docstring || !d->docstring[0] || !d->qualified_name || !d->name) { + continue; + } + if (twins && d->label && strcmp(d->label, L->twin_label) == 0) { + doclink_twin_key_t key = {d->start_line, d->name}; + if (bsearch(&key, twins, (size_t)twin_count, sizeof(twins[0]), twin_key_cmp)) { + /* the Field twin carries these references -- and so those of + * the declarators after it, which share this doc text */ + doc_line_ent_t *taken = doc_info_of(ctx, d->docstring); + if (taken) { + taken->parsed = true; + } + continue; + } + } + doc_line_ent_t *info = doc_info_of(ctx, d->docstring); + uint32_t doc_line = info ? info->line : 0; + bool csharp = ctx->language == CBM_LANG_CSHARP; + if (csharp && info && info->parsed) { + /* One C# doc comment documents every declarator of a field + * declaration (`int a, b, c;` share its text, extract_defs.c): its + * references are taken once, from the first declarator. Taken + * again per declarator, the tokens, their resolutions and the + * rows grow with references x declarators. */ + continue; + } + if (csharp) { + if (!cbm_doclink_cs_parse_doc_checked(ctx, d, d->docstring, + doc_line ? doc_line : d->start_line)) { + ctx->result->doc_links.failed = true; /* a reference value was lost */ + } + } else { + L->parse_doc(ctx, d, d->docstring, doc_line ? doc_line : d->start_line); + } + if (info) { + info->parsed = true; + } + } + /* The file's own doc: its references belong to the file. The parser gets + * a definition-shaped stand-in for the file; what it pushes is marked, and + * the resolving half takes the File node as the source. No qualified name + * for that node is computed on this side. */ + const char *file_doc = ctx->result->module_doc; + if (file_doc && file_doc[0]) { + CBMDocLinkArray *arr = &ctx->result->doc_links; + int first = arr->count; + CBMDefinition file_def = { + .name = ctx->rel_path, + .qualified_name = ctx->module_qn ? ctx->module_qn : "", + .label = "File", + .file_path = ctx->rel_path, + .start_line = SKIP_ONE, + .end_line = SKIP_ONE, + .docstring = file_doc, + }; + uint32_t doc_line = doc_line_of(ctx, file_doc); + L->parse_doc(ctx, &file_def, file_doc, doc_line ? doc_line : file_def.start_line); + for (int i = first; i < arr->count; i++) { + arr->items[i].flags |= CBM_DOCLINK_FLAG_FILE; + } + } + if (L->scan_scope) { + ctx->result->doc_scope = L->scan_scope(ctx); + } +} diff --git a/internal/cbm/doclink.h b/internal/cbm/doclink.h new file mode 100644 index 000000000..7b85ad45a --- /dev/null +++ b/internal/cbm/doclink.h @@ -0,0 +1,224 @@ +/* + * doclink.h — doc-comment references, extraction half. + * + * A definition's complete doc comment (extract_defs.c stores it whole) can + * name other code: C# ``, Java `{@link ...}`, Rust intra-doc + * links, and so on. Extraction only FINDS those references; whether one names + * a graph node is decided later, per file, by the pipeline's resolver + * (src/pipeline/doc_links.c), which turns them into MENTIONS edges or + * doc_link_unresolved rows. + * + * A language leg plugs in at two places of this half, and nowhere else: + * - its link families: lines of CBM_DOCLINK_FAMILY_LIST below (enum value, + * name, ship gate); + * - one row of the language table in doclink.c: + * parse_doc one documented definition's doc text -> CBMDocLink + * tokens + * scan_scope the file's doc-link SCOPE: a language-tagged text blob + * with what other files' resolution needs from this file + * (C#: namespaces, usings, type and member + * declarations). A pure function of the file's bytes, + * persisted with the file's LSP surface, and so + * identical on full and incremental runs. Optional. + * scope_tag, the blob's tag line and its persisted form (line + * portable_scope numbers dropped). Optional. + * State a language's hooks need between their calls for one file lives in + * ctx->doclink_state (cbm.h), never in a static or thread-local. + * A language without a row produces no tokens and no scope. The resolving + * half's hooks are a cbm_doclink_resolver_t (src/pipeline/doc_links.h). + */ +#ifndef CBM_DOCLINK_H +#define CBM_DOCLINK_H + +#include "cbm.h" + +#include +#include +#include + +/* Every link family, across languages: one line per (language, reference + * form), and the ONE place a language leg adds its families -- + * + * X(enum value, language, name, external, ships) + * + * name the "syntax" MENTIONS-edge property and doc_link_unresolved + * column, so it is public surface; with the language it is the + * tier a release audit judges. Names repeat across languages + * (`see` is C#'s, Java's, Kotlin's ...): the value, not the name, + * identifies the family, and a row's language is its file's. + * external outside the repository by construction (a URL): never looked up + * ships the SHIP GATE. true: a resolved reference becomes a MENTIONS + * edge. false: the tier is below the audit's bar; its references + * still resolve, but a resolved one stays a doc_link_unresolved + * row with reason below_bar_tier (an unresolved one keeps its own + * reason). Nothing else about the family changes. + * + * The enum below and the family table (doclink.c) are both generated from this + * list, so a value cannot exist without its name and its gate. */ +#define CBM_DOCLINK_FAMILY_LIST(X) \ + /* csharp */ \ + X(CBM_DOCLINK_CS_SEE, CBM_LANG_CSHARP, "see", false, true) \ + X(CBM_DOCLINK_CS_SEEALSO, CBM_LANG_CSHARP, "seealso", false, true) \ + X(CBM_DOCLINK_CS_EXCEPTION, CBM_LANG_CSHARP, "exception", false, true) \ + X(CBM_DOCLINK_CS_INHERITDOC, CBM_LANG_CSHARP, "inheritdoc", false, true) \ + /* any language (CBM_LANG_COUNT): a URL is never an edge */ \ + X(CBM_DOCLINK_HREF, CBM_LANG_COUNT, "href", true, false) + +typedef enum { + CBM_DOCLINK_NONE = 0, +#define CBM_DOCLINK_FAMILY_VALUE(id, lang, name, external, ships) id, + CBM_DOCLINK_FAMILY_LIST(CBM_DOCLINK_FAMILY_VALUE) +#undef CBM_DOCLINK_FAMILY_VALUE + CBM_DOCLINK_SYNTAX_COUNT +} CBMDocLinkSyntax; + +/* CBMDocLink.flags. */ +enum { + /* The reference is written in the FILE's own doc (CBMFileResult.module_doc: + * a Go package comment, Rust `//!` inner docs), not in a definition's. Its + * edge source is the file's File node, which the resolving half looks up + * from the file it is resolving; `source_qn` is not the source then. Set + * by the driver (cbm_doclinks_extract) on what a parser pushes for the + * file-level doc -- a parser never sets it. */ + CBM_DOCLINK_FLAG_FILE = 1, +}; + +/* One link family: a line of CBM_DOCLINK_FAMILY_LIST. */ +typedef struct { + CBMLanguage lang; /* the language whose doc comments write it; CBM_LANG_COUNT: any */ + const char *name; + bool external; + bool ships; +} CBMDocLinkFamily; + +/* The family of a syntax value; NULL for a value that names none. */ +const CBMDocLinkFamily *cbm_doclink_family(int syntax); + +/* "see", "seealso", "exception", "inheritdoc", "href"; "" for an unknown value. */ +const char *cbm_doclink_syntax_name(int syntax); + +/* True for a syntax whose reference is outside the repository by construction + * (a URL): the resolver records it as `external` without a lookup. */ +bool cbm_doclink_syntax_is_external(int syntax); + +/* True when resolved references of this family become edges (the ship gate). */ +bool cbm_doclink_syntax_ships(int syntax); + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +/* Test seam: ship or hold back one family whatever the table says, until + * cbm_doclink_test_reset_ships. Test builds only. */ +void cbm_doclink_test_set_ships(int syntax, bool ships); +void cbm_doclink_test_reset_ships(void); +#endif + +/* True when `lang` has a doc-reference parser. */ +bool cbm_doclink_lang_supported(CBMLanguage lang); + +/* Remember that the doc text `doc` (as returned by the doc-comment extractor) + * starts at 1-based source line `line`. Called by doc_run_text; a no-op for + * languages without a parser. */ +void cbm_doclink_note_doc_line(CBMExtractCtx *ctx, const char *doc, uint32_t line); + +/* Harvest the references of every documented definition into + * ctx->result->doc_links and the file's scope into ctx->result->doc_scope. + * Called once per file at the end of extraction, with the tree still alive. + * + * A file-level doc (ctx->result->module_doc) goes through the same parse_doc + * hook once more: `def` is then a stand-in for the FILE (label "File", name + * and file_path the relative path, qualified_name the file's module QN, + * start_line 1) and `doc_line` the doc's first line. The parser fills its + * tokens exactly as for a definition; the driver marks them + * CBM_DOCLINK_FLAG_FILE afterwards. */ +void cbm_doclinks_extract(CBMExtractCtx *ctx); + +/* Append one token (copies nothing: `raw` must live in the result arena). */ +void cbm_doclinks_push(CBMDocLinkArray *arr, CBMArena *a, CBMDocLink link); + +/* The scope as persisted with the file's LSP surface: line numbers zeroed, so + * an edit that only moves lines leaves it (and the surface hash) unchanged. + * Other files' references never read a file's lines; only the file's own + * references do, and those are resolved from its fresh extraction. Returns a + * memory-core block (release with cbm_free(CBM_MEM_CLASS_OTHER, p)), NULL on + * allocation failure. */ +char *cbm_doclink_portable_scope(const char *scope); + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +/* Fail one selected allocation attempt, then automatically disarm. */ +enum { + CBM_DOCLINK_ALLOC_SPAN, + CBM_DOCLINK_ALLOC_TEXT, + CBM_DOCLINK_ALLOC_VALUE, + CBM_DOCLINK_ALLOC_TOKENS, + CBM_DOCLINK_ALLOC_SCOPE, /* the C# scope scan, as if its builder ran out */ + CBM_DOCLINK_ALLOC_PROJECT, /* the project-file scan, likewise */ + CBM_DOCLINK_ALLOC_DOC_LINE, /* the doc-line map's growth */ + CBM_DOCLINK_ALLOC_KINDS, +}; +void cbm_doclink_test_fail_alloc_after(int kind, int nth); +bool cbm_doclink_test_fail_alloc(int kind); +void cbm_doclink_test_reset_alloc(void); + +/* The C# scope scan of the file at `rel_path` writes, after its records, one + * the resolver refuses -- as if writer and reader disagreed (NULL or "": + * none). What is stored for the file and what this run resolves both carry + * it. */ +void cbm_doclink_cs_test_spoil_scope(const char *rel_path); + +/* Actual comment memcpy bytes, input bytes submitted to the C# lexical parser, + * and bytes allocated for cleaned reference values; separate from token output. */ +void cbm_doclink_test_doc_work_reset(void); +void cbm_doclink_test_doc_work(uint64_t *copied, uint64_t *parse_input, uint64_t *cleaned); +void cbm_doclink_test_note_doc_work(uint64_t copied, uint64_t parse_input, uint64_t cleaned); +#endif + +/* ── C# (doclink_cs.c) ─────────────────────────────────────────────── */ + +void cbm_doclink_cs_parse_doc(CBMExtractCtx *ctx, const CBMDefinition *def, const char *doc, + uint32_t doc_line); +/* False if allocating a value or appending a token failed. Only a complete + * lexical result can be shared with another definition. */ +bool cbm_doclink_cs_parse_doc_checked(CBMExtractCtx *ctx, const CBMDefinition *def, const char *doc, + uint32_t doc_line); +const char *cbm_doclink_cs_scan_scope(CBMExtractCtx *ctx); +char *cbm_doclink_cs_portable_scope(const char *scope); + +/* MSBuild project files (*.csproj, *.props, *.targets) set the global usings + * of a C# project, so they have a scope blob too: the C# resolver evaluates + * those blobs and never opens a file. The blob carries the C# tag. The scan + * is gated by the file's name: any other XML file costs nothing and has no + * blob. A project file holds no doc references: its parse_doc does nothing. */ +void cbm_doclink_cs_project_parse_doc(CBMExtractCtx *ctx, const CBMDefinition *def, const char *doc, + uint32_t doc_line); +const char *cbm_doclink_cs_project_scan_scope(CBMExtractCtx *ctx); + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +/* Test seam: what the C# scope scans since the last reset cost -- the text + * positions, brace-stack entries and modifier children visited, and the bytes + * they took from the scratch arena. A test holds these against the size of its input, so that a + * scan whose cost grows faster than its input fails without a clock. Test + * builds only. */ +void cbm_doclink_cs_test_cost_reset(void); +void cbm_doclink_cs_test_cost(uint64_t *text_steps, uint64_t *scratch_bytes); +/* Completed C# scope-builder bytes since the cost reset, including any embedded + * terminator. A direct-extraction test compares this with the returned C string. */ +uint64_t cbm_doclink_cs_test_scope_bytes(void); +/* Import-index construction and a one-shot candidate-buffer failure. Query + * visits remain in the pipeline's existing test_work counter. */ +uint64_t cbm_doclink_cs_test_index_work(void); +void cbm_doclink_cs_test_fail_candidate_alloc(bool enabled); +bool cbm_doclink_cs_test_candidate_alloc_failed(void); +#endif + +/* Normalize one C# parameter type as written in a declaration or a cref + * parameter list: attributes, ref/out/in/params/this/scoped modifiers, type + * arguments, namespaces, nullable markers and a trailing parameter name are + * dropped, BCL names map to their keyword (Int32 -> int), array and pointer + * suffixes stay. "?" (a type nothing is known about) when nothing is left, when + * the text is too long to be a type, or when the result does not fit `out`: + * never a cut name. Writes a NUL-terminated string and returns its length. */ +size_t cbm_doclink_cs_norm_type(const char *in, size_t len, char *out, size_t cap); + +/* The C# scope blob starts with this tag line. */ +#define CBM_DOCLINK_CS_SCOPE_TAG "cs1" + +#endif /* CBM_DOCLINK_H */ diff --git a/internal/cbm/doclink_cs.c b/internal/cbm/doclink_cs.c new file mode 100644 index 000000000..57780da9c --- /dev/null +++ b/internal/cbm/doclink_cs.c @@ -0,0 +1,3790 @@ +/* + * doclink_cs.c — C# doc-comment references and the C# doc-link scope. + * + * Tokens: the cref of , , and , and the + * href of / (external). / name the + * definition's own parameters and are not references; pulls doc + * text from another file. + * + * Scope blob (one record per line, tab-separated, first line "cs1"; records + * in document order, so a member follows its type): + * X from to lines whose declarations could not be + * placed (the braces stop pairing, a + * namespace cannot be named, a block nests + * past a limit): a definition there has no + * known scope + * R id parent start end name + * namespace region: `name` as its + * declaration writes it (`A.B`), inside the + * region `parent` (0 = the file) + * U region kind alias target + * using: n namespace, s static, a alias; + * a `g` after it for a `global using` + * T region start end kind outer name tparams bases + * type: c class, s struct, i interface, e + * enum, r record class, t record struct, d + * delegate; then `p` when it is declared + * partial, then `!` when a parse error hides + * some of its members. `outer` is the ordinal + * of the enclosing type's T record (its + * position among the file's T records), `-` + * for none. tparams ','-joined, bases + * '|'-joined as written + * M start kind explicit type name tparams sig + * member of the type with ordinal `type`: c + * callable (method, constructor, primary + * constructor), v field (constant, enum + * member), p property (record parameter), e + * event, o operator (name: its token, or + * `implicit` / `explicit`), x indexer (name + * `this`); then `s` for what `using static` + * brings in (static or const, an enum's + * member, no extension method; a static + * constructor too). sig: '|'-joined normalized + * parameter types, '-' for v, p and e + * Q name a type that is declared, but whose + * namespace or outer type could not be + * established + * A record names its outer type and its owner by ordinal, never by a path, so + * the blob grows with the file and not with the nesting. Everything is + * derived from the tree alone, so the blob is a pure function of the file's + * bytes. Nesting is the tree's while the tree has no parse error, and the + * braces' otherwise (see "Braces" below). + */ +#include "doclink.h" + +#include "arena.h" +#include "foundation/constants.h" +#include "foundation/mem_core.h" +#include "tree_sitter/api.h" + +#include +#include +#include +#include +#include + +/* ── Small arena string builder ──────────────────────────────────── */ + +typedef struct { + CBMArena *a; + char *buf; + size_t len; + size_t cap; + bool failed; +} cs_sb_t; + +enum { CS_SB_INIT = 1024, CS_UINT_DIGITS = 16, CS_NAME_MAX = 512 }; + +static void sb_reserve(cs_sb_t *sb, size_t extra) { + if (sb->failed || sb->len + extra + SKIP_ONE <= sb->cap) { + return; + } + size_t ncap = sb->cap ? sb->cap : CS_SB_INIT; + while (ncap < sb->len + extra + SKIP_ONE) { + ncap *= PAIR_LEN; + } + char *grown = (char *)cbm_arena_alloc(sb->a, ncap); + if (!grown) { + sb->failed = true; + return; + } + if (sb->len > 0) { + memcpy(grown, sb->buf, sb->len); + } + sb->buf = grown; + sb->cap = ncap; +} + +static void sb_putn(cs_sb_t *sb, const char *s, size_t n) { + sb_reserve(sb, n); + if (sb->failed) { + return; + } + memcpy(sb->buf + sb->len, s, n); + sb->len += n; + sb->buf[sb->len] = '\0'; +} + +static void sb_puts(cs_sb_t *sb, const char *s) { + sb_putn(sb, s, strlen(s)); +} + +static void sb_putc(cs_sb_t *sb, char c) { + sb_putn(sb, &c, SKIP_ONE); +} + +static void sb_putu(cs_sb_t *sb, uint32_t v) { + char tmp[CS_UINT_DIGITS]; + int n = snprintf(tmp, sizeof(tmp), "%u", v); + if (n > 0) { + sb_putn(sb, tmp, (size_t)n); + } +} + +/* ── Well-formed text ───────────────────────────────────────────── + * + * A scope blob is stored as a JSON string, and a JSON writer refuses a + * string that is not UTF-8: every byte the scans write is well-formed UTF-8. + * A name or a text of a C# source that is not is not kept (a name: the + * declaration is not placed; a text: "?"); project-file text gets U+FFFD for + * every byte that is no part of a well-formed sequence. Both are a function + * of the file's bytes alone. */ + +static const char CS_REPLACEMENT[] = "\xEF\xBF\xBD"; /* U+FFFD */ + +enum { + CS_UTF8_TAIL_MASK = 0xC0, + CS_UTF8_TAIL = 0x80, +}; + +/* Length of the well-formed UTF-8 sequence at s[0, n), 1 to 4; 0 when there + * is none: a stray continuation byte, an overlong form, a surrogate, a code + * point past U+10FFFF, a sequence cut short. */ +static size_t cs_utf8_len(const unsigned char *s, size_t n) { + if (n == 0) { + return 0; + } + unsigned char c = s[0]; + if (c < 0x80) { + return SKIP_ONE; + } + size_t len = 0; + if (c >= 0xC2 && c <= 0xDF) { + len = PAIR_LEN; + } else if (c >= 0xE0 && c <= 0xEF) { + len = 3; + } else if (c >= 0xF0 && c <= 0xF4) { + len = 4; + } + if (len == 0 || n < len) { + return 0; + } + unsigned char c1 = s[1]; + if ((c == 0xE0 && c1 < 0xA0) || (c == 0xED && c1 > 0x9F) || (c == 0xF0 && c1 < 0x90) || + (c == 0xF4 && c1 > 0x8F)) { + return 0; /* overlong, surrogate, or past U+10FFFF */ + } + for (size_t k = SKIP_ONE; k < len; k++) { + if ((s[k] & CS_UTF8_TAIL_MASK) != CS_UTF8_TAIL) { + return 0; + } + } + return len; +} + +/* True when s[0, n) is well-formed UTF-8. */ +static bool cs_utf8_ok(const char *s, size_t n) { + for (size_t i = 0; i < n;) { + size_t l = cs_utf8_len((const unsigned char *)s + i, n - i); + if (l == 0) { + return false; + } + i += l; + } + return true; +} + +/* ── Doc-comment references ──────────────────────────────────────── */ + +static int cs_tag_syntax(const char *name, size_t len) { + static const struct { + const char *name; + int syntax; + } tags[] = { + {"see", CBM_DOCLINK_CS_SEE}, + {"seealso", CBM_DOCLINK_CS_SEEALSO}, + {"exception", CBM_DOCLINK_CS_EXCEPTION}, + {"inheritdoc", CBM_DOCLINK_CS_INHERITDOC}, + }; + for (size_t i = 0; i < sizeof(tags) / sizeof(tags[0]); i++) { + if (strlen(tags[i].name) == len && memcmp(tags[i].name, name, len) == 0) { + return tags[i].syntax; + } + } + return CBM_DOCLINK_NONE; +} + +static bool cs_attr_name_char(char c) { + return isalnum((unsigned char)c) || c == '_' || c == ':' || c == '.' || c == '-'; +} + +/* Decode the five XML entities, drop the comment prefix (`///`, ` * `) that + * follows a line break inside a value, collapse whitespace, trim. */ +static const char *cs_clean_value(CBMArena *a, const char *v, size_t n) { + char *out; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (cbm_doclink_test_fail_alloc(CBM_DOCLINK_ALLOC_VALUE)) { + out = NULL; + } else +#endif + { + out = (char *)cbm_arena_alloc(a, n + SKIP_ONE); + } +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (out) { + cbm_doclink_test_note_doc_work(0, 0, n + SKIP_ONE); + } +#endif + if (!out) { + return NULL; + } + static const struct { + const char *ent; + char ch; + } ents[] = {{"<", '<'}, {">", '>'}, {"&", '&'}, {""", '"'}, {"'", '\''}}; + size_t w = 0; + bool space = false; + size_t i = 0; + while (i < n) { + char c = v[i]; + if (c == '\n' || c == '\r') { + i++; + while (i < n && (v[i] == ' ' || v[i] == '\t' || v[i] == '\r' || v[i] == '\n')) { + i++; + } + static const char line_doc[] = "///"; + size_t ld = sizeof(line_doc) - SKIP_ONE; + if (i + ld <= n && memcmp(v + i, line_doc, ld) == 0) { + i += ld; /* the next `///` line of the same comment */ + } else if (i < n && v[i] == '*' && !(i + SKIP_ONE < n && v[i + SKIP_ONE] == '/')) { + i++; /* a block comment's leading star */ + } + space = true; + continue; + } + if (c == ' ' || c == '\t') { + space = true; + i++; + continue; + } + if (space && w > 0) { + out[w++] = ' '; + } + space = false; + if (c == '&') { + bool decoded = false; + for (size_t e = 0; e < sizeof(ents) / sizeof(ents[0]); e++) { + size_t el = strlen(ents[e].ent); + if (i + el <= n && memcmp(v + i, ents[e].ent, el) == 0) { + out[w++] = ents[e].ch; + i += el; + decoded = true; + break; + } + } + if (decoded) { + continue; + } + } + out[w++] = c; + i++; + } + out[w] = '\0'; + return out; +} + +bool cbm_doclink_cs_parse_doc_checked(CBMExtractCtx *ctx, const CBMDefinition *def, const char *doc, + uint32_t doc_line) { +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + cbm_doclink_test_note_doc_work(0, strlen(doc), 0); +#endif + bool complete = true; + const char *p = doc; + const char *counted = doc; + uint32_t line = doc_line; + while ((p = strchr(p, '<')) != NULL) { + /* what stands in an XML comment or a CDATA section is text, not + * markup: a cref written there is no reference */ + static const char comment_open[] = "" : "]]>"); + if (!end) { + break; /* it does not end: the rest of the text is inside it */ + } + p = end + 3; + continue; + } + const char *tag = p; + const char *q = p + SKIP_ONE; + while (*q == ' ' || *q == '\t') { + q++; + } + const char *name = q; + while (isalpha((unsigned char)*q)) { + q++; + } + int syntax = cs_tag_syntax(name, (size_t)(q - name)); + if (syntax == CBM_DOCLINK_NONE || + !(*q == ' ' || *q == '\t' || *q == '\n' || *q == '\r' || *q == '/' || *q == '>')) { + p = tag + SKIP_ONE; + continue; + } + const char *cref = NULL; + size_t cref_len = 0; + const char *href = NULL; + size_t href_len = 0; + bool closed = false; + while (*q) { + if (*q == '>') { + closed = true; + q++; + break; + } + if (*q == '<') { + break; /* malformed: another tag starts first */ + } + if (*q == '"' || *q == '\'') { + const char *end = strchr(q + SKIP_ONE, *q); + if (!end) { + break; + } + q = end + SKIP_ONE; + continue; + } + if (!cs_attr_name_char(*q)) { + q++; + continue; + } + const char *an = q; + while (cs_attr_name_char(*q)) { + q++; + } + size_t an_len = (size_t)(q - an); + const char *r = q; + while (*r == ' ' || *r == '\t' || *r == '\n' || *r == '\r') { + r++; + } + if (*r != '=') { + continue; + } + r++; + while (*r == ' ' || *r == '\t' || *r == '\n' || *r == '\r') { + r++; + } + if (*r != '"' && *r != '\'') { + q = r; + continue; + } + const char *vend = strchr(r + SKIP_ONE, *r); + if (!vend) { + break; + } + const char *val = r + SKIP_ONE; + size_t val_len = (size_t)(vend - val); + if (an_len == 4 && memcmp(an, "cref", 4) == 0) { + cref = val; + cref_len = val_len; + } else if (an_len == 4 && memcmp(an, "href", 4) == 0) { + href = val; + href_len = val_len; + } + q = vend + SKIP_ONE; + } + if (!closed) { + p = tag + SKIP_ONE; + continue; + } + for (; counted < tag; counted++) { + if (*counted == '\n') { + line++; + } + } + const char *raw = NULL; + int tok_syntax = CBM_DOCLINK_NONE; + if (cref) { + raw = cs_clean_value(ctx->arena, cref, cref_len); + tok_syntax = syntax; + } else if (href && (syntax == CBM_DOCLINK_CS_SEE || syntax == CBM_DOCLINK_CS_SEEALSO)) { + raw = cs_clean_value(ctx->arena, href, href_len); + tok_syntax = CBM_DOCLINK_HREF; + } + if (tok_syntax != CBM_DOCLINK_NONE && !raw) { + complete = false; + } + if (raw && raw[0]) { + CBMDocLink link = { + .source_qn = def->qualified_name, + .raw = raw, + .line = line, + .def_line = def->start_line, + .syntax = (uint16_t)tok_syntax, + }; + int before = ctx->result->doc_links.count; + cbm_doclinks_push(&ctx->result->doc_links, ctx->arena, link); + if (ctx->result->doc_links.count == before) { + complete = false; + } + } + p = q; + } + return complete; +} + +void cbm_doclink_cs_parse_doc(CBMExtractCtx *ctx, const CBMDefinition *def, const char *doc, + uint32_t doc_line) { + (void)cbm_doclink_cs_parse_doc_checked(ctx, def, doc, doc_line); +} + +/* ── Parameter type normalization ────────────────────────────────── */ + +typedef struct { + char *s; + size_t len; +} cs_str_t; + +static bool cs_ident_start(char c) { + return isalpha((unsigned char)c) || c == '_' || c == '@'; +} + +static bool cs_ident_char(char c) { + return isalnum((unsigned char)c) || c == '_'; +} + +/* Remove [start, end) from s. */ +static void cs_cut(cs_str_t *t, size_t start, size_t end) { + memmove(t->s + start, t->s + end, t->len - end + SKIP_ONE); + t->len -= end - start; +} + +static void cs_trim(cs_str_t *t) { + size_t b = 0; + while (b < t->len && isspace((unsigned char)t->s[b])) { + b++; + } + if (b > 0) { + cs_cut(t, 0, b); + } + while (t->len > 0 && isspace((unsigned char)t->s[t->len - SKIP_ONE])) { + t->s[--t->len] = '\0'; + } +} + +/* Index just past the bracket group opened at s[i] (one of < { [ ( ), + * matching nested groups of any of those kinds; t->len when unbalanced. */ +static size_t cs_group_end(const cs_str_t *t, size_t i) { + int depth = 0; + for (size_t k = i; k < t->len; k++) { + char c = t->s[k]; + if (c == '<' || c == '{' || c == '[' || c == '(') { + depth++; + } else if (c == '>' || c == '}' || c == ']' || c == ')') { + depth--; + if (depth == 0) { + return k + SKIP_ONE; + } + } + } + return t->len; +} + +/* Leading parameter attribute lists: [NotNull] [In] T x. */ +static void cs_drop_attributes(cs_str_t *t) { + cs_trim(t); + while (t->len > 0 && t->s[0] == '[') { + size_t e = cs_group_end(t, 0); + cs_cut(t, 0, e); + cs_trim(t); + } +} + +/* Nullable / Nullable{X} (optionally System.- or global::System.- + * qualified) -> X. */ +static void cs_unwrap_nullable(cs_str_t *t) { + static const char *const prefixes[] = {"global::System.Nullable", "System.Nullable", + "Nullable"}; + /* one pass: what an unwrapped group leaves at its place is looked at + * again (Nullable>), the text before it never is */ + for (size_t i = 0; i < t->len;) { + bool unwrapped = false; + bool boundary = i == 0 || !(cs_ident_char(t->s[i - SKIP_ONE]) || + t->s[i - SKIP_ONE] == '.' || t->s[i - SKIP_ONE] == ':'); + for (size_t pi = 0; boundary && !unwrapped && pi < sizeof(prefixes) / sizeof(prefixes[0]); + pi++) { + size_t pl = strlen(prefixes[pi]); + if (i + pl > t->len || memcmp(t->s + i, prefixes[pi], pl) != 0) { + continue; + } + size_t j = i + pl; + while (j < t->len && t->s[j] == ' ') { + j++; + } + if (j >= t->len || (t->s[j] != '<' && t->s[j] != '{')) { + continue; + } + size_t e = cs_group_end(t, j); + if (e > t->len || e <= j + SKIP_ONE) { + continue; + } + /* keep the inner text */ + size_t inner_len = e - j - PAIR_LEN; + memmove(t->s + i, t->s + j + SKIP_ONE, inner_len); + memmove(t->s + i + inner_len, t->s + e, t->len - e + SKIP_ONE); + t->len = i + inner_len + (t->len - e); + unwrapped = true; + } + if (!unwrapped) { + i++; + } + } +} + +static bool cs_word_at(const cs_str_t *t, size_t i, const char *w) { + size_t wl = strlen(w); + if (i + wl > t->len || memcmp(t->s + i, w, wl) != 0) { + return false; + } + if (i > 0 && cs_ident_char(t->s[i - SKIP_ONE])) { + return false; + } + return i + wl < t->len && isspace((unsigned char)t->s[i + wl]); +} + +static void cs_drop_modifiers(cs_str_t *t) { + static const char *const mods[] = {"ref", "out", "in", "params", + "this", "scoped", "readonly", "final"}; + for (size_t i = 0; i < t->len;) { + bool cut = false; + for (size_t m = 0; m < sizeof(mods) / sizeof(mods[0]); m++) { + if (cs_word_at(t, i, mods[m])) { + size_t e = i + strlen(mods[m]); + while (e < t->len && isspace((unsigned char)t->s[e])) { + e++; + } + cs_cut(t, i, e); + cut = true; + break; + } + } + if (!cut) { + i++; + } + } +} + +/* Doc-ID arity markers: `1, ``0. */ +static void cs_drop_backtick_arity(cs_str_t *t) { + for (size_t i = 0; i < t->len;) { + if (t->s[i] != '`') { + i++; + continue; + } + size_t e = i; + while (e < t->len && t->s[e] == '`') { + e++; + } + size_t d = e; + while (d < t->len && isdigit((unsigned char)t->s[d])) { + d++; + } + if (d > e) { + cs_cut(t, i, d); + } else { + i = e; + } + } +} + +enum { CS_NORM_WORK = 512 }; + +/* Remove every <...> and {...} group. An opener pairs with the next closer of + * its own kind that no unpaired opener stands before (`<` with `>`, `{` with + * `}`), so groups nest; what does not pair stays. Two passes over the text: + * pair, then copy what is outside. */ +static void cs_drop_type_args(cs_str_t *t) { + uint16_t past[CS_NORM_WORK]; /* for a paired opener: index just past its closer */ + uint16_t open[CS_NORM_WORK]; + size_t depth = 0; + if (t->len >= CS_NORM_WORK) { + return; + } + for (size_t i = 0; i < t->len; i++) { + char c = t->s[i]; + past[i] = 0; + if (c == '<' || c == '{') { + open[depth++] = (uint16_t)i; + } else if (depth > 0 && ((c == '>' && t->s[open[depth - SKIP_ONE]] == '<') || + (c == '}' && t->s[open[depth - SKIP_ONE]] == '{'))) { + past[open[--depth]] = (uint16_t)(i + SKIP_ONE); + } + } + size_t w = 0; + for (size_t i = 0; i < t->len;) { + if (past[i]) { + i = past[i]; + } else { + t->s[w++] = t->s[i++]; + } + } + t->s[w] = '\0'; + t->len = w; +} + +static const char *cs_bcl_alias(const char *base, size_t len) { + static const struct { + const char *bcl; + const char *kw; + } map[] = { + {"Int32", "int"}, {"Int64", "long"}, {"Int16", "short"}, {"Byte", "byte"}, + {"SByte", "sbyte"}, {"UInt32", "uint"}, {"UInt64", "ulong"}, {"UInt16", "ushort"}, + {"Single", "float"}, {"Double", "double"}, {"Decimal", "decimal"}, {"Boolean", "bool"}, + {"Char", "char"}, {"String", "string"}, {"Object", "object"}, {"IntPtr", "nint"}, + {"UIntPtr", "nuint"}, {"Void", "void"}, + }; + for (size_t i = 0; i < sizeof(map) / sizeof(map[0]); i++) { + if (strlen(map[i].bcl) == len && memcmp(map[i].bcl, base, len) == 0) { + return map[i].kw; + } + } + return NULL; +} + +size_t cbm_doclink_cs_norm_type(const char *in, size_t len, char *out, size_t cap) { + if (!out || cap == 0) { + return 0; + } + out[0] = '\0'; + char work[CS_NORM_WORK]; + if (!in || len == 0 || len >= sizeof(work)) { + snprintf(out, cap, "?"); + return strlen(out); + } + memcpy(work, in, len); + work[len] = '\0'; + cs_str_t t = {work, len}; + cs_drop_attributes(&t); + cs_unwrap_nullable(&t); + cs_drop_modifiers(&t); + cs_drop_backtick_arity(&t); + cs_drop_type_args(&t); + /* "..." -> "[]" (varargs spelling) */ + for (size_t i = 0; i + 2 < t.len; i++) { + if (t.s[i] == '.' && t.s[i + 1] == '.' && t.s[i + 2] == '.') { + t.s[i] = '['; + t.s[i + 1] = ']'; + cs_cut(&t, i + 2, i + 3); + } + } + while (t.len > 0 && t.s[t.len - SKIP_ONE] == '@') { + t.s[--t.len] = '\0'; + } + cs_trim(&t); + /* A trailing identifier after whitespace is the parameter name. */ + size_t last_ws = 0; + bool have_ws = false; + for (size_t i = 0; i < t.len; i++) { + if (isspace((unsigned char)t.s[i])) { + last_ws = i; + have_ws = true; + } + } + if (have_ws) { + size_t s0 = last_ws + SKIP_ONE; + bool ident = s0 < t.len && cs_ident_start(t.s[s0]); + for (size_t i = s0 + SKIP_ONE; ident && i < t.len; i++) { + ident = cs_ident_char(t.s[i]); + } + if (ident) { + t.s[last_ws] = '\0'; + t.len = last_ws; + } + } + /* Drop all whitespace and trailing ?/! markers. */ + size_t w = 0; + for (size_t i = 0; i < t.len; i++) { + if (!isspace((unsigned char)t.s[i])) { + t.s[w++] = t.s[i]; + } + } + t.s[w] = '\0'; + t.len = w; + while (t.len > 0 && (t.s[t.len - SKIP_ONE] == '?' || t.s[t.len - SKIP_ONE] == '!')) { + t.s[--t.len] = '\0'; + } + /* base, then array ([] [,]) and pointer (*) suffixes */ + size_t suf = t.len; + while (suf > 0 && t.s[suf - SKIP_ONE] == '*') { + suf--; + } + for (;;) { + if (suf > 0 && t.s[suf - SKIP_ONE] == ']') { + size_t k = suf - SKIP_ONE; + while (k > 0 && t.s[k - SKIP_ONE] == ',') { + k--; + } + if (k > 0 && t.s[k - SKIP_ONE] == '[') { + suf = k - SKIP_ONE; + continue; + } + } + break; + } + size_t base_end = suf; + while (base_end > 0 && t.s[base_end - SKIP_ONE] == '?') { + base_end--; + } + size_t base_start = 0; + for (size_t i = 0; i < base_end; i++) { + if (t.s[i] == '.') { + base_start = i + SKIP_ONE; + } else if (t.s[i] == ':' && i + SKIP_ONE < base_end && t.s[i + SKIP_ONE] == ':') { + base_start = i + PAIR_LEN; + } + } + if (base_start < base_end && t.s[base_start] == '@') { + base_start++; + } + size_t blen = base_end > base_start ? base_end - base_start : 0; + if (blen == 0) { + snprintf(out, cap, "?"); + return strlen(out); + } + const char *kw = cs_bcl_alias(t.s + base_start, blen); + int n = kw ? snprintf(out, cap, "%s%.*s", kw, (int)(t.len - suf), t.s + suf) + : snprintf(out, cap, "%.*s%.*s", (int)blen, t.s + base_start, (int)(t.len - suf), + t.s + suf); + if (n < 0 || (size_t)n >= cap) { + /* it does not fit: an unknown type, never a cut one (two long names + * cut to one prefix would compare equal) */ + snprintf(out, cap, "?"); + } + return strlen(out); +} + +/* ── Scope scan ──────────────────────────────────────────────────── */ + +/* An entry of the brace list: a structural brace, or the nesting depth a + * preprocessor branch starts from. */ +typedef struct { + uint32_t pos; /* byte offset */ + uint32_t row; /* 0-based */ + int match; /* index of the paired brace, CBM_NOT_FOUND when unpaired */ + int alias; /* the first branch's brace this one stands in for, or CBM_NOT_FOUND */ + int depth; /* nesting depth just after this entry */ + char kind; /* '{', '}', or '#' for a branch mark */ +} cs_brace_t; + +/* A declaration keyword (namespace, class, struct ...). */ +typedef struct { + uint32_t end; /* first byte after the keyword */ + uint32_t row; /* 0-based */ + char kind; /* N namespace; c s i e r as for types */ + bool partial; /* the word before it is `partial` */ +} cs_head_t; + +/* An open brace while the braces are paired. The open braces form a stack + * that is never copied: every open brace is one entry that names the one + * below it, so "the stack as it stood at the #if" is a single index, however + * deep the nesting, and going back to it costs nothing. */ +typedef struct { + int brace; /* its entry in the brace list */ + int below; /* the open brace under it, or CBM_NOT_FOUND */ +} cs_open_t; + +/* An open #if while the braces are paired. */ +typedef struct { + int at_if; /* the top open brace where the #if stands (CBM_NOT_FOUND: none) */ + int n_if; + int end1; /* ... and where its first branch ended; valid once has_end1 */ + int n_end1; + int first_open; /* first open-brace entry allocated in the current branch */ + bool has_end1; /* an #else was seen */ + int mark; /* its entry in the brace list */ +} cs_pp_t; + +enum { + CS_OWNER_NONE = -1, /* not in a type */ + CS_OWNER_LEXICAL = -2, /* found in an error node: whichever type's braces hold it */ + CS_ITEM_TEXT = -3, /* frame of a declaration read from the text */ + CS_PP_IF = 1, + CS_PP_ELSE, + CS_PP_ENDIF, + CS_PP_MAX = 32, /* nested #if */ + CS_CHAR_LITERAL_MAX = 12, /* '\U0010FFFF' */ + CS_RAW_QUOTES = 3, /* """ */ + CS_LEX_MAX_NEST = 64, /* interpolated strings inside interpolation holes */ + /* How deep a file may nest what the scope records. A block past either + * limit is not placed: its types are named in Q records, what is + * documented inside has no scope (an X range). Deeper than any program; + * the limits keep a lookup's walk over the enclosing types and namespaces + * bounded. */ + CS_MAX_TYPE_NEST = 64, /* types inside types */ + CS_MAX_NS_SEGMENTS = 64, /* segments of a namespace's full name */ +}; + +/* One declaration of the file. `node` is the declaration (a field's + * declaration for each of its declarators); a declaration read from the + * text, where the tree has no node for it, has none. */ +typedef struct { + TSNode node; + TSNode name; + TSNode tparams; + TSNode params; + const char *text_name; /* text declarations; M: an operator's or indexer's name */ + const char *text_tparams; /* text declarations: "T,U" */ + uint32_t start; /* byte offset: document order */ + uint32_t line; /* 1-based */ + int brace; /* text declarations: the brace opening the block, or CBM_NOT_FOUND */ + int tree_idx; /* position among the tree's items; CS_ITEM_TEXT for a text one */ + int owner; /* M: tree_idx of the type the tree nests it in, or CS_OWNER_* */ + char tag; /* N namespace block, F file-scoped namespace, U using, T type, + M member */ + char kind; /* T: c s i e r t d; M: c v p e o x */ + bool explicit_impl; + bool is_static; /* M: what `using static` brings in */ + bool partial; /* T: declared `partial` */ + bool broken; /* T: a parse error sits among its members */ + bool from_text; /* read from the text, not from a declaration node */ +} cs_item_t; + +typedef struct { + CBMExtractCtx *ctx; + CBMArena *tmp; /* names, paths, items, tokens: nothing of it outlives the scan */ + cs_sb_t sb; + cs_item_t *items; + int nitems; + int cap_items; + cs_brace_t *braces; + int nbraces; + int cap_braces; + cs_head_t *heads; + int nheads; + int cap_heads; + cs_open_t *open; /* brace pairing: every brace that was ever open */ + int nopen; + int cap_open; + int top; /* the innermost open brace (index into open), or CBM_NOT_FOUND */ + int sp; /* how many are open */ + cs_pp_t pp[CS_PP_MAX]; + int npp; + uint32_t row_pos; /* row cursor: the row of byte row_pos is `row` */ + uint32_t row; + uint64_t cost_steps; /* text positions and brace-stack entries visited */ + uint64_t cost_bytes; /* bytes taken from the scratch arena */ + bool failed; /* out of memory */ + bool lexical; /* the tree has parse errors: nesting is read from the braces */ + uint32_t untrusted; /* byte offset from which the braces do not pair up */ + uint32_t untrusted_row; /* its 0-based row */ + int next_region; + int types_out; /* T records written: a type's ordinal is its position among them */ + uint32_t root_end_byte; + uint32_t root_end_line; +} cs_scan_t; + +enum { + CS_ITEMS_INIT = 256, + CS_BRACES_INIT = 1024, + CS_HEADS_INIT = 64, + CS_HEADER_SCAN_MAX = 4096, /* bytes from a type's name to its `{` or `;` */ + CS_TPARAMS_SCAN_MAX = 1024, +}; + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +static _Atomic uint64_t cs_cost_text_steps; +static _Atomic uint64_t cs_cost_scratch_bytes; +static _Atomic uint64_t cs_cost_scope_bytes; + +void cbm_doclink_cs_test_cost_reset(void) { + atomic_store(&cs_cost_text_steps, 0); + atomic_store(&cs_cost_scratch_bytes, 0); + atomic_store(&cs_cost_scope_bytes, 0); +} + +void cbm_doclink_cs_test_cost(uint64_t *text_steps, uint64_t *scratch_bytes) { + *text_steps = atomic_load(&cs_cost_text_steps); + *scratch_bytes = atomic_load(&cs_cost_scratch_bytes); +} + +uint64_t cbm_doclink_cs_test_scope_bytes(void) { + return atomic_load(&cs_cost_scope_bytes); +} +#endif + +/* Hand a cost to the test seam (nothing in a product build). */ +static void cs_cost_add(uint64_t steps, uint64_t bytes) { +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + atomic_fetch_add(&cs_cost_text_steps, steps); + atomic_fetch_add(&cs_cost_scratch_bytes, bytes); +#else + (void)steps; + (void)bytes; +#endif +} + +/* Hand a finished scan's cost to the test seam. */ +static void cs_cost_publish(const cs_scan_t *s) { + cs_cost_add(s->cost_steps, s->cost_bytes); +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (!s->failed && !s->sb.failed && s->sb.buf) { + atomic_fetch_add(&cs_cost_scope_bytes, s->sb.len); + } +#endif +} + +static bool cs_kind_is(TSNode n, const char *kind) { + return strcmp(ts_node_type(n), kind) == 0; +} + +static TSNode cs_field(TSNode n, const char *field) { + return ts_node_child_by_field_name(n, field, (uint32_t)strlen(field)); +} + +/* The children of a node, in order. ts_node_child(n, i) walks from the first + * child on every call, so a loop over i costs the square of the child count + * (a parameter list, a base list and a class body are as long as the file + * makes them); a cursor steps from one child to the next. */ +typedef struct { + TSTreeCursor cur; + bool started; + bool done; +} cs_kids_t; + +static cs_kids_t cs_kids(TSNode parent) { + return (cs_kids_t){.cur = ts_tree_cursor_new(parent)}; +} + +/* The next child, named or not; false after the last one. */ +static bool cs_kids_next(cs_kids_t *k, TSNode *out) { + if (k->done) { + return false; + } + bool moved = k->started ? ts_tree_cursor_goto_next_sibling(&k->cur) + : ts_tree_cursor_goto_first_child(&k->cur); + k->started = true; + if (!moved) { + k->done = true; + return false; + } + *out = ts_tree_cursor_current_node(&k->cur); + return true; +} + +/* The next NAMED child; false after the last one. */ +static bool cs_kids_next_named(cs_kids_t *k, TSNode *out) { + while (cs_kids_next(k, out)) { + if (ts_node_is_named(*out)) { + return true; + } + } + return false; +} + +static void cs_kids_end(cs_kids_t *k) { + ts_tree_cursor_delete(&k->cur); +} + +/* The first named child of `n` of kind `kind`; a null node when it has none. */ +static TSNode cs_child_of_kind(TSNode n, const char *kind) { + TSNode found = {0}; + cs_kids_t k = cs_kids(n); + TSNode c; + while (cs_kids_next_named(&k, &c)) { + if (strcmp(ts_node_type(c), kind) == 0) { + found = c; + break; + } + } + cs_kids_end(&k); + return found; +} + +/* A declaration's type_parameter_list: a named field in some grammar + * versions, an unnamed child in others. */ +static TSNode cs_type_params(TSNode decl) { + TSNode tp = cs_field(decl, "type_parameters"); + if (!ts_node_is_null(tp)) { + return tp; + } + return cs_child_of_kind(decl, "type_parameter_list"); +} + +/* Node text without whitespace and without verbatim '@' markers, appended to + * sb. `global::` prefixes are dropped. Text that is too long to be a type + * name, that holds a scope separator or non-whitespace control byte, or that + * is not well-formed UTF-8, is written as "?" (an unresolvable name: the + * declaring type then counts as having an open hierarchy). The whole field is + * replaced before any copy. */ +static void cs_put_text_nows(cs_scan_t *s, TSNode n) { + const char *src = s->ctx->source; + uint32_t a = ts_node_start_byte(n); + uint32_t b = ts_node_end_byte(n); + if (b - a > 8 && memcmp(src + a, "global::", 8) == 0) { + a += 8; + } + bool ok = b >= a && b - a <= CS_NAME_MAX; + for (uint32_t i = a; ok && i < b; i++) { + unsigned char c = (unsigned char)src[i]; + ok = c != '|' && c != ';' && c != '{' && c != '}' && + !((c < 0x20 && !isspace(c)) || c == 0x7f); + } + if (!ok || !cs_utf8_ok(src + a, b - a)) { + sb_putc(&s->sb, '?'); + return; + } + for (uint32_t i = a; i < b; i++) { + char c = src[i]; + if (isspace((unsigned char)c) || c == '@') { + continue; + } + sb_putc(&s->sb, c); + } +} + +static void *cs_tmp_alloc(cs_scan_t *s, size_t n) { + s->cost_bytes += n; + void *p = cbm_arena_alloc(s->tmp, n ? n : SKIP_ONE); + if (!p) { + s->failed = true; + } + return p; +} + +/* Copy of source bytes [a, b) without whitespace and '@'. NULL unless it is a + * (dotted) identifier of sane length and well-formed UTF-8: an error-recovered + * parse can hand back a "name" spanning arbitrary code, which must not become + * a declaration, and a name that is not UTF-8 cannot be written. */ +static char *cs_ident_dup(cs_scan_t *s, uint32_t a, uint32_t b) { + const char *src = s->ctx->source; + if (b <= a || b - a > CS_NAME_MAX) { + return NULL; + } + char *out = (char *)cs_tmp_alloc(s, (size_t)(b - a) + SKIP_ONE); + if (!out) { + return NULL; + } + size_t w = 0; + for (uint32_t i = a; i < b; i++) { + unsigned char c = (unsigned char)src[i]; + if (isspace(c) || c == '@') { + continue; + } + if (!(isalnum(c) || c == '_' || c == '.' || c >= 0x80)) { + return NULL; /* not an identifier */ + } + out[w++] = (char)c; + } + out[w] = '\0'; + return w > 0 && cs_utf8_ok(out, w) ? out : NULL; +} + +static char *cs_name_dup(cs_scan_t *s, TSNode n) { + if (ts_node_is_null(n)) { + return NULL; + } + return cs_ident_dup(s, ts_node_start_byte(n), ts_node_end_byte(n)); +} + +static uint32_t cs_line(TSNode n) { + return ts_node_start_point(n).row + TS_LINE_OFFSET; +} + +static uint32_t cs_end_line(TSNode n) { + return ts_node_end_point(n).row + TS_LINE_OFFSET; +} + +/* type_parameter_list -> "T,U" into sb (nothing when absent). */ +static void cs_put_tparams(cs_scan_t *s, TSNode list) { + if (ts_node_is_null(list)) { + return; + } + bool first = true; + cs_kids_t k = cs_kids(list); + TSNode tp; + while (cs_kids_next_named(&k, &tp)) { + if (!cs_kind_is(tp, "type_parameter")) { + continue; + } + TSNode nm = cs_field(tp, "name"); + if (ts_node_is_null(nm)) { + continue; + } + if (!first) { + sb_putc(&s->sb, ','); + } + cs_put_text_nows(s, nm); + first = false; + } + cs_kids_end(&k); +} + +static void cs_put_sig(cs_scan_t *s, TSNode params) { + if (ts_node_is_null(params)) { + return; + } + const char *src = s->ctx->source; + bool first = true; + cs_kids_t k = cs_kids(params); + TSNode p; + while (cs_kids_next(&k, &p)) { + /* A parameter is a `parameter` node -- except the `params` one, which + * the grammar leaves inline in the list: its type is the list's own + * "type" field. */ + TSNode ty = {0}; + if (cs_kind_is(p, "parameter")) { + ty = cs_field(p, "type"); + } else { + const char *field = ts_tree_cursor_current_field_name(&k.cur); + if (!field || strcmp(field, "type") != 0) { + continue; + } + ty = p; + } + char norm[CBM_SZ_256]; + if (ts_node_is_null(ty)) { + snprintf(norm, sizeof(norm), "?"); + } else { + uint32_t a = ts_node_start_byte(ty); + uint32_t b = ts_node_end_byte(ty); + cbm_doclink_cs_norm_type(src + a, (size_t)(b - a), norm, sizeof(norm)); + if (!cs_utf8_ok(norm, strlen(norm))) { + snprintf(norm, sizeof(norm), "?"); /* a type nothing is known about */ + } + } + if (!first) { + sb_putc(&s->sb, '|'); + } + sb_puts(&s->sb, norm); + first = false; + } + cs_kids_end(&k); +} + +static bool cs_has_child_kind(TSNode n, const char *kind) { + return !ts_node_is_null(cs_child_of_kind(n, kind)); +} + +static char cs_type_kind(const char *k) { + if (strcmp(k, "class_declaration") == 0) { + return 'c'; + } + if (strcmp(k, "struct_declaration") == 0) { + return 's'; + } + if (strcmp(k, "interface_declaration") == 0) { + return 'i'; + } + if (strcmp(k, "enum_declaration") == 0) { + return 'e'; + } + if (strcmp(k, "record_declaration") == 0) { + return 'r'; /* a record class; cs_decl_kind tells a `record struct` apart */ + } + if (strcmp(k, "record_struct_declaration") == 0) { + return 't'; + } + if (strcmp(k, "delegate_declaration") == 0) { + return 'd'; + } + return 0; +} + +/* True when the anonymous token `word` is a direct child of `n`, or sits in + * one of its `modifier` children (`partial`, `static`). */ +static bool cs_has_word(TSNode n, const char *word) { + bool found = false; + uint64_t steps = 0; + cs_kids_t k = cs_kids(n); + TSNode c; + while (!found && cs_kids_next(&k, &c)) { + steps++; + const char *t = ts_node_type(c); + if (!ts_node_is_named(c)) { + found = strcmp(t, word) == 0; + } else if (strcmp(t, "modifier") == 0 && ts_node_child_count(c) > 0) { + found = strcmp(ts_node_type(ts_node_child(c, 0)), word) == 0; + } + } + cs_kids_end(&k); + cs_cost_add(steps, 0); + return found; +} + +/* The kind of a type declaration node: c class, s struct, i interface, e + * enum, r record class, t record struct, d delegate; 0 for any other node. */ +static char cs_decl_kind(TSNode decl) { + char kind = cs_type_kind(ts_node_type(decl)); + return (kind == 'r' && cs_has_word(decl, "struct")) ? 't' : kind; +} + +/* ── Tokens: the block structure the tree lost ─────────────────────── + * + * Error recovery closes blocks early and late, reports a block namespace as + * a file-scoped one, or gives up on a declaration and leaves its header as + * loose tokens in an error node. On the C# bench corpus 1,805 of 32,686 files + * have parse errors, and in 230 of them a namespace or type has a tree extent + * that is not its brace extent (the API reference files among them) -- which + * moves every declaration after the error into the wrong namespace or out of + * its outer type. A file whose tree has errors therefore takes its nesting + * from the text instead: the braces say where blocks begin and end, the + * declaration keywords say what the blocks are. The parser's own tokens + * cannot serve: once it loses the thread inside a string it lexes code as + * string content. So the braces are scanned here, with the lexical grammar a + * brace scanner needs -- comments, character and string literals in all + * their forms, interpolation holes, preprocessor branches. On the 30,881 + * error-free corpus files this scanner yields exactly the parser's brace + * tokens. Where the braces still do not pair up, the rest of the file is not + * placed at all. */ + +static uint32_t cs_row_of(cs_scan_t *s, uint32_t pos) { + const char *src = s->ctx->source; + if (pos < s->row_pos) { + s->row_pos = 0; + s->row = 0; + } + for (uint32_t i = s->row_pos; i < pos; i++) { + s->row += src[i] == '\n'; + } + s->row_pos = pos; + return s->row; +} + +/* A new entry of the brace list; its index, or CBM_NOT_FOUND. */ +static int cs_brace_push(cs_scan_t *s, uint32_t pos, char kind) { + if (s->failed) { + return CBM_NOT_FOUND; + } + if (s->nbraces >= s->cap_braces) { + int ncap = s->cap_braces ? s->cap_braces * PAIR_LEN : CS_BRACES_INIT; + cs_brace_t *grown = (cs_brace_t *)cs_tmp_alloc(s, (size_t)ncap * sizeof(*grown)); + if (!grown) { + return CBM_NOT_FOUND; + } + if (s->nbraces > 0) { + memcpy(grown, s->braces, (size_t)s->nbraces * sizeof(*grown)); + } + s->braces = grown; + s->cap_braces = ncap; + } + s->braces[s->nbraces] = (cs_brace_t){.pos = pos, + .row = cs_row_of(s, pos), + .match = CBM_NOT_FOUND, + .alias = CBM_NOT_FOUND, + .kind = kind}; + return s->nbraces++; +} + +/* Brace `b` is open now: it goes on top of the open braces. */ +static bool cs_open_push(cs_scan_t *s, int b) { + if (s->nopen >= s->cap_open) { + int ncap = s->cap_open ? s->cap_open * PAIR_LEN : CS_HEADS_INIT; + cs_open_t *grown = (cs_open_t *)cs_tmp_alloc(s, (size_t)ncap * sizeof(*grown)); + if (!grown) { + return false; + } + if (s->nopen > 0) { + memcpy(grown, s->open, (size_t)s->nopen * sizeof(*grown)); + } + s->open = grown; + s->cap_open = ncap; + } + s->open[s->nopen] = (cs_open_t){.brace = b, .below = s->top}; + s->top = s->nopen++; + s->sp++; + return true; +} + +static void cs_mark_untrusted_at(cs_scan_t *s, uint32_t pos, uint32_t row) { + if (pos < s->untrusted) { + s->untrusted = pos; + s->untrusted_row = row; + } +} + +static void cs_mark_untrusted(cs_scan_t *s, int brace) { + cs_mark_untrusted_at(s, s->braces[brace].pos, s->braces[brace].row); +} + +/* The scanner reports a brace: pair it. */ +static void cs_lex_brace(void *ud, uint32_t pos, bool open) { + cs_scan_t *s = (cs_scan_t *)ud; + int b = cs_brace_push(s, pos, open ? '{' : '}'); + if (b < 0) { + return; + } + if (open) { + (void)cs_open_push(s, b); + } else if (s->top < 0) { + cs_mark_untrusted(s, b); /* closes nothing */ + } else { + int o = s->open[s->top].brace; + s->top = s->open[s->top].below; + s->sp--; + if (s->braces[o].match < 0) { + s->braces[o].match = b; /* the first branch's close stands */ + } + s->braces[b].match = o; + } + s->braces[b].depth = s->sp; +} + +/* A later branch of a conditional ended. Where it leaves the same blocks + * open as the first one did, its open braces stand in for the first + * branch's (`class X : A {` / `#else` / `class X : B {` share one closing + * brace). Where the branches disagree the first one stands alone: a file + * can balance per configuration only (`#if A {` ... `#if A }`), and the + * braces left over at the end say whether this one does. + * + * Only entries opened in the current branch may acquire aliases. The first + * branch can replace a deep pre-existing stack; every empty later branch + * starts from that old stack. Walking it again would cost depth per branch. */ +static void cs_branch_merge(cs_scan_t *s, const cs_pp_t *f) { + if (s->sp != f->n_end1) { + return; + } + int a = s->top; + int b = f->end1; + while (a != b && a >= f->first_open && b >= 0) { + s->cost_steps++; + s->braces[s->open[a].brace].alias = s->open[b].brace; + a = s->open[a].below; + b = s->open[b].below; + } +} + +/* The scanner reports #if / #else (or #elif) / #endif. Every branch starts + * from the nesting the #if started from; after the #endif the first + * branch's result stands. */ +static void cs_lex_branch(void *ud, uint32_t pos, int what) { + cs_scan_t *s = (cs_scan_t *)ud; + int mark = cs_brace_push(s, pos, '#'); + if (mark < 0) { + return; + } + if (what == CS_PP_IF) { + if (s->npp >= CS_PP_MAX) { + cs_mark_untrusted(s, mark); + } else { + s->pp[s->npp++] = + (cs_pp_t){.at_if = s->top, .n_if = s->sp, .first_open = s->nopen, .mark = mark}; + } + } else if (s->npp > 0) { + cs_pp_t *f = &s->pp[s->npp - SKIP_ONE]; + if (f->has_end1) { + cs_branch_merge(s, f); + } else if (what == CS_PP_ELSE) { + f->end1 = s->top; + f->n_end1 = s->sp; + f->has_end1 = true; + } + if (what == CS_PP_ELSE) { + s->top = f->at_if; + s->sp = f->n_if; + f->first_open = s->nopen; + } else if (what == CS_PP_ENDIF) { + if (f->has_end1) { + s->top = f->end1; + s->sp = f->n_end1; + } + s->npp--; + } + } + s->braces[mark].depth = s->sp; +} + +/* The scanner reports a declaration keyword; `partial` when the word before + * it is that modifier. */ +static void cs_lex_keyword(void *ud, uint32_t kw_end, char kind, bool partial) { + cs_scan_t *s = (cs_scan_t *)ud; + if (s->failed) { + return; + } + if (s->nheads >= s->cap_heads) { + int ncap = s->cap_heads ? s->cap_heads * PAIR_LEN : CS_HEADS_INIT; + cs_head_t *grown = (cs_head_t *)cs_tmp_alloc(s, (size_t)ncap * sizeof(*grown)); + if (!grown) { + return; + } + if (s->nheads > 0) { + memcpy(grown, s->heads, (size_t)s->nheads * sizeof(*grown)); + } + s->heads = grown; + s->cap_heads = ncap; + } + s->heads[s->nheads++] = + (cs_head_t){.end = kw_end, .row = cs_row_of(s, kw_end), .kind = kind, .partial = partial}; +} + +/* --- the scanner ---------------------------------------------------- */ + +typedef struct { + const char *src; + uint32_t n; + void *ud; + uint64_t steps; /* positions looked at */ + int nest; /* interpolation holes the scan is inside of */ + uint32_t stop_pos; /* where the scan gave up (`stopped`) */ + bool stopped; /* holes nested deeper than CS_LEX_MAX_NEST: the rest is not read */ + bool prev_sep; /* the previous token was ':' or ',' */ + bool prev_partial; /* the previous token was `partial` (or `partial record`) */ +} cs_lex_t; + +static bool cs_lex_word(unsigned char c) { + return isalnum(c) || c == '_' || c >= 0x80; +} + +/* The declaration a keyword starts: N for `namespace`, a type kind, or 0. */ +static char cs_keyword_kind(const char *text, uint32_t len) { + static const struct { + const char *word; + char kind; + } words[] = {{"namespace", 'N'}, {"class", 'c'}, {"struct", 's'}, + {"interface", 'i'}, {"enum", 'e'}, {"record", 'r'}}; + for (size_t i = 0; i < sizeof(words) / sizeof(words[0]); i++) { + if (strlen(words[i].word) == len && memcmp(words[i].word, text, len) == 0) { + return words[i].kind; + } + } + return 0; +} + +static uint32_t cs_lex_code(cs_lex_t *lx, uint32_t i, bool hole); + +/* The code of an interpolation hole from src[i]: just past the brace that + * closes it. A hole can hold the next interpolated string, and each such + * level costs stack: past CS_LEX_MAX_NEST -- deeper than any program -- the + * scan stops for good, and what follows in the file is not placed. */ +static uint32_t cs_lex_hole(cs_lex_t *lx, uint32_t i) { + if (lx->nest >= CS_LEX_MAX_NEST) { + if (!lx->stopped) { + lx->stopped = true; + lx->stop_pos = i; + } + return lx->n; + } + lx->nest++; + uint32_t end = cs_lex_code(lx, i, true); + lx->nest--; + return end; +} + +/* Length of the run of `c` at src[i]. */ +static uint32_t cs_lex_run(cs_lex_t *lx, uint32_t i, char c) { + uint32_t j = i; + while (j < lx->n && lx->src[j] == c) { + j++; + } + lx->steps += (uint64_t)(j - i) + SKIP_ONE; + return j - i; +} + +/* 'x', '\n', 'A': past the literal, or past the lone quote when it is + * not one. */ +static uint32_t cs_lex_char(const cs_lex_t *lx, uint32_t i) { + uint32_t j = i + SKIP_ONE; + if (j < lx->n && lx->src[j] == '\\') { + j += PAIR_LEN; + } + while (j < lx->n && lx->src[j] != '\'' && lx->src[j] != '\n' && j - i < CS_CHAR_LITERAL_MAX) { + j++; + } + return (j < lx->n && lx->src[j] == '\'') ? j + SKIP_ONE : i + SKIP_ONE; +} + +/* "..." with backslash escapes; an unterminated one ends with its line. */ +static uint32_t cs_lex_string(const cs_lex_t *lx, uint32_t i) { + uint32_t j = i + SKIP_ONE; + while (j < lx->n) { + char c = lx->src[j]; + if (c == '\\') { + j += PAIR_LEN; + } else if (c == '"') { + return j + SKIP_ONE; + } else if (c == '\n') { + return j; + } else { + j++; + } + } + return lx->n; +} + +/* @"..." : a quote is written twice. `i` is at the opening quote. */ +static uint32_t cs_lex_verbatim(const cs_lex_t *lx, uint32_t i) { + uint32_t j = i + SKIP_ONE; + while (j < lx->n) { + if (lx->src[j] != '"') { + j++; + } else if (j + SKIP_ONE < lx->n && lx->src[j + SKIP_ONE] == '"') { + j += PAIR_LEN; + } else { + return j + SKIP_ONE; + } + } + return lx->n; +} + +/* """...""" (`quotes` >= 3 of them): ends with a run of at least as many. + * With `dollars`, a run of that many `{` opens a hole of code. */ +static uint32_t cs_lex_raw(cs_lex_t *lx, uint32_t i, uint32_t quotes, uint32_t dollars) { + uint32_t j = i + quotes; + while (j < lx->n) { + char c = lx->src[j]; + if (c == '"') { + uint32_t run = cs_lex_run(lx, j, '"'); + if (run >= quotes) { + return j + run; + } + j += run; + } else if (dollars > 0 && c == '{') { + uint32_t run = cs_lex_run(lx, j, '{'); + j += run; + if (run >= dollars) { + j = cs_lex_hole(lx, j); + j += cs_lex_run(lx, j, '}'); /* the rest of the closing run */ + } + } else { + j++; + } + } + return lx->n; +} + +/* $"..." / $@"..." : text with {holes} of code; {{ and }} are literal braces. + * `i` is at the opening quote. */ +static uint32_t cs_lex_interpolated(cs_lex_t *lx, uint32_t i, bool verbatim) { + uint32_t j = i + SKIP_ONE; + while (j < lx->n) { + char c = lx->src[j]; + if (c == '"') { + if (verbatim && j + SKIP_ONE < lx->n && lx->src[j + SKIP_ONE] == '"') { + j += PAIR_LEN; + continue; + } + return j + SKIP_ONE; + } + if (c == '\\' && !verbatim) { + j += PAIR_LEN; + } else if (c == '{' || c == '}') { + if (j + SKIP_ONE < lx->n && lx->src[j + SKIP_ONE] == c) { + j += PAIR_LEN; + } else if (c == '{') { + j = cs_lex_hole(lx, j + SKIP_ONE); + } else { + j++; + } + } else if (c == '\n' && !verbatim) { + return j; + } else { + j++; + } + } + return lx->n; +} + +/* A literal that starts with a quote, `@`, or `$` at src[i]; returns i itself + * when there is none there. A run of `$` is measured once: where it starts + * no literal the scan goes on behind it (or at its last `$`, when that one + * starts an ordinary interpolated string), never at its second character. */ +static uint32_t cs_lex_literal(cs_lex_t *lx, uint32_t i) { + const char *src = lx->src; + uint32_t n = lx->n; + char c = src[i]; + if (c == '"') { + uint32_t q = cs_lex_run(lx, i, '"'); + if (q >= CS_RAW_QUOTES) { + return cs_lex_raw(lx, i, q, 0); + } + return q == PAIR_LEN ? i + PAIR_LEN : cs_lex_string(lx, i); + } + if (c == '@' && i + SKIP_ONE < n && src[i + SKIP_ONE] == '"') { + return cs_lex_verbatim(lx, i + SKIP_ONE); + } + if (c == '@' && i + PAIR_LEN < n && src[i + SKIP_ONE] == '$' && src[i + PAIR_LEN] == '"') { + return cs_lex_interpolated(lx, i + PAIR_LEN, true); + } + if (c == '$') { + uint32_t d = cs_lex_run(lx, i, '$'); + uint32_t j = i + d; + if (j < n && src[j] == '"') { + uint32_t q = cs_lex_run(lx, j, '"'); + if (q >= CS_RAW_QUOTES) { + return cs_lex_raw(lx, j, q, d); + } + return (d == SKIP_ONE) ? cs_lex_interpolated(lx, j, false) : j - SKIP_ONE; + } + if (j + SKIP_ONE < n && src[j] == '@' && src[j + SKIP_ONE] == '"') { + return (d == SKIP_ONE) ? cs_lex_interpolated(lx, j + SKIP_ONE, true) : j - SKIP_ONE; + } + return j; + } + return i; +} + +static bool cs_lex_is(const char *src, uint32_t w, uint32_t len, const char *word) { + return strlen(word) == len && memcmp(src + w, word, len) == 0; +} + +/* A `#` directive at the start of a line: reports conditional branches and + * returns the end of the line. */ +static uint32_t cs_lex_directive(cs_lex_t *lx, uint32_t i, bool hole) { + const char *src = lx->src; + uint32_t j = i + SKIP_ONE; + while (j < lx->n && (src[j] == ' ' || src[j] == '\t')) { + j++; + } + uint32_t w = j; + while (j < lx->n && isalpha((unsigned char)src[j])) { + j++; + } + uint32_t len = j - w; + if (!hole) { + if (cs_lex_is(src, w, len, "if")) { + cs_lex_branch(lx->ud, i, CS_PP_IF); + } else if (cs_lex_is(src, w, len, "else") || cs_lex_is(src, w, len, "elif")) { + cs_lex_branch(lx->ud, i, CS_PP_ELSE); + } else if (cs_lex_is(src, w, len, "endif")) { + cs_lex_branch(lx->ud, i, CS_PP_ENDIF); + } + } + while (j < lx->n && src[j] != '\n') { + j++; + } + return j; +} + +/* A word at src[i] (an identifier, keyword or number, with an optional `@`): + * reports a declaration keyword and returns its end; i when there is none. + * *partial is set when the word leaves the `partial` modifier standing for + * the next keyword: `partial` itself, or `record` after it (`partial record + * struct`). */ +static uint32_t cs_lex_identifier(cs_lex_t *lx, uint32_t i, bool hole, bool *partial) { + static const char modifier[] = "partial"; + const char *src = lx->src; + bool verbatim = src[i] == '@'; + uint32_t a = verbatim ? i + SKIP_ONE : i; + uint32_t b = a; + while (b < lx->n && cs_lex_word((unsigned char)src[b])) { + b++; + } + if (b == a) { + return i; + } + char kind = (verbatim || hole) ? 0 : cs_keyword_kind(src + a, b - a); + if (kind && !lx->prev_sep) { + cs_lex_keyword(lx->ud, b, kind, lx->prev_partial); + } + *partial = !verbatim && !hole && + ((kind == 'r' && lx->prev_partial) || + (b - a == sizeof(modifier) - SKIP_ONE && memcmp(src + a, modifier, b - a) == 0)); + return b; +} + +/* Past a comment at src[i], or i when there is none. */ +static uint32_t cs_lex_comment(const cs_lex_t *lx, uint32_t i) { + const char *src = lx->src; + uint32_t n = lx->n; + if (src[i] != '/' || i + SKIP_ONE >= n) { + return i; + } + if (src[i + SKIP_ONE] == '/') { + while (i < n && src[i] != '\n') { + i++; + } + return i; + } + if (src[i + SKIP_ONE] == '*') { + i += PAIR_LEN; + while (i + SKIP_ONE < n && !(src[i] == '*' && src[i + SKIP_ONE] == '/')) { + i++; + } + return i + PAIR_LEN <= n ? i + PAIR_LEN : n; + } + return i; +} + +/* Scan code from src[i]. In a `hole` (the code of an interpolation) nothing + * is reported and the scan returns just past the brace that closes it. */ +static uint32_t cs_lex_code(cs_lex_t *lx, uint32_t i, bool hole) { + const char *src = lx->src; + uint32_t n = lx->n; + int depth = 0; + bool line_start = !hole; + while (i < n) { + unsigned char c = (unsigned char)src[i]; + lx->steps++; + if (isspace(c)) { + line_start = line_start || c == '\n'; + i++; + continue; + } + if (c == '#' && line_start) { + i = cs_lex_directive(lx, i, hole); + continue; + } + line_start = false; + uint32_t e = cs_lex_comment(lx, i); + if (e > i) { + i = e; + continue; + } + bool sep = c == ':' || c == ','; + bool partial = false; + if (c == '\'') { + e = cs_lex_char(lx, i); + } else if (c == '{' || c == '}') { + if (hole && c == '}' && depth == 0) { + return i + SKIP_ONE; + } + if (hole) { + depth += c == '{' ? SKIP_ONE : -SKIP_ONE; + } else { + cs_lex_brace(lx->ud, i, c == '{'); + } + e = i + SKIP_ONE; + } else { + e = (c == '"' || c == '$' || c == '@') ? cs_lex_literal(lx, i) : i; + if (e == i && (cs_lex_word(c) || c == '@')) { + e = cs_lex_identifier(lx, i, hole, &partial); + } + if (e == i) { + e = i + SKIP_ONE; + } + } + lx->prev_sep = sep; + lx->prev_partial = partial; + i = e; + } + return n; +} + +/* Scan the file's braces and declaration keywords, pair the braces. The + * first brace without a partner (or conditional whose branches disagree) + * starts the untrusted part of the file. */ +static void cs_scan_tokens(cs_scan_t *s) { + cs_lex_t lx = {.src = s->ctx->source, .n = s->root_end_byte, .ud = s}; + (void)cs_lex_code(&lx, 0, false); + s->cost_steps += lx.steps; + if (s->failed) { + return; + } + if (lx.stopped) { + cs_mark_untrusted_at(s, lx.stop_pos, cs_row_of(s, lx.stop_pos)); + } + /* the outermost brace that is still open */ + int unpaired = CBM_NOT_FOUND; + for (int o = s->top; o >= 0; o = s->open[o].below) { + unpaired = s->open[o].brace; + } + if (unpaired >= 0) { + cs_mark_untrusted(s, unpaired); + } + /* a later branch's brace closes where the first branch's does */ + for (int i = 0; i < s->nbraces; i++) { + cs_brace_t *b = &s->braces[i]; + int a = b->alias; + for (int hops = 0; a >= 0 && b->match < 0 && hops < CS_PP_MAX; hops++) { + b->match = s->braces[a].match; + a = s->braces[a].alias; + } + } +} + +/* Index of the first entry of the brace list at or after byte `pos`. */ +static int cs_brace_lower(const cs_scan_t *s, uint32_t pos) { + int lo = 0; + int hi = s->nbraces; + while (lo < hi) { + int mid = lo + ((hi - lo) / PAIR_LEN); + if (s->braces[mid].pos < pos) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + return lo; +} + +/* The open brace AT byte `pos`, or CBM_NOT_FOUND. */ +static int cs_open_brace_at(const cs_scan_t *s, uint32_t pos) { + int b = cs_brace_lower(s, pos); + return (b < s->nbraces && s->braces[b].pos == pos && s->braces[b].kind == '{') ? b + : CBM_NOT_FOUND; +} + +/* Brace nesting depth at byte `pos`. */ +static int cs_depth_at(const cs_scan_t *s, uint32_t pos) { + int i = cs_brace_lower(s, pos); + return i > 0 ? s->braces[i - SKIP_ONE].depth : 0; +} + +/* The extent of the block opened by brace `b`. false when it has no partner. */ +static bool cs_brace_extent(const cs_scan_t *s, int b, uint32_t *end, uint32_t *end_line, + int *inner_depth) { + if (b < 0 || b >= s->nbraces || s->braces[b].kind != '{' || s->braces[b].match < 0) { + return false; + } + const cs_brace_t *close = &s->braces[s->braces[b].match]; + *end = close->pos + SKIP_ONE; + *end_line = close->row + TS_LINE_OFFSET; + *inner_depth = s->braces[b].depth; + return true; +} + +/* ── Text ────────────────────────────────────────────────────────── */ + +/* Past whitespace and comments. */ +static uint32_t cs_skip_space(const char *src, uint32_t i, uint32_t n) { + for (;;) { + while (i < n && isspace((unsigned char)src[i])) { + i++; + } + if (i + SKIP_ONE >= n || src[i] != '/') { + return i; + } + if (src[i + SKIP_ONE] == '/') { + while (i < n && src[i] != '\n') { + i++; + } + } else if (src[i + SKIP_ONE] == '*') { + i += PAIR_LEN; + while (i + SKIP_ONE < n && !(src[i] == '*' && src[i + SKIP_ONE] == '/')) { + i++; + } + i = i + PAIR_LEN <= n ? i + PAIR_LEN : n; + } else { + return i; + } + } +} + +static bool cs_word_char(unsigned char c) { + return isalnum(c) || c == '_' || c >= 0x80; +} + +/* Recovery may omit punctuation before the name node. Start at the actual + * namespace header, not at that recovered node's first byte. */ +static uint32_t cs_namespace_name_start(const cs_scan_t *s, TSNode node) { + static const char keyword[] = "namespace"; + const uint32_t width = (uint32_t)(sizeof(keyword) - SKIP_ONE); + uint32_t a = ts_node_start_byte(node); + uint32_t n = s->root_end_byte; + const char *src = s->ctx->source; + if (a > n || n - a < width || memcmp(src + a, keyword, width) != 0) { + return n; + } + a += width; + uint32_t start = cs_skip_space(src, a, n); + return start > a || (a < n && src[a] == '@') ? start : n; +} + +/* Namespace names have nonempty identifier segments. Keep the scanner's + * existing Unicode-byte support, while allowing trivia between tokens and + * verbatim markers only at segment starts. NULL names remain unplaced items. */ +static char *cs_namespace_name_dup(cs_scan_t *s, uint32_t a, uint32_t b) { + if (b <= a || b > s->root_end_byte || b - a > CS_NAME_MAX) { + return NULL; + } + const char *src = s->ctx->source; + char *out = (char *)cs_tmp_alloc(s, (size_t)(b - a) + SKIP_ONE); + if (!out) { + return NULL; + } + size_t w = 0; + uint32_t i = a; + for (;;) { + i = cs_skip_space(src, i, b); + if (i < b && src[i] == '@') { + i++; + } + if (i == b || !cs_word_char((unsigned char)src[i]) || isdigit((unsigned char)src[i])) { + return NULL; + } + while (i < b && cs_word_char((unsigned char)src[i])) { + out[w++] = src[i++]; + } + i = cs_skip_space(src, i, b); + if (i == b) { + out[w] = '\0'; + return cs_utf8_ok(out, w) ? out : NULL; /* a name that cannot be written */ + } + if (src[i] != '.') { + return NULL; + } + out[w++] = '.'; + i++; + } +} + +/* Keep malformed dotted candidates until their known delimiter. Dropping an + * invalid file-scoped header would place its declarations in the root. The + * next recorded declaration bounds this scan even when no delimiter exists. */ +static uint32_t cs_namespace_header_end(cs_scan_t *s, uint32_t a, uint32_t stop) { + const char *src = s->ctx->source; + uint32_t i = a; + while (i < stop) { + uint32_t next = cs_skip_space(src, i, stop); + if (next != i) { + i = next; + continue; + } + unsigned char c = (unsigned char)src[i]; + if (!cs_word_char(c) && c != '@' && c != '.') { + break; + } + s->cost_steps++; + i++; + } + return i; +} + +/* End of the identifier starting at src[i] (i itself when there is none); + * `dotted` accepts a qualified name. */ +static uint32_t cs_ident_end(const char *src, uint32_t i, uint32_t n, bool dotted) { + uint32_t e = i; + if (e < n && src[e] == '@') { + e++; + } + uint32_t first = e; + while (e < n && (cs_word_char((unsigned char)src[e]) || + (dotted && src[e] == '.' && e > first && e + SKIP_ONE < n && + cs_word_char((unsigned char)src[e + SKIP_ONE])))) { + e++; + } + if (e == first || isdigit((unsigned char)src[first])) { + return i; + } + return e; +} + +/* Words that follow `class` / `record` ... without being a declared name. */ +static bool cs_not_a_name(const char *name) { + static const char *const words[] = {"class", "struct", "interface", "enum", "record", + "where", "in", "is", "as", "when", + "and", "or", "not", "with", "switch"}; + for (size_t i = 0; i < sizeof(words) / sizeof(words[0]); i++) { + if (strcmp(name, words[i]) == 0) { + return true; + } + } + return false; +} + +/* The type parameters written at src[*pos] (``): their names + * ','-joined, "" when there are none. *pos moves past the list. NULL when the + * list cannot be read. Nothing at or past `stop` is read. */ +static const char *cs_text_tparams(cs_scan_t *s, uint32_t *pos, uint32_t stop) { + const char *src = s->ctx->source; + uint32_t n = s->root_end_byte; + uint32_t i = cs_skip_space(src, *pos, n); + if (i >= n || src[i] != '<') { + return ""; + } + if (i >= stop) { + return NULL; /* the list is the next declaration's */ + } + uint32_t limit = stop - i > CS_TPARAMS_SCAN_MAX ? i + CS_TPARAMS_SCAN_MAX : stop; + char *out = (char *)cs_tmp_alloc(s, (size_t)(limit - i) + SKIP_ONE); + if (!out) { + return NULL; + } + size_t w = 0; + int square = 0; + uint32_t word_s = 0; + uint32_t word_e = 0; /* the last identifier of the current parameter */ + for (uint32_t k = i + SKIP_ONE; k < limit; k++) { + char c = src[k]; + s->cost_steps++; + if (c == '[') { + square++; + } else if (c == ']') { + square--; + } else if (square > 0) { + continue; /* an attribute on the parameter */ + } else if (c == ',' || c == '>') { + if (word_e == word_s) { + return NULL; + } + if (w > 0) { + out[w++] = ','; + } + memcpy(out + w, src + word_s, word_e - word_s); + w += word_e - word_s; + word_s = word_e = 0; + if (c == '>') { + out[w] = '\0'; + *pos = k + SKIP_ONE; + return cs_utf8_ok(out, w) ? out : NULL; /* names that cannot be written */ + } + } else if (cs_word_char((unsigned char)c)) { + uint32_t e = cs_ident_end(src, k, limit, false); + if (e == k) { + return NULL; + } + word_s = k; + word_e = e; + k = e - SKIP_ONE; + } else if (!isspace((unsigned char)c)) { + return NULL; + } + } + return NULL; +} + +/* From the end of a type's name and type parameters to the `{` that opens + * its body (returned as a brace index) or the `;` that ends a body-less + * declaration (*bodyless). CBM_NOT_FOUND with *bodyless false when neither is + * found: then this was no declaration. Nothing at or past `stop` is read. */ +static int cs_text_body(cs_scan_t *s, uint32_t from, uint32_t stop, bool *bodyless) { + const char *src = s->ctx->source; + uint32_t limit = from + CS_HEADER_SCAN_MAX < stop ? from + CS_HEADER_SCAN_MAX : stop; + int round = 0; + *bodyless = false; + for (uint32_t i = from; i < limit; i++) { + char c = src[i]; + s->cost_steps++; + if (c == '/' && i + SKIP_ONE < limit && + (src[i + SKIP_ONE] == '/' || src[i + SKIP_ONE] == '*')) { + uint32_t past = cs_skip_space(src, i, limit); + if (past <= i) { + return CBM_NOT_FOUND; + } + i = past - SKIP_ONE; + continue; + } + if (c == '(') { + round++; + } else if (c == ')') { + round--; + } else if (round > 0) { + continue; + } else if (c == ';') { + *bodyless = true; + return CBM_NOT_FOUND; + } else if (c == '{') { + return cs_open_brace_at(s, i); + } else if (c == '}' || c == '=') { + return CBM_NOT_FOUND; + } + } + return CBM_NOT_FOUND; +} + +/* ── Collection ──────────────────────────────────────────────────── */ + +/* A new item; returns its index or CBM_NOT_FOUND when memory ran out. */ +static int cs_item_new(cs_scan_t *s, char tag) { + if (s->failed) { + return CBM_NOT_FOUND; + } + if (s->nitems >= s->cap_items) { + int ncap = s->cap_items ? s->cap_items * PAIR_LEN : CS_ITEMS_INIT; + cs_item_t *grown = (cs_item_t *)cs_tmp_alloc(s, (size_t)ncap * sizeof(*grown)); + if (!grown) { + return CBM_NOT_FOUND; + } + if (s->nitems > 0) { + memcpy(grown, s->items, (size_t)s->nitems * sizeof(*grown)); + } + s->items = grown; + s->cap_items = ncap; + } + cs_item_t *it = &s->items[s->nitems]; + memset(it, 0, sizeof(*it)); + it->tag = tag; + it->owner = CS_OWNER_NONE; + it->brace = CBM_NOT_FOUND; + it->tree_idx = s->nitems; + return s->nitems++; +} + +static int cs_item_of_node(cs_scan_t *s, char tag, TSNode node) { + int i = cs_item_new(s, tag); + if (i >= 0) { + s->items[i].node = node; + s->items[i].start = ts_node_start_byte(node); + s->items[i].line = cs_line(node); + } + return i; +} + +/* True when the first parameter of `params` is written with `this`: an + * extension method. The text before the parameter's type says so, whatever + * shape the grammar gives the modifier. */ +static bool cs_first_param_this(const cs_scan_t *s, TSNode params) { + static const char word[] = "this"; + if (ts_node_is_null(params)) { + return false; + } + TSNode p = cs_child_of_kind(params, "parameter"); + if (ts_node_is_null(p)) { + return false; + } + TSNode ty = cs_field(p, "type"); + const char *src = s->ctx->source; + uint32_t a = ts_node_start_byte(p); + uint32_t b = ts_node_is_null(ty) ? ts_node_end_byte(p) : ts_node_start_byte(ty); + uint32_t wl = (uint32_t)(sizeof(word) - SKIP_ONE); + for (uint32_t i = a; i + wl <= b; i++) { + if (memcmp(src + i, word, wl) == 0 && + (i == a || !cs_word_char((unsigned char)src[i - SKIP_ONE])) && + (i + wl == b || !cs_word_char((unsigned char)src[i + wl]))) { + return true; + } + } + return false; +} + +/* True for a member `using static` brings in: one declared `static` or + * `const`, an enum's member -- and no extension method. */ +static bool cs_member_static(const cs_scan_t *s, TSNode decl, TSNode params) { + if (cs_kind_is(decl, "enum_member_declaration")) { + return true; + } + if (!cs_has_word(decl, "static") && !cs_has_word(decl, "const")) { + return false; + } + return !cs_first_param_this(s, params); +} + +static void cs_member_new(cs_scan_t *s, TSNode decl, char kind, bool explicit_impl, bool is_static, + int owner, TSNode name, TSNode tparams, TSNode params) { + if (ts_node_is_null(name)) { + return; + } + int i = cs_item_of_node(s, 'M', decl); + if (i < 0) { + return; + } + cs_item_t *it = &s->items[i]; + it->kind = kind; + it->explicit_impl = explicit_impl; + it->is_static = is_static; + it->owner = owner; + it->name = name; + it->tparams = tparams; + it->params = params; +} + +/* A member that has no identifier for a name: an operator (`name` is its + * token: +, ==, true, implicit ...) or an indexer (`this`). The graph has no + * node for either; the record says that the type declares one. */ +static void cs_member_unnamed(cs_scan_t *s, TSNode decl, char kind, int owner, const char *name, + TSNode params) { + int i = cs_item_of_node(s, 'M', decl); + if (i < 0) { + return; + } + cs_item_t *it = &s->items[i]; + it->kind = kind; + it->owner = owner; + it->text_name = name; + it->params = params; +} + +/* Every variable_declarator name under a field / event field declaration. */ +static void cs_collect_declarators(cs_scan_t *s, TSNode decl, char kind, int owner) { + TSNode null_node = {0}; + /* Modifiers belong to this declaration and are shared by all its names. */ + bool is_static = cs_member_static(s, decl, null_node); + cs_kids_t outer = cs_kids(decl); + TSNode vd; + while (cs_kids_next_named(&outer, &vd)) { + if (!cs_kind_is(vd, "variable_declaration")) { + continue; + } + cs_kids_t inner = cs_kids(vd); + TSNode d; + while (cs_kids_next_named(&inner, &d)) { + if (cs_kind_is(d, "variable_declarator")) { + cs_member_new(s, decl, kind, false, is_static, owner, cs_field(d, "name"), + null_node, null_node); + } + } + cs_kids_end(&inner); + } + cs_kids_end(&outer); +} + +/* A member whose header did not parse: an error node among its own parts, or + * (for a callable) an error anywhere outside its body. Its name and signature + * cannot be trusted then -- `public safe extern int M();` comes back as a + * method named `extern`. */ +static bool cs_header_broken(TSNode decl, bool callable) { + TSNode body = callable ? cs_field(decl, "body") : (TSNode){0}; + bool broken = false; + cs_kids_t k = cs_kids(decl); + TSNode ch; + while (!broken && cs_kids_next(&k, &ch)) { + if (!ts_node_is_null(body) && ts_node_eq(ch, body)) { + continue; + } + broken = cs_kind_is(ch, "ERROR") || ts_node_is_missing(ch) || + (callable && ts_node_has_error(ch)); + } + cs_kids_end(&k); + return broken; +} + +static void cs_collect_member(cs_scan_t *s, TSNode c, const char *k, int owner) { + TSNode null_node = {0}; + bool is_operator = strcmp(k, "operator_declaration") == 0; + bool is_conversion = strcmp(k, "conversion_operator_declaration") == 0; + bool is_indexer = strcmp(k, "indexer_declaration") == 0; + bool callable = strcmp(k, "method_declaration") == 0 || + strcmp(k, "constructor_declaration") == 0 || is_operator || is_conversion; + bool member = callable || is_indexer || strcmp(k, "property_declaration") == 0 || + strcmp(k, "field_declaration") == 0 || + strcmp(k, "event_field_declaration") == 0 || + strcmp(k, "event_declaration") == 0 || strcmp(k, "enum_member_declaration") == 0; + if (member && ts_node_has_error(c) && cs_header_broken(c, callable)) { + if (owner >= 0) { + s->items[owner].broken = true; /* a member of it is hidden */ + } + return; + } + if (strcmp(k, "method_declaration") == 0) { + TSNode params = cs_field(c, "parameters"); + cs_member_new(s, c, 'c', cs_has_child_kind(c, "explicit_interface_specifier"), + cs_member_static(s, c, params), owner, cs_field(c, "name"), cs_type_params(c), + params); + } else if (strcmp(k, "constructor_declaration") == 0) { + TSNode params = cs_field(c, "parameters"); + cs_member_new(s, c, 'c', false, cs_member_static(s, c, params), owner, cs_field(c, "name"), + null_node, params); + } else if (strcmp(k, "property_declaration") == 0) { + cs_member_new(s, c, 'p', cs_has_child_kind(c, "explicit_interface_specifier"), + cs_member_static(s, c, null_node), owner, cs_field(c, "name"), null_node, + null_node); + } else if (strcmp(k, "field_declaration") == 0) { + cs_collect_declarators(s, c, 'v', owner); + } else if (strcmp(k, "event_field_declaration") == 0) { + cs_collect_declarators(s, c, 'e', owner); + } else if (strcmp(k, "event_declaration") == 0) { + cs_member_new(s, c, 'e', cs_has_child_kind(c, "explicit_interface_specifier"), + cs_member_static(s, c, null_node), owner, cs_field(c, "name"), null_node, + null_node); + } else if (strcmp(k, "enum_member_declaration") == 0) { + cs_member_new(s, c, 'v', false, cs_member_static(s, c, null_node), owner, + cs_field(c, "name"), null_node, null_node); + } else if (is_operator) { + /* an anonymous token's type is its text */ + TSNode op = cs_field(c, "operator"); + if (!ts_node_is_null(op)) { + cs_member_unnamed(s, c, 'o', owner, ts_node_type(op), cs_field(c, "parameters")); + } + } else if (is_conversion) { + cs_member_unnamed(s, c, 'o', owner, cs_has_word(c, "implicit") ? "implicit" : "explicit", + cs_field(c, "parameters")); + } else if (is_indexer) { + cs_member_unnamed(s, c, 'x', owner, "this", cs_field(c, "parameters")); + } +} + +/* A node whose children are being collected. */ +typedef struct { + cs_kids_t kids; + int owner; /* the item of the type whose members the node holds, or CS_OWNER_* */ +} cs_walk_t; + +typedef struct { + cs_walk_t *frames; + int count; + int cap; +} cs_walk_stack_t; + +/* Go into `node`: its children are collected next. false when memory ran out. */ +static bool cs_walk_push(cs_scan_t *s, cs_walk_stack_t *w, TSNode node, int owner) { + if (ts_node_is_null(node)) { + return true; + } + if (w->count >= w->cap) { + int ncap = w->cap ? w->cap * PAIR_LEN : CS_HEADS_INIT; + cs_walk_t *grown = (cs_walk_t *)cs_tmp_alloc(s, (size_t)ncap * sizeof(*grown)); + if (!grown) { + return false; + } + if (w->count > 0) { + memcpy(grown, w->frames, (size_t)w->count * sizeof(*grown)); + } + w->frames = grown; + w->cap = ncap; + } + w->frames[w->count++] = (cs_walk_t){.kids = cs_kids(node), .owner = owner}; + return true; +} + +/* One child `c` of a collected node: record what it declares and go into it + * where declarations can sit. false when memory ran out. */ +static bool cs_collect_child(cs_scan_t *s, cs_walk_stack_t *w, TSNode c, int owner) { + const char *k = ts_node_type(c); + if (strcmp(k, "ERROR") == 0) { + if (owner >= 0) { + s->items[owner].broken = true; + } + return cs_walk_push(s, w, c, CS_OWNER_LEXICAL); + } + if (strncmp(k, "preproc_", 8) == 0 || strcmp(k, "declaration_list") == 0 || + strcmp(k, "enum_member_declaration_list") == 0) { + return cs_walk_push(s, w, c, owner); + } + if (strcmp(k, "using_directive") == 0) { + (void)cs_item_of_node(s, 'U', c); + return true; + } + bool file_scoped = strcmp(k, "file_scoped_namespace_declaration") == 0; + if (file_scoped || strcmp(k, "namespace_declaration") == 0) { + int ni = cs_item_of_node(s, file_scoped ? 'F' : 'N', c); + if (ni < 0) { + return false; + } + s->items[ni].name = cs_field(c, "name"); + /* a file-scoped namespace holds nothing itself -- unless it is a + * block namespace recovery could not parse, whose declarations + * then sit in an error node under it */ + TSNode body = file_scoped ? (TSNode){0} : cs_field(c, "body"); + return cs_walk_push(s, w, ts_node_is_null(body) ? c : body, CS_OWNER_NONE); + } + char tk = cs_decl_kind(c); + if (tk) { + int ti = cs_item_of_node(s, 'T', c); + if (ti < 0) { + return false; + } + s->items[ti].kind = tk; + s->items[ti].partial = cs_has_word(c, "partial"); + s->items[ti].name = cs_field(c, "name"); + s->items[ti].tparams = cs_type_params(c); + return tk == 'd' || cs_walk_push(s, w, cs_field(c, "body"), ti); + } + if (owner != CS_OWNER_NONE) { + cs_collect_member(s, c, k, owner); + } + return true; +} + +/* Collect the declarations under `root` in document order. `owner` is the + * item of the type whose members a node holds (CS_OWNER_NONE outside a type). + * Preprocessor blocks and error nodes are looked into: a declaration that + * parsed is a declaration wherever recovery left it, and its placement is + * checked against the braces afterwards. What sits in an error node belongs + * to whichever block's braces hold it. + * + * The walk keeps its own stack: how deep declarations nest is the file's + * choice, and must not be this thread's stack depth. */ +static void cs_collect(cs_scan_t *s, TSNode root, int owner) { + cs_walk_stack_t w = {0}; + bool ok = !s->failed && cs_walk_push(s, &w, root, owner); + while (w.count > 0) { + /* a child may push a frame, which moves the array: no pointer into it + * is kept across cs_collect_child */ + TSNode c; + if (!ok || s->failed || !cs_kids_next_named(&w.frames[w.count - SKIP_ONE].kids, &c)) { + cs_kids_end(&w.frames[w.count - SKIP_ONE].kids); + w.count--; + continue; + } + ok = cs_collect_child(s, &w, c, w.frames[w.count - SKIP_ONE].owner); + } +} + +/* ── Declarations the tree has no node for ───────────────────────── */ + +static int cs_u32_cmp(const void *a, const void *b) { + uint32_t x = *(const uint32_t *)a; + uint32_t y = *(const uint32_t *)b; + return (x > y) - (x < y); +} + +static int cs_item_start_cmp(const void *a, const void *b) { + const cs_item_t *x = (const cs_item_t *)a; + const cs_item_t *y = (const cs_item_t *)b; + if (x->start != y->start) { + return x->start < y->start ? -1 : 1; + } + /* declarators of one field declaration keep their order */ + return (x->tree_idx > y->tree_idx) - (x->tree_idx < y->tree_idx); +} + +/* Read the declaration a keyword starts from the text and add it as an item, + * unless the tree already has a node for it (`parsed`: the name positions of + * the tree's namespaces and types). A header ends where the next declaration + * keyword stands (`stop`): reading past it would read the file once per + * keyword. */ +static void cs_text_item(cs_scan_t *s, const cs_head_t *h, uint32_t stop, const uint32_t *parsed, + int nparsed) { + const char *src = s->ctx->source; + uint32_t n = s->root_end_byte; + uint32_t a = cs_skip_space(src, h->end, n); + if (a == h->end) { + return; /* the keyword runs into something: not a declaration */ + } + bool is_ns = h->kind == 'N'; + if (bsearch(&a, parsed, (size_t)nparsed, sizeof(uint32_t), cs_u32_cmp)) { + return; + } + char *name = NULL; + int brace = CBM_NOT_FOUND; + const char *tparams = ""; + char tag = 'T'; + if (is_ns) { + uint32_t after = cs_namespace_header_end(s, a, stop); + name = cs_namespace_name_dup(s, a, after); + if (name && src[a] != '@' && cs_not_a_name(name)) { + name = NULL; /* bare keywords stay unplaced; verbatim identifiers are names */ + } + if (after < n && src[after] == '{') { + brace = cs_open_brace_at(s, after); + tag = 'N'; + } else if (after < n && src[after] == ';') { + tag = 'F'; + } else { + return; + } + if (tag == 'N' && brace < 0) { + return; + } + /* An invalid name must still push an unplaced namespace frame. */ + } else { + uint32_t b = cs_ident_end(src, a, n, false); + name = cs_ident_dup(s, a, b); + if (!name || cs_not_a_name(name)) { + return; + } + uint32_t pos = b; + tparams = cs_text_tparams(s, &pos, stop); + bool bodyless = false; + brace = tparams ? cs_text_body(s, pos, stop, &bodyless) : CBM_NOT_FOUND; + if (!tparams || (brace < 0 && !bodyless)) { + return; + } + } + int i = cs_item_new(s, tag); + if (i < 0) { + return; + } + cs_item_t *it = &s->items[i]; + it->from_text = true; + it->tree_idx = CS_ITEM_TEXT; + it->partial = h->partial; + it->kind = is_ns ? 0 : h->kind; + it->text_name = name; + it->text_tparams = tparams; + it->brace = brace; + it->start = a; /* the name: after the modifiers, inside the same braces */ + it->line = h->row + TS_LINE_OFFSET; + it->broken = true; +} + +/* Add the declarations only the token stream shows and put all items in + * document order. */ +static void cs_add_text_items(cs_scan_t *s) { + if (s->failed || s->nheads == 0) { + return; + } + int tree_items = s->nitems; + uint32_t *parsed = + (uint32_t *)cs_tmp_alloc(s, (size_t)(tree_items + SKIP_ONE) * sizeof(uint32_t)); + if (!parsed) { + return; + } + int nparsed = 0; + for (int i = 0; i < tree_items; i++) { + const cs_item_t *it = &s->items[i]; + if (it->tag == 'N' || it->tag == 'F') { + parsed[nparsed++] = cs_namespace_name_start(s, it->node); + } else if (it->tag == 'T' && !ts_node_is_null(it->name)) { + parsed[nparsed++] = ts_node_start_byte(it->name); + } + } + qsort(parsed, (size_t)nparsed, sizeof(uint32_t), cs_u32_cmp); + for (int h = 0; h < s->nheads && !s->failed; h++) { + uint32_t stop = h + SKIP_ONE < s->nheads ? s->heads[h + SKIP_ONE].end : s->root_end_byte; + cs_text_item(s, &s->heads[h], stop, parsed, nparsed); + } + if (s->nitems > tree_items) { + qsort(s->items, (size_t)s->nitems, sizeof(cs_item_t), cs_item_start_cmp); + } +} + +/* ── Emission ────────────────────────────────────────────────────── */ + +static void cs_put_bases(cs_scan_t *s, TSNode type_decl, char kind) { + if (kind == 'e' || kind == 'd') { + return; /* an enum's base is its underlying integral type */ + } + TSNode bl = cs_child_of_kind(type_decl, "base_list"); + if (ts_node_is_null(bl)) { + return; + } + bool first = true; + cs_kids_t k = cs_kids(bl); + TSNode b; + while (cs_kids_next_named(&k, &b)) { + if (cs_kind_is(b, "primary_constructor_base_type")) { + TSNode ty = cs_field(b, "type"); + if (ts_node_is_null(ty) && ts_node_named_child_count(b) > 0) { + ty = ts_node_named_child(b, 0); + } + b = ty; + } else if (cs_kind_is(b, "argument_list") || cs_kind_is(b, "comment")) { + continue; + } + if (ts_node_is_null(b)) { + continue; + } + if (!first) { + sb_putc(&s->sb, '|'); + } + cs_put_text_nows(s, b); + first = false; + } + cs_kids_end(&k); +} + +/* What an M record says of its member. */ +typedef struct { + uint32_t line; + char kind; + bool explicit_impl; + bool is_static; + int type; /* the ordinal of the T record it belongs to */ + const char *name; +} cs_member_rec_t; + +/* A member record. A callable, an operator and an indexer carry their + * parameter types; every other member a '-'. */ +static void cs_emit_member(cs_scan_t *s, const cs_member_rec_t *m, TSNode tparams, TSNode params) { + if (!m->name || m->type < 0) { + return; + } + sb_puts(&s->sb, "M\t"); + sb_putu(&s->sb, m->line); + sb_putc(&s->sb, '\t'); + sb_putc(&s->sb, m->kind); + if (m->is_static) { + sb_putc(&s->sb, 's'); + } + sb_putc(&s->sb, '\t'); + sb_putc(&s->sb, m->explicit_impl ? '1' : '0'); + sb_putc(&s->sb, '\t'); + sb_putu(&s->sb, (uint32_t)m->type); + sb_putc(&s->sb, '\t'); + sb_puts(&s->sb, m->name); + sb_putc(&s->sb, '\t'); + cs_put_tparams(s, tparams); + sb_putc(&s->sb, '\t'); + if (m->kind == 'c' || m->kind == 'o' || m->kind == 'x') { + cs_put_sig(s, params); + } else { + sb_putc(&s->sb, '-'); + } + sb_putc(&s->sb, '\n'); +} + +static void cs_emit_using(cs_scan_t *s, TSNode u, int region) { + bool is_global = false; + bool is_static = false; + bool is_alias = false; + cs_kids_t k = cs_kids(u); + TSNode ch; + while (cs_kids_next(&k, &ch)) { + if (ts_node_is_named(ch)) { + continue; + } + const char *t = ts_node_type(ch); + if (strcmp(t, "global") == 0) { + is_global = true; + } else if (strcmp(t, "static") == 0) { + is_static = true; + } else if (strcmp(t, "=") == 0) { + is_alias = true; + } + } + cs_kids_end(&k); + TSNode alias = is_alias ? cs_field(u, "name") : (TSNode){0}; + TSNode target = {0}; + k = cs_kids(u); + while (cs_kids_next_named(&k, &ch)) { + if (is_alias && ts_node_eq(ch, alias)) { + continue; + } + if (cs_kind_is(ch, "comment")) { + continue; + } + target = ch; + } + cs_kids_end(&k); + if (ts_node_is_null(target)) { + return; + } + /* what the directive brings in, and whom it serves: `global using` (of a + * namespace, of a type's static members, of an alias alike) is in scope in + * every file of the project */ + sb_puts(&s->sb, "U\t"); + sb_putu(&s->sb, (uint32_t)region); + sb_putc(&s->sb, '\t'); + sb_putc(&s->sb, is_alias ? 'a' : (is_static ? 's' : 'n')); + if (is_global) { + sb_putc(&s->sb, 'g'); + } + sb_putc(&s->sb, '\t'); + if (is_alias && !ts_node_is_null(alias)) { + cs_put_text_nows(s, alias); + } else { + sb_putc(&s->sb, '-'); + } + sb_putc(&s->sb, '\t'); + cs_put_text_nows(s, target); + sb_putc(&s->sb, '\n'); +} + +/* Lines [from, to] hold declarations that could not be placed. */ +static void cs_emit_unplaced(cs_scan_t *s, uint32_t from, uint32_t to) { + sb_puts(&s->sb, "X\t"); + sb_putu(&s->sb, from); + sb_putc(&s->sb, '\t'); + sb_putu(&s->sb, to); + sb_putc(&s->sb, '\n'); +} + +/* A namespace region record: the name as its declaration writes it (`A.B`), + * under the region `parent`. Returns the new region id. */ +static int cs_emit_region(cs_scan_t *s, int parent, uint32_t start, uint32_t end, + const char *name) { + int id = s->next_region++; + sb_puts(&s->sb, "R\t"); + sb_putu(&s->sb, (uint32_t)id); + sb_putc(&s->sb, '\t'); + sb_putu(&s->sb, (uint32_t)parent); + sb_putc(&s->sb, '\t'); + sb_putu(&s->sb, start); + sb_putc(&s->sb, '\t'); + sb_putu(&s->sb, end); + sb_putc(&s->sb, '\t'); + sb_puts(&s->sb, name); + sb_putc(&s->sb, '\n'); + return id; +} + +/* A type record, the type's ordinal being `ord`. Its outer type is named by + * that type's ordinal, never by a path: a record's size does not grow with + * the nesting. */ +static void cs_emit_type(cs_scan_t *s, const cs_item_t *it, int region, uint32_t end_line, + int outer, const char *name, bool incomplete, int ord) { + sb_puts(&s->sb, "T\t"); + sb_putu(&s->sb, (uint32_t)region); + sb_putc(&s->sb, '\t'); + sb_putu(&s->sb, it->line); + sb_putc(&s->sb, '\t'); + sb_putu(&s->sb, end_line); + sb_putc(&s->sb, '\t'); + sb_putc(&s->sb, it->kind); + if (it->partial) { + sb_putc(&s->sb, 'p'); + } + if (incomplete) { + sb_putc(&s->sb, '!'); + } + sb_putc(&s->sb, '\t'); + if (outer >= 0) { + sb_putu(&s->sb, (uint32_t)outer); + } else { + sb_putc(&s->sb, '-'); + } + sb_putc(&s->sb, '\t'); + sb_puts(&s->sb, name); + sb_putc(&s->sb, '\t'); + if (it->from_text) { + /* its header did not parse: the bases are unknown, which the "?" + * records as a hierarchy that cannot be followed */ + sb_puts(&s->sb, it->text_tparams); + sb_puts(&s->sb, it->kind == 'e' ? "\t\n" : "\t?\n"); + return; + } + cs_put_tparams(s, it->tparams); + sb_putc(&s->sb, '\t'); + cs_put_bases(s, it->node, it->kind); + sb_putc(&s->sb, '\n'); + /* A parameter list on the type itself is its primary constructor + * (`record R(int X)`, `class C(int x)`): a constructor the graph has no + * node for. A record's parameters are properties as well. */ + bool record = it->kind == 'r' || it->kind == 't'; + if (!(record || it->kind == 'c' || it->kind == 's')) { + return; + } + TSNode pl = cs_child_of_kind(it->node, "parameter_list"); + if (ts_node_is_null(pl)) { + return; + } + TSNode null_node = {0}; + cs_member_rec_t ctor = {.line = it->line, .kind = 'c', .type = ord, .name = name}; + cs_emit_member(s, &ctor, null_node, pl); + cs_kids_t params = cs_kids(pl); + TSNode p; + while (record && cs_kids_next_named(¶ms, &p)) { + if (cs_kind_is(p, "parameter")) { + cs_member_rec_t prop = {.line = cs_line(p), + .kind = 'p', + .type = ord, + .name = cs_name_dup(s, cs_field(p, "name"))}; + cs_emit_member(s, &prop, null_node, null_node); + } + } + cs_kids_end(¶ms); +} + +/* An open block while the items are placed. */ +typedef struct { + int item; /* tree_idx of its declaration; CBM_NOT_FOUND for the file itself */ + uint32_t end; /* first byte after it */ + int inner_depth; /* brace depth of what it holds (files with parse errors) */ + int region; /* namespace region in effect inside */ + int ns_segments; /* segments of the namespace in effect inside */ + int type; /* a type's block: the ordinal of its T record (CBM_NOT_FOUND: none) */ + int type_depth; /* types that enclose what it holds */ + bool is_type; + bool bad; /* its own place or name is unknown: so is everything inside */ +} cs_frame_t; + +/* The open blocks. A block is pushed only on a placed one, and a placed + * block adds a type level or at least one namespace segment, so the two + * nesting limits bound the stack: the root, the placed blocks, and one + * unplaced block on top. */ +enum { CS_FRAMES = CS_MAX_TYPE_NEST + CS_MAX_NS_SEGMENTS + PAIR_LEN }; + +typedef struct { + cs_frame_t frames[CS_FRAMES]; + int sp; +} cs_stack_t; + +static void cs_frame_push(cs_scan_t *s, cs_stack_t *st, cs_frame_t f) { + if (st->sp >= CS_FRAMES) { + s->failed = true; /* cannot happen (see CS_FRAMES); no scope rather than a wrong one */ + return; + } + st->frames[st->sp++] = f; +} + +/* Segments of a dotted name. */ +static int cs_segments(const char *name) { + int n = SKIP_ONE; + for (const char *p = name; *p; p++) { + n += *p == '.'; + } + return n; +} + +/* What follows a namespace's name in a file with parse errors: the brace + * that opens its block (returned), or the `;` of a file-scoped namespace + * (*file_scoped). Recovery reports a block namespace it could not parse as a + * file-scoped one whose content is an error node; the text says which it is. + * Anything else -- `namespace A` / `#else` / `namespace B` / `#endif` / `{` + * -- is a namespace this scan cannot name: CBM_NOT_FOUND, not file-scoped. */ +static int cs_namespace_brace(const cs_scan_t *s, TSNode name, bool *file_scoped) { + const char *src = s->ctx->source; + uint32_t after = cs_skip_space(src, ts_node_end_byte(name), s->root_end_byte); + *file_scoped = after < s->root_end_byte && src[after] == ';'; + return cs_open_brace_at(s, after); +} + +static void cs_place_namespace(cs_scan_t *s, cs_stack_t *st, const cs_item_t *it, bool trusted) { + const cs_frame_t top = st->frames[st->sp - SKIP_ONE]; + if (top.bad) { + return; /* inside a block that is not placed: nothing is, and nothing nests */ + } + const char *name = it->from_text ? it->text_name : NULL; + if (!it->from_text && !ts_node_is_null(it->name)) { + name = cs_namespace_name_dup(s, cs_namespace_name_start(s, it->node), + ts_node_end_byte(it->name)); + } + int brace = it->brace; + bool file_scoped = it->tag == 'F'; + if (!it->from_text && s->lexical) { + /* the text decides what kind of namespace declaration this is */ + file_scoped = false; + brace = ts_node_is_null(it->name) ? CBM_NOT_FOUND + : cs_namespace_brace(s, it->name, &file_scoped); + } + /* a file-scoped namespace is the first declaration of its file */ + bool ok = trusted && !top.is_type && name && (!file_scoped || st->sp == SKIP_ONE); + uint32_t end = s->root_end_byte; + uint32_t end_line = s->root_end_line; + int inner = top.inner_depth; + if (!file_scoped) { + bool known = false; + if (brace >= 0) { + known = cs_brace_extent(s, brace, &end, &end_line, &inner); + } else if (!s->lexical && !it->from_text) { + end = ts_node_end_byte(it->node); + end_line = cs_end_line(it->node); + known = true; + } + if (!known) { + ok = false; + end = UINT32_MAX; /* where it ends is unknown: nothing after it is placed */ + } + } + /* a namespace nested past the limit is not placed */ + int segments = name ? top.ns_segments + cs_segments(name) : 0; + ok = ok && segments <= CS_MAX_NS_SEGMENTS; + if (!ok && it->start < s->untrusted) { + cs_emit_unplaced(s, it->line, end == UINT32_MAX ? s->root_end_line : end_line); + } + int region = ok ? cs_emit_region(s, top.region, it->line, end_line, name) : top.region; + cs_frame_push(s, st, + (cs_frame_t){.item = it->tree_idx, + .end = end, + .inner_depth = inner, + .region = region, + .ns_segments = ok ? segments : top.ns_segments, + .type = CBM_NOT_FOUND, + .type_depth = 0, + .is_type = false, + .bad = !ok}); +} + +static void cs_put_quarantine(cs_scan_t *s, const char *name) { + sb_puts(&s->sb, "Q\t"); + sb_puts(&s->sb, name); + sb_putc(&s->sb, '\n'); +} + +static void cs_place_type(cs_scan_t *s, cs_stack_t *st, const cs_item_t *it, bool trusted) { + const cs_frame_t top = st->frames[st->sp - SKIP_ONE]; + const char *name = it->from_text ? it->text_name : cs_name_dup(s, it->name); + if (top.bad) { + /* declared inside a block that is not placed: its name must not + * resolve to anything else; nothing nests under it */ + if (name) { + cs_put_quarantine(s, name); + } + return; + } + bool block = it->brace >= 0; + uint32_t end = it->start; + uint32_t end_line = it->line; + int inner = 0; + bool paired = true; + bool whole = false; /* the tree's extent is the block's extent */ + if (it->from_text) { + paired = !block || cs_brace_extent(s, it->brace, &end, &end_line, &inner); + } else { + TSNode body = it->kind == 'd' ? (TSNode){0} : cs_field(it->node, "body"); + block = !ts_node_is_null(body); + end = ts_node_end_byte(it->node); + end_line = cs_end_line(it->node); + whole = true; + if (block && s->lexical) { + uint32_t node_end = end; + paired = cs_brace_extent(s, cs_open_brace_at(s, ts_node_start_byte(body)), &end, + &end_line, &inner); + whole = paired && end == node_end; + } + } + /* a type nested past the limit is not placed */ + bool ok = trusted && paired && name && top.type_depth < CS_MAX_TYPE_NEST; + int ord = CBM_NOT_FOUND; + if (ok) { + ord = s->types_out++; + cs_emit_type(s, it, top.region, end_line, top.type, name, it->broken || !whole, ord); + } else if (name) { + /* declared, but where is unknown: its name must not resolve to + * anything else either */ + cs_put_quarantine(s, name); + } + if (!ok && block && it->start < s->untrusted) { + cs_emit_unplaced(s, it->line, paired ? end_line : s->root_end_line); + } + if (block) { + cs_frame_push(s, st, + (cs_frame_t){.item = it->tree_idx, + .end = paired ? end : UINT32_MAX, + .inner_depth = inner, + .region = top.region, + .ns_segments = top.ns_segments, + .type = ord, + .type_depth = top.type_depth + SKIP_ONE, + .is_type = true, + .bad = !ok}); + } +} + +/* Place every collected declaration in the block that holds it and write + * its record. */ +static void cs_emit_items(cs_scan_t *s) { + cs_stack_t *st = (cs_stack_t *)cs_tmp_alloc(s, sizeof(*st)); + if (!st) { + return; + } + st->sp = 0; + st->frames[st->sp++] = (cs_frame_t){.item = CBM_NOT_FOUND, + .end = UINT32_MAX, + .inner_depth = 0, + .region = 0, + .ns_segments = 0, + .type = CBM_NOT_FOUND, + .type_depth = 0, + .is_type = false, + .bad = false}; + for (int i = 0; i < s->nitems && !s->failed && !s->sb.failed; i++) { + const cs_item_t *it = &s->items[i]; + /* leave the blocks that ended; with parse errors also those the + * braces say this item is not in (an #else branch re-opening the + * block its #if branch opened) */ + int depth = s->lexical ? cs_depth_at(s, it->start) : 0; + while (st->sp > SKIP_ONE && + (st->frames[st->sp - SKIP_ONE].end <= it->start || + (s->lexical && st->frames[st->sp - SKIP_ONE].inner_depth > depth))) { + st->sp--; + } + const cs_frame_t top = st->frames[st->sp - SKIP_ONE]; + bool trusted = !top.bad; + if (trusted && s->lexical) { + trusted = it->start < s->untrusted && depth == top.inner_depth; + } + switch (it->tag) { + case 'U': + if (trusted && !top.is_type) { + cs_emit_using(s, it->node, top.region); + } + break; + case 'N': + case 'F': + cs_place_namespace(s, st, it, trusted); + break; + case 'T': + cs_place_type(s, st, it, trusted); + break; + case 'M': + /* the braces decide whose member it is when the tree has errors */ + if (trusted && top.is_type && (s->lexical || it->owner == top.item)) { + cs_member_rec_t rec = {.line = cs_line(it->node), + .kind = it->kind, + .explicit_impl = it->explicit_impl, + .is_static = it->is_static, + .type = top.type, + .name = it->text_name ? it->text_name + : cs_name_dup(s, it->name)}; + cs_emit_member(s, &rec, it->tparams, it->params); + } + break; + default: + break; + } + } +} + +/* Field positions (0-based, tag included) that hold line numbers. */ +static bool cs_line_field(char tag, int field) { + switch (tag) { + case 'R': + return field == 3 || field == 4; + case 'T': + return field == 2 || field == 3; + case 'M': + return field == 1; + case 'X': + return field == 1 || field == 2; + default: + return false; + } +} + +char *cbm_doclink_cs_portable_scope(const char *scope) { + size_t n = strlen(scope); + char *out = (char *)cbm_alloc(CBM_MEM_CLASS_OTHER, n + SKIP_ONE); + if (!out) { + return NULL; + } + size_t w = 0; + const char *p = scope; + while (*p) { + const char *nl = strchr(p, '\n'); + size_t len = nl ? (size_t)(nl - p) : strlen(p); + char tag = p[0]; + int field = 0; + size_t i = 0; + while (i < len) { + size_t fend = i; + while (fend < len && p[fend] != '\t') { + fend++; + } + /* a line number becomes 0; an empty field stays empty, so the + * copy is never longer than the scope it is made from */ + if (cs_line_field(tag, field) && fend > i) { + out[w++] = '0'; + } else { + memcpy(out + w, p + i, fend - i); + w += fend - i; + } + i = fend; + if (i < len) { + out[w++] = '\t'; + i++; + field++; + } + } + if (nl) { + out[w++] = '\n'; + p = nl + SKIP_ONE; + } else { + p += len; + } + } + out[w] = '\0'; + return out; +} + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +static char cs_test_spoiled_path[CBM_SZ_512]; + +void cbm_doclink_cs_test_spoil_scope(const char *rel_path) { + snprintf(cs_test_spoiled_path, sizeof(cs_test_spoiled_path), "%s", rel_path ? rel_path : ""); +} +#endif + +const char *cbm_doclink_cs_scan_scope(CBMExtractCtx *ctx) { + if (ts_node_is_null(ctx->root)) { + return NULL; + } + cs_scan_t s = {.ctx = ctx, + .tmp = ctx->scratch ? ctx->scratch : ctx->arena, + .sb = {.a = ctx->arena}, + .top = CBM_NOT_FOUND, + .untrusted = UINT32_MAX, + .next_region = SKIP_ONE}; + s.root_end_byte = ctx->source_len > 0 ? (uint32_t)ctx->source_len : 0; + s.root_end_line = cs_end_line(ctx->root); + s.lexical = ts_node_has_error(ctx->root); + if (s.lexical) { + cs_scan_tokens(&s); + } + /* the root itself is an error node when nothing of the file parsed */ + cs_collect(&s, ctx->root, cs_kind_is(ctx->root, "ERROR") ? CS_OWNER_LEXICAL : CS_OWNER_NONE); + cs_add_text_items(&s); + sb_puts(&s.sb, CBM_DOCLINK_CS_SCOPE_TAG "\n"); + cs_emit_items(&s); + if (s.untrusted != UINT32_MAX) { + cs_emit_unplaced(&s, s.untrusted_row + TS_LINE_OFFSET, s.root_end_line); + } + cs_cost_publish(&s); +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (cbm_doclink_test_fail_alloc(CBM_DOCLINK_ALLOC_SCOPE)) { + s.failed = true; + } + if (cs_test_spoiled_path[0] && ctx->rel_path && + strcmp(ctx->rel_path, cs_test_spoiled_path) == 0) { + sb_puts(&s.sb, "Z\tspoiled\n"); + } +#endif + if (s.failed || s.sb.failed || !s.sb.buf) { + /* memory ran out: no scope would read as a file that declares + * nothing, so the layer is told */ + if (ctx->result) { + ctx->result->doc_links.failed = true; + } + return NULL; + } + return s.sb.buf; +} + +/* ── MSBuild project files ─────────────────────────────────────────── + * + * A C# project's global usings come from its MSBuild files: the project + * file, the Directory.Build.props / .targets above it, and what those import + * (R1). The resolver must not open any of them: what it would read is a path + * the index never looked at (a symbolic link out of the repository, a named + * pipe), and it would read it again on every run. So a project file has a + * scope blob like a source file, and the resolver evaluates blobs. + * + * Project blob: the tag line "cs1", then one record per line. A field is + * empty for an attribute that is not there, else '=' and its text, XML + * entities decoded, with \\ \t \n \r escaped. + * P sdk state the first record. state '-': read; + * '!': a project file that could not be + * read; '>': one larger than a project + * file is (CSX_MAX_PROJECT_BYTES), not + * read. Nothing follows the last two + * I group-cond cond project sdk ; old blobs repeat a group's + * condition in the first field + * B cond start of an , condition once + * J (empty) cond project sdk a child import; shares the B condition + * E end of the import group + * G cond ; its properties follow + * V cond name value a property; value '?' when it is not + * plain text + * H cond that has items; + * they follow + * N cond include remove static alias + * K name a property set inside a construct that + * is not evaluated () + * Y a that is not evaluated (in a + * , with Update, with metadata + * elements) + * C what another construct that is not evaluated + * Document order is evaluation order. Nothing else of the file is in the + * blob (targets, other items), so an edit there leaves it as it is. */ + +enum { CSX_EOF = 0, CSX_OPEN, CSX_EMPTY, CSX_CLOSE, CSX_TEXT, CSX_BAD }; + +/* The size past which a file is no project file to this reader. */ +enum { CSX_MAX_PROJECT_BYTES = 1000000 }; + +enum { + CSX_A_CONDITION = 0, + CSX_A_PROJECT, + CSX_A_SDK, + CSX_A_INCLUDE, + CSX_A_REMOVE, + CSX_A_UPDATE, + CSX_A_STATIC, + CSX_A_ALIAS, + CSX_A_COUNT +}; + +typedef struct { + uint32_t s; + uint32_t e; + bool has; +} csx_span_t; + +typedef struct { + int kind; + csx_span_t name; /* element name without a namespace prefix */ + csx_span_t text; /* CSX_TEXT */ + bool raw; /* CSX_TEXT of a CDATA section: no entities in it */ + csx_span_t attr[CSX_A_COUNT]; +} csx_tok_t; + +typedef struct { + const char *src; + uint32_t n; + uint32_t i; +} csx_t; + +/* Index of `lit` in src[from, limit), or `limit`. */ +static uint32_t csx_find(const csx_t *x, uint32_t from, uint32_t limit, const char *lit) { + size_t ll = strlen(lit); + for (uint32_t k = from; k + ll <= limit; k++) { + if (x->src[k] == lit[0] && memcmp(x->src + k, lit, ll) == 0) { + return k; + } + } + return limit; +} + +static bool csx_name_char(unsigned char c) { + return isalnum(c) || c == '_' || c == ':' || c == '.' || c == '-' || c >= 0x80; +} + +static csx_span_t csx_local(const csx_t *x, uint32_t s, uint32_t e) { + for (uint32_t k = e; k > s; k--) { + if (x->src[k - SKIP_ONE] == ':') { + s = k; + break; + } + } + return (csx_span_t){.s = s, .e = e, .has = true}; +} + +static bool csx_is(const csx_t *x, csx_span_t v, const char *word) { + return v.has && strlen(word) == v.e - v.s && memcmp(x->src + v.s, word, v.e - v.s) == 0; +} + +/* The attributes the blob keeps. MSBuild reads attribute names without + * regard to case. */ +static int csx_attr_index(const char *name, size_t len) { + static const char *const names[CSX_A_COUNT] = {"condition", "project", "sdk", "include", + "remove", "update", "static", "alias"}; + for (int a = 0; a < CSX_A_COUNT; a++) { + if (strlen(names[a]) != len) { + continue; + } + size_t k = 0; + while (k < len && tolower((unsigned char)name[k]) == names[a][k]) { + k++; + } + if (k == len) { + return a; + } + } + return CBM_NOT_FOUND; +} + +/* Past markup that is no element at src[x->i] ('<' stands there): a comment, + * a processing instruction, a declaration. false when it does not end. A + * CDATA section is text: *cdata, and the cursor stays. */ +static bool csx_skip_markup(csx_t *x, bool *skipped, bool *cdata) { + const char *s = x->src; + uint32_t rest = x->n - x->i; + *skipped = true; + *cdata = false; + if (rest >= 4 && memcmp(s + x->i, ""); + x->i = e + 3; + return e < x->n; + } + if (rest >= 9 && memcmp(s + x->i, "= PAIR_LEN && s[x->i + SKIP_ONE] == '?') { + uint32_t e = csx_find(x, x->i + PAIR_LEN, x->n, "?>"); + x->i = e + PAIR_LEN; + return e < x->n; + } + if (rest >= PAIR_LEN && s[x->i + SKIP_ONE] == '!') { + /* , with an internal subset up to "]>" */ + uint32_t e = csx_find(x, x->i + PAIR_LEN, x->n, ">"); + uint32_t sub = csx_find(x, x->i + PAIR_LEN, e, "["); + if (sub < e) { + uint32_t close = csx_find(x, sub, x->n, "]>"); + e = close < x->n ? close + SKIP_ONE : x->n; + } + x->i = e + SKIP_ONE; + return e < x->n; + } + *skipped = false; + return true; +} + +/* The attributes of a start tag from src[p]; the tag's kind (CSX_OPEN, + * CSX_EMPTY) or CSX_BAD. Moves the cursor past the tag. */ +static int csx_attributes(csx_t *x, uint32_t p, csx_tok_t *t) { + const char *s = x->src; + for (;;) { + while (p < x->n && isspace((unsigned char)s[p])) { + p++; + } + if (p >= x->n) { + return CSX_BAD; + } + if (s[p] == '>') { + x->i = p + SKIP_ONE; + return CSX_OPEN; + } + if (s[p] == '/' && p + SKIP_ONE < x->n && s[p + SKIP_ONE] == '>') { + x->i = p + PAIR_LEN; + return CSX_EMPTY; + } + uint32_t as = p; + while (p < x->n && csx_name_char((unsigned char)s[p])) { + p++; + } + if (p == as) { + return CSX_BAD; + } + csx_span_t an = csx_local(x, as, p); + while (p < x->n && isspace((unsigned char)s[p])) { + p++; + } + if (p >= x->n || s[p] != '=') { + return CSX_BAD; + } + p++; + while (p < x->n && isspace((unsigned char)s[p])) { + p++; + } + if (p >= x->n || (s[p] != '"' && s[p] != '\'')) { + return CSX_BAD; + } + char quote = s[p++]; + uint32_t vs = p; + while (p < x->n && s[p] != quote) { + p++; + } + if (p >= x->n) { + return CSX_BAD; + } + int ai = csx_attr_index(s + an.s, an.e - an.s); + if (ai >= 0) { + t->attr[ai] = (csx_span_t){.s = vs, .e = p, .has = true}; + } + p++; + } +} + +/* The next token of the document. Every byte is passed once. */ +static void csx_next(csx_t *x, csx_tok_t *t) { + memset(t, 0, sizeof(*t)); + const char *s = x->src; + for (;;) { + if (x->i >= x->n) { + t->kind = CSX_EOF; + return; + } + if (s[x->i] != '<') { + uint32_t a = x->i; + while (x->i < x->n && s[x->i] != '<') { + x->i++; + } + t->kind = CSX_TEXT; + t->text = (csx_span_t){.s = a, .e = x->i, .has = true}; + return; + } + bool skipped = false; + bool cdata = false; + if (!csx_skip_markup(x, &skipped, &cdata)) { + t->kind = CSX_BAD; + return; + } + if (cdata) { + uint32_t e = csx_find(x, x->i + 9, x->n, "]]>"); + if (e >= x->n) { + t->kind = CSX_BAD; + return; + } + t->kind = CSX_TEXT; + t->raw = true; + t->text = (csx_span_t){.s = x->i + 9, .e = e, .has = true}; + x->i = e + 3; + return; + } + if (!skipped) { + break; + } + } + uint32_t p = x->i + SKIP_ONE; + bool close = p < x->n && s[p] == '/'; + if (close) { + p++; + } + uint32_t ns = p; + while (p < x->n && csx_name_char((unsigned char)s[p])) { + p++; + } + if (p == ns) { + t->kind = CSX_BAD; + return; + } + t->name = csx_local(x, ns, p); + if (!close) { + t->kind = csx_attributes(x, p, t); + return; + } + while (p < x->n && isspace((unsigned char)s[p])) { + p++; + } + if (p >= x->n || s[p] != '>') { + t->kind = CSX_BAD; + return; + } + x->i = p + SKIP_ONE; + t->kind = CSX_CLOSE; +} + +/* One character of a field: escaped where it would break the line format. A + * control character has no place in XML text; it is written as a space. */ +static void csx_put_char(cs_sb_t *sb, unsigned char c) { + if (c == '\\') { + sb_puts(sb, "\\\\"); + } else if (c == '\t') { + sb_puts(sb, "\\t"); + } else if (c == '\n') { + sb_puts(sb, "\\n"); + } else if (c == '\r') { + sb_puts(sb, "\\r"); + } else { + sb_putc(sb, c < 0x20 ? ' ' : (char)c); + } +} + +/* A code point as UTF-8. A surrogate (a numeric reference may name one) is + * no character: it is written as U+FFFD. */ +static void csx_put_codepoint(cs_sb_t *sb, uint32_t cp) { + if (cp >= 0xD800 && cp <= 0xDFFF) { + sb_puts(sb, CS_REPLACEMENT); + } else if (cp < 0x80) { + csx_put_char(sb, (unsigned char)cp); + } else if (cp < 0x800) { + sb_putc(sb, (char)(0xC0 | (cp >> 6))); + sb_putc(sb, (char)(0x80 | (cp & 0x3F))); + } else if (cp < 0x10000) { + sb_putc(sb, (char)(0xE0 | (cp >> 12))); + sb_putc(sb, (char)(0x80 | ((cp >> 6) & 0x3F))); + sb_putc(sb, (char)(0x80 | (cp & 0x3F))); + } else { + sb_putc(sb, (char)(0xF0 | ((cp >> 18) & 0x07))); + sb_putc(sb, (char)(0x80 | ((cp >> 12) & 0x3F))); + sb_putc(sb, (char)(0x80 | ((cp >> 6) & 0x3F))); + sb_putc(sb, (char)(0x80 | (cp & 0x3F))); + } +} + +/* The entity at src[i] ('&' stands there): its code point and the index past + * it; 0 when there is none. */ +static uint32_t csx_entity(const csx_t *x, uint32_t i, uint32_t end, uint32_t *cp) { + static const struct { + const char *ent; + char ch; + } ents[] = {{"<", '<'}, {">", '>'}, {"&", '&'}, {""", '"'}, {"'", '\''}}; + const char *s = x->src; + for (size_t e = 0; e < sizeof(ents) / sizeof(ents[0]); e++) { + size_t el = strlen(ents[e].ent); + if (i + el <= end && memcmp(s + i, ents[e].ent, el) == 0) { + *cp = (unsigned char)ents[e].ch; + return i + (uint32_t)el; + } + } + if (i + PAIR_LEN < end && s[i + SKIP_ONE] == '#') { + bool hex = s[i + PAIR_LEN] == 'x' || s[i + PAIR_LEN] == 'X'; + uint32_t k = i + PAIR_LEN + (hex ? SKIP_ONE : 0); + uint32_t v = 0; + uint32_t digits = 0; + while (k < end && digits < CBM_SZ_8 && + (hex ? isxdigit((unsigned char)s[k]) : isdigit((unsigned char)s[k]))) { + unsigned char d = (unsigned char)s[k]; + uint32_t dv = isdigit(d) ? (uint32_t)(d - '0') : (uint32_t)(tolower(d) - 'a') + 10U; + v = (v * (hex ? 16U : 10U)) + dv; + k++; + digits++; + } + if (digits > 0 && k < end && s[k] == ';' && v > 0 && v <= 0x10FFFF) { + *cp = v; + return k + SKIP_ONE; + } + } + return 0; +} + +/* The bytes src[i, end) of a project file that start at i into sb: a + * well-formed UTF-8 sequence as it stands, a byte that starts none as + * U+FFFD. Returns the index past what was taken. */ +static uint32_t csx_put_utf8(cs_sb_t *sb, const char *src, uint32_t i, uint32_t end) { + size_t len = cs_utf8_len((const unsigned char *)src + i, (size_t)(end - i)); + if (len == 0) { + sb_puts(sb, CS_REPLACEMENT); + return i + SKIP_ONE; + } + sb_putn(sb, src + i, len); + return i + (uint32_t)len; +} + +/* The text of `v`, entities decoded (unless `raw`), escaped. */ +static void csx_put_text(cs_sb_t *sb, const csx_t *x, csx_span_t v, bool raw) { + for (uint32_t i = v.s; i < v.e;) { + uint32_t cp = 0; + uint32_t past = (!raw && x->src[i] == '&') ? csx_entity(x, i, v.e, &cp) : 0; + if (past) { + csx_put_codepoint(sb, cp); + i = past; + } else if ((unsigned char)x->src[i] >= 0x80) { + i = csx_put_utf8(sb, x->src, i, v.e); + } else { + csx_put_char(sb, (unsigned char)x->src[i]); + i++; + } + } +} + +/* An element's name into sb, each byte that starts no well-formed UTF-8 + * sequence as U+FFFD (the tokenizer's names hold no separator). */ +static void csx_put_name(cs_sb_t *sb, const csx_t *x, csx_span_t name) { + for (uint32_t i = name.s; i < name.e;) { + i = csx_put_utf8(sb, x->src, i, name.e); + } +} + +/* A field: a tab, then nothing for an absent attribute, else '=' and its text. */ +static void csx_put_field(cs_sb_t *sb, const csx_t *x, csx_span_t v) { + sb_putc(sb, '\t'); + if (v.has) { + sb_putc(sb, '='); + csx_put_text(sb, x, v, false); + } +} + +/* What a project file's scan keeps between tokens. */ +typedef struct { + csx_t x; + cs_sb_t out; + cs_sb_t value; /* a property's text so far */ + int depth; /* open elements */ + char group; /* the child of the cursor is in: G H i c, or 0 */ + csx_span_t gcond; /* its Condition */ + bool h_written; /* the ItemGroup's H record is out */ + bool i_written; /* the ImportGroup's B record is out */ + bool prop_open; /* a property element is open ... */ + bool prop_complex; /* ... and holds elements, not just text */ + bool using_open; /* a with content is open ... */ + bool using_complex; /* ... and holds metadata elements */ + csx_tok_t pending; /* the open property's or 's start tag */ + int choose_props; /* in a : the depth of a 's children, or -1 */ +} csx_scan_t; + +static void csx_put_using(csx_scan_t *p, const csx_tok_t *t) { + if (!p->h_written) { + sb_putc(&p->out, 'H'); + csx_put_field(&p->out, &p->x, p->gcond); + sb_putc(&p->out, '\n'); + p->h_written = true; + } + sb_putc(&p->out, 'N'); + csx_put_field(&p->out, &p->x, t->attr[CSX_A_CONDITION]); + csx_put_field(&p->out, &p->x, t->attr[CSX_A_INCLUDE]); + csx_put_field(&p->out, &p->x, t->attr[CSX_A_REMOVE]); + csx_put_field(&p->out, &p->x, t->attr[CSX_A_STATIC]); + csx_put_field(&p->out, &p->x, t->attr[CSX_A_ALIAS]); + sb_putc(&p->out, '\n'); +} + +static void csx_put_import(csx_scan_t *p, const csx_tok_t *t, bool grouped) { + if (grouped && !p->i_written) { + sb_putc(&p->out, 'B'); + csx_put_field(&p->out, &p->x, p->gcond); + sb_putc(&p->out, '\n'); + p->i_written = true; + } + sb_putc(&p->out, grouped ? 'J' : 'I'); + csx_put_field(&p->out, &p->x, (csx_span_t){0}); + csx_put_field(&p->out, &p->x, t->attr[CSX_A_CONDITION]); + csx_put_field(&p->out, &p->x, t->attr[CSX_A_PROJECT]); + csx_put_field(&p->out, &p->x, t->attr[CSX_A_SDK]); + sb_putc(&p->out, '\n'); +} + +/* The open property ends: its record. */ +static void csx_put_property(csx_scan_t *p) { + const csx_tok_t *t = &p->pending; + sb_putc(&p->out, 'V'); + csx_put_field(&p->out, &p->x, t->attr[CSX_A_CONDITION]); + sb_putc(&p->out, '\t'); + csx_put_name(&p->out, &p->x, t->name); + sb_putc(&p->out, '\t'); + if (p->prop_complex) { + sb_putc(&p->out, '?'); + } else { + sb_putc(&p->out, '='); + if (p->value.len > 0) { + sb_putn(&p->out, p->value.buf, p->value.len); + } + } + sb_putc(&p->out, '\n'); +} + +/* An element inside a : what it could set is named, not evaluated. */ +static void csx_choose_child(csx_scan_t *p, const csx_tok_t *t) { + if (p->choose_props >= 0 && p->depth == p->choose_props) { + sb_puts(&p->out, "K\t"); + csx_put_name(&p->out, &p->x, t->name); + sb_putc(&p->out, '\n'); + } else if (csx_is(&p->x, t->name, "PropertyGroup") && p->choose_props < 0 && + t->kind == CSX_OPEN) { + p->choose_props = p->depth + SKIP_ONE; + } else if (csx_is(&p->x, t->name, "Using")) { + sb_puts(&p->out, "Y\n"); + } else if (csx_is(&p->x, t->name, "Import")) { + sb_puts(&p->out, "C\tImport\n"); + } +} + +/* A child of starts. */ +static void csx_project_child(csx_scan_t *p, const csx_tok_t *t) { + p->group = 0; + p->gcond = t->attr[CSX_A_CONDITION]; + if (csx_is(&p->x, t->name, "Import")) { + csx_put_import(p, t, false); + } else if (csx_is(&p->x, t->name, "PropertyGroup")) { + sb_putc(&p->out, 'G'); + csx_put_field(&p->out, &p->x, p->gcond); + sb_putc(&p->out, '\n'); + p->group = 'G'; + } else if (csx_is(&p->x, t->name, "ItemGroup")) { + p->group = 'H'; + p->h_written = false; + } else if (csx_is(&p->x, t->name, "ImportGroup")) { + p->group = 'i'; + p->i_written = false; + } else if (csx_is(&p->x, t->name, "Choose")) { + sb_puts(&p->out, "C\tChoose\n"); + p->group = 'c'; + p->choose_props = CBM_NOT_FOUND; + } + if (t->kind == CSX_EMPTY) { + p->group = 0; + } +} + +/* A grandchild of starts. */ +static void csx_group_child(csx_scan_t *p, const csx_tok_t *t) { + if (p->group == 'G') { + p->pending = *t; + p->value.len = 0; + p->prop_complex = false; + if (t->kind == CSX_EMPTY) { + csx_put_property(p); + } else { + p->prop_open = true; + } + } else if (p->group == 'H' && csx_is(&p->x, t->name, "Using")) { + if (t->attr[CSX_A_UPDATE].has) { + sb_puts(&p->out, "Y\n"); + } else if (t->kind == CSX_EMPTY) { + csx_put_using(p, t); + } else { + p->pending = *t; + p->using_open = true; + p->using_complex = false; + } + } else if (p->group == 'i' && csx_is(&p->x, t->name, "Import")) { + csx_put_import(p, t, true); + } +} + +/* A start tag below the root. */ +static void csx_element(csx_scan_t *p, const csx_tok_t *t) { + if (p->depth == SKIP_ONE) { + csx_project_child(p, t); + } else if (p->group == 'c') { + csx_choose_child(p, t); + } else if (p->depth == PAIR_LEN) { + csx_group_child(p, t); + } else { + p->prop_complex = p->prop_complex || p->prop_open; + p->using_complex = p->using_complex || p->using_open; + } + if (t->kind == CSX_OPEN) { + p->depth++; + } +} + +/* An end tag: `depth` is already the depth outside the element. */ +static void csx_element_end(csx_scan_t *p) { + if (p->depth == PAIR_LEN && p->prop_open) { + csx_put_property(p); + p->prop_open = false; + } else if (p->depth == PAIR_LEN && p->using_open) { + if (p->using_complex) { + sb_puts(&p->out, "Y\n"); + } else { + csx_put_using(p, &p->pending); + } + p->using_open = false; + } + if (p->depth == SKIP_ONE) { + if (p->group == 'i' && p->i_written) { + sb_puts(&p->out, "E\n"); + } + p->group = 0; + } + if (p->group == 'c' && p->choose_props == p->depth + SKIP_ONE) { + p->choose_props = CBM_NOT_FOUND; + } +} + +static bool csx_ci_suffix(const char *s, const char *sfx) { + size_t n = s ? strlen(s) : 0; + size_t sl = strlen(sfx); + if (n < sl) { + return false; + } + for (size_t i = 0; i < sl; i++) { + if (tolower((unsigned char)s[n - sl + i]) != sfx[i]) { + return false; + } + } + return true; +} + +void cbm_doclink_cs_project_parse_doc(CBMExtractCtx *ctx, const CBMDefinition *def, const char *doc, + uint32_t doc_line) { + /* a project file's comments hold no references to code */ + (void)ctx; + (void)def; + (void)doc; + (void)doc_line; +} + +const char *cbm_doclink_cs_project_scan_scope(CBMExtractCtx *ctx) { + /* The gate is the file's name, before a byte of it is looked at: only a + * *.csproj, *.props or *.targets file can be an MSBuild project file of a + * C# project. Every other XML file -- there are many, and large ones -- + * costs nothing here. */ + bool named_project = csx_ci_suffix(ctx->rel_path, ".csproj"); + if (!ctx->source || !(named_project || csx_ci_suffix(ctx->rel_path, ".props") || + csx_ci_suffix(ctx->rel_path, ".targets"))) { + return NULL; + } + /* A project file past the size a project file has is not read either, + * and its blob says so: what it holds is unknown, not absent. */ + if (ctx->source_len > CSX_MAX_PROJECT_BYTES) { + const char *blob = cbm_arena_strdup(ctx->arena, CBM_DOCLINK_CS_SCOPE_TAG "\nP\t\t>\n"); + if (!blob && ctx->result) { + ctx->result->doc_links.failed = true; /* memory ran out */ + } + return blob; + } + /* a *.csproj marks its directory as a C# project even when it cannot be + * read; a *.props or *.targets file has no blob only when it parses up to + * a root element that is not an MSBuild */ + csx_scan_t p = { + .x = {.src = ctx->source, .n = ctx->source_len > 0 ? (uint32_t)ctx->source_len : 0}, + .out = {.a = ctx->arena}, + .value = {.a = ctx->scratch ? ctx->scratch : ctx->arena}, + .choose_props = CBM_NOT_FOUND}; + static const char bom[] = "\xEF\xBB\xBF"; + if (p.x.n >= 3 && memcmp(p.x.src, bom, 3) == 0) { + p.x.i = 3; + } + bool is_project = false; + bool other_root = false; /* XML whose root element is not */ + bool bad = false; + csx_tok_t t; + for (;;) { + csx_next(&p.x, &t); + if (t.kind == CSX_EOF || t.kind == CSX_BAD) { + bad = t.kind == CSX_BAD || p.depth != 0; + break; + } + if (t.kind == CSX_TEXT) { + if (p.prop_open && p.depth == 3) { + csx_put_text(&p.value, &p.x, t.text, t.raw); + } + continue; + } + if (t.kind == CSX_CLOSE) { + if (p.depth == 0) { + bad = true; + break; + } + p.depth--; + csx_element_end(&p); + continue; + } + if (p.depth > 0) { + csx_element(&p, &t); + continue; + } + if (is_project) { + bad = true; /* a second root element */ + break; + } + if (!csx_is(&p.x, t.name, "Project")) { + other_root = true; + break; /* XML, but no MSBuild file */ + } + is_project = true; + sb_puts(&p.out, CBM_DOCLINK_CS_SCOPE_TAG "\nP"); + csx_put_field(&p.out, &p.x, t.attr[CSX_A_SDK]); + sb_puts(&p.out, "\t-\n"); + if (t.kind == CSX_OPEN) { + p.depth++; + } + } + cs_cost_add(p.x.i, 0); + /* What the scan could not read is unknown, not absent: a file malformed + * anywhere or ending before its root, and a *.csproj that is no MSBuild + * . With no blob, an import of such a file would count as one + * outside the repository, and as the nearest Directory.Build.* it would + * be passed over for the one above, which MSBuild does not read. */ + if (bad || (!is_project && (named_project || !other_root))) { + p.out.len = 0; + sb_puts(&p.out, CBM_DOCLINK_CS_SCOPE_TAG "\nP\t\t!\n"); + } else if (!is_project) { + return NULL; + } +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (cbm_doclink_test_fail_alloc(CBM_DOCLINK_ALLOC_PROJECT)) { + p.out.failed = true; + } +#endif + if (p.out.failed || p.value.failed || !p.out.buf) { + if (ctx->result) { + ctx->result->doc_links.failed = true; /* memory ran out, see above */ + } + return NULL; + } + return p.out.buf; +} diff --git a/internal/cbm/extract_defs.c b/internal/cbm/extract_defs.c index 281b38359..0621382d7 100644 --- a/internal/cbm/extract_defs.c +++ b/internal/cbm/extract_defs.c @@ -1,5 +1,6 @@ #include "cbm.h" -#include "arena.h" // CBMArena, cbm_arena_alloc/strdup/sprintf +#include "arena.h" // CBMArena, cbm_arena_alloc/strdup/sprintf +#include "doclink.h" // cbm_doclink_note_doc_line #include "helpers.h" #include "lang_specs.h" #include "foundation/constants.h" @@ -1361,6 +1362,7 @@ typedef struct { doc_span_t *items; /* trivia directly before the anchor, in source order */ int count; int cap; + bool failed; /* a missing span must not become shared documentation */ bool code_before; /* a non-trivia sibling precedes items[0] */ uint32_t code_erow; /* ... its effective end row */ uint32_t code_eb; /* ... its end byte (Kotlin gap scan) */ @@ -1488,8 +1490,17 @@ static doc_span_t doc_span_of(TSNode n, const char *src, uint8_t kind) { static void doc_push_span(CBMArena *a, doc_trivia_t *t, const doc_span_t *sp) { if (t->count == t->cap) { int ncap = t->cap ? t->cap * DOC_SPAN_GROW : DOC_SPAN_INIT_CAP; - doc_span_t *grown = (doc_span_t *)cbm_arena_alloc(a, (size_t)ncap * sizeof(doc_span_t)); + doc_span_t *grown; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (cbm_doclink_test_fail_alloc(CBM_DOCLINK_ALLOC_SPAN)) { + grown = NULL; + } else +#endif + { + grown = (doc_span_t *)cbm_arena_alloc(a, (size_t)ncap * sizeof(doc_span_t)); + } if (!grown) { + t->failed = true; return; } if (t->count > 0) { @@ -1912,6 +1923,15 @@ static void doc_collect_kotlin(CBMExtractCtx *ctx, TSNode parent, TSNode anchor, } } +/* A doc comment was lost because memory ran out. For a language whose doc + * comments are read for links, the doc-link layer must not then report a + * complete graph (CBMDocLinkArray.failed). */ +static void doc_lost(CBMExtractCtx *ctx) { + if (ctx->result && cbm_doclink_lang_supported(ctx->language)) { + ctx->result->doc_links.failed = true; + } +} + /* Leading trivia of `anchor`, in source order. */ static void doc_collect_trivia(CBMExtractCtx *ctx, TSNode anchor, doc_trivia_t *t) { memset(t, 0, sizeof(*t)); @@ -1928,12 +1948,15 @@ static void doc_collect_trivia(CBMExtractCtx *ctx, TSNode anchor, doc_trivia_t * found = doc_collect_cursor(ctx, parent, anchor, t); } if (!found) { + bool failed = t->failed; memset(t, 0, sizeof(*t)); - return; - } - if (t->count == 0 && ctx->language == CBM_LANG_KOTLIN) { + t->failed = failed; + } else if (t->count == 0 && ctx->language == CBM_LANG_KOTLIN) { doc_collect_kotlin(ctx, parent, anchor, t); } + if (t->failed) { + doc_lost(ctx); + } } /* go/ast CommentGroup.Text: a line comment with no space after the slashes is @@ -2029,11 +2052,15 @@ static const char *doc_run_text(CBMExtractCtx *ctx, const doc_trivia_t *t, int f int kept = 0; bool words = false; size_t total = 0; + uint32_t first_row = 0; for (int k = first; k <= last; k++) { const doc_span_t *sp = &t->items[k]; if (!doc_span_kept(src, sp, go_directives)) { continue; } + if (kept == 0) { + first_row = sp->srow; + } total += (size_t)(sp->eb - sp->sb) + SKIP_ONE; words = words || doc_has_words(src + sp->sb, sp->eb - sp->sb); kept++; @@ -2041,8 +2068,17 @@ static const char *doc_run_text(CBMExtractCtx *ctx, const doc_trivia_t *t, int f if (kept == 0 || !words) { return NULL; } - char *buf = (char *)cbm_arena_alloc(ctx->arena, total + SKIP_ONE); + char *buf; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (cbm_doclink_test_fail_alloc(CBM_DOCLINK_ALLOC_TEXT)) { + buf = NULL; + } else +#endif + { + buf = (char *)cbm_arena_alloc(ctx->arena, total + SKIP_ONE); + } if (!buf) { + doc_lost(ctx); return NULL; } size_t w = 0; @@ -2060,9 +2096,16 @@ static const char *doc_run_text(CBMExtractCtx *ctx, const doc_trivia_t *t, int f buf[w++] = '\n'; } memcpy(buf + w, src + sp->sb, eb - sp->sb); +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + cbm_doclink_test_note_doc_work(eb - sp->sb, 0, 0); +#endif w += eb - sp->sb; } buf[w] = '\0'; + /* Doc-link references need their source lines: the text's line k is the + * source line first_row + k (one comment per line, or a block keeping + * its own newlines). */ + cbm_doclink_note_doc_line(ctx, buf, first_row + SKIP_ONE); return buf; } @@ -2144,12 +2187,19 @@ static const char *doc_from_trivia(CBMExtractCtx *ctx, const doc_trivia_t *t, return doc_run_text(ctx, t, doc_run_first(lang, t, near), near, lang == CBM_LANG_GO); } -static const char *doc_for_anchor(CBMExtractCtx *ctx, TSNode anchor) { +static const char *doc_for_anchor_status(CBMExtractCtx *ctx, TSNode anchor, bool *complete) { doc_trivia_t t; doc_collect_trivia(ctx, anchor, &t); + if (complete) { + *complete = !t.failed; + } return doc_from_trivia(ctx, &t, ts_node_start_point(anchor).row); } +static const char *doc_for_anchor(CBMExtractCtx *ctx, TSNode anchor) { + return doc_for_anchor_status(ctx, anchor, NULL); +} + static bool doc_kind_is(TSNode n, const char *kind) { return !ts_node_is_null(n) && strcmp(ts_node_type(n), kind) == 0; } @@ -2636,11 +2686,19 @@ static const char *extract_docstring(CBMExtractCtx *ctx, TSNode node, const char } /* Doc of a Field, Variable, enum member or Macro (code languages only). */ -static const char *extract_member_docstring(CBMExtractCtx *ctx, TSNode node) { +static const char *extract_member_docstring_status(CBMExtractCtx *ctx, TSNode node, + bool *complete) { if (!doc_lang_member_docs(ctx->language)) { + if (complete) { + *complete = true; + } return NULL; } - return doc_for_anchor(ctx, doc_anchor(ctx, node)); + return doc_for_anchor_status(ctx, doc_anchor(ctx, node), complete); +} + +static const char *extract_member_docstring(CBMExtractCtx *ctx, TSNode node) { + return extract_member_docstring_status(ctx, node, NULL); } /* Go package comment: the comment group touching `package`, directives @@ -7462,8 +7520,8 @@ static void extract_elixir_call(CBMExtractCtx *ctx, TSNode node, const CBMLangSp * from `name` only where a language scopes a variable below the module — Nix, * whose binding names are attrpaths (`a.b.c = …` is name `c`, QN suffix `a.b.c`). * Pass NULL to use `name` for both. */ -static void push_var_def_qn(CBMExtractCtx *ctx, const char *name, const char *qn_name, - TSNode node) { +static void push_var_def_qn_doc(CBMExtractCtx *ctx, const char *name, const char *qn_name, + TSNode node, const char *doc) { if (!name || !name[0] || strcmp(name, "_") == 0) { return; } @@ -7485,10 +7543,18 @@ static void push_var_def_qn(CBMExtractCtx *ctx, const char *name, const char *qn def.start_line = ts_node_start_point(node).row + TS_LINE_OFFSET; def.end_line = ts_node_end_point(node).row + TS_LINE_OFFSET; def.is_exported = cbm_is_exported(name, ctx->language); - def.docstring = extract_member_docstring(ctx, node); + def.docstring = doc; cbm_defs_push(&ctx->result->defs, a, def); } +static void push_var_def_qn(CBMExtractCtx *ctx, const char *name, const char *qn_name, + TSNode node) { + if (!name || !name[0] || strcmp(name, "_") == 0) { + return; + } + push_var_def_qn_doc(ctx, name, qn_name, node, extract_member_docstring(ctx, node)); +} + static void push_var_def(CBMExtractCtx *ctx, const char *name, TSNode node) { push_var_def_qn(ctx, name, NULL, node); } @@ -7558,6 +7624,17 @@ static void extract_csharp_vars(CBMExtractCtx *ctx, TSNode node, CBMArena *a) { push_var_def(ctx, fname, node); return; } + /* All declarators have this field as their documentation anchor. Keep one + * immutable arena string, while each variable retains its own identity: + * the doc-link driver takes the references of that one text once, from + * the first declarator. A text that could not be collected whole is no + * doc of any of them (doc_lost has marked the file's doc links failed); + * it is not looked up again per declarator. */ + bool complete = false; + const char *doc = extract_member_docstring_status(ctx, node, &complete); + if (!complete) { + doc = NULL; + } uint32_t n = ts_node_named_child_count(node); for (uint32_t i = 0; i < n; i++) { TSNode child = ts_node_named_child(node, i); @@ -7573,7 +7650,8 @@ static void extract_csharp_vars(CBMExtractCtx *ctx, TSNode node, CBMArena *a) { id = cbm_find_child_by_kind(decl, "identifier"); } if (!ts_node_is_null(id)) { - push_var_def(ctx, cbm_node_text(a, id, ctx->source), decl); + const char *name = cbm_node_text(a, id, ctx->source); + push_var_def_qn_doc(ctx, name, NULL, decl, doc); } } } diff --git a/internal/cbm/result_compact.c b/internal/cbm/result_compact.c index c97400fec..9691b6c96 100644 --- a/internal/cbm/result_compact.c +++ b/internal/cbm/result_compact.c @@ -394,6 +394,12 @@ static void cr_walk(cr_ctx_t *c, CBMFileResult *r) { cr_str(c, &r->error_msg); cr_str(c, &r->error_ranges); cr_str(c, &r->module_doc); + cr_array(c, (void **)&r->doc_links.items, r->doc_links.count, sizeof(CBMDocLink)); + for (int i = 0; i < r->doc_links.count && r->doc_links.items; i++) { + cr_str(c, &r->doc_links.items[i].source_qn); + cr_str(c, &r->doc_links.items[i].raw); + } + cr_str(c, &r->doc_scope); cr_blob(c, (const void **)&r->source, r->source ? (size_t)r->source_len + SKIP_ONE : 0); } @@ -533,6 +539,7 @@ void cbm_result_compact(CBMFileResult *result) { tmp.infra_bindings.cap = tmp.infra_bindings.count; tmp.channels.cap = tmp.channels.count; tmp.field_types.cap = tmp.field_types.count; + tmp.doc_links.cap = tmp.doc_links.count; /* A composite kept its per-unit results only so shallow-copied strings * stayed valid; every string is now a copy of its own. */ diff --git a/scripts/memory-core-baseline.txt b/scripts/memory-core-baseline.txt index 145338803..a44ce0d2b 100644 --- a/scripts/memory-core-baseline.txt +++ b/scripts/memory-core-baseline.txt @@ -52,7 +52,7 @@ src/git/git_context.c 18 src/main.c 32 src/mcp/compact_out.c 39 src/mcp/index_supervisor.c 8 -src/mcp/mcp.c 778 +src/mcp/mcp.c 777 src/pipeline/artifact.c 38 src/pipeline/fqn.c 23 src/pipeline/lsp_resolve.h 4 diff --git a/src/cli/cli.c b/src/cli/cli.c index df51e723b..db1c15559 100644 --- a/src/cli/cli.c +++ b/src/cli/cli.c @@ -1566,8 +1566,8 @@ static const char skill_content[] = "## Edge Types\n" "CALLS, HTTP_CALLS, ASYNC_CALLS, DATA_FLOWS, IMPORTS, DEFINES, DEFINES_METHOD,\n" "HANDLES, IMPLEMENTS, OVERRIDE, USAGE, CALL_REFERENCE, CONFIGURES, REFERENCES_FILE,\n" - "FILE_CHANGES_WITH, SIMILAR_TO, SEMANTICALLY_RELATED, CONTAINS_FILE, CONTAINS_FOLDER,\n" - "CONTAINS_PACKAGE\n" + "MENTIONS (doc comment -> referenced code), FILE_CHANGES_WITH, SIMILAR_TO,\n" + "SEMANTICALLY_RELATED, CONTAINS_FILE, CONTAINS_FOLDER, CONTAINS_PACKAGE\n" "\n" "## Cypher Examples (for query_graph)\n" "```\n" diff --git a/src/mcp/mcp.c b/src/mcp/mcp.c index 1fe774c71..c799282f1 100644 --- a/src/mcp/mcp.c +++ b/src/mcp/mcp.c @@ -57,7 +57,8 @@ enum { #include "cypher/cypher.h" #include "discover/discover.h" #include "pipeline/pipeline.h" -#include "callable_sig.h" /* cbm_qn_callable_base_len */ +#include "callable_sig.h" /* cbm_qn_callable_base_len */ +#include "pipeline/doc_links.h" /* cbm_doclink_reason_name: the reasons the layer writes */ #include "pipeline/pass_cross_repo.h" #include "git/git_context.h" #include "cli/cli.h" @@ -741,8 +742,9 @@ static const tool_def_t TOOLS[] = { "\"project\"]}"}, {"index_status", - "Project readiness, counts, root, and coverage gaps. diagnostics adds coverage rows; verbose " - "adds Git paths. Best-effort only; verify cited paths with check_index_coverage.", + "Project readiness, counts, root, coverage gaps, and doc_links (doc-comment references: " + "MENTIONS edges, unresolved by reason). diagnostics adds coverage rows; verbose adds Git " + "paths. Best-effort only; verify cited paths with check_index_coverage.", "{\"type\":\"object\",\"properties\":{\"project\":{\"type\":\"string\"}," "\"verbose\":{\"type\":\"boolean\",\"default\":false,\"description\":\"Add worktree/" "shadow Git paths for index-location debugging.\"}," @@ -6177,6 +6179,335 @@ static char *handle_query_graph(cbm_mcp_server_t *srv, const char *args) { return res; } +/* Doc-comment references (index_status.doc_links): MENTIONS edges, the + * unresolved rows per reason, and the layer's status. "error" when the + * generation recorded a failed doc-link layer, when the doc_link_unresolved + * table is missing (an index from before the layer: reindex), or when it + * cannot be read. Samples (diagnostics=full) are rows ordered by reason, + * path and line. */ +#ifdef CBM_ENABLE_TEST_SEAMS +static atomic_int mcp_doc_links_sample_alloc_countdown = ATOMIC_VAR_INIT(CBM_NOT_FOUND); +static atomic_bool mcp_doc_links_sample_alloc_failed = ATOMIC_VAR_INIT(false); + +void cbm_mcp_doc_links_test_fail_sample_alloc_after(int successful_copies) { + atomic_store_explicit(&mcp_doc_links_sample_alloc_failed, false, memory_order_relaxed); + atomic_store_explicit(&mcp_doc_links_sample_alloc_countdown, + successful_copies < 0 ? CBM_NOT_FOUND : successful_copies, + memory_order_relaxed); +} + +bool cbm_mcp_doc_links_test_sample_alloc_failed(void) { + return atomic_load_explicit(&mcp_doc_links_sample_alloc_failed, memory_order_relaxed); +} +#endif + +/* The seam rejects an actual sample string allocation in yyjson's pool, + * independently of when that pool needs another backing allocation. */ +static yyjson_mut_val *doc_links_sample_string(yyjson_mut_doc *doc, const char *source, + size_t length) { + if (!doc || !source) { + return NULL; + } +#ifdef CBM_ENABLE_TEST_SEAMS + int remaining = + atomic_load_explicit(&mcp_doc_links_sample_alloc_countdown, memory_order_relaxed); + while (remaining >= 0) { + int next = remaining == 0 ? CBM_NOT_FOUND : remaining - 1; + if (atomic_compare_exchange_weak_explicit(&mcp_doc_links_sample_alloc_countdown, &remaining, + next, memory_order_relaxed, + memory_order_relaxed)) { + if (remaining == 0) { + atomic_store_explicit(&mcp_doc_links_sample_alloc_failed, true, + memory_order_relaxed); + return NULL; + } + break; + } + } +#endif + return yyjson_mut_strncpy(doc, source, length); +} + +typedef struct { + char text[CBM_DOC_LINK_PREVIEW_LONG_BYTES + 1]; + size_t length; + size_t included; + bool truncated; + bool escaped; +} doc_link_preview_t; + +/* A strict scalar decoder. Invalid input is represented one byte at a time + * by the caller, including incomplete sequences at the end of the source. */ +static size_t doc_link_scalar(const unsigned char *text, size_t length, uint32_t *scalar) { + if (!length) { + return 0; + } + unsigned char first = text[0]; + if (first < 0x80) { + *scalar = first; + return 1; + } + size_t width; + uint32_t value; + if (first >= 0xC2 && first <= 0xDF) { + width = 2; + value = first & 0x1F; + } else if (first >= 0xE0 && first <= 0xEF) { + width = 3; + value = first & 0x0F; + } else if (first >= 0xF0 && first <= 0xF4) { + width = 4; + value = first & 0x07; + } else { + return 0; + } + if (length < width || (first == 0xE0 && text[1] < 0xA0) || (first == 0xED && text[1] > 0x9F) || + (first == 0xF0 && text[1] < 0x90) || (first == 0xF4 && text[1] > 0x8F)) { + return 0; + } + for (size_t i = 1; i < width; i++) { + if ((text[i] & 0xC0) != 0x80) { + return 0; + } + value = (value << 6) | (text[i] & 0x3F); + } + *scalar = value; + return width; +} + +/* `unsigned long` (at least 32 bits by the standard), not uint32_t: the + * vendored yyjson.h has a fallback typedef chain for uint32_t, and an + * analysis that cannot evaluate it takes uint32_t for a 16-bit type and the + * tag block's bounds for out of range. */ +static bool doc_link_visible_escape(unsigned long scalar) { + return scalar < 0x20 || scalar == 0x7F || (scalar >= 0x200B && scalar <= 0x200F) || + (scalar >= 0x202A && scalar <= 0x202E) || (scalar >= 0x2066 && scalar <= 0x2069) || + (scalar >= 0xE0000 && scalar <= 0xE007F); +} + +/* Output tokens are complete scalars or visible escapes. Lookahead bytes do + * not count as included source, and the shared transport encoder is not used. */ +static bool doc_link_format_preview(const cbm_doc_link_preview_text_t *source, size_t budget, + doc_link_preview_t *out) { + memset(out, 0, sizeof(*out)); + if (!source->text || budget >= sizeof(out->text) || + source->length > budget + CBM_DOC_LINK_PREVIEW_LOOKAHEAD || + source->original_bytes < source->length) { + return false; + } + const unsigned char *text = (const unsigned char *)source->text; + bool reserved = (source->length >= 7 && memcmp(text, "@bytes:", 7) == 0) || + (source->length >= 6 && memcmp(text, "@utf8:", 6) == 0); + while (out->included < source->length && out->length < budget) { + const unsigned char *at = text + out->included; + uint32_t scalar = 0; + size_t consumed = doc_link_scalar(at, source->length - out->included, &scalar); + char escape[16]; + const char *token = (const char *)at; + size_t bytes = consumed; + bool escaped = false; + if (!consumed) { + int written = snprintf(escape, sizeof(escape), "\\x%02X", (unsigned int)at[0]); + if (written < 0 || (size_t)written >= sizeof(escape)) { + return false; + } + consumed = 1; + bytes = (size_t)written; + token = escape; + escaped = true; + } else if (scalar == '\\') { + token = "\\\\"; + bytes = 2; + escaped = true; + } else if (doc_link_visible_escape(scalar) || (out->included == 0 && reserved)) { + int written = snprintf(escape, sizeof(escape), "\\u{%04X}", (unsigned int)scalar); + if (written < 0 || (size_t)written >= sizeof(escape)) { + return false; + } + bytes = (size_t)written; + token = escape; + escaped = true; + } + if (bytes > budget - out->length) { + break; + } + memcpy(out->text + out->length, token, bytes); + out->length += bytes; + out->included += consumed; + out->escaped = out->escaped || escaped; + } + out->text[out->length] = '\0'; + out->truncated = out->included < source->original_bytes; + return true; +} + +/* Neither array is attached to the report until every row and metadata entry + * has been constructed. The JSON document owns all intermediate allocations. */ +static bool doc_link_preview_samples(yyjson_mut_doc *doc, const cbm_doc_link_preview_row_t *rows, + int count, yyjson_mut_val **samples_out, + yyjson_mut_val **metadata_out) { + yyjson_mut_val *samples = yyjson_mut_arr(doc); + yyjson_mut_val *metadata = yyjson_mut_arr(doc); + if (!samples || !metadata) { + return false; + } + static const char *const names[] = {"rel_path", "syntax", "raw", "reason"}; + int emitted = 0; + for (int i = 0; i < count; i++) { + if (!rows[i].rel_path.original_bytes) { + continue; + } + const cbm_doc_link_preview_text_t *fields[] = {&rows[i].rel_path, &rows[i].syntax, + &rows[i].raw, &rows[i].reason}; + yyjson_mut_val *sample = yyjson_mut_obj(doc); + if (!sample) { + return false; + } + for (int field = 0; field < 4; field++) { + size_t budget = field == 0 || field == 2 ? CBM_DOC_LINK_PREVIEW_LONG_BYTES + : CBM_DOC_LINK_PREVIEW_SHORT_BYTES; + doc_link_preview_t preview; + if (!doc_link_format_preview(fields[field], budget, &preview)) { + return false; + } + yyjson_mut_val *value = doc_links_sample_string(doc, preview.text, preview.length); + if (!value || !yyjson_mut_obj_add_val(doc, sample, names[field], value) || + (field == 0 && !yyjson_mut_obj_add_int(doc, sample, "line", rows[i].line))) { + return false; + } + if (preview.truncated || preview.escaped) { + yyjson_mut_val *entry = yyjson_mut_obj(doc); + if (!entry || !yyjson_mut_obj_add_int(doc, entry, "sample_index", emitted) || + !yyjson_mut_obj_add_str(doc, entry, "field", names[field]) || + !yyjson_mut_obj_add_uint(doc, entry, "original_bytes", + fields[field]->original_bytes) || + !yyjson_mut_obj_add_uint(doc, entry, "included_source_bytes", + preview.included) || + !yyjson_mut_obj_add_bool(doc, entry, "truncated", preview.truncated) || + !yyjson_mut_obj_add_bool(doc, entry, "escaped", preview.escaped) || + !yyjson_mut_arr_append(metadata, entry)) { + return false; + } + } + } + if (!yyjson_mut_arr_append(samples, sample)) { + return false; + } + emitted++; + } + *samples_out = samples; + *metadata_out = yyjson_mut_arr_size(metadata) ? metadata : NULL; + return true; +} + +/* True for a reason the doc-link layer writes. The table is read back from + * the database: any other text in it was not written by this layer, and + * never becomes a key of the report. */ +static bool doc_link_reason_known(const char *reason) { + for (int r = 0; r < CBM_DOCLINK_REASON_COUNT; r++) { + if (strcmp(reason, cbm_doclink_reason_name(r)) == 0) { + return true; + } + } + return false; +} + +static bool add_doc_links_report(yyjson_mut_doc *doc, yyjson_mut_val *root, cbm_store_t *store, + const char *project, bool with_samples) { + int mentions = cbm_store_count_edges_by_type(store, project, "MENTIONS"); + cbm_doc_link_reason_count_t *reasons = NULL; + int nreasons = 0; + cbm_doc_link_row_t *unused_samples = NULL; + int unused_count = 0; + bool present = false; + int rc = cbm_store_doc_links_summary(store, project, &reasons, &nreasons, &unused_samples, + &unused_count, 0, &present); + cbm_doc_link_preview_row_t *rows = NULL; + int count = 0; + yyjson_mut_val *samples = NULL; + yyjson_mut_val *metadata = NULL; + bool sample_failed = with_samples && rc != CBM_STORE_OK; + if (with_samples && rc == CBM_STORE_OK) { + bool preview_present = false; + int preview_rc = + cbm_store_doc_links_preview(store, project, &rows, &count, &preview_present); + sample_failed = preview_rc != CBM_STORE_OK || preview_present != present; + if (!sample_failed) { + sample_failed = !doc_link_preview_samples(doc, rows, count, &samples, &metadata); + } + } + bool error = rc != CBM_STORE_OK || mentions < 0 || !present || sample_failed; + bool built = false; + int64_t unrecognized = 0; + yyjson_mut_val *dl = yyjson_mut_obj(doc); + yyjson_mut_val *unresolved = yyjson_mut_obj(doc); + if (!dl || !unresolved) { + goto cleanup; + } + for (int i = 0; i < nreasons; i++) { + if (strcmp(reasons[i].reason, "error") == 0) { + error = true; + continue; + } + if (!doc_link_reason_known(reasons[i].reason)) { + /* rows this layer did not write: counted under one fixed key */ + unrecognized += reasons[i].count; + error = true; + continue; + } + yyjson_mut_val *key = yyjson_mut_strcpy(doc, reasons[i].reason); + yyjson_mut_val *number = yyjson_mut_int(doc, reasons[i].count); + if (!key || !number || !yyjson_mut_obj_add(unresolved, key, number)) { + goto cleanup; + } + } + if (unrecognized > 0 && + !yyjson_mut_obj_add_int(doc, unresolved, "unrecognized_reason", unrecognized)) { + goto cleanup; + } + if (!yyjson_mut_obj_add_int(doc, dl, "mentions", mentions < 0 ? 0 : mentions) || + !yyjson_mut_obj_add_val(doc, dl, "unresolved", unresolved) || + !yyjson_mut_obj_add_str(doc, dl, "status", error ? "error" : "ok")) { + goto cleanup; + } + if (error) { + const char *hint = + sample_failed && rc == CBM_STORE_OK + ? "Doc-link samples could not be prepared; retry index_status." + : (!present && rc == CBM_STORE_OK + ? "This index predates doc-comment links; re-run " + "index_repository(repo_path=...)." + : "The doc-link layer failed or its table could not be read; re-run " + "index_repository(repo_path=...)."); + if (!yyjson_mut_obj_add_str(doc, dl, "hint", hint)) { + goto cleanup; + } + } + if (with_samples && !sample_failed) { + if (!yyjson_mut_obj_add_val(doc, dl, "samples", samples)) { + goto cleanup; + } + if (metadata && + (!yyjson_mut_obj_add_val(doc, dl, "samples_preview", metadata) || + !yyjson_mut_obj_add_str(doc, dl, "samples_preview_note", + "Sample text is shown as previews; complete values remain " + "in doc_link_unresolved."))) { + goto cleanup; + } + } + /* A failed append leaves this whole report unattached. The caller discards + * the document and uses its existing MCP allocation-error response. */ + built = yyjson_mut_obj_add_val(doc, root, "doc_links", dl); +cleanup: + if (error && (rc != CBM_STORE_OK || sample_failed)) { + cbm_log_warn("index_status.doc_links", "project", project, "reason", "read_failed"); + } + cbm_store_free_doc_link_reasons(reasons, nreasons); + cbm_store_free_doc_links(unused_samples, unused_count); + cbm_store_free_doc_link_previews(rows, count); + return built; +} + /* Indexing-coverage report (#963), attached to index_status: the best-effort * signal from the separate index_coverage table (coverage is metadata ABOUT * the graph, stored outside it). Full per-project list, capped generously. */ @@ -6872,10 +7203,17 @@ static char *handle_index_status(cbm_mcp_server_t *srv, const char *args) { free(diagnostics); yyjson_mut_doc *doc = yyjson_mut_doc_new(NULL); - yyjson_mut_val *root = yyjson_mut_obj(doc); - yyjson_mut_doc_set_root(doc, root); + yyjson_mut_val *root = doc ? yyjson_mut_obj(doc) : NULL; + /* A failed allocation or report answers with the MCP allocation-error + * response, through the one exit below. */ + bool failed = !root; + if (root) { + yyjson_mut_doc_set_root(doc, root); + } - if (project) { + if (failed) { + /* nothing to report into */ + } else if (project) { int nodes = cbm_store_count_nodes(store, project); int edges = cbm_store_count_edges(store, project); /* A negative count is a failed read (CBM_STORE_ERR), not a small @@ -6902,10 +7240,14 @@ static char *handle_index_status(cbm_mcp_server_t *srv, const char *args) { } add_coverage_report(doc, root, store, project, have_proj_info ? proj_info.indexed_at : NULL, coverage_samples); + bool doc_links_built = + add_doc_links_report(doc, root, store, project, coverage_samples == COVERAGE_FILE_CAP); safe_str_free(&proj_info.name); safe_str_free(&proj_info.indexed_at); safe_str_free(&proj_info.root_path); - if (counts_unreadable) { + if (!doc_links_built) { + failed = true; + } else if (counts_unreadable) { const char *hint; if (nodes < 0 && edges < 0) { hint = "The nodes and edges tables could not be read; the database may be " @@ -6928,7 +7270,7 @@ static char *handle_index_status(cbm_mcp_server_t *srv, const char *args) { yyjson_mut_obj_add_str(doc, root, "status", "no_project"); } - char *json = yy_doc_to_str(doc); + char *json = failed ? NULL : yy_doc_to_str(doc); yyjson_mut_doc_free(doc); free(project); diff --git a/src/mcp/mcp_internal.h b/src/mcp/mcp_internal.h index 784e07145..ead1fca38 100644 --- a/src/mcp/mcp_internal.h +++ b/src/mcp/mcp_internal.h @@ -13,6 +13,11 @@ typedef bool (*cbm_mcp_quarantine_test_hook_fn)(void *context, const char *step); typedef bool (*cbm_mcp_command_test_hook_fn)(void *context, const char *command); #ifdef CBM_ENABLE_TEST_SEAMS +/* One-shot sample JSON string-copy allocation failure, not a backing-pool + * growth event: 0 is next, 4 is fifth, -1 disables. Setting clears consumption. */ +void cbm_mcp_doc_links_test_fail_sample_alloc_after(int successful_copies); +bool cbm_mcp_doc_links_test_sample_alloc_failed(void); + typedef void (*cbm_mcp_auto_index_count_test_hook_fn)(void *context); #endif diff --git a/src/pipeline/doc_links.c b/src/pipeline/doc_links.c new file mode 100644 index 000000000..8ac85fcb5 --- /dev/null +++ b/src/pipeline/doc_links.c @@ -0,0 +1,845 @@ +/* + * doc_links.c — doc-comment references -> MENTIONS edges: the language- + * independent core (run state, per-file resolution, edge collapse, rows). + * See doc_links.h; the C# resolver is doc_links_cs.c. + */ +#include "pipeline/doc_links.h" + +#include "doclink.h" +#include "foundation/arena.h" +#include "foundation/compat.h" /* cbm_clock_gettime */ +#include "foundation/constants.h" +#include "foundation/log.h" +#include "foundation/mem_core.h" +#include "result_spill.h" /* a parked result's header: is there a scope to read back? */ +#include "yyjson/yyjson.h" + +#include +#include +#include +#include +#include + +/* ── Reasons ─────────────────────────────────────────────────────── */ + +static const char *const DOCLINK_REASON_NAMES[CBM_DOCLINK_REASON_COUNT] = { + [CBM_DOCLINK_REASON_MISSING] = "missing", + [CBM_DOCLINK_REASON_AMBIGUOUS] = "ambiguous", + [CBM_DOCLINK_REASON_EXTERNAL] = "external", + [CBM_DOCLINK_REASON_TEST_ONLY] = "test_only_target", + [CBM_DOCLINK_REASON_NOT_INDEXED] = "not_indexed", + [CBM_DOCLINK_REASON_GRAPH_GAP] = "graph_gap", + [CBM_DOCLINK_REASON_UNPARSEABLE] = "unparseable", + [CBM_DOCLINK_REASON_BELOW_BAR] = "below_bar_tier", +}; + +const char *cbm_doclink_reason_name(int reason) { + if (reason < 0 || reason >= CBM_DOCLINK_REASON_COUNT) { + return "missing"; + } + return DOCLINK_REASON_NAMES[reason]; +} + +/* ── Resolver table ──────────────────────────────────────────────── */ + +/* One pointer per language with a resolver: the one line a language leg adds + * to this file. */ +static const cbm_doclink_resolver_t *const DOCLINK_RESOLVERS[] = { + &cbm_doclink_cs_resolver, +}; + +enum { DOCLINK_RESOLVER_COUNT = sizeof(DOCLINK_RESOLVERS) / sizeof(DOCLINK_RESOLVERS[0]) }; + +static bool resolver_has_lang(const cbm_doclink_resolver_t *R, CBMLanguage lang) { + for (int i = 0; i < R->lang_count && i < CBM_DOCLINK_RESOLVER_LANGS; i++) { + if (R->langs[i] == lang) { + return true; + } + } + return false; +} + +static int resolver_slot(CBMLanguage lang) { + for (int i = 0; i < DOCLINK_RESOLVER_COUNT; i++) { + if (resolver_has_lang(DOCLINK_RESOLVERS[i], lang)) { + return i; + } + } + return CBM_NOT_FOUND; +} + +/* True when the scope blob's tag line is `tag`. */ +static bool scope_has_tag(const char *scope, const char *tag) { + if (!scope || !tag) { + return false; + } + size_t tl = strlen(tag); + return strncmp(scope, tag, tl) == 0 && (scope[tl] == '\n' || scope[tl] == '\0'); +} + +/* The resolver whose scope blobs carry this blob's tag line; NULL for none. */ +static const cbm_doclink_resolver_t *resolver_of_scope(const char *scope) { + for (int i = 0; i < DOCLINK_RESOLVER_COUNT; i++) { + if (scope_has_tag(scope, DOCLINK_RESOLVERS[i]->scope_tag)) { + return DOCLINK_RESOLVERS[i]; + } + } + return NULL; +} + +/* ── Incremental scope rules ─────────────────────────────────────── */ + +bool cbm_doclinks_is_scope_input(const char *rel_path) { + for (int i = 0; rel_path && i < DOCLINK_RESOLVER_COUNT; i++) { + if (DOCLINK_RESOLVERS[i]->scope_input && DOCLINK_RESOLVERS[i]->scope_input(rel_path)) { + return true; + } + } + return false; +} + +int cbm_doclinks_scope_delta(const char *stored, const char *fresh, cbm_doclink_name_fn removed, + void *ud) { + if (!stored && !fresh) { + return CBM_DOCLINK_DELTA_LOCAL; + } + if (!stored || !fresh) { + return CBM_DOCLINK_DELTA_GLOBAL; /* a scope appeared, or left with its file */ + } + const cbm_doclink_resolver_t *R = resolver_of_scope(stored); + if (!R || R != resolver_of_scope(fresh) || !R->scope_delta) { + return CBM_DOCLINK_DELTA_GLOBAL; + } + return R->scope_delta(stored, fresh, removed, ud); +} + +const char *cbm_doclinks_storable_scope(const char *scope, char **owned) { + *owned = NULL; + const cbm_doclink_resolver_t *R = resolver_of_scope(scope); + if (!R || !R->scope_accepted || !R->rejected_scope) { + return scope; + } + int accepted = R->scope_accepted(scope); + if (accepted < 0) { + return NULL; + } + if (accepted) { + return scope; + } + *owned = R->rejected_scope(scope); + return *owned; +} + +/* ── Run state ───────────────────────────────────────────────────── */ + +typedef struct { + cbm_doc_link_row_t *items; + int count; +} doclink_file_rows_t; + +struct cbm_doclinks { + const cbm_file_info_t *files; /* borrowed: the run's file list */ + int file_count; + const char *project; /* borrowed: names the File node of a file-level source */ + void *index[DOCLINK_RESOLVER_COUNT]; + doclink_file_rows_t *rows; /* per run file */ + CBMArena arena; /* scope copies handed to the resolvers */ + _Atomic bool failed; + _Atomic int64_t edges; + _Atomic int64_t mentions; + _Atomic int64_t self_mentions; + _Atomic int64_t local_refs; + _Atomic int64_t no_source; + _Atomic int64_t reasons[CBM_DOCLINK_REASON_COUNT]; + /* what the layer cost, for the doc_links.timing log line (never a gate) */ + int64_t build_ns; /* the per-language indexes */ + _Atomic int64_t resolve_ns; /* summed over the files, i.e. over the workers */ + _Atomic int64_t resolved_files; /* files that had references */ + _Atomic int64_t resolved_tokens; /* references handed to a resolver */ +}; + +static const char *itoa64(int64_t v, char *buf, size_t n) { + snprintf(buf, n, "%lld", (long long)v); + return buf; +} + +static int64_t doclinks_now_ns(void) { + struct timespec ts; + cbm_clock_gettime(CLOCK_MONOTONIC, &ts); + return ((int64_t)ts.tv_sec * 1000000000LL) + (int64_t)ts.tv_nsec; +} + +static bool want_doc_scope(const CBMFileResult *header) { + return header->doc_scope != NULL; /* a parked header's pointer is only a presence bit */ +} + +static int file_cmp(const void *a, const void *b) { + return strcmp(((const cbm_doclink_file_t *)a)->rel_path, + ((const cbm_doclink_file_t *)b)->rel_path); +} + +/* True when file `i`'s result is parked on disk with a scope: an acquire that + * handed nothing out then means a failed read, not "no scope". A header that + * cannot be peeked counts as one. */ +static bool parked_scope_unread(const cbm_pipeline_ctx_t *ctx, CBMFileResult **cache, int i) { + if ((cache && cache[i]) || !ctx || !ctx->spill || !cbm_result_spill_has(ctx->spill, i)) { + return false; + } + CBMFileResult header; + return !cbm_result_spill_peek_header(ctx->spill, i, &header) || want_doc_scope(&header); +} + +/* The scope of run file `i`, copied into the run's arena; NULL when the file + * has none. *why is set when it has one that cannot be had (the copy fails, + * or its parked result does not load): a scope dropped silently would take + * the file's declarations out of the index, and references to them would + * read as missing. */ +static const char *run_file_scope(cbm_doclinks_t *dl, const cbm_pipeline_ctx_t *ctx, + CBMFileResult **cache, int i, const char **why) { + bool loaded = false; + CBMFileResult *r = cbm_pipeline_result_acquire(ctx, cache, i, want_doc_scope, &loaded); + const char *scope = NULL; + if (r && r->doc_scope) { + scope = cbm_arena_strdup(&dl->arena, r->doc_scope); + if (!scope) { + *why = "alloc"; + } + } else if (!r && parked_scope_unread(ctx, cache, i)) { + *why = "scope_unreadable"; + } + cbm_pipeline_result_release(r, loaded); + return scope; +} + +/* Build one language's index over this run's files of that language plus the + * base scopes tagged for it. Returns false, with *why, when a scope cannot be + * read or copied or the index cannot be built: the caller fails the layer. */ +static bool build_language(cbm_doclinks_t *dl, int slot, const cbm_pipeline_ctx_t *ctx, + const cbm_file_info_t *files, int file_count, CBMFileResult **cache, + const cbm_doclink_scope_t *base, int base_count, const cbm_gbuf_t *graph, + const char **why) { + const cbm_doclink_resolver_t *R = DOCLINK_RESOLVERS[slot]; + int cap = 0; + for (int i = 0; i < file_count; i++) { + cap += resolver_has_lang(R, files[i].language); + } + for (int i = 0; i < base_count; i++) { + cap += scope_has_tag(base[i].scope, R->scope_tag); + } + if (cap == 0) { + return true; + } + cbm_doclink_file_t *lf = + (cbm_doclink_file_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, (size_t)cap * sizeof(*lf)); + if (!lf) { + *why = "alloc"; + return false; + } + const char *failed = NULL; + int n = 0; + for (int i = 0; i < file_count; i++) { + if (!resolver_has_lang(R, files[i].language)) { + continue; + } + const char *scope = run_file_scope(dl, ctx, cache, i, &failed); + lf[n++] = + (cbm_doclink_file_t){.rel_path = files[i].rel_path, .scope = scope, .run_file = i}; + } + for (int i = 0; i < base_count; i++) { + if (!scope_has_tag(base[i].scope, R->scope_tag)) { + continue; + } + const char *rel_path = cbm_arena_strdup(&dl->arena, base[i].rel_path); + const char *scope = cbm_arena_strdup(&dl->arena, base[i].scope); + if (!rel_path || !scope) { + failed = "alloc"; + } + lf[n++] = + (cbm_doclink_file_t){.rel_path = rel_path, .scope = scope, .run_file = CBM_NOT_FOUND}; + } + if (failed) { + cbm_free(CBM_MEM_CLASS_OTHER, lf); + *why = failed; + return false; + } + qsort(lf, (size_t)n, sizeof(*lf), file_cmp); + cbm_doclink_build_in_t in = { + .ctx = ctx, .graph = graph, .files = lf, .file_count = n, .run_file_count = file_count}; + dl->index[slot] = R->build(&in); + cbm_free(CBM_MEM_CLASS_OTHER, lf); + if (!dl->index[slot]) { + *why = "index"; + } + return dl->index[slot] != NULL; +} + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +static atomic_bool doclinks_test_fail_build; + +void cbm_doclinks_test_fail_build_once(void) { + atomic_store(&doclinks_test_fail_build, true); +} +#endif + +cbm_doclinks_t *cbm_doclinks_build(const cbm_pipeline_ctx_t *ctx, const cbm_file_info_t *files, + int file_count, CBMFileResult **cache, + const cbm_doclink_scope_t *base, int base_count, + const cbm_gbuf_t *graph) { +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (atomic_exchange(&doclinks_test_fail_build, false)) { + cbm_log_error("doc_links.error", "phase", "build", "reason", "alloc"); + return NULL; + } +#endif + cbm_doclinks_t *dl = (cbm_doclinks_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*dl)); + if (!dl) { + cbm_log_error("doc_links.error", "phase", "build", "reason", "alloc"); + return NULL; + } + dl->files = files; + dl->file_count = file_count; + dl->project = ctx ? ctx->project_name : NULL; + dl->rows = file_count > 0 + ? (doclink_file_rows_t *)cbm_calloc( + CBM_MEM_CLASS_OTHER, (size_t)file_count * sizeof(doclink_file_rows_t)) + : NULL; + cbm_arena_init(&dl->arena); + if (file_count > 0 && !dl->rows) { + cbm_log_error("doc_links.error", "phase", "build", "reason", "alloc"); + cbm_doclinks_free(dl); + return NULL; + } + int64_t started = doclinks_now_ns(); + for (int s = 0; s < DOCLINK_RESOLVER_COUNT; s++) { + const char *why = "alloc"; + if (!build_language(dl, s, ctx, files, file_count, cache, base, base_count, graph, &why)) { + cbm_log_error("doc_links.error", "phase", "index_build", "reason", why); + cbm_doclinks_free(dl); + return NULL; + } + } + dl->build_ns = doclinks_now_ns() - started; + return dl; +} + +/* ── Per-file resolution ─────────────────────────────────────────── */ + +typedef struct { + int64_t src; + int64_t tgt; + uint32_t line; + uint16_t syntax; + bool exact; +} doclink_mention_t; + +static int mention_cmp(const void *a, const void *b) { + const doclink_mention_t *x = (const doclink_mention_t *)a; + const doclink_mention_t *y = (const doclink_mention_t *)b; + if (x->src != y->src) { + return x->src < y->src ? -1 : 1; + } + if (x->tgt != y->tgt) { + return x->tgt < y->tgt ? -1 : 1; + } + if (x->line != y->line) { + return x->line < y->line ? -1 : 1; + } + return (int)x->syntax - (int)y->syntax; +} + +/* Row memory is CBM_MEM_CLASS_STORE everywhere (the store reads and frees the + * same rows): cbm_store_free_doc_links is its one release. */ +static char *dup_or_empty(const char *s) { + return cbm_mem_strdup(CBM_MEM_CLASS_STORE, s ? s : ""); +} + +static void mark_failed(cbm_doclinks_t *dl, const char *rel, const char *why) { + if (!atomic_exchange(&dl->failed, true)) { + cbm_log_error("doc_links.error", "phase", "resolve", "path", rel ? rel : "", "reason", why); + } +} + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +static _Atomic int doclinks_test_edge_fail_after; + +void cbm_doclinks_test_fail_edge_insert_after(int nth) { + atomic_store(&doclinks_test_edge_fail_after, nth > 0 ? nth : 0); +} +#endif + +/* Insert one MENTIONS edge; 0 when it could not be stored. */ +static int64_t doclinks_insert_edge(cbm_gbuf_t *gb, int64_t src, int64_t tgt, const char *props) { +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + int n = atomic_load(&doclinks_test_edge_fail_after); + while (n > 0) { + if (atomic_compare_exchange_weak(&doclinks_test_edge_fail_after, &n, n - SKIP_ONE)) { + if (n == SKIP_ONE) { + return 0; + } + break; + } + } +#endif + return cbm_gbuf_insert_edge(gb, src, tgt, "MENTIONS", props); +} + +/* Emit one MENTIONS edge per (source, target): first line, its syntax, the + * mention count, and tier exact when any mention bound exactly. A failed + * insert fails the layer: the edge is not there. */ +static void emit_mentions(cbm_doclinks_t *dl, const char *rel, doclink_mention_t *m, int n, + cbm_gbuf_t *edge_out) { + qsort(m, (size_t)n, sizeof(*m), mention_cmp); + int i = 0; + while (i < n) { + int j = i; + bool exact = false; + while (j < n && m[j].src == m[i].src && m[j].tgt == m[i].tgt) { + exact = exact || m[j].exact; + j++; + } + char props[CBM_SZ_256]; + snprintf(props, sizeof(props), + "{\"via\":\"doc_comment\",\"syntax\":\"%s\",\"tier\":\"%s\",\"line\":%u," + "\"count\":%d}", + cbm_doclink_syntax_name(m[i].syntax), exact ? "exact" : "unique", m[i].line, + j - i); + if (doclinks_insert_edge(edge_out, m[i].src, m[i].tgt, props) == 0) { + mark_failed(dl, rel, "alloc"); /* an edge that is not there is not counted */ + } else { + atomic_fetch_add_explicit(&dl->edges, 1, memory_order_relaxed); + } + i = j; + } +} + +void cbm_doclinks_resolve_file(cbm_doclinks_t *dl, int file_idx, const CBMFileResult *result, + const cbm_gbuf_t *graph, cbm_gbuf_t *edge_out) { + if (!dl || !result || file_idx < 0 || file_idx >= dl->file_count) { + return; + } + if (result->doc_links.failed) { + /* the extraction lost doc-link data of this file to memory */ + mark_failed(dl, dl->files[file_idx].rel_path, "extract"); + } + if (result->doc_links.count == 0 || !result->doc_links.items) { + return; + } + int64_t started = doclinks_now_ns(); + const cbm_file_info_t *fi = &dl->files[file_idx]; + int slot = resolver_slot(fi->language); + const cbm_doclink_resolver_t *R = slot >= 0 ? DOCLINK_RESOLVERS[slot] : NULL; + const void *index = slot >= 0 ? dl->index[slot] : NULL; + int n = result->doc_links.count; + doclink_mention_t *mentions = + (doclink_mention_t *)cbm_alloc(CBM_MEM_CLASS_OTHER, (size_t)n * sizeof(*mentions)); + cbm_doc_link_row_t *rows = + (cbm_doc_link_row_t *)cbm_calloc(CBM_MEM_CLASS_STORE, (size_t)n * sizeof(*rows)); + if (!mentions || !rows) { + cbm_free(CBM_MEM_CLASS_OTHER, mentions); + cbm_free(CBM_MEM_CLASS_STORE, rows); + mark_failed(dl, fi->rel_path, "alloc"); + return; + } + int nm = 0; + int nr = 0; + /* the file's own node: looked up once, and only when a reference of the + * file-level doc resolves */ + const cbm_gbuf_node_t *file_node = NULL; + bool file_node_looked_up = false; + void *state = (R && index && R->file_begin) ? R->file_begin(index, file_idx) : NULL; + for (int i = 0; i < n; i++) { + const CBMDocLink *link = &result->doc_links.items[i]; + cbm_doclink_outcome_t out = {.kind = CBM_DOCLINK_UNRESOLVED, + .reason = CBM_DOCLINK_REASON_MISSING}; + if (cbm_doclink_syntax_is_external(link->syntax)) { + out.reason = CBM_DOCLINK_REASON_EXTERNAL; + } else if (!R || !index) { + continue; /* no resolver for this language: nothing to say */ + } else { + R->resolve(index, state, file_idx, link, graph, &out); + } + if (out.kind == CBM_DOCLINK_LOCAL) { + atomic_fetch_add_explicit(&dl->local_refs, 1, memory_order_relaxed); + continue; + } + if (out.kind == CBM_DOCLINK_EDGE && out.target) { + const cbm_gbuf_node_t *src = NULL; + if (link->flags & CBM_DOCLINK_FLAG_FILE) { + /* written in the file's own doc: the source is the File node */ + if (!file_node_looked_up) { + file_node = cbm_pipeline_file_node(graph, dl->project, fi->rel_path); + file_node_looked_up = true; + } + src = file_node; + } else { + src = cbm_gbuf_find_by_qn(graph, link->source_qn); + } + if (!src) { + atomic_fetch_add_explicit(&dl->no_source, 1, memory_order_relaxed); + continue; + } + if (src->id == out.target->id) { + atomic_fetch_add_explicit(&dl->mentions, 1, memory_order_relaxed); + atomic_fetch_add_explicit(&dl->self_mentions, 1, memory_order_relaxed); + continue; + } + if (cbm_doclink_syntax_ships(link->syntax)) { + atomic_fetch_add_explicit(&dl->mentions, 1, memory_order_relaxed); + mentions[nm++] = (doclink_mention_t){.src = src->id, + .tgt = out.target->id, + .line = link->line, + .syntax = link->syntax, + .exact = out.exact}; + continue; + } + /* the ship gate: resolved, but this link family is below the + * bar -- a row, not an edge */ + out.reason = CBM_DOCLINK_REASON_BELOW_BAR; + } + int reason = out.reason; + if (reason < 0 || reason >= CBM_DOCLINK_REASON_COUNT) { + reason = CBM_DOCLINK_REASON_MISSING; + } + cbm_doc_link_row_t *row = &rows[nr]; + row->rel_path = dup_or_empty(fi->rel_path); + row->line = (int)link->line; + row->syntax = dup_or_empty(cbm_doclink_syntax_name(link->syntax)); + row->raw = dup_or_empty(link->raw); + row->reason = dup_or_empty(cbm_doclink_reason_name(reason)); + nr++; + if (!row->rel_path || !row->syntax || !row->raw || !row->reason) { + mark_failed(dl, fi->rel_path, "alloc"); + break; + } + atomic_fetch_add_explicit(&dl->reasons[reason], 1, memory_order_relaxed); + } + if (state && R->file_end) { + R->file_end(state); + } + if (nm > 0) { + emit_mentions(dl, fi->rel_path, mentions, nm, edge_out); + } + cbm_free(CBM_MEM_CLASS_OTHER, mentions); + atomic_fetch_add_explicit(&dl->resolve_ns, doclinks_now_ns() - started, memory_order_relaxed); + atomic_fetch_add_explicit(&dl->resolved_files, 1, memory_order_relaxed); + atomic_fetch_add_explicit(&dl->resolved_tokens, n, memory_order_relaxed); + if (nr == 0) { + cbm_free(CBM_MEM_CLASS_STORE, rows); + return; + } + dl->rows[file_idx].items = rows; + dl->rows[file_idx].count = nr; +} + +/* ── Rows ────────────────────────────────────────────────────────── */ + +void cbm_doclinks_take_rows(cbm_doclinks_t *dl, cbm_doc_link_row_t **rows, int *count, + bool *failed) { + *rows = NULL; + *count = 0; + if (failed) { + *failed = dl ? atomic_load(&dl->failed) : true; + } + if (!dl) { + return; + } + int total = 0; + for (int i = 0; i < dl->file_count; i++) { + total += dl->rows[i].count; + } + cbm_doc_link_row_t *all = + total > 0 + ? (cbm_doc_link_row_t *)cbm_alloc(CBM_MEM_CLASS_STORE, (size_t)total * sizeof(*all)) + : NULL; + if (total > 0 && !all) { + mark_failed(dl, NULL, "alloc"); + if (failed) { + *failed = true; + } + return; /* the per-file rows are released by cbm_doclinks_free */ + } + int w = 0; + for (int i = 0; i < dl->file_count; i++) { + if (dl->rows[i].count > 0) { + memcpy(all + w, dl->rows[i].items, (size_t)dl->rows[i].count * sizeof(*all)); + w += dl->rows[i].count; + } + cbm_free(CBM_MEM_CLASS_STORE, dl->rows[i].items); + dl->rows[i].items = NULL; + dl->rows[i].count = 0; + } + *rows = all; + *count = w; + char b[8][CBM_SZ_32]; + cbm_log_info("doc_links.done", "edges", itoa64(atomic_load(&dl->edges), b[0], sizeof(b[0])), + "mentions", itoa64(atomic_load(&dl->mentions), b[1], sizeof(b[1])), "self", + itoa64(atomic_load(&dl->self_mentions), b[2], sizeof(b[2])), "local", + itoa64(atomic_load(&dl->local_refs), b[3], sizeof(b[3])), "unresolved", + itoa64(w, b[4], sizeof(b[4])), "no_source", + itoa64(atomic_load(&dl->no_source), b[5], sizeof(b[5]))); + cbm_log_info( + "doc_links.unresolved", "missing", + itoa64(atomic_load(&dl->reasons[CBM_DOCLINK_REASON_MISSING]), b[0], sizeof(b[0])), + "ambiguous", + itoa64(atomic_load(&dl->reasons[CBM_DOCLINK_REASON_AMBIGUOUS]), b[1], sizeof(b[1])), + "external", + itoa64(atomic_load(&dl->reasons[CBM_DOCLINK_REASON_EXTERNAL]), b[2], sizeof(b[2])), + "test_only_target", + itoa64(atomic_load(&dl->reasons[CBM_DOCLINK_REASON_TEST_ONLY]), b[3], sizeof(b[3])), + "graph_gap", + itoa64(atomic_load(&dl->reasons[CBM_DOCLINK_REASON_GRAPH_GAP]), b[4], sizeof(b[4])), + "unparseable", + itoa64(atomic_load(&dl->reasons[CBM_DOCLINK_REASON_UNPARSEABLE]), b[5], sizeof(b[5])), + "not_indexed", + itoa64(atomic_load(&dl->reasons[CBM_DOCLINK_REASON_NOT_INDEXED]), b[6], sizeof(b[6])), + "below_bar_tier", + itoa64(atomic_load(&dl->reasons[CBM_DOCLINK_REASON_BELOW_BAR]), b[7], sizeof(b[7]))); + /* resolve_cpu_ms is the time inside the resolvers added up over the + * files: with N workers the wall share is about a N-th of it */ + enum { NS_PER_MS = 1000000 }; + cbm_log_info("doc_links.timing", "index_build_ms", + itoa64(dl->build_ns / NS_PER_MS, b[0], sizeof(b[0])), "resolve_cpu_ms", + itoa64(atomic_load(&dl->resolve_ns) / NS_PER_MS, b[1], sizeof(b[1])), "files", + itoa64(atomic_load(&dl->resolved_files), b[2], sizeof(b[2])), "references", + itoa64(atomic_load(&dl->resolved_tokens), b[3], sizeof(b[3]))); +} + +void cbm_doclinks_free_rows(cbm_doc_link_row_t *rows, int count) { + cbm_store_free_doc_links(rows, count); +} + +void cbm_doclinks_free(cbm_doclinks_t *dl) { + if (!dl) { + return; + } + for (int s = 0; s < DOCLINK_RESOLVER_COUNT; s++) { + if (dl->index[s]) { + DOCLINK_RESOLVERS[s]->destroy(dl->index[s]); + } + } + if (dl->rows) { + for (int i = 0; i < dl->file_count; i++) { + cbm_doclinks_free_rows(dl->rows[i].items, dl->rows[i].count); + } + cbm_free(CBM_MEM_CLASS_OTHER, dl->rows); + } + cbm_arena_destroy(&dl->arena); + cbm_free(CBM_MEM_CLASS_OTHER, dl); +} + +/* ── Pipeline bracket ────────────────────────────────────────────── */ + +void cbm_doclinks_begin(cbm_pipeline_ctx_t *ctx, const cbm_file_info_t *files, int file_count, + CBMFileResult **cache) { + if (!ctx) { + return; + } + ctx->doc_links = cbm_doclinks_build(ctx, files, file_count, cache, ctx->doc_link_base, + ctx->doc_link_base_count, ctx->gbuf); + if (!ctx->doc_links) { + ctx->doc_links_failed = true; /* build logged doc_links.error */ + } +} + +void cbm_doclinks_end(cbm_pipeline_ctx_t *ctx) { + if (!ctx) { + return; + } + cbm_doc_link_row_t *rows = NULL; + int count = 0; + bool failed = ctx->doc_links_failed; + if (ctx->doc_links) { + bool run_failed = false; + cbm_doclinks_take_rows(ctx->doc_links, &rows, &count, &run_failed); + failed = failed || run_failed; + cbm_doclinks_free(ctx->doc_links); + ctx->doc_links = NULL; + } + cbm_pipeline_set_doc_link_rows(ctx->pipeline, rows, count, failed); + ctx->doc_links_failed = false; +} + +int cbm_pipeline_pass_doc_links(cbm_pipeline_ctx_t *ctx, const cbm_file_info_t *files, + int file_count) { + if (!ctx) { + return 0; + } + cbm_doclinks_begin(ctx, files, file_count, ctx->result_cache); + if (file_count > 0 && !ctx->result_cache) { + /* Without the result cache this route has nothing to read back. */ + cbm_log_error("doc_links.error", "phase", "sequential", "reason", "no_result_cache"); + ctx->doc_links_failed = true; + } + for (int i = 0; ctx->doc_links && ctx->result_cache && i < file_count; i++) { + bool loaded = false; + CBMFileResult *r = cbm_pipeline_result_acquire(ctx, ctx->result_cache, i, NULL, &loaded); + if (r) { + cbm_doclinks_resolve_file(ctx->doc_links, i, r, ctx->gbuf, ctx->gbuf); + } + cbm_pipeline_result_release(r, loaded); + } + cbm_doclinks_end(ctx); + return 0; +} + +/* ── Incremental carry-forward ───────────────────────────────────── */ + +static bool copy_row(cbm_doc_link_row_t *dst, const cbm_doc_link_row_t *src) { + dst->rel_path = dup_or_empty(src->rel_path); + dst->line = src->line; + dst->syntax = dup_or_empty(src->syntax); + dst->raw = dup_or_empty(src->raw); + dst->reason = dup_or_empty(src->reason); + return dst->rel_path && dst->syntax && dst->raw && dst->reason; +} + +int cbm_doclinks_merge_rows(const cbm_doc_link_row_t *old_rows, int old_count, + const CBMHashTable *replaced, const cbm_doc_link_row_t *fresh, + int fresh_count, cbm_doc_link_row_t **out, int *out_count) { + *out = NULL; + *out_count = 0; + int cap = old_count + fresh_count; + if (cap == 0) { + return 0; + } + cbm_doc_link_row_t *rows = + (cbm_doc_link_row_t *)cbm_calloc(CBM_MEM_CLASS_STORE, (size_t)cap * sizeof(*rows)); + if (!rows) { + return CBM_NOT_FOUND; + } + int n = 0; + for (int i = 0; i < old_count; i++) { + const cbm_doc_link_row_t *r = &old_rows[i]; + if (!r->rel_path || !r->rel_path[0]) { + continue; /* a previous generation's error marker */ + } + if (replaced && cbm_ht_get(replaced, r->rel_path)) { + continue; /* re-extracted or deleted: the fresh rows replace them */ + } + if (!copy_row(&rows[n++], r)) { + cbm_doclinks_free_rows(rows, n); + return CBM_NOT_FOUND; + } + } + for (int i = 0; i < fresh_count; i++) { + if (!copy_row(&rows[n++], &fresh[i])) { + cbm_doclinks_free_rows(rows, n); + return CBM_NOT_FOUND; + } + } + *out = rows; + *out_count = n; + return 0; +} + +/* ── Stored scopes ───────────────────────────────────────────────── */ + +/* The surface writer appends "dl" as the LAST top-level key, and a JSON + * string cannot contain an unescaped quote, so the last `"dl":"` in the row + * is that key. Only this tail is parsed: a row's "lsp" array is megabytes of + * definitions nobody needs here (the C# bench corpus holds ~1.5 GB of them). */ +int cbm_doclinks_scope_from_surface_json(const char *defs_json, char **out) { + *out = NULL; + if (!defs_json) { + return 0; + } + static const char key[] = "\"dl\":\""; + const size_t kl = sizeof(key) - SKIP_ONE; + size_t n = strlen(defs_json); + const char *hit = NULL; + for (size_t i = n; i >= kl; i--) { + if (defs_json[i - kl] == '"' && memcmp(defs_json + i - kl, key, kl) == 0) { + hit = defs_json + i - kl; + break; + } + } + if (!hit) { + return 0; + } + size_t tail = n - (size_t)(hit - defs_json); + char *buf = (char *)cbm_alloc(CBM_MEM_CLASS_OTHER, tail + PAIR_LEN); + if (!buf) { + return CBM_NOT_FOUND; + } + buf[0] = '{'; + memcpy(buf + SKIP_ONE, hit, tail + SKIP_ONE); + yyjson_doc *doc = yyjson_read(buf, tail + SKIP_ONE, 0); + yyjson_val *root = doc ? yyjson_doc_get_root(doc) : NULL; + const char *scope = root ? yyjson_get_str(yyjson_obj_get(root, "dl")) : NULL; + int rc = 0; + if (!scope) { + rc = CBM_NOT_FOUND; /* the key is there but not readable: a corrupt row */ + } else { + *out = cbm_mem_strdup(CBM_MEM_CLASS_OTHER, scope); + rc = *out ? 0 : CBM_NOT_FOUND; + } + yyjson_doc_free(doc); + cbm_free(CBM_MEM_CLASS_OTHER, buf); + return rc; +} + +int cbm_doclinks_scopes_from_surfaces(const cbm_lsp_surface_row_t *rows, int row_count, + const CBMHashTable *skip, cbm_doclink_scope_t **out, + int *count) { + *out = NULL; + *count = 0; + if (!rows || row_count <= 0) { + return 0; + } + int64_t started = doclinks_now_ns(); + cbm_doclink_scope_t *scopes = NULL; + int n = 0; + int cap = 0; + for (int i = 0; i < row_count; i++) { + const cbm_lsp_surface_row_t *row = &rows[i]; + if (!row->defs_json || !row->rel_path || (skip && cbm_ht_get(skip, row->rel_path))) { + continue; + } + char *scope = NULL; + if (cbm_doclinks_scope_from_surface_json(row->defs_json, &scope) != 0) { + cbm_doclinks_free_scopes(scopes, n); + return CBM_NOT_FOUND; + } + if (!scope) { + continue; /* no scope: a language without a scope scanner */ + } + if (n >= cap) { + int ncap = cap ? cap * PAIR_LEN : CBM_SZ_64; + cbm_doclink_scope_t *grown = (cbm_doclink_scope_t *)cbm_realloc( + CBM_MEM_CLASS_OTHER, scopes, (size_t)ncap * sizeof(*grown)); + if (!grown) { + cbm_free(CBM_MEM_CLASS_OTHER, scope); + cbm_doclinks_free_scopes(scopes, n); + return CBM_NOT_FOUND; + } + scopes = grown; + cap = ncap; + } + scopes[n].rel_path = cbm_mem_strdup(CBM_MEM_CLASS_OTHER, row->rel_path); + scopes[n].scope = scope; + n++; + if (!scopes[n - SKIP_ONE].rel_path) { + cbm_doclinks_free_scopes(scopes, n); + return CBM_NOT_FOUND; + } + } + *out = scopes; + *count = n; + /* an incremental run pays this instead of extracting the files again */ + char b[3][CBM_SZ_32]; + cbm_log_info("doc_links.stored_scopes", "rows", itoa64(row_count, b[0], sizeof(b[0])), "scopes", + itoa64(n, b[1], sizeof(b[1])), "elapsed_ms", + itoa64((doclinks_now_ns() - started) / 1000000, b[2], sizeof(b[2]))); + return 0; +} + +void cbm_doclinks_free_scopes(cbm_doclink_scope_t *scopes, int count) { + if (!scopes) { + return; + } + for (int i = 0; i < count; i++) { + cbm_free(CBM_MEM_CLASS_OTHER, (char *)scopes[i].rel_path); + cbm_free(CBM_MEM_CLASS_OTHER, (char *)scopes[i].scope); + } + cbm_free(CBM_MEM_CLASS_OTHER, scopes); +} diff --git a/src/pipeline/doc_links.h b/src/pipeline/doc_links.h new file mode 100644 index 000000000..20f9fbfaa --- /dev/null +++ b/src/pipeline/doc_links.h @@ -0,0 +1,269 @@ +/* + * doc_links.h — doc-comment references -> MENTIONS edges. + * + * Extraction (internal/cbm/doclink.h) leaves every documented definition's + * references in CBMFileResult.doc_links and every file's doc-link scope in + * CBMFileResult.doc_scope. This layer resolves them in the per-file resolve + * phase, beside CALLS: + * + * cbm_doclinks_build once per run, after the run's nodes exist: the + * per-language indexes over every file's scope + * (this run's results, plus the stored scopes of + * files an incremental run did not re-extract) + * cbm_doclinks_resolve_file per file, concurrently: MENTIONS edges into the + * caller's edge buffer, unresolved rows into the + * run's slot for that file + * cbm_doclinks_take_rows the run's unresolved rows, for publication into + * doc_link_unresolved + * + * Resolution is guess-free: a reference becomes an edge only when it names + * one entity EXACTLY (qualified name, doc ID, alias) or UNIQUELY at the first + * scope level of the language's own lookup rules. Everything else is a row + * with a reason. One MENTIONS edge per (documented definition, target): + * {"via":"doc_comment","syntax":..,"tier":..,"line":,"count":n}. + * + * Ship gate: a link family (doclink.h) whose tier is below the audit's bar + * does not ship. Its references still resolve, but a resolved one is written + * as a row with reason below_bar_tier instead of an edge; an unresolved one + * keeps its own reason. + * + * Per-language hooks: a cbm_doclink_resolver_t (C#: doc_links_cs.c). + */ +#ifndef CBM_PIPELINE_DOC_LINKS_H +#define CBM_PIPELINE_DOC_LINKS_H + +#include "pipeline/pipeline_internal.h" +#include "store/store.h" + +/* doc_link_unresolved.reason values. */ +typedef enum { + CBM_DOCLINK_REASON_MISSING = 0, /* named code that is not in the graph anywhere */ + CBM_DOCLINK_REASON_AMBIGUOUS, /* several entities, or an overload group */ + CBM_DOCLINK_REASON_EXTERNAL, /* outside the repository (URL, BCL, open scope) */ + CBM_DOCLINK_REASON_TEST_ONLY, /* product code naming a test-only entity */ + CBM_DOCLINK_REASON_NOT_INDEXED, /* on disk, but not in the graph */ + CBM_DOCLINK_REASON_GRAPH_GAP, /* declared in source, but without a graph node */ + CBM_DOCLINK_REASON_UNPARSEABLE, /* reference syntax not understood */ + CBM_DOCLINK_REASON_BELOW_BAR, /* resolved, but its link family does not ship + * (cbm_doclink_syntax_ships): below_bar_tier */ + CBM_DOCLINK_REASON_COUNT +} cbm_doclink_reason_t; + +const char *cbm_doclink_reason_name(int reason); + +typedef enum { + CBM_DOCLINK_EDGE = 0, /* target set: one MENTIONS edge */ + CBM_DOCLINK_UNRESOLVED, /* reason set: one doc_link_unresolved row */ + /* Neither an edge nor a row: what the parser took for a reference is no + * reference to code elsewhere. It names the definition's own parameter or + * type parameter, or it is text the language's doc tool renders as plain + * text, which only the resolver can tell (it needs the index). Counted as + * `local` in the doc_links.done log line, and nowhere in index_status. */ + CBM_DOCLINK_LOCAL, +} cbm_doclink_kind_t; + +typedef struct { + cbm_doclink_kind_t kind; + const cbm_gbuf_node_t *target; + bool exact; /* tier: "exact" (qualified/doc ID/alias) vs "unique" (scope lookup) */ + int reason; +} cbm_doclink_outcome_t; + +/* A file's stored doc-link scope (incremental runs: files not re-extracted). */ +typedef struct cbm_doclink_scope { + const char *rel_path; + const char *scope; +} cbm_doclink_scope_t; + +typedef struct cbm_doclinks cbm_doclinks_t; + +/* The resolve-phase bracket every pipeline route uses: begin builds the run + * state into ctx->doc_links (over ctx->doc_link_base for an incremental run); + * resolve_worker / the sequential pass call cbm_doclinks_resolve_file; end + * hands the rows to ctx->pipeline (cbm_pipeline_set_doc_link_rows) and frees + * the state. A failed build is logged and recorded, never silent. */ +void cbm_doclinks_begin(cbm_pipeline_ctx_t *ctx, const cbm_file_info_t *files, int file_count, + CBMFileResult **cache); +void cbm_doclinks_end(cbm_pipeline_ctx_t *ctx); + +/* Sequential pipelines: begin + resolve every file into ctx->gbuf + end. */ +int cbm_pipeline_pass_doc_links(cbm_pipeline_ctx_t *ctx, const cbm_file_info_t *files, + int file_count); + +/* Rows of the next generation on an incremental run: the previous rows of the + * files not re-extracted (paths in `replaced` are dropped, as is a previous + * error marker) followed by this run's rows. Strings are copied; free with + * cbm_doclinks_free_rows. Returns 0, -1 on allocation failure. */ +int cbm_doclinks_merge_rows(const cbm_doc_link_row_t *old_rows, int old_count, + const CBMHashTable *replaced, const cbm_doc_link_row_t *fresh, + int fresh_count, cbm_doc_link_row_t **out, int *out_count); + +/* Build the run's resolver state. `files[0..file_count)` / `cache` are the + * files extracted in this run (results read through the spill contract); + * `base` holds the scopes of the files it did not re-extract (NULL/0 on a full + * run); `graph` is the run's complete node set, read-only from here on. + * Returns NULL only when allocation failed (logged). */ +cbm_doclinks_t *cbm_doclinks_build(const cbm_pipeline_ctx_t *ctx, const cbm_file_info_t *files, + int file_count, CBMFileResult **cache, + const cbm_doclink_scope_t *base, int base_count, + const cbm_gbuf_t *graph); + +/* Resolve files[file_idx]'s references. Concurrent calls for different files + * are safe. Edges go to `edge_out` (the worker's buffer, or the graph itself + * on a sequential run). */ +void cbm_doclinks_resolve_file(cbm_doclinks_t *dl, int file_idx, const CBMFileResult *result, + const cbm_gbuf_t *graph, cbm_gbuf_t *edge_out); + +/* Move the run's unresolved rows out (file order, then line order) together + * with the failure flag; the caller frees them with cbm_doclinks_free_rows. + * Logs one summary line. */ +void cbm_doclinks_take_rows(cbm_doclinks_t *dl, cbm_doc_link_row_t **rows, int *count, + bool *failed); +void cbm_doclinks_free_rows(cbm_doc_link_row_t *rows, int count); +void cbm_doclinks_free(cbm_doclinks_t *dl); + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +/* Test seam: the next cbm_doclinks_build fails the way an allocation failure + * does (logged, NULL), so a test can follow a failed doc-link layer through + * publication and index_status. Test builds only. */ +void cbm_doclinks_test_fail_build_once(void); +#endif + +/* The stored doc-link scopes of a previous generation (lsp_surface rows, + * key "dl"), excluding `skip` paths. Strings are heap-owned by *out; free + * with cbm_doclinks_free_scopes. Returns 0, or -1 when a row is malformed. */ +int cbm_doclinks_scopes_from_surfaces(const cbm_lsp_surface_row_t *rows, int row_count, + const CBMHashTable *skip, cbm_doclink_scope_t **out, + int *count); +void cbm_doclinks_free_scopes(cbm_doclink_scope_t *scopes, int count); +/* The "dl" scope of one surface JSON: *out is a heap copy, or NULL when the + * row has none. Returns 0, -1 when the key is unreadable or memory ran out. */ +int cbm_doclinks_scope_from_surface_json(const char *defs_json, char **out); + +/* ── Per-language resolver hooks ─────────────────────────────────── */ + +typedef struct { + const char *rel_path; + const char *scope; /* the file's doc-link scope blob (NULL when it has none) */ + int run_file; /* index into the run's files[], or -1 for a base file */ +} cbm_doclink_file_t; + +/* What `build` is handed. LIFETIMES: the struct and its `files` array are + * valid only during the `build` call (the array is freed right after it): a + * resolver must not keep either pointer. The `rel_path` and `scope` strings + * the array points to stay valid until `destroy`, so an index may keep those + * without copying them. */ +typedef struct { + const cbm_pipeline_ctx_t *ctx; + const cbm_gbuf_t *graph; + const cbm_doclink_file_t *files; /* every file of the language, sorted by rel_path */ + int file_count; + int run_file_count; /* size of the run's files[] (run_file indexes into it) */ +} cbm_doclink_build_in_t; + +/* Receives one name a changed file no longer declares. false when it cannot be + * recorded: the hook then reports failure, and the caller fails closed. */ +typedef bool (*cbm_doclink_name_fn)(void *ud, const char *name, size_t len); + +/* What a changed file's scope means for the files an incremental run does NOT + * re-extract. */ +typedef enum { + CBM_DOCLINK_DELTA_LOCAL = 0, /* nothing outside the file resolves differently, except + * unresolved references that name a reported name */ + CBM_DOCLINK_DELTA_GLOBAL, /* any file may resolve differently: not repairable + * file by file */ +} cbm_doclink_delta_t; + +enum { CBM_DOCLINK_RESOLVER_LANGS = 4 }; + +/* Everything the resolving half needs from a language leg: one of these, and + * its pointer in doc_links.c's resolver table. */ +typedef struct { + /* The languages whose files it resolves, in ONE index: a leg whose + * references cross language values (TypeScript, TSX and JavaScript) lists + * them all. No language may be listed by two resolvers. */ + CBMLanguage langs[CBM_DOCLINK_RESOLVER_LANGS]; + int lang_count; + /* Tag line of its scope blobs (doclink.h); NULL: it has none. */ + const char *scope_tag; + /* The language's project-wide index; NULL on allocation failure. */ + void *(*build)(const cbm_doclink_build_in_t *in); + void (*destroy)(void *index); + /* Optional working state for the references of one file, which one thread + * resolves: made before its first reference, handed to every resolve call + * of the file, released after its last. A NULL state (no hook, or memory + * ran out) changes the cost of resolving, never an outcome. */ + void *(*file_begin)(const void *index, int run_file); + void (*file_end)(void *state); + /* Resolve one reference of run file `run_file` (its own scope is in the + * index). Thread-safe: the index is read-only after build; `state` is the + * file's own (file_begin) or NULL. */ + void (*resolve)(const void *index, void *state, int run_file, const CBMDocLink *link, + const cbm_gbuf_t *graph, cbm_doclink_outcome_t *out); + /* Incremental runs; both optional. + * scope_input true for a file that is no source of the language and has + * no scope blob, but sets the scope of its files. A change + * to one is GLOBAL. (C# needs none: its MSBuild project + * files have scope blobs of their own.) + * scope_delta compare a changed file's stored and fresh scope blobs (both + * of this language) and report the names it no longer + * declares through `removed`. Returns a cbm_doclink_delta_t, + * or -1 on failure. Without the hook every change is GLOBAL. */ + bool (*scope_input)(const char *rel_path); + int (*scope_delta)(const char *stored, const char *fresh, cbm_doclink_name_fn removed, + void *ud); + /* Optional, together: whether the reader takes a scope blob written in + * this run (1), refuses it (0), or could not tell for memory (-1); and + * the marker stored instead of a refused blob (a memory-core block of + * CBM_MEM_CLASS_OTHER, NULL when memory ran out). A run refuses the blob + * of its own file and goes on without the file's declarations; the marker + * makes a later run that reads the file's scope back do the same, where + * the blob itself would be a stored scope the reader refuses, which fails + * the run. */ + int (*scope_accepted)(const char *scope); + char *(*rejected_scope)(const char *scope); +} cbm_doclink_resolver_t; + +extern const cbm_doclink_resolver_t cbm_doclink_cs_resolver; + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +/* Test seam (doc_links_cs.c): the scope levels and overloads the C# resolver + * looked at since the last reset. A test holds it against the size of its + * input, so that a lookup whose cost grows faster than its input fails + * without a clock. Test builds only. */ +void cbm_doclink_cs_test_work_reset(void); +uint64_t cbm_doclink_cs_test_work(void); +/* The buckets the index build walked to empty its per-file scratch table + * (reset with the work above). */ +uint64_t cbm_doclink_cs_test_scratch_work(void); +/* Test seam (doc_links.c): the nth MENTIONS edge insert from now on fails as + * if memory ran out (0: none). */ +void cbm_doclinks_test_fail_edge_insert_after(int nth); +/* Test seam (doc_links_cs.c): ask the reader whether it takes a scope blob. + * (The scope of one file is spoiled where it is written: doclink.h.) */ +bool cbm_doclink_cs_test_scope_parses(const char *scope); +/* Test seam (doc_links_cs.c): false builds the C# indexes from now on without + * the run's memo of the units' using steps (every unit step walks its + * directives), so a run can be held against one with it; true restores it. */ +void cbm_doclink_cs_test_unit_memo(bool on); +#endif + +/* True when `rel_path` is a scope input of some language (scope_input). */ +bool cbm_doclinks_is_scope_input(const char *rel_path); + +/* The scope delta of one changed file (`fresh` set) or deleted file (`fresh` + * NULL): a cbm_doclink_delta_t, or -1 on failure. A file with no scope before + * and after is LOCAL; a scope that appears, disappears or changes its + * language is GLOBAL; otherwise the language's scope_delta hook decides. */ +int cbm_doclinks_scope_delta(const char *stored, const char *fresh, cbm_doclink_name_fn removed, + void *ud); + +/* What is stored for a scope blob written in this run (the surface row's + * `dl`, lsp_surface.c): the blob itself, or its language's rejected-scope + * marker when that language's reader refuses it (scope_accepted), made into + * *owned (release with cbm_free(CBM_MEM_CLASS_OTHER, *owned); NULL when the + * blob itself is returned). NULL when memory ran out deciding or making the + * marker. */ +const char *cbm_doclinks_storable_scope(const char *scope, char **owned); + +#endif /* CBM_PIPELINE_DOC_LINKS_H */ diff --git a/src/pipeline/doc_links_cs.c b/src/pipeline/doc_links_cs.c new file mode 100644 index 000000000..db5d8aeb7 --- /dev/null +++ b/src/pipeline/doc_links_cs.c @@ -0,0 +1,6751 @@ +/* + * doc_links_cs.c — C# cref resolution for doc_links.h. + * + * The project-wide index is built from every C# file's doc-link scope and + * every MSBuild project file's (internal/cbm/doclink_cs.c). Nothing is read + * from the disk. Graph nodes are only looked up by qualified name, never by + * line, so the closure-repair route (whose unchanged nodes are line-less + * proxies) resolves exactly as a full build does. + * + * Assemblies. The index does not know which files a project compiles; it + * goes by where a file stands. A project is a directory that holds a + * *.csproj -- one whose SDK compiles something: a project file of the + * NoTargets or Traversal SDK only runs build steps, and is the project of no + * file. A file belongs to the nearest project at or above it. An + * assembly is named by its project file: projects whose project files have + * the same name are one assembly (a reference source beside its + * implementation, the flavours of one library). A directory that holds + * several project files is an assembly of its own: which of them a file + * there is compiled into is not known. A file no project file stands above + * is of a shared tree -- the largest directory around it with no project + * file in or below it. A shared tree belongs to no assembly: the projects + * that include its files compile them. (A repository without any project + * file is one tree.) + * + * Entities. A type entity is (scope, name, generic arity, owner): the scope + * is its namespace, or its outer type's entity, so `Outer.Inner` and + * `Outer.Inner` are two types. The owner of a top-level type is its + * assembly: the declarations of one full name in one assembly are one type + * with several declarations (its parts, a stub beside the implementation, + * one per flavour), and declarations in two assemblies are two types, which + * see nothing of each other. The shared trees' declarations have no + * assembly. A complete declaration there is a type of its tree. The parts of + * a `partial` type there are one entity whatever tree they stand in, and + * they are parts of every assembly's type of that name whose implementation + * is itself declared partial: what a reference sees of a partial type is its + * own assembly's parts and the shared trees'. Seen without an assembly's own + * parts -- from a shared tree, or from an assembly that has none --, a name + * those parts declare is ambiguous: it is neither bound to what else has + * that name nor missing. + * + * Contracts. A declaration in a project directory named `ref` is a stub: + * `ref` is the .NET convention for reference-assembly sources. Inside one + * assembly the stub stands behind the implementation: the implementation's + * node binds, the stub's only where the implementation has none (its file + * did not parse, it lacks the member, or a parse error hides it). An + * assembly whose every declaration of a type is a stub holds only the + * contract, and the rule joins a contract to the one implementation it can + * belong to: the declaration of that full name and arity the shared trees + * hold, when they hold exactly one and no assembly holds another + * implementation of the name (entity_twins). The stub is then no second + * type; a binding that exists only through the join is never exact. + * Nothing is chosen between two implementations, two assemblies or two + * shared trees. + * + * What a reference sees of a name: its own assembly's type -- for a file of + * a shared tree, what its own tree declares. Else every shared tree's and + * every other assembly's type of that name alike: one binds, several are + * ambiguous. Which assemblies a project references is not known, so nothing + * chooses between two of them -- no stub, no nearer directory. + * + * Alternatives. Complete declarations of one type in two or more projects + * (the flavours of one assembly) are alternatives: the one of the + * referencing file's own project binds, and from anywhere else the + * reference is ambiguous. The same holds for one member declared in two or + * more projects or shared trees, and for the parts of a type in two or more + * shared trees. The parts of a partial type in one place are no + * alternatives: the type's node is the part's that stands nearest to the + * reference (a choice of presentation: every part is the type). + * + * Nodes. A declaration owns the node at . when it is the last + * declaration of that path in its file (the extractor keeps one node per + * qualified name and the last declaration wins, so `Foo` / `Foo` in one + * file leave one of them without a node: a graph gap, never a fallback to + * the other). The same holds for members: overloads share one node; `Foo.X` + * and `Foo.X` share one, and it belongs to the later declaration. + * Operators, indexers, events, delegates, implicit and primary constructors + * have no node at all: a reference to one that is declared is a graph gap. + * + * Lookup is the C# compiler's cref binding, and it guesses nothing: the first + * scope level that has the name decides, and more than one candidate there + * is ambiguous. + * - a simple name: the documented method's type parameters; then for every + * enclosing type, innermost first: its type parameters, its own nested + * types and members; then for every enclosing namespace, innermost + * first: the types and namespaces in it, then (where a namespace + * declaration of the file stands) that declaration's aliases and after + * them its usings. The file's own usings and the project's global ones + * belong to the outermost level. A using of an inner namespace + * declaration is therefore asked before an outer namespace + * - a qualified name: its first segment is looked up as a simple name that + * names a type or a namespace; every further segment is a member of what + * the one before named. There is no second try from another level, and + * no fully-qualified shortcut past a nearer namespace of that name + * - inherited members are not looked up: the compiler does not consider + * them in a cref, in a class or in an interface, by a simple name or + * through the derived type's name. Such a reference is a row + * (CS_BIND_INHERITED says what else it could be) + * - arity: a name without type arguments is the arity-0 type. A generic + * type of that name never stands in (no cross-arity fallback); type + * arguments select the types and the generic methods of that arity. A + * method name without type arguments takes methods of any arity, the + * non-generic ones first + * - the level that has the name decides also when a parameter list + * follows: a written signature binds the overload of THAT level with + * exactly these parameter types; a type variable matches by its position + * only (`{K}` ... `(K)` against `` ... `(TKey)`); only when no + * overload matches does one with a parameter type nothing is known about + * count, and several of those are ambiguous. Without a parameter list an + * overload group is ambiguous + * - constructors: a parameter list on a type's name (`Foo(int)`, + * `Ns.Foo(int)`, `Outer.Inner(int)`), `Foo.Foo`, and `Foo(int)` written + * inside a generic `Foo`; a bare `Foo` is the type; no constructor is + * inherited + * - `using static` brings in a type's nested types and its static members, + * extension methods excepted + * - explicit interface implementations are not addressable by name + * - visibility: product code never binds a test-only declaration + * (test_only_target, no fallback), and a namespace that only test code + * declares is no name in its scope; a test program does not bind another + * program's global-namespace test type + * + * Reasons. + * - external is structural: a name through a namespace the repository does + * not declare (a using of one, a qualifier that is one, an open + * hierarchy: a base outside the repository), a keyword alias the + * repository does not declare, a project whose global usings could not + * be evaluated. There is no list of well-known outside names; the one + * namespace treated as never the repository's own is `System` with what + * is under it (CS_STANDARD_ROOT) + * - missing: a name every scope level was asked for, in declared + * namespaces only; a name whose level has no overload with the written + * parameters; an inherited member named through a derived type or by a + * simple name (the compiler binds no such cref) when the hierarchy is + * the repository's + * - graph_gap: declared, and no node (see Nodes); a namespace; what the + * parser could not place (a type whose members a parse error hides + * answers a member it does not show with graph_gap; a name some file + * declares without an establishable scope is a gap wherever it is + * written) + * - ambiguous: several candidates at the level that has the name; two + * assemblies' (or two shared trees') types of one name; the flavours of + * one type or member seen from outside their projects; a name that parts + * of the type out of the reference's view declare; also an overload + * group -- or one project's complete declarations of one type -- larger + * than CS_MAX_OVERLOADS that would have to be compared, and a name more + * than CS_MAX_FOREIGN assemblies declare. The index's log line + * doc_links.cs.ambiguous counts the references by these causes + * - unparseable: a reference longer than CS_REF_BUF, with more segments + * than CS_MAX_SEGS or more parameters than CS_MAX_PARAMS + * No limit decides silently: every one of them ends in one of these rows. + */ +#include "pipeline/doc_links.h" + +#include "doclink.h" +#include "helpers.h" /* cbm_fqn_module_source_lang */ +#include "pipeline/doc_links_msbuild.h" +#include "foundation/arena.h" +#include "foundation/compat_thread.h" +#include "foundation/constants.h" +#include "foundation/hash_table.h" +#include "foundation/log.h" +#include "foundation/mem_core.h" + +#include +#include +#include +#include +#include + +enum { + CS_MAX_SEGS = 16, /* segments of a written reference */ + CS_MAX_PARAMS = 24, /* its parameters */ + CS_MAX_SUPERS = 64, /* supertypes one lookup follows */ + CS_MAX_OVERLOADS = 256, /* declarations of one name one lookup compares */ + CS_MAX_FOREIGN = 64, /* other owners' types of one name one lookup compares */ + CS_MAX_NEST = 64, /* the scanner's nesting limits (types, namespace segments) */ + CS_KEY_BUF = 2048, /* a node's qualified name */ + CS_NAME_BUF = 513, /* one name (the scanner's CS_NAME_MAX and its terminator) */ + CS_PARAM_BUF = 256, /* one normalized parameter type */ + CS_REF_BUF = 1024, /* a written reference */ + CS_ARITY_NONE = -1, + CS_NONE = -1, + CS_AMBIGUOUS = -2, + /* An entity's owner: an assembly (>= 0); CS_POOL, the shared trees' parts + * of a partial type; below it, the one shared tree that holds a complete + * declaration (shared_owner()). */ + CS_POOL = -1, + CS_FIELDS = 9, /* the most fields a scope record has (T) */ + CS_MAX_ROOTS = 3, /* Enum, ValueType, Object */ + /* 0: the compiler's rule, inherited members are never bound. 1: where the + * compiler's lookup binds nothing, the nearest supertype that has the + * member binds (a class's base classes, an interface's base interfaces, + * then the implicit roots the repository declares). Never a different + * binding than the compiler's: only one where it has none. */ + CS_BIND_INHERITED = 0, +}; + +/* ── Index data ──────────────────────────────────────────────────── */ + +/* A namespace of the repository: the global one (id 0), every declared one, + * and every namespace above a declared one. */ +typedef struct { + int parent; + int depth; /* segments of its full name (the global namespace: 0) */ + const char *name; /* its own segment */ + bool declared; /* a file declares it; else it is only above a declared one */ + bool prod; /* product code declares it, or a namespace under it */ + bool standard; /* `System`, or a namespace under it (see CS_STANDARD_ROOT) */ +} cs_ns_t; + +/* The one namespace no repository owns by declaring it. `System` is the + * standard library's namespace by the language standard, and user code adds + * to it (polyfills) without owning it: a name of `System`, or of a namespace + * under it, that the repository does not have is the standard library's -- + * outside -- also where the repository declares that namespace. No other + * root is treated so, and no list of names stands behind this: every other + * namespace is the repository's as soon as a file declares it. */ +static const char CS_STANDARD_ROOT[] = "System"; + +/* The IDs below retain the original directive order. Two written directives + * remain two candidates even when they name the same target. */ +typedef struct { + const char *name; + int id; +} cs_alias_id_t; + +typedef struct { + int scope; /* namespace or entity, according to the array */ + int id; +} cs_using_id_t; + +typedef struct { + cs_alias_id_t *aliases; /* by (name, directive ID) */ + size_t naliases; + cs_using_id_t *namespaces; /* by (namespace, directive ID) */ + size_t nnamespaces; + cs_using_id_t *entities; /* by (view entity, directive ID), at most two per using */ + size_t nentities; + bool ready; /* published only after all targets and arrays are complete */ +} cs_using_index_t; + +typedef struct { + const char *name; + int scope; + bool top; /* a namespace's top-level type, else an entity's named declaration */ +} cs_name_scope_t; + +typedef struct { + int parent; /* region index; CS_NONE for the file's own region 0 */ + int ns; + uint32_t start; /* its lines (see the note on lines below) */ + uint32_t end; + int u_lo; /* its usings: [u_lo, u_hi) of the file's */ + int u_hi; + cs_using_index_t using_index; + bool open; /* a using of its own, or of a region around it, names something + * outside the repository */ +} cs_region_t; + +typedef struct { + int region; + char kind; /* n namespace, s static, a alias */ + bool global; + const char *alias; + const char *target; /* as written */ + int ns; /* what it names: a namespace (CS_NONE: none of the repository) */ + int ent; /* ... or a type (CS_NONE; CS_AMBIGUOUS) */ + bool joined; /* ... one type only because a contract is joined to it */ +} cs_using_t; + +typedef struct { + int region; + uint32_t start; + uint32_t end; + char kind; /* c s i e r t d */ + bool partial; + bool incomplete; /* a parse error hides some of its members */ + bool owns_node; /* the last declaration of its path in the file */ + int outer; /* the enclosing type (index in the file), or CS_NONE */ + int depth; + const char *name; + const char *bases; + int arity; + int entity; + int gid; /* types of one path in the file share it */ + const cbm_gbuf_node_t *node; /* its graph node, when it owns one */ +} cs_type_t; + +/* Line numbers (start, end) are meaningful only in a file re-extracted by + * this run: the persisted scope of every other file carries zeros (so an edit + * that only moves lines keeps the file's surface). Lines are therefore read + * only for the SOURCE file's own context, never for a target. */ +typedef struct { + uint32_t start; + int type; /* the declaration it belongs to (index into the file's types) */ + int arity; /* generic method arity */ + char kind; /* c method or constructor, v field or enum member, p property, e event, + o operator, x indexer */ + bool explicit_impl; + bool is_static; /* what `using static` brings in: static, and no extension method */ + const char *name; + const char *sig; /* '|'-joined parameter types; NULL for v, p and e */ + int nparams; /* how many `sig` holds */ + const cbm_gbuf_node_t *node; /* the node a reference to it binds; NULL: none */ +} cs_member_t; + +/* A type parameter: of a type (owner = its index) or of a generic method + * (owner = the file's type count + the member's index). */ +typedef struct { + int owner; + int pos; + const char *name; +} cs_tparam_t; + +typedef struct { + uint32_t from; + uint32_t to; +} cs_span_lines_t; + +typedef struct { + const char *rel_path; + const char *module_qn; + bool is_test; + bool is_ref; /* of a project directory named `ref`: a reference assembly's source */ + int unit; + cs_span_lines_t *unplaced; /* line ranges without a scope: sorted, disjoint */ + int nunplaced; + cs_region_t *regions; + int nregions; + cs_using_t *usings; /* by region */ + int nusings; + cs_type_t *types; /* document order: index = the T record's ordinal */ + int ntypes; + cs_member_t *members; /* document order */ + int nmembers; + int *members_by_start; + cs_tparam_t *tparams; /* by (owner, name) */ + int ntparams; + /* Its scope was written in this run and this reader refuses it: the file + * declares nothing here, and its own references are graph gaps. */ + bool rejected; +} cs_file_t; + +typedef struct { + int file; + int type; +} cs_decl_t; + +/* Where a declaration stands for choosing the one a reference binds: with a + * node before without, an implementation before a reference assembly's stub, + * product code and test code apart. Entries of one class are in path order. */ +enum { CS_CLS_TEST = 1, CS_CLS_REF = 2, CS_CLS_NO_NODE = 4, CS_CLS_COUNT = 8 }; + +typedef struct { + int file; + int type; + unsigned char cls; +} cs_bind_t; + +/* A complete (not `partial`) declaration outside a reference assembly's + * source, by the project or shared tree it stands in. */ +typedef struct { + int unit; + int file; + int type; +} cs_full_t; + +typedef struct { + int ns; /* a top-level type's namespace; CS_NONE for a nested type */ + int parent; /* a nested type's outer entity; CS_NONE */ + int owner; /* a top-level type's: its assembly, CS_POOL, or a shared tree */ + int twin; /* the shared trees' parts that are parts of this type too; CS_NONE */ + /* For the shared trees' declaration, what the assemblies make of it: */ + bool used; /* an assembly's type has it as its twin */ + int stub; /* the assembly's type that has only stubs and takes it as its + * implementation; CS_NONE: none does, CS_AMBIGUOUS: several do */ + bool rival_known; /* `rival` was asked for */ + bool rival; /* an assembly holds an implementation of the name that is no part + * of it (see rival_implementation) */ + const char *name; + int arity; + char kind; + cs_decl_t *decls; + int ndecls; + int dcap; + int *units; /* the projects and shared trees that declare it: sorted, unique */ + int nunits; + int *bases; + int nbases; + cs_bind_t *binds; /* its declarations, by (class, file, type) */ + /* Its complete declarations, when two or more projects (or shared trees) + * hold one: alternatives, by (unit, file, type). NULL when at most one + * does. */ + cs_full_t *fulls; + int nfulls; + int full_units; /* how many units hold a complete declaration */ + int full_units_prod; /* ... one that is not test code */ + bool open; /* a base of its own is outside the repository */ + bool open_any; /* ... or one of a supertype's is */ + bool incomplete; /* a declaration of it has members a parse error hides */ + bool any_prod; + bool all_test; + bool has_impl; /* a declaration outside a reference assembly's source */ + bool impl_partial; /* ... that is declared partial */ + bool shared_parts; /* the shared trees' parts of a partial type, or nested in them */ + bool joined; /* stubs only, joined to the shared trees' one implementation (twin) */ +} cs_entity_t; + +/* A type by the scope that declares it: `scope` is a namespace for a + * top-level type and the outer entity for a nested one. */ +typedef struct { + int scope; + const char *name; + int arity; + int owner; /* the entity's (a nested type has its outer type's: 0 here) */ + int ent; +} cs_named_t; + +/* A member by the entity that declares it. */ +typedef struct { + int ent; + int file; + int midx; + unsigned char group; /* 0 value or event, 1 callable, 2 operator or indexer */ + unsigned char cls; +} cs_mref_t; + +/* A name that an assembly's own parts of a type declare beyond the shared + * trees' declaration `twin` they belong to: a nested type the shared trees + * do not have (`ent`), or members (`ent` and `arity` CS_NONE; one entry for + * all of a name, `prod`: one of them is no test declaration). */ +typedef struct { + int twin; + const char *name; + int arity; + int ent; + bool prod; +} cs_extra_t; + +/* A project (a directory with a *.csproj and what is below it, up to the + * next one), or a shared tree. */ +typedef struct { + const char *dir; + int group; /* its assembly; CS_NONE for a shared tree */ + bool is_ref; /* a project directory named `ref` */ + cs_using_t *usings; /* global usings: its files' and its project files' */ + int nusings; + int cap; + cs_using_index_t using_index; + bool open; /* a global using could not be evaluated, or names something outside */ +} cs_unit_t; + +/* A directory that holds project files. */ +typedef struct { + int count; /* its *.csproj files */ + const char *stem; /* the name of the first, without the extension */ +} cs_pdir_t; + +/* Why a reference is ambiguous. The row says `ambiguous`; the index's log + * line says how many references of each kind there were. */ +typedef enum { + CS_WHY_SCOPE = 0, /* by the language's rules: candidates of one scope level, overloads */ + CS_WHY_SHARED, /* two or more shared trees declare the type */ + CS_WHY_ASSEMBLIES, /* two or more assemblies do */ + CS_WHY_FLAVOURS, /* declarations of one type or member in several projects of one + assembly, seen from outside them */ + CS_WHY_PARTS, /* declared by a part of the type the reference does not see */ + CS_WHY_LIMIT, /* more candidates than one lookup compares */ + CS_WHY_COUNT, +} cs_why_t; + +/* Counted while references are resolved (by every worker at once). */ +typedef struct { + _Atomic uint64_t ambiguous[CS_WHY_COUNT]; + _Atomic uint64_t rejected; /* references of files whose fresh scope was refused */ +} cs_stats_t; + +/* What parsing one scope set in the tables all files share: so that a scope + * written in this run that this reader refuses can be taken back whole, and + * costs only its own file. Reset for every file. */ +typedef struct { + int ns; + bool declared; + bool prod; +} cs_ns_was_t; + +typedef struct { + CBMHashTable *ht; + const char *key; +} cs_mark_was_t; + +typedef struct { + int nnss; /* namespaces before the file */ + cs_ns_was_t *flags; + int nflags; + int cap_flags; + cs_mark_was_t *marks; + int nmarks; + int cap_marks; +} cs_undo_t; + +typedef struct { + CBMArena arena; + bool oom; /* an index allocation failed */ + const char *project; + cs_file_t *files; + int nfiles; + int *run_to_file; + int run_count; + cs_ns_t *nss; + int nnss; + int nscap; + CBMHashTable *ns_by_key; /* "\x1f" -> id + 1 */ + cs_entity_t *ents; + int nents; + int ecap; + CBMHashTable *ent_by_key; + cs_named_t *tops; /* by (namespace, name, arity, entity) */ + int ntops; + cs_named_t *kids; /* by (outer entity, name, arity, entity) */ + int nkids; + cs_mref_t *mrefs; /* by (entity, name, group, signature, class, file, order) */ + int nmrefs; + cs_extra_t *extras; /* by (twin, name, arity, entity): a name's members first */ + int nextras; + cs_name_scope_t *name_scopes; /* distinct (name, kind, scope) reverse postings */ + size_t nname_scopes; + bool name_scopes_ready; + /* "P\x1f": the entity declares an operator or indexer of + * that name outside test code; "A...": it declares one at all */ + CBMHashTable *specials; + CBMHashTable *type_names; /* every type's simple name */ + CBMHashTable *quarantine; /* names of types declared where no scope is known */ + CBMHashTable *quarantine_test; /* the same, declared by test code only */ + CBMHashTable *unit_by_dir; /* dir -> unit + 1 */ + cs_unit_t *units; + int nunits; + int ucap; + cbm_msb_t *msb; /* the repository's MSBuild project files */ + CBMHashTable *project_dirs; /* directory -> cs_pdir_t: the ones that hold a *.csproj */ + const char **projects; /* the *.csproj files, in path order */ + int nprojects; + CBMHashTable *project_above; /* directories with a *.csproj in or below them */ + CBMHashTable *group_by_stem; /* project file name -> assembly + 1 */ + int ngroups; /* assemblies */ + int nshared; /* shared trees */ + cs_stats_t *stats; + bool bind_inherited; /* CS_BIND_INHERITED */ + cs_undo_t undo; /* of the file being parsed */ + int rejected; /* files whose scope, written in this run, was refused */ + /* the unit part of the using steps, shared by the resolve workers; made + * once the index is complete (NULL before, and when memory ran out) */ + struct cs_unit_memo *unit_memo; +} cs_index_t; + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +/* Test seam: scope levels, supertypes and overloads the resolver looked at, + * and the directories it walked for the files' projects, since the last + * reset. A test holds it against the size of its input. */ +static _Atomic uint64_t cs_work_counter; +/* New import index construction is measured separately from lookup work. */ +static _Atomic uint64_t cs_index_work_counter; +/* Buckets of the node pass's scratch table walked when it is emptied. */ +static _Atomic uint64_t cs_scratch_work_counter; +static _Atomic bool cs_fail_candidate_alloc; +static _Atomic bool cs_candidate_alloc_failed; + +void cbm_doclink_cs_test_fail_candidate_alloc(bool enabled) { + atomic_store(&cs_candidate_alloc_failed, false); + atomic_store(&cs_fail_candidate_alloc, enabled); +} + +bool cbm_doclink_cs_test_candidate_alloc_failed(void) { + return atomic_load(&cs_candidate_alloc_failed); +} + +void cbm_doclink_cs_test_work_reset(void) { + atomic_store(&cs_work_counter, 0); + atomic_store(&cs_index_work_counter, 0); + atomic_store(&cs_scratch_work_counter, 0); +} + +uint64_t cbm_doclink_cs_test_work(void) { + return atomic_load(&cs_work_counter); +} + +uint64_t cbm_doclink_cs_test_index_work(void) { + return atomic_load(&cs_index_work_counter); +} + +uint64_t cbm_doclink_cs_test_scratch_work(void) { + return atomic_load(&cs_scratch_work_counter); +} + +static void cs_work(uint64_t n) { + atomic_fetch_add_explicit(&cs_work_counter, n, memory_order_relaxed); +} + +static void cs_index_work(uint64_t n) { + atomic_fetch_add_explicit(&cs_index_work_counter, n, memory_order_relaxed); +} + +static void cs_scratch_work(uint64_t n) { + atomic_fetch_add_explicit(&cs_scratch_work_counter, n, memory_order_relaxed); +} +#else +static void cs_work(uint64_t n) { + (void)n; +} + +static void cs_index_work(uint64_t n) { + (void)n; +} + +static void cs_scratch_work(uint64_t n) { + (void)n; +} +#endif + +/* ── Small helpers ───────────────────────────────────────────────── */ + +/* Index memory. A failed allocation is remembered: whatever a caller does + * with its NULL, the build as a whole reports failure instead of handing out + * an index that silently lacks a declaration. */ +static void *ix_alloc(cs_index_t *ix, size_t n) { + void *p = cbm_arena_alloc(&ix->arena, n ? n : SKIP_ONE); + ix->oom = ix->oom || !p; + return p; +} + +static void *ix_zalloc(cs_index_t *ix, size_t n) { + void *p = ix_alloc(ix, n); + if (p) { + memset(p, 0, n ? n : SKIP_ONE); + } + return p; +} + +static char *ix_strndup(cs_index_t *ix, const char *s, size_t n) { + char *p = cbm_arena_strndup(&ix->arena, s, n); + ix->oom = ix->oom || !p; + return p; +} + +static char *ix_strdup(cs_index_t *ix, const char *s) { + char *p = cbm_arena_strdup(&ix->arena, s ? s : ""); + ix->oom = ix->oom || !p; + return p; +} + +/* Add `key` to a name set; false when memory ran out. */ +static bool ht_mark(cs_index_t *ix, CBMHashTable *ht, const char *key) { + if (cbm_ht_get(ht, key)) { + return true; + } + char *k = ix_strdup(ix, key); + if (!k) { + return false; + } + cbm_ht_set(ht, k, (void *)k); + return true; +} + +/* Grow an undo array to hold one more entry. false when memory ran out. */ +static bool undo_room(cs_index_t *ix, void **arr, int n, int *cap, size_t size) { + if (n < *cap) { + return true; + } + int ncap = *cap ? *cap * PAIR_LEN : CBM_SZ_16; + void *grown = cbm_alloc(CBM_MEM_CLASS_OTHER, (size_t)ncap * size); + if (!grown) { + ix->oom = true; + return false; + } + if (n > 0) { + memcpy(grown, *arr, (size_t)n * size); + } + cbm_free(CBM_MEM_CLASS_OTHER, *arr); + *arr = grown; + *cap = ncap; + return true; +} + +/* Remember the flags of namespace `ns` before the file's scope sets them. */ +static bool undo_note_ns(cs_index_t *ix, int ns) { + cs_undo_t *u = &ix->undo; + if (ns >= u->nnss) { + return true; /* made by this file: taken back as a whole */ + } + if (!undo_room(ix, (void **)&u->flags, u->nflags, &u->cap_flags, sizeof(cs_ns_was_t))) { + return false; + } + u->flags[u->nflags++] = + (cs_ns_was_t){.ns = ns, .declared = ix->nss[ns].declared, .prod = ix->nss[ns].prod}; + return true; +} + +/* A type name declared in `f` at a place no namespace or outer type could be + * established for. Product code never binds test-only declarations, so a name + * only test files quarantine blocks references from test files only. */ +static bool quarantine_name(cs_index_t *ix, const cs_file_t *f, const char *name) { + CBMHashTable *ht = f->is_test ? ix->quarantine_test : ix->quarantine; + if (cbm_ht_get(ht, name)) { + return true; + } + char *k = ix_strdup(ix, name); + cs_undo_t *u = &ix->undo; + if (!k || !undo_room(ix, (void **)&u->marks, u->nmarks, &u->cap_marks, sizeof(cs_mark_was_t))) { + return false; + } + cbm_ht_set(ht, k, (void *)k); + u->marks[u->nmarks++] = (cs_mark_was_t){.ht = ht, .key = k}; + return true; +} + +static bool cs_is_test_path(const char *rel) { + /* C# test code: a directory named tests/test, or a *.Tests / *.UnitTests / + * *.FunctionalTests project directory (the prototype's rule). */ + const char *p = rel; + for (;;) { + const char *slash = strchr(p, '/'); + if (!slash) { + return false; + } + size_t n = (size_t)(slash - p); + char seg[CBM_SZ_256]; + if (n < sizeof(seg)) { + for (size_t i = 0; i < n; i++) { + seg[i] = (char)tolower((unsigned char)p[i]); + } + seg[n] = '\0'; + if (strcmp(seg, "tests") == 0 || strcmp(seg, "test") == 0) { + return true; + } + static const char *const sfx[] = {".tests", ".unittests", ".functionaltests"}; + for (size_t k = 0; k < sizeof(sfx) / sizeof(sfx[0]); k++) { + size_t sl = strlen(sfx[k]); + if (n >= sl && strcmp(seg + n - sl, sfx[k]) == 0) { + return true; + } + } + } + p = slash + SKIP_ONE; + } +} + +static bool cs_ci_suffix(const char *s, const char *sfx) { + size_t n = strlen(s); + size_t sl = strlen(sfx); + if (n < sl) { + return false; + } + for (size_t i = 0; i < sl; i++) { + if (tolower((unsigned char)s[n - sl + i]) != sfx[i]) { + return false; + } + } + return true; +} + +/* Split `s` in place at tabs into at most `max` fields; returns the count. */ +static int split_fields(char *s, char **out, int max) { + int n = 0; + out[n++] = s; + for (char *p = s; *p && n < max; p++) { + if (*p == '\t') { + *p = '\0'; + out[n++] = p + SKIP_ONE; + } + } + return n; +} + +static int count_list(const char *s, char sep) { + if (!s || !s[0]) { + return 0; + } + int n = 1; + for (const char *p = s; *p; p++) { + n += *p == sep; + } + return n; +} + +/* A decimal index of a scope record: its value when it is one below `limit`, + * else CS_NONE. A scope comes from the store: nothing in it is trusted to be + * in range. */ +static int field_index(const char *s, int limit) { + if (!s[0] || strlen(s) > CBM_SZ_8) { + return CS_NONE; + } + int v = 0; + for (const char *p = s; *p; p++) { + if (!isdigit((unsigned char)*p)) { + return CS_NONE; + } + v = (v * 10) + (*p - '0'); + } + return v < limit ? v : CS_NONE; +} + +/* ── Namespaces ──────────────────────────────────────────────────── */ + +static bool ns_key(char *key, size_t cap, int parent, const char *seg, size_t len) { + if (len == 0 || len >= CS_NAME_BUF) { + return false; + } + int kl = snprintf(key, cap, "%d\x1f%.*s", parent, (int)len, seg); + return kl > 0 && (size_t)kl < cap; +} + +/* The namespace `seg` directly under `parent`, or CS_NONE. */ +static int ns_find(const cs_index_t *ix, int parent, const char *seg, size_t len) { + char key[CS_NAME_BUF + CBM_SZ_16]; + if (!ns_key(key, sizeof(key), parent, seg, len)) { + return CS_NONE; + } + intptr_t v = (intptr_t)cbm_ht_get(ix->ns_by_key, key); + return v > 0 ? (int)(v - SKIP_ONE) : CS_NONE; +} + +/* The same, created when it is not there yet. CS_NONE for a name that is + * none, and when memory ran out. */ +static int ns_make(cs_index_t *ix, int parent, const char *seg, size_t len) { + int found = ns_find(ix, parent, seg, len); + char key[CS_NAME_BUF + CBM_SZ_16]; + if (found >= 0 || !ns_key(key, sizeof(key), parent, seg, len)) { + return found; + } + if (ix->nnss >= ix->nscap) { + int ncap = ix->nscap ? ix->nscap * PAIR_LEN : CBM_SZ_256; + cs_ns_t *grown = (cs_ns_t *)ix_alloc(ix, (size_t)ncap * sizeof(cs_ns_t)); + if (!grown) { + return CS_NONE; + } + if (ix->nnss > 0) { + memcpy(grown, ix->nss, (size_t)ix->nnss * sizeof(cs_ns_t)); + } + ix->nss = grown; + ix->nscap = ncap; + } + char *k = ix_strdup(ix, key); + char *name = ix_strndup(ix, seg, len); + if (!k || !name) { + return CS_NONE; + } + int id = ix->nnss++; + ix->nss[id] = + (cs_ns_t){.parent = parent, .depth = ix->nss[parent].depth + SKIP_ONE, .name = name}; + cbm_ht_set(ix->ns_by_key, k, (void *)(intptr_t)(id + SKIP_ONE)); + return id; +} + +/* The namespace the dotted `path` names under `from`, every segment created + * on the way; CS_NONE for a bad name, when memory ran out, and past the + * scanner's nesting limit: a scope read back from the store is held to the + * limit its writer keeps (every enclosing namespace is a lookup step). */ +static int ns_make_path(cs_index_t *ix, int from, const char *path) { + int ns = from; + for (const char *p = path; ns >= 0 && *p;) { + if (ix->nss[ns].depth >= CS_MAX_NEST) { + return CS_NONE; + } + const char *dot = strchr(p, '.'); + size_t n = dot ? (size_t)(dot - p) : strlen(p); + ns = ns_make(ix, ns, p, n); + p = dot ? dot + SKIP_ONE : p + n; + } + return ns; +} + +/* ── Scope blob parsing ──────────────────────────────────────────── */ + +static int tparam_cmp(const void *a, const void *b) { + const cs_tparam_t *x = (const cs_tparam_t *)a; + const cs_tparam_t *y = (const cs_tparam_t *)b; + if (x->owner != y->owner) { + return x->owner < y->owner ? -1 : 1; + } + int c = strcmp(x->name, y->name); + return c ? c : (x->pos > y->pos) - (x->pos < y->pos); +} + +/* Position of the type parameter `name` of `owner` in `f`, or CS_NONE. */ +static int tparam_find(const cs_file_t *f, int owner, const char *name, size_t len) { + int lo = 0; + int hi = f->ntparams; + while (lo < hi) { + int mid = lo + ((hi - lo) / PAIR_LEN); + const cs_tparam_t *t = &f->tparams[mid]; + int c = t->owner != owner ? (t->owner < owner ? -1 : 1) : strncmp(t->name, name, len); + if (c == 0 && t->name[len] != '\0') { + c = 1; + } + if (c < 0) { + lo = mid + SKIP_ONE; + } else if (c > 0) { + hi = mid; + } else { + return t->pos; + } + } + return CS_NONE; +} + +static int span_cmp(const void *a, const void *b) { + const cs_span_lines_t *x = (const cs_span_lines_t *)a; + const cs_span_lines_t *y = (const cs_span_lines_t *)b; + if (x->from != y->from) { + return x->from < y->from ? -1 : 1; + } + return (x->to > y->to) - (x->to < y->to); +} + +typedef struct { + uint32_t start; + bool plain; /* no generic method: those stand first among one line's members */ + int idx; +} cs_start_key_t; + +static int start_key_cmp(const void *a, const void *b) { + const cs_start_key_t *x = (const cs_start_key_t *)a; + const cs_start_key_t *y = (const cs_start_key_t *)b; + if (x->start != y->start) { + return x->start < y->start ? -1 : 1; + } + if (x->plain != y->plain) { + return x->plain ? 1 : -1; + } + return (x->idx > y->idx) - (x->idx < y->idx); +} + +static int using_region_cmp(const void *a, const void *b) { + const cs_using_t *x = (const cs_using_t *)a; + const cs_using_t *y = (const cs_using_t *)b; + if (x->region != y->region) { + return x->region < y->region ? -1 : 1; + } + /* document order within a region: the targets are fields of one buffer, + * in the order they were read */ + return (x->target > y->target) - (x->target < y->target); +} + +/* What a scan of the blob's lines found, to size the file's arrays. */ +typedef struct { + int regions; + int usings; + int types; + int members; + int unplaced; + int tparams; +} cs_counts_t; + +static void count_records(const char *buf, cs_counts_t *n) { + memset(n, 0, sizeof(*n)); + n->regions = SKIP_ONE; /* region 0: the file */ + for (const char *p = buf; *p;) { + const char *nl = strchr(p, '\n'); + switch (*p) { + case 'R': + n->regions++; + break; + case 'U': + n->usings++; + break; + case 'T': + n->types++; + break; + case 'M': + n->members++; + break; + case 'X': + n->unplaced++; + break; + default: + break; + } + /* an upper bound for the type parameters: one per record that can + * have a list, and one more per comma */ + if (*p == 'T' || *p == 'M') { + n->tparams++; + for (const char *q = p; *q && *q != '\n'; q++) { + n->tparams += *q == ','; + } + } + if (!nl) { + break; + } + p = nl + SKIP_ONE; + } +} + +/* Add the ','-joined type parameters `list` of `owner`. The list is cut at + * its commas in place, so every name is a string of its own afterwards. */ +static void add_tparams(cs_file_t *f, int owner, char *list) { + int pos = 0; + for (char *p = list; p && *p;) { + char *e = strchr(p, ','); + if (e) { + *e = '\0'; + } + f->tparams[f->ntparams++] = (cs_tparam_t){.owner = owner, .pos = pos++, .name = p}; + p = e ? e + SKIP_ONE : NULL; + } +} + +static bool parse_region(cs_index_t *ix, cs_file_t *f, char **fld, int n, int *next_region) { + /* R id parent start end name */ + if (n != 6 || !fld[5][0]) { + return false; /* the scanner names every region: an empty name would + * nest one without deepening the namespace */ + } + int id = field_index(fld[1], f->nregions); + int parent = field_index(fld[2], f->nregions); + if (id != *next_region || parent < 0 || parent >= id) { + return false; /* regions are numbered in order, each under an earlier one */ + } + (*next_region)++; + int ns = ns_make_path(ix, f->regions[parent].ns, fld[5]); + if (ns < 0 || !undo_note_ns(ix, ns)) { + return false; + } + ix->nss[ns].declared = true; + /* product code declares it, and with it every namespace above: one that + * is marked has its upper ones marked already */ + for (int up = ns; !f->is_test && up > 0 && !ix->nss[up].prod; up = ix->nss[up].parent) { + if (!undo_note_ns(ix, up)) { + return false; + } + ix->nss[up].prod = true; + } + f->regions[id] = (cs_region_t){.parent = parent, + .ns = ns, + .start = (uint32_t)strtoul(fld[3], NULL, 10), + .end = (uint32_t)strtoul(fld[4], NULL, 10)}; + return true; +} + +static bool parse_using(cs_file_t *f, char **fld, int n, int regions_seen) { + /* U region kind alias target */ + if (n != 5) { + return false; + } + int region = field_index(fld[1], regions_seen); + char kind = fld[2][0]; + bool global = kind && fld[2][1] == 'g'; + if (region < 0 || !kind || !strchr("nsa", kind) || (fld[2][1] && !(global && !fld[2][2]))) { + return false; + } + f->usings[f->nusings++] = (cs_using_t){.region = region, + .kind = kind, + .global = global, + .alias = strcmp(fld[3], "-") == 0 ? "" : fld[3], + .target = fld[4], + .ns = CS_NONE, + .ent = CS_NONE}; + return true; +} + +static bool parse_type(cs_file_t *f, char **fld, int n, int regions_seen) { + /* T region start end kind outer name tparams bases */ + if (n != 9) { + return false; + } + cs_type_t *t = &f->types[f->ntypes]; + memset(t, 0, sizeof(*t)); + t->region = field_index(fld[1], regions_seen); + t->start = (uint32_t)strtoul(fld[2], NULL, 10); + t->end = (uint32_t)strtoul(fld[3], NULL, 10); + t->kind = fld[4][0]; + const char *flags = t->kind ? fld[4] + SKIP_ONE : ""; + t->partial = flags[0] == 'p'; + flags += t->partial; + t->incomplete = flags[0] == '!'; + flags += t->incomplete; + bool top_level = strcmp(fld[5], "-") == 0; + t->outer = top_level ? CS_NONE : field_index(fld[5], f->ntypes); + t->depth = t->outer >= 0 ? f->types[t->outer].depth + SKIP_ONE : 0; + t->name = fld[6]; + t->bases = fld[8]; + t->arity = count_list(fld[7], ','); + t->entity = CS_NONE; + size_t nl = strlen(t->name); + if (t->region < 0 || !t->kind || !strchr("csiertd", t->kind) || flags[0] || + (t->outer < 0 && !top_level) || nl == 0 || nl >= CS_NAME_BUF || t->depth >= CS_MAX_NEST) { + return false; + } + add_tparams(f, f->ntypes, fld[7]); + f->ntypes++; + return true; +} + +static bool parse_member(cs_file_t *f, char **fld, int n) { + /* M start kind explicit type name tparams sig */ + if (n != 8) { + return false; + } + cs_member_t *m = &f->members[f->nmembers]; + memset(m, 0, sizeof(*m)); + m->start = (uint32_t)strtoul(fld[1], NULL, 10); + m->kind = fld[2][0]; + m->is_static = m->kind && fld[2][1] == 's'; + m->explicit_impl = fld[3][0] == '1'; + m->type = field_index(fld[4], f->ntypes); + m->name = fld[5]; + m->arity = count_list(fld[6], ','); + bool callable_like = m->kind == 'c' || m->kind == 'o' || m->kind == 'x'; + m->sig = callable_like ? fld[7] : NULL; + m->nparams = (m->sig && m->sig[0]) ? count_list(m->sig, '|') : 0; + size_t nl = strlen(m->name); + if (m->type < 0 || !m->kind || !strchr("cvpeox", m->kind) || + fld[2][m->is_static ? PAIR_LEN : SKIP_ONE] || nl == 0 || nl >= CS_NAME_BUF) { + return false; + } + /* owners of type parameters: the types come first, the members after + * the LAST type the blob has, whose count is known only at the end */ + add_tparams(f, -(f->nmembers + PAIR_LEN), fld[6]); + f->nmembers++; + return true; +} + +/* Everything the records of one blob say, into `f`. false for a blob this + * code did not write (or one the store damaged), and when memory ran out. */ +static bool parse_records(cs_index_t *ix, cs_file_t *f, char *buf) { + int next_region = SKIP_ONE; + bool first = true; + for (char *line = buf; line && *line;) { + char *nl = strchr(line, '\n'); + if (nl) { + *nl = '\0'; + } + bool ok = true; + if (first) { + first = false; + ok = strcmp(line, CBM_DOCLINK_CS_SCOPE_TAG) == 0; + } else { + char *fld[CS_FIELDS + SKIP_ONE]; + int n = split_fields(line, fld, CS_FIELDS + SKIP_ONE); + switch (line[0]) { + case 'R': + ok = parse_region(ix, f, fld, n, &next_region); + break; + case 'U': + ok = parse_using(f, fld, n, next_region); + break; + case 'T': + ok = parse_type(f, fld, n, next_region); + break; + case 'M': + ok = parse_member(f, fld, n); + break; + case 'X': + ok = n == 3; + if (ok) { + f->unplaced[f->nunplaced++] = + (cs_span_lines_t){.from = (uint32_t)strtoul(fld[1], NULL, 10), + .to = (uint32_t)strtoul(fld[2], NULL, 10)}; + } + break; + case 'Q': + ok = n == PAIR_LEN && fld[1][0] && quarantine_name(ix, f, fld[1]); + break; + default: + ok = false; + break; + } + } + if (!ok) { + return false; + } + line = nl ? nl + SKIP_ONE : NULL; + } + return !first && next_region == f->nregions; +} + +/* Sort the usings by region, give every region its range, and close the + * unplaced ranges into sorted, disjoint ones. */ +static void finish_ranges(cs_file_t *f) { + if (f->nusings > 1) { + qsort(f->usings, (size_t)f->nusings, sizeof(cs_using_t), using_region_cmp); + } + int u = 0; + for (int r = 0; r < f->nregions; r++) { + f->regions[r].u_lo = u; + while (u < f->nusings && f->usings[u].region == r) { + u++; + } + f->regions[r].u_hi = u; + } + if (f->nunplaced > 1) { + qsort(f->unplaced, (size_t)f->nunplaced, sizeof(cs_span_lines_t), span_cmp); + int w = 0; + for (int i = 1; i < f->nunplaced; i++) { + if (f->unplaced[i].from <= f->unplaced[w].to) { + if (f->unplaced[i].to > f->unplaced[w].to) { + f->unplaced[w].to = f->unplaced[i].to; + } + } else { + f->unplaced[++w] = f->unplaced[i]; + } + } + f->nunplaced = w + SKIP_ONE; + } +} + +/* The type parameters by (owner, name), and the members in line order -- of + * the members that start at one line, the generic methods first. + * false when memory ran out. */ +static bool finish_lookups(cs_index_t *ix, cs_file_t *f) { + /* a member's parameters were filed under -(index + 2) while the type + * count was still growing */ + for (int i = 0; i < f->ntparams; i++) { + if (f->tparams[i].owner < 0) { + f->tparams[i].owner = f->ntypes + (-f->tparams[i].owner - PAIR_LEN); + } + } + if (f->ntparams > 1) { + qsort(f->tparams, (size_t)f->ntparams, sizeof(cs_tparam_t), tparam_cmp); + } + if (f->nmembers == 0) { + return true; + } + f->members_by_start = (int *)ix_alloc(ix, (size_t)f->nmembers * sizeof(int)); + cs_start_key_t *keys = (cs_start_key_t *)cbm_alloc( + CBM_MEM_CLASS_OTHER, (size_t)f->nmembers * sizeof(cs_start_key_t)); + if (!f->members_by_start || !keys) { + cbm_free(CBM_MEM_CLASS_OTHER, keys); + ix->oom = true; + return false; + } + for (int i = 0; i < f->nmembers; i++) { + const cs_member_t *m = &f->members[i]; + keys[i] = (cs_start_key_t){ + .start = m->start, .plain = !(m->kind == 'c' && m->arity > 0), .idx = i}; + } + qsort(keys, (size_t)f->nmembers, sizeof(*keys), start_key_cmp); + for (int i = 0; i < f->nmembers; i++) { + f->members_by_start[i] = keys[i].idx; + } + cbm_free(CBM_MEM_CLASS_OTHER, keys); + return true; +} + +static bool parse_scope(cs_index_t *ix, cs_file_t *f, const char *blob) { + char *buf = ix_strdup(ix, blob); + if (!buf) { + return false; + } + cs_counts_t n; + count_records(buf, &n); + f->nregions = n.regions; + f->regions = (cs_region_t *)ix_zalloc(ix, (size_t)n.regions * sizeof(cs_region_t)); + f->usings = (cs_using_t *)ix_alloc(ix, (size_t)n.usings * sizeof(cs_using_t)); + f->types = (cs_type_t *)ix_alloc(ix, (size_t)n.types * sizeof(cs_type_t)); + f->members = (cs_member_t *)ix_alloc(ix, (size_t)n.members * sizeof(cs_member_t)); + f->unplaced = (cs_span_lines_t *)ix_alloc(ix, (size_t)n.unplaced * sizeof(cs_span_lines_t)); + f->tparams = (cs_tparam_t *)ix_alloc(ix, (size_t)n.tparams * sizeof(cs_tparam_t)); + if (ix->oom) { + return false; + } + f->regions[0].parent = CS_NONE; + if (!parse_records(ix, f, buf)) { + return false; + } + finish_ranges(f); + return finish_lookups(ix, f); +} + +/* ── Units (C# projects) and their global usings ─────────────────── */ + +/* Nothing here reads the disk: a project file is one the index holds (its + * path among the resolver's files, its content in its scope blob). What + * discovery did not take -- a symbolic link, a named pipe -- is no project + * file. */ + +static const char CS_PROJECT_EXT[] = ".csproj"; + +/* The name of a project directory that holds a reference assembly's source: + * the .NET convention (/ref beside /src). It tells the + * stub from the implementation inside one assembly, and nothing else. */ +static const char CS_REF_DIR[] = "ref"; + +/* Length of the directory above the one of length `len` in `path`. */ +static size_t parent_dir_len(const char *path, size_t len) { + while (len > 0 && path[len - SKIP_ONE] != '/') { + len--; + } + return len > 0 ? len - SKIP_ONE : 0; +} + +/* Take the project files out of the resolver's file list: every MSBuild blob + * goes to the evaluator, every *.csproj marks its directory as a project and + * names it. false when memory ran out. */ +static bool collect_projects(cs_index_t *ix, const cbm_doclink_build_in_t *in) { + int n = 0; + for (int i = 0; i < in->file_count; i++) { + n += cs_ci_suffix(in->files[i].rel_path, CS_PROJECT_EXT); + } + ix->projects = (const char **)ix_alloc(ix, (size_t)n * sizeof(char *)); + if (!ix->projects) { + return false; + } + for (int i = 0; i < in->file_count; i++) { + const cbm_doclink_file_t *src = &in->files[i]; + if (cbm_msb_is_project_scope(src->scope) && + !cbm_msb_add(ix->msb, src->rel_path, src->scope)) { + return false; + } + /* a project file of an SDK that compiles nothing (it runs build steps + * or builds other projects) is the project of no source file */ + if (!cs_ci_suffix(src->rel_path, CS_PROJECT_EXT) || + !cbm_msb_compiles(ix->msb, src->rel_path)) { + continue; + } + const char *slash = strrchr(src->rel_path, '/'); + const char *name = slash ? slash + SKIP_ONE : src->rel_path; + char *dir = ix_strndup(ix, src->rel_path, slash ? (size_t)(slash - src->rel_path) : 0); + char *path = ix_strdup(ix, src->rel_path); + if (!dir || !path) { + return false; + } + ix->projects[ix->nprojects++] = path; /* the file list is in path order */ + cs_pdir_t *pd = (cs_pdir_t *)cbm_ht_get(ix->project_dirs, dir); + if (pd) { + pd->count++; + continue; + } + pd = (cs_pdir_t *)ix_zalloc(ix, sizeof(*pd)); + char *stem = ix_strndup(ix, name, strlen(name) - (sizeof(CS_PROJECT_EXT) - SKIP_ONE)); + if (!pd || !stem) { + return false; + } + pd->count = SKIP_ONE; + pd->stem = stem; + cbm_ht_set(ix->project_dirs, dir, pd); + /* the directory and every one above it has a project in or below + * it; one that is marked has its upper ones marked already */ + for (size_t len = strlen(dir);; len = parent_dir_len(dir, len)) { + char *above = ix_strndup(ix, dir, len); + if (!above) { + return false; + } + if (cbm_ht_get(ix->project_above, above)) { + break; + } + cbm_ht_set(ix->project_above, above, above); + if (len == 0) { + break; + } + } + } + return true; +} + +/* The assembly of a project directory: the one its project file names -- + * projects whose project files have the same name are one assembly. A + * directory with several project files is an assembly of its own (which of + * them compiles a file there is not known). CS_NONE when memory ran out. */ +static int group_of(cs_index_t *ix, const cs_pdir_t *pd) { + if (pd->count != SKIP_ONE) { + return ix->ngroups++; + } + intptr_t g = (intptr_t)cbm_ht_get(ix->group_by_stem, pd->stem); + if (g <= 0) { + g = (intptr_t)ix->ngroups + SKIP_ONE; + ix->ngroups++; + cbm_ht_set(ix->group_by_stem, pd->stem, (void *)g); + } + return (int)(g - SKIP_ONE); +} + +/* The unit of directory `dir`, made on first sight: a project when `pd` says + * which project files the directory holds, else a shared tree. CS_NONE when + * memory ran out. */ +static int unit_get(cs_index_t *ix, const char *dir, const cs_pdir_t *pd) { + intptr_t v = (intptr_t)cbm_ht_get(ix->unit_by_dir, dir); + if (v > 0) { + return (int)(v - SKIP_ONE); + } + if (ix->nunits >= ix->ucap) { + int ncap = ix->ucap ? ix->ucap * PAIR_LEN : CBM_SZ_64; + cs_unit_t *grown = (cs_unit_t *)ix_alloc(ix, (size_t)ncap * sizeof(cs_unit_t)); + if (!grown) { + return CS_NONE; + } + if (ix->nunits > 0) { + memcpy(grown, ix->units, (size_t)ix->nunits * sizeof(cs_unit_t)); + } + ix->units = grown; + ix->ucap = ncap; + } + char *key = ix_strdup(ix, dir); + if (!key) { + return CS_NONE; + } + int id = ix->nunits++; + cs_unit_t *u = &ix->units[id]; + memset(u, 0, sizeof(*u)); + u->dir = key; + u->group = CS_NONE; + if (pd) { + const char *base = strrchr(key, '/'); + u->is_ref = strcmp(base ? base + SKIP_ONE : key, CS_REF_DIR) == 0; + u->group = group_of(ix, pd); + } else { + ix->nshared++; + } + cbm_ht_set(ix->unit_by_dir, key, (void *)(intptr_t)(id + SKIP_ONE)); + return id; +} + +/* The tree of the project-less directory rel[0, len): the length of its root + * -- the largest directory around it that has no project in or below it; the + * directory itself when a project stands below it (*whole stays unset: what + * is under it is not all of its tree). `memo` is what `dir_unit` holds for + * the directory the walk for a project stopped at: a tree's root when that + * directory is of a known tree, and then everything under it is of that tree + * too. `dir` is scratch for the paths asked. */ +static size_t tree_root(const cs_index_t *ix, const char *rel, size_t len, intptr_t memo, char *dir, + bool *whole) { + if (memo < CS_NONE) { + *whole = true; + return (size_t)(-(memo + PAIR_LEN)); + } + memcpy(dir, rel, len); + dir[len] = '\0'; + if (cbm_ht_get(ix->project_above, dir)) { + return len; + } + *whole = true; + size_t root = len; + while (root > 0) { + cs_work(SKIP_ONE); + size_t up = parent_dir_len(rel, root); + dir[up] = '\0'; + if (cbm_ht_get(ix->project_above, dir)) { + break; + } + root = up; + } + return root; +} + +/* The unit a file belongs to: its project -- the nearest directory at or + * above its own that holds a *.csproj. A file no project file stands above + * is of a shared tree (sources that projects elsewhere compile in): which + * programs it is part of is not known, and all such files of a repository + * are not one program. Its unit is the tree: the largest directory around it + * that has no project in or below it -- the file's own directory when a + * project stands below that. (A repository without any project file is one + * tree.) Every directory walked on the way up is remembered in `dir_unit`: + * its project's unit + 1; for one no project stands at or above, -1, or + * -(length of its tree's root + 2) when it has no project below it either. + * So a directory is walked once, for its project and for its tree, however + * many files it and the directories under it hold. CS_NONE when memory ran + * out. */ +static int unit_of(cs_index_t *ix, CBMHashTable *dir_unit, const char *rel) { + const char *slash = strrchr(rel, '/'); + size_t len = slash ? (size_t)(slash - rel) : 0; + char *dir = (char *)cbm_alloc(CBM_MEM_CLASS_OTHER, len + SKIP_ONE); + if (!dir) { + ix->oom = true; + return CS_NONE; + } + memcpy(dir, rel, len); + size_t cur = len; + int unit = CS_NONE; + intptr_t memo = 0; + bool projectless = false; + bool failed = false; + for (;;) { + cs_work(SKIP_ONE); + dir[cur] = '\0'; + memo = (intptr_t)cbm_ht_get(dir_unit, dir); + if (memo > 0) { + unit = (int)(memo - SKIP_ONE); + break; + } + const cs_pdir_t *pd = (const cs_pdir_t *)cbm_ht_get(ix->project_dirs, dir); + if (pd) { + unit = unit_get(ix, dir, pd); + failed = unit < 0; + break; + } + if (memo < 0 || cur == 0) { + projectless = true; + break; + } + cur = parent_dir_len(rel, cur); + } + bool whole = false; + size_t root = projectless ? tree_root(ix, rel, len, memo, dir, &whole) : 0; + for (size_t l = len; !failed;) { + memcpy(dir, rel, l); + dir[l] = '\0'; + if (!cbm_ht_get(dir_unit, dir)) { + char *key = ix_strdup(ix, dir); + if (!key) { + failed = true; + break; + } + intptr_t known = unit + SKIP_ONE; + if (projectless) { + known = (whole && l >= root) ? -(intptr_t)(root + PAIR_LEN) : CS_NONE; + } + cbm_ht_set(dir_unit, key, (void *)known); + } + if (l <= cur) { + break; + } + l = parent_dir_len(rel, l); + } + if (projectless && !failed) { + memcpy(dir, rel, root); + dir[root] = '\0'; + unit = unit_get(ix, dir, NULL); + } + cbm_free(CBM_MEM_CLASS_OTHER, dir); + return failed ? CS_NONE : unit; +} + +static bool unit_add_using(cs_index_t *ix, cs_unit_t *u, char kind, const char *alias, + const char *target) { + if (u->nusings >= u->cap) { + int ncap = u->cap ? u->cap * PAIR_LEN : CBM_SZ_8; + cs_using_t *grown = (cs_using_t *)ix_alloc(ix, (size_t)ncap * sizeof(cs_using_t)); + if (!grown) { + return false; + } + if (u->nusings > 0) { + memcpy(grown, u->usings, (size_t)u->nusings * sizeof(cs_using_t)); + } + u->usings = grown; + u->cap = ncap; + } + char *a = ix_strdup(ix, alias); + char *t = ix_strdup(ix, target); + if (!a || !t) { + return false; + } + u->usings[u->nusings++] = (cs_using_t){ + .kind = kind, .global = true, .alias = a, .target = t, .ns = CS_NONE, .ent = CS_NONE}; + return true; +} + +static int unit_using_cmp(const void *a, const void *b) { + const cs_using_t *x = (const cs_using_t *)a; + const cs_using_t *y = (const cs_using_t *)b; + if (x->kind != y->kind) { + return x->kind < y->kind ? -1 : 1; + } + int c = strcmp(x->alias, y->alias); + return c ? c : strcmp(x->target, y->target); +} + +/* What the project files could not tell, over all of them. */ +typedef struct { + int unevaluable; + int outside; +} cs_msb_totals_t; + +/* Global usings of every unit: the `global using` directives of its files + * (of a namespace, of a type's static members, of an alias alike) and the + * items of every project file in its directory -- all of them, in + * path order, so the union does not depend on how a directory is listed. + * false when memory ran out. */ +static bool units_collect_usings(cs_index_t *ix, cs_msb_totals_t *totals) { + for (int fi = 0; fi < ix->nfiles; fi++) { + const cs_file_t *f = &ix->files[fi]; + for (int u = 0; f->unit >= 0 && u < f->nusings; u++) { + const cs_using_t *us = &f->usings[u]; + if (us->global && + !unit_add_using(ix, &ix->units[f->unit], us->kind, us->alias, us->target)) { + return false; + } + } + } + cbm_msb_eval_context_t *eval_context = cbm_msb_eval_context_new(ix->msb); + if (!eval_context) { + return false; + } + for (int p = 0; p < ix->nprojects; p++) { + const char *slash = strrchr(ix->projects[p], '/'); + char *dir = ix_strndup(ix, ix->projects[p], slash ? (size_t)(slash - ix->projects[p]) : 0); + intptr_t unit = dir ? (intptr_t)cbm_ht_get(ix->unit_by_dir, dir) : 0; + if (unit <= 0) { + continue; /* a project directory without a C# file */ + } + cs_unit_t *u = &ix->units[unit - SKIP_ONE]; + if (!cbm_msb_has(ix->msb, ix->projects[p])) { + /* a *.csproj the extractor has no blob for: what it sets is not known */ + totals->unevaluable++; + u->open = true; + continue; + } + cbm_msb_result_t res; + if (!cbm_msb_eval_context_eval(eval_context, ix->projects[p], &res)) { + cbm_msb_eval_context_free(eval_context); + return false; + } + bool ok = true; + for (int k = 0; ok && k < res.count; k++) { + ok = unit_add_using(ix, u, res.usings[k].kind, res.usings[k].alias, + res.usings[k].target); + } + u->open = u->open || res.open; + totals->unevaluable += res.unevaluable; + totals->outside += res.outside; + cbm_msb_result_free(&res); + if (!ok) { + cbm_msb_eval_context_free(eval_context); + return false; + } + } + cbm_msb_eval_context_free(eval_context); + for (int ui = 0; ui < ix->nunits; ui++) { + cs_unit_t *u = &ix->units[ui]; + if (u->nusings < PAIR_LEN) { + continue; + } + qsort(u->usings, (size_t)u->nusings, sizeof(cs_using_t), unit_using_cmp); + int w = 0; + for (int i = 1; i < u->nusings; i++) { + if (unit_using_cmp(&u->usings[w], &u->usings[i]) != 0) { + u->usings[++w] = u->usings[i]; + } + } + u->nusings = w + SKIP_ONE; + } + return !ix->oom; +} + +/* ── Entities ────────────────────────────────────────────────────── */ + +/* The owner of a complete declaration in the shared tree `unit`. */ +static int shared_owner(int unit) { + return -(unit + PAIR_LEN); +} + +/* The entity of this scope, name, arity and owner, created on first sight. A + * top-level type has a namespace `ns` and an `owner`; a nested type has its + * outer entity `parent` (and the owner 0: its outer type's is the one that + * counts). CS_NONE when memory ran out. */ +static int entity_get(cs_index_t *ix, int ns, int parent, int owner, const char *name, int arity, + char kind) { + char key[CS_NAME_BUF + CBM_SZ_64]; + int kl = snprintf(key, sizeof(key), "%c%d\x1f%d\x1f%d\x1f%s", parent >= 0 ? 'E' : 'N', + parent >= 0 ? parent : ns, owner, arity, name); + if (kl < 0 || kl >= (int)sizeof(key)) { + ix->oom = true; /* cannot be: parse_type bounds the name */ + return CS_NONE; + } + intptr_t v = (intptr_t)cbm_ht_get(ix->ent_by_key, key); + if (v > 0) { + return (int)(v - SKIP_ONE); + } + if (ix->nents >= ix->ecap) { + int ncap = ix->ecap ? ix->ecap * PAIR_LEN : CBM_SZ_1K; + cs_entity_t *grown = + (cs_entity_t *)cbm_alloc(CBM_MEM_CLASS_OTHER, (size_t)ncap * sizeof(cs_entity_t)); + if (!grown) { + ix->oom = true; + return CS_NONE; + } + if (ix->nents > 0) { + memcpy(grown, ix->ents, (size_t)ix->nents * sizeof(cs_entity_t)); + } + cbm_free(CBM_MEM_CLASS_OTHER, ix->ents); + ix->ents = grown; + ix->ecap = ncap; + } + char *k = ix_strdup(ix, key); + if (!k) { + return CS_NONE; + } + int id = ix->nents++; + cs_entity_t *e = &ix->ents[id]; + memset(e, 0, sizeof(*e)); + e->ns = parent >= 0 ? CS_NONE : ns; + e->parent = parent >= 0 ? parent : CS_NONE; + e->owner = owner; + e->twin = CS_NONE; + e->stub = CS_NONE; + e->shared_parts = parent >= 0 ? ix->ents[parent].shared_parts : owner == CS_POOL; + e->name = name; + e->arity = arity; + e->kind = kind; + e->all_test = true; + cbm_ht_set(ix->ent_by_key, k, (void *)(intptr_t)(id + SKIP_ONE)); + return id; +} + +static bool entity_add_decl(cs_index_t *ix, cs_entity_t *e, int file, int type) { + if (e->ndecls >= e->dcap) { + int ncap = e->dcap ? e->dcap * PAIR_LEN : CBM_SZ_2; + cs_decl_t *grown = (cs_decl_t *)ix_alloc(ix, (size_t)ncap * sizeof(cs_decl_t)); + if (!grown) { + return false; + } + if (e->ndecls > 0) { + memcpy(grown, e->decls, (size_t)e->ndecls * sizeof(cs_decl_t)); + } + e->decls = grown; + e->dcap = ncap; + } + e->decls[e->ndecls++] = (cs_decl_t){.file = file, .type = type}; + const cs_file_t *f = &ix->files[file]; + e->any_prod = e->any_prod || !f->is_test; + e->all_test = e->all_test && f->is_test; + e->has_impl = e->has_impl || !f->is_ref; + e->impl_partial = e->impl_partial || (!f->is_ref && f->types[type].partial); + e->incomplete = e->incomplete || f->types[type].incomplete; + return true; +} + +/* A top-level type as one shared tree declares it. false when it does not + * fit (cannot be: parse_type bounds the name). */ +static bool tree_type_key(char *key, size_t cap, const cs_file_t *f, const cs_type_t *t) { + int n = snprintf(key, cap, "%d\x1f%d\x1f%d\x1f%s", f->regions[t->region].ns, f->unit, t->arity, + t->name); + return n > 0 && (size_t)n < cap; +} + +/* The top-level types the shared trees declare partial, by tree: into + * `partials` (keys in `keys`). false when memory ran out. */ +static bool tree_partials(const cs_index_t *ix, CBMHashTable *partials, CBMArena *keys) { + char key[CS_NAME_BUF + CBM_SZ_64]; + for (int fi = 0; fi < ix->nfiles; fi++) { + const cs_file_t *f = &ix->files[fi]; + for (int ti = 0; ix->units[f->unit].group < 0 && ti < f->ntypes; ti++) { + const cs_type_t *t = &f->types[ti]; + if (t->outer >= 0 || !t->partial || !tree_type_key(key, sizeof(key), f, t) || + cbm_ht_get(partials, key)) { + continue; + } + char *k = cbm_arena_strdup(keys, key); + if (!k) { + return false; + } + cbm_ht_set(partials, k, k); + } + } + return true; +} + +/* The owner of a top-level type declared in `f`: the file's assembly. For a + * file of a shared tree: the shared trees' parts of that name when the + * declaration is partial -- or stands in a tree that has partial ones of the + * name: one tree's declarations of a name are one type --, else the tree. */ +static int decl_owner(const cs_index_t *ix, const cs_file_t *f, const cs_type_t *t, + const CBMHashTable *partials) { + int group = ix->units[f->unit].group; + if (group >= 0) { + return group; + } + char key[CS_NAME_BUF + CBM_SZ_64]; + bool parts = t->partial || (tree_type_key(key, sizeof(key), f, t) && cbm_ht_get(partials, key)); + return parts ? CS_POOL : shared_owner(f->unit); +} + +/* The entity of every declared type. An outer type stands before the types + * it holds, so its entity is known when theirs is asked for. false when + * memory ran out. */ +static bool build_entities(cs_index_t *ix) { + CBMHashTable *partials = cbm_ht_create(CBM_SZ_1K); + CBMArena keys; + cbm_arena_init(&keys); + bool ok = partials && tree_partials(ix, partials, &keys); + for (int fi = 0; ok && fi < ix->nfiles; fi++) { + cs_file_t *f = &ix->files[fi]; + for (int ti = 0; ok && ti < f->ntypes; ti++) { + cs_type_t *t = &f->types[ti]; + int ent = t->outer >= 0 + ? entity_get(ix, CS_NONE, f->types[t->outer].entity, 0, t->name, t->arity, + t->kind) + : entity_get(ix, f->regions[t->region].ns, CS_NONE, + decl_owner(ix, f, t, partials), t->name, t->arity, t->kind); + ok = ent >= 0 && entity_add_decl(ix, &ix->ents[ent], fi, ti) && + ht_mark(ix, ix->type_names, t->name); + t->entity = ent; + } + } + cbm_ht_free(partials); + cbm_arena_destroy(&keys); + ix->oom = ix->oom || !ok; + return ok; +} + +static int int_cmp(const void *a, const void *b) { + int x = *(const int *)a; + int y = *(const int *)b; + return (x > y) - (x < y); +} + +static int full_cmp(const void *a, const void *b) { + const cs_full_t *x = (const cs_full_t *)a; + const cs_full_t *y = (const cs_full_t *)b; + if (x->unit != y->unit) { + return x->unit < y->unit ? -1 : 1; + } + if (x->file != y->file) { + return x->file < y->file ? -1 : 1; + } + return (x->type > y->type) - (x->type < y->type); +} + +/* The complete declarations of every entity that has them in two or more + * projects (or shared trees): the flavours of one type, alternatives of each + * other. A reference assembly's stubs stand behind and are not + * counted. false when memory ran out. */ +static bool entity_fulls(cs_index_t *ix) { + for (int i = 0; i < ix->nents; i++) { + cs_entity_t *e = &ix->ents[i]; + int n = 0; + for (int d = 0; d < e->ndecls; d++) { + const cs_file_t *f = &ix->files[e->decls[d].file]; + n += !f->is_ref && !f->types[e->decls[d].type].partial; + } + if (n < PAIR_LEN) { + continue; + } + cs_full_t *fulls = (cs_full_t *)ix_alloc(ix, (size_t)n * sizeof(cs_full_t)); + if (!fulls) { + return false; + } + int w = 0; + for (int d = 0; d < e->ndecls; d++) { + const cs_file_t *f = &ix->files[e->decls[d].file]; + if (!f->is_ref && !f->types[e->decls[d].type].partial) { + fulls[w++] = (cs_full_t){ + .unit = f->unit, .file = e->decls[d].file, .type = e->decls[d].type}; + } + } + qsort(fulls, (size_t)n, sizeof(cs_full_t), full_cmp); + int units = 0; + int prod = 0; + for (int k = 0; k < n;) { + int unit = fulls[k].unit; + bool product = false; + for (; k < n && fulls[k].unit == unit; k++) { + product = product || !ix->files[fulls[k].file].is_test; + } + units++; + prod += product; + } + if (units >= PAIR_LEN) { + e->fulls = fulls; + e->nfulls = n; + e->full_units = units; + e->full_units_prod = prod; + } + } + return true; +} + +/* The projects and shared trees that declare each entity: sorted, unique. */ +static bool entity_units(cs_index_t *ix) { + for (int i = 0; i < ix->nents; i++) { + cs_entity_t *e = &ix->ents[i]; + e->units = (int *)ix_alloc(ix, (size_t)e->ndecls * sizeof(int)); + if (!e->units) { + return false; + } + for (int d = 0; d < e->ndecls; d++) { + e->units[d] = ix->files[e->decls[d].file].unit; + } + qsort(e->units, (size_t)e->ndecls, sizeof(int), int_cmp); + int w = 0; + for (int d = 1; d < e->ndecls; d++) { + if (e->units[d] != e->units[w]) { + e->units[++w] = e->units[d]; + } + } + e->nunits = e->ndecls > 0 ? w + SKIP_ONE : 0; + } + return true; +} + +static bool ent_in_unit(const cs_entity_t *e, int unit) { + return bsearch(&unit, e->units, (size_t)e->nunits, sizeof(int), int_cmp) != NULL; +} + +/* ── Types by scope and name ─────────────────────────────────────── */ + +/* Order: scope, name, arity, then the owner -- the shared trees' (the + * directories' complete declarations, then the parts: CS_POOL) before the + * assemblies' -- then the entity. */ +static int named_key_cmp(const cs_named_t *e, int scope, const char *name, int arity) { + if (e->scope != scope) { + return e->scope < scope ? -1 : 1; + } + int c = strcmp(e->name, name); + if (c) { + return c; + } + return (e->arity > arity) - (e->arity < arity); +} + +static int named_cmp(const void *a, const void *b) { + const cs_named_t *x = (const cs_named_t *)a; + const cs_named_t *y = (const cs_named_t *)b; + int c = named_key_cmp(x, y->scope, y->name, y->arity); + if (c) { + return c; + } + if (x->owner != y->owner) { + return x->owner < y->owner ? -1 : 1; + } + return (x->ent > y->ent) - (x->ent < y->ent); +} + +/* The entry of `owner` within [lo, hi) (one scope, name and arity: sorted by + * owner, and an owner has one entity there), or CS_NONE. */ +static int named_of_owner(const cs_named_t *arr, int lo, int hi, int owner) { + while (lo < hi) { + int mid = lo + ((hi - lo) / PAIR_LEN); + if (arr[mid].owner < owner) { + lo = mid + SKIP_ONE; + } else if (arr[mid].owner > owner) { + hi = mid; + } else { + return mid; + } + } + return CS_NONE; +} + +/* Where the assemblies' entries start within [lo, hi): before it stand the + * shared trees'. */ +static int named_first_assembly(const cs_named_t *arr, int lo, int hi) { + while (lo < hi) { + int mid = lo + ((hi - lo) / PAIR_LEN); + if (arr[mid].owner < 0) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + return lo; +} + +/* [return, *hi) of arr[0..n): the entries of this scope, name and arity. */ +static int named_range(const cs_named_t *arr, int n, int scope, const char *name, int arity, + int *hi) { + int lo = 0; + int end = n; + while (lo < end) { + int mid = lo + ((end - lo) / PAIR_LEN); + if (named_key_cmp(&arr[mid], scope, name, arity) < 0) { + lo = mid + SKIP_ONE; + } else { + end = mid; + } + } + int a = lo; + end = n; + while (a < end) { + int mid = a + ((end - a) / PAIR_LEN); + if (named_key_cmp(&arr[mid], scope, name, arity) <= 0) { + a = mid + SKIP_ONE; + } else { + end = mid; + } + } + *hi = a; + return lo; +} + +/* The two tables: top-level types by namespace, nested types by their outer + * entity. */ +static bool build_named(cs_index_t *ix) { + for (int i = 0; i < ix->nents; i++) { + ix->ntops += ix->ents[i].parent < 0; + } + ix->nkids = ix->nents - ix->ntops; + ix->tops = (cs_named_t *)ix_alloc(ix, (size_t)ix->ntops * sizeof(cs_named_t)); + ix->kids = (cs_named_t *)ix_alloc(ix, (size_t)ix->nkids * sizeof(cs_named_t)); + if (!ix->tops || !ix->kids) { + return false; + } + int t = 0; + int k = 0; + for (int i = 0; i < ix->nents; i++) { + const cs_entity_t *e = &ix->ents[i]; + cs_named_t row = {.scope = e->parent >= 0 ? e->parent : e->ns, + .name = e->name, + .arity = e->arity, + .owner = e->owner, + .ent = i}; + if (e->parent >= 0) { + ix->kids[k++] = row; + } else { + ix->tops[t++] = row; + } + } + if (ix->ntops > 1) { + qsort(ix->tops, (size_t)ix->ntops, sizeof(cs_named_t), named_cmp); + } + if (ix->nkids > 1) { + qsort(ix->kids, (size_t)ix->nkids, sizeof(cs_named_t), named_cmp); + } + return true; +} + +/* True when an assembly among tops[lo, hi) -- the assemblies' types of the + * shared declaration `shared`'s name -- holds an implementation of the name + * that is no part of that declaration: a complete type, or parts where the + * shared declaration is no partial type. Every contract of the name asks; + * the assemblies are walked for the first, and the answer is kept with the + * shared declaration. */ +static bool rival_implementation(cs_index_t *ix, int lo, int hi, int shared) { + cs_entity_t *s = &ix->ents[shared]; + if (!s->rival_known) { + bool pool = s->owner == CS_POOL; + for (int i = lo; !s->rival && i < hi; i++) { + const cs_entity_t *e = &ix->ents[ix->tops[i].ent]; + s->rival = e->has_impl && !(pool && e->impl_partial); + } + s->rival_known = true; + } + return s->rival; +} + +/* The shared trees' declaration that belongs to an assembly's type: its + * twin. Two rules give a top-level type one: + * - parts: a type an assembly's implementation declares `partial` has the + * shared trees' parts of that name (CS_POOL) as parts of it -- what a + * reference sees of a partial type is its own assembly's parts and the + * shared trees'. A complete type has no further parts; + * - contract: an assembly whose every declaration of the type stands in a + * project directory named `ref` holds only the contract. `ref` is the + * .NET convention for reference-assembly sources, and the rule joins a + * contract to the one implementation it can belong to: the declaration + * the shared trees hold of that full name and arity, when they hold + * exactly one (the parts of one tree count as one) and no assembly + * holds another implementation of it. The stub is then no second type. + * Nothing is chosen between two trees' declarations, two + * implementations or two assemblies: no join there. + * A type nested in a type that has a twin has the one of its name nested in + * the twin. An entity is made after its outer type's, so one pass in order + * sees every outer type first. What the twin says of the type (test code, + * hidden members) is folded into it. */ +static void entity_twins(cs_index_t *ix) { + for (int i = 0; i < ix->nents; i++) { + cs_entity_t *e = &ix->ents[i]; + int hi = 0; + if (e->parent >= 0) { + int outer = ix->ents[e->parent].twin; + if (outer < 0) { + continue; + } + int lo = named_range(ix->kids, ix->nkids, outer, e->name, e->arity, &hi); + e->twin = lo < hi ? ix->kids[lo].ent : CS_NONE; + e->joined = e->twin >= 0 && ix->ents[e->parent].joined; + } else { + if (e->owner < 0 || (e->has_impl && !e->impl_partial)) { + continue; + } + int lo = named_range(ix->tops, ix->ntops, e->ns, e->name, e->arity, &hi); + int split = named_first_assembly(ix->tops, lo, hi); + int at = CS_NONE; + if (e->has_impl) { + at = named_of_owner(ix->tops, lo, split, CS_POOL); + } else if (split - lo == SKIP_ONE && ix->ents[ix->tops[lo].ent].nunits == SKIP_ONE && + !rival_implementation(ix, split, hi, ix->tops[lo].ent)) { + at = lo; + e->joined = true; + } + e->twin = at >= 0 ? ix->tops[at].ent : CS_NONE; + } + if (e->twin >= 0) { + cs_entity_t *parts = &ix->ents[e->twin]; + e->incomplete = e->incomplete || parts->incomplete; + e->any_prod = e->any_prod || parts->any_prod; + e->all_test = e->all_test && parts->all_test; + parts->used = true; + if (!e->has_impl) { + parts->stub = parts->stub == CS_NONE ? i : CS_AMBIGUOUS; + } + } + } +} + +/* ── Node binding ────────────────────────────────────────────────── */ + +static bool label_is_callable(const char *l) { + return l && (strcmp(l, "Method") == 0 || strcmp(l, "Function") == 0); +} + +static bool label_is_value(const char *l) { + return l && (strcmp(l, "Field") == 0 || strcmp(l, "Variable") == 0 || + strcmp(l, "Property") == 0 || strcmp(l, "Constant") == 0); +} + +/* .. of type `t` into `buf`, its length into *len. + * false when it does not fit: such a declaration has no node to be found. */ +static bool type_qn(const cs_file_t *f, int t, char *buf, size_t cap, size_t *len) { + size_t ml = strlen(f->module_qn); + size_t total = ml; + for (int x = t; x >= 0; x = f->types[x].outer) { + total += strlen(f->types[x].name) + SKIP_ONE; + } + if (total >= cap) { + return false; + } + size_t w = total; + buf[w] = '\0'; + for (int x = t; x >= 0; x = f->types[x].outer) { + size_t nl = strlen(f->types[x].name); + w -= nl; + memcpy(buf + w, f->types[x].name, nl); + buf[--w] = '.'; + } + memcpy(buf, f->module_qn, ml); + *len = total; + return true; +} + +/* The node a member's own qualified name has, when it is one a reference to + * a member of this kind can bind. */ +static const cbm_gbuf_node_t *member_node_at(const cbm_gbuf_t *g, const cs_member_t *m, char *qn, + size_t type_len, size_t cap) { + size_t nl = strlen(m->name); + if (m->kind == 'o' || m->kind == 'x' || type_len + nl + PAIR_LEN > cap) { + return NULL; /* operators and indexers have no node */ + } + qn[type_len] = '.'; + memcpy(qn + type_len + SKIP_ONE, m->name, nl + SKIP_ONE); + const cbm_gbuf_node_t *n = cbm_gbuf_find_by_qn(g, qn); + qn[type_len] = '\0'; + if (!n) { + return NULL; + } + return (m->kind == 'c' ? label_is_callable(n->label) : label_is_value(n->label)) ? n : NULL; +} + +/* Scratch tables of the node pass: emptied for every file. */ +typedef struct { + CBMHashTable *names; /* "\x1f" -> index + 1 */ + uint32_t sized; /* what the table is sized for: the most keys it held, or + * its initial capacity -- what emptying it walks */ + CBMArena keys; + int *last; /* per gid: the last type that has it */ + int cap_last; +} cs_node_pass_t; + +enum { CS_SCRATCH_MIN = 64, CS_SCRATCH_SLACK = 4 }; + +/* Empty the scratch table for a file that puts about `need` keys into it. + * Emptying a table walks every bucket it has, and a table never shrinks: one + * that a far larger file grew is made anew, so that one large file does not + * make every later file pay for its size. false when memory ran out. */ +static bool node_pass_reset(cs_index_t *ix, cs_node_pass_t *np, int need) { + uint32_t want = need > CS_SCRATCH_MIN ? (uint32_t)need : CS_SCRATCH_MIN; + if (np->sized > want * CS_SCRATCH_SLACK) { + cbm_ht_free(np->names); + np->names = cbm_ht_create(want); + np->sized = want; + if (!np->names) { + ix->oom = true; + return false; + } + } else { + cs_scratch_work(np->sized); + cbm_ht_clear(np->names); + } + cbm_arena_rewind(&np->keys); /* the emptied table held the only pointers into it */ + return true; +} + +/* Note how many keys the scratch table holds now. */ +static void node_pass_held(cs_node_pass_t *np) { + uint32_t held = cbm_ht_count(np->names); + np->sized = held > np->sized ? held : np->sized; +} + +/* The path group of every type of the file (types with one path -- `Foo` and + * `Foo`, and what is nested in them under one name -- share the node + * .), and which declaration owns that node: the last one. */ +static bool assign_gids(cs_index_t *ix, cs_file_t *f, cs_node_pass_t *np) { + if (f->ntypes > np->cap_last) { + cbm_free(CBM_MEM_CLASS_OTHER, np->last); + np->last = (int *)cbm_alloc(CBM_MEM_CLASS_OTHER, (size_t)f->ntypes * sizeof(int)); + np->cap_last = np->last ? f->ntypes : 0; + if (!np->last) { + ix->oom = true; + return false; + } + } + if (!node_pass_reset(ix, np, f->ntypes)) { + return false; + } + int gids = 0; + for (int t = 0; t < f->ntypes; t++) { + cs_type_t *ty = &f->types[t]; + char key[CS_NAME_BUF + CBM_SZ_16]; + snprintf(key, sizeof(key), "%d\x1f%s", ty->outer >= 0 ? f->types[ty->outer].gid : CS_NONE, + ty->name); + intptr_t v = (intptr_t)cbm_ht_get(np->names, key); + if (v > 0) { + ty->gid = (int)(v - SKIP_ONE); + } else { + char *k = cbm_arena_strdup(&np->keys, key); + if (!k) { + ix->oom = true; + return false; + } + ty->gid = gids++; + cbm_ht_set(np->names, k, (void *)(intptr_t)(ty->gid + SKIP_ONE)); + } + np->last[ty->gid] = t; + } + for (int t = 0; t < f->ntypes; t++) { + f->types[t].owns_node = np->last[f->types[t].gid] == t; + } + return true; +} + +/* The graph node of every type and member of the file that has one a + * reference may bind. A member's node .. belongs to the + * last declaration of that qualified name in the file: overloads of one type + * share it, but when `Foo` and `Foo` both declare the member, the earlier + * type's member has none of its own (the same rule as for the types). No + * node either where the slot holds a member of the other kind. */ +static bool bind_nodes(cs_index_t *ix, cs_file_t *f, const cbm_gbuf_t *g, cs_node_pass_t *np) { + if (!assign_gids(ix, f, np)) { + return false; + } + char qn[CS_KEY_BUF]; + size_t len = 0; + for (int t = 0; t < f->ntypes; t++) { + cs_type_t *ty = &f->types[t]; + if (ty->owns_node && ty->kind != 'd' && type_qn(f, t, qn, sizeof(qn), &len)) { + const cbm_gbuf_node_t *n = cbm_gbuf_find_by_qn(g, qn); + ty->node = (n && cbm_label_is_type_like(n->label)) ? n : NULL; + } + } + /* which member is the last of its (path, name) */ + node_pass_held(np); + if (!node_pass_reset(ix, np, f->nmembers)) { + return false; + } + for (int m = 0; m < f->nmembers; m++) { + char key[(CS_NAME_BUF) + CBM_SZ_16]; + snprintf(key, sizeof(key), "%d\x1f%s", f->types[f->members[m].type].gid, + f->members[m].name); + const char *k = cbm_ht_get_key(np->names, key); + if (!k) { + k = cbm_arena_strdup(&np->keys, key); + if (!k) { + ix->oom = true; + return false; + } + } + cbm_ht_set(np->names, k, (void *)(intptr_t)(m + SKIP_ONE)); + } + node_pass_held(np); + int cached = CS_NONE; + bool fits = false; + for (int m = 0; m < f->nmembers; m++) { + cs_member_t *mem = &f->members[m]; + char key[(CS_NAME_BUF) + CBM_SZ_16]; + snprintf(key, sizeof(key), "%d\x1f%s", f->types[mem->type].gid, mem->name); + intptr_t last = (intptr_t)cbm_ht_get(np->names, key); + if (last <= 0 || + f->types[f->members[last - SKIP_ONE].type].entity != f->types[mem->type].entity) { + continue; /* a declaration of another type took the name's node */ + } + if (mem->type != cached) { + cached = mem->type; + fits = type_qn(f, cached, qn, sizeof(qn), &len); + } + mem->node = fits ? member_node_at(g, mem, qn, len, sizeof(qn)) : NULL; + } + return true; +} + +static unsigned char decl_class(const cs_file_t *f, bool has_node) { + return (unsigned char)((has_node ? 0 : CS_CLS_NO_NODE) | (f->is_ref ? CS_CLS_REF : 0) | + (f->is_test ? CS_CLS_TEST : 0)); +} + +static int bind_cmp(const void *a, const void *b) { + const cs_bind_t *x = (const cs_bind_t *)a; + const cs_bind_t *y = (const cs_bind_t *)b; + if (x->cls != y->cls) { + return x->cls < y->cls ? -1 : 1; + } + if (x->file != y->file) { + return x->file < y->file ? -1 : 1; + } + return (x->type > y->type) - (x->type < y->type); +} + +/* Every entity's declarations by (class, path). */ +static bool build_binds(cs_index_t *ix) { + for (int i = 0; i < ix->nents; i++) { + cs_entity_t *e = &ix->ents[i]; + e->binds = (cs_bind_t *)ix_alloc(ix, (size_t)e->ndecls * sizeof(cs_bind_t)); + if (!e->binds) { + return false; + } + for (int d = 0; d < e->ndecls; d++) { + const cs_file_t *f = &ix->files[e->decls[d].file]; + e->binds[d] = + (cs_bind_t){.file = e->decls[d].file, + .type = e->decls[d].type, + .cls = decl_class(f, f->types[e->decls[d].type].node != NULL)}; + } + if (e->ndecls > 1) { + qsort(e->binds, (size_t)e->ndecls, sizeof(cs_bind_t), bind_cmp); + } + } + return true; +} + +/* ── Members by entity and name ──────────────────────────────────── */ + +/* One member for sorting: the sort keys carry their own comparison data (no + * global sort context). */ +typedef struct { + cs_mref_t ref; + const char *name; + const char *sig; +} cs_mkey_t; + +static int mkey_cmp(const void *a, const void *b) { + const cs_mkey_t *x = (const cs_mkey_t *)a; + const cs_mkey_t *y = (const cs_mkey_t *)b; + if (x->ref.ent != y->ref.ent) { + return x->ref.ent < y->ref.ent ? -1 : 1; + } + int c = strcmp(x->name, y->name); + if (c) { + return c; + } + if (x->ref.group != y->ref.group) { + return x->ref.group < y->ref.group ? -1 : 1; + } + c = strcmp(x->sig, y->sig); + if (c) { + return c; + } + if (x->ref.cls != y->ref.cls) { + return x->ref.cls < y->ref.cls ? -1 : 1; + } + if (x->ref.file != y->ref.file) { + return x->ref.file < y->ref.file ? -1 : 1; + } + return (x->ref.midx > y->ref.midx) - (x->ref.midx < y->ref.midx); +} + +/* The key of the operators and indexers named `name` that `ent` declares: + * those outside test code (`prod`), or all of them. false when it does not + * fit (cannot be for a declared one: parse_member bounds the name). */ +static bool special_key(char *key, size_t cap, bool prod, int ent, const char *name) { + int n = snprintf(key, cap, "%c%d\x1f%s", prod ? 'P' : 'A', ent, name); + return n > 0 && (size_t)n < cap; +} + +/* Note an operator or indexer of `ent`. false when memory ran out. */ +static bool special_mark(cs_index_t *ix, int ent, const char *name, unsigned char cls) { + char key[CS_NAME_BUF + CBM_SZ_64]; + if (!special_key(key, sizeof(key), false, ent, name)) { + return true; + } + if (!ht_mark(ix, ix->specials, key)) { + return false; + } + key[0] = 'P'; + return (cls & CS_CLS_TEST) != 0 || ht_mark(ix, ix->specials, key); +} + +/* Every member a name can address, by (entity, name, group, signature, + * class, path, order). Explicit interface implementations are left out: no + * name addresses one. */ +static bool build_mrefs(cs_index_t *ix) { + size_t total = 0; + for (int fi = 0; fi < ix->nfiles; fi++) { + total += (size_t)ix->files[fi].nmembers; + } + cs_mkey_t *keys = + (cs_mkey_t *)cbm_alloc(CBM_MEM_CLASS_OTHER, (total ? total : SKIP_ONE) * sizeof(cs_mkey_t)); + ix->mrefs = (cs_mref_t *)ix_alloc(ix, total * sizeof(cs_mref_t)); + if (!keys || !ix->mrefs) { + cbm_free(CBM_MEM_CLASS_OTHER, keys); + ix->oom = true; + return false; + } + size_t n = 0; + for (int fi = 0; fi < ix->nfiles; fi++) { + const cs_file_t *f = &ix->files[fi]; + for (int m = 0; m < f->nmembers; m++) { + const cs_member_t *mem = &f->members[m]; + if (mem->explicit_impl) { + continue; + } + unsigned char group = mem->kind == 'c' + ? SKIP_ONE + : ((mem->kind == 'o' || mem->kind == 'x') ? PAIR_LEN : 0); + unsigned char cls = decl_class(f, mem->node != NULL); + int ent = f->types[mem->type].entity; + if (group == PAIR_LEN && !special_mark(ix, ent, mem->name, cls)) { + cbm_free(CBM_MEM_CLASS_OTHER, keys); + return false; + } + keys[n++] = + (cs_mkey_t){.ref = {.ent = ent, .file = fi, .midx = m, .group = group, .cls = cls}, + .name = mem->name, + .sig = mem->sig ? mem->sig : ""}; + } + } + if (n > 1) { + qsort(keys, n, sizeof(cs_mkey_t), mkey_cmp); + } + for (size_t i = 0; i < n; i++) { + ix->mrefs[i] = keys[i].ref; + } + ix->nmrefs = (int)n; + cbm_free(CBM_MEM_CLASS_OTHER, keys); + return true; +} + +static const cs_member_t *mref_member(const cs_index_t *ix, const cs_mref_t *r) { + return &ix->files[r->file].members[r->midx]; +} + +/* Order of a member entry against a key (entity, name, and when `group` is + * not negative: group and signature). */ +static int mref_key_cmp(const cs_index_t *ix, const cs_mref_t *r, int ent, const char *name, + int group, const char *sig) { + if (r->ent != ent) { + return r->ent < ent ? -1 : 1; + } + const cs_member_t *m = mref_member(ix, r); + int c = strcmp(m->name, name); + if (c || group < 0) { + return c; + } + if (r->group != group) { + return (int)r->group < group ? -1 : 1; + } + return strcmp(m->sig ? m->sig : "", sig); +} + +/* [return, *hi): the entity's members named `name`; with `group` >= 0 only + * those of that group with exactly the signature `sig`. */ +static int mref_range(const cs_index_t *ix, int ent, const char *name, int group, const char *sig, + int *hi) { + int lo = 0; + int end = ix->nmrefs; + while (lo < end) { + int mid = lo + ((end - lo) / PAIR_LEN); + if (mref_key_cmp(ix, &ix->mrefs[mid], ent, name, group, sig) < 0) { + lo = mid + SKIP_ONE; + } else { + end = mid; + } + } + int a = lo; + end = ix->nmrefs; + while (a < end) { + int mid = a + ((end - a) / PAIR_LEN); + if (mref_key_cmp(ix, &ix->mrefs[mid], ent, name, group, sig) <= 0) { + a = mid + SKIP_ONE; + } else { + end = mid; + } + } + *hi = a; + return lo; +} + +/* ── What an assembly's own parts add to a shared declaration ────── */ + +static int extra_key_cmp(const cs_extra_t *x, int twin, const char *name, int arity) { + if (x->twin != twin) { + return x->twin < twin ? -1 : 1; + } + int c = strcmp(x->name, name); + if (c) { + return c; + } + return (x->arity > arity) - (x->arity < arity); +} + +static int extra_cmp(const void *a, const void *b) { + const cs_extra_t *x = (const cs_extra_t *)a; + const cs_extra_t *y = (const cs_extra_t *)b; + int c = extra_key_cmp(x, y->twin, y->name, y->arity); + return c ? c : (x->ent > y->ent) - (x->ent < y->ent); +} + +/* The shared trees' declaration beside which the type `ent` stands: a type + * nested in an assembly's own parts of a type, implemented there, that the + * shared trees' declaration of that type does not have. CS_NONE for every + * other type. */ +static int extra_type_twin(const cs_index_t *ix, int ent) { + const cs_entity_t *e = &ix->ents[ent]; + if (e->parent < 0 || !e->has_impl || e->twin >= 0) { + return CS_NONE; + } + return ix->ents[e->parent].twin; +} + +/* The same for a member: one an assembly's own parts of a type declare. A + * stub does not count: it declares what the implementation declares. */ +static int extra_member_twin(const cs_index_t *ix, const cs_mref_t *r) { + return (r->cls & CS_CLS_REF) ? CS_NONE : ix->ents[r->ent].twin; +} + +/* The names the assemblies' own parts declare beyond the shared trees' + * declarations they belong to, by that declaration and name: what a + * reference that sees the shared declaration without those parts cannot tell + * from what is not there (unseen_part_declares). One walk over the types and + * the members, so that no lookup walks the assemblies. false when memory ran + * out. */ +static bool build_extras(cs_index_t *ix) { + size_t n = 0; + for (int i = 0; i < ix->nents; i++) { + n += extra_type_twin(ix, i) >= 0; + } + for (int i = 0; i < ix->nmrefs; i++) { + n += extra_member_twin(ix, &ix->mrefs[i]) >= 0; + } + cs_extra_t *arr = (cs_extra_t *)ix_alloc(ix, n * sizeof(cs_extra_t)); + if (!arr) { + return false; + } + size_t w = 0; + for (int i = 0; i < ix->nents; i++) { + int twin = extra_type_twin(ix, i); + if (twin >= 0) { + arr[w++] = (cs_extra_t){ + .twin = twin, .name = ix->ents[i].name, .arity = ix->ents[i].arity, .ent = i}; + } + } + for (int i = 0; i < ix->nmrefs; i++) { + const cs_mref_t *r = &ix->mrefs[i]; + int twin = extra_member_twin(ix, r); + if (twin >= 0) { + arr[w++] = (cs_extra_t){.twin = twin, + .name = mref_member(ix, r)->name, + .arity = CS_NONE, + .ent = CS_NONE, + .prod = !(r->cls & CS_CLS_TEST)}; + } + } + if (n > 1) { + qsort(arr, n, sizeof(cs_extra_t), extra_cmp); + } + /* a name's members are one entry */ + size_t out = 0; + for (size_t i = 0; i < n; i++) { + if (out > 0 && arr[i].ent < 0 && arr[out - SKIP_ONE].ent < 0 && + extra_key_cmp(&arr[out - SKIP_ONE], arr[i].twin, arr[i].name, arr[i].arity) == 0) { + arr[out - SKIP_ONE].prod = arr[out - SKIP_ONE].prod || arr[i].prod; + } else { + arr[out++] = arr[i]; + } + } + ix->extras = arr; + ix->nextras = (int)out; + return true; +} + +/* [return, *hi): the extras of this shared declaration, name and arity + * (CS_NONE: the name's members). */ +static int extra_range(const cs_index_t *ix, int twin, const char *name, int arity, int *hi) { + int lo = 0; + int end = ix->nextras; + while (lo < end) { + int mid = lo + ((end - lo) / PAIR_LEN); + if (extra_key_cmp(&ix->extras[mid], twin, name, arity) < 0) { + lo = mid + SKIP_ONE; + } else { + end = mid; + } + } + int a = lo; + end = ix->nextras; + while (a < end) { + int mid = a + ((end - a) / PAIR_LEN); + if (extra_key_cmp(&ix->extras[mid], twin, name, arity) <= 0) { + a = mid + SKIP_ONE; + } else { + end = mid; + } + } + *hi = a; + return lo; +} + +/* ── Import candidate indexes ───────────────────────────────────── */ + +static int alias_id_cmp(const void *a, const void *b) { + const cs_alias_id_t *x = (const cs_alias_id_t *)a; + const cs_alias_id_t *y = (const cs_alias_id_t *)b; + cs_index_work(SKIP_ONE); + int order = strcmp(x->name, y->name); + return order ? order : (x->id > y->id) - (x->id < y->id); +} + +static int using_id_cmp(const void *a, const void *b) { + const cs_using_id_t *x = (const cs_using_id_t *)a; + const cs_using_id_t *y = (const cs_using_id_t *)b; + cs_index_work(SKIP_ONE); + if (x->scope != y->scope) { + return (x->scope > y->scope) - (x->scope < y->scope); + } + return (x->id > y->id) - (x->id < y->id); +} + +static int name_scope_cmp(const void *a, const void *b) { + const cs_name_scope_t *x = (const cs_name_scope_t *)a; + const cs_name_scope_t *y = (const cs_name_scope_t *)b; + cs_index_work(SKIP_ONE); + int order = strcmp(x->name, y->name); + if (order) { + return order; + } + if (x->top != y->top) { + return x->top ? 1 : -1; + } + return (x->scope > y->scope) - (x->scope < y->scope); +} + +static void *import_array(cs_index_t *ix, size_t n, size_t size) { + if (n > SIZE_MAX / size) { + ix->oom = true; + return NULL; + } + return n ? ix_alloc(ix, n * size) : NULL; +} + +/* One repository-wide pass, never one member-table expansion per import. + * Include test-only names and extras: the ordinary resolver, not this + * candidate filter, decides visibility, arity and unseen-part ambiguity. */ +static bool build_name_scopes(cs_index_t *ix) { + size_t n = 0; + const int counts[] = {ix->ntops, ix->nkids, ix->nmrefs, ix->nextras}; + for (size_t i = 0; i < sizeof(counts) / sizeof(counts[0]); i++) { + if ((size_t)counts[i] > SIZE_MAX - n) { + ix->oom = true; + return false; + } + n += (size_t)counts[i]; + } + cs_name_scope_t *rows = (cs_name_scope_t *)import_array(ix, n, sizeof(*rows)); + if (n && !rows) { + return false; + } + size_t w = 0; + for (int i = 0; i < ix->ntops; i++) { + cs_index_work(SKIP_ONE); + rows[w++] = + (cs_name_scope_t){.name = ix->tops[i].name, .scope = ix->tops[i].scope, .top = true}; + } + for (int i = 0; i < ix->nkids; i++) { + cs_index_work(SKIP_ONE); + rows[w++] = (cs_name_scope_t){.name = ix->kids[i].name, .scope = ix->kids[i].scope}; + } + for (int i = 0; i < ix->nmrefs; i++) { + cs_index_work(SKIP_ONE); + rows[w++] = (cs_name_scope_t){.name = mref_member(ix, &ix->mrefs[i])->name, + .scope = ix->mrefs[i].ent}; + } + for (int i = 0; i < ix->nextras; i++) { + cs_index_work(SKIP_ONE); + rows[w++] = (cs_name_scope_t){.name = ix->extras[i].name, .scope = ix->extras[i].twin}; + } + if (w > 1) { + qsort(rows, w, sizeof(*rows), name_scope_cmp); + } + size_t out = 0; + for (size_t i = 0; i < w; i++) { + if (!out || name_scope_cmp(&rows[out - SKIP_ONE], &rows[i])) { + rows[out++] = rows[i]; + } + } + ix->name_scopes = rows; + ix->nname_scopes = out; + ix->name_scopes_ready = true; + return true; +} + +/* Build only after these directives have resolved. A static import is + * associated with its own entity and its twin, without copying either + * entity's declarations into every region that imports it. */ +static bool build_using_index(cs_index_t *ix, const cs_using_t *us, int n, + cs_using_index_t *index) { + cs_using_index_t out = {0}; + for (int i = 0; i < n; i++) { + cs_index_work(SKIP_ONE); + const cs_using_t *u = &us[i]; + if (u->kind == 'a') { + out.naliases++; + } else if (u->kind == 'n' && u->ns >= 0) { + out.nnamespaces++; + } else if (u->kind == 's' && u->ent >= 0) { + size_t extra = ix->ents[u->ent].twin >= 0 ? PAIR_LEN : SKIP_ONE; + if (extra > SIZE_MAX - out.nentities) { + ix->oom = true; + return false; + } + out.nentities += extra; + } + } + out.aliases = (cs_alias_id_t *)import_array(ix, out.naliases, sizeof(*out.aliases)); + out.namespaces = (cs_using_id_t *)import_array(ix, out.nnamespaces, sizeof(*out.namespaces)); + out.entities = (cs_using_id_t *)import_array(ix, out.nentities, sizeof(*out.entities)); + if (ix->oom) { + return false; + } + size_t a = 0; + size_t ns = 0; + size_t e = 0; + for (int i = 0; i < n; i++) { + cs_index_work(SKIP_ONE); + const cs_using_t *u = &us[i]; + if (u->kind == 'a') { + out.aliases[a++] = (cs_alias_id_t){.name = u->alias, .id = i}; + } else if (u->kind == 'n' && u->ns >= 0) { + out.namespaces[ns++] = (cs_using_id_t){.scope = u->ns, .id = i}; + } else if (u->kind == 's' && u->ent >= 0) { + out.entities[e++] = (cs_using_id_t){.scope = u->ent, .id = i}; + int twin = ix->ents[u->ent].twin; + if (twin >= 0) { + out.entities[e++] = (cs_using_id_t){.scope = twin, .id = i}; + } + } + } + if (a > 1) { + qsort(out.aliases, a, sizeof(*out.aliases), alias_id_cmp); + } + if (ns > 1) { + qsort(out.namespaces, ns, sizeof(*out.namespaces), using_id_cmp); + } + if (e > 1) { + qsort(out.entities, e, sizeof(*out.entities), using_id_cmp); + } + out.ready = true; + *index = out; + return true; +} + +/* ── Choosing the declaration a reference binds ──────────────────── */ + +/* Number of leading directories two paths share. */ +static int common_dir_prefix(const char *a, const char *b) { + int n = 0; + for (;;) { + const char *sa = strchr(a, '/'); + const char *sb = strchr(b, '/'); + if (!sa || !sb) { + return n; + } + size_t la = (size_t)(sa - a); + if (la != (size_t)(sb - b) || memcmp(a, b, la) != 0) { + return n; + } + n++; + a = sa + SKIP_ONE; + b = sb + SKIP_ONE; + } +} + +/* Length of the first `dirs` directories of `path`, each with its '/'. */ +static size_t dir_prefix_len(const char *path, int dirs) { + const char *p = path; + for (int i = 0; i < dirs; i++) { + const char *slash = strchr(p, '/'); + if (!slash) { + break; + } + p = slash + SKIP_ONE; + } + return (size_t)(p - path); +} + +/* Of the declarations [lo, hi) -- one class, so in path order -- the one a + * reference written in file `src` binds (a choice of presentation among the + * parts of one type, made without a scan): the one in `src` itself; else the + * first one in the directory that shares the longest leading path with it; + * else the first. */ +static int nearest_bind(const cs_index_t *ix, const cs_bind_t *arr, int lo, int hi, int src) { + int a = lo; + int b = hi; + while (a < b) { + int mid = a + ((b - a) / PAIR_LEN); + if (arr[mid].file < src) { + a = mid + SKIP_ONE; + } else { + b = mid; + } + } + if (a < hi && arr[a].file == src) { + return a; + } + const char *sp = ix->files[src].rel_path; + int left = a > lo ? common_dir_prefix(ix->files[arr[a - SKIP_ONE].file].rel_path, sp) : 0; + int right = a < hi ? common_dir_prefix(ix->files[arr[a].file].rel_path, sp) : 0; + int best = left > right ? left : right; + if (best == 0) { + return lo; + } + /* the entries before `a` that share those directories are the last ones + * before it: the first of them */ + size_t plen = dir_prefix_len(sp, best); + int x = lo; + int y = a; + while (x < y) { + int mid = x + ((y - x) / PAIR_LEN); + if (strncmp(ix->files[arr[mid].file].rel_path, sp, plen) < 0) { + x = mid + SKIP_ONE; + } else { + y = mid; + } + } + return x; +} + +/* [return, *hi): the declarations of class `cls` among arr[0, n), which is + * sorted by class first. */ +static int class_range(const cs_bind_t *arr, int n, unsigned char cls, int *out_hi) { + int a = 0; + int b = n; + while (a < b) { + int mid = a + ((b - a) / PAIR_LEN); + if (arr[mid].cls < cls) { + a = mid + SKIP_ONE; + } else { + b = mid; + } + } + int first = a; + b = n; + while (a < b) { + int mid = a + ((b - a) / PAIR_LEN); + if (arr[mid].cls <= cls) { + a = mid + SKIP_ONE; + } else { + b = mid; + } + } + *out_hi = a; + return first; +} + +/* True when a declaration in file `fa` is the better one to bind than one in + * `fb`, for a reference written in `src`: its own file, then the longer + * shared path, then the path order. */ +static bool file_nearer(const cs_index_t *ix, int fa, int fb, int src) { + if ((fa == src) != (fb == src)) { + return fa == src; + } + const char *sp = ix->files[src].rel_path; + int pa = common_dir_prefix(ix->files[fa].rel_path, sp); + int pb = common_dir_prefix(ix->files[fb].rel_path, sp); + return pa != pb ? pa > pb : fa < fb; +} + +/* ── Resolution context ──────────────────────────────────────────── */ + +enum { CS_OK = 0, CS_UNRES, CS_LOCAL }; + +typedef struct { + int st; + const cbm_gbuf_node_t *node; + bool exact; + int reason; + cs_why_t why; /* of an ambiguous one */ +} cs_res_t; + +typedef struct cs_memo cs_memo_t; + +typedef struct { + const cs_index_t *ix; + int file; + const cs_file_t *f; + int type; /* the innermost type declaration around the definition, or CS_NONE */ + int member; /* the documented member, when it has type parameters; CS_NONE */ + int region; /* the namespace declaration around it */ + int unit; /* the file's project, or its directory in a shared tree */ + int group; /* its assembly; CS_NONE in a shared tree */ + bool prod; /* product code: test-only declarations are not bound */ + bool glob; /* the reference starts with global:: */ + /* The namespace declaration whose own aliases and usings are not asked + * (CS_NONE: none): a using directive's target is resolved without the + * directives beside it. */ + int skip_region; + /* Set when a namespace this context does not see was passed over (NULL: + * nobody asks). */ + bool *passed_over; + /* What the using directives of a scope level gave a query, remembered + * for the pass the caller runs (lookup); NULL: nothing is remembered. */ + cs_memo_t *memo; +} cs_ctx_t; + +static cs_res_t res_edge(const cbm_gbuf_node_t *n, bool exact) { + return (cs_res_t){.st = CS_OK, .node = n, .exact = exact}; +} + +static cs_res_t res_unres(int reason) { + return (cs_res_t){.st = CS_UNRES, .reason = reason}; +} + +static cs_res_t res_ambiguous(cs_why_t why) { + return (cs_res_t){.st = CS_UNRES, .reason = CBM_DOCLINK_REASON_AMBIGUOUS, .why = why}; +} + +static bool res_is(const cs_res_t *r, int reason) { + return r->st == CS_UNRES && r->reason == reason; +} + +/* What choosing among declarations came to. */ +typedef enum { CS_PICK_NODE = 0, CS_PICK_GAP, CS_PICK_AMBIGUOUS } cs_pick_t; + +/* An entity and the shared trees' declaration that belongs to it: what a + * reference sees of one type is in at most these two. */ +typedef struct { + int ent[PAIR_LEN]; + int n; +} cs_view_t; + +static cs_view_t view_of(const cs_index_t *ix, int ent) { + int twin = ix->ents[ent].twin; + return (cs_view_t){.ent = {ent, twin}, .n = twin >= 0 ? PAIR_LEN : SKIP_ONE}; +} + +/* The node of the nearest of `e`'s own declarations of one class -- an + * implementation's, or (`stub`) a reference assembly's stub's -- that has + * one; never a test declaration's for product code. NULL when none has. */ +static const cbm_gbuf_node_t *class_node(const cs_ctx_t *c, const cs_entity_t *e, bool stub) { + const cs_index_t *ix = c->ix; + const cs_bind_t *best = NULL; + for (int test = 0; test <= (c->prod ? 0 : SKIP_ONE); test++) { + int b = 0; + int a = + class_range(e->binds, e->ndecls, + (unsigned char)((stub ? CS_CLS_REF : 0) | (test ? CS_CLS_TEST : 0)), &b); + if (a >= b) { + continue; + } + const cs_bind_t *at = &e->binds[nearest_bind(ix, e->binds, a, b, c->file)]; + if (!best || file_nearer(ix, at->file, best->file, c->file)) { + best = at; + } + } + return best ? ix->files[best->file].types[best->type].node : NULL; +} + +/* True when one of `e`'s own declarations is in view, with a node or + * without: for product code one that is not test code. */ +static bool declared_in_view(const cs_ctx_t *c, const cs_entity_t *e) { + for (int cls = 0; cls < CS_CLS_COUNT; cls++) { + int b = 0; + if (!(c->prod && (cls & CS_CLS_TEST)) && + class_range(e->binds, e->ndecls, (unsigned char)cls, &b) < b) { + return true; + } + } + return false; +} + +/* ── Visibility and candidates ───────────────────────────────────── */ + +static bool ent_visible(const cs_ctx_t *c, int ent) { + const cs_index_t *ix = c->ix; + const cs_entity_t *e = &ix->ents[ent]; + if (c->prod) { + return e->any_prod; + } + /* test code: a global-namespace test type (and what it holds) is local + * to its own program */ + int top = ent; + while (ix->ents[top].parent >= 0) { + top = ix->ents[top].parent; + } + if (ix->ents[top].all_test && ix->ents[top].ns == 0) { + return ent_in_unit(&ix->ents[top], c->unit); + } + return true; +} + +/* What a name was found to be at one scope level. */ +typedef struct { + char kind; /* T type, M member(s) of entity `id`, N namespace, L type parameter, X outside */ + int id; +} cs_cand_t; + +typedef struct { + cs_cand_t first; + int n; /* distinct candidates of the deciding level; > 1 is ambiguous */ + bool invisible; /* a level had only what this code may not bind (test code) */ + bool exact; /* decided by an alias */ + bool in_namespace; /* decided by an enclosing namespace's own types and namespaces */ + bool joined; /* one type only because a contract is joined to its implementation */ + cs_why_t why; /* what made several of them, when it was not the scope's rules */ +} cs_found_t; + +/* The result for a name with several candidates. */ +static cs_res_t found_ambiguous(const cs_found_t *fd) { + return res_ambiguous(fd->why); +} + +static void found_add(cs_found_t *fd, char kind, int id) { + if (fd->n > 0 && fd->first.kind == kind && fd->first.id == id) { + return; + } + if (fd->n == 0) { + fd->first = (cs_cand_t){.kind = kind, .id = id}; + } + fd->n++; +} + +static void found_add_type(const cs_ctx_t *c, cs_found_t *fd, int ent) { + if (ent_visible(c, ent)) { + found_add(fd, 'T', ent); + } else { + fd->invisible = true; + } +} + +/* The namespace `seg` under `parent` as this context sees it. For product + * code a namespace that only test code declares does not exist: it is of no + * program product code is compiled with, so its name neither stands in the + * way of what the scope has further out nor makes a name under it the + * repository's. CS_NONE when none is in view. That one was passed over is + * noted in the context (what the reference names may then be test code's: + * resolve_ref asks). */ +static int ns_in_view(const cs_ctx_t *c, int parent, const char *seg, size_t len) { + int child = ns_find(c->ix, parent, seg, len); + if (child >= 0 && c->prod && !c->ix->nss[child].prod) { + if (c->passed_over) { + *c->passed_over = true; + } + return CS_NONE; + } + return child; +} + +/* Every type among tops[lo, hi) -- other owners' types of one name -- is a + * candidate: one binds, several are ambiguous (`why`). Nothing chooses + * between two of them. */ +static void add_owners(const cs_ctx_t *c, int lo, int hi, cs_why_t why, cs_found_t *fd) { + if (hi - lo > CS_MAX_FOREIGN) { + fd->n += PAIR_LEN; /* more of them than one lookup compares: ambiguous */ + fd->why = CS_WHY_LIMIT; + return; + } + int before = fd->n; + for (int i = lo; i < hi; i++) { + found_add_type(c, fd, c->ix->tops[i].ent); + } + if (fd->n - before > SKIP_ONE) { + fd->why = why; + } +} + +/* The entry among tops[lo, split) -- the shared trees' declarations of one + * name -- that is the context's own: what its own tree declares (a complete + * type, or a part of the shared trees' partial one). CS_NONE for a file of a + * project, and when the tree declares none. */ +static int own_shared(const cs_ctx_t *c, int lo, int split) { + const cs_index_t *ix = c->ix; + if (c->group >= 0 || c->unit < 0 || lo >= split) { + return CS_NONE; + } + int mine = named_of_owner(ix->tops, lo, split, shared_owner(c->unit)); + int last = split - SKIP_ONE; /* the parts stand last among the shared trees' */ + if (mine < 0 && ix->tops[last].owner == CS_POOL && + ent_in_unit(&ix->ents[ix->tops[last].ent], c->unit)) { + mine = last; + } + return mine; +} + +/* The top-level types of one namespace, name and arity a reference from this + * context can mean: its own assembly's type (for a file of a shared tree: + * what its own tree declares). Else every shared tree's and every other + * assembly's type of that name alike: one binds, several are ambiguous. An + * assembly's type that has the shared trees' declaration as its twin is that + * declaration seen from the assembly, and no second type. */ +static void add_top_types(const cs_ctx_t *c, int ns, const char *name, int arity, cs_found_t *fd) { + const cs_named_t *arr = c->ix->tops; + int hi = 0; + int lo = named_range(arr, c->ix->ntops, ns, name, arity, &hi); + if (lo >= hi) { + return; + } + int split = named_first_assembly(arr, lo, hi); + int mine = c->group >= 0 ? named_of_owner(arr, split, hi, c->group) : own_shared(c, lo, split); + if (mine >= 0) { + found_add_type(c, fd, arr[mine].ent); + return; + } + int before = fd->n; + add_owners(c, lo, split, CS_WHY_SHARED, fd); + int shared = fd->n - before; + if (hi - split > CS_MAX_FOREIGN) { + fd->n += PAIR_LEN; /* more of them than one lookup compares: ambiguous */ + fd->why = CS_WHY_LIMIT; + return; + } + bool joined = false; + for (int i = split; i < hi; i++) { + const cs_entity_t *e = &c->ix->ents[arr[i].ent]; + if (e->twin < 0 || !ent_visible(c, e->twin)) { + found_add_type(c, fd, arr[i].ent); + } else { + joined = joined || e->joined; + } + } + if (fd->n - before > SKIP_ONE && shared < PAIR_LEN) { + fd->why = CS_WHY_ASSEMBLIES; + } + /* the name is one type only because a stub was joined to it */ + fd->joined = fd->joined || (joined && fd->n - before == SKIP_ONE); +} + +/* The type of this name and arity nested in `outer`: declared in that type + * itself, or in the shared trees' declaration that belongs to it. */ +static void add_nested_types(const cs_ctx_t *c, int outer, const char *name, int arity, + cs_found_t *fd) { + const cs_index_t *ix = c->ix; + cs_view_t v = view_of(ix, outer); + for (int k = 0; k < v.n; k++) { + int hi = 0; + int lo = named_range(ix->kids, ix->nkids, v.ent[k], name, arity, &hi); + if (lo < hi) { + /* one entity per outer type; the one of the type itself has the + * twin's as its own twin */ + found_add_type(c, fd, ix->kids[lo].ent); + /* a contract's nested type that only the joined implementation has */ + fd->joined = fd->joined || (k > 0 && ix->ents[outer].joined); + return; + } + } +} + +static void add_types(const cs_ctx_t *c, bool top, int scope, const char *name, int arity, + cs_found_t *fd) { + if (top) { + add_top_types(c, scope, name, arity, fd); + } else { + add_nested_types(c, scope, name, arity, fd); + } +} + +/* What a lookup asks one scope for. A parameter list is no part of it: the + * compiler finds the name first and matches the overloads afterwards. */ +typedef struct { + const char *name; + int arity; /* written type arguments; CS_ARITY_NONE when none */ + bool types_only; /* a qualifier: a namespace or a type */ + bool statics; /* through `using static`: static members only */ + bool ctors; /* the type's own name: its constructors (`statics`: the static one) */ + char kind; /* 0, or the only member kind a doc ID names: c v p e */ +} cs_query_t; + +static int type_arity(int written) { + return written > 0 ? written : 0; +} + +/* Members of one type under one key: those the entity declares and those the + * shared trees' parts that belong to it declare -- at most two ranges of the + * member table. */ +typedef struct { + int lo[PAIR_LEN]; + int hi[PAIR_LEN]; + int n; + int total; +} cs_mspans_t; + +/* The members of `ent` named `name`; with `group` >= 0 only those of that + * group with exactly the signature `sig`. */ +static cs_mspans_t mref_spans(const cs_index_t *ix, int ent, const char *name, int group, + const char *sig) { + cs_mspans_t sp = {0}; + cs_view_t v = view_of(ix, ent); + for (int k = 0; k < v.n; k++) { + int hi = 0; + int lo = mref_range(ix, v.ent[k], name, group, sig, &hi); + if (lo < hi) { + sp.lo[sp.n] = lo; + sp.hi[sp.n] = hi; + sp.n++; + sp.total += hi - lo; + } + } + return sp; +} + +/* The members `ent` declares under the query's name. A type's own name names + * its constructors, which no lookup by name finds. */ +static cs_mspans_t member_spans(const cs_ctx_t *c, int ent, const cs_query_t *q) { + if (!q->ctors && strcmp(q->name, c->ix->ents[ent].name) == 0) { + return (cs_mspans_t){0}; + } + return mref_spans(c->ix, ent, q->name, CS_NONE, NULL); +} + +/* True when the member takes part in a lookup of this query: methods of any + * arity for a name without type arguments, of that arity with them; every + * other member only without them. */ +static bool member_viable(const cs_ctx_t *c, const cs_mref_t *r, const cs_query_t *q) { + const cs_member_t *m = mref_member(c->ix, r); + if (r->group == PAIR_LEN || (q->kind && m->kind != q->kind)) { + return false; /* an operator or indexer has no identifier */ + } + if (q->ctors) { + return m->kind == 'c' && m->is_static == q->statics; + } + if (q->statics && !m->is_static) { + return false; + } + return m->kind == 'c' ? (q->arity <= 0 || m->arity == q->arity) : q->arity <= 0; +} + +static bool mref_visible(const cs_ctx_t *c, const cs_mref_t *r) { + return !(c->prod && (r->cls & CS_CLS_TEST)); +} + +enum { CS_HAS_NONE = 0, CS_HAS_VISIBLE, CS_HAS_INVISIBLE }; + +/* Does `ent` itself declare a member the query finds? A group larger than a + * lookup compares counts as there (and comes out ambiguous). */ +static int members_named(const cs_ctx_t *c, int ent, const cs_query_t *q) { + const cs_index_t *ix = c->ix; + cs_mspans_t sp = member_spans(c, ent, q); + if (sp.total > CS_MAX_OVERLOADS) { + return CS_HAS_VISIBLE; + } + int state = CS_HAS_NONE; + for (int s = 0; s < sp.n; s++) { + for (int i = sp.lo[s]; i < sp.hi[s]; i++) { + cs_work(SKIP_ONE); + if (!member_viable(c, &ix->mrefs[i], q)) { + continue; + } + if (mref_visible(c, &ix->mrefs[i])) { + return CS_HAS_VISIBLE; + } + state = CS_HAS_INVISIBLE; + } + } + return state; +} + +/* True when `ent` -- or the shared trees' declaration that belongs to it -- + * declares an operator or an indexer by this name (its token, or `this`) + * that the context may see: asked of what was noted when the members were + * listed (special_mark), whatever the number of such members. */ +static bool special_declared(const cs_ctx_t *c, int ent, const char *name) { + const cs_index_t *ix = c->ix; + cs_view_t v = view_of(ix, ent); + char key[CS_NAME_BUF + CBM_SZ_64]; + for (int k = 0; k < v.n; k++) { + cs_work(SKIP_ONE); + if (special_key(key, sizeof(key), c->prod, v.ent[k], name) && + cbm_ht_get(ix->specials, key)) { + return true; + } + } + return false; +} + +/* The supertypes of `ent`, nearest first: a class's base classes, an + * interface's base interfaces. At most CS_MAX_SUPERS; *more when the + * hierarchy goes on behind them. (The compiler's cref lookup never comes + * here: see CS_BIND_INHERITED.) */ +static int supers_of(const cs_index_t *ix, int ent, int *out, bool *more) { + int n = 0; + *more = false; + bool iface = ix->ents[ent].kind == 'i'; + int cur = ent; + for (int head = -SKIP_ONE; head < n; head++) { + if (head >= 0) { + cur = out[head]; + } + /* the bases its own declarations write, and those the shared trees' + * parts of it write */ + cs_view_t v = view_of(ix, cur); + for (int k = 0; k < v.n; k++) { + const cs_entity_t *e = &ix->ents[v.ent[k]]; + for (int b = 0; b < e->nbases; b++) { + char bk = ix->ents[e->bases[b]].kind; + if (iface ? bk != 'i' : !(bk == 'c' || bk == 'r')) { + continue; + } + bool seen = e->bases[b] == ent; + for (int j = 0; !seen && j < n; j++) { + seen = out[j] == e->bases[b]; + } + if (seen) { + continue; + } + if (n >= CS_MAX_SUPERS) { + *more = true; + return n; + } + out[n++] = e->bases[b]; + } + } + } + return n; +} + +/* ── Reference syntax ────────────────────────────────────────────── */ + +typedef struct { + char name[CS_NAME_BUF]; + int arity; /* CS_ARITY_NONE: no type arguments written */ + const char *targs; /* the type arguments as written, between the brackets; NULL: none */ + size_t targs_len; +} cs_seg_t; + +enum { CS_OP_NAME = 16 }; + +typedef struct { + char text[CS_REF_BUF]; /* the reference, trimmed: the segments' type arguments are in it */ + cs_seg_t segs[CS_MAX_SEGS]; + int nsegs; + bool has_params; + char params[CS_MAX_PARAMS][CS_PARAM_BUF]; + int nparams; + char sig[CS_MAX_PARAMS * CS_PARAM_BUF]; /* the parameters as a declaration's signature */ + bool sig_unknown; /* a parameter type nothing is known about */ + char docid; /* 0, or T M P F E N; O: DocFX's overload group */ + bool glob; /* starts at the global namespace */ + bool op; /* an operator, a conversion or an indexer */ + char op_name[CS_OP_NAME]; /* the name the scope records it under */ + bool maybe_indexer; /* `Item(...)`: a method of that name, else the indexer */ + bool keyword; /* a keyword alias, rewritten to its System type */ +} cs_ref_t; + +static bool is_open_bracket(char c) { + return c == '<' || c == '{' || c == '[' || c == '('; +} + +static bool is_close_bracket(char c) { + return c == '>' || c == '}' || c == ']' || c == ')'; +} + +/* Index just past the bracket group opened at s[i] (< { [ ( nest together). */ +static size_t group_end(const char *s, size_t n, size_t i) { + int depth = 0; + for (size_t k = i; k < n; k++) { + if (is_open_bracket(s[k])) { + depth++; + } else if (is_close_bracket(s[k]) && --depth == 0) { + return k + SKIP_ONE; + } + } + return n; +} + +/* Count top-level comma-separated items of s[0..n). */ +static int count_top(const char *s, size_t n) { + bool any = false; + int items = 1; + int depth = 0; + for (size_t i = 0; i < n; i++) { + char c = s[i]; + if (is_open_bracket(c)) { + depth++; + } else if (is_close_bracket(c)) { + depth--; + } else if (c == ',' && depth == 0) { + items++; + } + any = any || !isspace((unsigned char)c); + } + return any ? items : 0; +} + +/* An identifier as the scanner takes one: a letter of any script, a digit + * after the first position, an underscore. */ +static bool ident_ok(const char *s) { + if (strcmp(s, "#ctor") == 0 || strcmp(s, "#cctor") == 0) { + return true; + } + unsigned char first = (unsigned char)s[0]; + if (!(isalpha(first) || first == '_' || first >= CBM_SZ_128)) { + return false; + } + for (const char *p = s + SKIP_ONE; *p; p++) { + unsigned char ch = (unsigned char)*p; + if (!(isalnum(ch) || ch == '_' || ch >= CBM_SZ_128)) { + return false; + } + } + return true; +} + +/* One dotted segment: `Name`, `Name{T,U}`, `Name`, `Name``2`. */ +static bool parse_seg(const char *s, size_t n, cs_seg_t *out) { + while (n > 0 && isspace((unsigned char)*s)) { + s++; + n--; + } + while (n > 0 && isspace((unsigned char)s[n - SKIP_ONE])) { + n--; + } + out->arity = CS_ARITY_NONE; + out->targs = NULL; + out->targs_len = 0; + size_t name_end = n; + for (size_t i = 0; i < n; i++) { + if (s[i] == '`') { + name_end = i; + size_t d = i; + while (d < n && s[d] == '`') { + d++; + } + out->arity = atoi(s + d); + break; + } + if (s[i] == '{' || s[i] == '<') { + name_end = i; + size_t e = group_end(s, n, i); + if (e < n) { + return false; /* text after the type arguments: no name */ + } + size_t inner = e > i + PAIR_LEN ? e - i - PAIR_LEN : 0; + out->arity = count_top(s + i + SKIP_ONE, inner); + out->targs = s + i + SKIP_ONE; + out->targs_len = inner; + break; + } + } + const char *name = s; + if (name_end > 0 && name[0] == '@') { + name++; + name_end--; + } + if (name_end == 0 || name_end >= sizeof(out->name)) { + return false; + } + memcpy(out->name, name, name_end); + out->name[name_end] = '\0'; + return ident_ok(out->name); +} + +/* Split a dotted path (dots inside type-argument groups do not split). */ +/* A path whose brackets do not pair is not read at all: its last segment + * would be dropped and the reference resolved by the segments before it. */ +static bool parse_path(const char *s, size_t n, cs_seg_t *segs, int *nsegs) { + *nsegs = 0; + size_t start = 0; + int depth = 0; + for (size_t i = 0; i <= n; i++) { + char c = i < n ? s[i] : '.'; + if (is_open_bracket(c)) { + depth++; + } else if (is_close_bracket(c)) { + if (--depth < 0) { + return false; + } + } else if (c == '.' && depth == 0) { + if (*nsegs >= CS_MAX_SEGS || !parse_seg(s + start, i - start, &segs[*nsegs])) { + return false; + } + (*nsegs)++; + start = i + SKIP_ONE; + } + } + return depth == 0 && *nsegs > 0; +} + +static const char *const CS_KEYWORD_TYPES[][2] = { + {"int", "Int32"}, {"string", "String"}, {"object", "Object"}, {"bool", "Boolean"}, + {"byte", "Byte"}, {"sbyte", "SByte"}, {"short", "Int16"}, {"ushort", "UInt16"}, + {"uint", "UInt32"}, {"long", "Int64"}, {"ulong", "UInt64"}, {"float", "Single"}, + {"double", "Double"}, {"decimal", "Decimal"}, {"char", "Char"}, {"nint", "IntPtr"}, + {"nuint", "UIntPtr"}, {"void", "Void"}, +}; + +static const char *keyword_type(const char *s) { + for (size_t i = 0; i < sizeof(CS_KEYWORD_TYPES) / sizeof(CS_KEYWORD_TYPES[0]); i++) { + if (strcmp(s, CS_KEYWORD_TYPES[i][0]) == 0) { + return CS_KEYWORD_TYPES[i][1]; + } + } + return NULL; +} + +/* The token a metadata operator name stands for (`op_Addition` -> `+`): the + * name the scope records an operator under. NULL for a name that is none. */ +static const char *operator_token(const char *name, size_t len) { + static const struct { + const char *meta; + const char *token; + } ops[] = { + {"Addition", "+"}, + {"UnaryPlus", "+"}, + {"Subtraction", "-"}, + {"UnaryNegation", "-"}, + {"Multiply", "*"}, + {"Division", "/"}, + {"Modulus", "%"}, + {"BitwiseAnd", "&"}, + {"BitwiseOr", "|"}, + {"ExclusiveOr", "^"}, + {"LeftShift", "<<"}, + {"RightShift", ">>"}, + {"UnsignedRightShift", ">>>"}, + {"Equality", "=="}, + {"Inequality", "!="}, + {"LessThan", "<"}, + {"GreaterThan", ">"}, + {"LessThanOrEqual", "<="}, + {"GreaterThanOrEqual", ">="}, + {"LogicalNot", "!"}, + {"OnesComplement", "~"}, + {"Increment", "++"}, + {"Decrement", "--"}, + {"True", "true"}, + {"False", "false"}, + {"Implicit", "implicit"}, + {"Explicit", "explicit"}, + {"CheckedAddition", "+"}, + {"CheckedSubtraction", "-"}, + {"CheckedMultiply", "*"}, + {"CheckedDivision", "/"}, + {"CheckedUnaryNegation", "-"}, + {"CheckedIncrement", "++"}, + {"CheckedDecrement", "--"}, + {"CheckedExplicit", "explicit"}, + }; + for (size_t i = 0; i < sizeof(ops) / sizeof(ops[0]); i++) { + if (strlen(ops[i].meta) == len && memcmp(ops[i].meta, name, len) == 0) { + return ops[i].token; + } + } + return NULL; +} + +static bool starts_word(const char *s, const char *word) { + size_t n = strlen(word); + return strncmp(s, word, n) == 0 && !isalnum((unsigned char)s[n]) && s[n] != '_'; +} + +static const char *skip_blanks(const char *s) { + while (*s == ' ') { + s++; + } + return s; +} + +/* The operator, conversion or indexer written at `q` (the start of a + * segment): its recorded name into `name`. false when `q` starts none. */ +static bool operator_at(const char *q, char name[CS_OP_NAME]) { + static const char op_prefix[] = "op_"; + q = skip_blanks(q); + bool implicit = starts_word(q, "implicit"); + if (implicit || starts_word(q, "explicit")) { + if (!starts_word(skip_blanks(q + strlen("implicit")), "operator")) { + return false; + } + snprintf(name, CS_OP_NAME, "%s", implicit ? "implicit" : "explicit"); + return true; + } + if (starts_word(q, "operator")) { + const char *t = skip_blanks(q + strlen("operator")); + if (starts_word(t, "checked")) { + t = skip_blanks(t + strlen("checked")); + } + size_t n = 0; + while (t[n] && t[n] != '(' && t[n] != ' ' && n + SKIP_ONE < CS_OP_NAME) { + n++; + } + snprintf(name, CS_OP_NAME, "%.*s", (int)n, n > 0 ? t : "?"); + return true; + } + if (starts_word(q, "this")) { + const char *t = skip_blanks(q + strlen("this")); + if (*t != '[' && *t != '\0') { + return false; + } + snprintf(name, CS_OP_NAME, "this"); + return true; + } + size_t pl = sizeof(op_prefix) - SKIP_ONE; + if (strncmp(q, op_prefix, pl) == 0 && isupper((unsigned char)q[pl])) { + size_t n = 0; + while (isalnum((unsigned char)q[pl + n])) { + n++; + } + const char *token = operator_token(q + pl, n); + if (!token) { + return false; /* `op_Custom`: an identifier like any other */ + } + snprintf(name, CS_OP_NAME, "%s", token); + return true; + } + return false; +} + +/* Where the reference's last segment starts an operator, a conversion or an + * indexer: its offset in `s`, or CS_NONE. */ +static int operator_start(const char *s, char name[CS_OP_NAME]) { + int depth = 0; + for (size_t i = 0; s[i]; i++) { + if ((i == 0 || (s[i - SKIP_ONE] == '.' && depth == 0)) && operator_at(s + i, name)) { + return (int)i; + } + if (is_open_bracket(s[i])) { + depth++; + } else if (is_close_bracket(s[i])) { + depth--; + } + } + return CS_NONE; +} + +/* A doc ID writes a type parameter by its position: `0 (of the type and the + * types around it), ``0 (of the method). Such a parameter is kept as it + * stands (without a by-reference mark): its position is what is compared. */ +static bool slot_param(const char *s, size_t n, char *out, size_t cap) { + while (n > 0 && isspace((unsigned char)*s)) { + s++; + n--; + } + while (n > 0 && (isspace((unsigned char)s[n - SKIP_ONE]) || s[n - SKIP_ONE] == '@')) { + n--; + } + size_t ticks = 0; + while (ticks < n && s[ticks] == '`') { + ticks++; + } + if (ticks == 0 || ticks > PAIR_LEN || ticks >= n || !isdigit((unsigned char)s[ticks])) { + return false; + } + size_t d = ticks; + while (d < n && isdigit((unsigned char)s[d])) { + d++; + } + for (size_t k = d; k < n; k++) { + if (!strchr("[],*", s[k])) { + return false; + } + } + if (n >= cap) { + return false; + } + memcpy(out, s, n); + out[n] = '\0'; + return true; +} + +/* The written parameter list s[from, to) into the reference: every parameter + * normalized, and all of them as one signature. false for more parameters + * than a reference may have. */ +static bool parse_params(cs_ref_t *r, const char *s, size_t from, size_t to) { + bool any = false; + for (size_t i = from; i < to && !any; i++) { + any = !isspace((unsigned char)s[i]); + } + size_t start = from; + int depth = 0; + size_t w = 0; + for (size_t i = from; any && i <= to; i++) { + char c = i < to ? s[i] : ','; + if (is_open_bracket(c)) { + depth++; + } else if (is_close_bracket(c)) { + if (--depth < 0) { + return false; /* a bracket that closes nothing */ + } + } else if (c == ',' && depth == 0) { + if (r->nparams >= CS_MAX_PARAMS) { + return false; + } + char *p = r->params[r->nparams++]; + if (!slot_param(s + start, i - start, p, CS_PARAM_BUF)) { + (void)cbm_doclink_cs_norm_type(s + start, i - start, p, CS_PARAM_BUF); + } + size_t pl = strlen(p); + r->sig_unknown = r->sig_unknown || strchr(p, '?') != NULL; + if (w > 0) { + r->sig[w++] = '|'; + } + memcpy(r->sig + w, p, pl); + w += pl; + start = i + SKIP_ONE; + } + } + r->sig[w] = '\0'; + /* a bracket left open would have swallowed the last parameter: the + * reference would be matched without it */ + return depth == 0; +} + +/* A written reference into its parts. false for one this code does not + * understand -- and for one longer than CS_REF_BUF, which is never cut and + * resolved by what is left of it. */ +static bool parse_cref(const char *raw, cs_ref_t *r) { + static const char global_prefix[] = "global::"; + r->nsegs = 0; + r->nparams = 0; + r->has_params = r->sig_unknown = r->glob = r->op = r->maybe_indexer = r->keyword = false; + r->docid = 0; + r->sig[0] = '\0'; + r->op_name[0] = '\0'; + const char *in = raw ? raw : ""; + while (isspace((unsigned char)*in)) { + in++; + } + size_t n = strlen(in); + while (n > 0 && isspace((unsigned char)in[n - SKIP_ONE])) { + n--; + } + if (n == 0 || n >= sizeof(r->text)) { + return false; + } + memcpy(r->text, in, n); + r->text[n] = '\0'; + char *s = r->text; + if (n > PAIR_LEN && s[1] == ':' && strchr("TMPFENO!", s[0])) { + char k = s[0]; + if (k == '!') { + return false; /* the compiler's error marker: it could not bind the reference */ + } + s += PAIR_LEN; + while (isspace((unsigned char)*s)) { + s++; + } + if (k == 'O') { /* DocFX's overload group: a member without a signature */ + char *paren = strchr(s, '('); + if (paren) { + *paren = '\0'; + } + } + r->docid = k; + r->glob = true; /* a doc ID is a full name */ + } + size_t gl = sizeof(global_prefix) - SKIP_ONE; + if (strncmp(s, global_prefix, gl) == 0) { + r->glob = true; + s += gl; + } + n = strlen(s); + int op = operator_start(s, r->op_name); + if (op >= 0) { + r->op = true; + return op == 0 || parse_path(s, (size_t)op - SKIP_ONE, r->segs, &r->nsegs); + } + /* `Path(params)`: the first top-level parenthesis starts the list. */ + size_t paren = n; + int depth = 0; + for (size_t i = 0; i < n; i++) { + char c = s[i]; + if (c == '<' || c == '{' || c == '[') { + depth++; + } else if (c == '>' || c == '}' || c == ']') { + depth--; + } else if (c == '(' && depth == 0) { + paren = i; + break; + } + } + if (!parse_path(s, paren, r->segs, &r->nsegs)) { + return false; + } + if (paren < n) { + r->has_params = true; + size_t close = group_end(s, n, paren); + bool closed = close > paren && s[close - SKIP_ONE] == ')'; + /* a parameter list that is not closed, or text after it, is not + * matched by what can be read of it */ + for (size_t k = close; closed && k < n; k++) { + closed = isspace((unsigned char)s[k]) != 0; + } + if (!closed || !parse_params(r, s, paren + SKIP_ONE, close - SKIP_ONE)) { + return false; + } + } + r->maybe_indexer = r->has_params && strcmp(r->segs[r->nsegs - SKIP_ONE].name, "Item") == 0; + return true; +} + +/* ── Results ─────────────────────────────────────────────────────── */ + +/* Where the complete declarations of `e` in units before `unit` end. */ +static int fulls_before(const cs_entity_t *e, int unit) { + int lo = 0; + int hi = e->nfulls; + while (lo < hi) { + int mid = lo + ((hi - lo) / PAIR_LEN); + if (e->fulls[mid].unit < unit) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + return lo; +} + +/* The complete declaration of `e` that stands in the referencing file's own + * project (or shared tree) and that it may bind: one with a node when + * there is one, the nearest of several. NULL when there is none -- and when + * that unit holds more of them than one lookup compares (*too_many). */ +static const cs_full_t *own_full(const cs_ctx_t *c, const cs_entity_t *e, bool *too_many) { + const cs_index_t *ix = c->ix; + int lo = fulls_before(e, c->unit); + int hi = fulls_before(e, c->unit + SKIP_ONE); + if (hi - lo > CS_MAX_OVERLOADS) { + *too_many = true; + return NULL; + } + const cs_full_t *best = NULL; + bool best_node = false; + for (int i = lo; i < hi; i++) { + const cs_full_t *d = &e->fulls[i]; + if (c->prod && ix->files[d->file].is_test) { + continue; + } + bool node = ix->files[d->file].types[d->type].node != NULL; + if (!best || (node && !best_node) || + (node == best_node && file_nearer(ix, d->file, best->file, c->file))) { + best = d; + best_node = node; + } + } + return best; +} + +/* The one assembly that has only stubs of the type and takes the shared + * trees' declaration `ent` as its implementation; CS_NONE when there is + * none, or more than one. Its stubs are the type's stubs: where the + * implementation has no node, or a parse error hides a member of it, the + * stub's stands in -- as it does inside one assembly. */ +static int stub_user(const cs_index_t *ix, int ent) { + int stub = ix->ents[ent].stub; + return stub >= 0 ? stub : CS_NONE; +} + +/* The implementation's node among one entity's own declarations. What is an + * alternative is not chosen between: + * - complete declarations of the type in two or more projects (the + * flavours of one assembly's type): the one of the referencing file's own + * project is meant -- and when that one has no node, no other flavour's + * stands in for it; + * - parts in two or more shared trees, which may or may not be compiled + * together: the ones of the referencing file's own tree. + * From anywhere else the type is AMBIGUOUS (*why). GAP when no + * implementation in view has a node. */ +static cs_pick_t impl_node(const cs_ctx_t *c, const cs_entity_t *e, const cbm_gbuf_node_t **node, + cs_why_t *why) { + if ((c->prod ? e->full_units_prod : e->full_units) >= PAIR_LEN) { + bool too_many = false; + const cs_full_t *own = own_full(c, e, &too_many); + if (!own) { + *why = too_many ? CS_WHY_LIMIT : CS_WHY_FLAVOURS; + return CS_PICK_AMBIGUOUS; + } + *node = c->ix->files[own->file].types[own->type].node; + return *node ? CS_PICK_NODE : CS_PICK_GAP; + } + if (e->shared_parts && e->nunits >= PAIR_LEN && !ent_in_unit(e, c->unit)) { + *why = CS_WHY_SHARED; + return CS_PICK_AMBIGUOUS; + } + *node = class_node(c, e, false); + return *node ? CS_PICK_NODE : CS_PICK_GAP; +} + +/* The type's node for a reference from this context: an implementation's -- + * of the type's own declarations, then of the shared trees' declaration that + * belongs to it -- and a reference assembly's stub's only when no + * implementation has one. Never a test declaration for product code. A + * visible type whose eligible declarations all lack a node is a graph gap. A + * node that is the type's only because a contract is joined to its + * implementation is never an exact binding. */ +static cs_res_t type_result(const cs_ctx_t *c, int ent, bool exact) { + const cs_index_t *ix = c->ix; + cs_view_t v = view_of(ix, ent); + const cbm_gbuf_node_t *node = NULL; + bool in_view = false; + for (int k = 0; k < v.n; k++) { + const cs_entity_t *e = &ix->ents[v.ent[k]]; + cs_why_t why = CS_WHY_SCOPE; + cs_pick_t p = impl_node(c, e, &node, &why); + if (p == CS_PICK_NODE) { + return res_edge(node, exact && !(k > 0 && ix->ents[ent].joined)); + } + if (p == CS_PICK_AMBIGUOUS) { + return res_ambiguous(why); + } + in_view = in_view || declared_in_view(c, e); + } + for (int k = 0; k < v.n; k++) { + node = class_node(c, &ix->ents[v.ent[k]], true); + if (node) { + return res_edge(node, exact); + } + } + int stubs = stub_user(ix, ent); + node = stubs >= 0 ? class_node(c, &ix->ents[stubs], true) : NULL; + if (node) { + return res_edge(node, false); + } + return res_unres(in_view ? CBM_DOCLINK_REASON_GRAPH_GAP : CBM_DOCLINK_REASON_TEST_ONLY); +} + +/* Bind the members in `sp` -- one name, group and signature: the + * declarations of one member. An implementation's node before a reference + * assembly's stub's; never a test declaration for product code; the nearest + * of several. Declarations of the member in two or more projects (or shared + * trees) are alternatives: the one of the referencing file's own is meant, + * and from anywhere else none can be chosen. More declarations than + * one lookup compares are ambiguous as well. `view`: the type the members + * were looked up in (CS_NONE: none to speak of) -- a member a contract has + * only through the implementation joined to it is never an exact binding. */ +static cs_res_t bind_members(const cs_ctx_t *c, const cs_mspans_t *sp, bool exact, int view) { + const cs_index_t *ix = c->ix; + if (sp->total > CS_MAX_OVERLOADS) { + return res_ambiguous(CS_WHY_LIMIT); + } + int first_unit = CS_NONE; + bool several = false; + bool own = false; + bool visible = false; + for (int s = 0; s < sp->n; s++) { + for (int i = sp->lo[s]; i < sp->hi[s]; i++) { + const cs_mref_t *r = &ix->mrefs[i]; + if (!mref_visible(c, r)) { + continue; + } + visible = true; + if (r->cls & CS_CLS_REF) { + continue; + } + int unit = ix->files[r->file].unit; + own = own || unit == c->unit; + several = several || (first_unit >= 0 && unit != first_unit); + first_unit = first_unit >= 0 ? first_unit : unit; + } + } + if (!visible) { + return res_unres(CBM_DOCLINK_REASON_TEST_ONLY); + } + if (several && !own) { + return res_ambiguous(CS_WHY_FLAVOURS); + } + /* an implementation's node (the own one of alternatives), else a stub's */ + const cs_mref_t *best[PAIR_LEN] = {NULL, NULL}; + for (int s = 0; s < sp->n; s++) { + for (int i = sp->lo[s]; i < sp->hi[s]; i++) { + const cs_mref_t *r = &ix->mrefs[i]; + bool stub = (r->cls & CS_CLS_REF) != 0; + if (!mref_visible(c, r) || (r->cls & CS_CLS_NO_NODE) || + (!stub && several && ix->files[r->file].unit != c->unit)) { + continue; + } + if (!best[stub] || file_nearer(ix, r->file, best[stub]->file, c->file)) { + best[stub] = r; + } + } + } + const cs_mref_t *pick = best[0] ? best[0] : best[SKIP_ONE]; + if (!pick) { + return res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); + } + bool through_join = view >= 0 && ix->ents[view].joined && pick->ent != view; + return res_edge(mref_member(ix, pick)->node, exact && !through_join); +} + +/* Bind the type's members of one name, group and signature. A member whose + * implementation has no node is its stub's, where one assembly's stubs stand + * for this declaration (stub_user): never an exact binding. */ +static cs_res_t bind_signature(const cs_ctx_t *c, int ent, const char *name, int group, + const char *sig, bool exact) { + cs_mspans_t sp = mref_spans(c->ix, ent, name, group, sig); + cs_res_t res = bind_members(c, &sp, exact, ent); + int stubs = res_is(&res, CBM_DOCLINK_REASON_GRAPH_GAP) ? stub_user(c->ix, ent) : CS_NONE; + if (stubs >= 0) { + cs_mspans_t of_stubs = mref_spans(c->ix, stubs, name, group, sig); + cs_res_t stub = bind_members(c, &of_stubs, false, CS_NONE); + if (stub.st == CS_OK) { + return stub; + } + } + return res; +} + +/* Which written segments name the declaring type and the member: a written + * type argument stands for the type parameter at its position there. */ +typedef struct { + int type_seg; /* CS_NONE: the type is not written (a simple name in scope) */ + int member_seg; /* CS_NONE: the member is not written (a constructor by its type) */ +} cs_where_t; + +/* How a found name is used. */ +typedef struct { + const cs_ref_t *r; + bool has_params; /* a parameter list follows the name */ + cs_where_t where; + bool exact; /* tier of a member bound through the written path */ + bool exact_type; /* ... and of a type the whole path names */ +} cs_use_t; + +static size_t suffix_start(const char *s, size_t n) { + size_t e = n; + while (e > 0 && strchr("[],*", s[e - SKIP_ONE])) { + e--; + } + return e; +} + +/* Position of `name` among the ','-separated type arguments of a segment. */ +static int targ_position(const cs_seg_t *seg, const char *name, size_t len) { + int pos = 0; + int depth = 0; + size_t start = 0; + for (size_t i = 0; seg->targs && i <= seg->targs_len; i++) { + char ch = i < seg->targs_len ? seg->targs[i] : ','; + if (is_open_bracket(ch)) { + depth++; + } else if (is_close_bracket(ch)) { + depth--; + } else if (ch == ',' && depth == 0) { + const char *a = seg->targs + start; + size_t al = i - start; + while (al > 0 && isspace((unsigned char)*a)) { + a++; + al--; + } + while (al > 0 && isspace((unsigned char)a[al - SKIP_ONE])) { + al--; + } + if (al == len && memcmp(a, name, len) == 0) { + return pos; + } + pos++; + start = i + SKIP_ONE; + } + } + return CS_NONE; +} + +/* The position a doc ID's `N (*method false) or ``N (*method true) names; + * CS_NONE for any other text. */ +static int slot_of(const char *s, size_t n, bool *method) { + size_t ticks = 0; + while (ticks < n && s[ticks] == '`') { + ticks++; + } + if (ticks == 0 || ticks > PAIR_LEN || ticks == n) { + return CS_NONE; + } + int pos = 0; + for (size_t i = ticks; i < n; i++) { + if (!isdigit((unsigned char)s[i]) || pos > CBM_SZ_4K) { + return CS_NONE; + } + pos = (pos * CBM_DECIMAL_BASE) + (s[i] - '0'); + } + *method = ticks == PAIR_LEN; + return pos; +} + +/* A written type variable and a declared one are the same when they stand at + * the same position of the same list: the method's own list, the declaring + * type's, its outer type's, and so on. A doc ID writes the position itself; + * there the positions of a type's list count on from its outer types'. */ +static bool same_type_variable(const cs_ctx_t *c, const cs_ref_t *r, cs_where_t w, + const cs_mref_t *mr, const char *written, size_t wl, + const char *declared, size_t dl) { + const cs_file_t *f = &c->ix->files[mr->file]; + bool slot_method = false; + int slot = slot_of(written, wl, &slot_method); + int pos = tparam_find(f, f->ntypes + mr->midx, declared, dl); + if (pos >= 0) { + if (slot >= 0) { + return slot_method && slot == pos; + } + return w.member_seg >= 0 && targ_position(&r->segs[w.member_seg], written, wl) == pos; + } + int seg = w.type_seg; + for (int t = f->members[mr->midx].type; t >= 0; t = f->types[t].outer, seg--) { + pos = tparam_find(f, t, declared, dl); + if (pos < 0) { + continue; + } + if (slot >= 0) { + int before = 0; + for (int o = f->types[t].outer; o >= 0; o = f->types[o].outer) { + before += f->types[o].arity; + } + return !slot_method && slot == before + pos; + } + return seg >= 0 && targ_position(&r->segs[seg], written, wl) == pos; + } + return false; +} + +enum { CS_FIT_NO = 0, CS_FIT_UNKNOWN, CS_FIT_EXACT }; + +/* How the written parameter list fits a declared signature: EXACT when every + * parameter type is the declared one (the same text, or the same type + * variable), UNKNOWN when the rest are types nothing is known about on + * either side ("?"), NO otherwise. */ +static int sig_fit(const cs_ctx_t *c, const cs_ref_t *r, cs_where_t w, const cs_mref_t *mr) { + const char *sig = mref_member(c->ix, mr)->sig; + /* counted when the declaration was read: a declared signature is not + * walked for every reference that cannot mean it */ + int nd = mref_member(c->ix, mr)->nparams; + if (nd != r->nparams) { + return CS_FIT_NO; + } + cs_work((uint64_t)nd); + int fit = CS_FIT_EXACT; + const char *p = sig; + for (int i = 0; i < nd; i++) { + const char *e = strchr(p, '|'); + size_t dn = e ? (size_t)(e - p) : strlen(p); + const char *a = r->params[i]; + size_t an = strlen(a); + if (!(an == dn && memcmp(a, p, an) == 0)) { + size_t ab = suffix_start(a, an); + size_t db = suffix_start(p, dn); + bool same_suffix = (an - ab) == (dn - db) && memcmp(a + ab, p + db, an - ab) == 0; + if (same_suffix && same_type_variable(c, r, w, mr, a, ab, p, db)) { + /* the same type variable */ + } else if ((an == SKIP_ONE && a[0] == '?') || (dn == SKIP_ONE && p[0] == '?')) { + fit = CS_FIT_UNKNOWN; + } else { + return CS_FIT_NO; + } + } + p = e ? e + SKIP_ONE : p + dn; + } + return fit; +} + +typedef enum { + CS_MB_NONE = 0, /* the entity declares no such member */ + CS_MB_INVISIBLE, /* only its test declarations do, and the reference is product code */ + CS_MB_MISMATCH, /* it declares the name; nothing of it takes the written parameters */ + CS_MB_RES, /* *res is the answer */ +} cs_mb_t; + +/* The selected members of one side (generic callables, non-generic ones, + * values): how many, and whether they are one declaration's worth -- one + * signature and arity. */ +typedef struct { + int n; + const char *sig; + int arity; + bool many; +} cs_side_t; + +static void side_add(cs_side_t *s, const cs_member_t *m) { + const char *sig = m->sig ? m->sig : ""; + if (s->n == 0) { + s->sig = sig; + s->arity = m->arity; + } else if (strcmp(s->sig, sig) != 0 || s->arity != m->arity) { + s->many = true; + } + s->n++; +} + +/* The members the query finds in `ent`, for a reference without a parameter + * list: one member binds; a method name without type arguments means the + * non-generic methods when there are any; an overload group, and a method + * beside a value of the same name, are ambiguous. */ +static cs_mb_t members_plain(const cs_ctx_t *c, int ent, const cs_query_t *q, bool exact, + cs_res_t *res) { + const cs_index_t *ix = c->ix; + cs_mspans_t sp = member_spans(c, ent, q); + if (sp.total == 0) { + return CS_MB_NONE; + } + if (sp.total > CS_MAX_OVERLOADS) { + *res = res_ambiguous(CS_WHY_LIMIT); /* more than one lookup compares */ + return CS_MB_RES; + } + cs_side_t plain = {0}; + cs_side_t generic = {0}; + cs_side_t values = {0}; + bool invisible = false; + for (int s = 0; s < sp.n; s++) { + for (int i = sp.lo[s]; i < sp.hi[s]; i++) { + cs_work(SKIP_ONE); + const cs_mref_t *mr = &ix->mrefs[i]; + if (!member_viable(c, mr, q)) { + continue; + } + if (!mref_visible(c, mr)) { + invisible = true; + continue; + } + const cs_member_t *m = mref_member(ix, mr); + side_add(m->kind != 'c' ? &values : (m->arity > 0 ? &generic : &plain), m); + } + } + const cs_side_t *calls = plain.n > 0 ? &plain : &generic; + if (calls->n + values.n == 0) { + return invisible ? CS_MB_INVISIBLE : CS_MB_NONE; + } + if ((calls->n > 0 && values.n > 0) || calls->many) { + *res = res_ambiguous(CS_WHY_SCOPE); + } else if (values.n > 0) { + *res = bind_signature(c, ent, q->name, 0, "", exact); + } else { + *res = bind_signature(c, ent, q->name, SKIP_ONE, calls->sig, exact); + } + return CS_MB_RES; +} + +/* The same for a reference WITH a parameter list: the overload with exactly + * the written parameter types (the same text, or the same type variables); + * only when there is none, one whose difference is a type nothing is known + * about. A non-generic match stands before a generic one; several distinct + * matches are ambiguous. A value never takes a parameter list. */ +static cs_mb_t members_signed(const cs_ctx_t *c, int ent, const cs_query_t *q, const cs_use_t *u, + cs_res_t *res) { + const cs_index_t *ix = c->ix; + cs_mspans_t sp = member_spans(c, ent, q); + if (sp.total == 0) { + return CS_MB_NONE; + } + if (sp.total > CS_MAX_OVERLOADS) { + /* too many to compare one by one: the written text itself can still + * be looked up */ + cs_mspans_t written = mref_spans(ix, ent, q->name, SKIP_ONE, u->r->sig); + *res = (written.total > 0 && !u->r->sig_unknown) ? bind_members(c, &written, u->exact, ent) + : res_ambiguous(CS_WHY_LIMIT); + return CS_MB_RES; + } + bool named = false; + bool invisible = false; + for (int pass = CS_FIT_EXACT; pass >= CS_FIT_UNKNOWN; pass--) { + cs_side_t plain = {0}; + cs_side_t generic = {0}; + for (int s = 0; s < sp.n; s++) { + for (int i = sp.lo[s]; i < sp.hi[s]; i++) { + cs_work(SKIP_ONE); + const cs_mref_t *mr = &ix->mrefs[i]; + if (!member_viable(c, mr, q)) { + continue; + } + if (!mref_visible(c, mr)) { + invisible = true; + continue; + } + named = true; + const cs_member_t *m = mref_member(ix, mr); + if (m->kind == 'c' && sig_fit(c, u->r, u->where, mr) == pass) { + side_add(m->arity > 0 ? &generic : &plain, m); + } + } + } + const cs_side_t *s = plain.n > 0 ? &plain : &generic; + if (s->n > 0) { + *res = s->many ? res_ambiguous(CS_WHY_SCOPE) + : bind_signature(c, ent, q->name, SKIP_ONE, s->sig, u->exact); + return CS_MB_RES; + } + } + if (named) { + return CS_MB_MISMATCH; + } + return invisible ? CS_MB_INVISIBLE : CS_MB_NONE; +} + +static cs_mb_t members_result(const cs_ctx_t *c, int ent, const cs_query_t *q, const cs_use_t *u, + cs_res_t *res) { + return u->has_params ? members_signed(c, ent, q, u, res) + : members_plain(c, ent, q, u->exact, res); +} + +/* True when a part of the type `ent` that this reference does not see + * declares `name`. `ent` is a shared trees' declaration that assemblies have + * as a part of their type, seen here without one of them: the reference + * stands in a shared tree, or in an assembly that has no part of its own. + * One of those assemblies' own parts declares an implementation of that name + * -- a nested type that the shared trees do not have, or (`types_only` + * unset) a member. Which assembly the reference is compiled into is not + * known, so the name is neither bound nor missing. A stub does not count: it + * declares what the implementation declares. The assemblies are not walked: + * what their parts add is looked up by the name (build_extras). *why: when + * more assemblies declare a nested type of the name than one lookup compares + * (and none of those compared is in view), that is said. */ +static bool unseen_part_declares(const cs_ctx_t *c, int ent, const char *name, int arity, + bool types_only, cs_why_t *why) { + const cs_index_t *ix = c->ix; + int hi = 0; + int lo = extra_range(ix, ent, name, type_arity(arity), &hi); + for (int i = lo; i < hi; i++) { + if (i - lo >= CS_MAX_FOREIGN) { + *why = CS_WHY_LIMIT; + return true; + } + cs_work(SKIP_ONE); + if (ent_visible(c, ix->extras[i].ent)) { + return true; + } + } + if (types_only) { + return false; + } + lo = extra_range(ix, ent, name, CS_NONE, &hi); + return lo < hi && (ix->extras[lo].prod || !c->prod); +} + +/* The reason for a member a type does not show (`name`, when it has one). */ +static cs_res_t member_unfound(const cs_ctx_t *c, int ent, const char *name, int arity) { + const cs_entity_t *e = &c->ix->ents[ent]; + cs_why_t why = CS_WHY_PARTS; + if (name && unseen_part_declares(c, ent, name, arity, false, &why)) { + return res_ambiguous(why); + } + if (e->incomplete) { + return res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); /* a parse error hides members */ + } + return res_unres(e->open_any ? CBM_DOCLINK_REASON_EXTERNAL : CBM_DOCLINK_REASON_MISSING); +} + +/* A member the implementation does not show because a parse error hides it, + * looked up in the stubs that stand for this declaration (stub_user). true + * when a stub binds it: never an exact binding. */ +static bool stub_member(const cs_ctx_t *c, int ent, const cs_query_t *q, const cs_use_t *u, + cs_res_t *res) { + int stubs = c->ix->ents[ent].incomplete ? stub_user(c->ix, ent) : CS_NONE; + cs_res_t of_stubs = res_unres(CBM_DOCLINK_REASON_MISSING); + cs_use_t via = *u; + via.exact = false; + via.exact_type = false; + if (stubs < 0 || members_result(c, stubs, q, &via, &of_stubs) != CS_MB_RES || + of_stubs.st != CS_OK) { + return false; + } + of_stubs.exact = false; + *res = of_stubs; + return true; +} + +/* True when `ent` declares an instance constructor without parameters. */ +static bool declares_default_ctor(const cs_ctx_t *c, int ent) { + const cs_index_t *ix = c->ix; + cs_mspans_t sp = mref_spans(ix, ent, ix->ents[ent].name, SKIP_ONE, ""); + for (int s = 0; s < sp.n; s++) { + for (int i = sp.lo[s]; i < sp.hi[s]; i++) { + if (!mref_member(ix, &ix->mrefs[i])->is_static) { + return true; + } + } + } + return false; +} + +/* A constructor of `ent`. A type has the instance constructors its source + * writes (a primary constructor among them: the scope records it), and ones + * no source writes: the parameterless one of a struct, and of a class or + * record that writes none; a record's copy constructor. Those are declared + * and have no node. No constructor is inherited. */ +static cs_res_t ctor_result(const cs_ctx_t *c, int ent, const cs_use_t *u) { + const cs_entity_t *e = &c->ix->ents[ent]; + cs_query_t q = {.name = e->name, .arity = CS_ARITY_NONE, .ctors = true, .kind = 'c'}; + cs_res_t res = res_unres(CBM_DOCLINK_REASON_MISSING); + cs_mb_t st = members_result(c, ent, &q, u, &res); + bool value_type = e->kind == 's' || e->kind == 't'; + bool record = e->kind == 'r' || e->kind == 't'; + if (st == CS_MB_RES) { + /* a struct has its parameterless constructor beside the one it writes */ + bool second = + !u->has_params && value_type && res.st == CS_OK && !declares_default_ctor(c, ent); + return second ? res_ambiguous(CS_WHY_SCOPE) : res; + } + if (st == CS_MB_INVISIBLE) { + return res_unres(CBM_DOCLINK_REASON_TEST_ONLY); + } + if (e->incomplete) { + return res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); + } + bool supplied = value_type || (st == CS_MB_NONE && (e->kind == 'c' || e->kind == 'r')); + if (supplied && (!u->has_params || u->r->nparams == 0)) { + return res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); + } + if (record && u->has_params && u->r->nparams == SKIP_ONE && + strcmp(u->r->params[0], e->name) == 0) { + return res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); + } + return res_unres(CBM_DOCLINK_REASON_MISSING); +} + +/* The static constructor of `ent` (a doc ID's `#cctor`). */ +static cs_res_t static_ctor_result(const cs_ctx_t *c, int ent, bool exact) { + cs_query_t q = {.name = c->ix->ents[ent].name, + .arity = CS_ARITY_NONE, + .statics = true, + .ctors = true, + .kind = 'c'}; + cs_res_t res = res_unres(CBM_DOCLINK_REASON_MISSING); + cs_mb_t st = members_plain(c, ent, &q, exact, &res); + if (st == CS_MB_RES) { + return res; + } + return st == CS_MB_INVISIBLE ? res_unres(CBM_DOCLINK_REASON_TEST_ONLY) + : member_unfound(c, ent, NULL, CS_ARITY_NONE); +} + +/* ── Scope lookup ────────────────────────────────────────────────── */ + +/* The types nested in `ent` and the members it declares itself, by the + * query. A doc ID's member kind names a member, never a nested type. */ +static void level_entity(const cs_ctx_t *c, int ent, const cs_query_t *q, cs_found_t *fd) { + /* the shared trees' declaration seen without an assembly's own parts of + * the type: a name those parts declare is not told from what is here */ + cs_why_t why = CS_WHY_PARTS; + if (c->ix->ents[ent].used && + unseen_part_declares(c, ent, q->name, q->arity, q->types_only, &why)) { + fd->n += PAIR_LEN; + fd->why = why; + return; + } + if (!q->kind) { + add_types(c, false, ent, q->name, type_arity(q->arity), fd); + } + if (q->types_only) { + return; + } + int has = members_named(c, ent, q); + if (has == CS_HAS_VISIBLE) { + found_add(fd, 'M', ent); + } else if (has == CS_HAS_INVISIBLE) { + fd->invisible = true; + } +} + +/* What an alias was resolved to, as a candidate. */ +static void add_alias_target(const cs_ctx_t *c, const cs_using_t *u, cs_found_t *fd) { + if (u->ent >= 0) { + found_add_type(c, fd, u->ent); + fd->joined = fd->joined || u->joined; + } else if (u->ent == CS_AMBIGUOUS) { + fd->n += PAIR_LEN; + } else if (u->ns >= 0) { + found_add(fd, 'N', u->ns); + } else { + found_add(fd, 'X', CS_NONE); + } +} + +/* True once a level has two candidates: the lookup is ambiguous whatever the + * directives after them bring, so they are not asked (S5). Which candidate + * came first, and whether there was one, never depends on the ones skipped. */ +static bool found_decided(const cs_found_t *fd) { + return fd->n > SKIP_ONE; +} + +static void level_aliases(const cs_ctx_t *c, const cs_using_t *us, int n, + const cs_using_index_t *index, const cs_query_t *q, cs_found_t *fd) { + /* an alias names a type or a namespace: no candidate for `Name{T}` */ + if (q->arity > 0) { + return; + } + if (!index || !index->ready) { + for (int i = 0; i < n && !found_decided(fd); i++) { + cs_work(SKIP_ONE); + if (us[i].kind == 'a' && strcmp(us[i].alias, q->name) == 0) { + add_alias_target(c, &us[i], fd); + } + } + return; + } + size_t lo = 0; + size_t hi = index->naliases; + while (lo < hi) { + cs_work(SKIP_ONE); + size_t mid = lo + (hi - lo) / PAIR_LEN; + if (strcmp(index->aliases[mid].name, q->name) < 0) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + for (size_t i = lo; i < index->naliases && !found_decided(fd); i++) { + cs_work(SKIP_ONE); + if (strcmp(index->aliases[i].name, q->name)) { + break; + } + add_alias_target(c, &us[index->aliases[i].id], fd); + } +} + +/* The original resolver is shared by the full scan and candidate path. */ +static void level_using(const cs_ctx_t *c, const cs_using_t *u, const cs_query_t *q, + bool a_type_name, cs_found_t *fd) { + cs_work(SKIP_ONE); + if (u->kind == 'n' && u->ns >= 0 && a_type_name) { + /* a using brings a namespace's types, not the namespaces in it */ + add_types(c, true, u->ns, q->name, type_arity(q->arity), fd); + } else if (u->kind == 's' && u->ent >= 0) { + cs_query_t statics = *q; + statics.statics = true; + level_entity(c, u->ent, &statics, fd); + } +} + +static size_t name_scope_range(const cs_index_t *ix, const char *name, size_t *end) { + size_t lo = 0; + size_t hi = ix->nname_scopes; + while (lo < hi) { + cs_work(SKIP_ONE); + size_t mid = lo + (hi - lo) / PAIR_LEN; + if (strcmp(ix->name_scopes[mid].name, name) < 0) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + size_t first = lo; + hi = ix->nname_scopes; + while (lo < hi) { + cs_work(SKIP_ONE); + size_t mid = lo + (hi - lo) / PAIR_LEN; + if (strcmp(ix->name_scopes[mid].name, name) <= 0) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + *end = lo; + return first; +} + +/* IDs are accumulated before any result is added: if growth fails, a full + * scan can retry without duplicating a partially resolved candidate set. */ +enum { CS_CANDIDATE_STACK = 32 }; +typedef struct { + int local[CS_CANDIDATE_STACK]; + int *ids; + size_t n; + size_t cap; +} cs_candidate_ids_t; + +static bool candidate_add(cs_candidate_ids_t *ids, int id) { + if (ids->n == ids->cap) { + if (ids->cap > SIZE_MAX / PAIR_LEN / sizeof(int)) { + return false; + } + size_t cap = ids->cap * PAIR_LEN; + int *grown = NULL; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + bool fail = atomic_exchange(&cs_fail_candidate_alloc, false); + if (fail) { + atomic_store(&cs_candidate_alloc_failed, true); + } else +#endif + { + grown = (int *)cbm_alloc(CBM_MEM_CLASS_OTHER, cap * sizeof(int)); + } + if (!grown) { + return false; + } + memcpy(grown, ids->ids, ids->n * sizeof(int)); + cs_work(ids->n); + if (ids->ids != ids->local) { + cbm_free(CBM_MEM_CLASS_OTHER, ids->ids); + } + ids->ids = grown; + ids->cap = cap; + } + ids->ids[ids->n++] = id; + return true; +} + +static bool candidate_scope(const cs_using_id_t *rows, size_t n, int scope, + cs_candidate_ids_t *ids) { + size_t lo = 0; + size_t hi = n; + while (lo < hi) { + cs_work(SKIP_ONE); + size_t mid = lo + (hi - lo) / PAIR_LEN; + if (rows[mid].scope < scope) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + for (size_t i = lo; i < n; i++) { + cs_work(SKIP_ONE); + if (rows[i].scope != scope) { + break; + } + if (!candidate_add(ids, rows[i].id)) { + return false; + } + } + return true; +} + +static int candidate_id_cmp(const void *a, const void *b) { + cs_work(SKIP_ONE); + return int_cmp(a, b); +} + +/* The directives of a using step that give a query anything, in order (R1): + * noted while the step is walked, for the unit memo. */ +typedef struct { + int *ids; + int n; + int cap; + bool failed; /* memory ran out: the list is not whole */ +} cs_active_rec_t; + +static void active_add(cs_active_rec_t *rec, int id) { + if (rec->n == rec->cap) { + int cap = rec->cap ? rec->cap * PAIR_LEN : CBM_SZ_16; + int *grown = (int *)cbm_alloc(CBM_MEM_CLASS_OTHER, (size_t)cap * sizeof(int)); + if (!grown) { + rec->failed = true; + return; + } + if (rec->n > 0) { + memcpy(grown, rec->ids, (size_t)rec->n * sizeof(int)); + } + cbm_free(CBM_MEM_CLASS_OTHER, rec->ids); + rec->ids = grown; + rec->cap = cap; + } + rec->ids[rec->n++] = id; +} + +/* Directive `id` of a step, for the query. With `rec`, it is noted when it + * gives the query anything at all: one that leaves an empty step empty adds + * no candidate and sets no flag whatever the step holds, so leaving it out + * changes no lookup. */ +static void using_visit(const cs_ctx_t *c, const cs_using_t *us, int id, const cs_query_t *q, + bool a_type_name, cs_found_t *fd, cs_active_rec_t *rec) { + if (rec) { + cs_found_t probe = {0}; + level_using(c, &us[id], q, a_type_name, &probe); + if (probe.n == 0 && !probe.invisible && !probe.joined && probe.why == CS_WHY_SCOPE) { + return; + } + active_add(rec, id); + } + level_using(c, &us[id], q, a_type_name, fd); +} + +/* False when none of a step's n directives can give the query anything: + * there are none, or the index shows no static candidates and no namespace + * that can contribute this name. *a_type_name: the name is some type's. */ +static bool usings_may_give(const cs_ctx_t *c, int n, const cs_using_index_t *index, + const cs_query_t *q, bool *a_type_name) { + if (!n) { + return false; + } + *a_type_name = cbm_ht_get(c->ix->type_names, q->name) != NULL; + bool indexed = index && index->ready && c->ix->name_scopes_ready; + return !(indexed && !index->nentities && (!index->nnamespaces || !*a_type_name)); +} + +static void level_usings(const cs_ctx_t *c, const cs_using_t *us, int n, + const cs_using_index_t *index, const cs_query_t *q, cs_found_t *fd, + cs_active_rec_t *rec) { + bool a_type_name = false; + if (!usings_may_give(c, n, index, q, &a_type_name)) { + return; + } + bool indexed = index && index->ready && c->ix->name_scopes_ready; + size_t hi = 0; + size_t lo = indexed ? name_scope_range(c->ix, q->name, &hi) : 0; + /* A common name in a large repository must not make a small scope walk + * more postings than it has directives. The baseline scan stays cheap. */ + if (!indexed || hi - lo >= (size_t)n) { + for (int i = 0; i < n && !found_decided(fd); i++) { + using_visit(c, us, i, q, a_type_name, fd, rec); + } + return; + } + cs_candidate_ids_t ids = {.cap = CS_CANDIDATE_STACK}; + ids.ids = ids.local; + bool complete = true; + for (size_t i = lo; complete && i < hi; i++) { + cs_work(SKIP_ONE); + const cs_name_scope_t *row = &c->ix->name_scopes[i]; + complete = row->top + ? candidate_scope(index->namespaces, index->nnamespaces, row->scope, &ids) + : candidate_scope(index->entities, index->nentities, row->scope, &ids); + } + if (complete) { + if (ids.n > 1) { + qsort(ids.ids, ids.n, sizeof(int), candidate_id_cmp); + } + for (size_t i = 0; i < ids.n && !found_decided(fd); i++) { + cs_work(SKIP_ONE); + /* Own and twin postings may select the SAME directive twice. + * Distinct directive IDs must still be resolved separately. */ + if (!i || ids.ids[i] != ids.ids[i - SKIP_ONE]) { + using_visit(c, us, ids.ids[i], q, a_type_name, fd, rec); + } + } + } + if (ids.ids != ids.local) { + cbm_free(CBM_MEM_CLASS_OTHER, ids.ids); + } + if (!complete) { + for (int i = 0; i < n && !found_decided(fd); i++) { + using_visit(c, us, i, q, a_type_name, fd, rec); + } + } +} + +/* The using step of a lookup -- what the directives of one region, and at + * the global namespace the unit's, give one query -- remembered for one pass + * over a file's references (or one pass of the build): a name looked up + * again in the same region costs one probe, not the directives (S5). The key + * holds everything the step reads: the file, the region and unit asked, the + * context's visibility (product code, unit, assembly) and the query. A memo + * that ran out of memory stops remembering; the lookups stay exact. */ +struct cs_memo { + CBMHashTable *steps; /* key -> cs_found_t in `arena` */ + CBMArena arena; + bool off; +}; + +enum { CS_MEMO_KEY = CS_NAME_BUF + CBM_SZ_128 }; + +/* Holds no memory until the first step is remembered: most files ask few + * using steps, and a memo is made for every file that has references. */ +static void memo_init(cs_memo_t *m) { + memset(m, 0, sizeof(*m)); + cbm_arena_init_lazy(&m->arena, CBM_ARENA_APPEND_BLOCK); +} + +static void memo_destroy(cs_memo_t *m) { + cbm_ht_free(m->steps); + cbm_arena_destroy(&m->arena); + memset(m, 0, sizeof(*m)); +} + +/* The key of one using step; false when it does not fit (not remembered). */ +static bool memo_key(char *buf, size_t cap, const cs_ctx_t *c, int region, int unit, + const cs_query_t *q) { + int n = snprintf(buf, cap, "%d|%d|%d|%d|%d|%d|%d|%d%d%d%c|%s", c->file, region, unit, c->unit, + c->group, (int)c->prod, q->arity, (int)q->types_only, (int)q->statics, + (int)q->ctors, q->kind ? q->kind : '-', q->name); + return n > 0 && (size_t)n < cap; +} + +static bool memo_get(const cs_memo_t *m, const char *key, cs_found_t *out) { + cs_work(SKIP_ONE); + const cs_found_t *hit = m->steps ? (const cs_found_t *)cbm_ht_get(m->steps, key) : NULL; + if (hit) { + *out = *hit; + } + return hit != NULL; +} + +static void memo_put(cs_memo_t *m, const char *key, const cs_found_t *step) { + if (!m->steps) { + m->steps = cbm_ht_create(CBM_SZ_16); + if (!m->steps) { + m->off = true; + return; + } + } + char *k = cbm_arena_strdup(&m->arena, key); + cs_found_t *v = (cs_found_t *)cbm_arena_alloc(&m->arena, sizeof(*v)); + if (!k || !v) { + m->off = true; + return; + } + *v = *step; + cbm_ht_set(m->steps, k, v); + m->off = cbm_ht_get(m->steps, k) != v; /* an insert that did not take */ +} + +/* The unit part of a using step (R1): which of a unit's directives give a + * query anything, in order, up to where they alone decide it. What a + * directive gives reads only the context's unit, assembly and product flag + * (level_using), so the list is the same for every file of the unit that + * asks the query: it is found once per run, shared by the resolve workers + * (guarded), and each file then asks only those directives after its own -- + * stopping where the walk of all of them would, which is never later than + * where they alone are decided. Each list is found once: by the first + * worker that asks for it, while the others that ask for the same key hold + * on until it is there (a step's cost never depends on the scheduling). A + * memo that ran out of memory stops remembering; the lookups stay exact. */ +typedef struct cs_unit_step { + cbm_mutex_t mu; /* held by the worker that finds the list, until it is there */ + int *ids; /* the directives, in order (memory-core block, or NULL) */ + int n; + bool failed; /* memory ran out finding it: askers walk all directives */ + struct cs_unit_step *next; /* every step, for the memo's release */ +} cs_unit_step_t; + +typedef struct cs_unit_memo { + cbm_mutex_t mu; /* guards steps, arena, all and off */ + CBMHashTable *steps; /* key -> cs_unit_step_t in `arena` */ + CBMArena arena; + cs_unit_step_t *all; + bool off; +} cs_unit_memo_t; + +static cs_unit_memo_t *unit_memo_new(void) { + cs_unit_memo_t *m = (cs_unit_memo_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*m)); + if (!m) { + return NULL; + } + cbm_mutex_init(&m->mu); + cbm_arena_init_lazy(&m->arena, CBM_ARENA_APPEND_BLOCK); + return m; +} + +static void unit_memo_free(cs_unit_memo_t *m) { + if (!m) { + return; + } + for (cs_unit_step_t *e = m->all; e; e = e->next) { + cbm_mutex_destroy(&e->mu); + cbm_free(CBM_MEM_CLASS_OTHER, e->ids); + } + cbm_ht_free(m->steps); + cbm_arena_destroy(&m->arena); + cbm_mutex_destroy(&m->mu); + cbm_free(CBM_MEM_CLASS_OTHER, m); +} + +/* The key of one unit step: everything it reads. false when it does not fit. */ +static bool unit_key(char *buf, size_t cap, const cs_ctx_t *c, const cs_query_t *q) { + int n = snprintf(buf, cap, "%d|%d|%d|%d|%d%d%d%c|%s", c->unit, c->group, (int)c->prod, q->arity, + (int)q->types_only, (int)q->statics, (int)q->ctors, q->kind ? q->kind : '-', + q->name); + return n > 0 && (size_t)n < cap; +} + +/* The step under `key` (the caller holds the memo's lock); a new one is + * added with its lock held, and *mine set: the caller finds its list. NULL + * when memory ran out. */ +static cs_unit_step_t *unit_step_at(cs_unit_memo_t *m, const char *key, bool *mine) { + *mine = false; + if (!m->steps) { + m->steps = cbm_ht_create(CBM_SZ_64); + if (!m->steps) { + return NULL; + } + } + cs_unit_step_t *had = (cs_unit_step_t *)cbm_ht_get(m->steps, key); + if (had) { + return had; + } + char *k = cbm_arena_strdup(&m->arena, key); + cs_unit_step_t *e = (cs_unit_step_t *)cbm_arena_alloc(&m->arena, sizeof(*e)); + if (!k || !e) { + return NULL; + } + memset(e, 0, sizeof(*e)); + cbm_mutex_init(&e->mu); + cbm_mutex_lock(&e->mu); + e->next = m->all; + m->all = e; + cbm_ht_set(m->steps, k, e); + if (cbm_ht_get(m->steps, k) != e) { + e->failed = true; /* an insert that did not take: kept for release only */ + cbm_mutex_unlock(&e->mu); + return NULL; + } + *mine = true; + return e; +} + +/* What the unit's directives give the query after the region's (`fd`). */ +static void unit_usings(const cs_ctx_t *c, const cs_unit_t *unit, const cs_query_t *q, + cs_found_t *fd) { + bool a_type_name = false; + if (!usings_may_give(c, unit->nusings, &unit->using_index, q, &a_type_name)) { + return; /* nothing to ask, nothing to remember */ + } + cs_unit_memo_t *m = c->ix->unit_memo; + char key[CS_MEMO_KEY]; + cs_unit_step_t *e = NULL; + bool mine = false; + if (m && unit_key(key, sizeof(key), c, q)) { + cs_work(SKIP_ONE); + cbm_mutex_lock(&m->mu); + e = m->off ? NULL : unit_step_at(m, key, &mine); + m->off = m->off || !e; + cbm_mutex_unlock(&m->mu); + } + if (e && mine) { + cs_active_rec_t rec = {0}; + cs_found_t alone = {0}; + level_usings(c, unit->usings, unit->nusings, &unit->using_index, q, &alone, &rec); + e->failed = rec.failed; + e->ids = rec.failed ? NULL : rec.ids; + e->n = rec.failed ? 0 : rec.n; + if (rec.failed) { + cbm_free(CBM_MEM_CLASS_OTHER, rec.ids); + } + cbm_mutex_unlock(&e->mu); + } else if (e) { + cbm_mutex_lock(&e->mu); /* the list is there once its finder lets go */ + cbm_mutex_unlock(&e->mu); + } + if (!e || e->failed) { + level_usings(c, unit->usings, unit->nusings, &unit->using_index, q, fd, NULL); + return; + } + for (int i = 0; i < e->n && !found_decided(fd); i++) { + level_using(c, &unit->usings[e->ids[i]], q, a_type_name, fd); + } +} + +/* What the directives of this scope level give the query: the region's own + * (`us`, when asked) and, at the global namespace, the unit's. Asked once per + * key when the context has a memo. */ +static void using_step(const cs_ctx_t *c, int region, const cs_using_t *us, int nus, + const cs_using_index_t *usi, const cs_unit_t *unit, const cs_query_t *q, + cs_found_t *step) { + char key[CS_MEMO_KEY]; + bool keyed = c->memo && !c->memo->off && + memo_key(key, sizeof(key), c, region, unit ? c->unit : CS_NONE, q); + if (keyed && memo_get(c->memo, key, step)) { + return; + } + level_usings(c, us, nus, usi, q, step, NULL); + if (unit && !found_decided(step)) { + unit_usings(c, unit, q, step); + } + if (keyed) { + memo_put(c->memo, key, step); + } +} + +/* True when the lookup is settled by what the last level added. A level that + * had only invisible candidates does not bind: the lookup goes on, and + * remembers. */ +static bool settled(cs_found_t *fd, bool *invisible) { + *invisible = *invisible || fd->invisible; + fd->invisible = false; + return fd->n > 0; +} + +/* Look a simple name up the way the compiler does in a cref: the documented + * method's type parameters; for every enclosing type its type parameters and + * what it declares itself; for every enclosing namespace its types and + * namespaces, and where a namespace declaration of the file stands, that + * declaration's aliases and then its usings. The first level that has the + * name decides. One step per enclosing type and per enclosing namespace: the + * scanner bounds both nestings. *statics: the name came through a `using + * static`. */ +static void lookup(const cs_ctx_t *c, const cs_query_t *q, cs_found_t *fd, bool *statics) { + const cs_index_t *ix = c->ix; + const cs_file_t *f = c->f; + size_t nl = strlen(q->name); + bool invisible = false; + memset(fd, 0, sizeof(*fd)); + *statics = false; + if (q->arity <= 0 && c->member >= 0 && + tparam_find(f, f->ntypes + c->member, q->name, nl) >= 0) { + found_add(fd, 'L', CS_NONE); + return; + } + for (int t = c->type; t >= 0; t = f->types[t].outer) { + cs_work(SKIP_ONE); + if (q->arity <= 0 && tparam_find(f, t, q->name, nl) >= 0) { + found_add(fd, 'L', CS_NONE); + return; + } + level_entity(c, f->types[t].entity, q, fd); + if (settled(fd, &invisible)) { + return; + } + } + int reg = c->region; + for (int ns = f->regions[reg].ns;; ns = ix->nss[ns].parent) { + cs_work(SKIP_ONE); + add_types(c, true, ns, q->name, type_arity(q->arity), fd); + int child = q->arity > 0 ? CS_NONE : ns_in_view(c, ns, q->name, nl); + if (child >= 0) { + found_add(fd, 'N', child); + } + /* a declaration of this namespace stands here: its directives are + * asked, unless they are the peers of the directive being resolved */ + bool here = reg >= 0 && f->regions[reg].ns == ns; + bool asked = here && reg != c->skip_region; + const cs_using_t *us = asked ? f->usings + f->regions[reg].u_lo : NULL; + int nus = asked ? f->regions[reg].u_hi - f->regions[reg].u_lo : 0; + const cs_using_index_t *usi = asked ? &f->regions[reg].using_index : NULL; + const cs_unit_t *unit = (ns == 0 && c->unit >= 0) ? &ix->units[c->unit] : NULL; + if (fd->n > 0) { + /* C# rejects an alias beside a type or namespace of its name */ + cs_found_t alias = {0}; + level_aliases(c, us, nus, usi, q, &alias); + if (unit) { + level_aliases(c, unit->usings, unit->nusings, &unit->using_index, q, &alias); + } + fd->n += alias.n > 0; + fd->in_namespace = true; + } + if (settled(fd, &invisible)) { + return; + } + level_aliases(c, us, nus, usi, q, fd); + if (unit) { + level_aliases(c, unit->usings, unit->nusings, &unit->using_index, q, fd); + } + if (settled(fd, &invisible)) { + fd->exact = true; + return; + } + if (nus > 0 || unit) { + /* Nothing is found yet (settled), and a step only adds: the step + * is asked on its own and its candidates are the level's. */ + cs_found_t step = {0}; + using_step(c, asked ? reg : CS_NONE, us, nus, usi, unit, q, &step); + fd->first = step.first; + fd->n = step.n; + fd->invisible = step.invisible; + fd->why = step.why; + fd->joined = fd->joined || step.joined; + } + if (settled(fd, &invisible)) { + *statics = fd->first.kind == 'M'; + return; + } + if (here) { + reg = f->regions[reg].parent; + } + if (ns == 0) { + break; + } + } + fd->invisible = invisible; +} + +/* ── Paths ───────────────────────────────────────────────────────── */ + +typedef enum { + CS_TP_TYPE = 0, /* the path names the type `ent` */ + CS_TP_NS, /* ... the namespace `ns` */ + CS_TP_PARTIAL, /* the type `ent` was reached; it has no type by the next segment */ + CS_TP_NSFAIL, /* the namespace `ns` was reached; it has nothing by the next segment */ + CS_TP_RES, /* `res` is the answer */ +} cs_tp_kind_t; + +typedef struct { + cs_tp_kind_t st; + int ent; + int ns; + bool exact; /* the path itself is a qualified name (or an alias, or absolute) */ + bool in_ns; /* its first segment is of an enclosing namespace (or absolute): one + * more segment makes a qualified name of it */ + const cs_seg_t *next; /* CS_TP_PARTIAL: the segment the type has nothing by */ + bool joined; /* a segment is one type only because a contract is joined to its + * implementation: nothing bound through the path is exact */ + cs_res_t res; +} cs_tp_t; + +static cs_tp_t tp_res(cs_res_t r) { + return (cs_tp_t){.st = CS_TP_RES, .res = r}; +} + +/* Follow segs[from, n) from where `cur` stands: each is a member of what the + * one before named -- a namespace or type of a namespace, a type nested in a + * type. There is no second try from anywhere else. */ +static cs_tp_t path_from(const cs_ctx_t *c, cs_tp_t cur, const cs_seg_t *segs, int from, int n) { + for (int i = from; i < n; i++) { + cs_found_t fd = {0}; + if (cur.st == CS_TP_NS) { + int child = segs[i].arity > 0 + ? CS_NONE + : ns_in_view(c, cur.ns, segs[i].name, strlen(segs[i].name)); + add_types(c, true, cur.ns, segs[i].name, type_arity(segs[i].arity), &fd); + if (child >= 0 && fd.n == 0 && !fd.invisible) { + cur.ns = child; + continue; + } + fd.n += child >= 0; + } else { + add_types(c, false, cur.ent, segs[i].name, type_arity(segs[i].arity), &fd); + } + if (fd.n > SKIP_ONE) { + return tp_res(found_ambiguous(&fd)); + } + if (fd.n == 0) { + if (fd.invisible) { + return tp_res(res_unres(CBM_DOCLINK_REASON_TEST_ONLY)); + } + cur.st = cur.st == CS_TP_NS ? CS_TP_NSFAIL : CS_TP_PARTIAL; + cur.next = &segs[i]; + return cur; + } + cur.st = CS_TP_TYPE; + cur.ent = fd.first.id; + if (fd.joined) { + cur.joined = true; + cur.exact = false; + cur.in_ns = false; + } + } + return cur; +} + +/* An open scope: a using in view names a namespace or type the repository + * does not declare, or the project's global usings could not be evaluated. + * A name found nowhere may come from there. */ +static bool scope_open(const cs_ctx_t *c) { + if (c->unit >= 0 && c->ix->units[c->unit].open) { + return true; + } + return c->f->regions[c->region].open; +} + +/* Why a simple name was found at no scope level. */ +static cs_res_t simple_unfound(const cs_ctx_t *c) { + const cs_index_t *ix = c->ix; + if (scope_open(c)) { + return res_unres(CBM_DOCLINK_REASON_EXTERNAL); + } + bool hidden = false; + for (int t = c->type; t >= 0; t = c->f->types[t].outer) { + const cs_entity_t *e = &ix->ents[c->f->types[t].entity]; + if (e->open_any) { + return res_unres(CBM_DOCLINK_REASON_EXTERNAL); /* a base outside the repository */ + } + hidden = hidden || e->incomplete; + } + /* An enclosing type with members a parse error hides: the name may be + * one of them. */ + return res_unres(hidden ? CBM_DOCLINK_REASON_GRAPH_GAP : CBM_DOCLINK_REASON_MISSING); +} + +/* The path segs[0, n): a type, a namespace, or how far it got. The first + * segment is a simple name that names a type or a namespace in scope; + * `global::` and a doc ID start at the global namespace instead. `qualifier`: + * the path qualifies a name that follows it. */ +static cs_tp_t resolve_path(const cs_ctx_t *c, const cs_seg_t *segs, int n, bool qualifier) { + if (c->glob) { + return path_from(c, (cs_tp_t){.st = CS_TP_NS, .ns = 0, .exact = true, .in_ns = true}, segs, + 0, n); + } + if (n <= 0) { + return tp_res(res_unres(CBM_DOCLINK_REASON_UNPARSEABLE)); + } + cs_query_t q = {.name = segs[0].name, .arity = segs[0].arity, .types_only = true}; + cs_found_t fd; + bool statics = false; + lookup(c, &q, &fd, &statics); + if (fd.n > SKIP_ONE) { + return tp_res(found_ambiguous(&fd)); + } + if (fd.n == 0) { + if (fd.invisible) { + return tp_res(res_unres(CBM_DOCLINK_REASON_TEST_ONLY)); + } + if (n == SKIP_ONE && !qualifier) { + return tp_res(simple_unfound(c)); + } + /* a qualifier that is no namespace and no type in scope: a type of + * the repository that is not in scope here, or a namespace the + * repository does not declare */ + return tp_res(res_unres(cbm_ht_get(c->ix->type_names, segs[0].name) + ? CBM_DOCLINK_REASON_MISSING + : CBM_DOCLINK_REASON_EXTERNAL)); + } + /* exact: a name through an alias, and a qualified name -- two segments + * or more, the first found among an enclosing namespace's own types and + * namespaces. A name found by its simple name alone, or through a using, + * is what the scope makes of it. */ + bool exact = (fd.exact || (n > SKIP_ONE && fd.in_namespace)) && !fd.joined; + bool in_ns = (fd.exact || fd.in_namespace) && !fd.joined; + switch (fd.first.kind) { + case 'T': + return path_from(c, + (cs_tp_t){.st = CS_TP_TYPE, + .ent = fd.first.id, + .exact = exact, + .in_ns = in_ns, + .joined = fd.joined}, + segs, SKIP_ONE, n); + case 'N': + return path_from( + c, (cs_tp_t){.st = CS_TP_NS, .ns = fd.first.id, .exact = exact, .in_ns = in_ns}, segs, + SKIP_ONE, n); + case 'L': + /* a type parameter: nothing is a member of one */ + return tp_res(n == SKIP_ONE ? (cs_res_t){.st = CS_LOCAL} + : res_unres(CBM_DOCLINK_REASON_MISSING)); + default: + return tp_res(res_unres(CBM_DOCLINK_REASON_EXTERNAL)); + } +} + +/* A namespace as a reference's target: declared, and without a node. */ +static cs_res_t namespace_result(void) { + return res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); +} + +/* The reason for a path that ended at a namespace without the next name: a + * namespace the repository declares has no such type (missing); one it only + * has namespaces under, and the global one, hold nothing of the repository + * (the name is outside) -- and so is what a standard-library namespace lacks, + * declared or not. */ +static cs_res_t namespace_unfound(const cs_ctx_t *c, int ns) { + const cs_ns_t *n = &c->ix->nss[ns]; + return res_unres((n->declared && !n->standard) ? CBM_DOCLINK_REASON_MISSING + : CBM_DOCLINK_REASON_EXTERNAL); +} + +/* The reason for a path that did not come to a type or a namespace. */ +static cs_res_t path_unfound(const cs_ctx_t *c, const cs_tp_t *tp) { + if (tp->st == CS_TP_RES) { + return tp->res; + } + if (tp->st == CS_TP_NSFAIL) { + return namespace_unfound(c, tp->ns); + } + return member_unfound(c, tp->ent, tp->next ? tp->next->name : NULL, + tp->next ? tp->next->arity : CS_ARITY_NONE); +} + +/* ── Members of a type ───────────────────────────────────────────── */ + +/* The implicit roots a type of this kind derives from without naming them, + * as far as the repository declares them. */ +static int implicit_roots(const cs_ctx_t *c, char kind, int *out) { + static const char *const enum_roots[] = {"Enum", "ValueType", "Object", NULL}; + static const char *const value_roots[] = {"ValueType", "Object", NULL}; + static const char *const object_root[] = {"Object", NULL}; + static const char system_ns[] = "System"; + const char *const *roots = object_root; + if (kind == 'e') { + roots = enum_roots; + } else if (kind == 's' || kind == 't') { + roots = value_roots; + } else if (kind == 'i' || kind == 'd') { + return 0; + } + int sys = ns_find(c->ix, 0, system_ns, sizeof(system_ns) - SKIP_ONE); + int n = 0; + for (int i = 0; sys >= 0 && roots[i]; i++) { + cs_found_t fd = {0}; + add_types(c, true, sys, roots[i], 0, &fd); + if (fd.n == SKIP_ONE) { + out[n++] = fd.first.id; + } + } + return n; +} + +/* NOT the compiler's rule (see CS_BIND_INHERITED): where the compiler's + * lookup has bound nothing, the member `seg` of the nearest supertype of + * `ent` that has one. false when the index does not bind inherited members, + * and when no supertype in view has the name. */ +static bool inherited_result(const cs_ctx_t *c, int ent, const cs_seg_t *seg, const cs_use_t *u, + char kind, cs_res_t *res) { + const cs_index_t *ix = c->ix; + if (!ix->bind_inherited) { + return false; + } + int sup[CS_MAX_SUPERS + CS_MAX_ROOTS]; + bool more = false; + int n = supers_of(ix, ent, sup, &more); + n += implicit_roots(c, ix->ents[ent].kind, sup + n); + cs_query_t q = {.name = seg->name, .arity = seg->arity, .kind = kind}; + for (int i = 0; i < n; i++) { + cs_work(SKIP_ONE); + cs_found_t fd = {0}; + level_entity(c, sup[i], &q, &fd); + if (fd.n > SKIP_ONE) { + *res = found_ambiguous(&fd); + return true; + } + if (fd.n == SKIP_ONE && fd.first.kind == 'T' && !u->has_params) { + *res = type_result(c, fd.first.id, u->exact); + return true; + } + cs_mb_t st = fd.n == SKIP_ONE && fd.first.kind == 'M' + ? members_result(c, sup[i], &q, u, res) + : CS_MB_NONE; + if (st == CS_MB_RES) { + return true; + } + if (st == CS_MB_INVISIBLE || fd.invisible) { + *res = res_unres(CBM_DOCLINK_REASON_TEST_ONLY); + return true; + } + } + if (more) { + *res = res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); /* a hierarchy past CS_MAX_SUPERS */ + return true; + } + return false; +} + +/* A property or an event by its accessor's name (get_X, set_X; add_X, + * remove_X). The compiler binds the accessor; the graph keeps an accessor in + * its property or event, so that is the node. An indexer's accessors + * (get_Item, set_Item) have none. */ +static bool accessor_result(const cs_ctx_t *c, int ent, const cs_seg_t *seg, bool exact, + cs_res_t *res) { + static const struct { + const char *prefix; + char kind; + } acc[] = {{"get_", 'p'}, {"set_", 'p'}, {"add_", 'e'}, {"remove_", 'e'}}; + for (size_t a = 0; seg->arity <= 0 && a < sizeof(acc) / sizeof(acc[0]); a++) { + size_t al = strlen(acc[a].prefix); + if (strncmp(seg->name, acc[a].prefix, al) != 0 || !seg->name[al]) { + continue; + } + cs_query_t q = {.name = seg->name + al, .arity = CS_ARITY_NONE, .kind = acc[a].kind}; + cs_mb_t st = members_plain(c, ent, &q, exact, res); + if (st == CS_MB_INVISIBLE) { + *res = res_unres(CBM_DOCLINK_REASON_TEST_ONLY); + } else if (st != CS_MB_RES && acc[a].kind == 'p' && strcmp(q.name, "Item") == 0 && + special_declared(c, ent, "this")) { + *res = res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); + } else if (st != CS_MB_RES) { + continue; + } + return true; + } + return false; +} + +/* `seg` as a member of the type `ent`, looked up in `ent` itself: a member + * it declares, a type nested in it (with a parameter list: that type's + * constructor), its own name (`Foo.Foo`: its constructor), a property or + * event by its accessor's name. `kind` is the member kind a doc ID names. */ +static cs_res_t member_in(const cs_ctx_t *c, int ent, const cs_seg_t *seg, const cs_use_t *u, + char kind) { + const cs_entity_t *e = &c->ix->ents[ent]; + cs_res_t res = res_unres(CBM_DOCLINK_REASON_MISSING); + cs_use_t ctor_use = *u; + ctor_use.where.member_seg = CS_NONE; + if (strcmp(seg->name, "#ctor") == 0) { + return ctor_result(c, ent, &ctor_use); + } + if (strcmp(seg->name, "#cctor") == 0) { + return static_ctor_result(c, ent, u->exact); + } + cs_query_t q = {.name = seg->name, .arity = seg->arity, .kind = kind}; + cs_found_t fd = {0}; + level_entity(c, ent, &q, &fd); + if (fd.n > SKIP_ONE) { + return found_ambiguous(&fd); + } + if (fd.n == SKIP_ONE && fd.first.kind == 'T') { + if (!u->has_params) { + return type_result(c, fd.first.id, u->exact_type && !fd.joined); + } + ctor_use.where.type_seg = u->where.member_seg; /* the nested type is the last segment */ + ctor_use.exact = u->exact_type && !fd.joined; + return ctor_result(c, fd.first.id, &ctor_use); + } + if (fd.n == SKIP_ONE) { + cs_mb_t st = members_result(c, ent, &q, u, &res); + if (st == CS_MB_RES) { + return res; + } + if (st == CS_MB_INVISIBLE) { + return res_unres(CBM_DOCLINK_REASON_TEST_ONLY); + } + /* it has the name, and nothing of it takes the written parameters */ + } else { + if (fd.invisible) { + return res_unres(CBM_DOCLINK_REASON_TEST_ONLY); + } + /* `Foo.Foo`: the constructor -- unless it is a bare `Foo{T}.Foo` + * written on that generic type itself, which the compiler leaves + * unbound */ + bool own_name = !kind && seg->arity <= 0 && strcmp(seg->name, e->name) == 0; + bool on_type = c->type >= 0 && c->f->types[c->type].entity == ent; + if (own_name && (u->has_params || e->arity == 0 || !on_type)) { + return ctor_result(c, ent, &ctor_use); + } + if (!kind && accessor_result(c, ent, seg, u->exact, &res)) { + return res; + } + if (u->r->maybe_indexer && (!kind || kind == 'p') && special_declared(c, ent, "this")) { + return res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); /* `Item(int)`: the indexer */ + } + } + if (stub_member(c, ent, &q, u, &res) || inherited_result(c, ent, seg, u, kind, &res)) { + return res; + } + return member_unfound(c, ent, seg->name, seg->arity); +} + +/* ── Simple names, qualified names, operators, doc IDs ───────────── */ + +/* A simple name no scope level has. */ +static cs_res_t simple_fallback(const cs_ctx_t *c, const cs_ref_t *r, const cs_use_t *u) { + const cs_file_t *f = c->f; + const cs_seg_t *seg = &r->segs[0]; + cs_res_t res = res_unres(CBM_DOCLINK_REASON_MISSING); + /* `Foo(int)` written in a generic `Foo`: no type `Foo` is in scope, + * and the compiler takes the constructor of the type the reference + * stands in */ + if (r->has_params && seg->arity <= 0 && c->type >= 0 && + strcmp(f->types[c->type].name, seg->name) == 0) { + cs_use_t ctor_use = *u; + ctor_use.where = (cs_where_t){.type_seg = CS_NONE, .member_seg = CS_NONE}; + return ctor_result(c, f->types[c->type].entity, &ctor_use); + } + cs_query_t q = {.name = seg->name, .arity = seg->arity}; + for (int t = c->type; t >= 0; t = f->types[t].outer) { + int ent = f->types[t].entity; + if (accessor_result(c, ent, seg, u->exact, &res)) { + return res; + } + if (r->maybe_indexer && special_declared(c, ent, "this")) { + return res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); /* `Item(int)`: the indexer */ + } + if (stub_member(c, ent, &q, u, &res) || inherited_result(c, ent, seg, u, 0, &res)) { + return res; + } + } + return simple_unfound(c); +} + +static cs_res_t resolve_simple(const cs_ctx_t *c, const cs_ref_t *r) { + const cs_seg_t *seg = &r->segs[0]; + cs_query_t q = {.name = seg->name, .arity = seg->arity}; + cs_found_t fd; + bool statics = false; + lookup(c, &q, &fd, &statics); + if (fd.n > SKIP_ONE) { + return found_ambiguous(&fd); + } + q.statics = statics; /* through a `using static`: its static members are meant */ + bool exact = fd.exact && !fd.joined; + cs_use_t u = {.r = r, + .has_params = r->has_params, + .where = {.type_seg = CS_NONE, .member_seg = 0}, + .exact = exact}; + if (fd.n == 0) { + return fd.invisible ? res_unres(CBM_DOCLINK_REASON_TEST_ONLY) : simple_fallback(c, r, &u); + } + cs_res_t res = res_unres(CBM_DOCLINK_REASON_MISSING); + switch (fd.first.kind) { + case 'L': + /* a type parameter of the definition: no reference to code elsewhere */ + return r->has_params ? res : (cs_res_t){.st = CS_LOCAL}; + case 'N': + return r->has_params ? res : namespace_result(); + case 'T': + if (!r->has_params) { + return type_result(c, fd.first.id, exact); + } + if (fd.exact) { + return res; /* the compiler matches no parameter list against an alias */ + } + /* a parameter list on a type's name: its constructor */ + u.where = (cs_where_t){.type_seg = 0, .member_seg = CS_NONE}; + return ctor_result(c, fd.first.id, &u); + case 'M': { + cs_mb_t st = members_result(c, fd.first.id, &q, &u, &res); + if (st == CS_MB_RES) { + return res; + } + if (st == CS_MB_INVISIBLE) { + return res_unres(CBM_DOCLINK_REASON_TEST_ONLY); + } + /* the level that has the name decides: nothing further out is asked */ + if (stub_member(c, fd.first.id, &q, &u, &res) || + inherited_result(c, fd.first.id, seg, &u, 0, &res)) { + return res; + } + return member_unfound(c, fd.first.id, seg->name, seg->arity); + } + default: + return res_unres(CBM_DOCLINK_REASON_EXTERNAL); + } +} + +/* `Path.Last`: the last segment is a member of what the path before it + * names. Of a type: member_in. Of a namespace: a type (with a parameter + * list: its constructor) or a namespace. */ +static cs_res_t resolve_qualified(const cs_ctx_t *c, const cs_ref_t *r, char kind, + bool has_params) { + int n = r->nsegs; + const cs_seg_t *last = &r->segs[n - SKIP_ONE]; + cs_tp_t tp = resolve_path(c, r->segs, n - SKIP_ONE, true); + cs_use_t u = {.r = r, + .has_params = has_params, + .where = {.type_seg = n - PAIR_LEN, .member_seg = n - SKIP_ONE}, + .exact = tp.exact, + .exact_type = tp.exact || tp.in_ns}; + if (tp.st == CS_TP_TYPE) { + return member_in(c, tp.ent, last, &u, kind); + } + if (tp.st != CS_TP_NS) { + return path_unfound(c, &tp); + } + cs_found_t fd = {0}; + if (!kind) { + add_types(c, true, tp.ns, last->name, type_arity(last->arity), &fd); + } + bool is_ns = + !kind && last->arity <= 0 && ns_in_view(c, tp.ns, last->name, strlen(last->name)) >= 0; + if (fd.n + (is_ns ? SKIP_ONE : 0) > SKIP_ONE) { + return found_ambiguous(&fd); + } + if (is_ns) { + return has_params ? res_unres(CBM_DOCLINK_REASON_MISSING) : namespace_result(); + } + if (fd.n == 0) { + return fd.invisible ? res_unres(CBM_DOCLINK_REASON_TEST_ONLY) : namespace_unfound(c, tp.ns); + } + u.exact_type = u.exact_type && !fd.joined; + if (!has_params) { + return type_result(c, fd.first.id, u.exact_type); + } + u.where = (cs_where_t){.type_seg = n - SKIP_ONE, .member_seg = CS_NONE}; + u.exact = u.exact_type; + return ctor_result(c, fd.first.id, &u); +} + +/* An operator, a conversion or an indexer: of the type the path before it + * names, or of the nearest enclosing type that declares one. None has a + * node: one that is declared is a graph gap. */ +static cs_res_t resolve_operator(const cs_ctx_t *c, const cs_ref_t *r) { + if (r->nsegs > 0) { + cs_tp_t tp = resolve_path(c, r->segs, r->nsegs, true); + if (tp.st == CS_TP_NS) { + return namespace_unfound(c, tp.ns); + } + if (tp.st != CS_TP_TYPE) { + return path_unfound(c, &tp); + } + return special_declared(c, tp.ent, r->op_name) + ? res_unres(CBM_DOCLINK_REASON_GRAPH_GAP) + : member_unfound(c, tp.ent, r->op_name, CS_ARITY_NONE); + } + for (int t = c->type; t >= 0; t = c->f->types[t].outer) { + if (special_declared(c, c->f->types[t].entity, r->op_name)) { + return res_unres(CBM_DOCLINK_REASON_GRAPH_GAP); + } + } + return c->type >= 0 ? member_unfound(c, c->f->types[c->type].entity, r->op_name, CS_ARITY_NONE) + : res_unres(CBM_DOCLINK_REASON_MISSING); +} + +/* A doc ID names its target by its full name, from the global namespace: + * `T:` a type, `N:` a namespace, `M:` `P:` `F:` `E:` a member of the kind + * the letter says. An `M:` without parentheses is the overload without + * parameters; DocFX's `O:` is the whole group. */ +static cs_res_t resolve_docid(const cs_ctx_t *c, const cs_ref_t *r) { + cs_ctx_t abs = *c; /* what stands around the reference is no part of the name */ + abs.type = CS_NONE; + abs.member = CS_NONE; + if (r->docid == 'T' || r->docid == 'N') { + cs_tp_t tp = resolve_path(&abs, r->segs, r->nsegs, false); + if (tp.st == CS_TP_RES) { + return tp.res; + } + if (r->docid == 'N') { + if (tp.st == CS_TP_NS) { + return namespace_result(); + } + /* a type, or a namespace the repository does not have */ + return res_unres(tp.st == CS_TP_NSFAIL ? CBM_DOCLINK_REASON_EXTERNAL + : CBM_DOCLINK_REASON_MISSING); + } + if (tp.st == CS_TP_TYPE) { + return type_result(&abs, tp.ent, !tp.joined); + } + return tp.st == CS_TP_NS ? res_unres(CBM_DOCLINK_REASON_MISSING) : path_unfound(&abs, &tp); + } + if (r->nsegs < PAIR_LEN) { + return res_unres(CBM_DOCLINK_REASON_UNPARSEABLE); + } + switch (r->docid) { + case 'M': + return resolve_qualified(&abs, r, 'c', true); + case 'O': + return resolve_qualified(&abs, r, 'c', false); + case 'P': + return resolve_qualified(&abs, r, 'p', r->has_params); + case 'E': + return resolve_qualified(&abs, r, 'e', false); + default: + return resolve_qualified(&abs, r, 'v', false); + } +} + +static cs_res_t resolve_form(const cs_ctx_t *c, const cs_ref_t *r) { + if (r->op) { + return resolve_operator(c, r); + } + if (r->docid) { + return resolve_docid(c, r); + } + if (r->nsegs == SKIP_ONE && !c->glob) { + return resolve_simple(c, r); + } + return resolve_qualified(c, r, 0, r->has_params); +} + +static bool reason_is_nothing(const cs_res_t *res) { + return res->st == CS_UNRES && (res->reason == CBM_DOCLINK_REASON_MISSING || + res->reason == CBM_DOCLINK_REASON_EXTERNAL); +} + +/* A reference in this context. For product code a namespace that only test + * code declares does not exist. When the reference came to nothing and such + * a namespace was passed over on the way, what it names may be a test + * declaration: asked once more the way test code sees it, a name that is + * there is test_only_target; one that is not there either keeps the reason + * it has without that namespace. */ +static cs_res_t resolve_ref(const cs_ctx_t *c, const cs_ref_t *r) { + bool passed_over = false; + cs_ctx_t seen = *c; + seen.passed_over = &passed_over; + cs_res_t res = resolve_form(&seen, r); + if (!passed_over || !reason_is_nothing(&res)) { + return res; + } + cs_ctx_t as_test = *c; + as_test.prod = false; + as_test.passed_over = NULL; + cs_res_t there = resolve_form(&as_test, r); + bool nothing = reason_is_nothing(&there) || + (there.st == CS_UNRES && there.reason == CBM_DOCLINK_REASON_UNPARSEABLE); + return nothing ? res : res_unres(CBM_DOCLINK_REASON_TEST_ONLY); +} + +/* ── Context ─────────────────────────────────────────────────────── */ + +/* The innermost type of `f` around `line`, or CS_NONE: the last type that + * starts at or before the line, or the nearest of its outer types that + * reaches the line (the types are in document order). */ +static int type_at(const cs_file_t *f, uint32_t line) { + int lo = 0; + int hi = f->ntypes; + while (lo < hi) { + int mid = lo + ((hi - lo) / PAIR_LEN); + if (f->types[mid].start <= line) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + for (int t = lo - SKIP_ONE; t >= 0; t = f->types[t].outer) { + if (f->types[t].end >= line) { + return t; + } + } + return CS_NONE; +} + +/* The innermost namespace declaration around `line` (0: the file itself). */ +static int region_at(const cs_file_t *f, uint32_t line) { + int lo = SKIP_ONE; + int hi = f->nregions; + while (lo < hi) { + int mid = lo + ((hi - lo) / PAIR_LEN); + if (f->regions[mid].start <= line) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + for (int r = lo - SKIP_ONE; r > 0; r = f->regions[r].parent) { + if (f->regions[r].end >= line) { + return r; + } + } + return 0; +} + +/* The generic method declared at `line`, or CS_NONE: its type parameters are + * in scope in its documentation. Of the members that start at one line the + * generic methods stand first (finish_lookups), so the first one tells -- + * however many members a line holds. */ +static int generic_method_at(const cs_file_t *f, uint32_t line) { + int lo = 0; + int hi = f->nmembers; + while (lo < hi) { + int mid = lo + ((hi - lo) / PAIR_LEN); + if (f->members[f->members_by_start[mid]].start < line) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + if (lo < f->nmembers) { + cs_work(SKIP_ONE); + int mi = f->members_by_start[lo]; + const cs_member_t *m = &f->members[mi]; + if (m->start == line && m->kind == 'c' && m->arity > 0) { + return mi; + } + } + return CS_NONE; +} + +/* A scope of file `file`: the namespace declaration `region`, the type + * `type` and the ones around it. */ +static void ctx_at(cs_ctx_t *c, const cs_index_t *ix, int file, int region, int type) { + memset(c, 0, sizeof(*c)); + c->ix = ix; + c->file = file; + c->f = &ix->files[file]; + c->unit = c->f->unit; + c->group = c->unit >= 0 ? ix->units[c->unit].group : CS_NONE; + c->prod = !c->f->is_test; + c->region = region; + c->type = type; + c->member = CS_NONE; + c->skip_region = CS_NONE; +} + +/* The scope of the definition that starts at `line` (the documented type is + * itself the innermost one: its members and type parameters are in scope in + * its own documentation). A file's own doc stands outside every declaration. */ +static void ctx_init(cs_ctx_t *c, const cs_index_t *ix, int file, uint32_t line, bool file_doc) { + const cs_file_t *f = &ix->files[file]; + int type = file_doc ? CS_NONE : type_at(f, line); + int region = file_doc ? 0 : (type >= 0 ? f->types[type].region : region_at(f, line)); + ctx_at(c, ix, file, region, type); + c->member = (type >= 0 && f->nmembers > 0) ? generic_method_at(f, line) : CS_NONE; +} + +/* True when `line` is in a part of the file whose declarations could not be + * placed (the ranges are sorted and disjoint). */ +static bool line_unplaced(const cs_file_t *f, uint32_t line) { + int lo = 0; + int hi = f->nunplaced; + while (lo < hi) { + int mid = lo + ((hi - lo) / PAIR_LEN); + if (f->unplaced[mid].to < line) { + lo = mid + SKIP_ONE; + } else { + hi = mid; + } + } + return lo < f->nunplaced && f->unplaced[lo].from <= line; +} + +/* ── Usings and base lists ───────────────────────────────────────── */ + +static const char CS_GLOBAL_PREFIX[] = "global::"; + +/* A written type or namespace name into `segs`; *glob when it starts at the + * global namespace. false for what is no dotted name (a tuple, an array, a + * pointer, a text this code did not understand). */ +static bool parse_name(const char *s, size_t n, cs_seg_t *segs, int *nsegs, bool *glob) { + size_t gl = sizeof(CS_GLOBAL_PREFIX) - SKIP_ONE; + *glob = n >= gl && strncmp(s, CS_GLOBAL_PREFIX, gl) == 0; + if (*glob) { + s += gl; + n -= gl; + } + return parse_path(s, n, segs, nsegs); +} + +/* What a using directive names: a namespace into u->ns, a type into u->ent + * (CS_AMBIGUOUS for several). Both stay CS_NONE for what the repository does + * not have. `c` is the scope the directive is resolved in. */ +static void resolve_using(cs_using_t *u, const cs_ctx_t *c) { + cs_seg_t segs[CS_MAX_SEGS]; + int n = 0; + cs_ctx_t at = *c; + bool glob = false; + u->ns = CS_NONE; + u->ent = CS_NONE; + if (!parse_name(u->target, strlen(u->target), segs, &n, &glob)) { + return; + } + at.glob = at.glob || glob; + cs_tp_t tp = resolve_path(&at, segs, n, false); + if (tp.st == CS_TP_NS && u->kind != 's') { + u->ns = tp.ns; + } else if (tp.st == CS_TP_TYPE && u->kind != 'n') { + u->ent = tp.ent; + u->joined = tp.joined; + } else if (tp.st == CS_TP_RES && res_is(&tp.res, CBM_DOCLINK_REASON_AMBIGUOUS) && + u->kind != 'n') { + u->ent = CS_AMBIGUOUS; + } +} + +/* True when the directive brings in names the repository does not declare: + * a namespace it has no declaration of (or a standard-library one, which it + * never has all of), a type it does not have. (An alias brings one name, and + * says so itself where it is used.) */ +static bool using_opens(const cs_index_t *ix, const cs_using_t *u) { + if (u->kind == 'n') { + return u->ns < 0 || !ix->nss[u->ns].declared || ix->nss[u->ns].standard; + } + return u->kind == 's' && u->ent == CS_NONE; +} + +/* What every using directive names, and which scopes are open. A directive + * at the top of a file, a `global using` and a project file's are + * resolved from the global namespace: nothing else is in scope there (the + * directives beside it are not). A directive inside a namespace declaration + * is resolved in that namespace, without that declaration's own directives. */ +static bool resolve_usings(cs_index_t *ix) { + for (int ui = 0; ui < ix->nunits; ui++) { + cs_unit_t *unit = &ix->units[ui]; + cs_ctx_t c = {.ix = ix, + .type = CS_NONE, + .member = CS_NONE, + .unit = ui, + .group = unit->group, + .glob = true, + .skip_region = CS_NONE}; + for (int i = 0; i < unit->nusings; i++) { + resolve_using(&unit->usings[i], &c); + unit->open = unit->open || using_opens(ix, &unit->usings[i]); + } + if (!build_using_index(ix, unit->usings, unit->nusings, &unit->using_index)) { + return false; + } + } + /* A lookup asks only regions above the directive's own, whose directives + * are resolved already: what it remembers of them stays true. */ + cs_memo_t memo; + memo_init(&memo); + for (int fi = 0; fi < ix->nfiles; fi++) { + cs_file_t *f = &ix->files[fi]; + /* Regions are in ancestor order. Publish an outer region's index + * before resolving its children's targets; a region's own imports + * remain excluded by skip_region during their resolution. */ + for (int r = 0; r < f->nregions; r++) { + cs_region_t *reg = &f->regions[r]; + for (int i = reg->u_lo; i < reg->u_hi; i++) { + cs_ctx_t c; + ctx_at(&c, ix, fi, r, CS_NONE); + c.prod = false; /* a directive names whatever the compiler bound */ + c.glob = r == 0; + c.skip_region = r; + c.memo = &memo; + resolve_using(&f->usings[i], &c); + } + const cs_using_t *us = reg->u_hi > reg->u_lo ? f->usings + reg->u_lo : NULL; + if (!build_using_index(ix, us, reg->u_hi - reg->u_lo, ®->using_index)) { + memo_destroy(&memo); + return false; + } + } + for (int r = 0; r < f->nregions; r++) { + cs_region_t *reg = &f->regions[r]; + reg->open = r > 0 && f->regions[reg->parent].open; + for (int i = reg->u_lo; !reg->open && i < reg->u_hi; i++) { + reg->open = using_opens(ix, &f->usings[i]); + } + } + } + memo_destroy(&memo); + return true; +} + +/* The entity a written base type names in the scope of its declaration, or + * CS_NONE. */ +static int base_entity(const cs_ctx_t *c, const char *s, size_t n) { + cs_seg_t segs[CS_MAX_SEGS]; + int nsegs = 0; + cs_ctx_t at = *c; + bool glob = false; + if (!parse_name(s, n, segs, &nsegs, &glob)) { + return CS_NONE; + } + at.glob = glob; + cs_tp_t tp = resolve_path(&at, segs, nsegs, false); + return tp.st == CS_TP_TYPE ? tp.ent : CS_NONE; +} + +/* The base types of every entity, from the base lists of all its + * declarations. A base that names nothing of the repository leaves the + * hierarchy open. false when memory ran out. */ +static bool resolve_bases(cs_index_t *ix) { + /* listed[b] == e + 1: b is in e's base list already (a partial type's + * declarations may each write the same base) */ + int *listed = + (int *)cbm_calloc(CBM_MEM_CLASS_OTHER, ((size_t)ix->nents + SKIP_ONE) * sizeof(int)); + bool ok = listed != NULL; + cs_memo_t memo; /* every directive is resolved: what a lookup remembers stays true */ + memo_init(&memo); + for (int ei = 0; ok && ei < ix->nents; ei++) { + cs_entity_t *e = &ix->ents[ei]; + int written = 0; + for (int d = 0; d < e->ndecls; d++) { + written += count_list(ix->files[e->decls[d].file].types[e->decls[d].type].bases, '|'); + } + if (written == 0) { + continue; + } + e->bases = (int *)ix_alloc(ix, (size_t)written * sizeof(int)); + ok = e->bases != NULL; + for (int d = 0; ok && d < e->ndecls; d++) { + const cs_type_t *t = &ix->files[e->decls[d].file].types[e->decls[d].type]; + cs_ctx_t c; + /* a base list is written outside the type it belongs to */ + ctx_at(&c, ix, e->decls[d].file, t->region, t->outer); + c.prod = false; /* a declared base is whatever the compiler bound */ + c.memo = &memo; + for (const char *p = t->bases; p && *p;) { + const char *bar = strchr(p, '|'); + size_t n = bar ? (size_t)(bar - p) : strlen(p); + int base = base_entity(&c, p, n); + if (base < 0) { + e->open = true; /* the hierarchy goes on outside the repository */ + } else if (base != ei && listed[base] != ei + SKIP_ONE) { + listed[base] = ei + SKIP_ONE; + e->bases[e->nbases++] = base; + } + p = bar ? bar + SKIP_ONE : NULL; + } + } + } + memo_destroy(&memo); + cbm_free(CBM_MEM_CLASS_OTHER, listed); + ix->oom = ix->oom || !ok; + return ok; +} + +/* How many entities `e` takes its openness from: its bases, and the shared + * trees' parts that belong to it (what they derive from, it derives from). */ +static int upper_count(const cs_entity_t *e) { + return e->nbases + (e->twin >= 0 ? SKIP_ONE : 0); +} + +static int upper_at(const cs_entity_t *e, int k) { + return k < e->nbases ? e->bases[k] : e->twin; +} + +/* open_any: a type whose own base, or a base of one of its supertypes, is + * outside the repository. Spread from the open types to everything derived + * from them, breadth first over the reversed base lists (a base list that + * goes round in a circle ends at the types already marked). false when + * memory ran out. */ +static bool spread_open(cs_index_t *ix) { + size_t n = (size_t)ix->nents; + int *first = (int *)cbm_calloc(CBM_MEM_CLASS_OTHER, (n + PAIR_LEN) * sizeof(int)); + int *queue = (int *)cbm_alloc(CBM_MEM_CLASS_OTHER, (n + SKIP_ONE) * sizeof(int)); + size_t edges = 0; + for (int i = 0; i < ix->nents; i++) { + edges += (size_t)upper_count(&ix->ents[i]); + } + int *derived = (int *)cbm_alloc(CBM_MEM_CLASS_OTHER, (edges + SKIP_ONE) * sizeof(int)); + bool ok = first && queue && derived; + if (ok) { + int qn = 0; + /* first[b + 2] counts the types derived from b; after the running + * sum first[b + 1] is where b's list starts, and filling the lists + * moves it to where the list ends: first[b] .. first[b + 1] */ + for (int i = 0; i < ix->nents; i++) { + for (int b = 0; b < upper_count(&ix->ents[i]); b++) { + first[upper_at(&ix->ents[i], b) + PAIR_LEN]++; + } + } + for (size_t i = PAIR_LEN; i < n + PAIR_LEN; i++) { + first[i] += first[i - SKIP_ONE]; + } + for (int i = 0; i < ix->nents; i++) { + for (int b = 0; b < upper_count(&ix->ents[i]); b++) { + derived[first[upper_at(&ix->ents[i], b) + SKIP_ONE]++] = i; + } + if (ix->ents[i].open) { + ix->ents[i].open_any = true; + queue[qn++] = i; + } + } + for (int head = 0; head < qn; head++) { + int b = queue[head]; + for (int k = first[b]; k < first[b + SKIP_ONE]; k++) { + if (!ix->ents[derived[k]].open_any) { + ix->ents[derived[k]].open_any = true; + queue[qn++] = derived[k]; + } + } + } + } + cbm_free(CBM_MEM_CLASS_OTHER, first); + cbm_free(CBM_MEM_CLASS_OTHER, queue); + cbm_free(CBM_MEM_CLASS_OTHER, derived); + ix->oom = ix->oom || !ok; + return ok; +} + +/* ── Index lifecycle and the resolver hooks ──────────────────────── */ + +/* What made the run's ambiguous references ambiguous, by kind: the rows all + * say `ambiguous`, and only the scope's rules and the limits are the same in + * every repository. Nothing is logged for a run without one. */ +static void log_ambiguous(const cs_stats_t *st) { + char b[CS_WHY_COUNT][CBM_SZ_32]; + uint64_t total = 0; + for (int i = 0; i < CS_WHY_COUNT; i++) { + uint64_t n = atomic_load(&st->ambiguous[i]); + total += n; + snprintf(b[i], sizeof(b[i]), "%llu", (unsigned long long)n); + } + if (total > 0) { + cbm_log_info("doc_links.cs.ambiguous", "scope_rules", b[CS_WHY_SCOPE], "shared_trees", + b[CS_WHY_SHARED], "assemblies", b[CS_WHY_ASSEMBLIES], "flavours", + b[CS_WHY_FLAVOURS], "unseen_parts", b[CS_WHY_PARTS], "limits", + b[CS_WHY_LIMIT]); + } +} + +static void cs_destroy(void *index) { + cs_index_t *ix = (cs_index_t *)index; + if (!ix) { + return; + } + if (ix->stats) { + log_ambiguous(ix->stats); + } + if (ix->rejected > 0) { + /* scopes written in this run that this reader refused: each cost + * only its own file, whose references are graph gaps */ + char files[CBM_SZ_32]; + char refs[CBM_SZ_32]; + snprintf(files, sizeof(files), "%d", ix->rejected); + snprintf(refs, sizeof(refs), "%llu", + (unsigned long long)(ix->stats ? atomic_load(&ix->stats->rejected) : 0)); + cbm_log_warn("doc_links.cs.rejected_scopes", "files", files, "references", refs); + } + unit_memo_free(ix->unit_memo); + cbm_free(CBM_MEM_CLASS_OTHER, ix->undo.flags); + cbm_free(CBM_MEM_CLASS_OTHER, ix->undo.marks); + cbm_free(CBM_MEM_CLASS_OTHER, ix->stats); + cbm_ht_free(ix->project_above); + cbm_ht_free(ix->ns_by_key); + cbm_ht_free(ix->ent_by_key); + cbm_ht_free(ix->type_names); + cbm_ht_free(ix->quarantine); + cbm_ht_free(ix->quarantine_test); + cbm_ht_free(ix->unit_by_dir); + cbm_ht_free(ix->project_dirs); + cbm_ht_free(ix->group_by_stem); + cbm_ht_free(ix->specials); + cbm_msb_free(ix->msb); + cbm_free(CBM_MEM_CLASS_OTHER, ix->ents); + cbm_free(CBM_MEM_CLASS_OTHER, ix->run_to_file); + cbm_arena_destroy(&ix->arena); + cbm_free(CBM_MEM_CLASS_OTHER, ix); +} + +/* Log why the index could not be built, and drop it. */ +static void *build_failed(cs_index_t *ix, const char *step, const char *reason, const char *path) { + cbm_log_error("doc_links.cs.error", "step", step, "reason", reason, "file", path ? path : ""); + cs_destroy(ix); + return NULL; +} + +/* The index's tables and the project files. false when memory ran out. */ +static bool build_tables(cs_index_t *ix, const cbm_doclink_build_in_t *in) { + cbm_arena_init(&ix->arena); + ix->project = in->ctx->project_name; + ix->bind_inherited = CS_BIND_INHERITED; + ix->ns_by_key = cbm_ht_create(CBM_SZ_1K); + ix->ent_by_key = cbm_ht_create(CBM_SZ_4K); + ix->type_names = cbm_ht_create(CBM_SZ_4K); + ix->quarantine = cbm_ht_create(CBM_SZ_64); + ix->quarantine_test = cbm_ht_create(CBM_SZ_64); + ix->unit_by_dir = cbm_ht_create(CBM_SZ_256); + ix->project_dirs = cbm_ht_create(CBM_SZ_256); + ix->project_above = cbm_ht_create(CBM_SZ_256); + ix->group_by_stem = cbm_ht_create(CBM_SZ_256); + ix->specials = cbm_ht_create(CBM_SZ_256); + ix->stats = (cs_stats_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(cs_stats_t)); + ix->msb = cbm_msb_new(); + ix->nfiles = in->file_count; + ix->files = (cs_file_t *)ix_zalloc(ix, (size_t)in->file_count * sizeof(cs_file_t)); + ix->run_count = in->run_file_count; + ix->run_to_file = (int *)cbm_alloc(CBM_MEM_CLASS_OTHER, + ((size_t)in->run_file_count + SKIP_ONE) * sizeof(int)); + if (!ix->ns_by_key || !ix->ent_by_key || !ix->type_names || !ix->quarantine || + !ix->quarantine_test || !ix->unit_by_dir || !ix->project_dirs || !ix->project_above || + !ix->group_by_stem || !ix->specials || !ix->stats || !ix->msb || !ix->files || + !ix->run_to_file) { + return false; + } + for (int i = 0; i < in->run_file_count; i++) { + ix->run_to_file[i] = CS_NONE; + } + /* namespace 0: the global one */ + ix->nss = (cs_ns_t *)ix_zalloc(ix, CBM_SZ_256 * sizeof(cs_ns_t)); + if (!ix->nss) { + return false; + } + ix->nscap = CBM_SZ_256; + ix->nnss = SKIP_ONE; + ix->nss[0] = (cs_ns_t){.parent = CS_NONE, .name = ""}; + return collect_projects(ix, in); +} + +/* Start the undo record of one file's scope. */ +static void undo_begin(cs_index_t *ix) { + ix->undo.nnss = ix->nnss; + ix->undo.nflags = 0; + ix->undo.nmarks = 0; +} + +/* Take back what the file's scope set in the shared tables: the names it + * quarantined, the flags it set on namespaces that were there before it, and + * the namespaces it made. */ +static void undo_scope(cs_index_t *ix) { + cs_undo_t *u = &ix->undo; + for (int i = u->nmarks - SKIP_ONE; i >= 0; i--) { + cbm_ht_delete(u->marks[i].ht, u->marks[i].key); + } + for (int i = u->nflags - SKIP_ONE; i >= 0; i--) { + ix->nss[u->flags[i].ns].declared = u->flags[i].declared; + ix->nss[u->flags[i].ns].prod = u->flags[i].prod; + } + for (int id = ix->nnss - SKIP_ONE; id >= u->nnss; id--) { + char key[CS_NAME_BUF + CBM_SZ_16]; + const cs_ns_t *n = &ix->nss[id]; + if (ns_key(key, sizeof(key), n->parent, n->name, strlen(n->name))) { + cbm_ht_delete(ix->ns_by_key, key); + } + } + ix->nnss = u->nnss; + u->nflags = 0; + u->nmarks = 0; +} + +/* Scope line fields (0-based, the tag is field 0; internal/cbm/doclink_cs.c): + * `U region kind alias target`, `T region start end kind outer name tparams + * bases`, `M start kind explicit type name tparams sig` and `Q name`. */ +enum { + CS_SCOPE_Q_NAME = 1, + CS_SCOPE_U_KIND = 2, + CS_SCOPE_M_NAME = 5, + CS_SCOPE_T_NAME = 6, + CS_SCOPE_T_BASES = 8 +}; + +static const char *delta_field(const char *line, size_t len, int idx, size_t *flen) { + int f = 0; + size_t s = 0; + for (size_t i = 0; i <= len; i++) { + if (i == len || line[i] == '\t') { + if (f == idx) { + *flen = i - s; + return line + s; + } + f++; + s = i + SKIP_ONE; + } + } + *flen = 0; + return NULL; +} + +/* True when s[0, n) is a name a reference can carry (ident_ok). */ +static bool cs_name_ok(const char *s, size_t n) { + char buf[CS_NAME_BUF]; + if (n == 0 || n >= sizeof(buf)) { + return false; + } + memcpy(buf, s, n); + buf[n] = '\0'; + return ident_ok(buf); +} + +/* The next type name a T or Q record of a scope blob carries, read leniently + * (the blob is one the reader refused): a line is taken when its name field + * is there and is a name, however the rest of it reads. Moves *cursor past + * the lines it read; false at the end of the blob. */ +static bool cs_next_type_name(const char **cursor, const char **name, size_t *len) { + while (**cursor) { + const char *line = *cursor; + const char *nl = strchr(line, '\n'); + size_t n = nl ? (size_t)(nl - line) : strlen(line); + *cursor = line + n + (nl ? SKIP_ONE : 0); + int idx = line[0] == 'T' ? CS_SCOPE_T_NAME : line[0] == 'Q' ? CS_SCOPE_Q_NAME : CS_NONE; + *name = idx > 0 && n > SKIP_ONE && line[SKIP_ONE] == '\t' ? delta_field(line, n, idx, len) + : NULL; + if (*name && cs_name_ok(*name, *len)) { + return true; + } + } + return false; +} + +/* What is stored, in place of its scope blob, for a file whose scope the + * reader refused in the run that wrote it (cs_scope_accepted, lsp_surface.c): + * the tag, one `!` record, and one Q record per type name the T and Q records + * of the refused blob carry (cs_next_type_name). A later run that reads it + * back treats the file as that run did -- it declares nothing, those names + * are quarantined, its references are graph gaps -- where the refused blob + * itself would fail that run as a stored scope this reader does not take. */ +static const char CS_REJECTED_SCOPE[] = CBM_DOCLINK_CS_SCOPE_TAG "\n!\trejected\n"; + +/* The marker, followed by nothing but Q records of names. */ +static bool cs_scope_marked_rejected(const char *scope) { + size_t n = sizeof(CS_REJECTED_SCOPE) - SKIP_ONE; + if (!scope || strncmp(scope, CS_REJECTED_SCOPE, n) != 0) { + return false; + } + for (const char *line = scope + n; *line;) { + const char *nl = strchr(line, '\n'); + if (!nl || line[0] != 'Q' || line[SKIP_ONE] != '\t' || + !cs_name_ok(line + PAIR_LEN, (size_t)(nl - line) - PAIR_LEN)) { + return false; + } + line = nl + SKIP_ONE; + } + return true; +} + +/* The marker stored for the refused blob `scope` (the resolver's + * rejected_scope): a memory-core block, NULL when memory ran out. A Q record + * is never longer than the line its name is read from plus a newline. */ +static char *cs_rejected_scope(const char *scope) { + size_t w = sizeof(CS_REJECTED_SCOPE) - SKIP_ONE; + char *out = (char *)cbm_alloc(CBM_MEM_CLASS_OTHER, w + strlen(scope) + PAIR_LEN); + if (!out) { + return NULL; + } + memcpy(out, CS_REJECTED_SCOPE, w); + const char *cursor = scope; + const char *name = NULL; + size_t len = 0; + while (cs_next_type_name(&cursor, &name, &len)) { + out[w++] = 'Q'; + out[w++] = '\t'; + memcpy(out + w, name, len); + w += len; + out[w++] = '\n'; + } + out[w] = '\0'; + return out; +} + +/* Quarantine the type names a refused scope blob, or its stored marker, + * carries: what the file declares is unknown, so a reference to one of them + * from any file is a graph gap. false when memory ran out. */ +static bool quarantine_refused(cs_index_t *ix, const cs_file_t *f, const char *scope) { + const char *cursor = scope; + const char *name = NULL; + size_t len = 0; + while (cs_next_type_name(&cursor, &name, &len)) { + char buf[CS_NAME_BUF]; + memcpy(buf, name, len); /* shorter than the buffer: cs_name_ok */ + buf[len] = '\0'; + if (!quarantine_name(ix, f, buf)) { + return false; + } + } + return true; +} + +/* Whether this reader takes a scope blob written in this run: its record + * checks, on an index of the blob's own (what makes a blob refused depends on + * the blob alone). 1 taken, 0 refused, -1 memory ran out. A project file's + * blob goes to the MSBuild evaluator, which takes every blob. */ +static int cs_scope_accepted(const char *scope) { + if (!scope || cbm_msb_is_project_scope(scope) || cs_scope_marked_rejected(scope)) { + return SKIP_ONE; + } + cs_index_t *ix = (cs_index_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*ix)); + if (!ix) { + return CBM_NOT_FOUND; + } + cbm_arena_init_lazy(&ix->arena, CBM_ARENA_APPEND_BLOCK); + ix->ns_by_key = cbm_ht_create(CBM_SZ_16); + ix->quarantine = cbm_ht_create(CBM_SZ_16); + ix->quarantine_test = cbm_ht_create(CBM_SZ_16); + ix->nss = (cs_ns_t *)ix_zalloc(ix, CBM_SZ_16 * sizeof(cs_ns_t)); + int accepted = CBM_NOT_FOUND; + if (ix->ns_by_key && ix->quarantine && ix->quarantine_test && ix->nss) { + ix->nscap = CBM_SZ_16; + ix->nnss = SKIP_ONE; + ix->nss[0] = (cs_ns_t){.parent = CS_NONE, .name = ""}; + cs_file_t f = {.rel_path = "Scope.cs"}; + undo_begin(ix); + bool parsed = parse_scope(ix, &f, scope); + accepted = ix->oom ? CBM_NOT_FOUND : (parsed ? SKIP_ONE : 0); + } + cbm_ht_free(ix->ns_by_key); + cbm_ht_free(ix->quarantine); + cbm_ht_free(ix->quarantine_test); + cbm_free(CBM_MEM_CLASS_OTHER, ix->undo.flags); + cbm_free(CBM_MEM_CLASS_OTHER, ix->undo.marks); + cbm_arena_destroy(&ix->arena); + cbm_free(CBM_MEM_CLASS_OTHER, ix); + return accepted; +} + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +/* Test seam: true when this reader takes the scope blob `scope`. A test holds + * every blob the scanner writes against it. */ +bool cbm_doclink_cs_test_scope_parses(const char *scope) { + return cs_scope_accepted(scope) == SKIP_ONE; +} + +/* Test seam: set, the indexes are built without the unit memo. */ +static atomic_bool cs_test_no_unit_memo; + +void cbm_doclink_cs_test_unit_memo(bool on) { + atomic_store(&cs_test_no_unit_memo, !on); +} +#endif + +/* Every file of the index with what its scope declares. A scope read back + * from the store that this reader refuses is not one this code wrote: the + * index is not built over it (nothing may be resolved around a declaration + * that is not known), and its path is returned. A scope written in this run + * that the reader refuses is the scanner's and this reader's disagreement, + * which costs only its own file: what its parse set is taken back, the file + * declares nothing, and its references are graph gaps (counted, logged). The + * names of its types are quarantined: what it declares is unknown, so a + * reference to one of them from any file is a graph gap too. A scope stored + * as the rejected marker is read as in the run that refused it. + * Returns "" when memory ran out, NULL when all is well. */ +static const char *build_files(cs_index_t *ix, const cbm_doclink_build_in_t *in) { + CBMHashTable *dir_unit = cbm_ht_create(CBM_SZ_1K); + const char *bad = dir_unit ? NULL : ""; + for (int i = 0; !bad && i < in->file_count; i++) { + const cbm_doclink_file_t *src = &in->files[i]; + cs_file_t *f = &ix->files[i]; + f->rel_path = src->rel_path; + f->module_qn = + cbm_fqn_module_source_lang(&ix->arena, ix->project, src->rel_path, CBM_LANG_CSHARP); + f->is_test = cs_is_test_path(src->rel_path); + /* a project file declares nothing: its blob went to the evaluator */ + const char *scope = cbm_msb_is_project_scope(src->scope) ? NULL : src->scope; + /* a scope its own run refused, stored as the marker: as in that run */ + bool marked = cs_scope_marked_rejected(scope); + undo_begin(ix); + if (marked || (scope && !parse_scope(ix, f, scope))) { + bool fresh = src->run_file >= 0 && src->run_file < in->run_file_count; + if (!marked && (ix->oom || !fresh)) { + bad = ix->oom ? "" : src->rel_path; + break; + } + undo_scope(ix); + *f = (cs_file_t){.rel_path = f->rel_path, + .module_qn = f->module_qn, + .is_test = f->is_test, + .rejected = true}; + ix->rejected++; + if (!quarantine_refused(ix, f, scope)) { + bad = ""; /* memory ran out */ + break; + } + scope = NULL; + } + if (!scope) { + /* no scope: an empty file (only its own region) */ + f->regions = (cs_region_t *)ix_zalloc(ix, sizeof(cs_region_t)); + f->nregions = SKIP_ONE; + if (f->regions) { + f->regions[0].parent = CS_NONE; + } + } + f->unit = unit_of(ix, dir_unit, src->rel_path); + f->is_ref = f->unit >= 0 && ix->units[f->unit].is_ref; + if (src->run_file >= 0 && src->run_file < in->run_file_count) { + ix->run_to_file[src->run_file] = i; + } + if (!f->module_qn || ix->oom) { + bad = ""; + } + } + cbm_ht_free(dir_unit); + /* which namespaces are the standard library's: a namespace is made after + * the one above it, so one pass in order sees every parent first */ + for (int i = SKIP_ONE; i < ix->nnss; i++) { + cs_ns_t *n = &ix->nss[i]; + n->standard = + n->parent == 0 ? strcmp(n->name, CS_STANDARD_ROOT) == 0 : ix->nss[n->parent].standard; + } + return bad; +} + +typedef struct { + const char *path; + int idx; +} cs_file_order_t; + +static int file_order_cmp(const void *a, const void *b) { + const cs_file_order_t *x = (const cs_file_order_t *)a; + const cs_file_order_t *y = (const cs_file_order_t *)b; + int c = strcmp(x->path ? x->path : "", y->path ? y->path : ""); + return c ? c : (x->idx > y->idx) - (x->idx < y->idx); +} + +/* The graph node of every declaration. false when memory ran out. Files are + * bound in path order: a file's bindings do not depend on the others, but the + * scratch table's cost does on the order (it is emptied per file), and the + * order the file system listed the files in differs between systems. */ +static bool build_nodes(cs_index_t *ix, const cbm_gbuf_t *g) { + cs_node_pass_t np = {.names = cbm_ht_create(CBM_SZ_256), .sized = CBM_SZ_256}; + cbm_arena_init(&np.keys); + cs_file_order_t *order = (cs_file_order_t *)cbm_alloc( + CBM_MEM_CLASS_OTHER, (size_t)(ix->nfiles ? ix->nfiles : 1) * sizeof(*order)); + bool ok = np.names != NULL && order != NULL; + if (!order) { + ix->oom = true; + } + for (int i = 0; ok && i < ix->nfiles; i++) { + order[i] = (cs_file_order_t){ix->files[i].rel_path, i}; + } + if (ok && ix->nfiles > 1) { + qsort(order, (size_t)ix->nfiles, sizeof(*order), file_order_cmp); + } + for (int i = 0; ok && i < ix->nfiles; i++) { + ok = bind_nodes(ix, &ix->files[order[i].idx], g, &np); + } + cbm_free(CBM_MEM_CLASS_OTHER, order); + cbm_ht_free(np.names); + cbm_arena_destroy(&np.keys); + cbm_free(CBM_MEM_CLASS_OTHER, np.last); + return ok; +} + +static void log_index(const cs_index_t *ix, const cs_msb_totals_t *msb) { + int incomplete = 0; + int open_units = 0; + for (int i = 0; i < ix->nents; i++) { + incomplete += ix->ents[i].incomplete; + } + for (int i = 0; i < ix->nunits; i++) { + open_units += ix->units[i].open; + } + char b[CBM_SZ_7][CBM_SZ_32]; + snprintf(b[0], sizeof(b[0]), "%d", ix->nfiles); + snprintf(b[1], sizeof(b[1]), "%d", ix->nents); + snprintf(b[2], sizeof(b[2]), "%d", ix->nunits - ix->nshared); + snprintf(b[3], sizeof(b[3]), "%d", incomplete); + snprintf(b[4], sizeof(b[4]), "%u", + (unsigned)(cbm_ht_count(ix->quarantine) + cbm_ht_count(ix->quarantine_test))); + snprintf(b[5], sizeof(b[5]), "%d", ix->ngroups); + snprintf(b[6], sizeof(b[6]), "%d", ix->nshared); + cbm_log_info("doc_links.cs.index", "files", b[0], "types", b[1], "projects", b[2], "assemblies", + b[5], "shared_trees", b[6], "incomplete_types", b[3], "quarantined_names", b[4]); + /* what the MSBuild project files could not tell: conditions, values and + * constructs that were not evaluated, and imports of files the index + * does not hold */ + snprintf(b[0], sizeof(b[0]), "%d", ix->nprojects); + snprintf(b[1], sizeof(b[1]), "%d", msb->unevaluable); + snprintf(b[2], sizeof(b[2]), "%d", msb->outside); + snprintf(b[3], sizeof(b[3]), "%d", open_units); + cbm_log_info("doc_links.cs.msbuild", "project_files", b[0], "unevaluable", b[1], + "imports_outside", b[2], "open_projects", b[3]); +} + +static void *cs_build(const cbm_doclink_build_in_t *in) { + cs_index_t *ix = (cs_index_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*ix)); + if (!ix) { + return NULL; + } + if (!build_tables(ix, in)) { + return build_failed(ix, "tables", "alloc", NULL); + } + const char *bad = build_files(ix, in); + if (bad) { + return build_failed(ix, "scopes", bad[0] ? "bad_scope" : "alloc", bad); + } + cs_msb_totals_t msb = {0}; + if (!units_collect_usings(ix, &msb)) { + return build_failed(ix, "projects", "alloc", NULL); + } + if (!build_entities(ix) || !entity_units(ix) || !entity_fulls(ix) || !build_named(ix)) { + return build_failed(ix, "entities", "alloc", NULL); + } + entity_twins(ix); + if (!build_nodes(ix, in->graph) || !build_binds(ix) || !build_mrefs(ix) || !build_extras(ix)) { + return build_failed(ix, "nodes", "alloc", NULL); + } + if (!build_name_scopes(ix) || !resolve_usings(ix)) { + return build_failed(ix, "usings", "alloc", NULL); + } + if (!resolve_bases(ix) || !spread_open(ix) || ix->oom) { + return build_failed(ix, "bases", "alloc", NULL); + } + /* only now: what the build's own lookups saw of the units' directives + * was not yet the whole index (NULL when memory ran out: no memo) */ +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (!atomic_load(&cs_test_no_unit_memo)) +#endif + { + ix->unit_memo = unit_memo_new(); + } + log_index(ix, &msb); + return ix; +} + +/* The reason for a reference this code does not understand: one that names a + * keyword type in a form that is no name (`int[]`, `string?`) is the BCL's; + * one too long to be read is never judged by its beginning. */ +static int unparsed_reason(const char *raw) { + const char *s = raw ? raw : ""; + while (isspace((unsigned char)*s)) { + s++; + } + char head[CBM_SZ_32]; + size_t n = strcspn(s, "({<.[?* \t"); + if (strlen(s) >= CS_REF_BUF || n == 0 || n >= sizeof(head)) { + return CBM_DOCLINK_REASON_UNPARSEABLE; + } + memcpy(head, s, n); + head[n] = '\0'; + return keyword_type(head) ? CBM_DOCLINK_REASON_EXTERNAL : CBM_DOCLINK_REASON_UNPARSEABLE; +} + +/* A keyword alias (`int`, `string.Empty`) names its System type whatever + * stands around the reference. */ +static void rewrite_keyword(cs_ref_t *r) { + static const char system_ns[] = "System"; + if (r->docid || r->nsegs == 0 || r->nsegs >= CS_MAX_SEGS || r->segs[0].arity > 0) { + return; + } + const char *bcl = keyword_type(r->segs[0].name); + if (!bcl) { + return; + } + memmove(&r->segs[1], &r->segs[0], (size_t)r->nsegs * sizeof(cs_seg_t)); + r->segs[0] = (cs_seg_t){.arity = CS_ARITY_NONE}; + snprintf(r->segs[0].name, sizeof(r->segs[0].name), "%s", system_ns); + snprintf(r->segs[1].name, sizeof(r->segs[1].name), "%s", bcl); + r->nsegs++; + r->keyword = true; + r->glob = true; +} + +/* True when the reference names a type some file declares at a place no + * scope could be established for: it could be that declaration. */ +static bool names_quarantined(const cs_index_t *ix, const cs_file_t *f, const cs_ref_t *r) { + for (int i = 0; i < r->nsegs; i++) { + if (cbm_ht_get(ix->quarantine, r->segs[i].name) || + (f->is_test && cbm_ht_get(ix->quarantine_test, r->segs[i].name))) { + return true; + } + } + return false; +} + +/* The working state of one file's references: the memo of their lookups. */ +static void *cs_file_begin(const void *index, int run_file) { + (void)index; + (void)run_file; + cs_memo_t *m = (cs_memo_t *)cbm_alloc(CBM_MEM_CLASS_OTHER, sizeof(*m)); + if (m) { + memo_init(m); + } + return m; +} + +static void cs_file_end(void *state) { + cs_memo_t *m = (cs_memo_t *)state; + if (m) { + memo_destroy(m); + cbm_free(CBM_MEM_CLASS_OTHER, m); + } +} + +static void cs_resolve(const void *index, void *state, int run_file, const CBMDocLink *link, + const cbm_gbuf_t *graph, cbm_doclink_outcome_t *out) { + const cs_index_t *ix = (const cs_index_t *)index; + (void)graph; /* every node was looked up when the index was built */ + out->kind = CBM_DOCLINK_UNRESOLVED; + out->reason = CBM_DOCLINK_REASON_MISSING; + out->target = NULL; + out->exact = false; + if (!ix || run_file < 0 || run_file >= ix->run_count || ix->run_to_file[run_file] < 0) { + return; + } + cs_ref_t r; + if (!parse_cref(link->raw, &r)) { + out->reason = unparsed_reason(link->raw); + return; + } + rewrite_keyword(&r); + /* What could not be placed is not resolved around: a definition in the + * part of its file where the braces stop pairing has no known scope, and + * a name some file declares without a known namespace could be that + * declaration. Both are declared-but-unplaced, i.e. graph gaps. */ + int file = ix->run_to_file[run_file]; + const cs_file_t *f = &ix->files[file]; + if (f->rejected) { + /* its scope was refused: nothing is known around the reference */ + out->reason = CBM_DOCLINK_REASON_GRAPH_GAP; + atomic_fetch_add_explicit(&ix->stats->rejected, 1, memory_order_relaxed); + return; + } + bool file_doc = (link->flags & CBM_DOCLINK_FLAG_FILE) != 0; + if ((!file_doc && line_unplaced(f, link->def_line)) || names_quarantined(ix, f, &r)) { + out->reason = CBM_DOCLINK_REASON_GRAPH_GAP; + return; + } + cs_ctx_t c; + ctx_init(&c, ix, file, link->def_line, file_doc); + c.glob = r.glob; + c.memo = (cs_memo_t *)state; + cs_res_t res = resolve_ref(&c, &r); + if (res.st == CS_LOCAL) { + out->kind = CBM_DOCLINK_LOCAL; + } else if (res.st == CS_OK && res.node) { + out->kind = CBM_DOCLINK_EDGE; + out->target = res.node; + out->exact = res.exact; + } else if (r.keyword && res.reason == CBM_DOCLINK_REASON_MISSING) { + out->reason = CBM_DOCLINK_REASON_EXTERNAL; /* the repository does not declare it */ + } else { + out->reason = res.reason; + if (res.reason == CBM_DOCLINK_REASON_AMBIGUOUS && res.why < CS_WHY_COUNT) { + atomic_fetch_add_explicit(&ix->stats->ambiguous[res.why], 1, memory_order_relaxed); + } + } +} + +/* ── Incremental scope rules ─────────────────────────────────────── */ + +/* A using directive only its own file sees: one that is not `global`. */ +static bool delta_local_using(const char *line, size_t len) { + if (line[0] != 'U') { + return false; + } + size_t klen = 0; + const char *kind = delta_field(line, len, CS_SCOPE_U_KIND, &klen); + return kind && klen > 0 && !memchr(kind, 'g', klen); +} + +/* The next scope line -- with `local_usings` only the file's own using + * directives, without it every other line; false at the end. */ +static bool delta_next_line(const char **cursor, bool local_usings, const char **line, + size_t *len) { + for (const char *p = *cursor; p && *p;) { + const char *nl = strchr(p, '\n'); + size_t n = nl ? (size_t)(nl - p) : strlen(p); + const char *next = nl ? nl + SKIP_ONE : p + n; + if (n > 0 && delta_local_using(p, n) == local_usings) { + *line = p; + *len = n; + *cursor = next; + return true; + } + p = next; + } + *cursor = NULL; + return false; +} + +/* True when the two scopes have the same own using directives, in order. */ +static bool delta_same_usings(const char *stored, const char *fresh) { + const char *sl = NULL; + const char *fl = NULL; + size_t slen = 0; + size_t flen = 0; + for (;;) { + bool hs = delta_next_line(&stored, true, &sl, &slen); + bool hf = delta_next_line(&fresh, true, &fl, &flen); + if (!hs || !hf) { + return hs == hf; + } + if (slen != flen || memcmp(sl, fl, slen) != 0) { + return false; + } + } +} + +/* True when the scope declares a type that has a base list. */ +static bool delta_has_bases(const char *scope) { + for (const char *p = scope; p && *p;) { + const char *nl = strchr(p, '\n'); + size_t n = nl ? (size_t)(nl - p) : strlen(p); + size_t blen = 0; + if (p[0] == 'T' && delta_field(p, n, CS_SCOPE_T_BASES, &blen) && blen > 0) { + return true; + } + p = nl ? nl + SKIP_ONE : NULL; + } + return false; +} + +/* Compare a changed file's stored and fresh scopes line by line, in order. + * Two differences leave every other file's resolution alone: + * - the file's own using directives (namespace, static, alias), as long as + * the file declares no type with a base list: they scope the file + * itself, and it is re-extracted anyway. A base list is resolved through + * them, and what a type derives from decides how OTHER files' references + * to its members come out; + * - a member the fresh scope no longer has. Every member line is a method, + * constructor, property, field, event, operator, indexer or enum member, + * so a reference that depended on it either bound it (an edge into this + * file) or names it in its unresolved row: the name is reported. + * Everything else is GLOBAL: a new or changed line (a type, a member, a + * signature, a namespace, a global using), a removed type (it may be another + * type's base or an alias target, which changes how references THROUGH those + * classify), a removed namespace, global using, quarantined name or unplaced + * range, and a changed order (same-path declarations own their node by + * order, and a record names its type by its position). + * + * An MSBuild project file's blob sets the global usings of every C# file of + * its project: any difference between two of those is GLOBAL. An edit that + * leaves the blob as it is -- a target, a package reference -- changes + * nobody's scope. */ +static int cs_scope_delta(const char *stored, const char *fresh, cbm_doclink_name_fn removed, + void *ud) { + /* a rejected file declares nothing: unchanged while it stays rejected */ + if (cbm_msb_is_project_scope(stored) || cbm_msb_is_project_scope(fresh) || + cs_scope_marked_rejected(stored) || cs_scope_marked_rejected(fresh)) { + return strcmp(stored, fresh) == 0 ? CBM_DOCLINK_DELTA_LOCAL : CBM_DOCLINK_DELTA_GLOBAL; + } + if (!delta_same_usings(stored, fresh) && (delta_has_bases(stored) || delta_has_bases(fresh))) { + return CBM_DOCLINK_DELTA_GLOBAL; + } + const char *sp = stored; + const char *fp = fresh; + const char *sl = NULL; + const char *fl = NULL; + size_t slen = 0; + size_t flen = 0; + bool hs = delta_next_line(&sp, false, &sl, &slen); + bool hf = delta_next_line(&fp, false, &fl, &flen); + while (hs || hf) { + if (hs && hf && slen == flen && memcmp(sl, fl, slen) == 0) { + hs = delta_next_line(&sp, false, &sl, &slen); + hf = delta_next_line(&fp, false, &fl, &flen); + continue; + } + if (!hs || sl[0] != 'M') { + return CBM_DOCLINK_DELTA_GLOBAL; + } + size_t nlen = 0; + const char *name = delta_field(sl, slen, CS_SCOPE_M_NAME, &nlen); + if (!name || nlen == 0 || !removed || !removed(ud, name, nlen)) { + return CBM_NOT_FOUND; /* not a member line this code wrote, or not recordable */ + } + hs = delta_next_line(&sp, false, &sl, &slen); + } + return CBM_DOCLINK_DELTA_LOCAL; +} + +/* XML is listed for the MSBuild project files: their scope blobs carry the C# + * tag, and the index needs the ones of this run as it needs the stored ones. + * An XML file that is no project file has no blob and no references. */ +const cbm_doclink_resolver_t cbm_doclink_cs_resolver = { + .langs = {CBM_LANG_CSHARP, CBM_LANG_XML}, + .lang_count = 2, + .scope_tag = CBM_DOCLINK_CS_SCOPE_TAG, + .build = cs_build, + .destroy = cs_destroy, + .file_begin = cs_file_begin, + .file_end = cs_file_end, + .resolve = cs_resolve, + .scope_delta = cs_scope_delta, + .scope_accepted = cs_scope_accepted, + .rejected_scope = cs_rejected_scope, +}; diff --git a/src/pipeline/doc_links_msbuild.c b/src/pipeline/doc_links_msbuild.c new file mode 100644 index 000000000..dfb5bd699 --- /dev/null +++ b/src/pipeline/doc_links_msbuild.c @@ -0,0 +1,3915 @@ +/* + * doc_links_msbuild.c — MSBuild global usings of a C# project (R1). + * + * A reader of data, never a build: nothing is executed, fetched or opened. + * The input is the scope blob the extractor writes for every MSBuild project + * file the index holds (internal/cbm/doclink_cs.c, "Project blob"). A file + * discovery did not take -- a symbolic link, a path outside the repository, + * an ignored file -- does not exist here. + * + * Evaluation model (MSBuild's passes, reduced): + * files the nearest Directory.Build.props above the project, the + * project file, the nearest Directory.Build.targets; an + * is evaluated where it stands, every file once + * pass 1 properties in evaluation order, later wins + * pass 2 items with the final properties; + * Static="true" names a type, Alias an alias; a Remove takes a + * target out wherever it was included + * implicit ImplicitUsings enable/true adds the default set of the + * project's SDK (Microsoft.NET.Sdk / .Web / .Worker) + * + * Nothing is guessed. A condition has three values: true, false, and UNKNOWN + * for what this reader cannot decide -- a property no evaluated file sets + * (the SDK, the environment or the command line may set it), a property + * function, an item or metadata reference, a relational operator, a function + * call, a comparison whose outcome depends on how MSBuild converts its + * operands. An element under an unknown condition is not applied and is + * counted. What it could have set is unknown from there on: its properties + * poison the conditions that read them, and a or an ImplicitUsings + * switch that cannot be evaluated leaves the project's usings open + * (cbm_msb_result_t.open). + * + * Paths: $(MSBuildThisFileDirectory) and $(MSBuildProjectDirectory) are + * absolute in MSBuild. Here they start with a mark byte followed by the + * repository-relative path, so an import through them is not joined to the + * importing file's directory a second time, and a path the file writes + * absolute itself (/usr/..., C:\...) stays what it is: outside. + */ +#include "pipeline/doc_links_msbuild.h" + +#include "doclink.h" /* CBM_DOCLINK_CS_SCOPE_TAG */ +#include "foundation/arena.h" +#include "foundation/constants.h" +#include "foundation/hash_table.h" +#include "foundation/mem_core.h" + +#include +#include +#include +#include +#include + +enum { + MSB_FIELDS = 5, + MSB_NAME_MAX = 256, /* a property name; a longer one never has a known value */ + MSB_VALUE_MAX = 4096, /* an expanded value; a longer one is unknown */ + MSB_COND_DEPTH = 32, /* parentheses of a condition; deeper is unknown */ + MSB_SIGNIFICANT = 15, /* the decimal digits a double holds exactly */ + MSB_PATH_MARK = 1, /* first byte of a repository-absolute path */ + MSB_INIT = 16, +}; + +typedef enum { MSB_FALSE = 0, MSB_TRUE = 1, MSB_UNKNOWN = 2 } msb_tri_t; + +/* ── The project files ───────────────────────────────────────────── */ + +typedef struct { + char tag; /* I G V W H N K Y C; W is a V whose value is no plain text */ + const char *f[MSB_FIELDS]; /* NULL: the attribute is not there */ +} msb_rec_t; + +typedef struct { + const char *rel_path; + const char *dir; /* "" for the repository root */ + const char *name; /* the file name */ + const char *stem; /* ... without its extension */ + const char *ext; /* ".csproj"; "" when it has none */ + const char *abs_dir; /* marked, with a trailing '/' */ + const char *abs_path; /* marked */ + const char *sdk; /* the attribute, or NULL */ + bool readable; + msb_rec_t *recs; + int nrecs; +} msb_file_t; + +struct cbm_msb { + CBMArena arena; + msb_file_t *files; + int nfiles; + int cap; + CBMHashTable *by_path; /* rel_path -> index + 1 */ + uint64_t generation; /* includes attempted additions that may partially mutate storage */ + bool oom; +}; + +bool cbm_msb_is_project_scope(const char *scope) { + static const char head[] = CBM_DOCLINK_CS_SCOPE_TAG "\nP\t"; + return scope && strncmp(scope, head, sizeof(head) - SKIP_ONE) == 0; +} + +cbm_msb_t *cbm_msb_new(void) { + cbm_msb_t *m = (cbm_msb_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*m)); + if (!m) { + return NULL; + } + cbm_arena_init(&m->arena); + m->by_path = cbm_ht_create(CBM_SZ_256); + if (!m->by_path) { + cbm_msb_free(m); + return NULL; + } + return m; +} + +void cbm_msb_free(cbm_msb_t *m) { + if (!m) { + return; + } + cbm_ht_free(m->by_path); + cbm_arena_destroy(&m->arena); + cbm_free(CBM_MEM_CLASS_OTHER, m); +} + +static void *msb_alloc(cbm_msb_t *m, size_t n) { + void *p = cbm_arena_alloc(&m->arena, n ? n : SKIP_ONE); + m->oom = m->oom || !p; + return p; +} + +/* a + b + c, in the arena. */ +static char *msb_join3(cbm_msb_t *m, const char *a, const char *b, const char *c) { + size_t al = strlen(a); + size_t bl = strlen(b); + size_t cl = strlen(c); + char *out = (char *)msb_alloc(m, al + bl + cl + SKIP_ONE); + if (out) { + memcpy(out, a, al); + memcpy(out + al, b, bl); + memcpy(out + al + bl, c, cl + SKIP_ONE); + } + return out; +} + +/* A blob field in place: NULL for an absent attribute, else the text after + * the '=' with its escapes resolved. */ +static char *msb_field(char *raw) { + if (!raw || raw[0] != '=') { + return NULL; + } + char *out = raw + SKIP_ONE; + char *w = out; + for (const char *r = out; *r; r++) { + if (*r == '\\' && r[1]) { + r++; + *w++ = *r == 't' ? '\t' : (*r == 'n' ? '\n' : (*r == 'r' ? '\r' : *r)); + } else { + *w++ = *r; + } + } + *w = '\0'; + return out; +} + +/* Cut a line (NUL-terminated, its tag in line[0]) into its fields in place. */ +static void msb_split(char *line, char *raw[MSB_FIELDS]) { + int n = 0; + for (int i = 0; i < MSB_FIELDS; i++) { + raw[i] = NULL; + } + for (char *p = line; *p && n < MSB_FIELDS; p++) { + if (*p == '\t') { + *p = '\0'; + raw[n++] = p + SKIP_ONE; + } + } +} + +/* One line of the blob as a record. */ +static void msb_parse_record(char *line, msb_rec_t *r) { + char *raw[MSB_FIELDS]; + msb_split(line, raw); + memset(r, 0, sizeof(*r)); + r->tag = line[0]; + if (r->tag == 'V') { + r->f[0] = msb_field(raw[0]); + r->f[1] = raw[1] ? raw[1] : ""; + if (raw[2] && raw[2][0] == '=') { + r->f[2] = msb_field(raw[2]); + } else { + r->tag = 'W'; + } + } else if (r->tag == 'K' || r->tag == 'C') { + r->f[0] = raw[0] ? raw[0] : ""; + } else { + for (int i = 0; i < MSB_FIELDS; i++) { + r->f[i] = msb_field(raw[i]); + } + } +} + +/* Fill the names derived from the file's path. */ +static void msb_file_names(cbm_msb_t *m, msb_file_t *f) { + const char *slash = strrchr(f->rel_path, '/'); + f->name = slash ? slash + SKIP_ONE : f->rel_path; + char *dir = + cbm_arena_strndup(&m->arena, f->rel_path, slash ? (size_t)(slash - f->rel_path) : 0); + const char *dot = strrchr(f->name, '.'); + char *stem = + cbm_arena_strndup(&m->arena, f->name, dot ? (size_t)(dot - f->name) : strlen(f->name)); + m->oom = m->oom || !dir || !stem; + f->dir = dir ? dir : ""; + f->stem = stem ? stem : ""; + f->ext = dot ? dot : ""; + static const char mark[] = {MSB_PATH_MARK, '/', '\0'}; + f->abs_dir = msb_join3(m, mark, f->dir, f->dir[0] ? "/" : ""); + f->abs_path = msb_join3(m, mark, f->rel_path, ""); +} + +bool cbm_msb_add(cbm_msb_t *m, const char *rel_path, const char *scope) { + if (!m || m->oom || !rel_path) { + return false; + } + if (!cbm_msb_is_project_scope(scope) || cbm_ht_get(m->by_path, rel_path)) { + return true; /* no project file, or one that is there already */ + } + /* Invalidate derived state before any allocation or partial mutation. */ + m->generation++; + if (m->nfiles >= m->cap) { + int ncap = m->cap ? m->cap * PAIR_LEN : CBM_SZ_64; + msb_file_t *grown = (msb_file_t *)msb_alloc(m, (size_t)ncap * sizeof(*grown)); + if (!grown) { + return false; + } + if (m->nfiles > 0) { + memcpy(grown, m->files, (size_t)m->nfiles * sizeof(*grown)); + } + m->files = grown; + m->cap = ncap; + } + msb_file_t *f = &m->files[m->nfiles]; + memset(f, 0, sizeof(*f)); + f->rel_path = cbm_arena_strdup(&m->arena, rel_path); + char *buf = cbm_arena_strdup(&m->arena, scope); + if (!f->rel_path || !buf) { + m->oom = true; + return false; + } + msb_file_names(m, f); + int lines = 0; + for (const char *p = buf; *p; p++) { + lines += *p == '\n'; + } + f->recs = (msb_rec_t *)msb_alloc(m, (size_t)(lines + SKIP_ONE) * sizeof(msb_rec_t)); + if (m->oom) { + return false; + } + /* line 0 is the tag, line 1 the P record */ + int line_no = 0; + bool import_group = false; + bool bad_group = false; + const char *group_cond = NULL; + for (char *line = buf; line && *line; line_no++) { + char *nl = strchr(line, '\n'); + if (nl) { + *nl = '\0'; + } + if (line_no == SKIP_ONE) { + char *raw[MSB_FIELDS]; + msb_split(line, raw); /* P sdk state */ + f->sdk = msb_field(raw[0]); + f->readable = raw[1] && raw[1][0] == '-'; + } else if (line_no > SKIP_ONE) { + msb_rec_t *r = &f->recs[f->nrecs]; + msb_parse_record(line, r); + if (import_group && r->tag != 'J' && r->tag != 'E') { + bad_group = true; + } + if (r->tag == 'B') { + bad_group = bad_group || line[1] || r->f[1] || r->f[2] || r->f[3] || r->f[4]; + import_group = true; + group_cond = r->f[0]; + } else if (r->tag == 'E') { + bad_group = bad_group || !import_group || line[1] || r->f[0] || r->f[1] || + r->f[2] || r->f[3] || r->f[4]; + import_group = false; + group_cond = NULL; + } else { + if (r->tag == 'J') { + bad_group = bad_group || !import_group || line[1] || r->f[0] || r->f[4]; + /* The decoded text lives in buf, not in the reused record. + * Each import still evaluates it at its original position. */ + r->tag = 'I'; + r->f[0] = group_cond; + } + f->nrecs++; + } + if (bad_group) { + break; + } + } + line = nl ? nl + SKIP_ONE : NULL; + } + if (bad_group || import_group || !f->readable) { + /* A malformed group is unknown, never an empty closed scope. Keep + * this project conservative without rejecting other project files. + * So is a file the scan did not read (malformed, or larger than a + * project file): what it holds is unknown, not absent -- the + * projects that evaluate it have an open scope. */ + f->readable = false; + f->recs[0] = (msb_rec_t){.tag = 'Y'}; + f->nrecs = 1; + } + cbm_ht_set(m->by_path, f->rel_path, (void *)(intptr_t)(m->nfiles + SKIP_ONE)); + m->nfiles++; + return !m->oom; +} + +static void msb_work(uint64_t n); + +static int msb_file_index(const cbm_msb_t *m, const char *rel_path) { + msb_work(SKIP_ONE); /* one logical path lookup, not hash-table bucket probes */ + intptr_t v = (intptr_t)cbm_ht_get(m->by_path, rel_path); + return v > 0 ? (int)(v - SKIP_ONE) : CBM_NOT_FOUND; +} + +bool cbm_msb_has(const cbm_msb_t *m, const char *rel_path) { + return m && rel_path && msb_file_index(m, rel_path) >= 0; +} + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +#include +static _Atomic uint64_t msb_record_counter; +static _Atomic uint64_t msb_work_counter; +static _Atomic uint64_t msb_nearest_counter; +static _Atomic uint64_t msb_peak_counter; +static _Atomic int msb_value_alloc_after; +static _Atomic bool msb_value_alloc_failed; +static _Atomic int msb_item_fail_operation; +static _Atomic int msb_item_fail_after; +static _Atomic bool msb_item_operation_failed; +static _Atomic int msb_node_fail_operation; +static _Atomic int msb_node_fail_after; +static _Atomic bool msb_node_alloc_failed; +static _Atomic uint64_t msb_state_allocations; +static _Atomic uint64_t msb_state_slot_reuses; +static _Atomic uint64_t msb_state_witnessed_slot_reuses; +static _Atomic uint64_t msb_state_same_length_value_changes; +static _Atomic uint64_t msb_state_witness_skips; +static _Atomic uint64_t msb_state_revision_errors; + +void cbm_msb_test_fail_node_alloc(cbm_msb_node_fail_operation_t operation, int nth) { + atomic_store(&msb_node_fail_operation, (int)operation); + atomic_store(&msb_node_fail_after, nth); + atomic_store(&msb_node_alloc_failed, false); +} + +bool cbm_msb_test_node_alloc_failed(void) { + return atomic_load(&msb_node_alloc_failed); +} + +void cbm_msb_test_state_stats(cbm_msb_state_test_stats_t *out) { + out->allocations = atomic_load(&msb_state_allocations); + out->slot_reuses = atomic_load(&msb_state_slot_reuses); + out->witnessed_slot_reuses = atomic_load(&msb_state_witnessed_slot_reuses); + out->same_length_value_changes = atomic_load(&msb_state_same_length_value_changes); + out->witness_skips = atomic_load(&msb_state_witness_skips); + out->revision_errors = atomic_load(&msb_state_revision_errors); +} + +void cbm_msb_test_fail_item_operation(cbm_msb_item_fail_operation_t operation, int nth) { + atomic_store(&msb_item_fail_operation, (int)operation); + atomic_store(&msb_item_fail_after, nth); + atomic_store(&msb_item_operation_failed, false); +} + +bool cbm_msb_test_item_operation_failed(void) { + return atomic_load(&msb_item_operation_failed); +} + +static bool msb_fail_item_operation(cbm_msb_item_fail_operation_t operation) { + if (operation == CBM_MSB_ITEM_FAIL_NONE || + atomic_load(&msb_item_fail_operation) != (int)operation) { + return false; + } + int n = atomic_load(&msb_item_fail_after); + while (n > 0) { + if (atomic_compare_exchange_weak(&msb_item_fail_after, &n, n - SKIP_ONE)) { + if (n == SKIP_ONE) { + atomic_store(&msb_item_operation_failed, true); + return true; + } + return false; + } + } + return false; +} + +void cbm_msb_test_fail_value_alloc_after(int nth) { + atomic_store(&msb_value_alloc_after, nth); + atomic_store(&msb_value_alloc_failed, false); +} + +bool cbm_msb_test_value_alloc_failed(void) { + return atomic_load(&msb_value_alloc_failed); +} + +static bool msb_fail_value_alloc(void) { + int n = atomic_load(&msb_value_alloc_after); + while (n > 0) { + if (atomic_compare_exchange_weak(&msb_value_alloc_after, &n, n - SKIP_ONE)) { + if (n == SKIP_ONE) { + atomic_store(&msb_value_alloc_failed, true); + return true; + } + return false; + } + } + return false; +} + +static _Atomic int msb_prop_insert_after; +static _Atomic bool msb_prop_insert_failed; +static _Atomic uint64_t msb_value_live_counter; + +void cbm_msb_test_fail_prop_insert_after(int nth) { + atomic_store(&msb_prop_insert_after, nth); + atomic_store(&msb_prop_insert_failed, false); +} + +bool cbm_msb_test_prop_insert_failed(void) { + return atomic_load(&msb_prop_insert_failed); +} + +uint64_t cbm_msb_test_value_live_bytes(void) { + return atomic_load(&msb_value_live_counter); +} + +static bool msb_fail_prop_insert(void) { + int n = atomic_load(&msb_prop_insert_after); + while (n > 0) { + if (atomic_compare_exchange_weak(&msb_prop_insert_after, &n, n - SKIP_ONE)) { + if (n == SKIP_ONE) { + atomic_store(&msb_prop_insert_failed, true); + return true; + } + return false; + } + } + return false; +} + +void cbm_msb_test_cost_reset(void) { + atomic_store(&msb_record_counter, 0); + atomic_store(&msb_work_counter, 0); + atomic_store(&msb_nearest_counter, 0); + atomic_store(&msb_peak_counter, 0); + atomic_store(&msb_state_allocations, 0); + atomic_store(&msb_state_slot_reuses, 0); + atomic_store(&msb_state_witnessed_slot_reuses, 0); + atomic_store(&msb_state_same_length_value_changes, 0); + atomic_store(&msb_state_witness_skips, 0); + atomic_store(&msb_state_revision_errors, 0); +} + +void cbm_msb_test_cost(uint64_t *records, uint64_t *peak_bytes) { + *records = atomic_load(&msb_record_counter); + *peak_bytes = atomic_load(&msb_peak_counter); +} + +uint64_t cbm_msb_test_work(void) { + return atomic_load(&msb_work_counter); +} + +uint64_t cbm_msb_test_nearest_steps(void) { + return atomic_load(&msb_nearest_counter); +} + +static void msb_nearest_step(void) { + atomic_fetch_add_explicit(&msb_nearest_counter, SKIP_ONE, memory_order_relaxed); +} + +static void msb_work(uint64_t n) { + atomic_fetch_add_explicit(&msb_work_counter, n, memory_order_relaxed); +} + +static void msb_record(void) { + atomic_fetch_add_explicit(&msb_record_counter, SKIP_ONE, memory_order_relaxed); + msb_work(SKIP_ONE); +} + +static void msb_peak(uint64_t bytes) { + uint64_t old = atomic_load(&msb_peak_counter); + while (old < bytes && !atomic_compare_exchange_weak(&msb_peak_counter, &old, bytes)) {} +} +#else +static void msb_nearest_step(void) {} +static void msb_work(uint64_t n) { + (void)n; +} +static void msb_record(void) {} +static bool msb_fail_item_operation(cbm_msb_item_fail_operation_t operation) { + (void)operation; + return false; +} +#endif + +static bool msb_fail_node_alloc(cbm_msb_node_fail_operation_t operation) { +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (operation == CBM_MSB_NODE_FAIL_NONE || + atomic_load(&msb_node_fail_operation) != (int)operation) + return false; + int n = atomic_load(&msb_node_fail_after); + while (n > 0) { + if (atomic_compare_exchange_weak(&msb_node_fail_after, &n, n - 1)) { + if (n == 1) { + atomic_store(&msb_node_alloc_failed, true); + return true; + } + return false; + } + } +#else + (void)operation; +#endif + return false; +} + +typedef struct st_node st_node_t; +typedef struct st_rope st_rope_t; +typedef struct st_component st_component_t; +typedef struct msb_state msb_state_t; +typedef struct { + st_node_t *root[3]; +} msb_view_t; + +/* ── One evaluation ──────────────────────────────────────────────── */ + +typedef struct { + char *value; /* owned current value; NULL: set, but to nothing this reader knows */ + size_t capacity; +} msb_prop_t; + +typedef struct { + const msb_rec_t *group; /* the 's H record */ + const msb_rec_t *item; /* the 's N record */ + int file; +} msb_item_t; + +typedef struct { + int file; + int rec; + st_component_t *owner; + msb_tri_t group; /* the condition of the being read */ + const msb_rec_t *hgroup; /* the being read */ + /* The being read: its condition's text (one string the + * group's imports share) and what it came to. MSBuild evaluates a + * group's condition once, where the group stands; evaluating it again + * for every import cost imports x condition length. */ + const char *igroup_text; + msb_tri_t igroup; +} msb_frame_t; + +enum { MSB_SEEDS = 5, MSB_LIVE_OWNERS = 5 }; +static const char *const MSB_SEED_NAMES[MSB_SEEDS] = { + "msbuildprojectname", "msbuildprojectfile", "msbuildprojectextension", "msbuildprojectfullpath", + "msbuildprojectdirectory"}; + +typedef struct { + bool present; /* absent and present-unknown are different dependencies */ + const char *value; + size_t bytes; + bool exception; /* expected state differs from the immutable input layers */ +} msb_prop_dep_t; + +typedef struct { + bool present; + bool exception; +} msb_set_dep_t; + +typedef struct { + CBMHashTable *props; + CBMHashTable *seen; + CBMHashTable *poisoned; + msb_prop_dep_t *seeds[MSB_SEEDS]; + size_t prop_exceptions; + size_t seen_exceptions; + size_t poisoned_exceptions; +} msb_inputs_t; + +typedef struct msb_eval msb_eval_t; +typedef struct { + const msb_eval_t *owners[MSB_LIVE_OWNERS]; /* prefix, target, project, seeds, items */ + const size_t *metadata_bytes; /* live directory-memo payloads, including pending insertion */ + const size_t *state_bytes; + const CBMArena *state_names; +} msb_live_t; + +struct msb_eval { + const cbm_msb_t *m; + msb_state_t *state; + msb_view_t view, state_inputs; + st_node_t *reuse_nodes; /* operation-local identity reuse, never a lookup layer */ + st_component_t *builder, *completed; + int main_project; + const msb_eval_t *base; /* immutable completed prefix, never written through */ + const msb_eval_t *seeds; /* this project's five initial properties */ + const msb_eval_t *effects; /* completed target writes, highest precedence in pass 2 */ + msb_eval_t *incoming; /* read-only project input while capturing effects or final items */ + const msb_live_t *live; /* simultaneous owners for peak accounting */ + msb_inputs_t inputs; + CBMHashTable *shared_removals; /* borrowed exact item result, only during publication */ + bool capture_inputs; + cbm_msb_item_fail_operation_t item_alloc_operation; + bool track_seed_reads; + bool seed_read; + CBMArena arena; /* keys, property owners, frames, items and output copies */ + CBMArena scratch; /* expressions and paths, released after their consumers finish */ + CBMHashTable *props; /* lower-cased name -> msb_prop_t* */ + CBMHashTable *seen; /* rel_path: files evaluated */ + CBMHashTable *poisoned; /* rel_path: files whose properties were made unknown */ + msb_item_t *items; + int nitems; + int cap_items; + msb_frame_t *frames; + int nframes; + int cap_frames; + bool open; + int unevaluable; + int outside; + size_t result_bytes; + size_t value_bytes; /* live property buffers, including a replacement before publication */ + bool oom; +}; + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +static size_t ev_owned_bytes(const msb_eval_t *ev) { + return ev ? cbm_arena_capacity(&ev->arena) + cbm_arena_capacity(&ev->scratch) + + ev->value_bytes + ev->result_bytes + : 0; +} +#endif + +static void ev_peak(const msb_eval_t *ev) { +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + size_t bytes = ev_owned_bytes(ev); + if (ev->live) { + bytes = ev->live->metadata_bytes ? *ev->live->metadata_bytes : 0; + if (ev->live->state_bytes) + bytes += *ev->live->state_bytes; + if (ev->live->state_names) + bytes += cbm_arena_capacity(ev->live->state_names); + for (int i = 0; i < MSB_LIVE_OWNERS; i++) { + bytes += ev_owned_bytes(ev->live->owners[i]); + } + } + msb_peak(bytes); +#else + (void)ev; +#endif +} + +static void *ev_alloc(msb_eval_t *ev, size_t n) { + void *p = msb_fail_item_operation(ev->item_alloc_operation) + ? NULL + : cbm_arena_alloc(&ev->arena, n ? n : SKIP_ONE); + ev->oom = ev->oom || !p; + ev_peak(ev); + return p; +} + +static char *ev_strndup(msb_eval_t *ev, const char *s, size_t n) { + char *p = msb_fail_item_operation(ev->item_alloc_operation) + ? NULL + : cbm_arena_strndup(&ev->arena, s, n); + if (p) { + msb_work(n + SKIP_ONE); + } + ev->oom = ev->oom || !p; + ev_peak(ev); + return p; +} + +static void *scratch_alloc(msb_eval_t *ev, size_t n) { + void *p = msb_fail_item_operation(ev->item_alloc_operation) + ? NULL + : cbm_arena_alloc(&ev->scratch, n ? n : SKIP_ONE); + ev->oom = ev->oom || !p; + ev_peak(ev); + return p; +} + +static char *scratch_strndup(msb_eval_t *ev, const char *s, size_t n) { + char *p = msb_fail_item_operation(ev->item_alloc_operation) + ? NULL + : cbm_arena_strndup(&ev->scratch, s, n); + if (p) { + msb_work(n + SKIP_ONE); + } + ev->oom = ev->oom || !p; + ev_peak(ev); + return p; +} + +/* Every property buffer is owned by a table entry. Account for the new and + * old buffers simultaneously until copying has finished and the old one is + * released; neither a replacement nor an unknown value retains history. */ +static char *value_alloc(msb_eval_t *ev, size_t n) { + bool fail = n > SIZE_MAX - ev->value_bytes; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + fail = fail || msb_fail_value_alloc(); +#endif + if (fail) { + ev->oom = true; + return NULL; + } + char *p = (char *)cbm_alloc(CBM_MEM_CLASS_OTHER, n); + if (!p) { + ev->oom = true; + return NULL; + } + ev->value_bytes += n; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + atomic_fetch_add_explicit(&msb_value_live_counter, n, memory_order_relaxed); +#endif + ev_peak(ev); + return p; +} + +static void value_clear(msb_eval_t *ev, msb_prop_t *p) { + if (p->value) { + cbm_free(CBM_MEM_CLASS_OTHER, p->value); + ev->value_bytes -= p->capacity; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + atomic_fetch_sub_explicit(&msb_value_live_counter, p->capacity, memory_order_relaxed); +#endif + } + p->value = NULL; + p->capacity = 0; +} + +static void prop_clear(const char *key, void *value, void *userdata) { + (void)key; + msb_work(SKIP_ONE); /* every property owner visited during teardown */ + value_clear((msb_eval_t *)userdata, (msb_prop_t *)value); +} + +/* The table key of a property name; false for a name too long to keep. */ +static bool msb_key(const char *name, size_t len, char key[MSB_NAME_MAX]) { + if (len == 0 || len >= MSB_NAME_MAX) { + return false; + } + for (size_t i = 0; i < len; i++) { + key[i] = (char)tolower((unsigned char)name[i]); + } + key[len] = '\0'; + msb_work(len + SKIP_ONE); /* lower-casing and the terminator */ + return true; +} + +static void st_set_property(msb_eval_t *ev, const char *name, const char *value); +static const msb_prop_t *st_property(msb_eval_t *ev, const char *key); + +/* Set a property; value NULL makes it unknown. */ +static void msb_set(msb_eval_t *ev, const char *name, const char *value) { + if (ev->state) { + st_set_property(ev, name, value); + return; + } + if (ev->oom) { + return; + } + char key[MSB_NAME_MAX]; + if (!msb_key(name, strlen(name), key)) { + return; + } + msb_work(SKIP_ONE); + msb_prop_t *p = (msb_prop_t *)cbm_ht_get(ev->props, key); + if (!p) { + p = (msb_prop_t *)ev_alloc(ev, sizeof(*p)); + if (!p) { + return; + } + memset(p, 0, sizeof(*p)); + char *k = ev_strndup(ev, key, strlen(key)); + if (!k) { + return; + } +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (!msb_fail_prop_insert()) +#endif + { + msb_work(SKIP_ONE); + cbm_ht_set(ev->props, k, p); + } + /* set returns the old value, so NULL alone cannot prove insertion. + * Acquire no heap buffer until cleanup can reach this owner. */ + msb_work(SKIP_ONE); + if (cbm_ht_get(ev->props, k) != p) { + ev->oom = true; + return; + } + } + if (!value) { + value_clear(ev, p); + return; + } + size_t bytes = strlen(value) + SKIP_ONE; + if (p->capacity == bytes) { + memmove(p->value, value, bytes); + msb_work(bytes); + return; + } + char *replacement = value_alloc(ev, bytes); + if (!replacement) { + return; /* sticky OOM prevents publishing a partial evaluation */ + } + memcpy(replacement, value, bytes); + msb_work(bytes); + value_clear(ev, p); + p->value = replacement; + p->capacity = bytes; +} + +/* Compare exact input state. Known-empty is a one-byte owned value; + * unknown and absent remain distinct even though both expand as unknown. */ +static bool prop_dep_matches(const msb_prop_dep_t *dep, const msb_prop_t *p) { + msb_work(SKIP_ONE); + if (dep->present != (p != NULL)) { + return false; + } + if (!p) { + return true; + } + if ((dep->value != NULL) != (p->value != NULL)) { + return false; + } + if (!p->value) { + return true; + } + if (dep->bytes != p->capacity) { + return false; + } + msb_work(dep->bytes); + return memcmp(dep->value, p->value, dep->bytes) == 0; +} + +/* Context-owned persistent state. Compressed radix branches split on property + * or file IDs; identity belongs to immutable content, never to a pool address. */ +enum { ST_PROPS, ST_SEEN, ST_POISON, ST_DOMAINS, ST_SLOTS = 64 }; +typedef struct st_value { + unsigned refs; + msb_prop_t prop; +} st_value_t; +typedef struct st_slab st_slab_t; +struct st_node { + st_node_t *child[2]; + st_node_t *next; + st_slab_t *slab; + st_value_t *value; + uint64_t revision; + uint64_t witness; + uint64_t epoch; + uint32_t key; + unsigned refs; + int bit; + unsigned char domain; + bool dependency; + bool present; + bool all_present; + bool all_absent; + bool witness_valid; + bool witnessed; +}; +struct st_slab { + st_slab_t *next, *prev; + st_slab_t *available_next, *available_prev; + st_slab_t *empty_next, *empty_prev; + bool on_empty; + st_node_t *free; + unsigned used; + st_node_t slots[ST_SLOTS]; +}; +struct st_rope { + st_rope_t *left, *right, *next; + msb_item_t item; + unsigned refs; + int height; + int count; +}; +struct st_component { + msb_view_t writes, inputs; + st_rope_t *items; + st_component_t *parent; + int file; + unsigned refs; + bool poison, retain; + bool open; + int unevaluable, outside; + bool start_open; + int start_unevaluable, start_outside; +}; +struct msb_state { + CBMHashTable *symbols; + CBMArena names; + uint32_t next_symbol; + st_slab_t *slabs, *available, *empty; + st_component_t **components; + int files; + uint64_t generation, epoch, revision, audited_revision; + size_t bytes; + st_rope_t *item_a, *item_b; + st_rope_t *capture_a, *capture_b; +}; + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +#define ST_STAT(name) atomic_fetch_add_explicit(&msb_state_##name, 1, memory_order_relaxed) +#else +#define ST_STAT(name) ((void)0) +#endif + +static st_node_t *st_hold(st_node_t *n) { + if (n) { + n->refs++; + msb_work(SKIP_ONE); + } + return n; +} +static st_value_t *st_value_hold(st_value_t *v) { + if (v) { + v->refs++; + msb_work(SKIP_ONE); + } + return v; +} +static void st_value_drop(msb_state_t *s, st_value_t *v) { + if (v) { + msb_work(SKIP_ONE); + if (--v->refs == 0) { + s->bytes -= sizeof(*v) + v->prop.capacity; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + atomic_fetch_sub_explicit(&msb_value_live_counter, v->prop.capacity, + memory_order_relaxed); +#endif + cbm_free(CBM_MEM_CLASS_OTHER, v); + } + } +} +static void st_available_remove(msb_state_t *s, st_slab_t *b) { + if (b->available_prev) + b->available_prev->available_next = b->available_next; + else + s->available = b->available_next; + if (b->available_next) + b->available_next->available_prev = b->available_prev; + b->available_next = b->available_prev = NULL; +} +static void st_available_add(msb_state_t *s, st_slab_t *b) { + b->available_next = s->available; + if (s->available) + s->available->available_prev = b; + s->available = b; +} +static void st_empty_remove(msb_state_t *s, st_slab_t *b) { + if (!b->on_empty) + return; + if (b->empty_prev) + b->empty_prev->empty_next = b->empty_next; + else + s->empty = b->empty_next; + if (b->empty_next) + b->empty_next->empty_prev = b->empty_prev; + b->empty_next = b->empty_prev = NULL; + b->on_empty = false; +} +static void st_empty_add(msb_state_t *s, st_slab_t *b) { + b->empty_next = s->empty; + b->empty_prev = NULL; + if (s->empty) + s->empty->empty_prev = b; + s->empty = b; + b->on_empty = true; +} +/* Intrusive zero-ref queue: neither radix nor component/rope depth consumes + * the C call stack, and returning slots cannot allocate. Slabs stay alive. */ +static void st_drop(msb_state_t *s, st_node_t *n) { + st_node_t *pending = NULL; + if (n && --n->refs == 0) { + n->next = pending; + pending = n; + } + while (pending) { + n = pending; + pending = n->next; + msb_work(SKIP_ONE); + for (int i = 0; i < PAIR_LEN; i++) { + st_node_t *c = n->child[i]; + if (c && --c->refs == 0) { + c->next = pending; + pending = c; + } + } + st_value_drop(s, n->value); + st_slab_t *b = n->slab; + if (!b->free) + st_available_add(s, b); + n->next = b->free; + b->free = n; + if (--b->used == 0) + st_empty_add(s, b); + } +} +static void st_view_drop(msb_state_t *s, msb_view_t *v) { + for (int d = 0; d < ST_DOMAINS; d++) { + st_drop(s, v->root[d]); + v->root[d] = NULL; + } +} +static st_node_t *st_new(msb_eval_t *ev, cbm_msb_node_fail_operation_t phase) { + msb_state_t *s = ev->state; + if (ev->oom || s->revision == UINT64_MAX) { + ev->oom = true; + return NULL; + } + /* This is a real slot acquisition. Empty/identity operations never call it. */ + if (msb_fail_node_alloc(phase) || msb_fail_item_operation(ev->item_alloc_operation)) { + ev->oom = true; + return NULL; + } + st_slab_t *b = s->available; + if (!b) { + b = (st_slab_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*b)); + if (!b) { + ev->oom = true; + return NULL; + } + s->bytes += sizeof(*b); + b->next = s->slabs; + if (s->slabs) + s->slabs->prev = b; + s->slabs = b; + for (int i = ST_SLOTS; i > 0;) { + st_node_t *n = &b->slots[--i]; + n->slab = b; + n->next = b->free; + b->free = n; + } + st_available_add(s, b); + msb_work(sizeof(*b)); + ev_peak(ev); + } + st_empty_remove(s, b); + st_node_t *n = b->free; + uint64_t previous = n->revision; + bool witnessed = n->witnessed; + b->free = n->next; + b->used++; + if (!b->free) + st_available_remove(s, b); + memset(n, 0, sizeof(*n)); + n->slab = b; + n->refs = 1; + n->bit = -1; + n->revision = ++s->revision; + n->epoch = s->epoch; + ST_STAT(allocations); + if (previous) { + ST_STAT(slot_reuses); + if (witnessed) + ST_STAT(witnessed_slot_reuses); + } + if (!n->revision || n->revision <= previous || n->revision <= s->audited_revision) { + ST_STAT(revision_errors); + } + s->audited_revision = n->revision; + msb_work(sizeof(*n)); + return n; +} +static int st_high_bit(uint32_t key) { + int bit = -1; + while (key) { + key >>= 1; + bit++; + msb_work(SKIP_ONE); + } + return bit; +} +static bool st_same_range(uint32_t a, uint32_t b, int bit) { + return bit == 31 || (a >> (bit + 1)) == (b >> (bit + 1)); +} +static st_node_t *st_get(st_node_t *n, uint32_t key) { + while (n && n->bit >= 0) { + msb_work(SKIP_ONE); + if (!st_same_range(n->key, key, n->bit)) + return NULL; + n = n->child[(key >> n->bit) & 1]; + } + msb_work(SKIP_ONE); + return n && n->key == key ? n : NULL; +} +/* Restrict a map to the exact dependency prefix, not merely its depth. */ +static st_node_t *st_region(st_node_t *n, uint32_t key, int bit) { + while (n && n->bit > bit) { + msb_work(SKIP_ONE); + if (!st_same_range(n->key, key, n->bit)) + return NULL; + n = n->child[(key >> n->bit) & 1]; + } + msb_work(SKIP_ONE); + return n && st_same_range(n->key, key, bit) ? n : NULL; +} +static bool st_value_equal(st_value_t *a, st_value_t *b) { + msb_work(SKIP_ONE); + if (a == b) + return true; + if (!a || !b || a->prop.capacity != b->prop.capacity) + return false; + msb_work(a->prop.capacity); + return memcmp(a->prop.value, b->prop.value, a->prop.capacity) == 0; +} +static bool st_expected_equal(st_node_t *a, st_node_t *b) { + return a->present == b->present && st_value_equal(a->value, b->value); +} +static void st_certify(msb_eval_t *ev, st_node_t *dep, st_node_t *view, int domain); +static st_node_t *st_branch(msb_eval_t *ev, st_node_t *l, st_node_t *r, int bit, bool dep, + int domain, cbm_msb_node_fail_operation_t phase) { + if (!l) + return st_hold(r); + if (!r) + return st_hold(l); + if (!dep && ev->reuse_nodes) { + /* One borrowed root exists only during this operation. A bounded + * radix lookup may reuse an exact state branch; witnesses do not + * establish equivalence, and no pair table is retained. */ + st_node_t *same = st_region(ev->reuse_nodes, l->key, bit); + if (same && !same->dependency && same->domain == domain && + same->epoch == ev->state->epoch && same->bit == bit && same->key == l->key && + same->child[0] == l && same->child[1] == r) + return st_hold(same); + } + st_node_t *n = st_new(ev, phase); + if (n) { + n->child[0] = st_hold(l); + n->child[1] = st_hold(r); + n->bit = bit; + n->key = l->key; + n->dependency = dep; + n->domain = (unsigned char)domain; + n->all_present = l->all_present && r->all_present; + n->all_absent = dep && l->all_absent && r->all_absent; + /* A fresh join starts without a witness. Certify every constraint + * explicitly before recording an aligned witness for this branch. */ + if (dep) + st_certify(ev, n, ev->view.root[domain], domain); + } + return n; +} +/* Immutable ordered overlay/union. Each recursive call eliminates a radix + * level. Unchanged/empty subtrees are shared; there is no pair-operation cache. */ +static st_node_t *st_union(msb_eval_t *ev, st_node_t *a, st_node_t *b, bool deps, int domain, + cbm_msb_node_fail_operation_t phase) { + msb_work(SKIP_ONE); + if (a == b || !b) + return st_hold(a); + if (!a) + return st_hold(b); + if (ev->oom) + return NULL; + int split = st_high_bit(a->key ^ b->key); + int top = a->bit > b->bit ? a->bit : b->bit; + if (split > top) { + return ((a->key >> split) & 1) ? st_branch(ev, b, a, split, deps, domain, phase) + : st_branch(ev, a, b, split, deps, domain, phase); + } + if (top < 0) { + if (deps && !st_expected_equal(a, b)) { + ev->oom = true; + return NULL; + } + return st_hold(a); + } + st_node_t *ac[2] = {NULL, NULL}, *bc[2] = {NULL, NULL}; + if (a->bit == top) { + ac[0] = a->child[0]; + ac[1] = a->child[1]; + } else + ac[(a->key >> top) & 1] = a; + if (b->bit == top) { + bc[0] = b->child[0]; + bc[1] = b->child[1]; + } else + bc[(b->key >> top) & 1] = b; + st_node_t *l = st_union(ev, ac[0], bc[0], deps, domain, phase); + st_node_t *r = st_union(ev, ac[1], bc[1], deps, domain, phase); + st_node_t *out = NULL; + if (!ev->oom) { + if (a->bit == top && l == a->child[0] && r == a->child[1]) + out = st_hold(a); + else if (b->bit == top && l == b->child[0] && r == b->child[1]) + out = st_hold(b); + else + out = st_branch(ev, l, r, top, deps, domain, phase); + } + st_drop(ev->state, l); + st_drop(ev->state, r); + if (out && out != a && a->bit == top && out->revision == a->revision) + ST_STAT(revision_errors); + if (out && out != b && b->bit == top && out->revision == b->revision) + ST_STAT(revision_errors); + return out; +} +static bool st_validate(msb_eval_t *ev, st_node_t *dep, st_node_t *view, int domain) { + msb_work(SKIP_ONE); + if (!dep) + return true; + st_node_t *at = st_region(view, dep->key, dep->bit); + /* Exact absence constraints match an empty aligned range independently + * of any historical revision witness. Present unknown is not absent. */ + if (!at && dep->dependency && dep->all_absent && dep->domain == domain && + dep->epoch == ev->state->epoch) + return true; + if (dep->witness_valid && dep->epoch == ev->state->epoch && dep->domain == domain && + dep->witness == (at ? at->revision : 0)) { + ST_STAT(witness_skips); + return true; + } + if (dep->bit < 0) { + if (dep->present != (at != NULL)) + return false; + return !at || st_value_equal(dep->value, at->value); + } + return st_validate(ev, dep->child[0], at, domain) && st_validate(ev, dep->child[1], at, domain); +} +/* Only unpublished nodes may gain a witness. All constraints, kind, epoch + * and aligned range are checked; irrelevant historical state is not retained. */ +static void st_certify(msb_eval_t *ev, st_node_t *dep, st_node_t *view, int domain) { + if (!dep || dep->refs != 1 || dep->witness_valid) + return; + if (st_validate(ev, dep, view, domain)) { + st_node_t *at = st_region(view, dep->key, dep->bit); + dep->witness = at ? at->revision : 0; + dep->witness_valid = true; + dep->epoch = ev->state->epoch; + dep->domain = (unsigned char)domain; + if (at) + at->witnessed = true; + } +} +static st_node_t *st_mask(msb_eval_t *ev, st_node_t *dep, st_node_t *writes, int domain) { + msb_work(SKIP_ONE); + if (!dep || !writes) + return st_hold(dep); + if (dep == writes) + return NULL; + st_node_t *at = st_region(writes, dep->key, dep->bit); + if (!at) + return st_hold(dep); + /* An aligned witness plus all-present constraints proves that every + * dependency key is written here, without walking a shared key domain. */ + if (dep->all_present && dep->witness_valid && dep->epoch == ev->state->epoch && + dep->domain == domain && dep->witness == at->revision) + return NULL; + if (dep->bit < 0) + return NULL; + st_node_t *l = st_mask(ev, dep->child[0], at, domain); + st_node_t *r = st_mask(ev, dep->child[1], at, domain); + st_node_t *out = NULL; + if (!ev->oom) { + if (l == dep->child[0] && r == dep->child[1]) + out = st_hold(dep); + else { + out = st_branch(ev, l, r, dep->bit, true, domain, CBM_MSB_NODE_FAIL_DEPENDENCY); + if (out && out->bit == dep->bit) { + out->witness_valid = dep->witness_valid; + out->witness = dep->witness; + out->epoch = dep->epoch; + } + } + } + st_drop(ev->state, l); + st_drop(ev->state, r); + return out; +} +static uint32_t st_symbol(msb_eval_t *ev, const char *key) { + msb_state_t *s = ev->state; + msb_work(SKIP_ONE); + uintptr_t id = (uintptr_t)cbm_ht_get(s->symbols, key); + if (id) + return (uint32_t)id; + if (s->next_symbol == UINT32_MAX) { + ev->oom = true; + return 0; + } + size_t n = strlen(key); + char *owned = cbm_arena_strndup(&s->names, key, n); + if (!owned) { + ev->oom = true; + return 0; + } + id = ++s->next_symbol; + cbm_ht_set(s->symbols, owned, (void *)id); + msb_work(n + 3); + if ((uintptr_t)cbm_ht_get(s->symbols, owned) != id) { + ev->oom = true; + return 0; + } + ev_peak(ev); + return (uint32_t)id; +} +static void st_record_input(msb_eval_t *ev, msb_view_t *inputs, int domain, uint32_t key) { + if (ev->oom || st_get(inputs->root[domain], key)) + return; + st_node_t *current = st_get(ev->view.root[domain], key); + st_node_t *dep = st_new(ev, CBM_MSB_NODE_FAIL_DEPENDENCY); + if (!dep) + return; + dep->dependency = true; + dep->domain = (unsigned char)domain; + dep->key = key; + dep->present = current != NULL; + dep->all_present = dep->present; + dep->all_absent = !dep->present; + dep->value = current ? st_value_hold(current->value) : NULL; + dep->witness_valid = true; + dep->witness = current ? current->revision : 0; + if (current) + current->witnessed = true; + st_node_t *next = + st_union(ev, inputs->root[domain], dep, true, domain, CBM_MSB_NODE_FAIL_DEPENDENCY); + st_drop(ev->state, dep); + if (!ev->oom) { + st_certify(ev, next, ev->view.root[domain], domain); + st_drop(ev->state, inputs->root[domain]); + inputs->root[domain] = next; + } else + st_drop(ev->state, next); +} +static const msb_prop_t *st_property(msb_eval_t *ev, const char *key) { + uint32_t id = st_symbol(ev, key); + if (!id) + return NULL; + st_node_t *n = st_get(ev->view.root[ST_PROPS], id); + if (ev->capture_inputs) + st_record_input(ev, &ev->state_inputs, ST_PROPS, id); + else if (ev->builder && !st_get(ev->builder->writes.root[ST_PROPS], id)) + st_record_input(ev, &ev->builder->inputs, ST_PROPS, id); + static const msb_prop_t unknown = {0}; + return n ? (n->value ? &n->value->prop : &unknown) : NULL; +} +static bool st_membership(msb_eval_t *ev, uint32_t file, int domain) { + bool has = st_get(ev->view.root[domain], file + 1) != NULL; + if (ev->builder && !st_get(ev->builder->writes.root[domain], file + 1)) + st_record_input(ev, &ev->builder->inputs, domain, file + 1); + return has; +} +static st_value_t *st_make_value(msb_eval_t *ev, const char *text) { + if (!text) + return NULL; + size_t bytes = strlen(text) + 1; + st_value_t *v = NULL; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (!msb_fail_value_alloc()) +#endif + { + v = (st_value_t *)cbm_alloc(CBM_MEM_CLASS_OTHER, sizeof(*v) + bytes); + } + if (!v) { + ev->oom = true; + return NULL; + } + v->refs = 1; + v->prop.value = (char *)(v + 1); + v->prop.capacity = bytes; + memcpy(v->prop.value, text, bytes); + ev->state->bytes += sizeof(*v) + bytes; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + atomic_fetch_add_explicit(&msb_value_live_counter, bytes, memory_order_relaxed); +#endif + msb_work(bytes + sizeof(*v)); + ev_peak(ev); + return v; +} +static void st_write(msb_eval_t *ev, uint32_t key, int domain, const char *text) { + if (ev->oom) + return; + st_node_t *old = st_get(ev->view.root[domain], key); +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + /* Independent before/after audit of every existing ancestor on this + * changed key's path. Snapshot IDs, not addresses that may be recycled. */ + struct { + uint32_t key; + int bit; + uint64_t revision; + } before[33]; + int before_count = 0; + for (st_node_t *n = ev->view.root[domain]; n && before_count < 33;) { + before[before_count].key = n->key; + before[before_count].bit = n->bit; + before[before_count++].revision = n->revision; + if (n->bit < 0 || !st_same_range(n->key, key, n->bit)) + break; + n = n->child[(key >> n->bit) & 1]; + } +#endif + if (domain == ST_PROPS && old && old->value && text && + old->value->prop.capacity == strlen(text) + 1 && strcmp(old->value->prop.value, text)) + ST_STAT(same_length_value_changes); + cbm_msb_node_fail_operation_t phase = + ev->builder && ev->builder->retain ? CBM_MSB_NODE_FAIL_CAPTURE : CBM_MSB_NODE_FAIL_NONE; + st_node_t *leaf = NULL; + if (old && + ((!text && !old->value) || (text && old->value && !strcmp(text, old->value->prop.value)))) + leaf = st_hold(old); + else { + leaf = st_new(ev, phase); + if (!leaf) + return; + leaf->key = key; + leaf->domain = (unsigned char)domain; + leaf->present = true; + leaf->value = domain == ST_PROPS ? st_make_value(ev, text) : NULL; + } + st_node_t *effect = NULL; + st_node_t *saved_reuse = ev->reuse_nodes; + if (!ev->oom && ev->builder) { + ev->reuse_nodes = ev->view.root[domain]; + effect = st_union(ev, leaf, ev->builder->writes.root[domain], false, domain, phase); + ev->reuse_nodes = saved_reuse; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + bool is_new = !st_get(ev->builder->writes.root[domain], key); + if (domain == ST_PROPS && is_new && msb_fail_prop_insert()) { + st_drop(ev->state, effect); + effect = st_hold(ev->builder->writes.root[domain]); + } +#endif + if (!st_get(effect, key)) + ev->oom = true; + } + ev->reuse_nodes = effect; + st_node_t *view = + !ev->oom ? st_union(ev, leaf, ev->view.root[domain], false, domain, phase) : NULL; + ev->reuse_nodes = saved_reuse; +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (!ev->builder && domain == ST_PROPS && !old && msb_fail_prop_insert()) { + st_drop(ev->state, view); + view = st_hold(ev->view.root[domain]); + } +#endif + if (!st_get(view, key)) + ev->oom = true; + if (!ev->oom) { + if (old && leaf != old && leaf->revision == old->revision) + ST_STAT(revision_errors); +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS + if (leaf != old) { + for (int i = 0; i < before_count; i++) { + /* Only old ranges containing the changed key are affected. */ + if (!st_same_range(before[i].key, key, before[i].bit)) + continue; + st_node_t *after = view; + while (after && after->bit > before[i].bit) { + if (!st_same_range(after->key, before[i].key, after->bit)) { + after = NULL; + break; + } + after = after->child[(before[i].key >> after->bit) & 1]; + } + if (after && !st_same_range(after->key, before[i].key, before[i].bit)) + after = NULL; + if (after && after->revision == before[i].revision) + ST_STAT(revision_errors); + } + } +#endif + st_drop(ev->state, ev->view.root[domain]); + ev->view.root[domain] = view; + if (ev->builder) { + st_drop(ev->state, ev->builder->writes.root[domain]); + ev->builder->writes.root[domain] = effect; + } + } else { + st_drop(ev->state, view); + st_drop(ev->state, effect); + } + st_drop(ev->state, leaf); +} +static void st_set_property(msb_eval_t *ev, const char *name, const char *value) { + char key[MSB_NAME_MAX]; + if (!msb_key(name, strlen(name), key)) + return; + uint32_t id = st_symbol(ev, key); + if (id) + st_write(ev, id, ST_PROPS, value); +} + +static int seed_slot(const char *key) { + for (int i = 0; i < MSB_SEEDS; i++) { + msb_work(SKIP_ONE); + if (strcmp(key, MSB_SEED_NAMES[i]) == 0) { + return i; + } + } + return CBM_NOT_FOUND; +} + +static msb_prop_dep_t *copy_prop_dep(msb_eval_t *ev, const msb_prop_t *p) { + msb_prop_dep_t *dep = (msb_prop_dep_t *)ev_alloc(ev, sizeof(*dep)); + if (!dep) { + return NULL; + } + *dep = (msb_prop_dep_t){.present = p != NULL}; + msb_work(sizeof(*dep)); + if (p && p->value) { + dep->bytes = p->capacity; + dep->value = ev_strndup(ev, p->value, dep->bytes - SKIP_ONE); + } + return ev->oom ? NULL : dep; +} + +/* Record reads that escape the capturing owner's local state. Incoming + * state is immutable for this pass. Target effects use the prefix baseline; + * final items also use completed target writes. Dependency values are owned + * copies, so no project buffer survives through a dependency pointer. */ +static void record_prop_input(msb_eval_t *ev, const char *key, const msb_prop_t *p) { + int seed = seed_slot(key); + if (seed >= 0) { + if (!ev->inputs.seeds[seed]) { + ev->inputs.seeds[seed] = copy_prop_dep(ev, p); + } + return; + } + msb_work(SKIP_ONE); + if (cbm_ht_get(ev->inputs.props, key)) { + return; + } + msb_prop_dep_t *dep = copy_prop_dep(ev, p); + char *owned_key = dep ? ev_strndup(ev, key, strlen(key)) : NULL; + if (!owned_key) { + return; + } + const msb_prop_t *baseline = NULL; + if (ev->incoming->effects) { + msb_work(SKIP_ONE); + baseline = (const msb_prop_t *)cbm_ht_get(ev->incoming->effects->props, key); + } + if (!baseline && ev->incoming->base) { + msb_work(SKIP_ONE); + baseline = (const msb_prop_t *)cbm_ht_get(ev->incoming->base->props, key); + } + dep->exception = !prop_dep_matches(dep, baseline); + msb_work(PAIR_LEN); + cbm_ht_set(ev->inputs.props, owned_key, dep); + if (cbm_ht_get(ev->inputs.props, owned_key) != dep) { + ev->oom = true; + return; + } + ev->inputs.prop_exceptions += dep->exception; +} + +/* A present unknown value shadows lower layers just like a known one. + * During target capture, incoming is a separate read-only owner; during + * pass 2, completed target writes are the highest-precedence layer. */ +static const msb_prop_t *prop_lookup(msb_eval_t *ev, const char *key) { + if (ev->state) + return st_property(ev, key); + const msb_prop_t *p = NULL; + if (ev->effects) { + msb_work(SKIP_ONE); + p = (const msb_prop_t *)cbm_ht_get(ev->effects->props, key); + if (p) { + return p; + } + } + if (ev->props) { + msb_work(SKIP_ONE); + p = (const msb_prop_t *)cbm_ht_get(ev->props, key); + } + if (!p && ev->incoming) { + p = prop_lookup(ev->incoming, key); + if (ev->capture_inputs) { + record_prop_input(ev, key, p); + } + return p; + } + if (!p && ev->base) { + msb_work(SKIP_ONE); + p = (const msb_prop_t *)cbm_ht_get(ev->base->props, key); + } + if (!p && ev->seeds) { + msb_work(SKIP_ONE); + p = (const msb_prop_t *)cbm_ht_get(ev->seeds->props, key); + if (p && ev->track_seed_reads) { + ev->seed_read = true; + } + } + return p; +} + +/* The value of $(name) read in `file`; NULL when it is not known. */ +static const char *msb_ref(msb_eval_t *ev, int file, const char *name, size_t len) { + char key[MSB_NAME_MAX]; + if (!msb_key(name, len, key)) { + return NULL; + } + static const char this_file[] = "msbuildthisfile"; + if (strncmp(key, this_file, sizeof(this_file) - SKIP_ONE) == 0) { + const msb_file_t *f = &ev->m->files[file]; + const char *rest = key + sizeof(this_file) - SKIP_ONE; + if (strcmp(rest, "directory") == 0) { + return f->abs_dir; + } + if (rest[0] == '\0') { + return f->name; + } + if (strcmp(rest, "name") == 0) { + return f->stem; + } + if (strcmp(rest, "extension") == 0) { + return f->ext; + } + if (strcmp(rest, "fullpath") == 0) { + return f->abs_path; + } + } + const msb_prop_t *p = prop_lookup(ev, key); + return p ? p->value : NULL; +} + +/* A value being put together; `over` once it is longer than a value may be. */ +typedef struct { + msb_eval_t *ev; + char *buf; + size_t len; + size_t cap; + bool over; +} msb_sb_t; + +static void msb_sb_put(msb_sb_t *b, const char *s, size_t n) { + if (b->over || b->ev->oom) { + return; + } + if (b->len + n >= MSB_VALUE_MAX) { + b->over = true; + return; + } + if (b->len + n + SKIP_ONE > b->cap) { + size_t ncap = b->cap ? b->cap * PAIR_LEN : CBM_SZ_64; + while (ncap < b->len + n + SKIP_ONE) { + ncap *= PAIR_LEN; + } + char *grown = (char *)scratch_alloc(b->ev, ncap); + if (!grown) { + return; + } + if (b->len > 0) { + memcpy(grown, b->buf, b->len); + msb_work(b->len); + } + b->buf = grown; + b->cap = ncap; + } + memcpy(b->buf + b->len, s, n); + b->len += n; + b->buf[b->len] = '\0'; + msb_work(n + SKIP_ONE); /* append plus its terminator */ +} + +/* `text` with its $(Property) references replaced. NULL when it has no value + * this reader knows: a property that is not known, a property function, an + * item or metadata reference, an escape, a value past MSB_VALUE_MAX. */ +static const char *msb_expand(msb_eval_t *ev, int file, const char *text) { + msb_sb_t b = {.ev = ev}; + for (const char *p = text; *p;) { + if (p[0] == '$' && p[1] == '(') { + const char *nm = p + PAIR_LEN; + const char *e = nm; + while (isalnum((unsigned char)*e) || *e == '_') { + e++; + } + if (e == nm || *e != ')' || isdigit((unsigned char)nm[0])) { + return NULL; + } + const char *v = msb_ref(ev, file, nm, (size_t)(e - nm)); + if (!v) { + return NULL; + } + msb_sb_put(&b, v, strlen(v)); + p = e + SKIP_ONE; + continue; + } + if ((p[0] == '@' || p[0] == '%') && p[1] == '(') { + return NULL; + } + if (p[0] == '%' && isxdigit((unsigned char)p[1]) && isxdigit((unsigned char)p[2])) { + return NULL; + } + msb_sb_put(&b, p, SKIP_ONE); + p++; + } + if (b.over || ev->oom) { + return NULL; + } + return b.buf ? b.buf : ""; +} + +/* ── Conditions ──────────────────────────────────────────────────── */ + +static msb_tri_t tri_not(msb_tri_t v) { + return v == MSB_UNKNOWN ? MSB_UNKNOWN : (v == MSB_TRUE ? MSB_FALSE : MSB_TRUE); +} + +static msb_tri_t tri_and(msb_tri_t a, msb_tri_t b) { + if (a == MSB_FALSE || b == MSB_FALSE) { + return MSB_FALSE; + } + return (a == MSB_TRUE && b == MSB_TRUE) ? MSB_TRUE : MSB_UNKNOWN; +} + +static msb_tri_t tri_or(msb_tri_t a, msb_tri_t b) { + if (a == MSB_TRUE || b == MSB_TRUE) { + return MSB_TRUE; + } + return (a == MSB_FALSE && b == MSB_FALSE) ? MSB_FALSE : MSB_UNKNOWN; +} + +static bool ci_eq_n(const char *a, size_t an, const char *b, size_t bn) { + if (an != bn) { + return false; + } + for (size_t i = 0; i < an; i++) { + if (tolower((unsigned char)a[i]) != tolower((unsigned char)b[i])) { + return false; + } + } + return true; +} + +static bool ci_is(const char *a, size_t an, const char *word) { + return ci_eq_n(a, an, word, strlen(word)); +} + +/* A string MSBuild converts to a boolean: 1, 0, or -1 for any other. */ +static int msb_bool(const char *s, size_t n) { + if (ci_is(s, n, "true") || ci_is(s, n, "on") || ci_is(s, n, "yes")) { + return SKIP_ONE; + } + if (ci_is(s, n, "false") || ci_is(s, n, "off") || ci_is(s, n, "no")) { + return 0; + } + return CBM_NOT_FOUND; +} + +/* A decimal number as its parts, leading and trailing zeros dropped. */ +typedef struct { + bool neg; + const char *ip; + size_t il; + const char *fp; + size_t fl; +} msb_num_t; + +/* [+-]digits[.digits] or [+-].digits, nothing else. */ +static bool msb_decimal(const char *s, size_t n, msb_num_t *out) { + size_t i = 0; + memset(out, 0, sizeof(*out)); + if (i < n && (s[i] == '+' || s[i] == '-')) { + out->neg = s[i] == '-'; + i++; + } + size_t is = i; + while (i < n && isdigit((unsigned char)s[i])) { + i++; + } + size_t ie = i; + size_t fs = i; + size_t fe = i; + if (i < n && s[i] == '.') { + i++; + fs = i; + while (i < n && isdigit((unsigned char)s[i])) { + i++; + } + fe = i; + } + if (i != n || (ie == is && fe == fs)) { + return false; + } + bool nonzero = false; + for (size_t k = is; k < fe; k++) { + nonzero = nonzero || (s[k] != '0' && s[k] != '.'); + } + while (is < ie && s[is] == '0') { + is++; + } + while (fe > fs && s[fe - SKIP_ONE] == '0') { + fe--; + } + out->ip = s + is; + out->il = ie - is; + out->fp = s + fs; + out->fl = fe - fs; + if (!nonzero) { + out->neg = false; /* -0 is 0 */ + } + return true; +} + +/* Text some conversion could read as a number although msb_decimal does not + * (hexadecimal, an exponent). */ +static bool msb_numberish(const char *s, size_t n) { + if (n == 0 || !(isdigit((unsigned char)s[0]) || s[0] == '+' || s[0] == '-' || s[0] == '.')) { + return false; + } + for (size_t i = 0; i < n; i++) { + unsigned char c = (unsigned char)s[i]; + if (!(isxdigit(c) || c == 'x' || c == 'X' || c == '.' || c == '+' || c == '-')) { + return false; + } + } + return true; +} + +/* MSBuild's `==` on two expanded operands, as far as it can be known. */ +static msb_tri_t msb_equal_spans(const char *a, size_t an, const char *b, size_t bn) { + if (ci_eq_n(a, an, b, bn)) { + return MSB_TRUE; /* the same text is equal under every conversion */ + } + /* An absolute path: where the repository lies is not known here, so its + * real text is not. Two of them differ as their marked texts do; one is + * never the empty string; any other comparison is open. */ + bool ma = an > 0 && a[0] == MSB_PATH_MARK; + bool mb = bn > 0 && b[0] == MSB_PATH_MARK; + if (memchr(a + ma, MSB_PATH_MARK, an - ma) || memchr(b + mb, MSB_PATH_MARK, bn - mb)) { + return MSB_UNKNOWN; + } + if (ma || mb) { + return ((ma && mb) || an == 0 || bn == 0) ? MSB_FALSE : MSB_UNKNOWN; + } + int ba = msb_bool(a, an); + int bb = msb_bool(b, bn); + if (ba >= 0 && bb >= 0) { + return ba == bb ? MSB_TRUE : MSB_FALSE; + } + msb_num_t na; + msb_num_t nb; + bool da = msb_decimal(a, an, &na); + bool db = msb_decimal(b, bn, &nb); + if (da && db) { + if (na.il + na.fl > MSB_SIGNIFICANT || nb.il + nb.fl > MSB_SIGNIFICANT) { + return MSB_UNKNOWN; /* MSBuild compares doubles: these may round together */ + } + bool same = na.neg == nb.neg && na.il == nb.il && na.fl == nb.fl && + memcmp(na.ip, nb.ip, na.il) == 0 && memcmp(na.fp, nb.fp, na.fl) == 0; + return same ? MSB_TRUE : MSB_FALSE; + } + if ((da || msb_numberish(a, an)) && (db || msb_numberish(b, bn))) { + return MSB_UNKNOWN; + } + return MSB_FALSE; +} + +static void trim_span(const char **s, size_t *n) { + while (*n > 0 && isspace((unsigned char)(*s)[0])) { + (*s)++; + (*n)--; + } + while (*n > 0 && isspace((unsigned char)(*s)[*n - SKIP_ONE])) { + (*n)--; + } +} + +/* A property's value keeps the white space its element was written with. + * Whether a comparison sees it is not something to guess: the two readings + * must agree. */ +static msb_tri_t msb_equal(const char *a, const char *b) { + size_t an = strlen(a); + size_t bn = strlen(b); + msb_tri_t as_written = msb_equal_spans(a, an, b, bn); + trim_span(&a, &an); + trim_span(&b, &bn); + msb_tri_t trimmed = msb_equal_spans(a, an, b, bn); + return as_written == trimmed ? as_written : MSB_UNKNOWN; +} + +/* A value standing alone as a condition. */ +static msb_tri_t msb_truth(const char *v) { + size_t n = strlen(v); + int as_written = msb_bool(v, n); + trim_span(&v, &n); + int trimmed = msb_bool(v, n); + if (as_written != trimmed || trimmed < 0) { + return MSB_UNKNOWN; + } + return trimmed ? MSB_TRUE : MSB_FALSE; +} + +typedef struct { + msb_eval_t *ev; + int file; + const char *s; + int depth; + bool bad; /* not a condition this reader can parse */ +} msb_cond_t; + +static void cond_ws(msb_cond_t *c) { + while (isspace((unsigned char)*c->s)) { + c->s++; + } +} + +/* The keyword `kw` (lower case) stands next: consume it. */ +static bool cond_keyword(msb_cond_t *c, const char *kw) { + cond_ws(c); + size_t n = strlen(kw); + for (size_t i = 0; i < n; i++) { + if (tolower((unsigned char)c->s[i]) != kw[i]) { + return false; + } + } + unsigned char after = (unsigned char)c->s[n]; + if (isalnum(after) || after == '_') { + return false; + } + c->s += n; + return true; +} + +/* Past the group opened by the '(' at c->s (quotes respected); false when it + * does not close. */ +static bool cond_skip_group(msb_cond_t *c) { + int depth = 0; + for (const char *p = c->s; *p; p++) { + if (*p == '\'') { + p = strchr(p + SKIP_ONE, '\''); + if (!p) { + return false; + } + } else if (*p == '(') { + depth++; + } else if (*p == ')' && --depth == 0) { + c->s = p + SKIP_ONE; + return true; + } + } + return false; +} + +/* An operand: 'text', $(Property), a bare word, or a function call. Returns + * its value, NULL when that is not known; *ok false when no operand stands + * here. */ +static const char *cond_value(msb_cond_t *c, bool *ok) { + cond_ws(c); + *ok = true; + const char *s = c->s; + if (*s == '\'') { + const char *e = strchr(s + SKIP_ONE, '\''); + if (!e) { + *ok = false; + return NULL; + } + c->s = e + SKIP_ONE; + const char *lit = scratch_strndup(c->ev, s + SKIP_ONE, (size_t)(e - s - SKIP_ONE)); + return lit ? msb_expand(c->ev, c->file, lit) : NULL; + } + if (s[0] == '$' && s[1] == '(') { + c->s = s + SKIP_ONE; + if (!cond_skip_group(c)) { + *ok = false; + return NULL; + } + const char *ref = scratch_strndup(c->ev, s, (size_t)(c->s - s)); + return ref ? msb_expand(c->ev, c->file, ref) : NULL; + } + const char *e = s; + while (isalnum((unsigned char)*e) || *e == '_' || *e == '.' || *e == '-' || *e == '+') { + e++; + } + if (e == s) { + *ok = false; + return NULL; + } + c->s = e; + cond_ws(c); + if (*c->s == '(') { + *ok = cond_skip_group(c); /* Exists(...), HasTrailingSlash(...): not evaluated */ + return NULL; + } + return scratch_strndup(c->ev, s, (size_t)(e - s)); +} + +static msb_tri_t cond_or(msb_cond_t *c); + +static msb_tri_t cond_comparison_value(msb_cond_t *c) { + bool ok = false; + const char *lhs = cond_value(c, &ok); + if (!ok) { + c->bad = true; + return MSB_UNKNOWN; + } + cond_ws(c); + char op0 = c->s[0]; + char op1 = op0 ? c->s[1] : '\0'; + if ((op0 == '=' || op0 == '!') && op1 == '=') { + c->s += PAIR_LEN; + const char *rhs = cond_value(c, &ok); + if (!ok) { + c->bad = true; + return MSB_UNKNOWN; + } + msb_tri_t v = (lhs && rhs) ? msb_equal(lhs, rhs) : MSB_UNKNOWN; + return op0 == '!' ? tri_not(v) : v; + } + if (op0 == '<' || op0 == '>') { + c->s += op1 == '=' ? PAIR_LEN : SKIP_ONE; /* an order of numbers or versions */ + (void)cond_value(c, &ok); + c->bad = c->bad || !ok; + return MSB_UNKNOWN; + } + return lhs ? msb_truth(lhs) : MSB_UNKNOWN; +} + +/* Both operands must survive through comparison, including every error + * return. Only the primitive result escapes to the surrounding condition. */ +static msb_tri_t cond_comparison(msb_cond_t *c) { + msb_tri_t result = cond_comparison_value(c); + cbm_arena_reset(&c->ev->scratch); + return result; +} + +static msb_tri_t cond_primary(msb_cond_t *c) { + cond_ws(c); + bool negate = false; + while (c->s[0] == '!' && c->s[1] != '=') { + negate = !negate; + c->s++; + cond_ws(c); + } + msb_tri_t v = MSB_UNKNOWN; + if (*c->s == '(') { + if (c->depth >= MSB_COND_DEPTH) { + c->bad = true; + return MSB_UNKNOWN; + } + c->s++; + c->depth++; + v = cond_or(c); + c->depth--; + cond_ws(c); + if (*c->s != ')') { + c->bad = true; + return MSB_UNKNOWN; + } + c->s++; + } else { + v = cond_comparison(c); + } + return negate ? tri_not(v) : v; +} + +static msb_tri_t cond_and(msb_cond_t *c) { + msb_tri_t v = cond_primary(c); + while (!c->bad && cond_keyword(c, "and")) { + v = tri_and(v, cond_primary(c)); + } + return v; +} + +static msb_tri_t cond_or(msb_cond_t *c) { + msb_tri_t v = cond_and(c); + while (!c->bad && cond_keyword(c, "or")) { + v = tri_or(v, cond_and(c)); + } + return v; +} + +/* A Condition attribute, read in `file`. `and` binds tighter than `or`. */ +static msb_tri_t msb_cond(msb_eval_t *ev, int file, const char *cond) { + if (!cond) { + return MSB_TRUE; + } + msb_cond_t c = {.ev = ev, .file = file, .s = cond}; + cond_ws(&c); + if (!*c.s) { + return MSB_TRUE; + } + msb_tri_t v = cond_or(&c); + cond_ws(&c); + return (c.bad || *c.s) ? MSB_UNKNOWN : v; +} + +/* ── Imports ─────────────────────────────────────────────────────── */ + +/* Resolve "." and ".." in a '/'-separated path, in place. false when it + * leaves the repository. */ +static bool msb_normalize(char *path) { + /* The text is read by index up to its original length: what is written + * (every kept segment and a '/' after it) never passes the read position, + * but it does overwrite the terminator. */ + size_t len = strlen(path); + size_t w = 0; + size_t i = 0; + while (i < len) { + size_t s = i; + while (i < len && path[i] != '/') { + i++; + } + size_t n = i - s; + i += i < len; /* past the '/' */ + if (n == PAIR_LEN && path[s] == '.' && path[s + SKIP_ONE] == '.') { + if (w == 0) { + return false; + } + w--; /* the '/' that ends the previous segment */ + while (w > 0 && path[w - SKIP_ONE] != '/') { + w--; + } + } else if (n > 0 && !(n == SKIP_ONE && path[s] == '.')) { + memmove(path + w, path + s, n); + msb_work(n + SKIP_ONE); /* copied segment and slash */ + w += n; + path[w++] = '/'; + } + } + path[w > 0 ? w - SKIP_ONE : 0] = '\0'; + return true; +} + +/* The file an written in `file` names: its index, or + * CBM_NOT_FOUND. What cannot be followed is counted: a path that is not + * known or names several files (unevaluable), a file the index does not hold + * (outside). */ +static int msb_import_target(msb_eval_t *ev, int file, const char *project) { + const char *value = project ? msb_expand(ev, file, project) : NULL; + if (!value) { + ev->unevaluable++; + return CBM_NOT_FOUND; + } + size_t n = strlen(value); + trim_span(&value, &n); + const msb_file_t *f = &ev->m->files[file]; + bool marked = n > 0 && value[0] == MSB_PATH_MARK; + size_t skip = marked ? SKIP_ONE : 0; + if (n == skip || memchr(value + skip, MSB_PATH_MARK, n - skip) || memchr(value, '*', n) || + memchr(value, '?', n)) { + ev->unevaluable++; + return CBM_NOT_FOUND; + } + bool absolute = + !marked && (value[0] == '/' || value[0] == '\\' || (n > SKIP_ONE && value[1] == ':')); + if (absolute) { + ev->outside++; + return CBM_NOT_FOUND; + } + size_t dl = marked ? 0 : strlen(f->dir); + char *path = (char *)scratch_alloc(ev, dl + n + PAIR_LEN); + if (!path) { + return CBM_NOT_FOUND; + } + memcpy(path, f->dir, dl); + path[dl] = '/'; + memcpy(path + dl + SKIP_ONE, value + skip, n - skip); + path[dl + SKIP_ONE + n - skip] = '\0'; + msb_work(dl + n - skip + PAIR_LEN); + for (char *p = path; *p; p++) { + if (*p == '\\') { + *p = '/'; + } + } + int target = msb_normalize(path) ? msb_file_index(ev->m, path) : CBM_NOT_FOUND; + ev->outside += target < 0; + return target; +} + +static bool set_has(const CBMHashTable *set, const char *key) { + msb_work(SKIP_ONE); + return cbm_ht_get(set, key) != NULL; +} + +/* Stable borrowed sentinel: a frozen set must not retain a stack address. */ +static const char MSB_SET_PRESENT = 0; + +static void record_set_input(msb_eval_t *ev, const char *key, bool present, bool poison) { + CBMHashTable *deps = poison ? ev->inputs.poisoned : ev->inputs.seen; + msb_work(SKIP_ONE); + if (cbm_ht_get(deps, key)) { + return; + } + msb_set_dep_t *dep = (msb_set_dep_t *)ev_alloc(ev, sizeof(*dep)); + char *owned_key = dep ? ev_strndup(ev, key, strlen(key)) : NULL; + if (!owned_key) { + return; + } + const msb_eval_t *base = ev->incoming->base; + bool baseline = base && set_has(poison ? base->poisoned : base->seen, key); + *dep = (msb_set_dep_t){.present = present, .exception = present != baseline}; + msb_work(sizeof(*dep) + PAIR_LEN); + cbm_ht_set(deps, owned_key, dep); + if (cbm_ht_get(deps, owned_key) != dep) { + ev->oom = true; + return; + } + if (poison) { + ev->inputs.poisoned_exceptions += dep->exception; + } else { + ev->inputs.seen_exceptions += dep->exception; + } +} + +static bool seen_has(msb_eval_t *ev, const char *key) { + if (ev->state) { + int file = msb_file_index(ev->m, key); + return file >= 0 && st_membership(ev, (uint32_t)file, ST_SEEN); + } + if (set_has(ev->seen, key)) { + return true; + } + bool present = + ev->incoming ? seen_has(ev->incoming, key) : ev->base && set_has(ev->base->seen, key); + if (ev->capture_inputs) { + record_set_input(ev, key, present, false); + } + return present; +} + +static bool poisoned_has(msb_eval_t *ev, const char *key) { + if (ev->state) { + int file = msb_file_index(ev->m, key); + return file >= 0 && st_membership(ev, (uint32_t)file, ST_POISON); + } + if (set_has(ev->poisoned, key)) { + return true; + } + bool present = ev->incoming ? poisoned_has(ev->incoming, key) + : ev->base && set_has(ev->base->poisoned, key); + if (ev->capture_inputs) { + record_set_input(ev, key, present, true); + } + return present; +} + +/* The nearest file called `name` in `dir` or above it; CBM_NOT_FOUND for none. */ +static int msb_nearest(msb_eval_t *ev, const char *dir, const char *name) { + size_t dl = strlen(dir); + size_t nl = strlen(name); + char *path = (char *)scratch_alloc(ev, dl + nl + PAIR_LEN); + if (!path) { + return CBM_NOT_FOUND; + } + for (;;) { + memcpy(path, dir, dl); + path[dl] = '/'; + memcpy(path + dl + (dl ? SKIP_ONE : 0), name, nl + SKIP_ONE); + msb_work(dl + nl + PAIR_LEN); + msb_nearest_step(); + int found = msb_file_index(ev->m, path); + if (found >= 0 || dl == 0) { + return found; + } + while (dl > 0 && dir[dl - SKIP_ONE] != '/') { + msb_work(SKIP_ONE); + dl--; + } + dl = dl > 0 ? dl - SKIP_ONE : 0; + } +} + +static bool msb_push(msb_eval_t *ev, int file) { + if (ev->nframes >= ev->cap_frames) { + int ncap = ev->cap_frames ? ev->cap_frames * PAIR_LEN : MSB_INIT; + msb_frame_t *grown = (msb_frame_t *)ev_alloc(ev, (size_t)ncap * sizeof(*grown)); + if (!grown) { + return false; + } + if (ev->nframes > 0) { + memcpy(grown, ev->frames, (size_t)ev->nframes * sizeof(*grown)); + msb_work((size_t)ev->nframes * sizeof(*grown)); + } + ev->frames = grown; + ev->cap_frames = ncap; + } + ev->frames[ev->nframes++] = (msb_frame_t){.file = file, .group = MSB_TRUE}; + msb_work(sizeof(msb_frame_t)); + return true; +} + +static st_rope_t *st_rope_hold(st_rope_t *r) { + if (r) { + r->refs++; + msb_work(1); + } + return r; +} +static void st_rope_drop(msb_state_t *s, st_rope_t *r) { + st_rope_t *pending = NULL; + if (r && --r->refs == 0) { + r->next = pending; + pending = r; + } + while (pending) { + r = pending; + pending = r->next; + msb_work(1); + st_rope_t *children[2] = {r->left, r->right}; + for (int i = 0; i < 2; i++) + if (children[i] && --children[i]->refs == 0) { + children[i]->next = pending; + pending = children[i]; + } + s->bytes -= sizeof(*r); + cbm_free(CBM_MEM_CLASS_OTHER, r); + } +} +static st_rope_t *st_rope_node(msb_eval_t *ev, st_rope_t *a, st_rope_t *b, const msb_item_t *item) { + if (ev->oom) + return NULL; + st_rope_t *r = (st_rope_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*r)); + if (!r) { + ev->oom = true; + return NULL; + } + ev->state->bytes += sizeof(*r); + r->refs = 1; + r->left = st_rope_hold(a); + r->right = st_rope_hold(b); + if (item) { + r->item = *item; + r->height = 1; + r->count = 1; + } else { + if (a->count > INT_MAX - b->count) { + ev->oom = true; + st_rope_drop(ev->state, r); + return NULL; + } + r->height = 1 + (a->height > b->height ? a->height : b->height); + r->count = a->count + b->count; + } + msb_work(sizeof(*r)); + ev_peak(ev); + return r; +} +static st_rope_t *st_rope_balance(msb_eval_t *ev, st_rope_t *a, st_rope_t *b) { + if (!a || !b) + return st_rope_hold(a ? a : b); + if (a->height > b->height + 1) { + st_rope_t *right = NULL, *left = NULL, *out = NULL; + if (a->left->height >= a->right->height) { + right = st_rope_node(ev, a->right, b, NULL); + if (right) + out = st_rope_node(ev, a->left, right, NULL); + } else { + left = st_rope_node(ev, a->left, a->right->left, NULL); + right = st_rope_node(ev, a->right->right, b, NULL); + if (left && right) + out = st_rope_node(ev, left, right, NULL); + } + st_rope_drop(ev->state, left); + st_rope_drop(ev->state, right); + return out; + } + if (b->height > a->height + 1) { + st_rope_t *right = NULL, *left = NULL, *out = NULL; + if (b->right->height >= b->left->height) { + left = st_rope_node(ev, a, b->left, NULL); + if (left) + out = st_rope_node(ev, left, b->right, NULL); + } else { + left = st_rope_node(ev, a, b->left->left, NULL); + right = st_rope_node(ev, b->left->right, b->right, NULL); + if (left && right) + out = st_rope_node(ev, left, right, NULL); + } + st_rope_drop(ev->state, left); + st_rope_drop(ev->state, right); + return out; + } + return st_rope_node(ev, a, b, NULL); +} +/* Join height-balanced nonempty spans. Empty import chains disappear here. */ +static st_rope_t *st_rope_join(msb_eval_t *ev, st_rope_t *a, st_rope_t *b) { + msb_work(1); + if (!a) + return st_rope_hold(b); + if (!b) + return st_rope_hold(a); + if (ev->oom) + return NULL; + if (a->height > b->height + 1) { + st_rope_t *tail = st_rope_join(ev, a->right, b); + st_rope_t *out = tail ? st_rope_balance(ev, a->left, tail) : NULL; + st_rope_drop(ev->state, tail); + return out; + } + if (b->height > a->height + 1) { + st_rope_t *head = st_rope_join(ev, a, b->left); + st_rope_t *out = head ? st_rope_balance(ev, head, b->right) : NULL; + st_rope_drop(ev->state, head); + return out; + } + return st_rope_node(ev, a, b, NULL); +} +static void st_component_drop(msb_state_t *s, st_component_t *c) { + if (!c) + return; + msb_work(1); + if (--c->refs) + return; + st_view_drop(s, &c->writes); + st_view_drop(s, &c->inputs); + st_rope_drop(s, c->items); + s->bytes -= sizeof(*c); + cbm_free(CBM_MEM_CLASS_OTHER, c); +} +static bool st_inputs_match(msb_eval_t *ev, const msb_view_t *inputs) { + for (int d = 0; d < ST_DOMAINS; d++) + if (!st_validate(ev, inputs->root[d], ev->view.root[d], d)) + return false; + return true; +} +/* B was selected against the actual after-A state before this call. Preserve + * every earlier A read, and mask only B's reads by A writes (unknown included). */ +static bool st_compose(msb_eval_t *ev, st_component_t *parent, st_component_t *child) { + if (!parent) + return true; + for (int d = 0; d < ST_DOMAINS && !ev->oom; d++) { + st_node_t *external = st_mask(ev, child->inputs.root[d], parent->writes.root[d], d); + st_node_t *inputs = + st_union(ev, parent->inputs.root[d], external, true, d, CBM_MSB_NODE_FAIL_DEPENDENCY); + st_drop(ev->state, external); + st_node_t *writes = st_union(ev, child->writes.root[d], parent->writes.root[d], false, d, + CBM_MSB_NODE_FAIL_OVERLAY); + if (!ev->oom) { + st_certify(ev, inputs, ev->view.root[d], d); + st_drop(ev->state, parent->inputs.root[d]); + parent->inputs.root[d] = inputs; + st_drop(ev->state, parent->writes.root[d]); + parent->writes.root[d] = writes; + } else { + st_drop(ev->state, inputs); + st_drop(ev->state, writes); + } + } + if (!ev->oom) { + st_rope_t *items = st_rope_join(ev, parent->items, child->items); + if (!ev->oom) { + st_rope_drop(ev->state, parent->items); + parent->items = items; + } else + st_rope_drop(ev->state, items); + } + return !ev->oom; +} +static bool st_apply(msb_eval_t *ev, st_component_t *c) { + if (!st_compose(ev, ev->builder, c)) + return false; + cbm_msb_node_fail_operation_t phase = + ev->builder ? CBM_MSB_NODE_FAIL_OVERLAY : CBM_MSB_NODE_FAIL_APPLY; + for (int d = 0; d < ST_DOMAINS && !ev->oom; d++) { + st_node_t *saved_reuse = ev->reuse_nodes; + ev->reuse_nodes = ev->builder ? ev->builder->writes.root[d] : NULL; + st_node_t *next = st_union(ev, c->writes.root[d], ev->view.root[d], false, d, phase); + ev->reuse_nodes = saved_reuse; + if (!ev->oom) { + st_drop(ev->state, ev->view.root[d]); + ev->view.root[d] = next; + } else + st_drop(ev->state, next); + } + if (ev->oom) + return false; + ev->open = ev->open || c->open; + ev->unevaluable += c->unevaluable; + ev->outside += c->outside; + if (!ev->builder) { + c->refs++; + ev->completed = c; + } + return true; +} +/* Returns true only when an interpreter frame was pushed. A validated cache + * hit applies the exact effect here, before the caller can mask dependencies. */ +static bool st_enter(msb_eval_t *ev, int file, bool poison) { + int domain = poison ? ST_POISON : ST_SEEN; + if (st_membership(ev, (uint32_t)file, domain) || ev->oom) + return false; + size_t slot = (size_t)file * 2 + (poison ? 1 : 0); + st_component_t *cached = ev->state->components[slot]; + if (cached && st_inputs_match(ev, &cached->inputs)) { + (void)st_apply(ev, cached); + return false; + } + st_component_t *c = (st_component_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*c)); + if (!c) { + ev->oom = true; + return false; + } + c->refs = 1; + c->parent = ev->builder; + c->file = file; + c->poison = poison; + c->retain = !cached && file != ev->main_project; + c->start_open = ev->open; + c->start_unevaluable = ev->unevaluable; + c->start_outside = ev->outside; + ev->state->bytes += sizeof(*c); + msb_work(sizeof(*c)); + ev_peak(ev); + ev->builder = c; + ev->open = false; + ev->unevaluable = 0; + ev->outside = 0; + if (!msb_push(ev, file)) { + ev->oom = true; + return false; + } + ev->frames[ev->nframes - 1].owner = c; + st_write(ev, (uint32_t)file + 1, domain, NULL); + /* a file that was not read is counted (and opens the scope) by its 'Y' + * record, as the walk reaches it */ + return !ev->oom; +} +static void st_finish(msb_eval_t *ev, st_component_t *c) { + c->open = ev->open; + c->unevaluable = ev->unevaluable; + c->outside = ev->outside; + ev->open = c->start_open || c->open; + ev->unevaluable += c->start_unevaluable; + ev->outside += c->start_outside; + ev->builder = c->parent; + c->parent = NULL; + if (!st_compose(ev, ev->builder, c)) { + st_component_drop(ev->state, c); + return; + } + if (c->retain) { + size_t slot = (size_t)c->file * 2 + (c->poison ? 1 : 0); + if (!ev->state->components[slot]) { + c->refs++; + ev->state->components[slot] = c; + } + } + if (ev->builder) + st_component_drop(ev->state, c); + else + ev->completed = c; +} +static void st_add_item(msb_eval_t *ev, const msb_rec_t *group, const msb_rec_t *item, int file) { + msb_item_t it = {.group = group, .item = item, .file = file}; + st_rope_t *leaf = st_rope_node(ev, NULL, NULL, &it); + st_rope_t *next = leaf ? st_rope_join(ev, ev->builder->items, leaf) : NULL; + st_rope_drop(ev->state, leaf); + if (!ev->oom) { + st_rope_drop(ev->state, ev->builder->items); + ev->builder->items = next; + } else + st_rope_drop(ev->state, next); +} + +/* `file` may or may not be imported: what it (and what it imports, under any + * condition) sets is unknown from here on. */ +static void msb_poison(msb_eval_t *ev, int file) { + int base = ev->nframes; + if (ev->state) { + if (!st_enter(ev, file, true)) + return; + } else { + if (poisoned_has(ev, ev->m->files[file].rel_path) || !msb_push(ev, file)) + return; + msb_work(SKIP_ONE); + cbm_ht_set(ev->poisoned, ev->m->files[file].rel_path, (void *)&MSB_SET_PRESENT); + } + while (ev->nframes > base && !ev->oom) { + msb_frame_t *fr = &ev->frames[ev->nframes - SKIP_ONE]; + const msb_file_t *f = &ev->m->files[fr->file]; + if (fr->rec >= f->nrecs) { + st_component_t *owner = fr->owner; + ev->nframes--; + if (ev->state && owner) + st_finish(ev, owner); + continue; + } + const msb_rec_t *r = &f->recs[fr->rec++]; + msb_record(); + int at = fr->file; + if (r->tag == 'V' || r->tag == 'W') { + msb_set(ev, r->f[1], NULL); + } else if (r->tag == 'K') { + msb_set(ev, r->f[0], NULL); + } else if (r->tag == 'N' || r->tag == 'Y') { + ev->open = true; + } else if (r->tag == 'I' && !r->f[3]) { + int target = msb_import_target(ev, at, r->f[2]); + if (target >= 0) { + if (ev->state) + (void)st_enter(ev, target, true); + else if (!poisoned_has(ev, ev->m->files[target].rel_path) && msb_push(ev, target)) { + msb_work(SKIP_ONE); + cbm_ht_set(ev->poisoned, ev->m->files[target].rel_path, + (void *)&MSB_SET_PRESENT); + } + } + } + cbm_arena_reset(&ev->scratch); + } +} + +static void msb_add_item(msb_eval_t *ev, const msb_rec_t *group, const msb_rec_t *item, int file) { + if (ev->state) { + st_add_item(ev, group, item, file); + return; + } + if (ev->nitems >= ev->cap_items) { + int ncap = ev->cap_items ? ev->cap_items * PAIR_LEN : MSB_INIT; + msb_item_t *grown = (msb_item_t *)ev_alloc(ev, (size_t)ncap * sizeof(*grown)); + if (!grown) { + return; + } + if (ev->nitems > 0) { + memcpy(grown, ev->items, (size_t)ev->nitems * sizeof(*grown)); + msb_work((size_t)ev->nitems * sizeof(*grown)); + } + ev->items = grown; + ev->cap_items = ncap; + } + ev->items[ev->nitems++] = (msb_item_t){.group = group, .item = item, .file = file}; + msb_work(sizeof(msb_item_t)); +} + +/* An standing in the file on top of the frame stack, under its + * 's condition `group` (MSB_TRUE outside a group). */ +static void msb_import(msb_eval_t *ev, int file, const msb_rec_t *r, msb_tri_t group) { + msb_tri_t c = tri_and(group, msb_cond(ev, file, r->f[1])); + if (c == MSB_FALSE || r->f[3]) { + return; /* not taken; or an SDK's file, which is no file of the repository */ + } + int target = msb_import_target(ev, file, r->f[2]); + if (c == MSB_UNKNOWN) { + ev->unevaluable++; + if (target >= 0) { + msb_poison(ev, target); + } + return; + } + if (ev->state) { + if (target >= 0) + (void)st_enter(ev, target, false); + return; + } + if (target >= 0 && !seen_has(ev, ev->m->files[target].rel_path) && msb_push(ev, target)) { + msb_work(SKIP_ONE); + cbm_ht_set(ev->seen, ev->m->files[target].rel_path, (void *)&MSB_SET_PRESENT); + /* a file that was not read (malformed, or too large) is counted and + * opens the scope by its 'Y' record (cbm_msb_add) */ + } +} + +/* A property record under its group's condition `group`. */ +static void msb_property(msb_eval_t *ev, int file, const msb_rec_t *r, msb_tri_t group) { + msb_tri_t own = msb_cond(ev, file, r->f[0]); + msb_tri_t c = tri_and(group, own); + if (c == MSB_FALSE) { + return; + } + ev->unevaluable += own == MSB_UNKNOWN; + const char *value = + (c == MSB_TRUE && r->tag == 'V') ? msb_expand(ev, file, r->f[2] ? r->f[2] : "") : NULL; + msb_set(ev, r->f[1], value); +} + +/* Pass 1 over `file` and what it imports: properties in order, the + * items collected for pass 2. No recursion: an import chain is as long as + * the repository makes it. */ +static void msb_pass1(msb_eval_t *ev, int file) { + int base = ev->nframes; + const char *rel = ev->m->files[file].rel_path; + if (ev->state) { + if (!st_enter(ev, file, false)) + return; + } else { + if (seen_has(ev, rel) || !msb_push(ev, file)) + return; + msb_work(SKIP_ONE); + cbm_ht_set(ev->seen, rel, (void *)&MSB_SET_PRESENT); + } + while (ev->nframes > base && !ev->oom) { + /* an import pushes a frame, which may move the array: the frame is + * read into locals before the record is handled */ + msb_frame_t *fr = &ev->frames[ev->nframes - SKIP_ONE]; + const msb_file_t *f = &ev->m->files[fr->file]; + if (fr->rec >= f->nrecs) { + st_component_t *owner = fr->owner; + ev->nframes--; + if (ev->state && owner) + st_finish(ev, owner); + continue; + } + const msb_rec_t *r = &f->recs[fr->rec++]; + msb_record(); + int at = fr->file; + switch (r->tag) { + case 'G': + fr->group = msb_cond(ev, at, r->f[0]); + ev->unevaluable += fr->group == MSB_UNKNOWN; + break; + case 'V': + case 'W': + msb_property(ev, at, r, fr->group); + break; + case 'K': + msb_set(ev, r->f[0], NULL); + break; + case 'H': + fr->hgroup = r; + break; + case 'N': + msb_add_item(ev, fr->hgroup, r, at); + break; + case 'Y': + ev->open = true; + ev->unevaluable++; + break; + case 'C': + ev->unevaluable++; + break; + case 'I': { + /* the group's condition once per group: its imports share the + * text (cbm_msb_add); a legacy blob's copies are evaluated each */ + msb_tri_t group = MSB_TRUE; + if (r->f[0]) { + if (fr->igroup_text != r->f[0]) { + fr->igroup_text = r->f[0]; + fr->igroup = msb_cond(ev, at, r->f[0]); + } + group = fr->igroup; + } + msb_import(ev, at, r, group); /* may push a frame: fr is not used after it */ + break; + } + default: + break; + } + cbm_arena_reset(&ev->scratch); + } +} + +/* ── Usings ──────────────────────────────────────────────────────── */ + +typedef struct { + cbm_msb_using_t *v; + int n; + int cap; + CBMHashTable *unique; /* capture only: exact tuples, or removal targets */ + bool target_only; +} msb_ulist_t; + +static void ulist_add(msb_eval_t *ev, msb_ulist_t *l, char kind, const char *alias, + const char *target) { + if (ev->oom) { + return; + } + if (l->n >= l->cap) { + if (l->cap > INT_MAX / PAIR_LEN) { + ev->oom = true; + return; + } + int ncap = l->cap ? l->cap * PAIR_LEN : MSB_INIT; + if ((size_t)ncap > SIZE_MAX / sizeof(cbm_msb_using_t)) { + ev->oom = true; + return; + } + cbm_msb_using_t *grown = (cbm_msb_using_t *)ev_alloc(ev, (size_t)ncap * sizeof(*grown)); + if (!grown) { + return; + } + if (l->n > 0) { + memcpy(grown, l->v, (size_t)l->n * sizeof(*grown)); + msb_work((size_t)l->n * sizeof(*grown)); + } + l->v = grown; + l->cap = ncap; + } + l->v[l->n++] = (cbm_msb_using_t){.kind = kind, .alias = alias, .target = target}; + msb_work(sizeof(cbm_msb_using_t)); +} + +/* The tuple key has a fixed-width hexadecimal target length, followed by + * the target and alias bytes. It cannot confuse embedded separators or a + * different target/alias boundary. Only a new tuple acquires owned strings. */ +static void ulist_add_unique(msb_eval_t *ev, msb_ulist_t *l, char kind, const char *alias, + size_t an, const char *target, size_t tn) { + size_t header = SKIP_ONE + PAIR_LEN * sizeof(size_t); + if (tn > SIZE_MAX - header - SKIP_ONE || an > SIZE_MAX - header - tn - SKIP_ONE) { + ev->oom = true; + return; + } + size_t bytes = l->target_only ? tn : header + tn + an; + char *key = (char *)scratch_alloc(ev, bytes + SKIP_ONE); + if (!key) { + return; + } + if (l->target_only) { + memcpy(key, target, tn); + } else { + static const char digits[] = "0123456789abcdef"; + key[0] = kind; + size_t length = tn; + for (size_t i = header; i > SKIP_ONE;) { + key[--i] = digits[length & 15]; + length >>= 4; + } + memcpy(key + header, target, tn); + memcpy(key + header + tn, alias, an); + } + key[bytes] = '\0'; + msb_work(bytes + PAIR_LEN); /* key construction and unique lookup */ + if (cbm_ht_get(l->unique, key)) { + return; + } + char *owned_key = ev_strndup(ev, key, bytes); + const char *owned_target = l->target_only ? owned_key : ev_strndup(ev, target, tn); + const char *owned_alias = an ? ev_strndup(ev, alias, an) : ""; + if (ev->oom) { + return; + } + msb_work(SKIP_ONE); + if (!msb_fail_item_operation(CBM_MSB_ITEM_FAIL_UNIQUE_INSERT)) { + msb_work(SKIP_ONE); + cbm_ht_set(l->unique, owned_key, owned_key); + } + if (cbm_ht_get(l->unique, owned_key) != owned_key) { + ev->oom = true; + return; + } + if (!l->target_only) { + ulist_add(ev, l, kind, owned_alias, owned_target); + } +} + +/* Add every ';'-separated entry. During capture, neither aliases nor targets + * are copied into persistent storage before exact duplicate detection. */ +static void ulist_add_split(msb_eval_t *ev, msb_ulist_t *l, char kind, const char *alias, size_t an, + const char *list) { + if (!l->unique && an > 0) { + alias = ev_strndup(ev, alias, an); + } + for (const char *p = list; !ev->oom && p;) { + const char *e = strchr(p, ';'); + size_t n = e ? (size_t)(e - p) : strlen(p); + const char *s = p; + trim_span(&s, &n); + if (n > 0) { + if (l->unique) { + ulist_add_unique(ev, l, kind, alias, an, s, n); + } else { + ulist_add(ev, l, kind, an ? alias : "", ev_strndup(ev, s, n)); + } + } + p = e ? e + SKIP_ONE : NULL; + } +} + +static int using_cmp(const void *a, const void *b) { + msb_work(SKIP_ONE); + const cbm_msb_using_t *x = (const cbm_msb_using_t *)a; + const cbm_msb_using_t *y = (const cbm_msb_using_t *)b; + if (x->kind != y->kind) { + return x->kind < y->kind ? -1 : 1; + } + int c = strcmp(x->target, y->target); + return c ? c : strcmp(x->alias, y->alias); +} + +static int target_cmp(const void *a, const void *b) { + msb_work(SKIP_ONE); + return strcmp(((const cbm_msb_using_t *)a)->target, ((const cbm_msb_using_t *)b)->target); +} + +/* An 's condition for the items of one pass over them. MSBuild + * evaluates it once, where the group stands, and the final properties the + * items see do not change while they are read: evaluating it again for + * every item cost items x condition length. */ +typedef struct { + const msb_rec_t *group; + msb_tri_t value; +} msb_group_memo_t; + +static msb_tri_t item_group_cond(msb_eval_t *ev, const msb_item_t *it, msb_group_memo_t *memo) { + if (!it->group) { + return MSB_TRUE; + } + if (memo->group != it->group) { + memo->group = it->group; + memo->value = msb_cond(ev, it->file, it->group->f[0]); + } + return memo->value; +} + +/* One item with the final properties: into `inc` or `rem`. */ +static void msb_using(msb_eval_t *ev, const msb_item_t *it, msb_ulist_t *inc, msb_ulist_t *rem, + msb_group_memo_t *memo) { + const msb_rec_t *r = it->item; + msb_record(); + msb_tri_t c = tri_and(item_group_cond(ev, it, memo), msb_cond(ev, it->file, r->f[0])); + if (c == MSB_FALSE) { + return; + } + const char *include = r->f[1] ? msb_expand(ev, it->file, r->f[1]) : ""; + const char *remove = r->f[2] ? msb_expand(ev, it->file, r->f[2]) : ""; + const char *is_static = r->f[3] ? msb_expand(ev, it->file, r->f[3]) : ""; + const char *alias = r->f[4] ? msb_expand(ev, it->file, r->f[4]) : ""; + if (c == MSB_UNKNOWN || !include || !remove || !is_static || !alias) { + ev->open = true; + ev->unevaluable++; + return; + } + size_t sn = strlen(is_static); + trim_span(&is_static, &sn); + size_t an = strlen(alias); + trim_span(&alias, &an); + char kind = an > 0 ? 'a' : (ci_is(is_static, sn, "true") ? 's' : 'n'); + ulist_add_split(ev, inc, kind, alias, an, include); + ulist_add_split(ev, rem, 'n', "", 0, remove); +} + +static const char *const SDK_DEFAULT[] = { + "System", "System.Collections.Generic", "System.IO", "System.Linq", "System.Net.Http", + "System.Threading", "System.Threading.Tasks", NULL}; +static const char *const SDK_WEB[] = {"System", + "System.Collections.Generic", + "System.IO", + "System.Linq", + "System.Net.Http", + "System.Net.Http.Json", + "System.Threading", + "System.Threading.Tasks", + "Microsoft.AspNetCore.Builder", + "Microsoft.AspNetCore.Hosting", + "Microsoft.AspNetCore.Http", + "Microsoft.AspNetCore.Routing", + "Microsoft.Extensions.Configuration", + "Microsoft.Extensions.DependencyInjection", + "Microsoft.Extensions.Hosting", + "Microsoft.Extensions.Logging", + NULL}; +static const char *const SDK_WORKER[] = {"System", + "System.Collections.Generic", + "System.IO", + "System.Linq", + "System.Net.Http", + "System.Threading", + "System.Threading.Tasks", + "Microsoft.Extensions.Configuration", + "Microsoft.Extensions.DependencyInjection", + "Microsoft.Extensions.Hosting", + "Microsoft.Extensions.Logging", + NULL}; + +/* True when the ';'-separated SDK list names `sdk` (a version after '/' is + * no part of the name). */ +static bool sdk_listed(const char *list, const char *sdk) { + size_t sl = strlen(sdk); + for (const char *p = list; p;) { + const char *e = strchr(p, ';'); + size_t n = e ? (size_t)(e - p) : strlen(p); + const char *s = p; + trim_span(&s, &n); + const char *slash = memchr(s, '/', n); + size_t name = slash ? (size_t)(slash - s) : n; + if (name == sl && memcmp(s, sdk, sl) == 0) { + return true; + } + p = e ? e + SKIP_ONE : NULL; + } + return false; +} + +/* The SDK a project file names: its attribute, else the first + * import that names one. NULL when it names none. */ +static const char *msb_sdk(const msb_file_t *f) { + const char *sdk = f->sdk; + for (int i = 0; !sdk && i < f->nrecs; i++) { + sdk = f->recs[i].tag == 'I' ? f->recs[i].f[3] : NULL; + } + return sdk; +} + +bool cbm_msb_compiles(const cbm_msb_t *m, const char *rel_path) { + /* These two SDKs are defined as producing no assembly (one runs build + * steps, the other builds other projects): that is why they are named. */ + static const char *const idle[] = {"Microsoft.Build.NoTargets", "Microsoft.Build.Traversal"}; + int at = (m && rel_path) ? msb_file_index(m, rel_path) : CBM_NOT_FOUND; + const char *sdk = at >= 0 ? msb_sdk(&m->files[at]) : NULL; + for (size_t i = 0; sdk && i < sizeof(idle) / sizeof(idle[0]); i++) { + if (sdk_listed(sdk, idle[i])) { + return false; + } + } + return true; +} + +/* The usings ImplicitUsings brings in, or NULL. An ImplicitUsings that some + * file sets to a value this reader does not know leaves the usings open. */ +static const char *const *msb_implicit(msb_eval_t *ev, int project) { + static const char name[] = "ImplicitUsings"; + char key[MSB_NAME_MAX]; + (void)msb_key(name, sizeof(name) - SKIP_ONE, key); + const msb_prop_t *p = prop_lookup(ev, key); + if (!p) { + return NULL; + } + msb_tri_t on = + p->value ? tri_or(msb_equal(p->value, "enable"), msb_equal(p->value, "true")) : MSB_UNKNOWN; + if (on == MSB_UNKNOWN) { + ev->open = true; + ev->unevaluable++; + } + if (on != MSB_TRUE) { + return NULL; + } + const char *sdk = msb_sdk(&ev->m->files[project]); + if (sdk && sdk_listed(sdk, "Microsoft.NET.Sdk.Web")) { + return SDK_WEB; + } + return (sdk && sdk_listed(sdk, "Microsoft.NET.Sdk.Worker")) ? SDK_WORKER : SDK_DEFAULT; +} + +/* The evaluation's usings as one heap block: the array, then its strings. */ +static bool msb_result(msb_eval_t *ev, msb_ulist_t *inc, const msb_ulist_t *rem, + cbm_msb_result_t *out) { + if (rem->n > 0) { + qsort(rem->v, (size_t)rem->n, sizeof(rem->v[0]), target_cmp); + } + if (inc->n > 0) { + qsort(inc->v, (size_t)inc->n, sizeof(inc->v[0]), using_cmp); + } + int kept = 0; + size_t bytes = 0; + for (int i = 0; i < inc->n; i++) { + const cbm_msb_using_t *u = &inc->v[i]; + bool removed = + rem->n > 0 && bsearch(u, rem->v, (size_t)rem->n, sizeof(rem->v[0]), target_cmp) != NULL; + if (!removed && ev->shared_removals) { + msb_work(SKIP_ONE); + removed = cbm_ht_get(ev->shared_removals, u->target) != NULL; + } + bool dup = kept > 0 && using_cmp(&inc->v[kept - SKIP_ONE], u) == 0; + if (removed || dup) { + continue; + } + inc->v[kept++] = *u; + msb_work(sizeof(*u)); + bytes += strlen(u->alias) + strlen(u->target) + PAIR_LEN; + } + out->open = ev->open; + out->unevaluable = ev->unevaluable; + out->outside = ev->outside; + if (kept == 0) { + return !ev->oom; + } + size_t head = (size_t)kept * sizeof(cbm_msb_using_t); + char *mem = msb_fail_item_operation(CBM_MSB_ITEM_FAIL_PUBLISH_ALLOC) + ? NULL + : (char *)cbm_alloc(CBM_MEM_CLASS_OTHER, head + bytes); + if (!mem) { + return false; + } + ev->result_bytes = head + bytes; + ev_peak(ev); + cbm_msb_using_t *arr = (cbm_msb_using_t *)(void *)mem; + char *w = mem + head; + for (int i = 0; i < kept; i++) { + size_t al = strlen(inc->v[i].alias) + SKIP_ONE; + size_t tl = strlen(inc->v[i].target) + SKIP_ONE; + memcpy(w, inc->v[i].alias, al); + memcpy(w + al, inc->v[i].target, tl); + arr[i] = (cbm_msb_using_t){.kind = inc->v[i].kind, .alias = w, .target = w + al}; + msb_work(al + tl + sizeof(arr[i])); + w += al + tl; + } + out->usings = arr; + out->count = kept; + out->mem = mem; + return !ev->oom; +} + +/* The properties MSBuild sets before it reads the project file. */ +static void msb_seed(msb_eval_t *ev, int project) { + const msb_file_t *f = &ev->m->files[project]; + msb_set(ev, "MSBuildProjectName", f->stem); + msb_set(ev, "MSBuildProjectFile", f->name); + msb_set(ev, "MSBuildProjectExtension", f->ext); + msb_set(ev, "MSBuildProjectFullPath", f->abs_path); + /* the directory without its trailing '/' */ + size_t dl = strlen(f->abs_dir); + char *dir = scratch_strndup(ev, f->abs_dir, dl > SKIP_ONE ? dl - SKIP_ONE : dl); + if (dir) { + msb_set(ev, "MSBuildProjectDirectory", dir); + } +} + +static bool eval_init(msb_eval_t *ev) { + cbm_arena_init(&ev->arena); + cbm_arena_init(&ev->scratch); + ev_peak(ev); + ev->props = cbm_ht_create(CBM_SZ_64); + ev->seen = cbm_ht_create(MSB_INIT); + ev->poisoned = cbm_ht_create(MSB_INIT); + ev->oom = !ev->props || !ev->seen || !ev->poisoned; + return !ev->oom; +} + +static void eval_clear(msb_eval_t *ev) { + if (ev->state) + st_view_drop(ev->state, &ev->state_inputs); + if (ev->props) { + cbm_ht_foreach(ev->props, prop_clear, ev); + } + cbm_ht_free(ev->props); + cbm_ht_free(ev->seen); + cbm_ht_free(ev->poisoned); + cbm_ht_free(ev->inputs.props); + cbm_ht_free(ev->inputs.seen); + cbm_ht_free(ev->inputs.poisoned); + cbm_arena_destroy(&ev->arena); + cbm_arena_destroy(&ev->scratch); +} + +bool cbm_msb_eval(const cbm_msb_t *m, const char *project_rel, cbm_msb_result_t *out) { + memset(out, 0, sizeof(*out)); + int project = (m && project_rel) ? msb_file_index(m, project_rel) : CBM_NOT_FOUND; + if (project < 0) { + return false; + } + msb_eval_t ev = {.m = m}; + cbm_arena_init(&ev.arena); + cbm_arena_init(&ev.scratch); + ev_peak(&ev); + ev.props = cbm_ht_create(CBM_SZ_64); + ev.seen = cbm_ht_create(MSB_INIT); + ev.poisoned = cbm_ht_create(MSB_INIT); + bool ok = ev.props && ev.seen && ev.poisoned; + if (ok) { + const msb_file_t *f = &m->files[project]; + msb_seed(&ev, project); + cbm_arena_reset(&ev.scratch); + int props = msb_nearest(&ev, f->dir, "Directory.Build.props"); + cbm_arena_reset(&ev.scratch); + if (props >= 0) { + msb_pass1(&ev, props); + } + msb_pass1(&ev, project); + int targets = msb_nearest(&ev, f->dir, "Directory.Build.targets"); + cbm_arena_reset(&ev.scratch); + if (targets >= 0) { + msb_pass1(&ev, targets); + } + msb_ulist_t inc = {0}; + msb_ulist_t rem = {0}; + const char *const *implicit = msb_implicit(&ev, project); + for (int i = 0; implicit && implicit[i]; i++) { + ulist_add(&ev, &inc, 'n', "", implicit[i]); + } + msb_group_memo_t memo = {0}; + for (int i = 0; i < ev.nitems; i++) { + msb_using(&ev, &ev.items[i], &inc, &rem, &memo); + /* All four expansions remain live until aliases and targets + * have been copied into the persistent evaluation arena. */ + cbm_arena_reset(&ev.scratch); + } + ok = !ev.oom && msb_result(&ev, &inc, &rem, out); + } + eval_clear(&ev); + if (!ok) { + cbm_msb_result_free(out); + } + return ok; +} + +/* One immutable prefix and one exact target effect. The target effect's + * dependency keys are indexed, so validation walks CURRENT local overrides, + * not every property or imported file in a large shared closure. */ +typedef struct { + bool known[PAIR_LEN]; + int file[PAIR_LEN]; +} msb_nearest_entry_t; + +struct cbm_msb_eval_context { + const cbm_msb_t *m; + msb_state_t *state; + uint64_t generation; + int root; + bool ready; + msb_eval_t prefix; + int target_root; + const msb_eval_t *target_base; + bool target_live; + bool target_ready; + msb_eval_t target; + bool items_live; + bool items_ready; + msb_eval_t item_owner; /* lazy arena/dependencies; no property or import state */ + msb_ulist_t item_includes; + msb_ulist_t item_removals; + CBMHashTable *nearest; /* immutable model-owned directory -> two nearest-file results */ + size_t nearest_bytes; +}; + +static void nearest_entry_free(const char *key, void *value, void *userdata) { + (void)key; + cbm_msb_eval_context_t *context = (cbm_msb_eval_context_t *)userdata; + cbm_free(CBM_MEM_CLASS_OTHER, value); + context->nearest_bytes -= sizeof(msb_nearest_entry_t); + msb_work(SKIP_ONE); +} + +static void context_drop_items(cbm_msb_eval_context_t *context) { + if (context->items_live) { + msb_work(SKIP_ONE + context->item_owner.arena.nblocks + + context->item_owner.scratch.nblocks); + cbm_ht_free(context->item_includes.unique); + cbm_ht_free(context->item_removals.unique); + eval_clear(&context->item_owner); + memset(&context->item_owner, 0, sizeof(context->item_owner)); + context->item_includes = (msb_ulist_t){0}; + context->item_removals = (msb_ulist_t){0}; + msb_work(sizeof(context->item_owner) + sizeof(msb_ulist_t) * PAIR_LEN); + context->items_live = false; + context->items_ready = false; + } +} + +static void context_drop_target(cbm_msb_eval_context_t *context) { + if (context->target_live) { + context_drop_items(context); + eval_clear(&context->target); + memset(&context->target, 0, sizeof(context->target)); + msb_work(sizeof(context->target)); + context->target_live = false; + context->target_ready = false; + context->target_base = NULL; + } +} + +static void context_drop_prefix(cbm_msb_eval_context_t *context) { + if (context->ready) { + context_drop_items(context); + /* Target dependencies can borrow the immutable prefix identity. */ + context_drop_target(context); + eval_clear(&context->prefix); + memset(&context->prefix, 0, sizeof(context->prefix)); + msb_work(sizeof(context->prefix)); + context->ready = false; + } +} + +static void context_clear_nearest(cbm_msb_eval_context_t *context) { + cbm_ht_foreach(context->nearest, nearest_entry_free, context); + cbm_ht_clear(context->nearest); + msb_work(SKIP_ONE); +} + +static msb_state_t *st_state_new(const cbm_msb_t *m); +static void st_state_free(msb_state_t *s); +static bool st_context_eval(cbm_msb_eval_context_t *context, const char *project_rel, + cbm_msb_result_t *out); + +cbm_msb_eval_context_t *cbm_msb_eval_context_new(const cbm_msb_t *m) { + if (!m) { + return NULL; + } + cbm_msb_eval_context_t *context = + (cbm_msb_eval_context_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*context)); + if (!context) { + return NULL; + } + msb_work(sizeof(*context)); + context->m = m; + context->generation = m->generation; + context->root = CBM_NOT_FOUND; + context->target_root = CBM_NOT_FOUND; + context->nearest = cbm_ht_create(MSB_INIT); + context->state = st_state_new(m); + if (!context->nearest || !context->state) { + st_state_free(context->state); + cbm_ht_free(context->nearest); + cbm_free(CBM_MEM_CLASS_OTHER, context); + return NULL; + } + return context; +} + +void cbm_msb_eval_context_free(cbm_msb_eval_context_t *context) { + if (context) { + context_drop_items(context); + context_drop_target(context); + context_drop_prefix(context); + st_state_free(context->state); + cbm_ht_foreach(context->nearest, nearest_entry_free, context); + cbm_ht_free(context->nearest); + cbm_free(CBM_MEM_CLASS_OTHER, context); + } +} + +/* Cache metadata only. Directory keys live in the model, and an absent + * result is as reusable as a present one until the model generation changes. */ +static int context_nearest(cbm_msb_eval_context_t *context, msb_eval_t *ev, const char *dir, + bool targets) { + msb_work(SKIP_ONE); + msb_nearest_entry_t *entry = (msb_nearest_entry_t *)cbm_ht_get(context->nearest, dir); + if (!entry) { + if (sizeof(*entry) > SIZE_MAX - context->nearest_bytes) { + ev->oom = true; + return CBM_NOT_FOUND; + } + entry = (msb_nearest_entry_t *)cbm_alloc(CBM_MEM_CLASS_OTHER, sizeof(*entry)); + if (!entry) { + ev->oom = true; + return CBM_NOT_FOUND; + } + context->nearest_bytes += sizeof(*entry); + ev_peak(ev); + memset(entry, 0, sizeof(*entry)); + msb_work(sizeof(*entry) + PAIR_LEN); + cbm_ht_set(context->nearest, dir, entry); + if (cbm_ht_get(context->nearest, dir) != entry) { + nearest_entry_free(dir, entry, context); + ev->oom = true; + return CBM_NOT_FOUND; + } + } + int which = targets ? SKIP_ONE : 0; + msb_work(SKIP_ONE); + if (!entry->known[which]) { + int found = + msb_nearest(ev, dir, targets ? "Directory.Build.targets" : "Directory.Build.props"); + cbm_arena_reset(&ev->scratch); + if (ev->oom) { + return CBM_NOT_FOUND; + } + entry->file[which] = found; + entry->known[which] = true; + } + return entry->file[which]; +} + +static void attach_prefix(msb_eval_t *ev, const msb_eval_t *prefix) { + ev->base = prefix; + ev->open = prefix->open; + ev->unevaluable = prefix->unevaluable; + ev->outside = prefix->outside; + msb_work(SKIP_ONE); + ev_peak(ev); +} + +typedef struct { + const msb_inputs_t *inputs; + const msb_eval_t *effects; /* final-item inputs: target writes hide project entries */ + size_t props; + size_t seen; + size_t poisoned; + bool match; +} msb_input_check_t; + +static void check_prop_input(const char *key, void *value, void *userdata) { + msb_input_check_t *check = (msb_input_check_t *)userdata; + msb_work(SKIP_ONE); /* every CURRENT local entry visited, including irrelevant ones */ + if (!check->match) { + return; + } + if (check->effects) { + msb_work(SKIP_ONE); + if (cbm_ht_get(check->effects->props, key)) { + return; + } + } + msb_work(SKIP_ONE); + const msb_prop_dep_t *dep = (const msb_prop_dep_t *)cbm_ht_get(check->inputs->props, key); + if (dep) { + check->match = prop_dep_matches(dep, (const msb_prop_t *)value); + check->props += check->match && dep->exception; + } +} + +static void check_set_input(const char *key, msb_input_check_t *check, bool poison) { + msb_work(SKIP_ONE); + if (!check->match) { + return; + } + msb_work(SKIP_ONE); + const msb_set_dep_t *dep = (const msb_set_dep_t *)cbm_ht_get( + poison ? check->inputs->poisoned : check->inputs->seen, key); + if (dep) { + check->match = dep->present; + if (poison) { + check->poisoned += check->match && dep->exception; + } else { + check->seen += check->match && dep->exception; + } + } +} + +static void check_seen_input(const char *key, void *value, void *userdata) { + (void)value; + check_set_input(key, (msb_input_check_t *)userdata, false); +} + +static void check_poisoned_input(const char *key, void *value, void *userdata) { + (void)value; + check_set_input(key, (msb_input_check_t *)userdata, true); +} + +static bool seed_inputs_match(const msb_inputs_t *inputs, msb_eval_t *incoming) { + for (int i = 0; i < MSB_SEEDS; i++) { + msb_work(SKIP_ONE); + const msb_prop_dep_t *dep = inputs->seeds[i]; + if (dep && !prop_dep_matches(dep, prop_lookup(incoming, MSB_SEED_NAMES[i]))) { + return false; + } + } + return true; +} + +/* A dependency equal to the immutable baseline needs no local override. + * Every other dependency needs a matching CURRENT local entry. Counting + * matched exceptions catches missing inputs without scanning cached maps. */ +static bool target_inputs_match(const msb_eval_t *target, msb_eval_t *incoming) { + msb_input_check_t check = {.inputs = &target->inputs, .match = true}; + cbm_ht_foreach(incoming->props, check_prop_input, &check); + cbm_ht_foreach(incoming->seen, check_seen_input, &check); + cbm_ht_foreach(incoming->poisoned, check_poisoned_input, &check); + if (!check.match || check.props != target->inputs.prop_exceptions || + check.seen != target->inputs.seen_exceptions || + check.poisoned != target->inputs.poisoned_exceptions) { + return false; + } + return seed_inputs_match(&target->inputs, incoming); +} + +static bool apply_targets(cbm_msb_eval_context_t *context, msb_eval_t *ev, int root) { + if (root < 0) { + context_drop_target(context); + return !ev->oom; + } + msb_work(SKIP_ONE); + bool hit = context->target_ready && context->target_root == root && + context->target_base == ev->base && target_inputs_match(&context->target, ev); + if (!hit) { + context_drop_items(context); + context_drop_target(context); + msb_eval_t *target = &context->target; + target->m = ev->m; + target->incoming = ev; + target->live = ev->live; + target->capture_inputs = true; + context->target_live = true; + bool ok = eval_init(target); + target->inputs.props = cbm_ht_create(MSB_INIT); + target->inputs.seen = cbm_ht_create(MSB_INIT); + target->inputs.poisoned = cbm_ht_create(MSB_INIT); + target->oom = target->oom || !target->inputs.props || !target->inputs.seen || + !target->inputs.poisoned; + if (ok && !target->oom) { + msb_pass1(target, root); + } + if (target->oom || target->nframes) { + ev->oom = true; + context_drop_target(context); + return false; + } + target->incoming = NULL; + target->live = NULL; + target->capture_inputs = false; + context->target_root = root; + context->target_base = ev->base; + context->target_ready = true; + } + ev->effects = &context->target; + ev->open = ev->open || context->target.open; + ev->unevaluable += context->target.unevaluable; + ev->outside += context->target.outside; + msb_work(SKIP_ONE); + ev_peak(ev); + return true; +} + +static void eval_items(msb_eval_t *ev, const msb_eval_t *owner, msb_ulist_t *inc, + msb_ulist_t *rem) { + msb_group_memo_t memo = {0}; + for (int i = 0; !ev->oom && i < owner->nitems; i++) { + msb_using(ev, &owner->items[i], inc, rem, &memo); + cbm_arena_reset(&ev->scratch); + } +} + +/* Item dependencies see final properties. A present target value, even + * unknown, hides a project override. Matching exception counts detect inputs + * missing from CURRENT locals without walking the retained dependency map. */ +static bool item_inputs_match(const msb_eval_t *items, msb_eval_t *incoming) { + if (incoming->state) + return st_inputs_match(incoming, &items->state_inputs); + msb_input_check_t check = { + .inputs = &items->inputs, .effects = incoming->effects, .match = true}; + cbm_ht_foreach(incoming->props, check_prop_input, &check); + return check.match && check.props == items->inputs.prop_exceptions && + seed_inputs_match(&items->inputs, incoming); +} + +static void st_eval_rope(msb_eval_t *ev, st_rope_t *rope, msb_ulist_t *inc, msb_ulist_t *rem); + +static bool capture_items(cbm_msb_eval_context_t *context, msb_eval_t *ev) { + msb_eval_t *items = &context->item_owner; + items->m = ev->m; + items->incoming = ev; + items->live = ev->live; + items->capture_inputs = true; + if (ev->state) { + items->state = ev->state; + items->view = ev->view; + } + items->item_alloc_operation = CBM_MSB_ITEM_FAIL_CAPTURE_ALLOC; + context->items_live = true; + cbm_arena_init_lazy(&items->arena, CBM_ARENA_DEFAULT_BLOCK_SIZE); + cbm_arena_init_lazy(&items->scratch, CBM_ARENA_DEFAULT_BLOCK_SIZE); + items->inputs.props = cbm_ht_create(MSB_INIT); + context->item_includes.unique = cbm_ht_create(MSB_INIT); + context->item_removals.unique = cbm_ht_create(MSB_INIT); + context->item_removals.target_only = true; + items->oom = + !items->inputs.props || !context->item_includes.unique || !context->item_removals.unique; + msb_work(sizeof(*items)); + if (ev->state) { + st_eval_rope(items, ev->state->capture_a, &context->item_includes, &context->item_removals); + st_eval_rope(items, ev->state->capture_b, &context->item_includes, &context->item_removals); + } else { + if (!items->oom && ev->base) + eval_items(items, ev->base, &context->item_includes, &context->item_removals); + if (!items->oom && ev->effects) + eval_items(items, ev->effects, &context->item_includes, &context->item_removals); + } + if (items->oom) { + ev->oom = true; + context_drop_items(context); + return false; + } + /* Shared removals apply globally. Prune shared includes once, but retain + * the removal index for project and implicit includes at publication. */ + msb_ulist_t *inc = &context->item_includes; + int kept = 0; + for (int i = 0; i < inc->n; i++) { + msb_work(SKIP_ONE); + if (!cbm_ht_get(context->item_removals.unique, inc->v[i].target)) { + inc->v[kept++] = inc->v[i]; + msb_work(sizeof(inc->v[i])); + } + } + inc->n = kept; + cbm_ht_free(inc->unique); + inc->unique = NULL; + msb_work(SKIP_ONE + items->scratch.nblocks); + cbm_arena_destroy(&items->scratch); + items->incoming = NULL; + items->view = (msb_view_t){0}; + items->live = NULL; + items->capture_inputs = false; + items->item_alloc_operation = CBM_MSB_ITEM_FAIL_NONE; + context->items_ready = true; + return true; +} + +static bool apply_items(cbm_msb_eval_context_t *context, msb_eval_t *ev, msb_ulist_t *inc, + msb_ulist_t *rem) { + if (ev->state) { + if (!ev->state->capture_a && !ev->state->capture_b) + return !ev->oom; + } else if ((!ev->base || !ev->base->nitems) && (!ev->effects || !ev->effects->nitems)) { + return !ev->oom; + } + msb_work(SKIP_ONE); + if (context->items_ready && !item_inputs_match(&context->item_owner, ev)) { + /* Keep the first exact variant for this owner epoch. A different + * project's final inputs do not churn retained item storage. */ + if (ev->state) { + st_eval_rope(ev, ev->state->capture_a, inc, rem); + st_eval_rope(ev, ev->state->capture_b, inc, rem); + } else { + if (ev->base) + eval_items(ev, ev->base, inc, rem); + if (ev->effects) + eval_items(ev, ev->effects, inc, rem); + } + return !ev->oom; + } + if (!context->items_ready && !capture_items(context, ev)) { + return false; + } + const msb_ulist_t *shared = &context->item_includes; + if (shared->n > 0) { + if (inc->n > INT_MAX - shared->n || + (size_t)(inc->n + shared->n) > SIZE_MAX / sizeof(*inc->v)) { + ev->oom = true; + return false; + } + int n = inc->n + shared->n; + cbm_msb_item_fail_operation_t previous = ev->item_alloc_operation; + ev->item_alloc_operation = CBM_MSB_ITEM_FAIL_APPLY_ALLOC; + cbm_msb_using_t *v = (cbm_msb_using_t *)ev_alloc(ev, (size_t)n * sizeof(*v)); + ev->item_alloc_operation = previous; + if (!v) { + return false; + } + if (inc->n) { + memcpy(v, inc->v, (size_t)inc->n * sizeof(*v)); + } + memcpy(v + inc->n, shared->v, (size_t)shared->n * sizeof(*v)); + msb_work((size_t)n * sizeof(*v)); + inc->v = v; + inc->n = n; + inc->cap = n; + } + ev->shared_removals = context->item_removals.unique; + ev->open = ev->open || context->item_owner.open; + ev->unevaluable += context->item_owner.unevaluable; + ev->outside += context->item_owner.outside; + msb_work(SKIP_ONE); + ev_peak(ev); + return !ev->oom; +} + +bool cbm_msb_eval_context_eval(cbm_msb_eval_context_t *context, const char *project_rel, + cbm_msb_result_t *out) { + if (context && context->state) + return st_context_eval(context, project_rel, out); + memset(out, 0, sizeof(*out)); + if (!context) { + return false; + } + const cbm_msb_t *m = context->m; + msb_work(SKIP_ONE); + if (context->generation != m->generation) { + context_drop_items(context); + context_drop_target(context); + context_drop_prefix(context); + context_clear_nearest(context); + context->generation = m->generation; + } + int project = project_rel ? msb_file_index(m, project_rel) : CBM_NOT_FOUND; + if (project < 0) { + return false; + } + msb_eval_t seeds = {.m = m}; + msb_eval_t ev = {.m = m, .seeds = &seeds, .base = context->ready ? &context->prefix : NULL}; + msb_live_t live = { + .owners = {&context->prefix, &context->target, &ev, &seeds, &context->item_owner}, + .metadata_bytes = &context->nearest_bytes}; + ev.live = &live; + seeds.live = &live; + bool ok = eval_init(&ev); + bool seeds_ok = eval_init(&seeds); + if (ok && seeds_ok) { + msb_seed(&seeds, project); + cbm_arena_reset(&seeds.scratch); + ok = !seeds.oom; + } else { + ok = false; + } + if (ok) { + const msb_file_t *f = &m->files[project]; + int props = context_nearest(context, &ev, f->dir, false); + msb_work(SKIP_ONE); + if (context->ready && context->root == props) { + attach_prefix(&ev, &context->prefix); + } else { + ev.base = NULL; + /* No retained prefix means a stable empty baseline. In + * particular, no-props calls must not discard target effects. */ + context_drop_prefix(context); + ev.track_seed_reads = true; + if (props >= 0) { + msb_pass1(&ev, props); + } + ev.track_seed_reads = false; + if (props >= 0 && !ev.oom && !ev.seed_read && ev.nframes == 0) { + context_drop_items(context); + context_drop_target(context); + context->prefix = ev; + msb_work(sizeof(ev)); + context->prefix.seeds = NULL; + context->prefix.base = NULL; + context->prefix.live = NULL; + context->root = props; + context->ready = true; + memset(&ev, 0, sizeof(ev)); + msb_work(sizeof(ev)); + ev.m = m; + ev.seeds = &seeds; + ev.base = &context->prefix; + ev.live = &live; + ok = eval_init(&ev); + attach_prefix(&ev, &context->prefix); + } + } + if (ok && !ev.oom) { + msb_pass1(&ev, project); + int targets = context_nearest(context, &ev, f->dir, true); + ok = !ev.oom && apply_targets(context, &ev, targets); + if (ok) { + msb_ulist_t inc = {0}; + msb_ulist_t rem = {0}; + const char *const *implicit = msb_implicit(&ev, project); + for (int i = 0; !ev.oom && implicit && implicit[i]; i++) { + ulist_add(&ev, &inc, 'n', "", implicit[i]); + } + ok = !ev.oom && apply_items(context, &ev, &inc, &rem); + if (ok) { + eval_items(&ev, &ev, &inc, &rem); + ok = !ev.oom && msb_result(&ev, &inc, &rem, out); + } + } + } else { + ok = false; + } + } + eval_clear(&ev); + eval_clear(&seeds); + if (!ok) { + cbm_msb_result_free(out); + } + return ok; +} + +/* Traverse source spans only. Empty chains collapse at construction; the + * bounded explicit stack follows the balanced rope, not the import graph. */ +static void st_eval_rope(msb_eval_t *ev, st_rope_t *rope, msb_ulist_t *inc, msb_ulist_t *rem) { + st_rope_t *stack[CBM_SZ_64]; + int depth = 0; + msb_group_memo_t memo = {0}; + while (!ev->oom && (rope || depth)) { + if (!rope) { + rope = stack[--depth]; + continue; + } + msb_work(1); + if (!rope->left) { + msb_using(ev, &rope->item, inc, rem, &memo); + cbm_arena_reset(&ev->scratch); + rope = NULL; + } else { + if (depth == CBM_SZ_64) { + ev->oom = true; + break; + } + stack[depth++] = rope->right; + rope = rope->left; + } + } +} +static void st_trim(msb_state_t *s) { + while (s->empty) { + st_slab_t *b = s->empty; + st_empty_remove(s, b); + st_available_remove(s, b); + if (b->prev) + b->prev->next = b->next; + else + s->slabs = b->next; + if (b->next) + b->next->prev = b->prev; + s->bytes -= sizeof(*b); + cbm_free(CBM_MEM_CLASS_OTHER, b); + msb_work(1); + } +} +static void st_state_clear(msb_state_t *s) { + st_rope_drop(s, s->item_a); + st_rope_drop(s, s->item_b); + s->item_a = s->item_b = s->capture_a = s->capture_b = NULL; + for (int i = 0; i < s->files; i++) + for (int mode = 0; mode < 2; mode++) { + st_component_drop(s, s->components[(size_t)i * 2 + mode]); + msb_work(1); + } + s->bytes -= (size_t)s->files * 2 * sizeof(*s->components); + cbm_free(CBM_MEM_CLASS_OTHER, s->components); + s->components = NULL; + s->files = 0; + st_trim(s); + cbm_ht_free(s->symbols); + s->symbols = NULL; + cbm_arena_destroy(&s->names); + s->next_symbol = 0; +} +static bool st_state_epoch(msb_state_t *s, const cbm_msb_t *m) { + if (s->epoch == UINT64_MAX || (size_t)m->nfiles > SIZE_MAX / 2 / sizeof(*s->components)) + return false; + s->epoch++; + s->symbols = cbm_ht_create(CBM_SZ_64); + cbm_arena_init_lazy(&s->names, CBM_ARENA_APPEND_BLOCK); + s->components = (st_component_t **)cbm_calloc(CBM_MEM_CLASS_OTHER, + (size_t)m->nfiles * 2 * sizeof(*s->components)); + if (!s->symbols || (!s->components && m->nfiles)) + return false; + s->generation = m->generation; + s->files = m->nfiles; + s->bytes += (size_t)s->files * 2 * sizeof(*s->components); + msb_work((size_t)s->files * 2 * sizeof(*s->components)); + return true; +} +static msb_state_t *st_state_new(const cbm_msb_t *m) { + msb_state_t *s = (msb_state_t *)cbm_calloc(CBM_MEM_CLASS_OTHER, sizeof(*s)); + if (!s) + return NULL; + s->bytes = sizeof(*s); + msb_work(sizeof(*s)); + if (!st_state_epoch(s, m)) { + st_state_free(s); + return NULL; + } + return s; +} +static void st_state_free(msb_state_t *s) { + if (!s) + return; + st_state_clear(s); + cbm_free(CBM_MEM_CLASS_OTHER, s); +} +static st_component_t *st_run(msb_eval_t *ev, int file) { + ev->completed = NULL; + if (file >= 0 && !ev->oom) + msb_pass1(ev, file); + st_component_t *out = ev->completed; + ev->completed = NULL; + return out; +} +static bool st_context_eval(cbm_msb_eval_context_t *context, const char *project_rel, + cbm_msb_result_t *out) { + memset(out, 0, sizeof(*out)); + const cbm_msb_t *m = context->m; + msb_state_t *s = context->state; + if (s->generation != m->generation) { + context_drop_items(context); + st_state_clear(s); + context_clear_nearest(context); + if (!st_state_epoch(s, m)) { + st_state_clear(s); + return false; + } + } + int project = project_rel ? msb_file_index(m, project_rel) : CBM_NOT_FOUND; + if (project < 0 || !s->symbols || !s->components) + return false; + msb_eval_t ev = {.m = m, .state = s, .main_project = project}; + msb_live_t live = {.owners = {&ev, &context->item_owner}, + .metadata_bytes = &context->nearest_bytes, + .state_bytes = &s->bytes, + .state_names = &s->names}; + ev.live = &live; + /* One call-local arena pair; components never acquire an arena pair. */ + cbm_arena_init(&ev.arena); + cbm_arena_init(&ev.scratch); + ev_peak(&ev); + msb_seed(&ev, project); + cbm_arena_reset(&ev.scratch); + int props = + !ev.oom ? context_nearest(context, &ev, m->files[project].dir, false) : CBM_NOT_FOUND; + st_component_t *prefix = st_run(&ev, props); + st_component_t *local = st_run(&ev, project); + int targets = + !ev.oom ? context_nearest(context, &ev, m->files[project].dir, true) : CBM_NOT_FOUND; + st_component_t *target = st_run(&ev, targets); + bool ok = false; + if (!ev.oom) { + st_rope_t *a = prefix ? prefix->items : NULL, *b = target ? target->items : NULL; + if (s->item_a != a || s->item_b != b) { + context_drop_items(context); + st_rope_drop(s, s->item_a); + st_rope_drop(s, s->item_b); + s->item_a = st_rope_hold(a); + s->item_b = st_rope_hold(b); + } + s->capture_a = a; + s->capture_b = b; + msb_ulist_t inc = {0}, rem = {0}; + const char *const *implicit = msb_implicit(&ev, project); + for (int i = 0; !ev.oom && implicit && implicit[i]; i++) + ulist_add(&ev, &inc, 'n', "", implicit[i]); + ok = !ev.oom && apply_items(context, &ev, &inc, &rem); + if (ok) { + st_eval_rope(&ev, local ? local->items : NULL, &inc, &rem); + ok = !ev.oom && msb_result(&ev, &inc, &rem, out); + } + s->capture_a = s->capture_b = NULL; + } + while (ev.builder) { + st_component_t *c = ev.builder; + ev.builder = c->parent; + c->parent = NULL; + st_component_drop(s, c); + } + st_component_drop(s, prefix); + st_component_drop(s, local); + st_component_drop(s, target); + st_component_drop(s, ev.completed); + st_view_drop(s, &ev.view); + eval_clear(&ev); + st_trim(s); + if (!ok) + cbm_msb_result_free(out); + return ok; +} + +void cbm_msb_result_free(cbm_msb_result_t *r) { + if (!r) { + return; + } + cbm_free(CBM_MEM_CLASS_OTHER, r->mem); + memset(r, 0, sizeof(*r)); +} diff --git a/src/pipeline/doc_links_msbuild.h b/src/pipeline/doc_links_msbuild.h new file mode 100644 index 000000000..bdb618976 --- /dev/null +++ b/src/pipeline/doc_links_msbuild.h @@ -0,0 +1,146 @@ +/* + * doc_links_msbuild.h — MSBuild global usings of a C# project (R1), read from + * the scope blobs of the repository's project files. Private to the C# leg + * (doc_links_cs.c); no file is opened here. + */ +#ifndef CBM_PIPELINE_DOC_LINKS_MSBUILD_H +#define CBM_PIPELINE_DOC_LINKS_MSBUILD_H + +#include +#include + +/* The project files of one repository. */ +typedef struct cbm_msb cbm_msb_t; + +/* True when `scope` is the scope blob of an MSBuild project file (written by + * cbm_doclink_cs_project_scan_scope), not of a C# source file. */ +bool cbm_msb_is_project_scope(const char *scope); + +cbm_msb_t *cbm_msb_new(void); +void cbm_msb_free(cbm_msb_t *m); + +/* Add the project file `rel_path` (repository-relative, '/'-separated) with + * its blob; both are copied. false when memory ran out. */ +bool cbm_msb_add(cbm_msb_t *m, const char *rel_path, const char *scope); + +/* True when `rel_path` was added. */ +bool cbm_msb_has(const cbm_msb_t *m, const char *rel_path); + +/* False when the project file `rel_path` names an SDK that compiles nothing + * (Microsoft.Build.NoTargets, Microsoft.Build.Traversal: projects that run + * build steps or build other projects). Such a file is the project of no + * source file. True for every other project file, and for one that was not + * added: nothing is known about it. */ +bool cbm_msb_compiles(const cbm_msb_t *m, const char *rel_path); + +typedef struct { + char kind; /* n: a namespace, s: a type (using static), a: an alias */ + const char *alias; /* kind a: the alias; "" otherwise */ + const char *target; +} cbm_msb_using_t; + +typedef struct { + cbm_msb_using_t *usings; /* sorted, without duplicates; NULL when there are none */ + int count; + /* A item or the ImplicitUsings switch could not be evaluated: the + * project may have a global using this list lacks. */ + bool open; + int unevaluable; /* conditions, values and constructs that were not evaluated */ + int outside; /* imports of files the index does not hold */ + void *mem; +} cbm_msb_result_t; + +/* The global usings the project file `project_rel` gives its C# files: the + * nearest Directory.Build.props above it, the file itself, the nearest + * Directory.Build.targets, and what they import, evaluated in that order. + * false when memory ran out or `project_rel` was not added. The result is + * released with cbm_msb_result_free. */ +bool cbm_msb_eval(const cbm_msb_t *m, const char *project_rel, cbm_msb_result_t *out); +void cbm_msb_result_free(cbm_msb_result_t *r); + +/* A single-caller evaluation context; the model must outlive it. The first + * completed normal/poison effect per imported or nearest file is retained. + * Exact dependencies validate reuse; mismatches interpret normally without + * replacing that variant. Ordered persistent state shares unchanged subtrees + * across roots. Project-local evaluation remains call-local, and source item + * spans are evaluated with final properties using one first-variant result + * cache. Results own an independent publication block. Nearest-file metadata + * includes absent files. Model additions invalidate the complete reuse epoch. + * Large overlapping effects or changed inputs can still require more work. */ +typedef struct cbm_msb_eval_context cbm_msb_eval_context_t; +cbm_msb_eval_context_t *cbm_msb_eval_context_new(const cbm_msb_t *m); +void cbm_msb_eval_context_free(cbm_msb_eval_context_t *context); +bool cbm_msb_eval_context_eval(cbm_msb_eval_context_t *context, const char *project_rel, + cbm_msb_result_t *out); + +/* Operation labels stay available to the module in production builds; + * the controls and failure behavior exist only with test seams enabled. */ +typedef enum { + CBM_MSB_ITEM_FAIL_NONE = 0, + CBM_MSB_ITEM_FAIL_CAPTURE_ALLOC = 1, + CBM_MSB_ITEM_FAIL_UNIQUE_INSERT = 2, + CBM_MSB_ITEM_FAIL_APPLY_ALLOC = 3, + CBM_MSB_ITEM_FAIL_PUBLISH_ALLOC = 4, +} cbm_msb_item_fail_operation_t; + +typedef enum { + CBM_MSB_NODE_FAIL_NONE = 0, + CBM_MSB_NODE_FAIL_CAPTURE = 1, + CBM_MSB_NODE_FAIL_OVERLAY = 2, + CBM_MSB_NODE_FAIL_DEPENDENCY = 3, + CBM_MSB_NODE_FAIL_APPLY = 4, +} cbm_msb_node_fail_operation_t; + +#if defined(CBM_ENABLE_TEST_SEAMS) && CBM_ENABLE_TEST_SEAMS +/* Records consumed and peak owned string/array storage: arena capacities + * plus requested heap bytes, including result publication and directory + * memo payloads, packed state-pool capacities (including free slots), symbol + * storage and retained component/value/rope objects. Immutable input, + * hash-table bookkeeping and allocator + * metadata are excluded. */ +void cbm_msb_test_cost_reset(void); +void cbm_msb_test_cost(uint64_t *records, uint64_t *peak_bytes); +/* Deterministic work units since cost_reset: one per record, logical hash + * operation or property cleanup entry; bytes copied/appended by expression, + * property, string, array and publication operations; dependency comparison + * bytes and current-entry validation visits; owner initialization, + * transfer and clearing bytes; unique-item key construction, comparison and + * filtering operations; and bytes processed when forming property keys and + * walking directory paths; persistent-node lookup, overlay, masking, + * validation, source-span composition, allocation and iterative release. + * Test-only revision auditing is observation, not evaluator work. Hash operations + * count API probes, not internal bucket visits. Failed copies add no bytes. + * This measures selected operations, not time or every CPU instruction. */ +uint64_t cbm_msb_test_work(void); +/* Actual directory candidate probes made by msb_nearest since cost_reset. */ +uint64_t cbm_msb_test_nearest_steps(void); +void cbm_msb_test_fail_value_alloc_after(int nth); +bool cbm_msb_test_value_alloc_failed(void); +void cbm_msb_test_fail_prop_insert_after(int nth); +bool cbm_msb_test_prop_insert_failed(void); +uint64_t cbm_msb_test_value_live_bytes(void); +/* One-shot failure of the nth real operation of the selected kind. Zero + * disables it; failed insertions skip the actual set and require verification. */ +void cbm_msb_test_fail_item_operation(cbm_msb_item_fail_operation_t operation, int nth); +bool cbm_msb_test_item_operation_failed(void); + +/* One-shot failure immediately before acquiring a real component node slot. + * Empty/identity fast paths consume no allocation. NONE/nonpositive nth disable. */ +void cbm_msb_test_fail_node_alloc(cbm_msb_node_fail_operation_t operation, int nth); +bool cbm_msb_test_node_alloc_failed(void); + +/* Observations at real state operations; reset clears counters only, never + * revision IDs, pool history, witness metadata or armed fault controls. */ +typedef struct { + uint64_t allocations; + uint64_t slot_reuses; + uint64_t witnessed_slot_reuses; + uint64_t same_length_value_changes; + uint64_t witness_skips; + uint64_t revision_errors; +} cbm_msb_state_test_stats_t; +void cbm_msb_test_state_stats(cbm_msb_state_test_stats_t *out); + +#endif + +#endif /* CBM_PIPELINE_DOC_LINKS_MSBUILD_H */ diff --git a/src/pipeline/lsp_surface.c b/src/pipeline/lsp_surface.c index 4c3cde336..c96d48ad9 100644 --- a/src/pipeline/lsp_surface.c +++ b/src/pipeline/lsp_surface.c @@ -23,9 +23,12 @@ #include #include -#include "cbm.h" /* cbm_label_is_relation — reg-only surface membership */ +#include "cbm.h" /* cbm_label_is_relation — reg-only surface membership */ +#include "doclink.h" /* cbm_doclink_portable_scope — the "dl" key */ #include "foundation/log.h" +#include "foundation/mem_core.h" #include "foundation/sha256.h" +#include "pipeline/doc_links.h" /* cbm_doclinks_storable_scope */ #include "pipeline/worker_pool.h" #include "yyjson/yyjson.h" @@ -150,6 +153,30 @@ static char *surface_file_to_json(const CBMFileResult *result, const CBMLSPDef * yyjson_mut_obj_add_val(doc, root, "http", http); } + /* The doc-link scope (doclink.h): what other files' doc-comment + * references resolve against -- namespaces, usings, type and member + * declarations -- without line numbers, so a body edit keeps it. A + * changed scope changes the hash, and the closure planner reads it to + * decide whether a repair can stay local. Written only for files that + * have one: every other file's surface bytes stay exactly as they were. + * The defs decoder ignores this key; cbm_doclinks_scopes_from_surfaces + * reads it back for the files an incremental run does not re-extract. */ + if (result && result->doc_scope) { + /* a scope this run's reader refuses is stored as its language's + * rejected marker: a later run then treats the file as this one does + * (doc_links.h, scope_accepted) */ + char *marker = NULL; + const char *kept = cbm_doclinks_storable_scope(result->doc_scope, &marker); + char *portable = kept ? cbm_doclink_portable_scope(kept) : NULL; + cbm_free(CBM_MEM_CLASS_OTHER, marker); + if (!portable) { + yyjson_mut_doc_free(doc); + return NULL; /* a missing scope would diverge silently: fail the row */ + } + yyjson_mut_obj_add_strcpy(doc, root, "dl", portable); + cbm_free(CBM_MEM_CLASS_OTHER, portable); + } + char *json = yyjson_mut_write(doc, 0, out_len); yyjson_mut_doc_free(doc); return json; diff --git a/src/pipeline/pass_parallel.c b/src/pipeline/pass_parallel.c index 5c1d69089..8e8ea8fdd 100644 --- a/src/pipeline/pass_parallel.c +++ b/src/pipeline/pass_parallel.c @@ -70,6 +70,7 @@ enum { PP_CSHARP_M_PREFIX_LEN = 2 }; #define PP_RETAIN_PER_FILE_HARD_MAX_BYTES (32ULL * 1024 * 1024) /* 32 MiB per file */ #include "pipeline/pipeline.h" #include "pipeline/pipeline_internal.h" +#include "pipeline/doc_links.h" /* cbm_doclinks_resolve_file: MENTIONS beside CALLS */ #include "result_spill.h" #include "foundation/platform.h" /* cbm_resolve_cache_dir */ #include "pipeline/pass_lsp_cross.h" /* cbm_pxc_* helpers for fused cross-file LSP */ @@ -2728,14 +2729,20 @@ static void file_node_cache_clear(void) { tl_file_node = NULL; } +const cbm_gbuf_node_t *cbm_pipeline_file_node(const cbm_gbuf_t *gbuf, const char *project, + const char *rel) { + char *file_qn = cbm_pipeline_fqn_compute(project, rel, "__file__"); + const cbm_gbuf_node_t *node = cbm_gbuf_find_by_qn(gbuf, file_qn); + free(file_qn); + return node; +} + static const cbm_gbuf_node_t *file_node_for(const cbm_gbuf_t *gbuf, const char *project, const char *rel) { if (tl_file_node_gbuf == gbuf && tl_file_node_rel == rel) { return tl_file_node; } - char *file_qn = cbm_pipeline_fqn_compute(project, rel, "__file__"); - const cbm_gbuf_node_t *node = cbm_gbuf_find_by_qn(gbuf, file_qn); - free(file_qn); + const cbm_gbuf_node_t *node = cbm_pipeline_file_node(gbuf, project, rel); tl_file_node_gbuf = gbuf; tl_file_node_rel = rel; tl_file_node = node; @@ -3956,6 +3963,12 @@ static void resolve_worker(int worker_id, void *ctx_ptr) { atomic_fetch_add_explicit(&rc->time_ns_semantic, extract_now_ns() - _ph_t0, memory_order_relaxed); + /* ── MENTIONS (doc-comment references) ─────────────────── */ + if (rc->pctx && rc->pctx->doc_links) { + cbm_doclinks_resolve_file(rc->pctx->doc_links, file_idx, result, rc->main_gbuf, + ws->local_edge_buf); + } + cbm_registry_reach_cache_end(); cbm_registry_import_map_cache_end(); cbm_registry_resolve_cache_end(); diff --git a/src/pipeline/pipeline.c b/src/pipeline/pipeline.c index b5073a321..3e7094daa 100644 --- a/src/pipeline/pipeline.c +++ b/src/pipeline/pipeline.c @@ -14,11 +14,12 @@ #include "foundation/constants.h" -enum { CBM_DIR_PERMS = 0755, PL_RING = 4, PL_RING_MASK = 3, PL_SEQ_PASSES = 6 }; +enum { CBM_DIR_PERMS = 0755, PL_RING = 4, PL_RING_MASK = 3, PL_SEQ_PASSES = 7 }; #define PL_NSEC_PER_SEC 1000000000LL #include "pipeline/pipeline.h" #include "pipeline/artifact.h" #include "pipeline/pipeline_internal.h" +#include "pipeline/doc_links.h" #include "pipeline/lsp_surface.h" #include "pipeline/pass_lsp_cross.h" #include "pipeline/pass_ensemble_routing.h" @@ -260,6 +261,13 @@ struct cbm_pipeline { cbm_lsp_surface_row_t *surface_rows; int surface_row_count; + /* This run's doc_link_unresolved rows (doc_links.h), handed over by the + * resolve phase; doc_links_ran stays false until a doc-link phase did. */ + cbm_doc_link_row_t *doc_link_rows; + int doc_link_row_count; + bool doc_links_failed; + bool doc_links_ran; + /* Deterministic test-only seam at the final publication boundary. Kept * per pipeline so concurrent test/process activity cannot cross-trigger. */ void (*before_publish_hook)(cbm_pipeline_t *, const char *, void *); @@ -431,6 +439,42 @@ void cbm_pipeline_set_lsp_surfaces(cbm_pipeline_t *p, cbm_lsp_surface_row_t *row p->surface_row_count = count; } +void cbm_pipeline_set_doc_link_rows(cbm_pipeline_t *p, cbm_doc_link_row_t *rows, int count, + bool failed) { + if (!p) { + cbm_doclinks_free_rows(rows, count); + return; + } + cbm_doclinks_free_rows(p->doc_link_rows, p->doc_link_row_count); + p->doc_link_rows = rows; + p->doc_link_row_count = count; + p->doc_links_failed = failed; + p->doc_links_ran = true; +} + +void cbm_pipeline_take_doc_link_rows(cbm_pipeline_t *p, cbm_doc_link_row_t **rows, int *count, + bool *failed, bool *ran) { + *rows = p ? p->doc_link_rows : NULL; + *count = p ? p->doc_link_row_count : 0; + *failed = p ? p->doc_links_failed : true; + *ran = p ? p->doc_links_ran : false; + if (p) { + p->doc_link_rows = NULL; + p->doc_link_row_count = 0; + p->doc_links_failed = false; + p->doc_links_ran = false; + } +} + +/* Forget a previous run's doc-link rows (a pipeline object can run twice). */ +static void pipeline_reset_doc_links(cbm_pipeline_t *p) { + cbm_doclinks_free_rows(p->doc_link_rows, p->doc_link_row_count); + p->doc_link_rows = NULL; + p->doc_link_row_count = 0; + p->doc_links_failed = false; + p->doc_links_ran = false; +} + void cbm_pipeline_free(cbm_pipeline_t *p) { if (!p) { return; @@ -461,6 +505,7 @@ void cbm_pipeline_free(cbm_pipeline_t *p) { cbm_store_free_lsp_surfaces(p->surface_rows, p->surface_row_count); p->surface_rows = NULL; p->surface_row_count = 0; + pipeline_reset_doc_links(p); cbm_git_context_free(&p->git_ctx); /* gbuf, store, registry freed during/after run */ /* Defensively free userconfig in case run() was never called or panicked */ @@ -1445,6 +1490,9 @@ static int run_sequential_pipeline(cbm_pipeline_t *p, cbm_pipeline_ctx_t *ctx, {cbm_pipeline_pass_calls, "calls", false}, {cbm_pipeline_pass_usages, "usages", false}, {cbm_pipeline_pass_semantic, "semantic", false}, + /* doc-comment references: a failure is recorded for publication + * (doc_links.status), never a failed index */ + {cbm_pipeline_pass_doc_links, "doc_links", true}, }; int rc = 0; for (int si = 0; si < PL_SEQ_PASSES && rc == 0; si++) { @@ -1693,9 +1741,16 @@ static int run_parallel_pipeline(cbm_pipeline_t *p, cbm_pipeline_ctx_t *ctx, cbm_log_info("pass.timing", "pass", "lsp_cross_prepare", "elapsed_ms", itoa_buf((int)elapsed_ms(*t))); pipeline_phase_mark("lsp_cross_prepare"); + /* Doc-comment references resolve beside CALLS in the resolve workers; + * their indexes need every file's scope and the complete node set. */ + cbm_clock_gettime(CLOCK_MONOTONIC, t); + cbm_doclinks_begin(ctx, files, file_count, cache); + cbm_log_info("pass.timing", "pass", "doc_links_prepare", "elapsed_ms", + itoa_buf((int)elapsed_ms(*t))); cbm_clock_gettime(CLOCK_MONOTONIC, t); rc = cbm_parallel_resolve(ctx, files, file_count, cache, &shared_ids, worker_count, all_defs, def_count, def_modules, module_def_index, &cross_registries); + cbm_doclinks_end(ctx); cbm_log_info("pass.timing", "pass", "parallel_resolve", "elapsed_ms", itoa_buf((int)elapsed_ms(*t))); pipeline_phase_mark("parallel_resolve"); @@ -2253,6 +2308,31 @@ int cbm_pipeline_publish_generation(const cbm_pipeline_generation_t *generation) return cbm_pipeline_publish_staged(stage_path, generation, true, false); } +/* Write the generation's doc_link_unresolved rows (plus the error marker when + * the doc-link layer failed). */ +static int publish_doc_links(cbm_store_t *store, const cbm_pipeline_generation_t *generation) { + if (!generation->doc_links_failed) { + return cbm_store_doc_links_replace(store, generation->project, generation->doc_link_rows, + generation->doc_link_row_count); + } + cbm_log_error("doc_links.error", "phase", "publish", "reason", "layer_failed", "project", + generation->project); + int n = generation->doc_link_row_count; + cbm_doc_link_row_t *rows = (cbm_doc_link_row_t *)cbm_alloc( + CBM_MEM_CLASS_STORE, (size_t)(n + SKIP_ONE) * sizeof(*rows)); + if (!rows) { + return CBM_STORE_ERR; + } + if (n > 0) { + memcpy(rows, generation->doc_link_rows, (size_t)n * sizeof(*rows)); + } + rows[n] = (cbm_doc_link_row_t){ + .rel_path = "", .line = 0, .syntax = "", .raw = "doc-link layer failed", .reason = "error"}; + int rc = cbm_store_doc_links_replace(store, generation->project, rows, n + SKIP_ONE); + cbm_free(CBM_MEM_CLASS_STORE, rows); /* the array only: the strings are borrowed */ + return rc; +} + /* Complete and publish an already-materialized staging database: metadata * writes, FTS policy, integrity, seal, then the shared finalize leg. Takes * ownership of stage_path (frees it on every path). fts_wholesale selects @@ -2325,6 +2405,15 @@ int cbm_pipeline_publish_staged(char *stage_path, const cbm_pipeline_generation_ cbm_log_info("publish.timing", "block", "coverage_replace", "elapsed_ms", itoa_buf((int)elapsed_ms(t_pub))); cbm_clock_gettime(CLOCK_MONOTONIC, &t_pub); + /* Unresolved doc-comment references belong to the generation, like its + * coverage rows. A failed doc-link layer is recorded as an error marker + * row (rel_path "", reason "error") so index_status can say so. */ + if (ok && publish_doc_links(store, generation) != CBM_STORE_OK) { + ok = false; + } + cbm_log_info("publish.timing", "block", "doc_links", "elapsed_ms", + itoa_buf((int)elapsed_ms(t_pub))); + cbm_clock_gettime(CLOCK_MONOTONIC, &t_pub); /* The column list lives in cbm_store_fts_rebuild() alone — see the delta * merge, which must index the SAME columns or prose goes missing on the * warm path while a full reindex looks perfect. */ @@ -2553,6 +2642,10 @@ static int dump_and_persist_hashes(cbm_pipeline_t *p, const cbm_file_hash_t *bas }, .surface_rows = p->surface_rows, .surface_row_count = p->surface_row_count, + .doc_link_rows = p->doc_link_rows, + .doc_link_row_count = p->doc_link_row_count, + /* every full route runs a doc-link phase; none having run is a fault */ + .doc_links_failed = p->doc_links_failed || !p->doc_links_ran, }; free(db_dir); @@ -2728,6 +2821,7 @@ static int cbm_pipeline_run_staged(cbm_pipeline_t *p) { bool restore_requested_discovery = false; p->mode = p->requested_mode; + pipeline_reset_doc_links(p); bool mode_promoted = promote_mode_to_existing_coverage(p); /* cbm_pipeline_new() may precede the actual run by an arbitrary interval. diff --git a/src/pipeline/pipeline_incremental.c b/src/pipeline/pipeline_incremental.c index 99195cc2c..8ac3db44b 100644 --- a/src/pipeline/pipeline_incremental.c +++ b/src/pipeline/pipeline_incremental.c @@ -20,17 +20,20 @@ enum { INCR_RING_BUF = 4, INCR_RING_MASK = 3, INCR_TS_BUF = 24 }; #include "sqlite3.h" #include "yyjson/yyjson.h" #include "pipeline/pipeline_internal.h" +#include "pipeline/doc_links.h" #include "store/store.h" #include "graph_buffer/graph_buffer.h" #include "discover/discover.h" #include "foundation/log.h" #include "foundation/hash_table.h" +#include "foundation/mem_core.h" #include "foundation/compat.h" #include "foundation/compat_fs.h" #include "foundation/compat_thread.h" #include "foundation/platform.h" #include "foundation/sha256.h" +#include #include #include #include @@ -1126,14 +1129,34 @@ typedef struct { int n_dependents; cbm_lsp_surface_row_t *stored_rows; /* whole previous generation */ int stored_count; + /* The previous generation's doc_link_unresolved rows: carried forward for + * the files the repair does not re-extract. */ + cbm_doc_link_row_t *doc_rows; + int doc_row_count; } closure_plan_t; static void closure_plan_free(closure_plan_t *plan) { free(plan->files); cbm_store_free_lsp_surfaces(plan->stored_rows, plan->stored_count); + cbm_store_free_doc_links(plan->doc_rows, plan->doc_row_count); memset(plan, 0, sizeof(*plan)); } +/* Add a heap copy of `name` to a heap-keyed name set (value = key; released + * by surface_name_set_free). No-op when present; false when the copy could + * not be made. */ +static bool surface_name_set_put(CBMHashTable *set, const char *name) { + if (!name[0] || cbm_ht_get(set, name)) { + return true; + } + char *copy = strdup(name); + if (!copy) { + return false; + } + cbm_ht_set(set, copy, copy); + return true; +} + /* Short names a surface JSON defines ("lsp"[].sn plus "reg"[].n), as a * heap-keyed set. Returns NULL on parse failure — callers decline to FULL. */ static CBMHashTable *surface_name_set(const char *defs_json) { @@ -1159,11 +1182,8 @@ static CBMHashTable *surface_name_set(const char *defs_json) { yyjson_val *item; yyjson_arr_foreach(arrs[a], idx, max, item) { const char *name = yyjson_get_str(yyjson_obj_get(item, arr_keys[a])); - if (name && name[0] && !cbm_ht_get(set, name)) { - char *copy = strdup(name); - if (copy) { - cbm_ht_set(set, copy, copy); - } + if (name) { + (void)surface_name_set_put(set, name); } } } @@ -1217,6 +1237,218 @@ static int surface_added_names(const char *stored_json, const char *fresh_json, return rc; } +/* ── Doc-link closure rules ─────────────────────────────────────── + * + * MENTIONS edges are owned by their source file like CALLS: a re-extracted + * file rewrites its own edges and doc_link_unresolved rows, and the files + * with edges INTO a changed file are dependents. What the edge set cannot + * see is decided by the language's resolver hooks (doc_links.h), never here: + * - a scope input (a file without a scope of its own that sets the scope of + * a language's files; no language has one today): a change declines to + * FULL; + * - a scope delta ("dl" surface key) the language calls GLOBAL can re-route + * or re-classify references in files with no edge into the changed file: + * FULL, exactly like an added name. A deleted file with a scope is GLOBAL. + * C#'s MSBuild project files come in here: each has a scope blob, and any + * change to it is GLOBAL (it sets the global usings of a whole project); + * - a REMOVED name can make another file's unresolved reference resolve (an + * overload group shrinks) or change its reason: the files whose rows + * mention it re-resolve with the closure. */ + +/* Add s[0..n) to a heap-keyed name set. false when it cannot be recorded (too + * long for a key, or out of memory): callers fail closed. */ +static bool name_set_add_n(CBMHashTable *set, const char *s, size_t n) { + char key[CBM_SZ_1K]; + if (n >= sizeof(key)) { + return false; + } + memcpy(key, s, n); + key[n] = '\0'; + return surface_name_set_put(set, key); +} + +/* The "dl" scope of a surface JSON (a CBM_MEM_CLASS_OTHER block; NULL when it + * has none). false when the row's scope cannot be read: the caller declines. */ +static bool surface_dl_scope(const char *defs_json, char **out) { + return cbm_doclinks_scope_from_surface_json(defs_json, out) == 0; +} + +/* cbm_doclink_name_fn over a heap-keyed name set. */ +static bool removed_name_put(void *ud, const char *name, size_t len) { + return name_set_add_n((CBMHashTable *)ud, name, len); +} + +typedef struct { + CBMHashTable *fresh; + CBMHashTable *removed; + bool failed; +} removed_walk_t; + +static void removed_name_visitor(const char *key, void *value, void *userdata) { + (void)value; + removed_walk_t *walk = (removed_walk_t *)userdata; + if ((!walk->fresh || !cbm_ht_get(walk->fresh, key)) && + !name_set_add_n(walk->removed, key, strlen(key))) { + walk->failed = true; + } +} + +/* Short names of `stored_json` absent from `fresh_json` (all of them when + * fresh_json is NULL: a deleted file). */ +static int surface_removed_names(const char *stored_json, const char *fresh_json, + CBMHashTable *removed) { + CBMHashTable *stored_set = surface_name_set(stored_json); + CBMHashTable *fresh_set = fresh_json ? surface_name_set(fresh_json) : NULL; + if (!stored_set || (fresh_json && !fresh_set)) { + surface_name_set_free(stored_set); + surface_name_set_free(fresh_set); + return CBM_NOT_FOUND; + } + removed_walk_t walk = {.fresh = fresh_set, .removed = removed, .failed = false}; + cbm_ht_foreach(stored_set, removed_name_visitor, &walk); + surface_name_set_free(stored_set); + surface_name_set_free(fresh_set); + return walk.failed ? CBM_NOT_FOUND : 0; +} + +/* The doc-link view of one changed (fresh_json set) or deleted (fresh_json + * NULL) file: the names it no longer declares go to `removed`. false when the + * change cannot be repaired file by file (the rules above) or the stored data + * cannot be read -- either way the caller declines. */ +static bool doc_scope_repairable(const char *stored_json, const char *fresh_json, + CBMHashTable *removed) { + char *stored_dl = NULL; + char *fresh_dl = NULL; + bool ok = surface_dl_scope(stored_json, &stored_dl) && + (!fresh_json || surface_dl_scope(fresh_json, &fresh_dl)) && + cbm_doclinks_scope_delta(stored_dl, fresh_dl, removed_name_put, removed) == + CBM_DOCLINK_DELTA_LOCAL && + surface_removed_names(stored_json, fresh_json, removed) == 0; + cbm_free(CBM_MEM_CLASS_OTHER, stored_dl); + cbm_free(CBM_MEM_CLASS_OTHER, fresh_dl); + return ok; +} + +typedef struct { + CBMHashTable *closure; + CBMHashTable *files; + int added; +} row_dep_walk_t; + +/* Add a row-named dependent to the closure when discovery still has it (a + * deleted file's rows are dropped by the carry-forward anyway). */ +static void row_dep_visitor(const char *key, void *value, void *userdata) { + (void)value; + row_dep_walk_t *walk = (row_dep_walk_t *)userdata; + if (!cbm_ht_get(walk->files, key) || cbm_ht_get(walk->closure, key)) { + return; + } + cbm_ht_set(walk->closure, key, (void *)key); + walk->added++; +} + +/* Files whose unresolved doc-link rows mention a removed name as an + * identifier token. Keys are borrowed from `rows`. */ +/* A byte of an identifier as the resolvers read one: a letter, a digit, '_', + * and every byte of a multi-byte character (doc_links_cs.c ident_ok). */ +static bool doc_row_ident_byte(unsigned char c) { + return isalnum(c) || c == '_' || c >= CBM_SZ_128; +} + +/* True when the identifier raw[0, n) is one of the removed names -- of any + * length. When memory runs out the answer is yes: re-resolving a file that + * did not need it costs time, keeping a stale row costs correctness. */ +static bool doc_row_token_removed(const CBMHashTable *removed, const char *s, size_t n) { + char small[CBM_SZ_512]; + char *tok = n < sizeof(small) ? small : (char *)cbm_alloc(CBM_MEM_CLASS_OTHER, n + SKIP_ONE); + if (!tok) { + return true; + } + memcpy(tok, s, n); + tok[n] = '\0'; + bool hit = cbm_ht_get(removed, tok) != NULL; + if (tok != small) { + cbm_free(CBM_MEM_CLASS_OTHER, tok); + } + return hit; +} + +static void doc_row_name_dependents(const cbm_doc_link_row_t *rows, int n, + const CBMHashTable *removed, CBMHashTable *out_paths) { + if (!removed || cbm_ht_count(removed) == 0) { + return; + } + for (int i = 0; i < n; i++) { + const char *raw = rows[i].raw; + const char *rel = rows[i].rel_path; + if (!raw || !rel || !rel[0] || cbm_ht_get(out_paths, rel)) { + continue; + } + for (const char *p = raw; *p;) { + while (*p && !doc_row_ident_byte((unsigned char)*p)) { + p++; + } + const char *s = p; + while (*p && doc_row_ident_byte((unsigned char)*p)) { + p++; + } + size_t tl = (size_t)(p - s); + if (tl > 0 && doc_row_token_removed(removed, s, tl)) { + cbm_ht_set(out_paths, rel, (void *)rel); + break; + } + } + } +} + +/* Doc-link carry-forward state of the legacy partial route: the previous + * rows, the stored scopes of the files it does not re-extract, and the set + * of re-extracted or deleted paths whose old rows it replaces. */ +typedef struct { + cbm_doc_link_row_t *old_rows; + int old_count; + cbm_doclink_scope_t *scopes; + int scope_count; + CBMHashTable *replaced; /* heap keys (value = key) */ + bool ok; +} legacy_doc_t; + +static void legacy_doc_load(legacy_doc_t *d, cbm_store_t *store, const char *project, + const cbm_file_info_t *changed, int ci, char *const *deleted, + int deleted_count) { + memset(d, 0, sizeof(*d)); + d->replaced = cbm_ht_create(CBM_SZ_64); + d->ok = d->replaced != NULL; + for (int i = 0; d->ok && i < ci; i++) { + d->ok = name_set_add_n(d->replaced, changed[i].rel_path, strlen(changed[i].rel_path)); + } + for (int i = 0; d->ok && i < deleted_count; i++) { + d->ok = name_set_add_n(d->replaced, deleted[i], strlen(deleted[i])); + } + if (d->ok && cbm_store_doc_links_get(store, project, &d->old_rows, &d->old_count, NULL) != + CBM_STORE_OK) { + d->ok = false; + } + cbm_lsp_surface_row_t *surf = NULL; + int surf_count = 0; + if (d->ok && cbm_store_get_lsp_surfaces(store, project, &surf, &surf_count) == CBM_STORE_OK && + cbm_doclinks_scopes_from_surfaces(surf, surf_count, d->replaced, &d->scopes, + &d->scope_count) != 0) { + d->ok = false; + } + cbm_store_free_lsp_surfaces(surf, surf_count); + if (!d->ok) { + cbm_log_error("doc_links.error", "phase", "legacy_carry_forward", "reason", "read"); + } +} + +static void legacy_doc_free(legacy_doc_t *d) { + cbm_store_free_doc_links(d->old_rows, d->old_count); + cbm_doclinks_free_scopes(d->scopes, d->scope_count); + surface_name_set_free(d->replaced); + memset(d, 0, sizeof(*d)); +} + /* Run parallel or sequential extract+resolve for changed files. Any failure * aborts before persistence: the caller discards this in-memory graph and * preserves the old on-disk database and its retryable hashes. */ @@ -1367,9 +1599,11 @@ static int run_extract_resolve(cbm_pipeline_ctx_t *ctx, cbm_file_info_t *changed "elapsed_ms", itoa_buf((int)elapsed_ms(t))); } cbm_clock_gettime(CLOCK_MONOTONIC, &t); + cbm_doclinks_begin(ctx, changed_files, ci, cache); rc = cbm_parallel_resolve(ctx, changed_files, ci, cache, &shared_ids, worker_count, all_defs, all_def_count, closure ? closure->def_modules : NULL, module_def_index, registries_arg); + cbm_doclinks_end(ctx); if (module_def_index) { cbm_pxc_free_module_def_index(module_def_index); } @@ -1423,6 +1657,9 @@ static int run_extract_resolve(cbm_pipeline_ctx_t *ctx, cbm_file_info_t *changed if (rc == 0) { rc = cbm_pipeline_check_cancel(ctx); } + if (rc == 0) { + (void)cbm_pipeline_pass_doc_links(ctx, changed_files, ci); + } if (owns_cache) { free_incremental_result_cache(cache, ci); ctx->result_cache = prior_cache; @@ -1515,12 +1752,19 @@ static int run_postpasses(cbm_pipeline_ctx_t *ctx, cbm_file_info_t *changed_file } /* Publish the test-only legacy partial result through the same atomic * generation boundary as full indexing. */ +typedef struct { + const cbm_doc_link_row_t *rows; + int count; + bool failed; +} legacy_doc_rows_t; + static int dump_and_persist(cbm_gbuf_t *gbuf, const char *db_path, const char *project, atomic_int *cancelled, const cbm_file_hash_t *manifest, int manifest_count, const char *adr_content, const cbm_coverage_row_t *cov, int cov_count, const cbm_coverage_meta_t *meta_template, - const cbm_lsp_surface_row_t *surface_rows, int surface_row_count) { + const cbm_lsp_surface_row_t *surface_rows, int surface_row_count, + legacy_doc_rows_t doc) { struct timespec t; cbm_clock_gettime(CLOCK_MONOTONIC, &t); cbm_pipeline_generation_t generation = { @@ -1536,6 +1780,9 @@ static int dump_and_persist(cbm_gbuf_t *gbuf, const char *db_path, const char *p .coverage_meta = meta_template ? *meta_template : (cbm_coverage_meta_t){0}, .surface_rows = surface_rows, .surface_row_count = surface_row_count, + .doc_link_rows = doc.rows, + .doc_link_row_count = doc.count, + .doc_links_failed = doc.failed, }; int rc = cbm_pipeline_publish_generation(&generation); cbm_log_info("incremental.dump", "rc", itoa_buf(rc), "elapsed_ms", @@ -1718,7 +1965,12 @@ static int closure_try_plan(cbm_pipeline_t *p, cbm_store_t *store, const char *p cbm_path_alias_collection_t *plan_aliases = NULL; int n_changed = 0; int n_deleted = 0; - if (!fresh_by_path || !files_by_path || !stored_by_path || !closure_set) { + int n_row_dependents = 0; + /* names a changed or deleted file no longer declares (doc-link rows) */ + CBMHashTable *removed_names = cbm_ht_create(CBM_SZ_64); + CBMHashTable *row_deps = cbm_ht_create(CBM_SZ_64); + if (!fresh_by_path || !files_by_path || !stored_by_path || !closure_set || !removed_names || + !row_deps) { decline = "alloc"; goto done; } @@ -1795,6 +2047,14 @@ static int closure_try_plan(cbm_pipeline_t *p, cbm_store_t *store, const char *p decline = "no_file_delta"; goto done; } + /* A doc-link scope input scopes files other than itself and has no scope + * blob the surface comparison below could judge. */ + for (int i = 0; i < n_changed + n_deleted; i++) { + if (cbm_doclinks_is_scope_input(changed_paths[i])) { + decline = "doc_scope_input_changed"; + goto done; + } + } /* Load the previous generation's surfaces and probe the changed files' * fresh ones. Missing rows fail closed. */ @@ -1804,6 +2064,27 @@ static int closure_try_plan(cbm_pipeline_t *p, cbm_store_t *store, const char *p decline = "no_surface_rows"; goto done; } + /* The previous doc_link_unresolved rows: carried forward by the repair. + * A generation without them, or one whose doc-link layer failed, cannot + * be repaired file by file. */ + { + bool doc_present = false; + if (cbm_store_doc_links_get(store, project, &plan->doc_rows, &plan->doc_row_count, + &doc_present) != CBM_STORE_OK) { + decline = "doc_links_read_failed"; + goto done; + } + if (!doc_present) { + decline = "doc_links_missing"; + goto done; + } + for (int i = 0; i < plan->doc_row_count; i++) { + if (!plan->doc_rows[i].rel_path || !plan->doc_rows[i].rel_path[0]) { + decline = "doc_links_error"; + goto done; + } + } + } CBMHashTable *rows_by_path = cbm_ht_create((size_t)plan->stored_count * PAIR_LEN); if (!rows_by_path) { decline = "alloc"; @@ -1866,18 +2147,30 @@ static int closure_try_plan(cbm_pipeline_t *p, cbm_store_t *store, const char *p continue; /* body edit: the file re-resolves, nobody else does */ } bool added = false; + const char *why = NULL; if (surface_added_names(stored_row->defs_json, fresh_row->defs_json, &added) != 0 || added) { + why = "added_definition_names"; + } else if (!doc_scope_repairable(stored_row->defs_json, fresh_row->defs_json, + removed_names)) { + why = "doc_scope_changed"; + } + if (why) { free(dep_targets); cbm_ht_free(rows_by_path); - decline = "added_definition_names"; + decline = why; goto done; } n_surface_changed++; dep_targets[dep_target_count++] = changed_paths[i]; } + bool gone_scope_changed = false; /* a deleted file takes its declarations along */ for (int i = 0; i < n_deleted; i++) { dep_targets[dep_target_count++] = changed_paths[n_changed + i]; + const cbm_lsp_surface_row_t *gone = cbm_ht_get(rows_by_path, changed_paths[n_changed + i]); + if (gone && !doc_scope_repairable(gone->defs_json, NULL, removed_names)) { + gone_scope_changed = true; + } } cbm_ht_free(rows_by_path); @@ -1889,6 +2182,10 @@ static int closure_try_plan(cbm_pipeline_t *p, cbm_store_t *store, const char *p goto done; } free(dep_targets); + if (gone_scope_changed) { + decline = "doc_scope_changed"; + goto done; + } /* Closure = changed ∪ dependents. Every member must be in the current * discovery (a dependent outside it cannot be re-resolved). */ @@ -1906,6 +2203,14 @@ static int closure_try_plan(cbm_pipeline_t *p, cbm_store_t *store, const char *p cbm_ht_set(closure_set, dependents[i], dependents[i]); } } + /* Files whose unresolved doc-link references name something a changed or + * deleted file no longer declares re-resolve too. */ + doc_row_name_dependents(plan->doc_rows, plan->doc_row_count, removed_names, row_deps); + { + row_dep_walk_t walk = {.closure = closure_set, .files = files_by_path, .added = 0}; + cbm_ht_foreach(row_deps, row_dep_visitor, &walk); + n_row_dependents = walk.added; + } /* closure_count == 0 is legitimate: a deleted-only delta with no * dependents has nothing to re-parse, but the purge itself still needs * the executor. */ @@ -1936,7 +2241,8 @@ static int closure_try_plan(cbm_pipeline_t *p, cbm_store_t *store, const char *p cbm_log_info("incremental.closure_plan", "changed", itoa_buf(n_changed), "surface_changed", itoa_buf(n_surface_changed), "deleted", itoa_buf(n_deleted), "dependents", itoa_buf(dependent_count)); - cbm_log_info("incremental.closure_plan_done", "closure", itoa_buf(plan->count), "elapsed_ms", + cbm_log_info("incremental.closure_plan_done", "closure", itoa_buf(plan->count), + "doc_link_dependents", itoa_buf(n_row_dependents), "elapsed_ms", itoa_buf((int)elapsed_ms(t))); done: @@ -1949,10 +2255,13 @@ static int closure_try_plan(cbm_pipeline_t *p, cbm_store_t *store, const char *p cbm_ht_free(files_by_path); cbm_ht_free(stored_by_path); cbm_ht_free(closure_set); + cbm_ht_free(row_deps); /* keys borrowed from plan->doc_rows */ + surface_name_set_free(removed_names); if (decline) { cbm_log_info("incremental.closure_decline", "reason", decline, "elapsed_ms", itoa_buf((int)elapsed_ms(t))); cbm_store_free_lsp_surfaces(plan->stored_rows, plan->stored_count); + cbm_store_free_doc_links(plan->doc_rows, plan->doc_row_count); memset(plan, 0, sizeof(*plan)); return 0; } @@ -1995,6 +2304,13 @@ static int run_closure_delta(cbm_pipeline_t *p, const char *db_path, const char int manifest_count = 0; cbm_coverage_row_t *cov = NULL; int cov_n = 0; + cbm_doclink_scope_t *doc_base = NULL; + int doc_base_count = 0; + cbm_doc_link_row_t *doc_fresh = NULL; + int doc_fresh_count = 0; + cbm_doc_link_row_t *doc_rows = NULL; + int doc_row_count = 0; + bool doc_failed = false; cbm_clock_gettime(CLOCK_MONOTONIC, &t); if (cbm_delta_stage_clone(db_path, &stage) != 0) { @@ -2167,6 +2483,12 @@ static int run_closure_delta(cbm_pipeline_t *p, const char *db_path, const char cbm_pipeline_get_excluded(p, &excluded_dirs, &excluded_count); path_aliases = cbm_load_path_aliases_excluded(cbm_pipeline_repo_path(p), excluded_dirs, excluded_count); + /* Doc-link scopes of every file this repair does not re-extract. */ + if (cbm_doclinks_scopes_from_surfaces(plan->stored_rows, plan->stored_count, stale_surface_set, + &doc_base, &doc_base_count) != 0) { + cbm_log_error("delta.err", "phase", "doc_link_scopes"); + goto out; + } cbm_pipeline_ctx_t ctx = { .project_name = project, .repo_path = cbm_pipeline_repo_path(p), @@ -2178,6 +2500,8 @@ static int run_closure_delta(cbm_pipeline_t *p, const char *db_path, const char .path_aliases = path_aliases, .excluded_dirs = excluded_dirs, .excluded_count = excluded_count, + .doc_link_base = doc_base, + .doc_link_base_count = doc_base_count, }; for (int i = 0; i < ci; i++) { char *file_qn = cbm_pipeline_fqn_compute(project, changed_files[i].rel_path, "__file__"); @@ -2221,6 +2545,19 @@ static int run_closure_delta(cbm_pipeline_t *p, const char *db_path, const char } cbm_log_info("delta.repair", "files", itoa_buf(ci), "elapsed_ms", itoa_buf((int)elapsed_ms(t))); + /* Doc-link rows: the previous rows of every file not re-extracted (or + * deleted), then this repair's. */ + { + bool doc_ran = false; + cbm_pipeline_take_doc_link_rows(p, &doc_fresh, &doc_fresh_count, &doc_failed, &doc_ran); + doc_failed = doc_failed || !doc_ran; + if (cbm_doclinks_merge_rows(plan->doc_rows, plan->doc_row_count, stale_surface_set, + doc_fresh, doc_fresh_count, &doc_rows, &doc_row_count) != 0) { + cbm_log_error("delta.err", "phase", "doc_link_rows"); + goto out; + } + } + cbm_clock_gettime(CLOCK_MONOTONIC, &t); if (cbm_delta_patch(staging, project, gbuf, max_db_id, snapshot, snapshot_count) != 0) { goto out; @@ -2365,6 +2702,9 @@ static int run_closure_delta(cbm_pipeline_t *p, const char *db_path, const char .surface_rows = NULL, .surface_row_count = 0, .surfaces_in_place = true, + .doc_link_rows = doc_rows, + .doc_link_row_count = doc_row_count, + .doc_links_failed = doc_failed, }; cbm_store_close(staging); staging = NULL; @@ -2391,6 +2731,9 @@ static int run_closure_delta(cbm_pipeline_t *p, const char *db_path, const char } cbm_delta_free_snapshot(snapshot, snapshot_count); surface_name_set_free(stale_surface_set); + cbm_doclinks_free_scopes(doc_base, doc_base_count); + cbm_doclinks_free_rows(doc_fresh, doc_fresh_count); + cbm_doclinks_free_rows(doc_rows, doc_row_count); if (cr_arena_live) { cbm_store_free_lsp_surfaces(cr.fresh_rows, cr.fresh_count); for (int i = 0; i < cr.def_module_count; i++) { @@ -2426,6 +2769,25 @@ static int run_closure_delta(cbm_pipeline_t *p, const char *db_path, const char /* ── Incremental pipeline entry point ────────────────────────────── */ +/* The stored generation's doc-link layer is usable: its doc_link_unresolved + * table exists and records no failure of the layer. */ +static bool incr_doc_links_current(cbm_store_t *store, const char *project) { + cbm_doc_link_reason_count_t *reasons = NULL; + int reason_count = 0; + cbm_doc_link_row_t *samples = NULL; + int sample_count = 0; + bool present = false; + bool current = cbm_store_doc_links_summary(store, project, &reasons, &reason_count, &samples, + &sample_count, 0, &present) == CBM_STORE_OK && + present; + for (int i = 0; current && i < reason_count; i++) { + current = strcmp(reasons[i].reason, "error") != 0; + } + cbm_store_free_doc_link_reasons(reasons, reason_count); + cbm_store_free_doc_links(samples, sample_count); + return current; +} + int cbm_pipeline_run_incremental(cbm_pipeline_t *p, const char *db_path, cbm_file_info_t *files, int file_count, const cbm_file_hash_t *baseline_manifest, int baseline_count, bool force_full_on_mismatch) { @@ -2478,7 +2840,11 @@ int cbm_pipeline_run_incremental(cbm_pipeline_t *p, const char *db_path, cbm_fil meta.coverage_version == CBM_SEMANTIC_INDEX_VERSION && meta.hash_records_complete && meta.index_mode && strcmp(meta.index_mode, mode_name) == 0; - bool exact = metadata_current && + /* A generation whose doc-link layer is missing or failed is not + * current even with identical inputs: index_status tells the user to + * re-run, and the re-run has to rebuild. */ + bool doc_links_current = metadata_current && incr_doc_links_current(store, project); + bool exact = doc_links_current && cbm_pipeline_semantic_manifests_equal(stored, stored_count, baseline_manifest, baseline_count); cbm_store_coverage_meta_clear(&meta); @@ -2520,7 +2886,9 @@ int cbm_pipeline_run_incremental(cbm_pipeline_t *p, const char *db_path, cbm_fil #if defined(CBM_INCREMENTAL_TEST_API) && CBM_INCREMENTAL_TEST_API incr_test_set_last_route(CBM_INCREMENTAL_ROUTE_FORCED_FULL); #endif - cbm_log_info("incremental.force_full", "reason", "semantic_manifest_changed"); + cbm_log_info("incremental.force_full", "reason", + metadata_current && !doc_links_current ? "doc_links_not_current" + : "semantic_manifest_changed"); return CBM_PIPELINE_FORCE_FULL_REINDEX; } } @@ -2684,6 +3052,8 @@ int cbm_pipeline_run_incremental(cbm_pipeline_t *p, const char *db_path, cbm_fil return CBM_PIPELINE_ABORT_PRESERVE_DB; } + legacy_doc_t legacy_doc; + legacy_doc_load(&legacy_doc, store, project, changed_files, ci, deleted, deleted_count); cbm_store_close(store); /* Snapshot inbound cross-file edges into changed files BEFORE purging, so @@ -2751,6 +3121,8 @@ int cbm_pipeline_run_incremental(cbm_pipeline_t *p, const char *db_path, cbm_fil .path_aliases = path_aliases, .excluded_dirs = excluded_dirs, .excluded_count = excluded_count, + .doc_link_base = legacy_doc.scopes, + .doc_link_base_count = legacy_doc.scope_count, }; for (int i = 0; i < ci; i++) { @@ -2810,9 +3182,29 @@ int cbm_pipeline_run_incremental(cbm_pipeline_t *p, const char *db_path, cbm_fil free_mode_skipped(mode_skipped, mode_skipped_count); free(saved_adr); cbm_gbuf_free(existing); + legacy_doc_free(&legacy_doc); return CBM_PIPELINE_ABORT_PRESERVE_DB; } + /* Doc-link rows: previous rows of the files not re-extracted, then this + * run's. A failed carry-forward read is a failed doc-link layer. */ + cbm_doc_link_row_t *doc_rows = NULL; + int doc_row_count = 0; + bool doc_failed = false; + { + cbm_doc_link_row_t *fresh = NULL; + int fresh_count = 0; + bool ran = false; + cbm_pipeline_take_doc_link_rows(p, &fresh, &fresh_count, &doc_failed, &ran); + doc_failed = doc_failed || !ran || !legacy_doc.ok; + if (cbm_doclinks_merge_rows(legacy_doc.old_rows, legacy_doc.old_count, legacy_doc.replaced, + fresh, fresh_count, &doc_rows, &doc_row_count) != 0) { + doc_failed = true; + } + cbm_doclinks_free_rows(fresh, fresh_count); + } + legacy_doc_free(&legacy_doc); + /* Coverage rows (#963): merge = previous FAILURE rows for files NOT * re-extracted this run + this run's fresh entries (changed files replace * their old rows — a file that parses cleanly now simply contributes @@ -2902,6 +3294,7 @@ int cbm_pipeline_run_incremental(cbm_pipeline_t *p, const char *db_path, cbm_fil free_mode_skipped(mode_skipped, mode_skipped_count); free(saved_adr); cbm_gbuf_free(existing); + cbm_doclinks_free_rows(doc_rows, doc_row_count); return manifest_rc == CBM_DISCOVER_LIMIT_EXCEEDED ? CBM_PIPELINE_RESOURCE_LIMIT : CBM_PIPELINE_ABORT_PRESERVE_DB; } @@ -2929,9 +3322,11 @@ int cbm_pipeline_run_incremental(cbm_pipeline_t *p, const char *db_path, cbm_fil * re-parsed files have no codec output, and publishing a stale row * would satisfy a future closure plan with yesterday's surface; an * empty table just routes the next incremental to a full rebuild. */ + legacy_doc_rows_t doc_out = {.rows = doc_rows, .count = doc_row_count, .failed = doc_failed}; int persist_rc = dump_and_persist(existing, db_path, project, cbm_pipeline_cancelled_ptr(p), manifest, - manifest_count, saved_adr, cov, cov_n, &coverage_meta, NULL, 0); + manifest_count, saved_adr, cov, cov_n, &coverage_meta, NULL, 0, doc_out); + cbm_doclinks_free_rows(doc_rows, doc_row_count); cbm_pipeline_free_semantic_manifest(manifest, manifest_count); free(saved_adr); free(cov); diff --git a/src/pipeline/pipeline_internal.h b/src/pipeline/pipeline_internal.h index a196dd5c5..62b9c4c41 100644 --- a/src/pipeline/pipeline_internal.h +++ b/src/pipeline/pipeline_internal.h @@ -156,6 +156,16 @@ typedef struct { * the incremental and probe routes still hand the cache array to passes * that index it directly, so they keep results in memory (follow-up). */ bool spill_allowed; + + /* Doc-comment references -> MENTIONS (doc_links.h). doc_links is the + * run's resolver state while the resolve phase runs (NULL otherwise); + * doc_link_base holds the stored scopes of the files an incremental run + * does not re-extract (borrowed from the route that loaded them); + * doc_links_failed records a failed build so publication can mark it. */ + struct cbm_doclinks *doc_links; + const struct cbm_doclink_scope *doc_link_base; + int doc_link_base_count; + bool doc_links_failed; } cbm_pipeline_ctx_t; /* ── Result-cache access contract (spill mode) ──────────────────────── @@ -174,6 +184,12 @@ void cbm_pipeline_result_release(CBMFileResult *r, bool loaded); /* Log the store counters, close and delete the store, drop the latch. */ void cbm_pipeline_spill_close(cbm_pipeline_ctx_t *ctx); +/* The File node of `rel` in `gbuf` (NULL when it has none): the one lookup + * for "this file as an edge source", by the name cbm_pipeline_fqn_compute + * gives every File node. */ +const cbm_gbuf_node_t *cbm_pipeline_file_node(const cbm_gbuf_t *gbuf, const char *project, + const char *rel); + /* Transcode an ObjectScript Studio Export XML file and compose every generated * UDL class into one cacheable result. The returned result owns all child * extraction arenas and is released with the ordinary cbm_free_result(). */ @@ -829,8 +845,11 @@ int cbm_pipeline_build_fresh_semantic_manifest(cbm_pipeline_t *p, const char *pr * (the enum name is no longer a segment); typedef names, anonymous-enum * constants and macro-prefixed functions are nodes; a bodyless * `struct X` is no node. An index written before this holds the old - * QNs for every unchanged file, so it is rebuilt in full once. */ -enum { CBM_SEMANTIC_INDEX_VERSION = 4 }; + * QNs for every unchanged file, so it is rebuilt in full once. + * 5: doc-comment references became MENTIONS edges and doc_link_unresolved + * rows, and C# LSP surfaces carry the doc-link scope ("dl"); an index + * built before has neither, so it rebuilds once on upgrade. */ +enum { CBM_SEMANTIC_INDEX_VERSION = 5 }; typedef struct { cbm_gbuf_t *gbuf; @@ -853,6 +872,13 @@ typedef struct { * into the staging store (delta patch); publish then skips the * wholesale delete+rewrite. */ bool surfaces_in_place; + /* The generation's doc_link_unresolved rows (complete: an incremental + * route passes the merge of carried-forward and fresh rows), and whether + * the doc-link layer failed for it (publish then adds the error marker + * row that index_status reports as doc_links.status = "error"). */ + const cbm_doc_link_row_t *doc_link_rows; + int doc_link_row_count; + bool doc_links_failed; } cbm_pipeline_generation_t; /* Serialize and fully populate a sibling staging database, then atomically @@ -915,6 +941,15 @@ void cbm_pipeline_discard_stage(const char *stage_path); * Takes ownership; dump_and_persist_hashes writes them into the staging * store and cbm_pipeline_free releases them. Passing NULL/0 clears. */ void cbm_pipeline_set_lsp_surfaces(cbm_pipeline_t *p, cbm_lsp_surface_row_t *rows, int count); +/* The run's doc_link_unresolved rows and failure flag (doc_links.h), taken + * over by the pipeline (set replaces and frees earlier rows; NULL p frees). + * A full run publishes them from dump_and_persist_hashes; an incremental + * route takes them back for its carry-forward merge. `ran` stays false until + * a doc-link phase hands rows over, so a route that never resolved can tell. */ +void cbm_pipeline_set_doc_link_rows(cbm_pipeline_t *p, cbm_doc_link_row_t *rows, int count, + bool failed); +void cbm_pipeline_take_doc_link_rows(cbm_pipeline_t *p, cbm_doc_link_row_t **rows, int *count, + bool *failed, bool *ran); /* Pipeline accessors for incremental use */ const char *cbm_pipeline_repo_path(const cbm_pipeline_t *p); diff --git a/src/store/store.c b/src/store/store.c index a8a2ab21c..7acb10e87 100644 --- a/src/store/store.c +++ b/src/store/store.c @@ -77,6 +77,7 @@ enum { #include "foundation/platform.h" #include "foundation/compat.h" #include "foundation/log.h" +#include "foundation/mem_core.h" #include "foundation/compat_regex.h" #include "callable_sig.h" /* cbm_qn_callable_base_len: base-match tier */ #include "foundation/mem_core.h" /* cbm_alloc: pattern buffers */ @@ -2548,6 +2549,8 @@ int cbm_store_list_projects(cbm_store_t *s, cbm_project_t **out, int *count) { return CBM_STORE_OK; } +static int doc_links_table_state(cbm_store_t *s); + int cbm_store_delete_project(cbm_store_t *s, const char *name) { if (!s || !s->db || !name) { return CBM_STORE_ERR; @@ -2561,6 +2564,30 @@ int cbm_store_delete_project(cbm_store_t *s, const char *name) { "DELETE FROM index_coverage_meta WHERE project = ?1;", "DELETE FROM projects WHERE name = ?1 || '::missed';", }; + /* doc_link_unresolved exists only in databases published by a build that + * writes it; an older database has nothing to clean. */ + int doc_links_state = doc_links_table_state(s); + if (doc_links_state < 0) { + (void)exec_sql(s, "ROLLBACK;"); + return CBM_STORE_ERR; + } + if (doc_links_state > 0) { + sqlite3_stmt *dl = NULL; + if (sqlite3_prepare_v2(s->db, "DELETE FROM doc_link_unresolved WHERE project = ?1;", + CBM_NOT_FOUND, &dl, NULL) != SQLITE_OK) { + store_set_error_sqlite(s, "delete project doc_links prepare"); + (void)exec_sql(s, "ROLLBACK;"); + return CBM_STORE_ERR; + } + bind_text(dl, SKIP_ONE, name); + int dl_rc = sqlite3_step(dl); + sqlite3_finalize(dl); + if (dl_rc != SQLITE_DONE) { + store_set_error_sqlite(s, "delete project doc_links"); + (void)exec_sql(s, "ROLLBACK;"); + return CBM_STORE_ERR; + } + } for (size_t i = 0; i < sizeof(cleanup_sql) / sizeof(cleanup_sql[0]); i++) { sqlite3_stmt *cleanup = NULL; if (sqlite3_prepare_v2(s->db, cleanup_sql[i], CBM_NOT_FOUND, &cleanup, NULL) != SQLITE_OK) { @@ -4363,6 +4390,509 @@ void cbm_store_free_coverage(cbm_coverage_row_t *rows, int count) { free(rows); } +/* ── Doc-link unresolved references ─────────────────────────────── */ + +/* Created at publish, not in init_schema: a database written by an older + * build simply has no table, and readers treat that as "no doc-link data" + * instead of failing (no index-format change). */ +static const char DOC_LINKS_DDL[] = "CREATE TABLE IF NOT EXISTS doc_link_unresolved (" + " project TEXT NOT NULL," + " rel_path TEXT NOT NULL," + " line INTEGER NOT NULL DEFAULT 0," + " syntax TEXT NOT NULL DEFAULT ''," + " raw TEXT NOT NULL DEFAULT ''," + " reason TEXT NOT NULL" + ");" + "CREATE INDEX IF NOT EXISTS idx_doc_link_unresolved_path " + "ON doc_link_unresolved(project, rel_path);"; + +/* 1 = the table exists, 0 = it does not, -1 = the probe failed. */ +static int doc_links_table_state(cbm_store_t *s) { + sqlite3_stmt *stmt = NULL; + if (sqlite3_prepare_v2(s->db, + "SELECT 1 FROM sqlite_master WHERE type = 'table' " + "AND name = 'doc_link_unresolved';", + CBM_NOT_FOUND, &stmt, NULL) != SQLITE_OK) { + store_set_error_sqlite(s, "doc_links probe prepare"); + return CBM_NOT_FOUND; + } + int rc = sqlite3_step(stmt); + sqlite3_finalize(stmt); + if (rc == SQLITE_ROW) { + return SKIP_ONE; + } + if (rc == SQLITE_DONE) { + return 0; + } + store_set_error_sqlite(s, "doc_links probe"); + return CBM_NOT_FOUND; +} + +int cbm_store_doc_links_replace(cbm_store_t *s, const char *project, const cbm_doc_link_row_t *rows, + int count) { + if (!s || !s->db || !project || count < 0 || (count > 0 && !rows)) { + return CBM_STORE_ERR; + } + if (exec_sql(s, DOC_LINKS_DDL) != CBM_STORE_OK) { + return CBM_STORE_ERR; + } + if (exec_sql(s, "BEGIN;") != CBM_STORE_OK) { + return CBM_STORE_ERR; + } + sqlite3_stmt *del = NULL; + if (sqlite3_prepare_v2(s->db, "DELETE FROM doc_link_unresolved WHERE project = ?1;", + CBM_NOT_FOUND, &del, NULL) != SQLITE_OK) { + store_set_error_sqlite(s, "doc_links delete prepare"); + (void)exec_sql(s, "ROLLBACK;"); + return CBM_STORE_ERR; + } + bind_text(del, SKIP_ONE, project); + int rc = sqlite3_step(del); + sqlite3_finalize(del); + if (rc != SQLITE_DONE) { + store_set_error_sqlite(s, "doc_links delete"); + (void)exec_sql(s, "ROLLBACK;"); + return CBM_STORE_ERR; + } + sqlite3_stmt *ins = NULL; + if (sqlite3_prepare_v2(s->db, + "INSERT INTO doc_link_unresolved " + "(project, rel_path, line, syntax, raw, reason) " + "VALUES (?1, ?2, ?3, ?4, ?5, ?6);", + CBM_NOT_FOUND, &ins, NULL) != SQLITE_OK) { + store_set_error_sqlite(s, "doc_links insert prepare"); + (void)exec_sql(s, "ROLLBACK;"); + return CBM_STORE_ERR; + } + for (int i = 0; i < count; i++) { + if (!rows[i].rel_path || !rows[i].reason) { + continue; + } + bind_text(ins, SKIP_ONE, project); + bind_text(ins, ST_COL_2, rows[i].rel_path); + sqlite3_bind_int(ins, ST_COL_3, rows[i].line); + bind_text(ins, CBM_SZ_4, rows[i].syntax ? rows[i].syntax : ""); + bind_text(ins, CBM_SZ_5, rows[i].raw ? rows[i].raw : ""); + bind_text(ins, CBM_SZ_6, rows[i].reason); + if (sqlite3_step(ins) != SQLITE_DONE) { + store_set_error_sqlite(s, "doc_links insert"); + sqlite3_finalize(ins); + (void)exec_sql(s, "ROLLBACK;"); + return CBM_STORE_ERR; + } + sqlite3_reset(ins); + } + sqlite3_finalize(ins); + return exec_sql(s, "COMMIT;"); +} + +#ifdef CBM_ENABLE_TEST_SEAMS +static atomic_uint_fast64_t doc_links_sample_field_copies = ATOMIC_VAR_INIT(0); +static atomic_uint_fast64_t doc_links_sample_copied_bytes = ATOMIC_VAR_INIT(0); +static atomic_uint_fast64_t doc_links_sample_requested_bytes = ATOMIC_VAR_INIT(0); +static atomic_uint_fast64_t doc_links_sample_max_request_bytes = ATOMIC_VAR_INIT(0); +static atomic_int doc_links_sample_alloc_countdown = ATOMIC_VAR_INIT(CBM_NOT_FOUND); +static atomic_bool doc_links_sample_alloc_failed = ATOMIC_VAR_INIT(false); + +void cbm_store_doc_links_test_sample_stats_reset(void) { + atomic_store_explicit(&doc_links_sample_field_copies, 0, memory_order_relaxed); + atomic_store_explicit(&doc_links_sample_copied_bytes, 0, memory_order_relaxed); + atomic_store_explicit(&doc_links_sample_requested_bytes, 0, memory_order_relaxed); + atomic_store_explicit(&doc_links_sample_max_request_bytes, 0, memory_order_relaxed); +} + +void cbm_store_doc_links_test_sample_stats(cbm_doc_links_sample_test_stats_t *out) { + if (!out) { + return; + } + out->field_copies = atomic_load_explicit(&doc_links_sample_field_copies, memory_order_relaxed); + out->copied_bytes = atomic_load_explicit(&doc_links_sample_copied_bytes, memory_order_relaxed); + out->requested_bytes = + atomic_load_explicit(&doc_links_sample_requested_bytes, memory_order_relaxed); + out->max_request_bytes = + atomic_load_explicit(&doc_links_sample_max_request_bytes, memory_order_relaxed); +} + +void cbm_store_doc_links_test_fail_sample_alloc_after(int successful_copies) { + atomic_store_explicit(&doc_links_sample_alloc_failed, false, memory_order_relaxed); + atomic_store_explicit(&doc_links_sample_alloc_countdown, + successful_copies < 0 ? CBM_NOT_FOUND : successful_copies, + memory_order_relaxed); +} + +bool cbm_store_doc_links_test_sample_alloc_failed(void) { + return atomic_load_explicit(&doc_links_sample_alloc_failed, memory_order_relaxed); +} +#endif + +/* Copy the exact bytes requested by the sample query, including embedded NUL. + * Counters observe this allocation and copy, never the original column length. */ +static char *doc_links_sample_copy(const void *source, size_t length) { + if (length == SIZE_MAX || (length && !source)) { + return NULL; + } + size_t bytes = length + SKIP_ONE; +#ifdef CBM_ENABLE_TEST_SEAMS + atomic_fetch_add_explicit(&doc_links_sample_requested_bytes, bytes, memory_order_relaxed); + uint_fast64_t maximum = + atomic_load_explicit(&doc_links_sample_max_request_bytes, memory_order_relaxed); + while (maximum < bytes && !atomic_compare_exchange_weak_explicit( + &doc_links_sample_max_request_bytes, &maximum, bytes, + memory_order_relaxed, memory_order_relaxed)) {} + if (graph_compare_test_countdown_fires(&doc_links_sample_alloc_countdown)) { + atomic_store_explicit(&doc_links_sample_alloc_failed, true, memory_order_relaxed); + return NULL; + } +#endif + char *copy = cbm_alloc(CBM_MEM_CLASS_STORE, bytes); + if (copy) { + if (length) { + memcpy(copy, source, length); + } + copy[length] = '\0'; +#ifdef CBM_ENABLE_TEST_SEAMS + atomic_fetch_add_explicit(&doc_links_sample_field_copies, 1, memory_order_relaxed); + atomic_fetch_add_explicit(&doc_links_sample_copied_bytes, bytes, memory_order_relaxed); +#endif + } + return copy; +} + +/* Full getters and the original summary API retain their C-string behavior. */ +static char *doc_links_field_copy(const char *source, bool sample) { + if (!sample || !source) { + return cbm_mem_strdup(CBM_MEM_CLASS_STORE, source); + } + return doc_links_sample_copy(source, strlen(source)); +} + +/* Run a row query (columns rel_path, line, syntax, raw, reason) with the + * project bound to ?1 and an optional integer limit bound to ?2. */ +static int doc_links_query(cbm_store_t *s, const char *sql, const char *project, int limit, + cbm_doc_link_row_t **out, int *count) { + sqlite3_stmt *stmt = NULL; + if (sqlite3_prepare_v2(s->db, sql, CBM_NOT_FOUND, &stmt, NULL) != SQLITE_OK) { + store_set_error_sqlite(s, "doc_links get prepare"); + return CBM_STORE_ERR; + } + bind_text(stmt, SKIP_ONE, project); + if (limit >= 0) { + sqlite3_bind_int(stmt, ST_COL_2, limit); + } + int cap = ST_INIT_CAP_16; + int n = 0; + cbm_doc_link_row_t *arr = cbm_alloc(CBM_MEM_CLASS_STORE, (size_t)cap * sizeof(*arr)); + if (!arr) { + sqlite3_finalize(stmt); + return CBM_STORE_ERR; + } + int scan_rc; + while ((scan_rc = sqlite3_step(stmt)) == SQLITE_ROW) { + if (n >= cap) { + cbm_doc_link_row_t *grown = + cbm_realloc(CBM_MEM_CLASS_STORE, arr, (size_t)cap * ST_GROWTH * sizeof(*arr)); + if (!grown) { + sqlite3_finalize(stmt); + cbm_store_free_doc_links(arr, n); + return CBM_STORE_ERR; + } + arr = grown; + cap *= ST_GROWTH; + } + cbm_doc_link_row_t *r = &arr[n]; + r->rel_path = doc_links_field_copy((const char *)sqlite3_column_text(stmt, 0), limit >= 0); + r->line = sqlite3_column_int(stmt, SKIP_ONE); + r->syntax = + doc_links_field_copy((const char *)sqlite3_column_text(stmt, ST_COL_2), limit >= 0); + r->raw = + doc_links_field_copy((const char *)sqlite3_column_text(stmt, ST_COL_3), limit >= 0); + r->reason = + doc_links_field_copy((const char *)sqlite3_column_text(stmt, CBM_SZ_4), limit >= 0); + n++; + if (!r->rel_path || !r->syntax || !r->raw || !r->reason) { + sqlite3_finalize(stmt); + cbm_store_free_doc_links(arr, n); + return CBM_STORE_ERR; + } + } + sqlite3_finalize(stmt); + if (scan_rc != SQLITE_DONE) { + store_set_error_sqlite(s, "doc_links scan"); + cbm_store_free_doc_links(arr, n); + return CBM_STORE_ERR; + } + *out = arr; + *count = n; + return CBM_STORE_OK; +} + +int cbm_store_doc_links_get(cbm_store_t *s, const char *project, cbm_doc_link_row_t **out, + int *count, bool *table_present) { + if (!out || !count) { + return CBM_STORE_ERR; + } + *out = NULL; + *count = 0; + if (table_present) { + *table_present = false; + } + if (!s || !s->db || !project) { + return CBM_STORE_ERR; + } + int state = doc_links_table_state(s); + if (state < 0) { + return CBM_STORE_ERR; + } + if (state == 0) { + return CBM_STORE_OK; + } + if (table_present) { + *table_present = true; + } + return doc_links_query(s, + "SELECT rel_path, line, syntax, raw, reason FROM doc_link_unresolved " + "WHERE project = ?1 ORDER BY rel_path, line, raw, syntax, reason;", + project, CBM_NOT_FOUND, out, count); +} + +int cbm_store_doc_links_summary(cbm_store_t *s, const char *project, + cbm_doc_link_reason_count_t **reasons, int *reason_count, + cbm_doc_link_row_t **samples, int *sample_count, int sample_limit, + bool *table_present) { + if (!reasons || !reason_count || !samples || !sample_count) { + return CBM_STORE_ERR; + } + *reasons = NULL; + *reason_count = 0; + *samples = NULL; + *sample_count = 0; + if (table_present) { + *table_present = false; + } + if (!s || !s->db || !project) { + return CBM_STORE_ERR; + } + int state = doc_links_table_state(s); + if (state < 0) { + return CBM_STORE_ERR; + } + if (state == 0) { + return CBM_STORE_OK; + } + if (table_present) { + *table_present = true; + } + sqlite3_stmt *stmt = NULL; + if (sqlite3_prepare_v2(s->db, + "SELECT reason, COUNT(*) FROM doc_link_unresolved WHERE project = ?1 " + "GROUP BY reason ORDER BY reason;", + CBM_NOT_FOUND, &stmt, NULL) != SQLITE_OK) { + store_set_error_sqlite(s, "doc_links summary prepare"); + return CBM_STORE_ERR; + } + bind_text(stmt, SKIP_ONE, project); + int cap = ST_INIT_CAP_8; + int n = 0; + cbm_doc_link_reason_count_t *arr = cbm_alloc(CBM_MEM_CLASS_STORE, (size_t)cap * sizeof(*arr)); + if (!arr) { + sqlite3_finalize(stmt); + return CBM_STORE_ERR; + } + int scan_rc; + while ((scan_rc = sqlite3_step(stmt)) == SQLITE_ROW) { + if (n >= cap) { + cbm_doc_link_reason_count_t *grown = + cbm_realloc(CBM_MEM_CLASS_STORE, arr, (size_t)cap * ST_GROWTH * sizeof(*arr)); + if (!grown) { + sqlite3_finalize(stmt); + cbm_store_free_doc_link_reasons(arr, n); + return CBM_STORE_ERR; + } + arr = grown; + cap *= ST_GROWTH; + } + arr[n].reason = + cbm_mem_strdup(CBM_MEM_CLASS_STORE, (const char *)sqlite3_column_text(stmt, 0)); + arr[n].count = sqlite3_column_int(stmt, SKIP_ONE); + n++; + if (!arr[n - SKIP_ONE].reason) { + sqlite3_finalize(stmt); + cbm_store_free_doc_link_reasons(arr, n); + return CBM_STORE_ERR; + } + } + sqlite3_finalize(stmt); + if (scan_rc != SQLITE_DONE) { + store_set_error_sqlite(s, "doc_links summary scan"); + cbm_store_free_doc_link_reasons(arr, n); + return CBM_STORE_ERR; + } + if (sample_limit > 0 && + doc_links_query(s, + "SELECT rel_path, line, syntax, raw, reason FROM doc_link_unresolved " + "WHERE project = ?1 ORDER BY reason, rel_path, line, raw LIMIT ?2;", + project, sample_limit, samples, sample_count) != CBM_STORE_OK) { + cbm_store_free_doc_link_reasons(arr, n); + return CBM_STORE_ERR; + } + *reasons = arr; + *reason_count = n; + return CBM_STORE_OK; +} + +/* Both substr and length operate on bytes; ORDER BY retains the original + * columns, so projection cannot change which fifty source rows are sampled. */ +static bool doc_links_preview_field(sqlite3_stmt *stmt, int column, size_t cap, + cbm_doc_link_preview_text_t *out) { + /* SQLite's BLOB substr returns SQL NULL for a zero-byte BLOB. Its + * separately computed integer length distinguishes that from a NULL + * source or a failed nonempty projection. */ + if (sqlite3_column_type(stmt, column + SKIP_ONE) != SQLITE_INTEGER) { + return false; + } + sqlite3_int64 original = sqlite3_column_int64(stmt, column + SKIP_ONE); + int projection_type = sqlite3_column_type(stmt, column); + if (original < 0 || + (projection_type != SQLITE_BLOB && (projection_type != SQLITE_NULL || original != 0))) { + return false; + } + const void *source = sqlite3_column_blob(stmt, column); + int bytes = sqlite3_column_bytes(stmt, column); + if (bytes < 0 || (size_t)bytes > cap || (bytes && !source)) { + return false; + } + uint64_t expected = (uint64_t)original < cap ? (uint64_t)original : cap; + if ((uint64_t)bytes != expected) { + return false; + } + out->text = doc_links_sample_copy(source, (size_t)bytes); + if (!out->text) { + return false; + } + out->length = (size_t)bytes; + out->original_bytes = (uint64_t)original; + return true; +} + +int cbm_store_doc_links_preview(cbm_store_t *s, const char *project, + cbm_doc_link_preview_row_t **out, int *count, bool *table_present) { + if (!out || !count) { + return CBM_STORE_ERR; + } + *out = NULL; + *count = 0; + if (table_present) { + *table_present = false; + } + if (!s || !s->db || !project) { + return CBM_STORE_ERR; + } + int state = doc_links_table_state(s); + if (state < 0) { + return CBM_STORE_ERR; + } + if (!state) { + return CBM_STORE_OK; + } + if (table_present) { + *table_present = true; + } + sqlite3_stmt *stmt = NULL; + static const char sql[] = + "SELECT substr(CAST(d.rel_path AS BLOB),1,?2),length(CAST(d.rel_path AS BLOB))," + "d.line,substr(CAST(d.syntax AS BLOB),1,?3),length(CAST(d.syntax AS BLOB))," + "substr(CAST(d.raw AS BLOB),1,?2),length(CAST(d.raw AS BLOB))," + "substr(CAST(d.reason AS BLOB),1,?3),length(CAST(d.reason AS BLOB)) " + "FROM doc_link_unresolved AS d WHERE d.project=?1 " + "ORDER BY d.reason,d.rel_path,d.line,d.raw LIMIT ?4;"; + if (sqlite3_prepare_v2(s->db, sql, CBM_NOT_FOUND, &stmt, NULL) != SQLITE_OK) { + store_set_error_sqlite(s, "doc_links preview prepare"); + return CBM_STORE_ERR; + } + const int long_cap = CBM_DOC_LINK_PREVIEW_LONG_BYTES + CBM_DOC_LINK_PREVIEW_LOOKAHEAD; + const int short_cap = CBM_DOC_LINK_PREVIEW_SHORT_BYTES + CBM_DOC_LINK_PREVIEW_LOOKAHEAD; + if (bind_text(stmt, SKIP_ONE, project) != SQLITE_OK || + sqlite3_bind_int(stmt, ST_COL_2, long_cap) != SQLITE_OK || + sqlite3_bind_int(stmt, ST_COL_3, short_cap) != SQLITE_OK || + sqlite3_bind_int(stmt, ST_COL_4, CBM_DOC_LINK_PREVIEW_ROWS) != SQLITE_OK) { + store_set_error_sqlite(s, "doc_links preview bind"); + sqlite3_finalize(stmt); + return CBM_STORE_ERR; + } + cbm_doc_link_preview_row_t *rows = + cbm_calloc(CBM_MEM_CLASS_STORE, CBM_DOC_LINK_PREVIEW_ROWS * sizeof(*rows)); + if (!rows) { + sqlite3_finalize(stmt); + return CBM_STORE_ERR; + } + int n = 0; + int rc; + while ((rc = sqlite3_step(stmt)) == SQLITE_ROW) { + if (n == CBM_DOC_LINK_PREVIEW_ROWS) { + store_set_error(s, "doc_links preview row limit"); + sqlite3_finalize(stmt); + cbm_store_free_doc_link_previews(rows, n); + return CBM_STORE_ERR; + } + cbm_doc_link_preview_row_t *row = &rows[n++]; + row->line = sqlite3_column_int(stmt, ST_COL_2); + if (!doc_links_preview_field(stmt, 0, (size_t)long_cap, &row->rel_path) || + !doc_links_preview_field(stmt, ST_COL_3, (size_t)short_cap, &row->syntax) || + !doc_links_preview_field(stmt, ST_COL_5, (size_t)long_cap, &row->raw) || + !doc_links_preview_field(stmt, ST_COL_7, (size_t)short_cap, &row->reason)) { + store_set_error(s, "doc_links preview copy failed"); + sqlite3_finalize(stmt); + cbm_store_free_doc_link_previews(rows, n); + return CBM_STORE_ERR; + } + } + sqlite3_finalize(stmt); + if (rc != SQLITE_DONE) { + store_set_error_sqlite(s, "doc_links preview scan"); + cbm_store_free_doc_link_previews(rows, n); + return CBM_STORE_ERR; + } + *out = rows; + *count = n; + return CBM_STORE_OK; +} + +void cbm_store_free_doc_link_previews(cbm_doc_link_preview_row_t *rows, int count) { + if (!rows) { + return; + } + for (int i = 0; i < count; i++) { + cbm_free(CBM_MEM_CLASS_STORE, (char *)rows[i].rel_path.text); + cbm_free(CBM_MEM_CLASS_STORE, (char *)rows[i].syntax.text); + cbm_free(CBM_MEM_CLASS_STORE, (char *)rows[i].raw.text); + cbm_free(CBM_MEM_CLASS_STORE, (char *)rows[i].reason.text); + } + cbm_free(CBM_MEM_CLASS_STORE, rows); +} + +void cbm_store_free_doc_links(cbm_doc_link_row_t *rows, int count) { + if (!rows) { + return; + } + for (int i = 0; i < count; i++) { + cbm_free(CBM_MEM_CLASS_STORE, (char *)rows[i].rel_path); + cbm_free(CBM_MEM_CLASS_STORE, (char *)rows[i].syntax); + cbm_free(CBM_MEM_CLASS_STORE, (char *)rows[i].raw); + cbm_free(CBM_MEM_CLASS_STORE, (char *)rows[i].reason); + } + cbm_free(CBM_MEM_CLASS_STORE, rows); +} + +void cbm_store_free_doc_link_reasons(cbm_doc_link_reason_count_t *reasons, int count) { + if (!reasons) { + return; + } + for (int i = 0; i < count; i++) { + cbm_free(CBM_MEM_CLASS_STORE, (char *)reasons[i].reason); + } + cbm_free(CBM_MEM_CLASS_STORE, reasons); +} + /* ── FindNodesByFileOverlap ─────────────────────────────────────── */ int cbm_store_find_nodes_by_file_overlap(cbm_store_t *s, const char *project, const char *file_path, diff --git a/src/store/store.h b/src/store/store.h index 916fa3a65..61b555177 100644 --- a/src/store/store.h +++ b/src/store/store.h @@ -717,6 +717,99 @@ void cbm_store_coverage_shadow_project(char *dst, size_t dstsz, const char *proj void cbm_store_free_coverage(cbm_coverage_row_t *rows, int count); +/* ── Doc-link unresolved references ─────────────────────────────── */ + +/* One doc-comment reference that did not become a MENTIONS edge, stored in + * doc_link_unresolved (project, rel_path, line, syntax, raw, reason). The + * table is created at publish when missing, so databases written by older + * builds stay readable without an index-format change. `reason` is one of + * missing, ambiguous, external, test_only_target, not_indexed, graph_gap, + * unparseable, below_bar_tier (it resolved, but its link family does not + * ship); a row with rel_path "" and reason "error" records that the + * doc-link layer itself failed for the generation. Rows returned by the + * getters own their strings; rows and strings are memory-core blocks of + * CBM_MEM_CLASS_STORE, released only by cbm_store_free_doc_links. */ +typedef struct { + const char *rel_path; + int line; + const char *syntax; + const char *raw; + const char *reason; +} cbm_doc_link_row_t; + +/* Replace the project's rows in one transaction, creating the table first + * when it does not exist. */ +int cbm_store_doc_links_replace(cbm_store_t *s, const char *project, const cbm_doc_link_row_t *rows, + int count); + +/* All rows of the project ordered by (rel_path, line, raw). *table_present + * (optional) is false — with zero rows and CBM_STORE_OK — for a database that + * has no doc_link_unresolved table yet. */ +int cbm_store_doc_links_get(cbm_store_t *s, const char *project, cbm_doc_link_row_t **out, + int *count, bool *table_present); + +/* Row count per reason (ordered by reason; reasons owned by the result) and + * up to `sample_limit` rows ordered by (reason, rel_path, line). Same + * table_present contract as above. */ +typedef struct { + const char *reason; + int count; +} cbm_doc_link_reason_count_t; +int cbm_store_doc_links_summary(cbm_store_t *s, const char *project, + cbm_doc_link_reason_count_t **reasons, int *reason_count, + cbm_doc_link_row_t **samples, int *sample_count, int sample_limit, + bool *table_present); + +/* Diagnostic display budgets; the query copies three extra bytes per field + * so a formatter can inspect a complete UTF-8 scalar at the display boundary. */ +enum { + CBM_DOC_LINK_PREVIEW_ROWS = 50, + CBM_DOC_LINK_PREVIEW_LONG_BYTES = 1024, + CBM_DOC_LINK_PREVIEW_SHORT_BYTES = 128, + CBM_DOC_LINK_PREVIEW_LOOKAHEAD = 3, +}; +typedef struct { + const char *text; + size_t length; /* copied prefix bytes, excluding the added NUL */ + uint64_t original_bytes; +} cbm_doc_link_preview_text_t; +typedef struct { + cbm_doc_link_preview_text_t rel_path; + int line; + cbm_doc_link_preview_text_t syntax; + cbm_doc_link_preview_text_t raw; + cbm_doc_link_preview_text_t reason; +} cbm_doc_link_preview_row_t; +/* Up to fifty rows ordered by original (reason, rel_path, line, raw), before + * empty-path marker filtering. Text is an explicit-length byte prefix and may + * contain NUL or invalid UTF-8. Missing-table semantics match the full getter. + * On failure, no rows are returned. All allocations belong to STORE. */ +int cbm_store_doc_links_preview(cbm_store_t *s, const char *project, + cbm_doc_link_preview_row_t **out, int *count, bool *table_present); +void cbm_store_free_doc_link_previews(cbm_doc_link_preview_row_t *rows, int count); + +#ifdef CBM_ENABLE_TEST_SEAMS +/* Sample-field copies only: exclude full getters, row arrays and reason keys. + * Byte counts include the trailing NUL. Requests include failed allocations; + * field_copies/copied_bytes count only successful copies. Reset leaves fault + * controls and their consumed flag unchanged. */ +typedef struct { + uint64_t field_copies; + uint64_t copied_bytes; + uint64_t requested_bytes; + uint64_t max_request_bytes; +} cbm_doc_links_sample_test_stats_t; +void cbm_store_doc_links_test_sample_stats_reset(void); +void cbm_store_doc_links_test_sample_stats(cbm_doc_links_sample_test_stats_t *out); +/* One-shot field-copy allocation failure: 0 is next, 4 is fifth, -1 disables. + * Setting the control clears its consumed flag. */ +void cbm_store_doc_links_test_fail_sample_alloc_after(int successful_copies); +bool cbm_store_doc_links_test_sample_alloc_failed(void); +#endif + +void cbm_store_free_doc_links(cbm_doc_link_row_t *rows, int count); +void cbm_store_free_doc_link_reasons(cbm_doc_link_reason_count_t *reasons, int count); + /* ── Search ─────────────────────────────────────────────────────── */ int cbm_store_search(cbm_store_t *s, const cbm_search_params_t *params, cbm_search_output_t *out); diff --git a/tests/test_doc_mentions.c b/tests/test_doc_mentions.c new file mode 100644 index 000000000..9f4697a96 --- /dev/null +++ b/tests/test_doc_mentions.c @@ -0,0 +1,10614 @@ +/* + * test_doc_mentions.c — doc-comment references -> MENTIONS edges. + * + * Extraction (doclink.c / doclink_cs.c), the C# resolver (doc_links_cs.c), + * MSBuild global usings, publication into doc_link_unresolved, the + * index_status block, delete_project, and incremental == full across edits. + * Every pipeline test indexes a real fixture through cbm_pipeline_run and + * reads the published database (helpers: test_doc_mentions_helpers.h). + */ +#include "../src/foundation/compat.h" +#include "test_framework.h" +#include "test_helpers.h" +#include "test_doc_mentions_helpers.h" + +#include "cbm.h" +#include "helpers.h" /* cbm_kind_in_set_free_cache */ +#include "doclink.h" +#include "lang_specs.h" +#include "foundation/compat_thread.h" +#include "foundation/mem_core.h" +#include "mcp/mcp.h" +#include "mcp/mcp_internal.h" +#include "pipeline/doc_links.h" +#include "pipeline/doc_links_msbuild.h" +#include "pipeline/lsp_surface.h" +#include "pipeline/pipeline.h" +#include "pipeline/pipeline_internal.h" +#include "store/store.h" +#include "sqlite3.h" + +#include +#include +#include +#ifndef _WIN32 +#include /* mkfifo */ +#include /* symlink */ +#endif + +/* ── extraction ──────────────────────────────────────────────────── */ + +TEST(doc_mentions_extract_cs_tokens) { + const char *src = + "namespace N\n" /* 1 */ + "{\n" /* 2 */ + " /// A and\n" /* 3 */ + " /// , not \n" /* 4 */ + " /// or or .\n" /* 5 */ + " /// \n" /* 6 */ + " /// \n" /* 7 */ + " /// bad\n" /* 8 */ + " /// \n" /* 10 */ + " public class C\n" /* 11 */ + " {\n" + " /// Field .\n" + " public const int K = 1;\n" + " }\n" + "}\n"; + CBMFileResult *r = + cbm_extract_file(src, (int)strlen(src), CBM_LANG_CSHARP, "p", "C.cs", 0, NULL, NULL); + ASSERT_NOT_NULL(r); + const CBMDocLink *foo = dm_find_token(r, "Foo"); + ASSERT_NOT_NULL(foo); + ASSERT_EQ(foo->line, 3); + ASSERT_EQ(foo->syntax, CBM_DOCLINK_CS_SEE); + ASSERT_STR_EQ(foo->source_qn, "p.C.C"); + ASSERT_EQ(foo->def_line, 11); + const CBMDocLink *baz = dm_find_token(r, "Bar.Baz(int)"); + ASSERT_NOT_NULL(baz); + ASSERT_EQ(baz->line, 4); + ASSERT_EQ(baz->syntax, CBM_DOCLINK_CS_SEEALSO); + /* local parameter references and keywords are not references */ + ASSERT_NULL(dm_find_token(r, "x")); + ASSERT_NULL(dm_find_token(r, "T")); + ASSERT_NULL(dm_find_token(r, "null")); + const CBMDocLink *href = dm_find_token(r, "https://x.org"); + ASSERT_NOT_NULL(href); + ASSERT_EQ(href->syntax, CBM_DOCLINK_HREF); + ASSERT_EQ(href->line, 6); + /* XML entities are decoded */ + ASSERT_NOT_NULL(dm_find_token(r, "List")); + const CBMDocLink *ex = dm_find_token(r, "Oops"); + ASSERT_NOT_NULL(ex); + ASSERT_EQ(ex->syntax, CBM_DOCLINK_CS_EXCEPTION); + ASSERT_EQ(ex->line, 8); + const CBMDocLink *inh = dm_find_token(r, "Base.M"); + ASSERT_NOT_NULL(inh); + ASSERT_EQ(inh->syntax, CBM_DOCLINK_CS_INHERITDOC); + /* a tag spanning two comment lines */ + const CBMDocLink *wrapped = dm_find_token(r, "Wrapped.Name"); + ASSERT_NOT_NULL(wrapped); + ASSERT_EQ(wrapped->line, 9); + /* the constant's Field and its Variable twin carry the doc once */ + ASSERT_EQ(dm_count_tokens(r, "Foo"), 2); /* the class's and the constant's */ + cbm_free_result(r); + PASS(); +} + +TEST(doc_mentions_cs_scope_blob) { + const char *src = "global using Acme.G; global using static Acme.GS;\n" /* 1 */ + "using static Acme.S;\n" /* 2 */ + "using A = Acme.Util.Helper;\n" /* 3 */ + "namespace Outer.Inner\n" /* 4 */ + "{\n" /* 5 */ + " using Acme.Local;\n" /* 6 */ + " public partial class W : Base, IThing\n" /* 7 */ + " {\n" /* 8 */ + " void IThing.Do(int x) { }\n" /* 9 */ + " public void Go(ref string s, params int[] rest) { }\n" /* 10 */ + " public event System.EventHandler Fired;\n" /* 11 */ + " public int P { get; set; }\n" /* 12 */ + " public record R(int Width);\n" /* 13 */ + " public delegate void D();\n" /* 14 */ + " public enum E { One }\n" /* 15 */ + " public static W operator +(W a, int b) => a;\n" /* 16 */ + " public static implicit operator int(W w) => 0;\n" /* 17 */ + " public int this[int i] => i;\n" /* 18 */ + " public static int Count(string s) => 0;\n" /* 19 */ + " public static int Ext(this string s) => 0;\n" /* 20 */ + " public const int Max = 1;\n" /* 21 */ + " public int field;\n" /* 22 */ + " static W() { }\n" /* 23 */ + " public W(T first) { }\n" /* 24 */ + " public readonly record struct RS(int X);\n" /* 25 */ + " }\n" /* 26 */ + "}\n"; /* 27 */ + CBMFileResult *r = + cbm_extract_file(src, (int)strlen(src), CBM_LANG_CSHARP, "p", "W.cs", 0, NULL, NULL); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + const char *s = r->doc_scope; + ASSERT(strncmp(s, "cs1\n", 4) == 0); + /* a using says what it brings in (n namespace, s static, a alias), and + * `g` when it is a `global using`: of whatever kind */ + ASSERT_NOT_NULL(strstr(s, "U\t0\tng\t-\tAcme.G\n")); + ASSERT_NOT_NULL(strstr(s, "U\t0\ts\t-\tAcme.S\n")); + ASSERT_NOT_NULL(strstr(s, "U\t0\ta\tA\tAcme.Util.Helper\n")); + ASSERT_NOT_NULL(strstr(s, "U\t0\tsg\t-\tAcme.GS\n")); + /* a region carries its own name and the region it stands in */ + ASSERT_NOT_NULL(strstr(s, "R\t1\t0\t4\t27\tOuter.Inner\n")); + ASSERT_NOT_NULL(strstr(s, "U\t1\tn\t-\tAcme.Local\n")); /* a namespace-block using */ + /* a type: `p` for partial, `-` for no outer type; a nested one names its + * outer type by ordinal, a member its type */ + ASSERT_NOT_NULL(strstr(s, "T\t1\t7\t26\tcp\t-\tW\tT,U\tBase|IThing\n")); + ASSERT_NOT_NULL(strstr(s, "M\t9\tc\t1\t0\tDo\t\tint\n")); /* explicit implementation */ + ASSERT_NOT_NULL(strstr(s, "M\t10\tc\t0\t0\tGo\t\tstring|int[]\n")); + ASSERT_NOT_NULL(strstr(s, "M\t11\te\t0\t0\tFired\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t12\tp\t0\t0\tP\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t13\t13\tr\t0\tR\t\t\n")); + ASSERT_NOT_NULL(strstr(s, "M\t13\tc\t0\t1\tR\t\tint\n")); /* its primary constructor */ + ASSERT_NOT_NULL(strstr(s, "M\t13\tp\t0\t1\tWidth\t\t-\n")); /* positional record property */ + ASSERT_NOT_NULL(strstr(s, "T\t1\t14\t14\td\t0\tD\t\t\n")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t15\t15\te\t0\tE\t\t\n")); + ASSERT_NOT_NULL(strstr(s, "M\t15\tvs\t0\t3\tOne\t\t-\n")); /* an enum's member is static */ + /* operators, conversions and indexers are declared, under their token */ + ASSERT_NOT_NULL(strstr(s, "M\t16\to\t0\t0\t+\t\tW|int\n")); + ASSERT_NOT_NULL(strstr(s, "M\t17\to\t0\t0\timplicit\t\tW\n")); + ASSERT_NOT_NULL(strstr(s, "M\t18\tx\t0\t0\tthis\t\tint\n")); + /* `s`: what a `using static` brings in -- static, and no extension method */ + ASSERT_NOT_NULL(strstr(s, "M\t19\tcs\t0\t0\tCount\t\tstring\n")); + ASSERT_NOT_NULL(strstr(s, "M\t20\tc\t0\t0\tExt\t\tstring\n")); + ASSERT_NOT_NULL(strstr(s, "M\t21\tvs\t0\t0\tMax\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t22\tv\t0\t0\tfield\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t23\tcs\t0\t0\tW\t\t\n")); /* the static constructor */ + ASSERT_NOT_NULL(strstr(s, "M\t24\tc\t0\t0\tW\t\tT\n")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t25\t25\tt\t0\tRS\t\t\n")); /* a record struct */ + /* the persisted scope drops every line number, nothing else */ + char *portable = cbm_doclink_portable_scope(s); + ASSERT_NOT_NULL(portable); + ASSERT_NOT_NULL(strstr(portable, "R\t1\t0\t0\t0\tOuter.Inner\n")); + ASSERT_NOT_NULL(strstr(portable, "T\t1\t0\t0\tcp\t-\tW\tT,U\tBase|IThing\n")); + ASSERT_NOT_NULL(strstr(portable, "M\t0\tc\t0\t0\tGo\t\tstring|int[]\n")); + cbm_free(CBM_MEM_CLASS_OTHER, portable); + cbm_free_result(r); + PASS(); +} + +/* Every name in a field/event declaration shares its static/const modifiers. */ +TEST(doc_mentions_cs_declarator_modifiers) { + const char *src = "class C {\n" + "[System.Obsolete] public int a, b;\n" + "[System.Obsolete] public static int c, d;\n" + "[System.Obsolete] public const int e = 1, f = 2;\n" + "[System.Obsolete] public event System.Action G, H;\n" + "[System.Obsolete] public static event System.Action I, J;\n" + "}\n"; + CBMFileResult *r = dm_extract(src, CBM_LANG_CSHARP, "Fields.cs"); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + const char *s = r->doc_scope; + ASSERT_NOT_NULL(strstr(s, "M\t2\tv\t0\t0\ta\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t2\tv\t0\t0\tb\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t3\tvs\t0\t0\tc\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t3\tvs\t0\t0\td\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t4\tvs\t0\t0\te\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t4\tvs\t0\t0\tf\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t5\te\t0\t0\tG\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t5\te\t0\t0\tH\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t6\tes\t0\t0\tI\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "M\t6\tes\t0\t0\tJ\t\t-\n")); + cbm_free_result(r); + PASS(); +} + +/* The scope blob of one C# source; the result owns it. */ +static CBMFileResult *dm_scope(const char *src) { + return dm_extract(src, CBM_LANG_CSHARP, "S.cs"); +} + +/* Controlled parser/scanner observation inputs. No fixture is executed as C#. + * Each name is exercised in both block and file-scoped declarations. */ +static const struct { + const char *id; + const char *name; + bool valid; + bool recovery; +} dm_namespace_cases[] = { + {"leading_dot", ".Bad", false, false}, + {"interior_empty", "Bad..Inner", false, false}, + {"trailing_dot", "Bad.", false, false}, + {"dot_only", ".", false, false}, + {"spaced_empty", "Bad. .Inner", false, false}, + {"verbatim_empty", "@Bad..@Inner", false, false}, + {"valid_dotted", "Good.Inner", true, false}, + {"valid_verbatim", "@Good.@class", true, false}, + {"valid_unicode", "Gr\xC3\xBCne.Inner", true, false}, + {"valid_recovery", "Good.Recovered", true, true}, + {"valid_prefix_comment", "/* note */ Good.Inner", true, false}, + {"valid_segment_comments", "Good /* note */ . /* note */ Inner", true, false}, + {"comment_empty", "Bad. /* note */ .Inner", false, false}, + {"trailing_recovery", "Bad.", false, true}, + {"verbatim_recovery", "@Good.@class", true, true}, +}; + +static bool dm_namespace_source(char *out, size_t cap, size_t index, bool file_scoped) { + const char *recovery = dm_namespace_cases[index].recovery + ? "public class ParseEdge { unsafe void M(void* p) { " + "_r = ref *(int*)p; } }\n" + : ""; + int n = snprintf(out, cap, + "namespace %s%s" + "/// \n" + "public class FromBad { }\n" + "public class Hidden { }\n" + "%s%s", + dm_namespace_cases[index].name, file_scoped ? ";\n" : " {\n", recovery, + file_scoped ? "" : "}\n"); + return n >= 0 && (size_t)n < cap; +} + +/* Real parser inputs, including both AST and lexical recovery paths. Valid + * controls prevent treating every file with a parse error as unplaceable. */ +TEST(doc_mentions_cs_namespace_scopes) { + static const char *normalized[] = {NULL, + NULL, + NULL, + NULL, + NULL, + NULL, + "Good.Inner", + "Good.class", + "Gr\xC3\xBCne.Inner", + "Good.Recovered", + "Good.Inner", + "Good.Inner", + NULL, + NULL, + "Good.class"}; + bool correct = true; + for (size_t i = 0; i < sizeof(dm_namespace_cases) / sizeof(dm_namespace_cases[0]); i++) { + for (int file_scoped = 0; file_scoped < 2; file_scoped++) { + char source[2048]; + bool made = dm_namespace_source(source, sizeof(source), i, file_scoped != 0); + CBMFileResult *result = made ? dm_extract(source, CBM_LANG_CSHARP, "Bad.cs") : NULL; + const char *scope = result ? result->doc_scope : NULL; + int tokens = result ? dm_count_tokens(result, "global::Good.Target") : -1; + bool valid = dm_namespace_cases[i].valid; + bool shape = false; + if (scope && valid) { + char name[128]; + snprintf(name, sizeof(name), "\t%s\n", normalized[i]); + shape = strstr(scope, "\nR\t1\t0\t") && strstr(scope, name) && + strstr(scope, "\nT\t1\t3\t3\tc\t-\tFromBad\t\t\n") && + strstr(scope, "\nT\t1\t4\t4\tc\t-\tHidden\t\t\n") && + !strstr(scope, "\nX\t") && !strstr(scope, "\nQ\t"); + } else if (scope) { + shape = strstr(scope, "\nX\t") && strstr(scope, "\nQ\tFromBad\n") && + strstr(scope, "\nQ\tHidden\n") && !strstr(scope, "\nR\t") && + !strstr(scope, "\nT\t"); + } + bool one = made && scope && tokens == 1 && shape; + fprintf(stderr, + "doc namespace scope case=%s file_scoped=%d valid=%d tokens=%d shape=%d " + "correct=%d\n", + dm_namespace_cases[i].id, file_scoped, valid, tokens, shape, one); + correct = one && correct; + if (result) { + cbm_free_result(result); + } + } + } + ASSERT(correct); + PASS(); +} + +/* Shared real source fixture: a malformed namespace must not erase the + * independent Target link or permit Hidden to bind around an unknown declaration. */ +static const char *dm_namespace_healthy_source = "namespace Good {\n" + "public class Target { }\n" + "public class Hidden { }\n" + "/// \n" + "public class Uses { }\n" + "/// \n" + "public class TriesHidden { }\n" + "}\n"; + +/* Check the persisted layer and the public status. No assertion here may skip + * the caller's environment restoration or fixture cleanup. */ +static bool dm_namespace_published(const char *db, const char *project, bool valid, + const char *label) { + char props[512], inside_reason[64], conflict_reason[64]; + int healthy = 0, inside = 0, conflict = 0; + dm_edge(db, "Healthy.Uses", "Healthy.Target", props, sizeof(props), &healthy); + dm_edge(db, "Bad.FromBad", "Healthy.Target", props, sizeof(props), &inside); + dm_edge(db, "Healthy.TriesHidden", "Healthy.Hidden", props, sizeof(props), &conflict); + dm_row(db, "Bad.cs", "global::Good.Target", inside_reason, sizeof(inside_reason), NULL, 0); + dm_row(db, "Healthy.cs", "Hidden", conflict_reason, sizeof(conflict_reason), NULL, 0); + int errors = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'"); + char *status = dm_index_status(project, false); + const char *block = status ? strstr(status, "doc_links:\n") : NULL; + bool status_ok = block && strstr(block, "\n status: ok"); + int from_bad = dm_mentions_from(db, "Bad.FromBad"); + bool inside_ok = valid + ? inside == 1 && from_bad == 1 && !inside_reason[0] + : inside == 0 && from_bad == 0 && strcmp(inside_reason, "graph_gap") == 0; + bool conflict_ok = valid ? conflict == 1 && !conflict_reason[0] + : conflict == 0 && strcmp(conflict_reason, "graph_gap") == 0; + bool correct = errors == 0 && healthy == 1 && inside_ok && conflict_ok && status_ok; + fprintf(stderr, + "doc namespace published case=%s valid=%d errors=%d healthy=%d inside=%d " + "inside_reason=%s conflict=%d conflict_reason=%s status_ok=%d correct=%d\n", + label, valid, errors, healthy, inside, inside_reason, conflict, conflict_reason, + status_ok, correct); + free(status); + return correct; +} + +TEST(doc_mentions_cs_namespace_publication) { + char tmp[256] = "/tmp/cbm_dm_ns_XXXXXX"; + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400], cache[400], db[1024], full_db[512]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + snprintf(cache, sizeof(cache), "%s/cache", tmp); + snprintf(full_db, sizeof(full_db), "%s/full.db", tmp); + cbm_mkdir_p(cache, 0700); + cbm_mkdir_p(repo, 0700); /* project identity canonicalizes the existing path */ + char *project = cbm_project_name_from_path(repo); + const char *saved_cache = getenv("CBM_CACHE_DIR"); + char *saved_copy = saved_cache ? strdup(saved_cache) : NULL; + bool correct = project && (!saved_cache || saved_copy); + if (!correct) { + free(saved_copy); + free(project); + th_rmtree(tmp); + ASSERT(correct); + } + snprintf(db, sizeof(db), "%s/%s.db", cache, project); + cbm_setenv("CBM_CACHE_DIR", cache, 1); + for (size_t i = 0; i < sizeof(dm_namespace_cases) / sizeof(dm_namespace_cases[0]); i++) { + for (int file_scoped = 0; file_scoped < 2; file_scoped++) { + char source[2048], label[128]; + bool made = dm_namespace_source(source, sizeof(source), i, file_scoped != 0); + if (!made) { + correct = false; + continue; + } + snprintf(label, sizeof(label), "%s/%s", dm_namespace_cases[i].id, + file_scoped ? "file" : "block"); + dm_unlink_db(db); + dm_unlink_db(full_db); + th_write_file(TH_PATH(repo, "Healthy.cs"), dm_namespace_healthy_source); + th_write_file(TH_PATH(repo, "Bad.cs"), source); + char *indexed_project = NULL; + bool indexed = dm_index(repo, db, &indexed_project) == 0; + bool identity = indexed_project && strcmp(project, indexed_project) == 0; + free(indexed_project); + bool published = + dm_namespace_published(db, project, dm_namespace_cases[i].valid, label); + char *before = dm_doclink_state(db); + cbm_pipeline_incremental_test_reset_faults(); + bool repeated = dm_index(repo, db, NULL) == 0; + cbm_incremental_route_t route = cbm_pipeline_incremental_test_last_route(); + char *after = dm_doclink_state(db); + bool stable = before && after && strcmp(before, after) == 0; + bool noop = route == CBM_INCREMENTAL_ROUTE_NOOP; + fprintf(stderr, + "doc namespace unchanged case=%s indexed=%d identity=%d repeated=%d stable=%d " + "noop=%d " + "route=%d\n", + label, indexed, identity, repeated, stable, noop, (int)route); + correct = indexed && identity && published && repeated && stable && noop && correct; + free(before); + free(after); + + /* These three cases cover clipped names, accepted trailing dots, + * and scope blobs that formerly failed the entire layer. Exercise + * both reuse of stored scopes and edits into/out of the bad region. */ + if (i == 0 || i == 2 || i == 4) { + char edited[2048]; + snprintf(edited, sizeof(edited), "%s\n// body-only edit\n", + dm_namespace_healthy_source); + th_write_file(TH_PATH(repo, "Healthy.cs"), edited); + int step = dm_step(repo, db, full_db, label, CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR); + bool state = dm_namespace_published(db, project, false, label); + correct = step == 0 && state && correct; + + bool repaired = dm_namespace_source(edited, sizeof(edited), 6, file_scoped != 0); + if (repaired) { + th_write_file(TH_PATH(repo, "Bad.cs"), edited); + step = dm_step(repo, db, full_db, "namespace repaired", + CBM_INCREMENTAL_ROUTE_FORCED_FULL); + state = dm_namespace_published(db, project, true, "namespace repaired"); + correct = step == 0 && state && correct; + th_write_file(TH_PATH(repo, "Bad.cs"), source); + step = dm_step(repo, db, full_db, "namespace malformed again", + CBM_INCREMENTAL_ROUTE_FORCED_FULL); + state = dm_namespace_published(db, project, false, label); + correct = step == 0 && state && correct; + } else { + correct = false; + } + } + } + } + if (saved_copy) { + cbm_setenv("CBM_CACHE_DIR", saved_copy, 1); + } else { + cbm_unsetenv("CBM_CACHE_DIR"); + } + free(saved_copy); + free(project); + dm_unlink_db(db); + dm_unlink_db(full_db); + th_rmtree(tmp); + ASSERT(correct); + PASS(); +} + +/* Observe a controlled fixture's recovery path without borrowing CBM's parser + * or retaining Tree-sitter objects across extraction. */ +static bool dm_namespace_parser_state(const char *source, size_t length, bool *has_namespace, + bool *has_error) { + TSParser *parser = ts_parser_new(); + TSTree *tree = NULL; + TSTreeCursor cursor = {0}; + bool cursor_live = false, correct = false; + *has_namespace = false; + *has_error = false; + if (!parser || !ts_parser_set_language(parser, cbm_ts_language(CBM_LANG_CSHARP))) { + goto cleanup; + } + tree = ts_parser_parse_string(parser, NULL, source, (uint32_t)length); + if (!tree) { + goto cleanup; + } + TSNode root = ts_tree_root_node(tree); + *has_error = ts_node_has_error(root); + cursor = ts_tree_cursor_new(root); + cursor_live = true; + size_t visited = 0; + for (;;) { + if (++visited > 4096) { + goto cleanup; + } + const char *kind = ts_node_type(ts_tree_cursor_current_node(&cursor)); + if (strcmp(kind, "namespace_declaration") == 0 || + strcmp(kind, "file_scoped_namespace_declaration") == 0) { + *has_namespace = true; + correct = true; + goto cleanup; + } + if (ts_tree_cursor_goto_first_child(&cursor)) { + continue; + } + while (!ts_tree_cursor_goto_next_sibling(&cursor)) { + if (!ts_tree_cursor_goto_parent(&cursor)) { + correct = true; + goto cleanup; + } + } + } +cleanup: + if (cursor_live) { + ts_tree_cursor_delete(&cursor); + } + if (tree) { + ts_tree_delete(tree); + } + if (parser) { + ts_parser_delete(parser); + } + return correct; +} + +/* Namespace token boundaries must preserve verbatim identifiers without + * accepting bare type keywords as recovered namespace names. */ +TEST(doc_mentions_cs_namespace_boundaries) { + static const struct { + const char *id, *header, *footer, *name; + } cases[] = { + {"bare_keyword", "namespace class{\n", + "public class ParseEdge { unsafe void M(void* p) { _r = ref *(int*)p; } }\n}\n", NULL}, + {"adjacent_file", "namespace@Boundary;\n", "", "Boundary"}, + {"adjacent_block", "namespace@Boundary {\n", "}\n", "Boundary"}, + {"verbatim_keyword_file", "namespace @class;\n", "", "class"}, + {"verbatim_keyword_block", "namespace @class {\n", "}\n", "class"}, + }; + bool correct = true; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + char source[2048], normalized[128]; + int n = snprintf(source, sizeof(source), + "%s/// \n" + "public class FromBad { }\npublic class Hidden { }\n%s", + cases[i].header, cases[i].footer); + bool has_namespace = false, has_error = false; + bool recovery = cases[i].name || + (n > 0 && (size_t)n < sizeof(source) && + dm_namespace_parser_state(source, (size_t)n, &has_namespace, &has_error) && + !has_namespace && has_error); + CBMFileResult *result = n > 0 && (size_t)n < sizeof(source) + ? dm_extract(source, CBM_LANG_CSHARP, "Bad.cs") + : NULL; + const char *scope = result ? result->doc_scope : NULL; + int tokens = result ? dm_count_tokens(result, "global::Good.Target") : -1; + bool placed = scope && strstr(scope, "\nR\t"); + bool quarantined = scope && strstr(scope, "\nX\t") && strstr(scope, "\nQ\tFromBad\n") && + strstr(scope, "\nQ\tHidden\n") && !strstr(scope, "\nT\t"); + bool shape = false; + if (scope && cases[i].name) { + snprintf(normalized, sizeof(normalized), "\t%s\n", cases[i].name); + shape = placed && strstr(scope, normalized) && + strstr(scope, "\nT\t1\t3\t3\tc\t-\tFromBad\t\t\n") && + strstr(scope, "\nT\t1\t4\t4\tc\t-\tHidden\t\t\n") && !strstr(scope, "\nX\t") && + !strstr(scope, "\nQ\t"); + } else if (scope) { + shape = !placed && quarantined; + } + bool one = recovery && tokens == 1 && shape; + fprintf(stderr, + "doc namespace boundary case=%s tokens=%d placed=%d quarantined=%d " + "recovery=%d namespace_node=%d parser_error=%d correct=%d\n", + cases[i].id, tokens, placed, quarantined, recovery, has_namespace, has_error, one); + correct = one && correct; + if (result) { + cbm_free_result(result); + } + } + ASSERT(correct); + PASS(); +} + +static const struct { + const char *id, *text; +} dm_control_cases[] = { + {"using_name", "using Goo~d.Inner;\n"}, + {"using_dot_before", "using Good~.Inner;\n"}, + {"using_dot_after", "using Good.~Inner;\n"}, + {"using_alias", "using Al~ias = Good.Target;\n"}, + {"using_target", "using Alias = Good.Tar~get;\n"}, + {"using_comment", "using Alias = Good./*~*/Target;\n"}, + {"base_name", "public class Before : Goo~d.Base { }\n"}, + {"base_generic", "public class Before : Good.Ba~se { }\n"}, + {"base_comment", "public class Before : Good./*~*/Base { }\n"}, + {"type_parameter", "public class Before { }\n"}, + {"method_parameter", "public class Before { public void M() { } }\n"}, + {"signature", "public class Before { public void M(Good.Tar~get x) { } }\n"}, +}; + +/* The returned length includes the replaced marker even when byte is NUL. */ +static int dm_control_source(char *source, size_t capacity, size_t which, unsigned byte) { + int n = snprintf(source, capacity, + "%s/// \n" + "public class Tail { }\nnamespace . { public class Hidden { } }\n", + dm_control_cases[which].text); + if (n <= 0 || (size_t)n >= capacity) { + return -1; + } + char *mark = strchr(source, '~'); + if (!mark) { + return -1; + } + *mark = (char)byte; + return n; +} + +static bool dm_control_write(const char *path, const char *source, size_t length) { + FILE *file = cbm_fopen(path, "wb"); + if (!file) { + return false; + } + bool complete = fwrite(source, 1, length, file) == length; + return fclose(file) == 0 && complete; +} + +/* Establish the routing contract using complete byte spans, including NUL. + * Own imports are local; changes to a declared base can affect other files. */ +static bool dm_control_delta(const char *before, int before_length, const char *after, + int after_length, int expected, const char *label) { + CBMFileResult *a = + cbm_extract_file(before, before_length, CBM_LANG_CSHARP, "p", "Bad.cs", 0, NULL, NULL); + CBMFileResult *b = + cbm_extract_file(after, after_length, CBM_LANG_CSHARP, "p", "Bad.cs", 0, NULL, NULL); + char *pa = a && a->doc_scope ? cbm_doclink_portable_scope(a->doc_scope) : NULL; + char *pb = b && b->doc_scope ? cbm_doclink_portable_scope(b->doc_scope) : NULL; + dm_names_t names = {{0}}; + int delta = + pa && pb ? cbm_doclinks_scope_delta(pa, pb, dm_name_put, &names) : DM_DELTA_SCAN_FAILED; + bool changed = pa && pb && strcmp(pa, pb) != 0; + bool correct = changed && delta == expected; + fprintf(stderr, "doc control delta case=%s changed=%d delta=%d expected=%d correct=%d\n", label, + changed, delta, expected, correct); + cbm_free(CBM_MEM_CLASS_OTHER, pa); + cbm_free(CBM_MEM_CLASS_OTHER, pb); + if (a) { + cbm_free_result(a); + } + if (b) { + cbm_free_result(b); + } + return correct; +} + +TEST(doc_mentions_cs_control_scopes) { + static const unsigned bytes[] = {0, 1, 127, 32}; + bool correct = true; + for (size_t i = 0; i < sizeof(dm_control_cases) / sizeof(dm_control_cases[0]); i++) { + for (size_t j = 0; j < sizeof(bytes) / sizeof(bytes[0]); j++) { + char source[2048]; + int length = dm_control_source(source, sizeof(source), i, bytes[j]); + cbm_doclink_cs_test_cost_reset(); + CBMFileResult *result = length > 0 ? cbm_extract_file(source, length, CBM_LANG_CSHARP, + "p", "Control.cs", 0, NULL, NULL) + : NULL; + uint64_t built = cbm_doclink_cs_test_scope_bytes(); + const char *scope = result ? result->doc_scope : NULL; + size_t visible = scope ? strlen(scope) : 0; + bool clean = scope != NULL; + for (size_t k = 0; k < visible; k++) { + unsigned char c = (unsigned char)scope[k]; + clean = clean && ((c >= 32 && c != 127) || c == '\t' || c == '\n'); + } + int tokens = result ? dm_count_tokens(result, "global::Good.Target") : -1; + bool tail = scope && strstr(scope, "\tTail\t"); + bool quarantine = scope && strstr(scope, "\nQ\tHidden\n") && strstr(scope, "\nX\t"); + bool marker = true; + if (bytes[j] != 32 && (i == 1 || i == 2)) { + marker = scope && strstr(scope, "\nU\t0\tn\t-\t?\n"); + } else if (bytes[j] != 32 && i == 4) { + marker = scope && strstr(scope, "\nU\t0\ta\tAlias\t?\n"); + } else if (bytes[j] != 32 && (i == 6 || i == 7 || i == 8)) { + marker = scope && strstr(scope, "\tBefore\t\t?\n"); + } + bool one = + scope && built == visible && clean && tokens == 1 && tail && quarantine && marker; + fprintf(stderr, + "doc control scope case=%s byte=%u built=%llu visible=%zu clean=%d " + "tokens=%d tail=%d quarantine=%d marker=%d correct=%d\n", + dm_control_cases[i].id, bytes[j], (unsigned long long)built, visible, clean, + tokens, tail, quarantine, marker, one); + correct = one && correct; + if (result) { + cbm_free_result(result); + } + } + } + ASSERT(correct); + PASS(); +} + +static bool dm_control_published(const char *db, const char *project, const char *label) { + char props[512], tail_reason[64], conflict_reason[64]; + int healthy = 0, tail = 0, conflict = 0; + dm_edge(db, "Healthy.Uses", "Healthy.Target", props, sizeof(props), &healthy); + dm_edge(db, "Bad.Tail", "Healthy.Target", props, sizeof(props), &tail); + dm_edge(db, "Healthy.TriesHidden", "Healthy.Hidden", props, sizeof(props), &conflict); + dm_row(db, "Bad.cs", "global::Good.Target", tail_reason, sizeof(tail_reason), NULL, 0); + dm_row(db, "Healthy.cs", "Hidden", conflict_reason, sizeof(conflict_reason), NULL, 0); + int errors = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'"); + char *status = dm_index_status(project, false); + const char *block = status ? strstr(status, "doc_links:\n") : NULL; + bool status_ok = block && strstr(block, "\n status: ok"); + bool correct = healthy == 1 && tail == 1 && !tail_reason[0] && conflict == 0 && + strcmp(conflict_reason, "graph_gap") == 0 && errors == 0 && status_ok; + fprintf(stderr, + "doc control published case=%s healthy=%d tail=%d tail_reason=%s conflict=%d " + "conflict_reason=%s errors=%d status_ok=%d correct=%d\n", + label, healthy, tail, tail_reason, conflict, conflict_reason, errors, status_ok, + correct); + free(status); + return correct; +} + +TEST(doc_mentions_cs_control_publication) { + static const size_t cases[] = {1, 2, 4, 6, 7, 8}; + static const unsigned bytes[] = {0, 1, 127, 32}; + char tmp[256] = "/tmp/cbm_dm_control_XXXXXX"; + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400], cache[400], db[1024], full_db[512]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + snprintf(cache, sizeof(cache), "%s/cache", tmp); + snprintf(full_db, sizeof(full_db), "%s/full.db", tmp); + cbm_mkdir_p(cache, 0700); + cbm_mkdir_p(repo, 0700); + char *project = cbm_project_name_from_path(repo); + const char *saved_cache = getenv("CBM_CACHE_DIR"); + char *saved_copy = saved_cache ? strdup(saved_cache) : NULL; + bool correct = project && (!saved_cache || saved_copy); + if (!correct) { + free(saved_copy); + free(project); + th_rmtree(tmp); + ASSERT(correct); + } + snprintf(db, sizeof(db), "%s/%s.db", cache, project); + cbm_setenv("CBM_CACHE_DIR", cache, 1); + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + for (size_t j = 0; j < sizeof(bytes) / sizeof(bytes[0]); j++) { + char source[2048], label[128]; + int length = dm_control_source(source, sizeof(source), cases[i], bytes[j]); + snprintf(label, sizeof(label), "%s/%u", dm_control_cases[cases[i]].id, bytes[j]); + dm_unlink_db(db); + dm_unlink_db(full_db); + th_write_file(TH_PATH(repo, "Healthy.cs"), dm_namespace_healthy_source); + bool written = + length > 0 && dm_control_write(TH_PATH(repo, "Bad.cs"), source, (size_t)length); + char *indexed_project = NULL; + bool indexed = written && dm_index(repo, db, &indexed_project) == 0; + bool identity = indexed_project && strcmp(project, indexed_project) == 0; + free(indexed_project); + bool published = dm_control_published(db, project, label); + char *before = dm_doclink_state(db); + cbm_pipeline_incremental_test_reset_faults(); + bool repeated = dm_index(repo, db, NULL) == 0; + cbm_incremental_route_t route = cbm_pipeline_incremental_test_last_route(); + char *after = dm_doclink_state(db); + bool stable = before && after && strcmp(before, after) == 0; + bool noop = route == CBM_INCREMENTAL_ROUTE_NOOP; + fprintf(stderr, + "doc control unchanged case=%s written=%d indexed=%d identity=%d repeated=%d " + "stable=%d noop=%d route=%d\n", + label, written, indexed, identity, repeated, stable, noop, (int)route); + correct = written && indexed && identity && published && repeated && stable && noop && + correct; + free(before); + free(after); + + /* Cover both an import and a base field across stored-scope reuse, + * a repaired field and reintroduction of the embedded NUL. */ + if (bytes[j] == 0 && (cases[i] == 1 || cases[i] == 7)) { + char repaired[2048], edited[2048], step_label[160]; + cbm_incremental_route_t changed_route = cases[i] == 1 + ? CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR + : CBM_INCREMENTAL_ROUTE_FORCED_FULL; + int changed_delta = + cases[i] == 1 ? CBM_DOCLINK_DELTA_LOCAL : CBM_DOCLINK_DELTA_GLOBAL; + snprintf(edited, sizeof(edited), "%s\n// body-only edit\n", + dm_namespace_healthy_source); + th_write_file(TH_PATH(repo, "Healthy.cs"), edited); + snprintf(step_label, sizeof(step_label), "%s/reuse", label); + int step = + dm_step(repo, db, full_db, step_label, CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR); + bool state = dm_control_published(db, project, label); + correct = step == 0 && state && correct; + + int repaired_length = dm_control_source(repaired, sizeof(repaired), cases[i], 32); + bool repaired_written = + repaired_length > 0 && + dm_control_write(TH_PATH(repo, "Bad.cs"), repaired, (size_t)repaired_length); + if (repaired_written) { + snprintf(step_label, sizeof(step_label), "%s/repaired", label); + bool delta = dm_control_delta(source, length, repaired, repaired_length, + changed_delta, step_label); + step = dm_step(repo, db, full_db, step_label, changed_route); + state = dm_control_published(db, project, step_label); + correct = delta && step == 0 && state && correct; + bool restored = + dm_control_write(TH_PATH(repo, "Bad.cs"), source, (size_t)length); + if (restored) { + snprintf(step_label, sizeof(step_label), "%s/reintroduced", label); + delta = dm_control_delta(repaired, repaired_length, source, length, + changed_delta, step_label); + step = dm_step(repo, db, full_db, step_label, changed_route); + state = dm_control_published(db, project, label); + correct = delta && step == 0 && state && correct; + } else { + correct = false; + } + } else { + correct = false; + } + } + } + } + if (saved_copy) { + cbm_setenv("CBM_CACHE_DIR", saved_copy, 1); + } else { + cbm_unsetenv("CBM_CACHE_DIR"); + } + free(saved_copy); + free(project); + dm_unlink_db(db); + dm_unlink_db(full_db); + th_rmtree(tmp); + ASSERT(correct); + PASS(); +} + +/* S10: extract real source bytes, then run the surface writer of the run + * with the file's complete definitions, without its scope, and with only + * its scope. Every run must succeed, twice alike, and the stored `dl` must + * be the portable scope. `expect`: 1 the inserted bytes stay in the scope + * (well-formed UTF-8), 0 they do not (malformed), -1 not asked (an entity). */ +static bool dm_scope_utf8_ok(const char *scope); + +static bool dm_utf8_probe_surface(CBMFileResult *bad, CBMFileResult *healthy, CBMLanguage language, + const char *path, const char *label, const char *bytes, + int expect) { + CBMFileResult *cache[] = {bad, healthy}; + cbm_file_info_t files[] = {{.rel_path = (char *)path, .language = language}, + {.rel_path = "Healthy.cs", .language = CBM_LANG_CSHARP}}; + CBMArena arena; + cbm_arena_init(&arena); + char *modules[2] = {NULL, NULL}; + int starts[3] = {0}, def_count = 0; + CBMLSPDef *defs = + cbm_pxc_collect_all_defs(NULL, &arena, cache, files, 2, "p", modules, &def_count, starts); + cbm_lsp_surface_row_t *rows = NULL, *again = NULL; + int count = 0, again_count = 0; + int full_rc = + cbm_lsp_surface_build_rows(NULL, "p", cache, files, 2, defs, starts, &rows, &count); + int full_count = count; + int again_rc = + cbm_lsp_surface_build_rows(NULL, "p", cache, files, 2, defs, starts, &again, &again_count); + bool repeat = full_rc == again_rc && count == again_count; + bool dl_equal = false; + char *portable = bad->doc_scope ? cbm_doclink_portable_scope(bad->doc_scope) : NULL; + if (full_rc == 0 && count == 2 && again_rc == 0 && again_count == 2) { + for (int i = 0; i < count; i++) { + repeat = repeat && strcmp(rows[i].defs_json, again[i].defs_json) == 0 && + strcmp(rows[i].surface_sha, again[i].surface_sha) == 0; + } + yyjson_doc *doc = yyjson_read(rows[0].defs_json, strlen(rows[0].defs_json), 0); + yyjson_val *dl = doc ? yyjson_obj_get(yyjson_doc_get_root(doc), "dl") : NULL; + dl_equal = portable && yyjson_is_str(dl) && strcmp(portable, yyjson_get_str(dl)) == 0; + yyjson_doc_free(doc); + } + cbm_store_free_lsp_surfaces(rows, count); + cbm_store_free_lsp_surfaces(again, again_count); + rows = NULL; + count = 0; + const char *saved_scope = bad->doc_scope; + bad->doc_scope = NULL; + int nodl_rc = + cbm_lsp_surface_build_rows(NULL, "p", cache, files, 2, defs, starts, &rows, &count); + int nodl_count = count; + bad->doc_scope = saved_scope; + cbm_store_free_lsp_surfaces(rows, count); + rows = NULL; + count = 0; + CBMFileResult scope_only = {.doc_scope = saved_scope}; + cache[0] = &scope_only; + int scope_rc = + cbm_lsp_surface_build_rows(NULL, "p", cache, files, 2, NULL, NULL, &rows, &count); + bool retained = saved_scope && strstr(saved_scope, bytes) != NULL; + bool written = full_rc == 0 && full_count == 2 && again_rc == 0 && dl_equal && nodl_rc == 0 && + nodl_count == 2 && scope_rc == 0 && count == 2; + bool kept = expect < 0 || retained == (expect == 1); + bool utf8 = saved_scope && dm_scope_utf8_ok(saved_scope); + bool correct = saved_scope != NULL && def_count > 0 && repeat && written && kept && utf8; + if (!correct) { + printf("doc_utf8_probe case=%s scope=%d retained=%d utf8=%d defs=%d full_rc=%d " + "full_count=%d nodl_rc=%d nodl_count=%d scope_rc=%d scope_count=%d repeat=%d " + "dl_equal=%d\n", + label, saved_scope != NULL, retained, utf8, def_count, full_rc, full_count, nodl_rc, + nodl_count, scope_rc, count, repeat, dl_equal); + } + cbm_store_free_lsp_surfaces(rows, count); + cbm_free(CBM_MEM_CLASS_OTHER, portable); + free(defs); + free(modules[0]); + free(modules[1]); + cbm_arena_destroy(&arena); + return correct; +} + +/* S10: a malformed byte sequence in any place a scope keeps text -- a name + * or a text of a C# source, an element, a value or an attribute of a project + * file, a numeric reference to a surrogate -- leaves a well-formed scope, and + * the surface writer of the run succeeds. Well-formed sequences stay as they + * are; malformed ones do not reach the scope. */ +TEST(doc_mentions_cs_utf8_surfaces) { + static const struct { + const char *label, *bytes; + } sequences[] = { + {"ascii", "A"}, + {"valid2", "\xC3\xA9"}, + {"valid3", "\xE6\xBC\xA2"}, + {"valid4", "\xF0\x90\x90\x80"}, + {"continuation", "\x80"}, + {"overlong", "\xC0\xAF"}, + {"surrogate", "\xED\xA0\x80"}, + {"above_limit", "\xF4\x90\x80\x80"}, + {"truncated", "\xE2\x82"}, + }; + static const struct { + const char *label, *before, *after; + CBMLanguage language; + } positions[] = { + {"using_target", "using Acme.", "Name;\nclass Local {}\n", CBM_LANG_CSHARP}, + {"alias_target", "using Alias = Acme.", "Name;\nclass Local {}\n", CBM_LANG_CSHARP}, + {"namespace_name", "namespace N", "Name { class Local {} }\n", CBM_LANG_CSHARP}, + {"type_name", "class N", "Name {}\n", CBM_LANG_CSHARP}, + {"method_name", "class Local { void N", "Name() {} }\n", CBM_LANG_CSHARP}, + {"parameter_type", "class Local { void M(N", "Name arg) {} }\n", CBM_LANG_CSHARP}, + {"type_parameter", "class Local {}\n", CBM_LANG_CSHARP}, + {"method_parameter", "class Local { void M() {} }\n", CBM_LANG_CSHARP}, + {"field_name", "class Local { int N", "Name; }\n", CBM_LANG_CSHARP}, + {"base_type", "class Local : Acme.N", "Name {}\n", CBM_LANG_CSHARP}, + {"property_value", "

N", "Name

", + CBM_LANG_XML}, + {"property_condition", "

Value

", CBM_LANG_XML}, + {"using_include", "", CBM_LANG_XML}, + {"using_alias", "", CBM_LANG_XML}, + {"import_project", "", + CBM_LANG_XML}, + }; + const char *tail = "/// \nclass Tail {}\n"; + CBMFileResult *healthy = + dm_extract("namespace Good { public class Target {} }\n", CBM_LANG_CSHARP, "Healthy.cs"); + ASSERT_NOT_NULL(healthy); + bool correct = healthy->doc_scope != NULL; + cbm_file_info_t healthy_file = {.rel_path = "Healthy.cs", .language = CBM_LANG_CSHARP}; + cbm_lsp_surface_row_t *healthy_rows = NULL; + int healthy_count = 0; + int healthy_rc = cbm_lsp_surface_build_rows(NULL, "p", &healthy, &healthy_file, 1, NULL, NULL, + &healthy_rows, &healthy_count); + correct = correct && healthy_rc == 0 && healthy_count == 1; + cbm_store_free_lsp_surfaces(healthy_rows, healthy_count); + for (size_t p = 0; p < sizeof(positions) / sizeof(positions[0]); p++) { + for (size_t s = 0; s < sizeof(sequences) / sizeof(sequences[0]); s++) { + char source[2048], label[96]; + const char *path = positions[p].language == CBM_LANG_CSHARP ? "Bad.cs" : "App.csproj"; + size_t a = strlen(positions[p].before), b = strlen(sequences[s].bytes); + size_t c = strlen(positions[p].after); + size_t d = positions[p].language == CBM_LANG_CSHARP ? strlen(tail) : 0; + memcpy(source, positions[p].before, a); + memcpy(source + a, sequences[s].bytes, b); + memcpy(source + a + b, positions[p].after, c); + if (d) { + memcpy(source + a + b + c, tail, d); + } + source[a + b + c + d] = '\0'; + CBMFileResult *bad = cbm_extract_file(source, (int)(a + b + c + d), + positions[p].language, "p", path, 0, NULL, NULL); + snprintf(label, sizeof(label), "%s/%s", positions[p].label, sequences[s].label); + bool well_formed = s < 4; /* ascii, valid2, valid3, valid4 */ + bool setup = bad && dm_utf8_probe_surface(bad, healthy, positions[p].language, path, + label, sequences[s].bytes, well_formed); + correct = setup && correct; + cbm_free_result(bad); + } + } + static const char *entities[] = {"�", "�", "�", "�", + "é", "𐐀", "😀"}; + for (int position = 0; position < 2; position++) { + for (size_t e = 0; e < sizeof(entities) / sizeof(entities[0]); e++) { + char source[1024], label[96]; + const char *format = + position == 0 ? "

N%sName

" + : ""; + int length = snprintf(source, sizeof(source), format, entities[e]); + CBMFileResult *bad = + cbm_extract_file(source, length, CBM_LANG_XML, "p", "App.csproj", 0, NULL, NULL); + snprintf(label, sizeof(label), "entity_%s/%zu", position == 0 ? "property" : "using", + e); + bool setup = bad && dm_utf8_probe_surface(bad, healthy, CBM_LANG_XML, "App.csproj", + label, entities[e], -1); + correct = setup && correct; + cbm_free_result(bad); + } + } + cbm_free_result(healthy); + ASSERT(correct); + PASS(); +} + +/* A file whose tree has parse errors takes its nesting from the braces: error + * recovery closes blocks early and late, which would otherwise move the + * declarations after the error into another namespace or outer type. */ +TEST(doc_mentions_cs_scope_parse_errors) { + /* `ref *(int*)p` is beyond the vendored grammar: the method swallows the + * class's closing brace, `P` becomes a local, `Two` a nested class of + * `One`, and the namespace an error node. */ + const char *late = "namespace A\n" /* 1 */ + "{\n" /* 2 */ + " public class One\n" /* 3 */ + " {\n" /* 4 */ + " unsafe void M(void* p) { _r = ref *(int*)p; }\n" /* 5 */ + " public int P;\n" /* 6 */ + " }\n" /* 7 */ + " public class Two { }\n" /* 8 */ + "}\n" /* 9 */ + "namespace B\n" /* 10 */ + "{\n" + " public class Three { }\n" /* 12 */ + "}\n"; /* 13 */ + CBMFileResult *r = dm_scope(late); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + const char *s = r->doc_scope; + ASSERT_NOT_NULL(strstr(s, "R\t1\t0\t1\t9\tA\n")); /* the block, not the tree's node */ + ASSERT_NOT_NULL( + strstr(s, "T\t1\t3\t7\tc!\t-\tOne\t\t\n")); /* ends at its own brace; a member is hidden */ + ASSERT_NOT_NULL(strstr(s, "M\t5\tc\t0\t0\tM\t\tvoid*\n")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t8\t8\tc\t-\tTwo\t\t\n")); /* a sibling in A, not in One */ + ASSERT_NOT_NULL(strstr(s, "R\t2\t0\t10\t13\tB\n")); + ASSERT_NOT_NULL(strstr(s, "T\t2\t12\t12\tc\t-\tThree\t\t\n")); + ASSERT_NULL(strstr(s, "\nQ\t")); + ASSERT_NULL(strstr(s, "\nX\t")); + cbm_free_result(r); + + /* `ref partial struct` is not parsed as a declaration: its header is read + * from the text (kind, name; bases unknown), and the types after it stay + * in the namespace its closing brace seemed to end. */ + const char *early = "namespace Acme\n" /* 1 */ + "{\n" + " public class Before { }\n" /* 3 */ + " public ref partial struct Iter\n" /* 4 */ + " {\n" + " public void End() { }\n" + " }\n" /* 7 */ + " public class After\n" /* 8 */ + " {\n" + " public int Size;\n" /* 10 */ + " }\n" /* 11 */ + "}\n"; /* 12 */ + r = dm_scope(early); + ASSERT_NOT_NULL(r); + s = r->doc_scope; + ASSERT_NOT_NULL(s); + ASSERT_NOT_NULL(strstr(s, "R\t1\t0\t1\t12\tAcme\n")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t3\t3\tc\t-\tBefore\t\t\n")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t4\t7\tsp!\t-\tIter\tT\t?\n")); /* partial, incomplete */ + ASSERT_NOT_NULL(strstr(s, "T\t1\t8\t11\tc\t-\tAfter\t\t\n")); + ASSERT_NOT_NULL(strstr(s, "M\t10\tv\t0\t2\tSize\t\t-\n")); /* of the third type */ + ASSERT_NULL(strstr(s, "T\t0\t")); /* nothing fell into the global namespace */ + cbm_free_result(r); + + /* Both branches of a conditional open the same block: one closing brace + * serves both, and the members belong to the type either way. */ + const char *branches = "namespace Net\n" /* 1 */ + "{\n" + "#if DEBUG\n" + " internal abstract class Cred : DebugHandle {\n" /* 4 */ + "#else\n" + " internal abstract class Cred : PlainHandle {\n" /* 6 */ + "#endif\n" + " public int Size;\n" /* 8 */ + " }\n" /* 9 */ + " public class After { }\n" /* 10 */ + "}\n"; + r = dm_scope(branches); + ASSERT_NOT_NULL(r); + s = r->doc_scope; + ASSERT_NOT_NULL(s); + ASSERT_NOT_NULL(strstr(s, "R\t1\t0\t1\t11\tNet\n")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t4\t9\tc!\t-\tCred\t\t")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t6\t9\tc!\t-\tCred\t\t")); + ASSERT_NOT_NULL(strstr(s, "M\t8\tv\t0\t1\tSize\t\t-\n")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t10\t10\tc\t-\tAfter\t\t\n")); + ASSERT_NULL(strstr(s, "\nX\t")); + cbm_free_result(r); + + /* A string the parser's own lexer loses the thread in ($@"..." with a + * doubled quote): it then reads code as string content, so the braces come + * from this scan's own reading of the text. */ + const char *derailed = "namespace Net\n" /* 1 */ + "{\n" + " public class Verb\n" /* 3 */ + " {\n" + " void M() { s += $@\", K = \"\"{K.Name}\"\"\"; }\n" /* 5 */ + " public int Q;\n" + " }\n" /* 7 */ + " public class After { }\n" /* 8 */ + "}\n"; /* 9 */ + r = dm_scope(derailed); + ASSERT_NOT_NULL(r); + s = r->doc_scope; + ASSERT_NOT_NULL(s); + ASSERT_NOT_NULL(strstr(s, "R\t1\t0\t1\t9\tNet\n")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t3\t7\tc!\t-\tVerb\t\t")); + ASSERT_NOT_NULL(strstr(s, "T\t1\t8\t8\tc")); + ASSERT_NOT_NULL(strstr(s, "\tAfter\t\t")); + ASSERT_NULL(strstr(s, "\nX\t")); + cbm_free_result(r); + + /* Braces that do not pair: nothing after the first unpaired one is + * placed; the type names are kept, to be resolved to nothing else. */ + const char *open = "namespace Acme\n" /* 1 */ + "{\n" /* 2 */ + " public class Ok { }\n" + " public class Broken\n" + " {\n" + " public void M() {\n" + " public class Lost { }\n"; + r = dm_scope(open); + ASSERT_NOT_NULL(r); + s = r->doc_scope; + ASSERT_NOT_NULL(s); + ASSERT_NOT_NULL(strstr(s, "\nX\t2\t")); /* from the namespace's brace to the end */ + ASSERT_NOT_NULL(strstr(s, "Q\tOk\n")); + ASSERT_NOT_NULL(strstr(s, "Q\tBroken\n")); + ASSERT_NOT_NULL(strstr(s, "Q\tLost\n")); + ASSERT_NULL(strstr(s, "\nT\t")); + ASSERT_NULL(strstr(s, "\nR\t")); + char *portable = cbm_doclink_portable_scope(s); + ASSERT_NOT_NULL(portable); + ASSERT_NOT_NULL(strstr(portable, "\nX\t0\t0\n")); /* line numbers like the others */ + cbm_free(CBM_MEM_CLASS_OTHER, portable); + cbm_free_result(r); + + /* A namespace only one configuration can name is not named at all. */ + const char *either = "#if GEN\n" + "namespace Gen.Interop\n" + "#else\n" + "namespace Run.Interop\n" + "#endif\n" + "{\n" + " public enum Mode { One }\n" + "}\n"; + r = dm_scope(either); + ASSERT_NOT_NULL(r); + s = r->doc_scope; + ASSERT_NOT_NULL(s); + ASSERT_NULL(strstr(s, "\nR\t")); + ASSERT_NOT_NULL(strstr(s, "Q\tMode\n")); + ASSERT_NOT_NULL(strstr(s, "\nX\t")); /* and what is documented there has no scope */ + cbm_free_result(r); + + /* A member whose header did not parse is hidden, and its type says so: + * `safe extern` comes back as a method named `extern`. */ + const char *hidden = "namespace C\n" + "{\n" + " public class Holey\n" /* 3 */ + " {\n" + " public safe extern int Hidden();\n" /* 5 */ + " public int Seen;\n" /* 6 */ + " }\n" /* 7 */ + "}\n"; + r = dm_scope(hidden); + ASSERT_NOT_NULL(r); + s = r->doc_scope; + ASSERT_NOT_NULL(s); + ASSERT_NOT_NULL(strstr(s, "T\t1\t3\t7\tc!\t-\tHoley\t\t\n")); + ASSERT_NOT_NULL(strstr(s, "M\t6\tv\t0\t0\tSeen\t\t-\n")); + ASSERT_NULL(strstr(s, "\textern\t")); + ASSERT_NULL(strstr(s, "\tHidden\t")); + cbm_free_result(r); + PASS(); +} + +TEST(doc_mentions_cs_norm_type) { + struct { + const char *in; + const char *out; + } cases[] = { + {"ref int x", "int"}, + {"params string[] args", "string[]"}, + {"[NotNull] System.Collections.Generic.List xs", "List"}, + {"Nullable{System.Int32}", "int"}, + {"System.String", "string"}, + {"T?", "T"}, + {"byte*", "byte*"}, + {"int[,]", "int[,]"}, + {"``0", "?"}, + {"", "?"}, + }; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + char out[128]; + cbm_doclink_cs_norm_type(cases[i].in, strlen(cases[i].in), out, sizeof(out)); + if (strcmp(out, cases[i].out) != 0) { + printf(" norm(%s) = %s, want %s\n", cases[i].in, out, cases[i].out); + FAIL("normalized type"); + } + } + /* a name that does not fit the caller's buffer is a type nothing is known + * about, never a cut name: two long names with one beginning would + * compare equal */ + char small[8]; + const char *long_name = "LongTypeNameOne"; + ASSERT_EQ(cbm_doclink_cs_norm_type(long_name, strlen(long_name), small, sizeof(small)), 1); + ASSERT_STR_EQ(small, "?"); + ASSERT_EQ(cbm_doclink_cs_norm_type("Fits", 4, small, sizeof(small)), 4); + ASSERT_STR_EQ(small, "Fits"); + PASS(); +} + +/* ── MSBuild global usings (R1) ──────────────────────────────────── */ + +/* A project file of a test: its path in the repository and its text. */ +typedef struct { + const char *rel_path; + const char *xml; +} dm_project_file_t; + +/* Evaluate `project` over `files`, each through the extractor's scope blob: + * no file is written, none is read. The result is released with + * cbm_msb_result_free. false when a file yields no blob or memory runs out. */ +static bool dm_msb_eval(const dm_project_file_t *files, int n, const char *project, + cbm_msb_result_t *out) { + cbm_msb_t *m = cbm_msb_new(); + bool ok = m != NULL; + for (int i = 0; ok && i < n; i++) { + CBMFileResult *r = dm_extract(files[i].xml, CBM_LANG_XML, files[i].rel_path); + ok = r && r->doc_scope && cbm_msb_is_project_scope(r->doc_scope) && + cbm_msb_add(m, files[i].rel_path, r->doc_scope); + if (r) { + cbm_free_result(r); + } + } + ok = ok && cbm_msb_eval(m, project, out); + cbm_msb_free(m); + return ok; +} + +/* True when the result holds a using of this kind (n namespace, s static, + * a alias) and target. */ +static bool dm_has_using(const cbm_msb_result_t *r, char kind, const char *target) { + for (int i = 0; i < r->count; i++) { + if (r->usings[i].kind == kind && strcmp(r->usings[i].target, target) == 0) { + return true; + } + } + return false; +} + +/* Repeated expansion must not retain every superseded value or condition + * operand. Measure real evaluation arena capacity, not a wall-clock sample. */ +static bool dm_msb_add_xml(cbm_msb_t *m, const char *path, const char *xml) { + CBMFileResult *source = dm_extract(xml, CBM_LANG_XML, path); + bool ok = source && source->doc_scope && cbm_msb_add(m, path, source->doc_scope); + if (source) { + cbm_free_result(source); + } + return ok; +} + +static bool dm_msb_same(const cbm_msb_result_t *a, const cbm_msb_result_t *b) { + if (a->count != b->count || a->open != b->open || a->unevaluable != b->unevaluable || + a->outside != b->outside) { + return false; + } + for (int i = 0; i < a->count; i++) { + if (a->usings[i].kind != b->usings[i].kind || + strcmp(a->usings[i].alias, b->usings[i].alias) || + strcmp(a->usings[i].target, b->usings[i].target)) { + return false; + } + } + return true; +} + +static bool dm_msb_context_matches(cbm_msb_eval_context_t *context, const cbm_msb_t *m, + const char *project) { + cbm_msb_result_t actual = {0}; + cbm_msb_result_t reference = {0}; + bool ok = cbm_msb_eval_context_eval(context, project, &actual) && + cbm_msb_eval(m, project, &reference) && dm_msb_same(&actual, &reference); + if (!ok) { + fprintf(stderr, "MSBuild context differs from uncached evaluation: %s\n", project); + } + cbm_msb_result_free(&actual); + cbm_msb_result_free(&reference); + return ok; +} + +/* Include construction, per-project work and context destruction. A hit must + * not replay/copy all shared properties behind a constant record counter. */ +/* The hit-phase comparison grows shared dependencies while keeping local + * inputs/output fixed. Scanning the entire cached dependency set must fail. */ +/* Dependence on project input is exact, including absence versus unknown. + * Input owners are gone before the next call; cached effects must own any + * captured local values and preserve import/poison history. */ +TEST(doc_mentions_msbuild_targets_isolation) { + const dm_project_file_t files[] = { + {"Directory.Build.props", + "Prefix" + ""}, + {"PrefixOnly.props", "PrefixStable" + ""}, + {"Directory.Build.targets", + "" + "" + "$(Input)$(Empty)Tail" + "$(Custom)First$(Own)" + "Final" + "" + "" + "" + "" + "" + "" + "" + "" + "" + ""}, + {"Seen.targets", "SharedSeen" + ""}, + {"Maybe.targets", "Hidden" + ""}, + {"PoisonInput.props", + "Hidden" + "HiddenHidden"}, + {"Irrelevant.props", "Anything" + ""}, + {"A.csproj", "SameCommon" + "Project"}, + {"B.csproj", "SameCommon" + ""}, + {"Changed.csproj", + "Changed" + "DifferentNonempty"}, + {"Missing.csproj", ""}, + {"Empty.csproj", "" + ""}, + {"Unknown.csproj", ""}, + {"EarlySeen.csproj", "" + "LocalSeen"}, + {"EarlyPoison.csproj", "" + "Restored"}, + {"EarlyRoot.csproj", "" + "AfterAfter" + ""}, + {"Seed/Directory.Build.targets", + "" + "$(MSBuildProjectName)" + ""}, + {"Seed/A.csproj", ""}, + {"Seed/B.csproj", ""}, + {"Seed/left/A.csproj", + "Fixed" + ""}, + {"Seed/right/A.csproj", ""}, + {"Seed/Unknown.csproj", ""}, + {"Seed/SeedPoison.props", + "Hidden" + ""}, + {"Seed/Override.csproj", + "Fixed" + ""}, + {"Later/Directory.Build.targets", ""}, + {"Later/A.csproj", ""}, + {"Later/B.csproj", ""}, + }; + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + for (size_t i = 0; i < sizeof(files) / sizeof(files[0]); i++) { + ASSERT_TRUE(dm_msb_add_xml(m, files[i].rel_path, files[i].xml)); + } + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + const char *projects[] = {"A.csproj", + "B.csproj", + "A.csproj", + "Changed.csproj", + "A.csproj", + "Missing.csproj", + "Empty.csproj", + "Unknown.csproj", + "Missing.csproj", + "A.csproj", + "EarlySeen.csproj", + "Missing.csproj", + "EarlyPoison.csproj", + "Missing.csproj", + "EarlyRoot.csproj", + "A.csproj", + "Seed/A.csproj", + "Seed/B.csproj", + "Seed/Override.csproj", + "Seed/A.csproj", + "Seed/left/A.csproj", + "Seed/right/A.csproj", + "Seed/Unknown.csproj", + "Seed/A.csproj", + "Later/A.csproj", + "Later/B.csproj"}; + for (size_t i = 0; i < sizeof(projects) / sizeof(projects[0]); i++) { + ASSERT_TRUE(dm_msb_context_matches(context, m, projects[i])); + } + ASSERT_TRUE(dm_msb_add_xml(m, "Later/Later.targets", + "" + "")); + ASSERT_TRUE(dm_msb_context_matches(context, m, "Later/A.csproj")); + ASSERT_TRUE(dm_msb_context_matches(context, m, "Later/B.csproj")); + ASSERT_TRUE(dm_msb_context_matches(context, m, "A.csproj")); + cbm_msb_eval_context_free(context); + ASSERT_EQ(cbm_msb_test_value_live_bytes(), 0); + cbm_msb_free(m); + PASS(); +} + +/* A target captures project-owned values before it fails. The same context + * must be usable again, and later failures must leave retained owners sound. */ +TEST(doc_mentions_msbuild_targets_failure) { + char xml[8192]; + size_t w = (size_t)snprintf(xml, sizeof(xml), ""); + for (int i = 0; i < 64; i++) { + w += (size_t)snprintf(xml + w, sizeof(xml) - w, "$(Input)%02d", i, i, i); + } + w += (size_t)snprintf(xml + w, sizeof(xml) - w, + "" + ""); + ASSERT_TRUE(w < sizeof(xml)); + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE( + dm_msb_add_xml(m, "Directory.Build.props", + "Prefix")); + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.targets", xml)); + ASSERT_TRUE(dm_msb_add_xml(m, "A.csproj", + "Shared" + "")); + ASSERT_TRUE(dm_msb_add_xml(m, "B.csproj", + "Changed" + "")); + int failures = 0; + for (int mode = 0; mode < 2; mode++) { + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + cbm_msb_test_fail_value_alloc_after(mode == 0 ? 16 : 0); + cbm_msb_test_fail_prop_insert_after(mode == 1 ? 16 : 0); + cbm_msb_result_t r = {0}; + bool ok = cbm_msb_eval_context_eval(context, "A.csproj", &r); + bool consumed = + mode == 0 ? cbm_msb_test_value_alloc_failed() : cbm_msb_test_prop_insert_failed(); + failures += ok || !consumed || r.mem != NULL || r.usings != NULL || r.count != 0; + cbm_msb_result_free(&r); + cbm_msb_test_fail_value_alloc_after(0); + cbm_msb_test_fail_prop_insert_after(0); + failures += !dm_msb_context_matches(context, m, "A.csproj"); + uint64_t retained = cbm_msb_test_value_live_bytes(); + cbm_msb_test_fail_value_alloc_after(mode == 0 ? 1 : 0); + cbm_msb_test_fail_prop_insert_after(mode == 1 ? 1 : 0); + ok = cbm_msb_eval_context_eval(context, "B.csproj", &r); + consumed = + mode == 0 ? cbm_msb_test_value_alloc_failed() : cbm_msb_test_prop_insert_failed(); + failures += ok || !consumed || r.mem != NULL || r.usings != NULL || r.count != 0 || + cbm_msb_test_value_live_bytes() > retained; + cbm_msb_result_free(&r); + cbm_msb_test_fail_value_alloc_after(0); + cbm_msb_test_fail_prop_insert_after(0); + failures += !dm_msb_context_matches(context, m, "A.csproj"); + failures += !dm_msb_context_matches(context, m, "B.csproj"); + failures += !dm_msb_context_matches(context, m, "A.csproj"); + cbm_msb_eval_context_free(context); + failures += cbm_msb_test_value_live_bytes() != 0; + fprintf(stderr, "msbuild targets fault mode=%d retained=%llu failures=%d\n", mode, + (unsigned long long)retained, failures); + } + cbm_msb_free(m); + ASSERT_EQ(failures, 0); + PASS(); +} + +/* Final shared items read project inputs after target writes. Cached includes + * and removals must compose with local items and retain diagnostic counts. */ +TEST(doc_mentions_msbuild_items_isolation) { + const dm_project_file_t files[] = { + {"Directory.Build.props", + "" + "" + "" + ""}, + {"Parts/One.props", "" + "" + ""}, + {"Directory.Build.targets", + "Targets.Fixed" + "" + "" + "" + ""}, + {"Parts/Two.targets", "" + "" + "" + ""}, + {"Poison.props", "HiddenHidden" + "HiddenHiddenHidden" + "Hidden"}, + {"A.csproj", "Same.InputSameAlias" + "falseShared.Dropon" + "Project.A" + "" + ""}, + {"B.csproj", + "Same.InputSameAlias" + "falseShared.Dropon" + "Project.BDifferent" + "" + ""}, + {"Changed.csproj", + "Changed.InputChangedAlias" + "trueTargets.Fixedoff" + "on"}, + {"Empty.csproj", "" + "" + ""}, + {"Missing.csproj", ""}, + {"Unknown.csproj", ""}, + {"Other/Directory.Build.props", "" + "" + ""}, + {"Other/A.csproj", ""}, + {"Target/Directory.Build.targets", "" + "" + "" + ""}, + {"Target/A.csproj", ""}, + {"Later/A.csproj", ""}, + {"Later/B.csproj", ""}, + }; + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + for (size_t i = 0; i < sizeof(files) / sizeof(files[0]); i++) { + ASSERT_TRUE(dm_msb_add_xml(m, files[i].rel_path, files[i].xml)); + } + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + const char *projects[] = { + "A.csproj", "B.csproj", "A.csproj", "Changed.csproj", "A.csproj", + "Empty.csproj", "Missing.csproj", "Unknown.csproj", "Missing.csproj", "B.csproj", + "Other/A.csproj", "Other/A.csproj", "A.csproj", "Target/A.csproj", "Target/A.csproj", + "A.csproj", "Later/A.csproj", "Later/B.csproj"}; + for (size_t i = 0; i < sizeof(projects) / sizeof(projects[0]); i++) { + ASSERT_TRUE(dm_msb_context_matches(context, m, projects[i])); + } + ASSERT_TRUE(dm_msb_add_xml(m, "Later/Directory.Build.props", + "" + "")); + ASSERT_TRUE(dm_msb_context_matches(context, m, "Later/A.csproj")); + ASSERT_TRUE(dm_msb_context_matches(context, m, "Later/B.csproj")); + ASSERT_TRUE(dm_msb_context_matches(context, m, "A.csproj")); + cbm_msb_eval_context_free(context); + ASSERT_EQ(cbm_msb_test_value_live_bytes(), 0); + cbm_msb_free(m); + PASS(); +} + +/* The seams fail real allocation/insertion operations. Context-owned state + * must remain usable, and all allocator-tracked storage must be released. */ +TEST(doc_mentions_msbuild_items_failure) { + char xml[16384]; + size_t w = (size_t)snprintf(xml, sizeof(xml), ""); + for (int i = 0; i < 64; i++) { + w += (size_t)snprintf(xml + w, sizeof(xml) - w, + "", i); + } + w += (size_t)snprintf(xml + w, sizeof(xml) - w, ""); + ASSERT_TRUE(w < sizeof(xml)); + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.props", xml)); + ASSERT_TRUE(dm_msb_add_xml(m, "A.csproj", + "Shared" + "")); + ASSERT_TRUE(dm_msb_add_xml(m, "B.csproj", + "Shared" + "Different")); + ASSERT_TRUE(dm_msb_add_xml(m, "C.csproj", + "Changed" + "")); + const struct { + cbm_msb_item_fail_operation_t operation; + int nth; + bool warm; + } cases[] = { + {CBM_MSB_ITEM_FAIL_CAPTURE_ALLOC, 1, false}, {CBM_MSB_ITEM_FAIL_CAPTURE_ALLOC, 9, false}, + {CBM_MSB_ITEM_FAIL_CAPTURE_ALLOC, 33, false}, {CBM_MSB_ITEM_FAIL_UNIQUE_INSERT, 1, false}, + {CBM_MSB_ITEM_FAIL_UNIQUE_INSERT, 9, false}, {CBM_MSB_ITEM_FAIL_UNIQUE_INSERT, 33, false}, + {CBM_MSB_ITEM_FAIL_APPLY_ALLOC, 1, true}, {CBM_MSB_ITEM_FAIL_PUBLISH_ALLOC, 1, true}}; + int failures = 0; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + size_t before = cbm_mem_tracked_live_bytes(); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + if (cases[i].warm) { + failures += !dm_msb_context_matches(context, m, "A.csproj"); + } + size_t retained = cbm_mem_tracked_live_bytes(); + cbm_msb_test_fail_item_operation(cases[i].operation, cases[i].nth); + cbm_msb_result_t r = {0}; + bool ok = cbm_msb_eval_context_eval(context, cases[i].warm ? "B.csproj" : "A.csproj", &r); + bool consumed = cbm_msb_test_item_operation_failed(); + failures += ok || !consumed || r.mem != NULL || r.usings != NULL || r.count != 0; + cbm_msb_result_free(&r); + cbm_msb_test_fail_item_operation(CBM_MSB_ITEM_FAIL_NONE, 0); + if (cases[i].warm) { + failures += cbm_mem_tracked_live_bytes() != retained; + } + failures += !dm_msb_context_matches(context, m, "A.csproj"); + failures += !dm_msb_context_matches(context, m, "B.csproj"); + failures += !dm_msb_context_matches(context, m, "C.csproj"); + failures += !dm_msb_context_matches(context, m, "A.csproj"); + cbm_msb_eval_context_free(context); + size_t after = cbm_mem_tracked_live_bytes(); + failures += after != before || cbm_msb_test_value_live_bytes() != 0; + fprintf(stderr, + "msbuild items fault operation=%d nth=%d consumed=%d " + "before=%llu after=%llu failures=%d\n", + (int)cases[i].operation, cases[i].nth, consumed, (unsigned long long)before, + (unsigned long long)after, failures); + } + cbm_msb_free(m); + ASSERT_EQ(failures, 0); + PASS(); +} + +/* A published result owns its strings independently of the retained cache, + * its replacement and the model. A mismatch must keep the first variant. */ +TEST(doc_mentions_msbuild_items_lifetime) { + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.props", + "" + "" + "" + "")); + ASSERT_TRUE(dm_msb_add_xml(m, "A.csproj", + "First.Value" + "")); + ASSERT_TRUE(dm_msb_add_xml(m, "B.csproj", + "Other.Value" + "")); + ASSERT_TRUE(dm_msb_add_xml(m, "Other/Directory.Build.props", + "" + "")); + ASSERT_TRUE(dm_msb_add_xml(m, "Other/A.csproj", "")); + cbm_msb_result_t expected = {0}; + ASSERT_TRUE(cbm_msb_eval(m, "A.csproj", &expected)); + ASSERT_EQ(expected.count, 3); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + cbm_msb_result_t held = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, "A.csproj", &held)); + ASSERT_TRUE(dm_msb_same(&held, &expected)); + cbm_msb_result_t hit = {0}; + cbm_msb_test_cost_reset(); + ASSERT_TRUE(cbm_msb_eval_context_eval(context, "A.csproj", &hit)); + uint64_t before = 0; + uint64_t peak = 0; + cbm_msb_test_cost(&before, &peak); + ASSERT_TRUE(dm_msb_same(&hit, &expected)); + cbm_msb_result_free(&hit); + ASSERT_TRUE(dm_msb_context_matches(context, m, "B.csproj")); + cbm_msb_test_cost_reset(); + ASSERT_TRUE(cbm_msb_eval_context_eval(context, "A.csproj", &hit)); + uint64_t after = 0; + cbm_msb_test_cost(&after, &peak); + bool same_hit_work = before == after; + ASSERT_TRUE(dm_msb_same(&hit, &expected)); + cbm_msb_result_free(&hit); + ASSERT_TRUE(dm_msb_context_matches(context, m, "Other/A.csproj")); + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.targets", + "" + "")); + ASSERT_TRUE(dm_msb_context_matches(context, m, "A.csproj")); + cbm_msb_eval_context_free(context); + cbm_msb_free(m); + bool independent = dm_msb_same(&held, &expected); + cbm_msb_result_free(&held); + cbm_msb_result_free(&expected); + fprintf(stderr, "msbuild items retained variant before=%llu after=%llu independent=%d\n", + (unsigned long long)before, (unsigned long long)after, independent); + ASSERT_TRUE(same_hit_work && independent); + ASSERT_EQ(cbm_msb_test_value_live_bytes(), 0); + PASS(); +} + +/* Shared final-property items can produce no output, or the same unique + * output, despite a large input span. Warm work must not scale with that + * span after an exact reusable result is available. */ +TEST(doc_mentions_msbuild_items_work) { + enum { CAP = 131072, PROJECTS = 24 }; + char *xml = malloc(CAP); + ASSERT_NOT_NULL(xml); + uint64_t hits[9][2] = {{0}}; + bool correct = true; + bool storage_bounded = true; + char long_alias[4096]; + memset(long_alias, 'A', sizeof(long_alias) - 1); + long_alias[sizeof(long_alias) - 1] = '\0'; + for (int mode = 0; mode < 9; mode++) { + for (int size = 0; size < 2; size++) { + int items = 512 << size; + size_t w = (size_t)snprintf(xml, CAP, ""); + if (mode == 6 || mode == 7) { + for (int i = 0; i < items; i++) { + w += (size_t)snprintf(xml + w, CAP - w, "off", i, i); + } + } else if (mode == 8) { + w += (size_t)snprintf(xml + w, CAP - w, "%s", long_alias); + } + w += (size_t)snprintf(xml + w, CAP - w, ""); + for (int i = 0; i < items; i++) { + if (mode == 0) { + w += (size_t)snprintf(xml + w, CAP - w, + "", i); + } else if (mode == 1) { + w += (size_t)snprintf(xml + w, CAP - w, + ""); + } else if (mode == 2) { + w += (size_t)snprintf(xml + w, CAP - w, + "", + i); + } else if (mode == 3) { + w += (size_t)snprintf(xml + w, CAP - w, + "", + i); + } else if (mode == 4) { + w += (size_t)snprintf(xml + w, CAP - w, "", i); + } else if (mode == 5) { + w += (size_t)snprintf(xml + w, CAP - w, + "" + "", + i, i); + } else if (mode == 6 || mode == 7) { + w += (size_t)snprintf(xml + w, CAP - w, + "", + i, i); + } else { + w += + (size_t)snprintf(xml + w, CAP - w, + ""); + } + } + w += (size_t)snprintf(xml + w, CAP - w, ""); + ASSERT_TRUE(w < CAP); + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.props", xml)); + if (mode == 7) { + /* All final K values come from targets; the changing local + * K0000 value must not invalidate their item dependencies. */ + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.targets", xml)); + } + ASSERT_TRUE(dm_msb_add_xml( + m, "src/Warm.csproj", + mode == 4 ? "off" + "" + : "off" + "")); + for (int i = 0; i < PROJECTS; i++) { + char path[64]; + char project[256]; + snprintf(path, sizeof(path), "src/P%02d.csproj", i); + snprintf(project, sizeof(project), + "offP%02d%s" + "%s", + i, mode == 7 ? "on" : "", + mode == 4 ? "" + : ""); + ASSERT_TRUE(dm_msb_add_xml(m, path, project)); + } + cbm_msb_test_cost_reset(); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + cbm_msb_result_t warm = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, "src/Warm.csproj", &warm)); + correct = correct && warm.count == (mode == 1 || mode == 4 || mode == 8 ? 1 : 0) && + warm.open == (mode == 2) && warm.unevaluable == (mode == 2 ? items : 0) && + !warm.outside; + cbm_msb_result_free(&warm); + uint64_t captured = cbm_msb_test_work(); + for (int i = 0; i < PROJECTS; i++) { + char path[64]; + snprintf(path, sizeof(path), "src/P%02d.csproj", i); + cbm_msb_result_t r = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, path, &r)); + correct = correct && r.count == (mode == 1 || mode == 4 || mode == 8 ? 1 : 0) && + r.open == (mode == 2) && r.unevaluable == (mode == 2 ? items : 0) && + !r.outside; + if (mode == 1 || mode == 8) { + correct = + correct && r.count == 1 && r.usings[0].kind == 'a' && + strcmp(r.usings[0].alias, mode == 8 ? long_alias : "SameAlias") == 0 && + strcmp(r.usings[0].target, "Same.Target") == 0; + } else if (mode == 4) { + correct = correct && r.count == 1 && r.usings[0].kind == 'n' && + strcmp(r.usings[0].alias, "") == 0 && + strcmp(r.usings[0].target, "Project.Keep") == 0; + } + cbm_msb_result_free(&r); + } + hits[mode][size] = cbm_msb_test_work() - captured; + cbm_msb_eval_context_free(context); + uint64_t records = 0; + uint64_t peak = 0; + cbm_msb_test_cost(&records, &peak); + if (mode == 8) { + /* One long value and one unique output must not leave a + * persistent alias copy per duplicate item during capture. */ + uint64_t bound = 16 * (uint64_t)w + 1024 * 1024; + storage_bounded = storage_bounded && peak <= bound; + fprintf(stderr, "msbuild items duplicate storage items=%d peak=%llu bound=%llu\n", + items, (unsigned long long)peak, (unsigned long long)bound); + } + fprintf(stderr, + "msbuild items mode=%d items=%d projects=%d interpreted=%llu " + "work=%llu hit_work=%llu peak=%llu live=%llu semantics=%d\n", + mode, items, PROJECTS, (unsigned long long)records, + (unsigned long long)cbm_msb_test_work(), (unsigned long long)hits[mode][size], + (unsigned long long)peak, (unsigned long long)cbm_msb_test_value_live_bytes(), + correct); + correct = correct && cbm_msb_test_value_live_bytes() == 0; + cbm_msb_free(m); + } + } + free(xml); + ASSERT_TRUE(correct); + bool bounded = true; + for (int mode = 0; mode < 9; mode++) { + ASSERT_TRUE(hits[mode][0] > 0); + double ratio = (double)hits[mode][1] / (double)hits[mode][0]; + fprintf(stderr, "msbuild items shared-size hit ratio mode=%d ratio=%.3f\n", mode, ratio); + bounded = bounded && ratio >= 0.9 && ratio <= 1.1; + } + ASSERT_TRUE(bounded && storage_bounded); + PASS(); +} + +TEST(doc_mentions_msbuild_targets_work) { + enum { CAP = 131072 }; + char *props = malloc(CAP); + char *targets = malloc(CAP); + ASSERT_NOT_NULL(props); + ASSERT_NOT_NULL(targets); + uint64_t work[4][2][2] = {{{0}}}; + uint64_t hits[4][2][2] = {{{0}}}; + bool correct = true; + for (int mode = 0; mode < 4; mode++) { + for (int size = 0; size < 2; size++) { + int records = 512 << size; + size_t p = (size_t)snprintf(props, CAP, ""); + size_t t = (size_t)snprintf(targets, CAP, ""); + for (int i = 0; i < records; i++) { + if (mode == 1) { + p += (size_t)snprintf(props + p, CAP - p, "Shared%04d", i, i, i); + t += (size_t)snprintf(targets + t, CAP - t, "$(K%04d)", i, i, i); + } else if (mode == 2) { + t += (size_t)snprintf(targets + t, CAP - t, "$(Custom)%04d", i, + i, i); + } else { + t += (size_t)snprintf(targets + t, CAP - t, "Shared%04d", i, i, + i); + } + } + p += (size_t)snprintf(props + p, CAP - p, + "Prefix"); + t += (size_t)snprintf( + targets + t, CAP - t, + "" + "" + "" + "", + records - 1); + ASSERT_TRUE(p < CAP && t < CAP); + for (int count = 0; count < 2; count++) { + int projects = 16 << count; + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + /* No prefix file in mode 3: absence is a stable baseline, + * not a reason to discard a reusable target on every call. */ + if (mode != 3) { + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.props", props)); + } + ASSERT_TRUE(dm_msb_add_xml(m, "Shared.targets", targets)); + ASSERT_TRUE( + dm_msb_add_xml(m, "Directory.Build.targets", + "")); + for (int i = 0; i < projects; i++) { + char path[64]; + char xml[512]; + snprintf(path, sizeof(path), "src/P%02d.csproj", i); + snprintf( + xml, sizeof(xml), + "" + "Shared0000SharedP%02d" + "", + i); + ASSERT_TRUE(dm_msb_add_xml(m, path, xml)); + } + /* Capture many redundant overrides in mode 1; later projects + * omit them. They must not create required local exceptions. */ + ASSERT_TRUE(dm_msb_add_xml(m, "src/Warm.csproj", + mode == 2 ? "Shared" + : props)); + cbm_msb_test_cost_reset(); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + cbm_msb_result_t warm = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, "src/Warm.csproj", &warm)); + correct = correct && warm.count == 3 && !warm.open && !warm.unevaluable && + !warm.outside && dm_has_using(&warm, 'a', "Shared0000"); + cbm_msb_result_free(&warm); + uint64_t captured = cbm_msb_test_work(); + for (int i = 0; i < projects; i++) { + char path[64]; + char name[32]; + char last[32]; + snprintf(path, sizeof(path), "src/P%02d.csproj", i); + snprintf(name, sizeof(name), "P%02d", i); + snprintf(last, sizeof(last), "Shared%04d", records - 1); + cbm_msb_result_t r = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, path, &r)); + correct = correct && r.count == 3 && !r.open && !r.unevaluable && !r.outside && + dm_has_using(&r, 'a', "Shared0000") && dm_has_using(&r, 'a', last) && + dm_has_using(&r, 'a', name); + cbm_msb_result_free(&r); + } + hits[mode][size][count] = cbm_msb_test_work() - captured; + cbm_msb_eval_context_free(context); + work[mode][size][count] = cbm_msb_test_work(); + uint64_t consumed = 0; + uint64_t peak = 0; + cbm_msb_test_cost(&consumed, &peak); + fprintf( + stderr, + "msbuild targets mode=%d records=%d projects=%d " + "interpreted=%llu work=%llu hit_work=%llu peak=%llu live=%llu semantics=%d\n", + mode, records, projects, (unsigned long long)consumed, + (unsigned long long)work[mode][size][count], + (unsigned long long)hits[mode][size][count], (unsigned long long)peak, + (unsigned long long)cbm_msb_test_value_live_bytes(), correct); + correct = correct && cbm_msb_test_value_live_bytes() == 0; + cbm_msb_free(m); + } + } + } + free(props); + free(targets); + ASSERT_TRUE(correct); + bool bounded = true; + for (int mode = 0; mode < 4; mode++) { + for (int size = 0; size < 2; size++) { + ASSERT_TRUE(work[mode][size][0] > 0); + double ratio = (double)work[mode][size][1] / (double)work[mode][size][0]; + fprintf(stderr, "msbuild targets project ratio mode=%d records=%d ratio=%.3f\n", mode, + 512 << size, ratio); + bounded = bounded && ratio >= 0.9 && ratio <= 1.4; + } + for (int count = 0; count < 2; count++) { + ASSERT_TRUE(hits[mode][0][count] > 0); + double ratio = (double)hits[mode][1][count] / (double)hits[mode][0][count]; + fprintf(stderr, + "msbuild targets shared-size hit ratio mode=%d projects=%d ratio=%.3f\n", mode, + 16 << count, ratio); + bounded = bounded && ratio >= 0.9 && ratio <= 1.1; + } + } + ASSERT_TRUE(bounded); + PASS(); +} + +TEST(doc_mentions_msbuild_nearest_work) { + enum { DEPTH = 48, PROJECTS = 24 }; + char dir[512] = ""; + for (int i = 0; i < DEPTH; i++) { + strcat(dir, "deep/"); + } + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.props", "")); + for (int i = 0; i < PROJECTS; i++) { + char path[600]; + snprintf(path, sizeof(path), "%sP%02d.csproj", dir, i); + ASSERT_TRUE(dm_msb_add_xml(m, path, "")); + } + cbm_msb_test_cost_reset(); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + bool correct = true; + for (int i = 0; i < PROJECTS; i++) { + char path[600]; + snprintf(path, sizeof(path), "%sP%02d.csproj", dir, i); + cbm_msb_result_t r = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, path, &r)); + correct = correct && r.count == 0 && !r.open && !r.unevaluable && !r.outside; + cbm_msb_result_free(&r); + } + uint64_t probes = cbm_msb_test_nearest_steps(); + char target[600]; + snprintf(target, sizeof(target), "%sDirectory.Build.targets", dir); + ASSERT_TRUE(dm_msb_add_xml(m, target, + "" + "")); + char project[600]; + snprintf(project, sizeof(project), "%sP00.csproj", dir); + cbm_msb_result_t r = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, project, &r)); + correct = correct && r.count == 1 && !r.open && !r.unevaluable && !r.outside && + dm_has_using(&r, 'n', "New.Target"); + cbm_msb_result_free(&r); + uint64_t fresh = cbm_msb_test_nearest_steps() - probes; + cbm_msb_eval_context_free(context); + correct = correct && cbm_msb_test_value_live_bytes() == 0; + cbm_msb_free(m); + fprintf(stderr, + "msbuild nearest depth=%d projects=%d probes=%llu after_add=%llu semantics=%d\n", DEPTH, + PROJECTS, (unsigned long long)probes, (unsigned long long)fresh, correct); + ASSERT_TRUE(correct); + ASSERT_TRUE(probes <= 2 * (DEPTH + 1)); + ASSERT_TRUE(fresh > 0 && fresh <= 2 * (DEPTH + 1)); + PASS(); +} + +/* Distinct nearest-file roots must not replay the common imported state. + * All models are built before measurement. The warm root and measured roots + * differ; doubling shared size must not change subsequent per-root work. */ +TEST(doc_mentions_msbuild_shared_closure_work) { + enum { CAP = 131072 }; + char *xml = malloc(CAP); + ASSERT_NOT_NULL(xml); + char *definitions = malloc(CAP); + ASSERT_NOT_NULL(definitions); + uint64_t work[4][2][2] = {{{0}}}; + uint64_t hits[4][2][2] = {{{0}}}; + bool correct = true; + for (int mode = 0; mode < 4; mode++) { + const char *suffix = mode % 2 == 0 ? "props" : "targets"; + for (int size = 0; size < 2; size++) { + int records = 512 << size; + size_t w = (size_t)snprintf(xml, CAP, ""); + size_t d = (size_t)snprintf(definitions, CAP, ""); + for (int i = 0; i < records; i++) { + if (mode >= 2) { + d += (size_t)snprintf(definitions + d, CAP - d, "Shared%04d", i, + i, i); + w += (size_t)snprintf(xml + w, CAP - w, "$(K%04d)", i, i, i); + } else { + w += (size_t)snprintf(xml + w, CAP - w, "Shared%04d", i, i, i); + } + } + d += (size_t)snprintf(definitions + d, CAP - d, ""); + w += (size_t)snprintf( + xml + w, CAP - w, + "" + "" + "" + "", + mode >= 2 ? 'T' : 'K', mode >= 2 ? 'T' : 'K', records - 1); + ASSERT_TRUE(w < CAP && d < CAP); + for (int count = 0; count < 2; count++) { + int projects = 16 << count; + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + char shared[64]; + char wrapper[256]; + snprintf(shared, sizeof(shared), "Shared.%s", suffix); + if (mode >= 2) { + char define_path[64]; + snprintf(define_path, sizeof(define_path), "Definitions.%s", suffix); + ASSERT_TRUE(dm_msb_add_xml(m, define_path, definitions)); + snprintf(wrapper, sizeof(wrapper), + "" + "", + suffix, suffix); + } else { + snprintf(wrapper, sizeof(wrapper), + "", suffix); + } + ASSERT_TRUE(dm_msb_add_xml(m, shared, xml)); + for (int i = 0; i <= projects; i++) { + char root[96]; + char project[96]; + snprintf(root, sizeof(root), "src/R%02d/Directory.Build.%s", i, suffix); + snprintf(project, sizeof(project), "src/R%02d/P%02d.csproj", i, i); + ASSERT_TRUE(dm_msb_add_xml(m, root, wrapper)); + ASSERT_TRUE(dm_msb_add_xml(m, project, "")); + } + cbm_msb_test_cost_reset(); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + uint64_t captured = 0; + for (int i = 0; i <= projects; i++) { + char path[96]; + char name[32]; + char last[32]; + snprintf(path, sizeof(path), "src/R%02d/P%02d.csproj", i, i); + snprintf(name, sizeof(name), "P%02d", i); + snprintf(last, sizeof(last), "Shared%04d", records - 1); + cbm_msb_result_t r = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, path, &r)); + cbm_msb_using_t expected_usings[] = { + {.kind = 'a', .alias = "Project", .target = name}, + {.kind = 'a', .alias = "First", .target = "Shared0000"}, + {.kind = 'a', .alias = "Last", .target = last}, + }; + cbm_msb_result_t expected = {.usings = expected_usings, .count = 3}; + correct = correct && dm_msb_same(&r, &expected); + cbm_msb_result_free(&r); + if (i == 0) { + captured = cbm_msb_test_work(); + } + } + hits[mode][size][count] = cbm_msb_test_work() - captured; + cbm_msb_eval_context_free(context); + work[mode][size][count] = cbm_msb_test_work(); + uint64_t interpreted = 0; + uint64_t peak = 0; + cbm_msb_test_cost(&interpreted, &peak); + correct = correct && cbm_msb_test_value_live_bytes() == 0; + fprintf(stderr, + "msbuild shared closure mode=%d records=%d projects=%d interpreted=%llu " + "work=%llu hit_work=%llu peak=%llu live=%llu semantics=%d\n", + mode, records, projects, (unsigned long long)interpreted, + (unsigned long long)work[mode][size][count], + (unsigned long long)hits[mode][size][count], (unsigned long long)peak, + (unsigned long long)cbm_msb_test_value_live_bytes(), correct); + cbm_msb_free(m); + } + } + } + free(xml); + free(definitions); + ASSERT_TRUE(correct); + bool bounded = true; + for (int mode = 0; mode < 4; mode++) { + for (int count = 0; count < 2; count++) { + ASSERT_TRUE(hits[mode][0][count] > 0); + double ratio = (double)hits[mode][1][count] / (double)hits[mode][0][count]; + fprintf(stderr, + "msbuild shared closure shared-size hit ratio mode=%d projects=%d ratio=%.3f\n", + mode, 16 << count, ratio); + bounded = bounded && ratio >= 0.9 && ratio <= 1.1; + } + /* Include construction and teardown: sharing must not shift the same + * product cost into retained-state copying or cleanup. */ + ASSERT_TRUE(work[mode][1][0] > work[mode][0][0]); + ASSERT_TRUE(work[mode][1][1] > work[mode][0][1]); + uint64_t small = work[mode][1][0] - work[mode][0][0]; + uint64_t large = work[mode][1][1] - work[mode][0][1]; + double ratio = (double)large / (double)small; + fprintf(stderr, "msbuild shared closure extra-size ratio mode=%d ratio=%.3f\n", mode, + ratio); + bounded = bounded && ratio >= 0.9 && ratio <= 1.2; + } + ASSERT_TRUE(bounded); + PASS(); +} + +/* Grow the number of shared files, with fixed tiny outputs. Component lookup, + * item traversal, dependency composition and cleanup must not replay the chain. */ +TEST(doc_mentions_msbuild_shared_file_work) { + uint64_t work[2][2][2] = {{{0}}}; + uint64_t hits[2][2][2] = {{{0}}}; + bool correct = true; + bool storage_bounded = true; + for (int mode = 0; mode < 2; mode++) { + const char *suffix = mode == 0 ? "props" : "targets"; + for (int size = 0; size < 2; size++) { + int files = 128 << size; + for (int count = 0; count < 2; count++) { + int projects = 16 << count; + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + size_t source_bytes = 0; + for (int i = 0; i < files; i++) { + char path[96]; + char next[96] = ""; + char xml[512]; + snprintf(path, sizeof(path), "Shared/Chain%04d.props", i); + if (i + 1 < files) { + snprintf(next, sizeof(next), "", + i + 1); + } + int n = snprintf(xml, sizeof(xml), + "Shared%04d" + "%s", + i, i, i, next); + ASSERT_TRUE(n > 0 && (size_t)n < sizeof(xml)); + source_bytes += (size_t)n; + ASSERT_TRUE(dm_msb_add_xml(m, path, xml)); + } + char root_xml[1024]; + int n = snprintf(root_xml, sizeof(root_xml), + "" + "" + "" + "" + "", + files - 1); + ASSERT_TRUE(n > 0 && (size_t)n < sizeof(root_xml)); + source_bytes += (size_t)n; + ASSERT_TRUE(dm_msb_add_xml(m, "Shared/Root.props", root_xml)); + const char wrapper[] = + ""; + const char project_xml[] = ""; + for (int i = 0; i <= projects; i++) { + char root[96]; + char project[96]; + snprintf(root, sizeof(root), "src/R%02d/Directory.Build.%s", i, suffix); + snprintf(project, sizeof(project), "src/R%02d/P%02d.csproj", i, i); + ASSERT_TRUE(dm_msb_add_xml(m, root, wrapper)); + ASSERT_TRUE(dm_msb_add_xml(m, project, project_xml)); + source_bytes += sizeof(wrapper) + sizeof(project_xml) - 2; + } + cbm_msb_test_cost_reset(); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + uint64_t captured = 0; + for (int i = 0; i <= projects; i++) { + char path[96]; + char name[32]; + char last[32]; + snprintf(path, sizeof(path), "src/R%02d/P%02d.csproj", i, i); + snprintf(name, sizeof(name), "P%02d", i); + snprintf(last, sizeof(last), "Shared%04d", files - 1); + cbm_msb_result_t actual = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, path, &actual)); + cbm_msb_using_t expected_usings[] = { + {.kind = 'a', .alias = "Project", .target = name}, + {.kind = 'a', .alias = "First", .target = "Shared0000"}, + {.kind = 'a', .alias = "Last", .target = last}, + }; + cbm_msb_result_t expected = {.usings = expected_usings, .count = 3}; + correct = dm_msb_same(&actual, &expected) && correct; + cbm_msb_result_free(&actual); + if (i == 0) { + captured = cbm_msb_test_work(); + } + } + hits[mode][size][count] = cbm_msb_test_work() - captured; + cbm_msb_eval_context_free(context); + work[mode][size][count] = cbm_msb_test_work(); + uint64_t interpreted = 0; + uint64_t peak = 0; + cbm_msb_test_cost(&interpreted, &peak); + correct = correct && cbm_msb_test_value_live_bytes() == 0; + /* Allow fixed per-file metadata and bounded-depth state nodes, + * but reject a default64KiB retained arena for every tiny file. */ + uint64_t bound = (uint64_t)source_bytes * 64 + 1024 * 1024; + storage_bounded = storage_bounded && peak <= bound; + fprintf(stderr, + "msbuild shared files mode=%d files=%d projects=%d interpreted=%llu " + "work=%llu hit_work=%llu peak=%llu bound=%llu live=%llu semantics=%d\n", + mode, files, projects, (unsigned long long)interpreted, + (unsigned long long)work[mode][size][count], + (unsigned long long)hits[mode][size][count], (unsigned long long)peak, + (unsigned long long)bound, + (unsigned long long)cbm_msb_test_value_live_bytes(), correct); + cbm_msb_free(m); + } + } + } + ASSERT_TRUE(correct); + bool bounded = true; + for (int mode = 0; mode < 2; mode++) { + for (int count = 0; count < 2; count++) { + ASSERT_TRUE(hits[mode][0][count] > 0); + double ratio = (double)hits[mode][1][count] / (double)hits[mode][0][count]; + fprintf(stderr, "msbuild shared files size hit ratio mode=%d projects=%d ratio=%.3f\n", + mode, 16 << count, ratio); + bounded = bounded && ratio >= 0.9 && ratio <= 1.1; + } + ASSERT_TRUE(work[mode][1][0] > work[mode][0][0]); + ASSERT_TRUE(work[mode][1][1] > work[mode][0][1]); + uint64_t small = work[mode][1][0] - work[mode][0][0]; + uint64_t large = work[mode][1][1] - work[mode][0][1]; + double ratio = (double)large / (double)small; + fprintf(stderr, "msbuild shared files extra-size ratio mode=%d ratio=%.3f\n", mode, ratio); + bounded = bounded && ratio >= 0.9 && ratio <= 1.2; + } + ASSERT_TRUE(bounded && storage_bounded); + PASS(); +} + +/* Shared components must preserve import order and history across distinct + * wrappers; the ordinary evaluator remains the exact semantic oracle. */ +TEST(doc_mentions_msbuild_shared_closure_isolation) { + const dm_project_file_t common[] = { + {"Shared/State.props", + "Shared$(Before)" + "" + "" + "" + "" + "" + "" + "" + ""}, + {"Shared/Leaf.props", "$(Value).Leaf" + "" + "" + "" + ""}, + {"Shared/Diamond.props", + "" + ""}, + {"Shared/Cycle.props", "" + ""}, + }; + bool correct = true; + for (int mode = 0; mode < 2; mode++) { + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + for (size_t i = 0; i < sizeof(common) / sizeof(common[0]); i++) { + ASSERT_TRUE(dm_msb_add_xml(m, common[i].rel_path, common[i].xml)); + } + const char *suffix = mode == 0 ? "props" : "targets"; + char paths[5][96]; + for (int i = 0; i < 5; i++) { + char root[96]; + char wrapper[2048]; + char project[512]; + char name = (char)('A' + i); + snprintf(root, sizeof(root), "Roots/%c/Directory.Build.%s", name, suffix); + snprintf(paths[i], sizeof(paths[i]), "Roots/%c/%c.csproj", name, name); + snprintf(wrapper, sizeof(wrapper), + "%sPre.%c" + "%s" + "" + "Post.%cAfter.%c", + i == 2 ? "Changed" : "Same", name, + i == 3 ? "" + : "", + name, name); + snprintf(project, sizeof(project), + "Project.%c" + "%s", + name, i == 4 ? "" : ""); + ASSERT_TRUE(dm_msb_add_xml(m, root, wrapper)); + ASSERT_TRUE(dm_msb_add_xml(m, paths[i], project)); + } + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + cbm_msb_result_t held = {0}; + cbm_msb_result_t expected = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, paths[0], &held)); + ASSERT_TRUE(cbm_msb_eval(m, paths[0], &expected)); + correct = correct && dm_msb_same(&held, &expected) && held.count == 9 && + dm_has_using(&held, 'a', "Shared.Leaf") && + dm_has_using(&held, 'a', "File.State") && dm_has_using(&held, 'a', "Leaf.Leaf") && + dm_has_using(&held, 'a', mode == 0 ? "Project.A" : "Post.A"); + const int order[] = {0, 1, 2, 0, 3, 1, 4, 0}; + for (size_t i = 0; i < sizeof(order) / sizeof(order[0]); i++) { + correct = dm_msb_context_matches(context, m, paths[order[i]]) && correct; + } + ASSERT_TRUE(dm_msb_add_xml(m, "Shared/Later.props", + "")); + for (int i = 0; i < 5; i++) { + correct = dm_msb_context_matches(context, m, paths[i]) && correct; + } + cbm_msb_eval_context_free(context); + cbm_msb_free(m); + correct = correct && dm_msb_same(&held, &expected) && cbm_msb_test_value_live_bytes() == 0; + cbm_msb_result_free(&held); + cbm_msb_result_free(&expected); + } + /* Preserve the parent's read of entry X while masking the child's X read + * after X is overwritten. Y escapes from the child and must join the input + * requirements. A child-state witness cannot certify the earlier X read. */ + for (int mode = 0; mode < 2; mode++) { + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE( + dm_msb_add_xml(m, "Common/Union.props", + "$(X)Forced" + "" + "" + "" + "")); + ASSERT_TRUE(dm_msb_add_xml(m, "Common/Child.props", + "$(X)$(Y)" + "")); + const char *xs[] = {"Old", "Forced", "Old", "Old"}; + const char *ys[] = {"One", "One", "Two", NULL}; + const char *y_records[] = {"One", "One", "Two", ""}; + char paths[4][96]; + for (int i = 0; i < 4; i++) { + char root[96]; + char xml[512]; + snprintf(root, sizeof(root), "Roots/R%d/Directory.Build.%s", i, + mode == 0 ? "props" : "targets"); + snprintf(paths[i], sizeof(paths[i]), "Roots/R%d/P.csproj", i); + snprintf(xml, sizeof(xml), + "%s%s" + "", + xs[i], y_records[i]); + ASSERT_TRUE(dm_msb_add_xml(m, root, xml)); + ASSERT_TRUE(dm_msb_add_xml(m, paths[i], "")); + } + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + const int order[] = {0, 1, 2, 0, 3, 0}; + for (size_t i = 0; i < sizeof(order) / sizeof(order[0]); i++) { + int which = order[i]; + cbm_msb_result_t actual = {0}; + cbm_msb_result_t reference = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, paths[which], &actual)); + ASSERT_TRUE(cbm_msb_eval(m, paths[which], &reference)); + correct = dm_msb_same(&actual, &reference) && correct; + bool found[3] = {false, false, false}; + for (int j = 0; j < actual.count; j++) { + const cbm_msb_using_t *u = &actual.usings[j]; + found[0] = found[0] || (u->kind == 'a' && strcmp(u->alias, "Snapshot") == 0 && + strcmp(u->target, xs[which]) == 0); + found[1] = found[1] || (u->kind == 'a' && strcmp(u->alias, "FromX") == 0 && + strcmp(u->target, "Forced") == 0); + found[2] = + found[2] || (ys[which] && u->kind == 'a' && strcmp(u->alias, "FromY") == 0 && + strcmp(u->target, ys[which]) == 0); + } + correct = correct && found[0] && found[1] && found[2] == (ys[which] != NULL) && + actual.count == (ys[which] ? 3 : 2) && actual.open == (ys[which] == NULL) && + actual.unevaluable == (ys[which] ? 0 : 1) && actual.outside == 0; + cbm_msb_result_free(&actual); + cbm_msb_result_free(&reference); + } + cbm_msb_eval_context_free(context); + cbm_msb_free(m); + correct = correct && cbm_msb_test_value_live_bytes() == 0; + } + /* Same-length call-local values exercise revision identity independently + * of value length. Repeat callers after their temporary state is freed. */ + for (int mode = 0; mode < 2; mode++) { + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE( + dm_msb_add_xml(m, "Read.props", + "$(Input)" + "" + "")); + const char *values[] = {"Old", "New", "Alt"}; + char paths[3][32]; + for (int i = 0; i < 3; i++) { + char xml[512]; + snprintf(paths[i], sizeof(paths[i]), "P%d.csproj", i); + snprintf(xml, sizeof(xml), + "Tmp%s" + "%s", + values[i], mode == 0 ? "" : ""); + ASSERT_TRUE(dm_msb_add_xml(m, paths[i], xml)); + } + if (mode != 0) { + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.targets", + "")); + } + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + const int order[] = {0, 1, 0, 2, 1, 0}; + for (size_t i = 0; i < sizeof(order) / sizeof(order[0]); i++) { + int which = order[i]; + cbm_msb_result_t actual = {0}; + cbm_msb_result_t reference = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, paths[which], &actual)); + ASSERT_TRUE(cbm_msb_eval(m, paths[which], &reference)); + cbm_msb_using_t expected_using = { + .kind = 'a', .alias = "Value", .target = values[which]}; + cbm_msb_result_t expected = {.usings = &expected_using, .count = 1}; + correct = + dm_msb_same(&actual, &reference) && dm_msb_same(&actual, &expected) && correct; + cbm_msb_result_free(&actual); + cbm_msb_result_free(&reference); + } + cbm_msb_eval_context_free(context); + cbm_msb_free(m); + correct = correct && cbm_msb_test_value_live_bytes() == 0; + } + ASSERT_TRUE(correct); + PASS(); +} + +/* Node acquisition failures must take the ordinary OOM path. The retained + * component remains usable and every transient owner is released. */ +TEST(doc_mentions_msbuild_components_failure) { + char xml[16384]; + size_t w = (size_t)snprintf(xml, sizeof(xml), ""); + for (int i = 0; i < 64; i++) { + w += + (size_t)snprintf(xml + w, sizeof(xml) - w, "$(I%03d).N%03d", i, i, i, i); + } + w += (size_t)snprintf(xml + w, sizeof(xml) - w, + "" + "" + "" + ""); + ASSERT_TRUE(w < sizeof(xml)); + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(dm_msb_add_xml(m, "Shared.props", xml)); + ASSERT_TRUE(dm_msb_add_xml(m, "Child.props", + "$(K063)" + "Forced")); + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.targets", + "Before" + "" + "After")); + for (int project = 0; project < 3; project++) { + char path[32]; + snprintf(path, sizeof(path), "%c.csproj", 'A' + project); + w = (size_t)snprintf(xml, sizeof(xml), "P%d", + project); + for (int i = 0; i < 64; i++) { + w += (size_t)snprintf(xml + w, sizeof(xml) - w, "%s", i, + project == 2 ? "BBBB" : "AAAA", i); + } + w += (size_t)snprintf(xml + w, sizeof(xml) - w, ""); + ASSERT_TRUE(w < sizeof(xml)); + ASSERT_TRUE(dm_msb_add_xml(m, path, xml)); + } + const struct { + cbm_msb_node_fail_operation_t operation; + int nth; + bool warm; + } cases[] = { + {CBM_MSB_NODE_FAIL_CAPTURE, 1, false}, {CBM_MSB_NODE_FAIL_CAPTURE, 5, false}, + {CBM_MSB_NODE_FAIL_OVERLAY, 1, false}, {CBM_MSB_NODE_FAIL_OVERLAY, 5, false}, + {CBM_MSB_NODE_FAIL_DEPENDENCY, 1, false}, {CBM_MSB_NODE_FAIL_DEPENDENCY, 5, false}, + {CBM_MSB_NODE_FAIL_APPLY, 1, true}, {CBM_MSB_NODE_FAIL_APPLY, 5, true}, + }; + int failures = 0; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + size_t before = cbm_mem_tracked_live_bytes(); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + if (cases[i].warm) { + failures += !dm_msb_context_matches(context, m, "A.csproj"); + } + cbm_msb_test_fail_node_alloc(cases[i].operation, cases[i].nth); + cbm_msb_result_t actual = {0}; + bool ok = cbm_msb_eval_context_eval(context, "A.csproj", &actual); + bool consumed = cbm_msb_test_node_alloc_failed(); + failures += + ok || !consumed || actual.mem != NULL || actual.usings != NULL || actual.count != 0; + cbm_msb_result_free(&actual); + cbm_msb_test_fail_node_alloc(CBM_MSB_NODE_FAIL_NONE, 0); + const char *order[] = {"A.csproj", "B.csproj", "C.csproj", "A.csproj"}; + for (size_t j = 0; j < sizeof(order) / sizeof(order[0]); j++) { + failures += !dm_msb_context_matches(context, m, order[j]); + } + cbm_msb_eval_context_free(context); + size_t after = cbm_mem_tracked_live_bytes(); + failures += after != before || cbm_msb_test_value_live_bytes() != 0; + fprintf(stderr, + "msbuild components fault operation=%d nth=%d consumed=%d " + "before=%llu after=%llu failures=%d\n", + (int)cases[i].operation, cases[i].nth, consumed, (unsigned long long)before, + (unsigned long long)after, failures); + } + cbm_msb_free(m); + ASSERT_EQ(failures, 0); + PASS(); +} + +/* Revision witnesses must remain exact when short-lived state reuses slots. + * The observations certify that this fixture actually reaches those paths. */ +TEST(doc_mentions_msbuild_components_revision) { + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.targets", + "" + "")); + ASSERT_TRUE(dm_msb_add_xml(m, "Constants.props", + "Stable.Fixed" + "")); + ASSERT_TRUE( + dm_msb_add_xml(m, "Read.props", + "$(Input)$(Stable)" + "" + "" + "")); + const char *values[] = {"AAAA", "BBBB", "CCCC"}; + char paths[3][32]; + for (int i = 0; i < 3; i++) { + char xml[512]; + snprintf(paths[i], sizeof(paths[i]), "P%d.csproj", i); + snprintf(xml, sizeof(xml), + "Temp%s" + "Unchanged", + values[i]); + ASSERT_TRUE(dm_msb_add_xml(m, paths[i], xml)); + } + size_t before = cbm_mem_tracked_live_bytes(); + cbm_msb_test_cost_reset(); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + const int order[] = {0, 0, 1, 0, 2, 1, 0}; + bool correct = true; + for (size_t i = 0; i < sizeof(order) / sizeof(order[0]); i++) { + int which = order[i]; + cbm_msb_result_t actual = {0}; + cbm_msb_result_t reference = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, paths[which], &actual)); + ASSERT_TRUE(cbm_msb_eval(m, paths[which], &reference)); + cbm_msb_using_t expected_usings[] = { + {.kind = 'a', .alias = "Value", .target = values[which]}, + {.kind = 'a', .alias = "Constant", .target = "Stable.Fixed"}, + }; + cbm_msb_result_t expected = {.usings = expected_usings, .count = 2}; + correct = dm_msb_same(&actual, &reference) && dm_msb_same(&actual, &expected) && correct; + cbm_msb_result_free(&actual); + cbm_msb_result_free(&reference); + } + cbm_msb_state_test_stats_t stats = {0}; + cbm_msb_test_state_stats(&stats); + cbm_msb_eval_context_free(context); + size_t after = cbm_mem_tracked_live_bytes(); + correct = correct && before == after && cbm_msb_test_value_live_bytes() == 0; + cbm_msb_free(m); + fprintf(stderr, + "msbuild components revisions allocations=%llu reused=%llu witnessed_reused=%llu " + "same_length=%llu witness_skips=%llu revision_errors=%llu semantics=%d\n", + (unsigned long long)stats.allocations, (unsigned long long)stats.slot_reuses, + (unsigned long long)stats.witnessed_slot_reuses, + (unsigned long long)stats.same_length_value_changes, + (unsigned long long)stats.witness_skips, (unsigned long long)stats.revision_errors, + correct); + ASSERT_TRUE(correct); + ASSERT_TRUE(stats.allocations > 0 && stats.slot_reuses > 0 && stats.witnessed_slot_reuses > 0 && + stats.same_length_value_changes > 0 && stats.witness_skips > 0 && + stats.revision_errors == 0); + PASS(); +} + +/* Mixed presence constraints must not use the all-absent empty-range shortcut. + * Insertion order places A/B at seen keys 4/5, with all other visited files + * outside that aligned range. Warm Shared sees A present and B absent; Cold + * reaches the same retained component with the entire range empty. */ +TEST(doc_mentions_msbuild_components_mixed_absence) { + const dm_project_file_t files[] = { + {"Warm.csproj", "" + ""}, + {"Cold.csproj", ""}, + {"Padding.props", ""}, + {"A.props", "" + ""}, + {"B.props", "" + ""}, + {"Shared.props", "" + ""}, + }; + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + for (size_t i = 0; i < sizeof(files) / sizeof(files[0]); i++) { + ASSERT_TRUE(dm_msb_add_xml(m, files[i].rel_path, files[i].xml)); + } + cbm_msb_using_t expected_usings[] = { + {.kind = 'n', .alias = "", .target = "A.Effect"}, + {.kind = 'n', .alias = "", .target = "B.Effect"}, + }; + cbm_msb_result_t expected = {.usings = expected_usings, .count = 2}; + size_t before = cbm_mem_tracked_live_bytes(); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + bool correct = true; + const char *order[] = {"Warm.csproj", "Cold.csproj", "Cold.csproj", "Warm.csproj", + "Cold.csproj"}; + for (size_t i = 0; i < sizeof(order) / sizeof(order[0]); i++) { + cbm_msb_result_t actual = {0}; + cbm_msb_result_t reference = {0}; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, order[i], &actual)); + ASSERT_TRUE(cbm_msb_eval(m, order[i], &reference)); + correct = dm_msb_same(&actual, &reference) && dm_msb_same(&actual, &expected) && correct; + cbm_msb_result_free(&actual); + cbm_msb_result_free(&reference); + } + cbm_msb_eval_context_free(context); + size_t after = cbm_mem_tracked_live_bytes(); + correct = correct && before == after && cbm_msb_test_value_live_bytes() == 0; + cbm_msb_free(m); + fprintf(stderr, "msbuild components mixed absence semantics=%d before=%llu after=%llu\n", + correct, (unsigned long long)before, (unsigned long long)after); + ASSERT_TRUE(correct); + PASS(); +} + +TEST(doc_mentions_msbuild_prefix_work) { + enum { CAP = 131072 }; + char *xml = malloc(CAP); + ASSERT_NOT_NULL(xml); + uint64_t work[2][2] = {{0}}; + bool correct = true; + for (int size = 0; size < 2; size++) { + int records = 512 << size; + for (int count = 0; count < 2; count++) { + int projects = 16 << count; + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + size_t w = (size_t)snprintf(xml, CAP, ""); + for (int i = 0; i < records; i++) { + w += (size_t)snprintf(xml + w, CAP - w, "Shared%04d", i, i, i); + } + w += (size_t)snprintf( + xml + w, CAP - w, + "" + "" + "" + "", + records - 1); + ASSERT_TRUE(w < CAP); + ASSERT_TRUE(dm_msb_add_xml(m, "Shared.props", xml)); + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.props", + "")); + for (int i = 0; i < projects; i++) { + char path[64]; + snprintf(path, sizeof(path), "src/P%02d.csproj", i); + ASSERT_TRUE(dm_msb_add_xml(m, path, "")); + } + cbm_msb_test_cost_reset(); + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + for (int i = 0; i < projects; i++) { + char path[64]; + char project[32]; + char last[32]; + snprintf(path, sizeof(path), "src/P%02d.csproj", i); + snprintf(project, sizeof(project), "P%02d", i); + snprintf(last, sizeof(last), "Shared%04d", records - 1); + cbm_msb_result_t r; + ASSERT_TRUE(cbm_msb_eval_context_eval(context, path, &r)); + correct = correct && r.count == 3 && !r.open && r.unevaluable == 0 && + r.outside == 0 && dm_has_using(&r, 'a', "Shared0000") && + dm_has_using(&r, 'a', last) && dm_has_using(&r, 'a', project); + cbm_msb_result_free(&r); + } + cbm_msb_eval_context_free(context); + work[size][count] = cbm_msb_test_work(); + uint64_t consumed = 0; + uint64_t peak = 0; + cbm_msb_test_cost(&consumed, &peak); + fprintf(stderr, + "msbuild prefix records=%d projects=%d interpreted=%llu work=%llu " + "peak=%llu live=%llu semantics=%d\n", + records, projects, (unsigned long long)consumed, + (unsigned long long)work[size][count], (unsigned long long)peak, + (unsigned long long)cbm_msb_test_value_live_bytes(), correct); + correct = correct && cbm_msb_test_value_live_bytes() == 0; + cbm_msb_free(m); + } + } + free(xml); + ASSERT_TRUE(correct); + bool bounded = true; + for (int size = 0; size < 2; size++) { + ASSERT_TRUE(work[size][0] > 0); + double ratio = (double)work[size][1] / (double)work[size][0]; + fprintf(stderr, "msbuild prefix project ratio records=%d ratio=%.3f\n", 512 << size, ratio); + bounded = bounded && ratio >= 0.9 && ratio <= 1.4; + } + ASSERT_TRUE(bounded); + PASS(); +} + +/* The uncached evaluator is the semantic oracle, including diagnostics and + * ordering. Exercise shared properties and items across project/root changes. */ +TEST(doc_mentions_msbuild_prefix_isolation) { + const dm_project_file_t files[] = { + {"Directory.Build.props", + "Prefix" + "" + "" + "" + "" + ""}, + {"Common.props", "Yes" + ""}, + {"Directory.Build.targets", + "$(Flavor)" + "" + "" + "" + ""}, + {"A.csproj", + "AA" + "Nonempty"}, + {"B.csproj", "B"}, + {"Poison.csproj", "B" + ""}, + {"Poison.props", "HiddenHidden" + ""}, + {"Write/Directory.Build.props", + "" + "Forced$(MSBuildProjectName)" + "" + ""}, + {"Write/A.csproj", ""}, + {"Write/B.csproj", ""}, + {"Read/Directory.Build.props", + "" + "$(MSBuildProjectName)" + "" + ""}, + {"Read/A.csproj", ""}, + {"Read/B.csproj", ""}, + {"Read/A.props", ""}, + {"Read/B.props", ""}, + {"History/Directory.Build.props", + "Seed" + "" + ""}, + {"History/Owned.csproj", "$(V).Again" + ""}, + {"History/Other.csproj", + "Known" + ""}, + {"History/Maybe.props", "Hidden"}, + {"Condition/Directory.Build.props", + "" + "" + "For.A" + "" + "For.B" + ""}, + {"Condition/A.csproj", ""}, + {"Condition/B.csproj", ""}, + {"PoisonSeed/Directory.Build.props", + "Known" + "" + ""}, + {"PoisonSeed/A.csproj", ""}, + {"PoisonSeed/B.csproj", ""}, + {"PoisonSeed/A.props", "Hidden" + ""}, + {"PoisonSeed/B.props", "Hidden" + ""}, + {"Implicit/Directory.Build.props", + "" + "enable"}, + {"Implicit/A.csproj", ""}, + {"Implicit/B.csproj", ""}, + {"Implicit/C.csproj", "disable" + ""}, + }; + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + for (size_t i = 0; i < sizeof(files) / sizeof(files[0]); i++) { + ASSERT_TRUE(dm_msb_add_xml(m, files[i].rel_path, files[i].xml)); + } + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + const char *projects[] = {"A.csproj", + "B.csproj", + "Poison.csproj", + "B.csproj", + "Write/A.csproj", + "Write/B.csproj", + "Read/A.csproj", + "Read/B.csproj", + "History/Owned.csproj", + "History/Other.csproj", + "A.csproj", + "Write/A.csproj", + "B.csproj", + "History/Owned.csproj", + "Write/B.csproj", + "Condition/A.csproj", + "Condition/B.csproj", + "Condition/A.csproj", + "PoisonSeed/A.csproj", + "PoisonSeed/B.csproj", + "PoisonSeed/A.csproj", + "Implicit/A.csproj", + "Implicit/B.csproj", + "Implicit/C.csproj", + "Implicit/A.csproj"}; + for (size_t i = 0; i < sizeof(projects) / sizeof(projects[0]); i++) { + ASSERT_TRUE(dm_msb_context_matches(context, m, projects[i])); + } + ASSERT_TRUE(dm_msb_context_matches(context, m, "A.csproj")); + ASSERT_TRUE(dm_msb_add_xml(m, "Later.props", + "" + "")); + ASSERT_TRUE(dm_msb_context_matches(context, m, "A.csproj")); + ASSERT_TRUE(dm_msb_context_matches(context, m, "B.csproj")); + cbm_msb_eval_context_free(context); + ASSERT_EQ(cbm_msb_test_value_live_bytes(), 0); + cbm_msb_free(m); + PASS(); +} + +/* Construction failure must leave no partial prefix, and a failed project + * must not mutate the already retained prefix. Retry on the same context. */ +TEST(doc_mentions_msbuild_prefix_failure) { + char xml[8192]; + size_t w = (size_t)snprintf(xml, sizeof(xml), ""); + for (int i = 0; i < 64; i++) { + w += (size_t)snprintf(xml + w, sizeof(xml) - w, "Shared%02d", i, i, i); + } + w += (size_t)snprintf(xml + w, sizeof(xml) - w, + "" + ""); + ASSERT_TRUE(w < sizeof(xml)); + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(dm_msb_add_xml(m, "Directory.Build.props", xml)); + ASSERT_TRUE(dm_msb_add_xml(m, "A.csproj", "")); + ASSERT_TRUE(dm_msb_add_xml(m, "B.csproj", + "" + "Local")); + int failures = 0; + for (int mode = 0; mode < 2; mode++) { + cbm_msb_eval_context_t *context = cbm_msb_eval_context_new(m); + ASSERT_NOT_NULL(context); + cbm_msb_test_fail_value_alloc_after(mode == 0 ? 8 : 0); + cbm_msb_test_fail_prop_insert_after(mode == 1 ? 8 : 0); + cbm_msb_result_t r = {0}; + bool ok = cbm_msb_eval_context_eval(context, "A.csproj", &r); + bool consumed = + mode == 0 ? cbm_msb_test_value_alloc_failed() : cbm_msb_test_prop_insert_failed(); + failures += ok || !consumed || r.mem != NULL || r.usings != NULL || r.count != 0 || + cbm_msb_test_value_live_bytes() != 0; + cbm_msb_result_free(&r); + cbm_msb_test_fail_value_alloc_after(0); + cbm_msb_test_fail_prop_insert_after(0); + failures += !dm_msb_context_matches(context, m, "A.csproj"); + uint64_t retained = cbm_msb_test_value_live_bytes(); + cbm_msb_test_fail_value_alloc_after(mode == 0 ? 1 : 0); + cbm_msb_test_fail_prop_insert_after(mode == 1 ? 1 : 0); + ok = cbm_msb_eval_context_eval(context, "B.csproj", &r); + consumed = + mode == 0 ? cbm_msb_test_value_alloc_failed() : cbm_msb_test_prop_insert_failed(); + failures += ok || !consumed || r.mem != NULL || r.usings != NULL || r.count != 0 || + cbm_msb_test_value_live_bytes() != retained; + cbm_msb_result_free(&r); + cbm_msb_test_fail_value_alloc_after(0); + cbm_msb_test_fail_prop_insert_after(0); + failures += !dm_msb_context_matches(context, m, "A.csproj"); + failures += !dm_msb_context_matches(context, m, "B.csproj"); + cbm_msb_eval_context_free(context); + failures += cbm_msb_test_value_live_bytes() != 0; + fprintf(stderr, "msbuild prefix fault mode=%d retained=%llu failures=%d\n", mode, + (unsigned long long)retained, failures); + } + cbm_msb_free(m); + ASSERT_EQ(failures, 0); + PASS(); +} + +TEST(doc_mentions_msbuild_eval_storage) { + enum { CAP = 262144, LONG_LEN = 2048 }; + char *xml = malloc(CAP); + char *long_value = malloc(LONG_LEN + 1); + ASSERT_NOT_NULL(xml); + ASSERT_NOT_NULL(long_value); + memset(long_value, 'A', LONG_LEN); + long_value[LONG_LEN] = '\0'; + int failures = 0; + for (int mode = 0; mode < 3; mode++) { + for (int repeats = 256; repeats <= 512; repeats *= 2) { + size_t w = + (size_t)snprintf(xml, CAP, "%s", long_value); + if (mode == 0) { + for (int i = 0; i < repeats; i++) { + w += (size_t)snprintf( + xml + w, CAP - w, + "$(Long)"); + } + w += (size_t)snprintf( + xml + w, CAP - w, + "$(Last)Short" + "" + ""); + } else if (mode == 1) { + w += (size_t)snprintf(xml + w, CAP - w, "Kept
" + "
"); + } else { + w += (size_t)snprintf(xml + w, CAP - w, "
"); + for (int i = 0; i < repeats; i++) { + w += (size_t)snprintf( + xml + w, CAP - w, + ""); + } + w += (size_t)snprintf(xml + w, CAP - w, + "
"); + } + ASSERT_TRUE(w < CAP); + CBMFileResult *source = dm_extract(xml, CBM_LANG_XML, "App.csproj"); + ASSERT_NOT_NULL(source); + ASSERT_NOT_NULL(source->doc_scope); + size_t blob_bytes = strlen(source->doc_scope); + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(cbm_msb_add(m, "App.csproj", source->doc_scope)); + cbm_free_result(source); + cbm_msb_test_cost_reset(); + cbm_msb_result_t r; + ASSERT_TRUE(cbm_msb_eval(m, "App.csproj", &r)); + uint64_t records = 0; + uint64_t peak = 0; + cbm_msb_test_cost(&records, &peak); + bool semantics = !r.open && r.unevaluable == 0 && r.outside == 0; + if (mode == 0) { + semantics = semantics && r.count == 2 && dm_has_using(&r, 'a', long_value) && + dm_has_using(&r, 'n', "Short"); + for (int i = 0; i < r.count; i++) { + if (r.usings[i].kind == 'a') { + semantics = semantics && strcmp(r.usings[i].alias, "Saved") == 0; + } + } + } else { + semantics = semantics && r.count == 1 && dm_has_using(&r, 'n', "Kept"); + } + uint64_t limit = 131072 + 4 * (uint64_t)blob_bytes; + fprintf(stderr, + "msbuild storage mode=%d repeats=%d blob=%zu records=%llu " + "peak=%llu limit=%llu semantics=%d\n", + mode, repeats, blob_bytes, (unsigned long long)records, + (unsigned long long)peak, (unsigned long long)limit, semantics); + failures += !semantics || !records || !peak || peak > limit; + cbm_msb_result_free(&r); + cbm_msb_free(m); + } + } + free(long_value); + free(xml); + ASSERT_EQ(failures, 0); + PASS(); +} + +/* A growing self-assignment reads the old value before replacing it. A + * failed value allocation returns no partial result and does not poison a + * subsequent independent evaluation on the same input model. */ +TEST(doc_mentions_msbuild_value_allocation) { + enum { LONG_LEN = 2048, CAP = 8192 }; + char long_value[LONG_LEN + 2]; + long_value[0] = 'x'; + memset(long_value + 1, 'A', LONG_LEN); + long_value[LONG_LEN + 1] = '\0'; + char xml[CAP]; + int written = snprintf(xml, sizeof(xml), + "x$(V)" + "$(V)%s$(V)Short" + "" + "" + "" + "", + long_value + 1); + ASSERT_TRUE(written > 0 && written < CAP); + const dm_project_file_t files[] = { + {"App.csproj", xml}, + {"More.props", "$(V).Imported" + ""}, + }; + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + for (int i = 0; i < 2; i++) { + CBMFileResult *source = dm_extract(files[i].xml, CBM_LANG_XML, files[i].rel_path); + ASSERT_NOT_NULL(source); + ASSERT_NOT_NULL(source->doc_scope); + ASSERT_TRUE(cbm_msb_add(m, files[i].rel_path, source->doc_scope)); + cbm_free_result(source); + } + const int fault_after[] = {0, 1, 0, 8, 0, 0, 0}; + const int insert_after[] = {0, 0, 0, 0, 0, 6, 0}; + int failures = 0; + for (size_t i = 0; i < sizeof(fault_after) / sizeof(fault_after[0]); i++) { + cbm_msb_test_fail_value_alloc_after(fault_after[i]); + cbm_msb_test_fail_prop_insert_after(insert_after[i]); + cbm_msb_result_t r; + bool ok = cbm_msb_eval(m, "App.csproj", &r); + bool consumed = cbm_msb_test_value_alloc_failed(); + bool insert_failed = cbm_msb_test_prop_insert_failed(); + bool injected = fault_after[i] || insert_after[i]; + failures += consumed != (fault_after[i] > 0) || insert_failed != (insert_after[i] > 0) || + cbm_msb_test_value_live_bytes() != 0; + if (injected) { + failures += ok || r.mem != NULL || r.usings != NULL || r.count != 0; + } else { + failures += !ok || consumed || r.open || r.count != 3 || !dm_has_using(&r, 'a', "x") || + !dm_has_using(&r, 'a', long_value) || + !dm_has_using(&r, 'n', "Short.Imported"); + for (int k = 0; k < r.count; k++) { + if (r.usings[k].kind == 'a') { + const char *alias = + strcmp(r.usings[k].target, "x") == 0 ? "Saved" : "LongSaved"; + failures += strcmp(r.usings[k].alias, alias) != 0; + } + } + } + fprintf(stderr, + "msbuild value allocation nth=%d consumed=%d insert=%d/%d " + "live=%llu success=%d failures=%d\n", + fault_after[i], consumed, insert_after[i], insert_failed, + (unsigned long long)cbm_msb_test_value_live_bytes(), ok, failures); + cbm_msb_result_free(&r); + } + cbm_msb_test_fail_value_alloc_after(0); + cbm_msb_test_fail_prop_insert_after(0); + cbm_msb_free(m); + ASSERT_EQ(failures, 0); + PASS(); +} + +/* Scratch resets must not change copied properties, output strings, or the + * poisoning caused by an unknown import (including an imported child). */ +TEST(doc_mentions_msbuild_value_lifetimes) { + const dm_project_file_t files[] = { + {"App.csproj", + "" + "One;Two$(Seed)Changed" + "ReadyAliasKept" + "$(Empty)Tail" + "" + "Recovered$(Empty)" + "$(Empty)Again" + "" + "" + "" + "" + "" + "" + "" + ""}, + {"Unknown.props", "Changed" + ""}, + {"Child.props", + "Unknown.ValuePoison" + ""}, + }; + cbm_msb_result_t r; + ASSERT_TRUE(dm_msb_eval(files, 3, "App.csproj", &r)); + ASSERT_EQ(r.count, 6); + ASSERT_TRUE(dm_has_using(&r, 'a', "One")); + ASSERT_TRUE(dm_has_using(&r, 'a', "Two")); + ASSERT_TRUE(dm_has_using(&r, 's', "Static.Kept")); + ASSERT_TRUE(dm_has_using(&r, 'n', "Tail.Kept")); + ASSERT_TRUE(dm_has_using(&r, 'n', "Tail")); + ASSERT_TRUE(dm_has_using(&r, 'n', "Again")); + for (int i = 0; i < r.count; i++) { + if (r.usings[i].kind == 'a') { + ASSERT_STR_EQ(r.usings[i].alias, "AliasKept"); + } + } + ASSERT_TRUE(r.open); + ASSERT_EQ(r.unevaluable, 3); + ASSERT_EQ(r.outside, 0); + cbm_msb_result_free(&r); + PASS(); +} + +TEST(doc_mentions_msbuild_usings) { + const dm_project_file_t files[] = { + {"Directory.Build.props", "\n" + " web\n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + "\n"}, + {"src/App/App.csproj", + "\xEF\xBB\xBF\r\n" + "\r\n" + "\r\n" + " enable\r\n" + " \r\n" + " \r\n" + " \r\n" + " \r\n" + " \r\n" + " \r\n" + " \r\n" + " \r\n" + " \r\n" + "\r\n"}, + }; + cbm_msb_result_t r; + ASSERT_TRUE(dm_msb_eval(files, 2, "src/App/App.csproj", &r)); + ASSERT_TRUE(dm_has_using(&r, 'n', "System")); /* ImplicitUsings: the SDK default set */ + ASSERT_TRUE(dm_has_using(&r, 'n', "Acme.Extra")); /* */ + ASSERT_TRUE(dm_has_using(&r, 'n', "From.Props")); /* Directory.Build.props, condition true */ + ASSERT_FALSE(dm_has_using(&r, 'n', "System.Linq")); /* */ + ASSERT_FALSE(dm_has_using(&r, 'n', "Never.Here")); /* condition false */ + /* Static="true" names a type and Alias an alias: neither is a namespace */ + ASSERT_FALSE(dm_has_using(&r, 'n', "Acme.Static")); + ASSERT_TRUE(dm_has_using(&r, 's', "Acme.Static")); + ASSERT_FALSE(dm_has_using(&r, 'n', "Acme.Aliased")); + ASSERT_TRUE(dm_has_using(&r, 'a', "Acme.Aliased")); + /* an Include is a ';'-separated list */ + ASSERT_TRUE(dm_has_using(&r, 'n', "Acme.One")); + ASSERT_TRUE(dm_has_using(&r, 'n', "Acme.Two")); + ASSERT_FALSE(dm_has_using(&r, 'n', "Acme.One; Acme.Two ;;")); + /* comments, other items and targets are no usings */ + ASSERT_FALSE(dm_has_using(&r, 'n', "In.A.Comment")); + ASSERT_FALSE(dm_has_using(&r, 'n', "Not.A.Using")); + ASSERT_FALSE(dm_has_using(&r, 'n', "In.A.Target")); + /* everything was evaluated: nothing is open */ + ASSERT_FALSE(r.open); + ASSERT_EQ(r.unevaluable, 0); + ASSERT_EQ(r.outside, 0); + cbm_msb_result_free(&r); + PASS(); +} + +/* : a path is relative to the importing file; the two directory + * properties are absolute and are not joined to it again; what the index + * does not hold is not imported, and is counted. */ +TEST(doc_mentions_msbuild_imports) { + const dm_project_file_t files[] = { + {"src/Directory.Build.props", + "\n" + " $(MSBuildThisFileDirectory)eng/\n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + "\n"}, + {"src/eng/ByThisFile.props", + "\n" + " \n" /* itself: taken once */ + "\n"}, + {"src/eng/ByProperty.props", + "\n"}, + {"src/eng/Relative.props", + "\n"}, + {"src/eng/NotTaken.props", + "\n"}, + {"Up.props", "\n"}, + {"src/App/App.csproj", "\n" + " \n" + "\n"}, + {"src/App/Local.props", + "\n"}, + }; + cbm_msb_result_t r; + ASSERT_TRUE(dm_msb_eval(files, 8, "src/App/App.csproj", &r)); + ASSERT_TRUE(dm_has_using(&r, 'n', "By.ThisFile")); + ASSERT_TRUE(dm_has_using(&r, 'n', "By.Property")); + ASSERT_TRUE(dm_has_using(&r, 'n', "By.Relative")); + ASSERT_TRUE(dm_has_using(&r, 'n', "By.Up")); + ASSERT_TRUE(dm_has_using(&r, 'n', "By.ProjectDir")); + ASSERT_FALSE(dm_has_using(&r, 'n', "Not.Taken")); /* its group's condition is false */ + /* the path through an unknown property and the wildcard */ + ASSERT_EQ(r.unevaluable, 2); + /* above the repository, absolute, and not in the index */ + ASSERT_EQ(r.outside, 3); + ASSERT_FALSE(r.open); /* an import that cannot be followed leaves nothing open */ + cbm_msb_result_free(&r); + PASS(); +} + +/* One condition over fixed properties: the usings it lets through. */ +static bool dm_cond_using(const char *cond, cbm_msb_result_t *out) { + char xml[1024]; + snprintf(xml, sizeof(xml), + "\n" + " \n" + " xxytrue\n" + " x 1\n" + " \n" + " \n" + " \n" + " \n" + "\n", + cond); + const dm_project_file_t files[] = {{"P.csproj", xml}}; + return dm_msb_eval(files, 1, "P.csproj", out); +} + +/* A condition is true, false, or not known. What is not known is never + * taken for one of the two: the item is not applied, the project's usings + * are open, and the count says so. */ +TEST(doc_mentions_msbuild_conditions) { + static const struct { + const char *cond; + char want; /* T applied, F not applied, U not known */ + } cases[] = { + {"'$(A)' == 'x'", 'T'}, + {"'$(A)' == 'X'", 'T'}, /* comparison ignores case */ + {"'$(A)' != 'x'", 'F'}, + {"'$(A)' == '$(B)'", 'T'}, /* both sides are expanded */ + {"'$(A)' == '$(C)'", 'F'}, + {"$(A) == x", 'T'}, /* unquoted operands */ + {"'$(A)-$(C)' == 'x-y'", 'T'}, + {"'$(Empty)' == ''", 'T'}, + {"$(Yes)", 'T'}, + {"!$(Yes)", 'F'}, + {"'$(Yes)' == 'on'", 'T'}, /* booleans compare as booleans */ + {"'$(One)' == '1.0'", 'T'}, /* numbers as numbers */ + {"'$(One)' == '2'", 'F'}, + /* `and` binds tighter than `or`, parentheses group */ + {"'e' == 'e' or 'a' == 'b' and 'c' == 'd'", 'T'}, + {"('e' == 'e' or 'a' == 'b') and 'c' == 'd'", 'F'}, + {"'a' == 'a' or ('b' == 'c' and 'd' == 'e')", 'T'}, + {"!('a' == 'b')", 'T'}, + /* not known: a property no file sets ... */ + {"'$(Undefined)' == ''", 'U'}, + {"'$(Undefined)' != 'v'", 'U'}, + /* ... unless the other operand decides it */ + {"'a' == 'b' and '$(Undefined)' == 'v'", 'F'}, + {"'a' == 'a' or '$(Undefined)' == 'v'", 'T'}, + {"'a' == 'a' and '$(Undefined)' == 'v'", 'U'}, + /* property functions, item lists, function calls, an order */ + {"'$(A.StartsWith('x'))' == 'true'", 'U'}, + {"'@(Compile)' == ''", 'U'}, + {"Exists('x.props')", 'U'}, + {"'a' == 'b' and Exists('x.props')", 'F'}, + {"'$(One)' > '0'", 'U'}, + /* white space a property's element was written with */ + {"'$(Spaced)' == 'x'", 'U'}, + {"'$(Spaced)' == 'y'", 'F'}, + /* no condition this reader can parse */ + {"'a' == ", 'U'}, + {"('a' == 'a'", 'U'}, + {"'a' == 'a' 'b'", 'U'}, + }; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + cbm_msb_result_t r; + if (!dm_cond_using(cases[i].cond, &r)) { + printf(" [%s]: no evaluation\n", cases[i].cond); + FAIL("condition"); + } + bool applied = dm_has_using(&r, 'n', "Under.Condition"); + char got = applied ? 'T' : ((r.open && r.unevaluable > 0) ? 'U' : 'F'); + bool consistent = + applied ? (!r.open && r.unevaluable == 0) : (r.open == (r.unevaluable > 0)); + cbm_msb_result_free(&r); + if (got != cases[i].want || !consistent) { + printf(" [%s]: %c, want %c%s\n", cases[i].cond, got, cases[i].want, + consistent ? "" : " (open and the count disagree)"); + FAIL("condition"); + } + } + PASS(); +} + +/* What an element under an unknown condition could have set is not known + * afterwards, and neither is anything that reads it. */ +TEST(doc_mentions_msbuild_unknown_spreads) { + const dm_project_file_t files[] = { + {"Maybe.props", "\n" + " v\n" + " \n" + "\n"}, + {"P.csproj", + "\n" + " \n" + " plain\n" + " legacy\n" + " yes\n" + " enable\n" + " \n" + " \n" + " \n" + " 1\n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + " \n" + "\n"}, + }; + cbm_msb_result_t r; + ASSERT_TRUE(dm_msb_eval(files, 2, "P.csproj", &r)); + /* Mode is `plain` or `legacy`: nothing that depends on it is applied */ + ASSERT_FALSE(dm_has_using(&r, 'n', "Reads.Poisoned")); + ASSERT_FALSE(dm_has_using(&r, 'n', "plain.Ns")); + ASSERT_FALSE(dm_has_using(&r, 'n', "System")); /* ImplicitUsings: not known either */ + /* a file imported under an unknown condition: its properties and usings */ + ASSERT_FALSE(dm_has_using(&r, 'n', "Reads.Maybe")); + ASSERT_FALSE(dm_has_using(&r, 'n', "From.Maybe")); + /* a property a sets */ + ASSERT_FALSE(dm_has_using(&r, 'n', "Reads.Choose")); + /* what does not depend on any of it stands */ + ASSERT_TRUE(dm_has_using(&r, 'n', "Reads.Sure")); + ASSERT_TRUE(r.open); + ASSERT_GT(r.unevaluable, 5); + cbm_msb_result_free(&r); + PASS(); +} + +/* Growing one group's condition must not copy it once per child import. */ +TEST(doc_mentions_msbuild_import_group_blob_growth) { + enum { IMPORTS = 64, CONDITION_GROWTH = 512 }; + size_t sizes[2] = {0}; + for (int sample = 0; sample < 2; sample++) { + char xml[4096]; + const char *head = "", 2); + used += 2; + for (int i = 0; i < IMPORTS; i++) { + const char *child = ""; + size_t n = strlen(child); + ASSERT_LT(used + n, sizeof(xml)); + memcpy(xml + used, child, n); + used += n; + } + const char *tail = ""; + ASSERT_LT(used + strlen(tail), sizeof(xml)); + memcpy(xml + used, tail, strlen(tail) + 1); + CBMFileResult *r = dm_extract(xml, CBM_LANG_XML, "App.csproj"); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + sizes[sample] = strlen(r->doc_scope); + cbm_free_result(r); + } + ASSERT_GTE(sizes[1], sizes[0]); + ASSERT_LTE(sizes[1] - sizes[0], 2 * CONDITION_GROWTH); + PASS(); +} + +TEST(doc_mentions_msbuild_import_group_roundtrip) { + const dm_project_file_t files[] = { + {"App.csproj", + "" + "" + "" + "" + ""}, + {"A.props", "" + ""}, + {"B.props", ""}, + {"Nested.props", "" + "" + ""}, + {"Skip.props", ""}, + }; + cbm_msb_result_t r; + ASSERT_TRUE(dm_msb_eval(files, 5, "App.csproj", &r)); + ASSERT_TRUE(dm_has_using(&r, 'n', "From.A")); + ASSERT_TRUE(dm_has_using(&r, 'n', "From.B")); + ASSERT_TRUE(dm_has_using(&r, 'n', "From.Nested")); + ASSERT_FALSE(dm_has_using(&r, 'n', "Not.Taken")); + ASSERT_EQ(r.count, 3); + ASSERT_FALSE(r.open); + cbm_msb_result_free(&r); + PASS(); +} + +/* The evaluator side of the same product: an 's condition is + * evaluated once, where the group stands, not again for every import in it. + * Doubling both the imports and the condition's length at most doubles the + * evaluator's work; evaluated per import, it would quadruple. */ +TEST(doc_mentions_msbuild_import_group_condition_work) { + enum { IMPORTS = 100, TERMS = 100, CAP = 64 * 1024 }; + /* Flag is defined (an undefined property is unknown); no term holds */ + static const char term[] = "'$(Flag)' == 'aaaaaaaaaaaaaaaa'"; + uint64_t work[2] = {0}; + char *xml = malloc(CAP); + ASSERT_NOT_NULL(xml); + for (int sample = 0; sample < 2; sample++) { + int imports = IMPORTS << sample; + int terms = TERMS << sample; + size_t used = (size_t)snprintf(xml, CAP, + "b" + ""); + for (int i = 0; i < imports; i++) { + used += (size_t)snprintf(xml + used, CAP - used, ""); + } + used += (size_t)snprintf(xml + used, CAP - used, + "" + ""); + ASSERT_LT(used, (size_t)CAP); + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(dm_msb_add_xml(m, "App.csproj", xml)); + ASSERT_TRUE(dm_msb_add_xml( + m, "X.props", + "")); + cbm_msb_test_cost_reset(); + cbm_msb_result_t r; + ASSERT_TRUE(cbm_msb_eval(m, "App.csproj", &r)); + work[sample] = cbm_msb_test_work(); + ASSERT_TRUE(dm_has_using(&r, 'n', "Kept")); + ASSERT_FALSE(dm_has_using(&r, 'n', "Not.Taken")); + ASSERT_EQ(r.count, 1); + ASSERT_FALSE(r.open); + ASSERT_EQ(r.unevaluable, 0); + cbm_msb_result_free(&r); + cbm_msb_free(m); + } + free(xml); + fprintf(stderr, "msbuild import group condition work: %llu -> %llu\n", + (unsigned long long)work[0], (unsigned long long)work[1]); + ASSERT_GT(work[0], 0); + ASSERT_LTE(work[1], 2 * work[0]); + PASS(); +} + +/* The same for an : its condition is evaluated once for the + * items of a pass, not again for every in it. */ +TEST(doc_mentions_msbuild_item_group_condition_work) { + enum { USINGS = 100, TERMS = 100, CAP = 64 * 1024 }; + static const char term[] = "'$(Flag)' == 'aaaaaaaaaaaaaaaa'"; + uint64_t work[2] = {0}; + char *xml = malloc(CAP); + ASSERT_NOT_NULL(xml); + for (int sample = 0; sample < 2; sample++) { + int usings = USINGS << sample; + int terms = TERMS << sample; + size_t used = (size_t)snprintf(xml, CAP, + "b" + ""); + for (int i = 0; i < usings; i++) { + used += + (size_t)snprintf(xml + used, CAP - used, "", i); + } + used += (size_t)snprintf(xml + used, CAP - used, + "" + ""); + ASSERT_LT(used, (size_t)CAP); + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(dm_msb_add_xml(m, "App.csproj", xml)); + cbm_msb_test_cost_reset(); + cbm_msb_result_t r; + ASSERT_TRUE(cbm_msb_eval(m, "App.csproj", &r)); + work[sample] = cbm_msb_test_work(); + ASSERT_TRUE(dm_has_using(&r, 'n', "Kept")); + ASSERT_EQ(r.count, 1); + ASSERT_FALSE(r.open); + ASSERT_EQ(r.unevaluable, 0); + cbm_msb_result_free(&r); + cbm_msb_free(m); + } + free(xml); + fprintf(stderr, "msbuild item group condition work: %llu -> %llu\n", + (unsigned long long)work[0], (unsigned long long)work[1]); + ASSERT_GT(work[0], 0); + ASSERT_LTE(work[1], 2 * work[0]); + PASS(); +} + +/* MSBuild evaluates an 's condition once, before its imports: + * a property the group's first import sets does not take the second import + * away. */ +TEST(doc_mentions_msbuild_import_group_condition_once) { + const dm_project_file_t files[] = { + /* a property no file defines is unknown (the SDK or the + * environment may set it): Stop is defined first */ + {"App.csproj", "no" + "" + "" + "" + "" + ""}, + {"First.props", "yes" + ""}, + {"Second.props", + ""}, + {"Third.props", + ""}, + }; + cbm_msb_result_t r; + ASSERT_TRUE(dm_msb_eval(files, 4, "App.csproj", &r)); + ASSERT_TRUE(dm_has_using(&r, 'n', "From.First")); + ASSERT_TRUE(dm_has_using(&r, 'n', "From.Second")); + /* a later group with the same text is evaluated where it stands */ + ASSERT_FALSE(dm_has_using(&r, 'n', "From.Third")); + ASSERT_EQ(r.count, 2); + ASSERT_FALSE(r.open); + cbm_msb_result_free(&r); + PASS(); +} + +TEST(doc_mentions_msbuild_legacy_import_blob) { + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(cbm_msb_add(m, "App.csproj", + "cs1\nP\t\t-\n" + "I\t=true\t\t=Take.props\t\n" + "I\t=false\t\t=Skip.props\t\n")); + ASSERT_TRUE(cbm_msb_add(m, "Take.props", "cs1\nP\t\t-\nH\t\nN\t\t=Taken\t\t\t\n")); + ASSERT_TRUE(cbm_msb_add(m, "Skip.props", "cs1\nP\t\t-\nH\t\nN\t\t=Skipped\t\t\t\n")); + cbm_msb_result_t r; + ASSERT_TRUE(cbm_msb_eval(m, "App.csproj", &r)); + ASSERT_TRUE(dm_has_using(&r, 'n', "Taken")); + ASSERT_FALSE(dm_has_using(&r, 'n', "Skipped")); + ASSERT_EQ(r.count, 1); + ASSERT_FALSE(r.open); + cbm_msb_result_free(&r); + cbm_msb_free(m); + PASS(); +} + +TEST(doc_mentions_msbuild_bad_import_group_blob) { + const char *bad[] = { + "J\t\t\t=Take.props\t\n", /* orphan import */ + "B\t=true\nJ\t\t\t=Take.props\t\n", /* missing end */ + "B\t=true\nB\t=false\nE\nE\n", /* nested group */ + "E\n", /* orphan end */ + "B\t=true\nI\t\t\t=Take.props\t\nE\n", /* wrong child */ + "B\t=true\nJ\t=false\t\t=Take.props\t\nE\n", /* inline group condition */ + }; + for (size_t i = 0; i < sizeof(bad) / sizeof(bad[0]); i++) { + char blob[256]; + int n = snprintf(blob, sizeof(blob), "cs1\nP\t\t-\n%s", bad[i]); + ASSERT_GT(n, 0); + ASSERT_LT((size_t)n, sizeof(blob)); + cbm_msb_t *m = cbm_msb_new(); + ASSERT_NOT_NULL(m); + ASSERT_TRUE(cbm_msb_add(m, "App.csproj", blob)); + ASSERT_TRUE(cbm_msb_add(m, "Take.props", "cs1\nP\t\t-\nH\t\nN\t\t=Taken\t\t\t\n")); + cbm_msb_result_t r; + ASSERT_TRUE(cbm_msb_eval(m, "App.csproj", &r)); + ASSERT_TRUE(r.open); + ASSERT_GT(r.unevaluable, 0); + ASSERT_EQ(r.count, 0); + cbm_msb_result_free(&r); + cbm_msb_free(m); + } + PASS(); +} + +/* What the scanner makes of a project file. */ +TEST(doc_mentions_msbuild_blob) { + CBMFileResult *r = dm_extract("\n" + " \n" + " one\ttwo\n" + " & ]]>tail\n" + " \n" + " a\\bAB\n" + " \n" + " \n" + " \n" + " A\n" + " \n" + " \n" + "\n", + CBM_LANG_XML, "src/App.csproj"); + ASSERT_NOT_NULL(r); + const char *s = r->doc_scope; + ASSERT_NOT_NULL(s); + ASSERT_TRUE(cbm_msb_is_project_scope(s)); + ASSERT(strncmp(s, "cs1\nP\t=Microsoft.NET.Sdk/8.0\t-\n", 30) == 0); + ASSERT_NOT_NULL(strstr(s, "\nG\t='$(A)' < 'b'\n")); /* entities are decoded */ + ASSERT_NOT_NULL(strstr(s, "\nV\t\tPlain\t=one\\ttwo\n")); /* a tab is escaped */ + ASSERT_NOT_NULL(strstr(s, "\nV\t\tCdata\t= & tail\n")); + ASSERT_NOT_NULL(strstr(s, "\nV\t\tNested\t?\n")); /* no plain text */ + ASSERT_NOT_NULL(strstr(s, "\nV\t\tBack\t=a\\\\bAB\n")); + ASSERT_NULL(strstr(s, "Reference")); /* an item group without a is not kept */ + ASSERT_NOT_NULL(strstr(s, "\nY\nY\n")); /* metadata elements, Update: not evaluated */ + ASSERT_NULL(strstr(s, "\nN\t")); + /* the persisted form is the blob itself: it has no line numbers */ + char *portable = cbm_doclink_portable_scope(s); + ASSERT_NOT_NULL(portable); + ASSERT_STR_EQ(portable, s); + cbm_free(CBM_MEM_CLASS_OTHER, portable); + cbm_free_result(r); + + /* XML that is no MSBuild project has no scope: by its name ... */ + r = dm_extract("\n", + CBM_LANG_XML, "src/app.config.xml"); + ASSERT_NOT_NULL(r); + ASSERT_NULL(r->doc_scope); + cbm_free_result(r); + /* ... or, under a project file's name, by its root element */ + r = dm_extract("\n", CBM_LANG_XML, + "src/Settings.props"); + ASSERT_NOT_NULL(r); + ASSERT_NULL(r->doc_scope); + cbm_free_result(r); + /* Directory.Build.props and .targets files are project files */ + r = dm_extract("\n", + CBM_LANG_XML, "Directory.Build.targets"); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + ASSERT_NOT_NULL(strstr(r->doc_scope, "\nN\t\t=X\t\t\t\n")); + cbm_free_result(r); + /* a *.csproj that cannot be read is still a project: its blob says so */ + r = dm_extract("1\n", CBM_LANG_XML, "Bad.csproj"); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + ASSERT_STR_EQ(r->doc_scope, "cs1\nP\t\t!\n"); + cbm_free_result(r); + r = dm_extract("not xml at all", CBM_LANG_XML, "Worse.CSPROJ"); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + ASSERT_STR_EQ(r->doc_scope, "cs1\nP\t\t!\n"); + cbm_free_result(r); + PASS(); +} + +/* ── the C# resolver end to end ──────────────────────────────────── */ + +static void dm_write_resolver_fixture(const char *tmp) { + th_write_file(TH_PATH(tmp, "src/Lib/Lib.csproj"), "\n" + " \n" + " \n" + " \n" + " \n" + " \n" + "\n"); + th_write_file(TH_PATH(tmp, "src/Lib/Extra.cs"), "namespace Acme.Extra\n" + "{\n" + " public class Tool { }\n" + "}\n" + "namespace Acme.Removed\n" + "{\n" + " public class Gone { }\n" + "}\n"); + th_write_file(TH_PATH(tmp, "src/Lib/Helper.cs"), + "namespace Acme.Util\n" + "{\n" + " public static class Helper\n" + " {\n" + " public static void Run(int n) { }\n" + " public static void Run(string s) { }\n" + " public static void Once(int n) { }\n" + " }\n" + "}\n"); + th_write_file(TH_PATH(tmp, "src/Lib/Global.cs"), "public class Size { }\n"); + /* the usual polyfill: it makes `System` a namespace of the repository */ + th_write_file(TH_PATH(tmp, "src/Lib/Shim.cs"), "namespace System.Runtime.CompilerServices\n" + "{\n" + " internal static class IsExternalInit { }\n" + "}\n"); + th_write_file(TH_PATH(tmp, "src/Lib/tests/OnlyInTests.cs"), + "namespace Acme.Core\n" + "{\n" + " public class TestOnlyThing { }\n" + "}\n"); + th_write_file( + TH_PATH(tmp, "src/Lib/Widgets.cs"), + "using Acme.Util;\n" + "using H = Acme.Util.Helper;\n" + "\n" + "namespace Acme.Core\n" + "{\n" + " /// \n" + " /// A widget: \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// when bad\n" + " public class Widget\n" + " {\n" + " /// \n" + " public void Spin() { }\n" + "\n" + " /// Group , one\n" + " /// , wrong .\n" + " /// \n" + " public void Go() { }\n" + "\n" + " /// Twice , ; self\n" + " /// ; member .\n" + " public void Self() { }\n" + "\n" + " /// Arity: and .\n" + " public int Size;\n" + " }\n" + "\n" + " /// Gadget.\n" + " public class Gadget\n" + " {\n" + " public void Spin() { }\n" + " }\n" + "\n" + " public class WidgetError { }\n" + "\n" + " public class Box { }\n" + " public class Box { }\n" + "\n" + " public interface IPinger { void Ping(int n); }\n" + "\n" + " public class Ping { }\n" + "\n" + " public class Pinger : IPinger\n" + " {\n" + " void IPinger.Ping(int n) { }\n" + "\n" + " /// Explicit implementations are not addressable, and an\n" + " /// interface's members are not the class's: ,\n" + " /// .\n" + " public void Other() { }\n" + " }\n" + "}\n"); + th_write_file( + TH_PATH(tmp, "src/Other/Plain.cs"), + "namespace Acme.Plain\n" + "{\n" + " /// and .\n" + " public class Plain { }\n" + "}\n"); + /* arity, constructors, generic methods, same-path declarations */ + th_write_file(TH_PATH(tmp, "src/Other/Lists.cs"), + "namespace Coll\n" + "{\n" + " public interface IList { int IndexOf(object o); }\n" + "}\n"); + th_write_file(TH_PATH(tmp, "src/Other/GenericLists.cs"), "namespace Coll.Generic\n" + "{\n" + " public interface IList { }\n" + " public class OnlyGeneric { }\n" + "}\n"); + th_write_file( + TH_PATH(tmp, "src/Other/Arity.cs"), + "using Coll;\n" + "using Coll.Generic;\n" + "\n" + "namespace Acme.Arity\n" + "{\n" + " public class Pair { public int Left; }\n" + " public class Pair { public int Left; public int Right; }\n" + "\n" + " public class Conv\n" + " {\n" + " public Conv() { }\n" + " public Conv(int x) { }\n" + " public void To(int x) { }\n" + " public void To(T x) { }\n" + " public void Many(string first, params int[] rest) { }\n" + " }\n" + "\n" + " public class Derived : Conv { }\n" + "\n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " public class Uses : Conv { }\n" + "}\n"); + /* declarations the parser cannot place */ + th_write_file( + TH_PATH(tmp, "src/Other/Recovered.cs"), + "namespace Acme.Rec\n" + "{\n" + " public class Before { }\n" + " public ref partial struct Iter\n" + " {\n" + " public void End() { }\n" + " }\n" + " public class Holey\n" + " {\n" + " public safe extern int Hidden();\n" + " /// \n" + " public int Seen;\n" + " }\n" + " /// \n" + " /// \n" + " public class Middle { }\n" + "}\n"); + th_write_file(TH_PATH(tmp, "src/Other/Unbalanced.cs"), "namespace Acme.Rec\n" + "{\n" + " public class Lost { }\n" + " public class Open\n" + " {\n" + " public void M() {\n"); + /* a namespace only one build configuration can name: its block is not + * placed, though the declarations in it are extracted */ + th_write_file(TH_PATH(tmp, "src/Other/Either.cs"), + "#if GEN\n" + "namespace Gen.Interop\n" + "#else\n" + "namespace Run.Interop\n" + "#endif\n" + "{\n" + " /// \n" + " public class InEither { }\n" + "}\n"); + /* an incomplete type in a file that imports a namespace the repository + * does not declare */ + th_write_file(TH_PATH(tmp, "src/Other/OpenScope.cs"), + "using Outside.Lib;\n" + "\n" + "namespace Acme.Rec2\n" + "{\n" + " public class Holey2\n" + " {\n" + " public safe extern int Hidden2();\n" + " /// \n" + " public int Seen2;\n" + " }\n" + "}\n"); + /* a field and a method of one name (two declarations of a partial type + * can do that to a reader who sees no build configuration) */ + th_write_file(TH_PATH(tmp, "src/Other/Kinds.cs"), + "namespace Acme.Kinds\n" + "{\n" + " public class Mixed\n" + " {\n" + " public int Both;\n" + " public void Both(int x) { }\n" + " public int Only;\n" + " }\n" + " /// \n" + " /// \n" + " public class UsesKinds { }\n" + "}\n"); + /* two test programs, each with types in the global namespace */ + th_write_file(TH_PATH(tmp, "tests/ProgA/ProgA.csproj"), + "\n\n"); + th_write_file(TH_PATH(tmp, "tests/ProgA/Shared.cs"), + "public class Shared { }\n" + "/// \n" + "public class UsesOwn { }\n"); + th_write_file(TH_PATH(tmp, "tests/ProgB/ProgB.csproj"), + "\n\n"); + th_write_file(TH_PATH(tmp, "tests/ProgB/Uses.cs"), + "/// \n" + "public class UsesOther { }\n"); + /* ... and one only test code fails to place: no concern of product code */ + th_write_file(TH_PATH(tmp, "src/Other/tests/Half.cs"), "namespace Acme.Core\n" + "{\n" + " public class Gadget { }\n" + " public class Cut\n" + " {\n" + " public void M() {\n"); +} + +TEST(doc_mentions_resolver_rules) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_res_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + dm_write_resolver_fixture(tmp); + char db[512]; + snprintf(db, sizeof(db), "%s/res.db", tmp); + char *project = NULL; + ASSERT_EQ(dm_index(tmp, db, &project), 0); + free(project); + char props[512]; + char reason[64]; + char syntax[64]; + int n = 0; + const char *widgets = "src/Lib/Widgets.cs"; + + /* every cref-bearing tag becomes an edge with its syntax */ + dm_edge(db, "Widgets.Widget", "Widgets.Gadget", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + ASSERT_NOT_NULL(strstr(props, "\"via\":\"doc_comment\"")); + ASSERT_NOT_NULL(strstr(props, "\"syntax\":\"see\"")); + ASSERT_NOT_NULL(strstr(props, "\"tier\":\"unique\"")); + ASSERT_NOT_NULL(strstr(props, "\"line\":7")); + dm_edge(db, "Widgets.Widget", "Helper.Helper.Once", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + ASSERT_NOT_NULL(strstr(props, "\"syntax\":\"seealso\"")); + dm_edge(db, "Widgets.Widget", "Widgets.WidgetError", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + ASSERT_NOT_NULL(strstr(props, "\"syntax\":\"exception\"")); + dm_edge(db, "Widgets.Widget.Spin", "Widgets.Gadget.Spin", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + ASSERT_NOT_NULL(strstr(props, "\"syntax\":\"inheritdoc\"")); + /* alias: exact */ + dm_edge(db, "Widgets.Widget", "Helper.Helper", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + ASSERT_NOT_NULL(strstr(props, "\"tier\":\"exact\"")); + /* R1: a namespace imported only by the project's MSBuild */ + dm_edge(db, "Widgets.Widget", "Extra.Tool", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + /* ... and not one the project file removes again */ + dm_edge(db, "Widgets.Widget", "Extra.Gone", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + dm_row(db, widgets, "Gone", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + /* local and keyword references are neither edges nor rows */ + dm_row(db, widgets, "x", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, ""); + /* a URL is external */ + dm_row(db, widgets, "https://example.org/docs", reason, sizeof(reason), syntax, sizeof(syntax)); + ASSERT_STR_EQ(reason, "external"); + ASSERT_STR_EQ(syntax, "href"); + /* R4: keyword aliases and System.* names outside the corpus -- although a + * polyfill (Shim.cs) makes `System` one of the repository's namespaces */ + dm_row(db, widgets, "string", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "external"); + dm_row(db, widgets, "System.Text.StringBuilder", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "external"); + /* a type the repository's own namespace does not declare is doc rot, not + * a gap of the graph */ + dm_row(db, widgets, "Acme.Core.NoSuchType", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + /* a namespace itself is declared code without a node */ + dm_row(db, widgets, "Acme.Util", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "graph_gap"); + /* product code never binds a test-only declaration */ + dm_row(db, widgets, "TestOnlyThing", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "test_only_target"); + /* a test program binds its own global-namespace types, never another + * program's */ + dm_edge(db, "Shared.UsesOwn", "Shared.Shared", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + ASSERT_EQ(dm_mentions_from(db, "Uses.UsesOther"), 0); + dm_row(db, "tests/ProgB/Uses.cs", "Shared", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "test_only_target"); + dm_row(db, widgets, "Bad Syntax!", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "unparseable"); + + /* overloads: a group without a signature is ambiguous, a signature picks */ + dm_row(db, widgets, "Helper.Run", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "ambiguous"); + dm_edge(db, "Widgets.Widget.Go", "Helper.Helper.Run", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_row(db, widgets, "Helper.Run(double)", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + + /* one edge per (source, target): count and the first line; no self edge */ + dm_edge(db, "Widgets.Widget.Self", "Widgets.Gadget", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + ASSERT_NOT_NULL(strstr(props, "\"count\":2")); + ASSERT_NOT_NULL(strstr(props, "\"line\":24")); + dm_edge(db, "Widgets.Widget.Self", "Widgets.Widget.Self", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + dm_row(db, widgets, "Self", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, ""); + /* a single segment never takes a qualified shortcut: the member Size, + * not the global-namespace class Size */ + dm_edge(db, "Widgets.Widget.Self", "Widgets.Widget.Size", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_edge(db, "Widgets.Widget.Self", "Global.Size", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + + /* R3: type arguments select the generic type; R2: `Box` names the + * arity-0 type, whose node the generic twin took -- a graph gap, never a + * fallback to Box */ + dm_edge(db, "Widgets.Widget.Size", "Widgets.Box", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_row(db, widgets, "Box", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "graph_gap"); + + /* An explicit interface implementation is not addressable by name, and a + * class does not find the members of an interface it implements (the + * compiler looks up no inherited member in a cref): `Ping` is the class of + * that name in the namespace, `Ping(int)` a constructor it does not have. */ + dm_edge(db, "Widgets.Pinger.Other", "Widgets.Ping", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_edge(db, "Widgets.Pinger.Other", "Widgets.IPinger.Ping", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + dm_edge(db, "Widgets.Pinger.Other", "Widgets.Pinger.Ping", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + dm_row(db, widgets, "Ping(int)", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + + /* a closed scope: missing */ + dm_row(db, "src/Other/Plain.cs", "Nowhere", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + dm_row(db, "src/Other/Plain.cs", "Plain.Missing", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + + dm_unlink_db(db); + th_rmtree(tmp); + PASS(); +} + +/* Arity (R2/R3), constructors, generic methods, `params`, and declarations + * that share one qualified name. */ +TEST(doc_mentions_resolver_arity_and_members) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_ar_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + dm_write_resolver_fixture(tmp); + char db[512]; + snprintf(db, sizeof(db), "%s/res.db", tmp); + ASSERT_EQ(dm_index(tmp, db, NULL), 0); + char props[512]; + char reason[64]; + int n = 0; + const char *arity = "src/Other/Arity.cs"; + + /* R2: `IList` is the arity-0 interface, wherever one is in scope -- the + * generic IList of another imported namespace is no rival */ + dm_edge(db, "Arity.Uses", "Lists.IList.IndexOf", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_row(db, arity, "IList.IndexOf", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, ""); + /* ... and a generic type never stands in for a name written without type + * arguments, not even when no arity-0 type of that name is in scope: the + * compiler binds `OnlyGeneric` to nothing. Only `OnlyGeneric{T}` names it. */ + dm_row(db, arity, "OnlyGeneric", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + dm_edge(db, "Arity.Uses", "GenericLists.OnlyGeneric", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + ASSERT_NOT_NULL(strstr(props, "\"count\":1")); /* OnlyGeneric{T} only */ + /* R3: type arguments also select a generic METHOD of that arity */ + dm_edge(db, "Arity.Uses", "Arity.Conv.To", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + /* `Pair.Left` and `Pair.Left` share one node, and it is the later + * declaration's: the arity-0 member is a graph gap, like its type */ + dm_row(db, arity, "Pair.Left", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "graph_gap"); + dm_edge(db, "Arity.Uses", "Arity.Pair.Left", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + ASSERT_NOT_NULL(strstr(props, "\"count\":1")); /* Pair{T}.Left only */ + dm_edge(db, "Arity.Uses", "Arity.Pair.Right", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + /* a bare `Conv` is the type: the constructors of a base class are neither + * inherited nor what a name without a parameter list means */ + dm_edge(db, "Arity.Uses", "Arity.Conv", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_row(db, arity, "Conv", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, ""); + /* a parameter list names the constructor */ + dm_edge(db, "Arity.Uses", "Arity.Conv.Conv", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + /* ... of that type only */ + dm_row(db, arity, "Derived.Conv", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + /* a `params` parameter is part of the signature */ + dm_edge(db, "Arity.Uses", "Arity.Conv.Many", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* member kind: a field and a method of one name are ambiguous without a + * parameter list; with one, only the callable is meant */ + const char *kinds = "src/Other/Kinds.cs"; + dm_row(db, kinds, "Mixed.Both", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "ambiguous"); + dm_row(db, kinds, "Mixed.Both(int)", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, ""); + dm_edge(db, "Kinds.UsesKinds", "Kinds.Mixed.Both", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + ASSERT_NOT_NULL(strstr(props, "\"count\":1")); + dm_edge(db, "Kinds.UsesKinds", "Kinds.Mixed.Only", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + dm_unlink_db(db); + th_rmtree(tmp); + PASS(); +} + +/* Declarations the parser cannot place are never resolved around. */ +TEST(doc_mentions_resolver_parse_errors) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_pe_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + dm_write_resolver_fixture(tmp); + char db[512]; + snprintf(db, sizeof(db), "%s/res.db", tmp); + ASSERT_EQ(dm_index(tmp, db, NULL), 0); + char props[512]; + char reason[64]; + int n = 0; + const char *rec = "src/Other/Recovered.cs"; + + /* `Middle` follows a struct the grammar cannot parse; the tree puts it in + * the global namespace, the braces keep it in Acme.Rec, where `Before` is */ + dm_edge(db, "Recovered.Middle", "Recovered.Before", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + /* the unparsed struct is declared (its header is read from the text) but + * has no node */ + dm_row(db, rec, "Iter", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "graph_gap"); + /* a member behind a parse error: a gap, not doc rot */ + dm_row(db, rec, "Holey.Hidden", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "graph_gap"); + dm_edge(db, "Recovered.Middle", "Recovered.Holey.Seen", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + /* inside the incomplete type: a simple name found nowhere may be the + * hidden member -- a gap, where it would otherwise be reported missing */ + dm_row(db, rec, "Hidden", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "graph_gap"); + /* ... but a keyword alias is never a member, */ + dm_row(db, rec, "int", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "external"); + /* ... and a name an imported namespace outside the repository may supply + * stays external */ + dm_row(db, "src/Other/OpenScope.cs", "Thing", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "external"); + /* a type in a file whose braces do not pair: its name resolves to nothing */ + dm_row(db, rec, "Lost", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "graph_gap"); + /* what is documented in a block that could not be placed has no scope to + * resolve in: a gap, not a lookup from the global namespace (which would + * report `Before`, a class of Acme.Rec, as missing) */ + dm_row(db, "src/Other/Either.cs", "Before", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "graph_gap"); + ASSERT_EQ(dm_mentions_from(db, "Either.InEither"), 0); + /* ... unless only TEST code fails to place the name: product code never + * binds test declarations anyway (tests/Half.cs quarantines `Gadget`) */ + dm_edge(db, "Widgets.Widget", "Widgets.Gadget", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_row(db, rec, "Gadget", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); /* Acme.Core.Gadget is not in Acme.Rec's scope */ + + dm_unlink_db(db); + th_rmtree(tmp); + PASS(); +} + +/* The ship gate: a link family that is switched off still resolves, but its + * resolved references are rows (below_bar_tier), not edges. */ +TEST(doc_mentions_ship_gate) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_gate_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + dm_write_resolver_fixture(tmp); + char db[512]; + snprintf(db, sizeof(db), "%s/gate.db", tmp); + const char *widgets = "src/Lib/Widgets.cs"; + char props[512]; + char reason[64]; + char syntax[64]; + int n = 0; + + /* the families and their names */ + ASSERT_STR_EQ(cbm_doclink_syntax_name(CBM_DOCLINK_CS_SEE), "see"); + ASSERT_STR_EQ(cbm_doclink_syntax_name(CBM_DOCLINK_CS_SEEALSO), "seealso"); + ASSERT_STR_EQ(cbm_doclink_syntax_name(CBM_DOCLINK_CS_EXCEPTION), "exception"); + ASSERT_STR_EQ(cbm_doclink_syntax_name(CBM_DOCLINK_CS_INHERITDOC), "inheritdoc"); + for (int s = CBM_DOCLINK_CS_SEE; s <= CBM_DOCLINK_CS_INHERITDOC; s++) { + const CBMDocLinkFamily *f = cbm_doclink_family(s); + ASSERT_NOT_NULL(f); + ASSERT_EQ(f->lang, CBM_LANG_CSHARP); + ASSERT_TRUE(cbm_doclink_syntax_ships(s)); /* every implemented family ships */ + } + ASSERT_FALSE(cbm_doclink_syntax_ships(CBM_DOCLINK_HREF)); /* a URL is never an edge */ + ASSERT_NULL(cbm_doclink_family(CBM_DOCLINK_NONE)); + ASSERT_NULL(cbm_doclink_family(CBM_DOCLINK_SYNTAX_COUNT)); + + cbm_doclink_test_set_ships(CBM_DOCLINK_CS_SEEALSO, false); + int rc = dm_index(tmp, db, NULL); + cbm_doclink_test_reset_ships(); + ASSERT_EQ(rc, 0); + /* the resolved seealso is a row with the gate's reason and no edge */ + dm_edge(db, "Widgets.Widget", "Helper.Helper.Once", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + dm_row(db, widgets, "Helper.Once(int)", reason, sizeof(reason), syntax, sizeof(syntax)); + ASSERT_STR_EQ(reason, "below_bar_tier"); + ASSERT_STR_EQ(syntax, "seealso"); + /* a seealso that does not resolve keeps its own reason */ + dm_row(db, widgets, "NoSuchThing", reason, sizeof(reason), syntax, sizeof(syntax)); + ASSERT_STR_EQ(reason, "missing"); + ASSERT_STR_EQ(syntax, "seealso"); + /* the other families are untouched */ + dm_edge(db, "Widgets.Widget", "Widgets.Gadget", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_edge(db, "Widgets.Widget", "Widgets.WidgetError", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_edge(db, "Widgets.Widget.Spin", "Widgets.Gadget.Spin", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* with the family shipping again, the same reference is an edge */ + dm_unlink_db(db); + ASSERT_EQ(dm_index(tmp, db, NULL), 0); + dm_edge(db, "Widgets.Widget", "Helper.Helper.Once", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_row(db, widgets, "Helper.Once(int)", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, ""); + + dm_unlink_db(db); + th_rmtree(tmp); + PASS(); +} + +/* A reference written in a file's own doc has the file's File node as its + * source. The resolver finds that node with the pipeline's one lookup; this + * holds it against the graph the pipeline published: the lookup must name the + * File node of exactly that file. (No language of this suite has file-level + * docs; the language legs test the references themselves.) */ +TEST(doc_mentions_file_node_lookup) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_fn_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + dm_write_resolver_fixture(tmp); + char db[512]; + snprintf(db, sizeof(db), "%s/res.db", tmp); + char *project = NULL; + ASSERT_EQ(dm_index(tmp, db, &project), 0); + ASSERT_NOT_NULL(project); + cbm_gbuf_t *gb = cbm_gbuf_new(project, tmp); + ASSERT_NOT_NULL(gb); + ASSERT_EQ(cbm_gbuf_load_from_db(gb, db, project), 0); + const char *paths[] = {"src/Lib/Widgets.cs", "src/Other/tests/Half.cs", "src/Lib/Lib.csproj"}; + for (size_t i = 0; i < sizeof(paths) / sizeof(paths[0]); i++) { + const cbm_gbuf_node_t *n = cbm_pipeline_file_node(gb, project, paths[i]); + ASSERT_NOT_NULL(n); + ASSERT_STR_EQ(n->label, "File"); + ASSERT_STR_EQ(n->file_path, paths[i]); + } + ASSERT_NULL(cbm_pipeline_file_node(gb, project, "src/Lib/NoSuchFile.cs")); + cbm_gbuf_free(gb); + free(project); + dm_unlink_db(db); + th_rmtree(tmp); + PASS(); +} + +/* ── publication: index_status, delete_project, older databases ──── */ + +/* The checks of doc_mentions_index_status_and_delete; the caller owns the + * environment and the directories. */ +static int dm_index_status_checks(const char *tmp, const char *repo, const char *cache_dir) { + char *project = cbm_project_name_from_path(repo); + ASSERT_NOT_NULL(project); + char db[1024]; + snprintf(db, sizeof(db), "%s/%s.db", cache_dir, project); + ASSERT_EQ(dm_index(repo, db, NULL), 0); + int edges = dm_count(db, "SELECT COUNT(*) FROM edges WHERE type = 'MENTIONS'"); + ASSERT_GT(edges, 0); + + /* doc_links: the edge count, every reason with its row count, the status */ + char *text = dm_index_status(project, false); + ASSERT_NOT_NULL(text); + const char *block = strstr(text, "doc_links:\n"); + ASSERT_NOT_NULL(block); + char want[64]; + snprintf(want, sizeof(want), "\n mentions: %d\n", edges); + ASSERT_NOT_NULL(strstr(block, want)); + ASSERT_NOT_NULL(strstr(block, "\n unresolved:\n")); + ASSERT_GT(dm_reason_lines(db, block), 4); + ASSERT_NOT_NULL(strstr(block, "\n test_only_target: 2\n")); + ASSERT_NOT_NULL(strstr(block, "\n unparseable: 1\n")); + ASSERT_NOT_NULL(strstr(block, "\n status: ok")); + ASSERT_NULL(strstr(block, "hint")); + ASSERT_NULL(strstr(block, "samples")); /* only under diagnostics=full */ + free(text); + text = dm_index_status(project, true); + ASSERT_NOT_NULL(text); + block = strstr(text, "doc_links:\n"); + ASSERT_NOT_NULL(block); + ASSERT_NOT_NULL(strstr(block, "samples")); + ASSERT_NOT_NULL(strstr(block, "TestOnlyThing")); + ASSERT_NOT_NULL(strstr(block, "src/Lib/Widgets.cs")); + free(text); + + /* A doc-link layer that fails does not fail the index, and is not + * published as "no references" either: the generation carries the error, + * index_status reports it with what to do, and doing it rebuilds. */ + dm_unlink_db(db); + cbm_doclinks_test_fail_build_once(); + ASSERT_EQ(dm_index(repo, db, NULL), 0); + ASSERT_EQ(dm_count(db, "SELECT COUNT(*) FROM edges WHERE type = 'MENTIONS'"), 0); + ASSERT_EQ(dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"), 1); + ASSERT_EQ(dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'"), 1); + text = dm_index_status(project, false); + ASSERT_NOT_NULL(text); + block = strstr(text, "doc_links:\n"); + ASSERT_NOT_NULL(block); + ASSERT_NOT_NULL(strstr(block, "\n mentions: 0\n")); + ASSERT_NOT_NULL(strstr(block, "\n status: error")); + ASSERT_NOT_NULL(strstr(block, "index_repository")); + ASSERT_NULL(strstr(block, "\n error: ")); + free(text); + ASSERT_EQ(dm_index(repo, db, NULL), 0); /* no file changed */ + ASSERT_EQ(cbm_pipeline_incremental_test_last_route(), CBM_INCREMENTAL_ROUTE_FORCED_FULL); + ASSERT_EQ(dm_count(db, "SELECT COUNT(*) FROM edges WHERE type = 'MENTIONS'"), edges); + ASSERT_EQ(dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'"), 0); + + /* the marker row is the status, never a reference: with rows beside it + * the counts and the samples are theirs alone */ + cbm_store_t *s = cbm_store_open_path(db); + ASSERT_NOT_NULL(s); + cbm_doc_link_row_t marker[2] = { + {.rel_path = "src/Lib/Widgets.cs", + .line = 9, + .syntax = "see", + .raw = "Gone", + .reason = "missing"}, + {.rel_path = "", + .line = 0, + .syntax = "", + .raw = "doc-link layer failed", + .reason = "error"}, + }; + ASSERT_EQ(cbm_store_doc_links_replace(s, project, marker, 2), CBM_STORE_OK); + cbm_store_close(s); + text = dm_index_status(project, true); + ASSERT_NOT_NULL(text); + block = strstr(text, "doc_links:\n"); + ASSERT_NOT_NULL(block); + ASSERT_NOT_NULL(strstr(block, "\n status: error")); + ASSERT_NOT_NULL(strstr(block, "hint")); + ASSERT_NOT_NULL(strstr(block, "\n missing: 1\n")); + ASSERT_NULL(strstr(block, "\n error: ")); + /* nor is it a sample */ + ASSERT_NOT_NULL(strstr(block, "\n samples: 1 (cols: rel_path line syntax raw reason)\n" + " src/Lib/Widgets.cs 9 see Gone missing\n")); + free(text); + /* ... and the re-run the hint asks for rebuilds, although no file changed */ + int rows_before = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"); + ASSERT_EQ(rows_before, 2); + ASSERT_EQ(dm_index(repo, db, NULL), 0); + ASSERT_EQ(cbm_pipeline_incremental_test_last_route(), CBM_INCREMENTAL_ROUTE_FORCED_FULL); + ASSERT_GT(dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"), rows_before); + ASSERT_EQ(dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'"), 0); + + /* an index from before the layer has no table: index_status reports an + * error and what to do, not zero unresolved references */ + sqlite3 *h = NULL; + ASSERT_EQ(sqlite3_open_v2(db, &h, SQLITE_OPEN_READWRITE, NULL), SQLITE_OK); + ASSERT_EQ(sqlite3_exec(h, "DROP TABLE doc_link_unresolved", NULL, NULL, NULL), SQLITE_OK); + sqlite3_close(h); + text = dm_index_status(project, false); + ASSERT_NOT_NULL(text); + block = strstr(text, "doc_links:\n"); + ASSERT_NOT_NULL(block); + ASSERT_NOT_NULL(strstr(block, "\n status: error")); + ASSERT_NOT_NULL(strstr(block, "predates")); + free(text); + /* ... and the next index run creates the table again */ + ASSERT_EQ(dm_index(repo, db, NULL), 0); + ASSERT_EQ(cbm_pipeline_incremental_test_last_route(), CBM_INCREMENTAL_ROUTE_FORCED_FULL); + ASSERT_GT(dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"), rows_before); + text = dm_index_status(project, false); + ASSERT_NOT_NULL(text); + block = strstr(text, "doc_links:\n"); + ASSERT_NOT_NULL(block); + ASSERT_NOT_NULL(strstr(block, "\n status: ok")); + free(text); + /* a healthy generation with unchanged inputs is current */ + ASSERT_EQ(dm_index(repo, db, NULL), 0); + ASSERT_EQ(cbm_pipeline_incremental_test_last_route(), CBM_INCREMENTAL_ROUTE_NOOP); + + /* store-level delete_project removes the rows */ + s = cbm_store_open_path(db); + ASSERT_NOT_NULL(s); + cbm_doc_link_row_t *rows = NULL; + int count = 0; + bool present = false; + ASSERT_EQ(cbm_store_doc_links_get(s, project, &rows, &count, &present), CBM_STORE_OK); + ASSERT_TRUE(present); + ASSERT_GT(count, rows_before); + cbm_store_free_doc_links(rows, count); + ASSERT_EQ(cbm_store_delete_project(s, project), CBM_STORE_OK); + ASSERT_EQ(cbm_store_doc_links_get(s, project, &rows, &count, &present), CBM_STORE_OK); + ASSERT_EQ(count, 0); + cbm_store_free_doc_links(rows, count); + cbm_store_close(s); + + /* a database without the table reads as "no data" at the store level, and + * delete_project still works */ + char old_db[600]; + snprintf(old_db, sizeof(old_db), "%s/old.db", tmp); + cbm_store_t *o = cbm_store_open_path(old_db); + ASSERT_NOT_NULL(o); + ASSERT_EQ(cbm_store_doc_links_get(o, "x", &rows, &count, &present), CBM_STORE_OK); + ASSERT_FALSE(present); + ASSERT_EQ(count, 0); + cbm_doc_link_reason_count_t *reasons = NULL; + int nreasons = 0; + cbm_doc_link_row_t *samples = NULL; + int nsamples = 0; + ASSERT_EQ( + cbm_store_doc_links_summary(o, "x", &reasons, &nreasons, &samples, &nsamples, 5, &present), + CBM_STORE_OK); + ASSERT_FALSE(present); + ASSERT_EQ(cbm_store_delete_project(o, "x"), CBM_STORE_OK); + cbm_store_close(o); + + dm_unlink_db(db); + dm_unlink_db(old_db); + free(project); + return 0; +} + +/* Each preview test starts from a real private index. Restore process state and + * remove the fixture even when a deliberately RED check returns failure. */ +static int dm_preview_fixture(int (*check)(const char *db, const char *project)) { + char tmp[256] = "/tmp/cbm_dm_preview_XXXXXX"; + if (!cbm_mkdtemp(tmp)) { + return 1; + } + char repo[400], cache_dir[400], db[1024]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + snprintf(cache_dir, sizeof(cache_dir), "%s/cache", tmp); + dm_write_resolver_fixture(repo); + cbm_mkdir_p(cache_dir, 0700); + char *project = cbm_project_name_from_path(repo); + const char *saved = getenv("CBM_CACHE_DIR"); + char *saved_copy = saved ? strdup(saved) : NULL; + if (!project || (saved && !saved_copy)) { + free(project); + free(saved_copy); + th_rmtree(tmp); + return 1; + } + snprintf(db, sizeof(db), "%s/%s.db", cache_dir, project); + cbm_setenv("CBM_CACHE_DIR", cache_dir, 1); + int rc = dm_index(repo, db, NULL); + if (rc == 0) { + rc = check(db, project); + } + if (saved_copy) { + cbm_setenv("CBM_CACHE_DIR", saved_copy, 1); + } else { + cbm_unsetenv("CBM_CACHE_DIR"); + } + free(saved_copy); + dm_unlink_db(db); + free(project); + th_rmtree(tmp); + return rc; +} + +/* Return the actual MCP envelope, including structuredContent for JSON, so a + * test exercises both transport boundaries instead of a formatter in isolation. */ +static yyjson_doc *dm_preview_status(const char *project, bool json) { + cbm_mcp_server_t *srv = cbm_mcp_server_new(NULL); + if (!srv) { + return NULL; + } + char args[1200]; + snprintf(args, sizeof(args), "{\"project\":\"%s\",\"diagnostics\":\"full\"%s}", project, + json ? ",\"format\":\"json\"" : ""); + char *response = cbm_mcp_handle_tool(srv, "index_status", args); + yyjson_doc *out = response ? yyjson_read(response, strlen(response), 0) : NULL; + free(response); + cbm_mcp_server_free(srv); + return out; +} + +static const char *dm_preview_text(yyjson_doc *envelope) { + yyjson_val *root = envelope ? yyjson_doc_get_root(envelope) : NULL; + yyjson_val *content = yyjson_obj_get(root, "content"); + return yyjson_get_str(yyjson_obj_get(yyjson_arr_get(content, 0), "text")); +} + +static yyjson_val *dm_preview_report(yyjson_doc *envelope) { + yyjson_val *root = envelope ? yyjson_doc_get_root(envelope) : NULL; + return yyjson_obj_get(yyjson_obj_get(root, "structuredContent"), "doc_links"); +} + +static bool dm_preview_status_is(yyjson_doc *envelope, const char *status) { + const char *actual = yyjson_get_str(yyjson_obj_get(dm_preview_report(envelope), "status")); + return actual && strcmp(actual, status) == 0; +} + +static bool dm_preview_json_consistent(yyjson_doc *envelope) { + const char *text = dm_preview_text(envelope); + yyjson_doc *payload = text ? yyjson_read(text, strlen(text), 0) : NULL; + yyjson_val *root = envelope ? yyjson_doc_get_root(envelope) : NULL; + yyjson_val *structured = yyjson_obj_get(root, "structuredContent"); + char *a = structured ? yyjson_val_write(structured, 0, NULL) : NULL; + char *b = payload ? yyjson_val_write(yyjson_doc_get_root(payload), 0, NULL) : NULL; + bool correct = a && b && strcmp(a, b) == 0; + free(a); + free(b); + yyjson_doc_free(payload); + return correct; +} + +static bool dm_preview_compact_status(yyjson_doc *env, const char *status) { + const char *text = dm_preview_text(env); + const char *report = text ? strstr(text, "doc_links:\n") : NULL; + char expected[50]; + snprintf(expected, sizeof(expected), "\n status: %s\n", status); + return report && strstr(report, expected); +} + +static bool dm_preview_compact_metadata(yyjson_doc *env, size_t sample, const char *field, + size_t original, size_t included, bool truncated, + bool escaped) { + const char *text = dm_preview_text(env); + const char *table = text ? strstr(text, "\n samples_preview:") : NULL; + const char *columns = table ? strstr(table, "(cols: sample_index field original_bytes " + "included_source_bytes truncated escaped)\n") + : NULL; + const char *end = table ? strchr(table + 1, '\n') : NULL; + char row[220]; + snprintf(row, sizeof(row), "\n %zu %s %zu %zu %s %s\n", sample, field, original, included, + truncated ? "true" : "false", escaped ? "true" : "false"); + return columns && end && columns < end && strstr(end, row); +} + +static yyjson_val *dm_preview_metadata(yyjson_val *report, size_t sample, const char *field) { + yyjson_val *entries = yyjson_obj_get(report, "samples_preview"); + for (size_t i = 0; i < yyjson_arr_size(entries); i++) { + yyjson_val *entry = yyjson_arr_get(entries, i); + const char *name = yyjson_get_str(yyjson_obj_get(entry, "field")); + yyjson_val *index = yyjson_obj_get(entry, "sample_index"); + if (name && yyjson_is_uint(index) && yyjson_get_uint(index) == sample && + strcmp(name, field) == 0) { + return entry; + } + } + return NULL; +} + +static bool dm_preview_metadata_matches(yyjson_val *entry, size_t original, size_t included, + bool truncated, bool escaped) { + yyjson_val *orig = yyjson_obj_get(entry, "original_bytes"); + yyjson_val *shown = yyjson_obj_get(entry, "included_source_bytes"); + yyjson_val *cut = yyjson_obj_get(entry, "truncated"); + yyjson_val *quoted = yyjson_obj_get(entry, "escaped"); + return yyjson_is_uint(orig) && yyjson_get_uint(orig) == original && yyjson_is_uint(shown) && + yyjson_get_uint(shown) == included && yyjson_is_bool(cut) && + yyjson_get_bool(cut) == truncated && yyjson_is_bool(quoted) && + yyjson_get_bool(quoted) == escaped; +} + +static bool dm_preview_full_row(cbm_store_t *store, const char *project, + const cbm_doc_link_row_t *expected) { + cbm_doc_link_row_t *rows = NULL; + int count = 0; + bool present = false; + bool correct = + cbm_store_doc_links_get(store, project, &rows, &count, &present) == CBM_STORE_OK && + present && count == 1; + if (correct) { + correct = rows[0].line == expected->line && + strcmp(rows[0].rel_path, expected->rel_path) == 0 && + strcmp(rows[0].syntax, expected->syntax) == 0 && + strcmp(rows[0].raw, expected->raw) == 0 && + strcmp(rows[0].reason, expected->reason) == 0; + } + cbm_store_free_doc_links(rows, count); + return correct; +} + +static int dm_preview_bounds_checks(const char *db, const char *project) { + cbm_store_t *store = cbm_store_open_path(db); + if (!store) { + return 1; + } + const char *fields[] = {"rel_path", "syntax", "raw", "reason"}; + const size_t limits[] = {1024, 128, 1024, 128}; + const cbm_doc_link_row_t ordinary = {.rel_path = "src/Normal.cs", + .line = 7, + .syntax = "see", + .raw = "Gone", + .reason = "missing"}; + bool shape = cbm_store_doc_links_replace(store, project, &ordinary, 1) == CBM_STORE_OK; + for (int format = 0; format < 2; format++) { + yyjson_doc *env = dm_preview_status(project, format != 0); + const char *text = dm_preview_text(env); + if (format) { + yyjson_val *report = dm_preview_report(env); + yyjson_val *samples = yyjson_obj_get(report, "samples"); + shape = dm_preview_status_is(env, "ok") && dm_preview_json_consistent(env) && + yyjson_arr_size(samples) == 1 && + yyjson_obj_size(yyjson_arr_get(samples, 0)) == 5 && + !yyjson_obj_get(report, "samples_preview") && + !yyjson_obj_get(report, "samples_preview_note") && shape; + } else { + shape = text && + strstr(text, "samples: 1 (cols: rel_path line syntax raw reason)\n" + " src/Normal.cs 7 see Gone missing\n") && + !strstr(text, "samples_preview") && shape; + } + yyjson_doc_free(env); + } + bool bounded = true, copies = true, preserved = true; + for (int size = 0; size < 3; size++) { + size_t length = size == 1 ? 65536 : 8192; + bool escaped = size == 2; + char *values[4] = {0}; + bool allocated = true; + for (int field = 0; field < 4; field++) { + values[field] = malloc(length + 1); + if (!values[field]) { + allocated = false; + break; + } + memset(values[field], escaped ? '\\' : 'a' + field, length); + values[field][length] = '\0'; + } + if (!allocated) { + for (int field = 0; field < 4; field++) { + free(values[field]); + } + cbm_store_close(store); + return 1; + } + cbm_doc_link_row_t row = {.rel_path = values[0], + .line = 42, + .syntax = values[1], + .raw = values[2], + .reason = values[3]}; + preserved = + cbm_store_doc_links_replace(store, project, &row, 1) == CBM_STORE_OK && preserved; + for (int format = 0; format < 2; format++) { + cbm_store_doc_links_test_sample_stats_reset(); + yyjson_doc *env = dm_preview_status(project, format != 0); + cbm_doc_links_sample_test_stats_t stats = {0}; + cbm_store_doc_links_test_sample_stats(&stats); + copies = stats.field_copies == 4 && stats.copied_bytes <= 2320 && + stats.requested_bytes == stats.copied_bytes && + stats.max_request_bytes <= 1028 && copies; + const char *text = dm_preview_text(env); + if (format) { + yyjson_val *report = dm_preview_report(env); + yyjson_val *samples = yyjson_obj_get(report, "samples"); + yyjson_val *sample = yyjson_arr_get(samples, 0); + yyjson_val *metadata = yyjson_obj_get(report, "samples_preview"); + const char *note = yyjson_get_str(yyjson_obj_get(report, "samples_preview_note")); + /* the row's reason is no reason this layer writes: the + * status is error (S12); its sample is still shown bounded */ + bounded = dm_preview_status_is(env, "error") && dm_preview_json_consistent(env) && + yyjson_arr_size(samples) == 1 && yyjson_obj_size(sample) == 5 && + yyjson_arr_size(metadata) == 4 && note && + strstr(note, "doc_link_unresolved") && bounded; + for (int field = 0; field < 4; field++) { + yyjson_val *value = yyjson_obj_get(sample, fields[field]); + const char *actual = yyjson_get_str(value); + bounded = actual && yyjson_get_len(value) == limits[field] && + memcmp(actual, values[field], limits[field]) == 0 && + dm_preview_metadata_matches( + dm_preview_metadata(report, 0, fields[field]), length, + limits[field] / (escaped ? 2 : 1), true, escaped) && + bounded; + } + char *encoded = samples ? yyjson_val_write(samples, 0, NULL) : NULL; + bounded = encoded && strlen(encoded) <= 5000 && bounded; + free(encoded); + } else { + const char *table = text ? strstr(text, "\n samples:") : NULL; + const char *metadata = table ? strstr(table, "\n samples_preview:") : NULL; + size_t bytes = table ? (metadata ? (size_t)(metadata - table) : strlen(table)) : 0; + bounded = table && metadata && bytes <= 5000 && + dm_preview_compact_status(env, "error") && + strstr(metadata, "\n samples_preview_note:") && + strstr(table, "(cols: rel_path line syntax raw reason)") && bounded; + for (int field = 0; field < 4; field++) { + bounded = dm_preview_compact_metadata(env, 0, fields[field], length, + limits[field] / (escaped ? 2 : 1), true, + escaped) && + bounded; + } + } + fprintf(stderr, + "doc preview bounds source=%llu escaped=%d json=%d copies=%llu copied=%llu " + "requested=%llu " + "max_request=%llu bounded=%d copy_bound=%d\n", + (unsigned long long)length, escaped, format, + (unsigned long long)stats.field_copies, (unsigned long long)stats.copied_bytes, + (unsigned long long)stats.requested_bytes, + (unsigned long long)stats.max_request_bytes, bounded, copies); + yyjson_doc_free(env); + } + preserved = dm_preview_full_row(store, project, &row) && preserved; + for (int field = 0; field < 4; field++) { + free(values[field]); + } + } + cbm_store_close(store); + fprintf(stderr, "doc preview shape=%d full_rows_preserved=%d\n", shape, preserved); + return !(shape && bounded && copies && preserved); +} + +TEST(doc_mentions_index_status_preview_bounds) { + int result = dm_preview_fixture(dm_preview_bounds_checks); + ASSERT_EQ(result, 0); + PASS(); +} + +/* Check binary storage through SQLite, without asking the legacy C-string + * getter to promise NUL support. All non-NUL cases also exercise that getter. */ +static bool dm_preview_raw_bytes(const char *db, const char *project, const void *bytes, + size_t length, bool write) { + sqlite3 *sql = NULL; + sqlite3_stmt *stmt = NULL; + bool correct = sqlite3_open_v2(db, &sql, write ? SQLITE_OPEN_READWRITE : SQLITE_OPEN_READONLY, + NULL) == SQLITE_OK; + const char *query = write + ? "UPDATE doc_link_unresolved SET raw=?2 WHERE project=?1;" + : "SELECT CAST(raw AS BLOB) FROM doc_link_unresolved WHERE project=?1;"; + correct = correct && sqlite3_prepare_v2(sql, query, -1, &stmt, NULL) == SQLITE_OK && + sqlite3_bind_text(stmt, 1, project, -1, SQLITE_TRANSIENT) == SQLITE_OK; + if (correct && write) { + correct = sqlite3_bind_text(stmt, 2, bytes, (int)length, SQLITE_TRANSIENT) == SQLITE_OK && + sqlite3_step(stmt) == SQLITE_DONE && sqlite3_changes(sql) == 1; + } else if (correct) { + correct = sqlite3_step(stmt) == SQLITE_ROW && + sqlite3_column_bytes(stmt, 0) == (int)length && + (length == 0 || memcmp(sqlite3_column_blob(stmt, 0), bytes, length) == 0) && + sqlite3_step(stmt) == SQLITE_DONE; + } + sqlite3_finalize(stmt); + sqlite3_close(sql); + return correct; +} + +/* Expected preview strings are independent, literal test data. This helper + * only applies the compact table's quote/backslash framing to an expectation. */ +static bool dm_preview_compact_raw(yyjson_doc *env, const char *expected) { + const char *text = dm_preview_text(env); + const char *table = + text ? strstr(text, "samples: 1 (cols: rel_path line syntax raw reason)\n") : NULL; + const char *row = table ? strchr(table, '\n') + 1 : NULL; + const char *end = row ? strchr(row, '\n') : NULL; + if (!row || !end || (size_t)(end - row) > 2200) { + return false; + } + char unquoted[2200], quoted[2200]; + snprintf(unquoted, sizeof(unquoted), " src/Text.cs 9 see %s missing", expected); + size_t used = (size_t)snprintf(quoted, sizeof(quoted), " src/Text.cs 9 see \""); + for (const unsigned char *p = (const unsigned char *)expected; *p; p++) { + if (*p == '\\' || *p == '"') { + quoted[used++] = '\\'; + } + quoted[used++] = (char)*p; + } + snprintf(quoted + used, sizeof(quoted) - used, "\" missing"); + size_t n = (size_t)(end - row); + return (strlen(unquoted) == n && memcmp(row, unquoted, n) == 0) || + (strlen(quoted) == n && memcmp(row, quoted, n) == 0); +} + +static bool dm_preview_text_case(cbm_store_t *store, const char *db, const char *project, + const char *name, const char *source, size_t length, + const char *expected, size_t included, bool escaped) { + cbm_doc_link_row_t row = { + .rel_path = "src/Text.cs", .line = 9, .syntax = "see", .raw = source, .reason = "missing"}; + bool correct = cbm_store_doc_links_replace(store, project, &row, 1) == CBM_STORE_OK && + dm_preview_raw_bytes(db, project, source, length, true); + bool truncated = included < length; + for (int format = 0; format < 2; format++) { + yyjson_doc *env = dm_preview_status(project, format != 0); + bool rendered; + if (format) { + yyjson_val *report = dm_preview_report(env); + yyjson_val *samples = yyjson_obj_get(report, "samples"); + yyjson_val *sample = yyjson_arr_get(samples, 0); + yyjson_val *raw = yyjson_obj_get(sample, "raw"); + const char *actual = yyjson_get_str(raw); + yyjson_val *metadata = yyjson_obj_get(report, "samples_preview"); + rendered = dm_preview_status_is(env, "ok") && dm_preview_json_consistent(env) && + yyjson_arr_size(samples) == 1 && yyjson_obj_size(sample) == 5 && actual && + yyjson_get_len(raw) == strlen(expected) && strcmp(actual, expected) == 0; + if (truncated || escaped) { + const char *note = yyjson_get_str(yyjson_obj_get(report, "samples_preview_note")); + rendered = rendered && yyjson_arr_size(metadata) == 1 && note && + strstr(note, "doc_link_unresolved") && + dm_preview_metadata_matches(dm_preview_metadata(report, 0, "raw"), + length, included, truncated, escaped); + } else { + rendered = rendered && !metadata && !yyjson_obj_get(report, "samples_preview_note"); + } + } else { + rendered = + dm_preview_compact_raw(env, expected) && dm_preview_compact_status(env, "ok"); + const char *text = dm_preview_text(env); + if (truncated || escaped) { + rendered = dm_preview_compact_metadata(env, 0, "raw", length, included, truncated, + escaped) && + text && strstr(text, "\n samples_preview_note:") && rendered; + } else { + rendered = text && !strstr(text, "samples_preview") && rendered; + } + } + correct = rendered && correct; + fprintf(stderr, "doc preview text case=%s json=%d rendered=%d\n", name, format, rendered); + yyjson_doc_free(env); + } + correct = dm_preview_raw_bytes(db, project, source, length, false) && correct; + if (!memchr(source, 0, length)) { + correct = dm_preview_full_row(store, project, &row) && correct; + } + return correct; +} + +static int dm_preview_text_checks(const char *db, const char *project) { + cbm_store_t *store = cbm_store_open_path(db); + if (!store) { + return 1; + } + const struct { + const char *name; + const char *source; + size_t length; + const char *expected; + bool escaped; + } cases[] = { + {"ordinary_utf8", "A\xC3\xA9\xE2\x82\xAC\xF0\x9F\x99\x82Z", 11, + "A\xC3\xA9\xE2\x82\xAC\xF0\x9F\x99\x82Z", false}, + {"c0_del", "A\x01\t\n\r\x1F\x7FZ", 8, + "A\\u{0001}\\u{0009}\\u{000A}\\u{000D}\\u{001F}\\u{007F}Z", true}, + {"nul", "A\0B", 3, "A\\u{0000}B", true}, + {"literal_escapes", "\\u{000A}\\xFF", 12, "\\\\u{000A}\\\\xFF", true}, + {"reserved_bytes", "@bytes:AA", 9, "\\u{0040}bytes:AA", true}, + {"reserved_utf8", "@utf8:AA", 8, "\\u{0040}utf8:AA", true}, + {"middle_prefix", "A@utf8:AA", 9, "A@utf8:AA", false}, + {"malformed", "\xC0\xAF\xED\xA0\x80\xF4\x90\x80\x80\x80\xC2", 11, + "\\xC0\\xAF\\xED\\xA0\\x80\\xF4\\x90\\x80\\x80\\x80\\xC2", true}, + {"format_controls", + "\xE2\x80\x8B\xE2\x80\x8F\xE2\x80\xAA\xE2\x80\xAE" + "\xE2\x81\xA6\xE2\x81\xA9\xF3\xA0\x80\x80\xF3\xA0\x81\xBF", + 26, "\\u{200B}\\u{200F}\\u{202A}\\u{202E}\\u{2066}\\u{2069}\\u{E0000}\\u{E007F}", true}, + {"adjacent_unicode", + "\xE2\x80\x8A\xE2\x80\x90\xE2\x80\xA9\xE2\x80\xAF" + "\xE2\x81\xA5\xE2\x81\xAA\xF3\x9F\xBF\xBF\xF3\xA0\x82\x80", + 26, + "\xE2\x80\x8A\xE2\x80\x90\xE2\x80\xA9\xE2\x80\xAF" + "\xE2\x81\xA5\xE2\x81\xAA\xF3\x9F\xBF\xBF\xF3\xA0\x82\x80", + false}, + }; + bool correct = true; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + correct = dm_preview_text_case(store, db, project, cases[i].name, cases[i].source, + cases[i].length, cases[i].expected, cases[i].length, + cases[i].escaped) && + correct; + } + /* One expected token must either fit entirely or contribute no source + * bytes. The suffix guarantees truncation even for the exactly-fit case. */ + const struct { + const char *name, *source, *expected; + size_t source_bytes, display_bytes; + bool escape; + } tokens[] = { + {"utf8_2", "\xC3\xA9", "\xC3\xA9", 2, 2, false}, + {"utf8_3", "\xE2\x82\xAC", "\xE2\x82\xAC", 3, 3, false}, + {"utf8_4", "\xF0\x9F\x99\x82", "\xF0\x9F\x99\x82", 4, 4, false}, + {"newline", "\n", "\\u{000A}", 1, 8, true}, + {"backslash", "\\", "\\\\", 1, 2, true}, + {"invalid", "\xFF", "\\xFF", 1, 4, true}, + {"tag", "\xF3\xA0\x80\x81", "\\u{E0001}", 4, 9, true}, + }; + for (size_t i = 0; i < sizeof(tokens) / sizeof(tokens[0]); i++) { + for (int fit = 0; fit < 2; fit++) { + char source[1100], expected[1100], name[60]; + size_t prefix = fit ? 1024 - tokens[i].display_bytes : 1023; + memset(source, 'p', prefix); + memcpy(source + prefix, tokens[i].source, tokens[i].source_bytes); + memcpy(source + prefix + tokens[i].source_bytes, "TAIL", 5); + memset(expected, 'p', prefix); + size_t shown = prefix; + if (fit) { + memcpy(expected + shown, tokens[i].expected, tokens[i].display_bytes); + shown += tokens[i].display_bytes; + } + expected[shown] = '\0'; + snprintf(name, sizeof(name), "%s_%s", tokens[i].name, fit ? "fits" : "stops"); + correct = dm_preview_text_case(store, db, project, name, source, + prefix + tokens[i].source_bytes + 4, expected, + prefix + (fit ? tokens[i].source_bytes : 0), + fit && tokens[i].escape) && + correct; + } + } + cbm_store_close(store); + return !correct; +} + +TEST(doc_mentions_index_status_preview_text) { + int result = dm_preview_fixture(dm_preview_text_checks); + ASSERT_EQ(result, 0); + PASS(); +} + +static int dm_preview_order_checks(const char *db, const char *project) { + cbm_store_t *store = cbm_store_open_path(db); + if (!store) { + return 1; + } + cbm_doc_link_row_t rows[52] = {0}; + char paths[51][40], raw[1026]; + memset(raw, 'r', sizeof(raw) - 1); + raw[sizeof(raw) - 1] = '\0'; + /* The marker sorts first and consumes one of the original SQL LIMIT 50. + * Insert references backwards, so a storage-order accident cannot pass. */ + rows[0] = + (cbm_doc_link_row_t){.rel_path = "", .line = 0, .syntax = "", .raw = "", .reason = "error"}; + for (int i = 0; i < 51; i++) { + snprintf(paths[i], sizeof(paths[i]), "src/Case%03d.cs", 50 - i); + rows[i + 1] = (cbm_doc_link_row_t){ + .rel_path = paths[i], .line = 5, .syntax = "see", .raw = raw, .reason = "missing"}; + } + bool correct = cbm_store_doc_links_replace(store, project, rows, 52) == CBM_STORE_OK; + for (int format = 0; format < 2; format++) { + cbm_store_doc_links_test_sample_stats_reset(); + yyjson_doc *env = dm_preview_status(project, format != 0); + cbm_doc_links_sample_test_stats_t stats = {0}; + cbm_store_doc_links_test_sample_stats(&stats); + bool ordered = stats.field_copies >= 196 && stats.field_copies <= 200 && + stats.copied_bytes <= 116000 && + stats.requested_bytes == stats.copied_bytes && + stats.max_request_bytes <= 1028; + if (format) { + yyjson_val *report = dm_preview_report(env); + yyjson_val *samples = yyjson_obj_get(report, "samples"); + ordered = dm_preview_status_is(env, "error") && dm_preview_json_consistent(env) && + yyjson_arr_size(samples) == 49 && + yyjson_arr_size(yyjson_obj_get(report, "samples_preview")) == 49 && ordered; + for (int i = 0; i < 49; i++) { + char path[40]; + snprintf(path, sizeof(path), "src/Case%03d.cs", i); + yyjson_val *sample = yyjson_arr_get(samples, (size_t)i); + const char *actual = yyjson_get_str(yyjson_obj_get(sample, "rel_path")); + ordered = actual && strcmp(actual, path) == 0 && yyjson_obj_size(sample) == 5 && + yyjson_get_len(yyjson_obj_get(sample, "raw")) == 1024 && + dm_preview_metadata_matches(dm_preview_metadata(report, (size_t)i, "raw"), + 1025, 1024, true, false) && + ordered; + } + } else { + const char *text = dm_preview_text(env); + const char *cursor = + text ? strstr(text, "samples: 49 (cols: rel_path line syntax raw reason)\n") + : NULL; + ordered = cursor && ordered; + for (int i = 0; i < 49; i++) { + char prefix[60]; + snprintf(prefix, sizeof(prefix), "\n src/Case%03d.cs 5 see ", i); + const char *next = cursor ? strstr(cursor, prefix) : NULL; + ordered = next && ordered; + cursor = next ? next + strlen(prefix) : NULL; + } + ordered = cursor && !strstr(cursor, "src/Case049.cs") && + !strstr(cursor, "src/Case050.cs") && ordered; + } + fprintf(stderr, "doc preview order json=%d copies=%llu copied=%llu ordered=%d\n", format, + (unsigned long long)stats.field_copies, (unsigned long long)stats.copied_bytes, + ordered); + correct = ordered && correct; + yyjson_doc_free(env); + } + cbm_doc_link_row_t *full = NULL; + int count = 0; + bool present = false; + correct = cbm_store_doc_links_get(store, project, &full, &count, &present) == CBM_STORE_OK && + present && count == 52 && correct; + for (int i = 0; i < count; i++) { + if (full[i].rel_path && full[i].rel_path[0]) { + correct = strcmp(full[i].raw, raw) == 0 && correct; + } + } + cbm_store_free_doc_links(full, count); + /* Sort the complete source before projection. These two rows have the same + * displayed raw prefix, but the longer original sorts first (aa before z). */ + char first[1030], second[1029]; + memset(first, 't', 1027); + memcpy(first + 1027, "aa", 3); + memset(second, 't', 1027); + memcpy(second + 1027, "z", 2); + cbm_doc_link_row_t ties[] = { + {.rel_path = "src/Tie.cs", .line = 2, .syntax = "see", .raw = second, .reason = "missing"}, + {.rel_path = "src/Tie.cs", .line = 2, .syntax = "see", .raw = first, .reason = "missing"}, + {.rel_path = "src/Tie.cs", .line = 1, .syntax = "see", .raw = second, .reason = "missing"}, + {.rel_path = "src/Z.cs", .line = 3, .syntax = "see", .raw = first, .reason = "ambiguous"}, + }; + correct = cbm_store_doc_links_replace(store, project, ties, 4) == CBM_STORE_OK && correct; + yyjson_doc *env = dm_preview_status(project, true); + yyjson_val *report = dm_preview_report(env); + yyjson_val *ordered = yyjson_obj_get(report, "samples"); + const size_t original[] = {1029, 1028, 1029, 1028}; + const int lines[] = {3, 1, 2, 2}; + bool ties_correct = dm_preview_status_is(env, "ok") && dm_preview_json_consistent(env) && + yyjson_arr_size(ordered) == 4; + for (size_t i = 0; i < 4; i++) { + yyjson_val *sample = yyjson_arr_get(ordered, i); + const char *path = yyjson_get_str(yyjson_obj_get(sample, "rel_path")); + ties_correct = path && strcmp(path, i ? "src/Tie.cs" : "src/Z.cs") == 0 && + yyjson_get_int(yyjson_obj_get(sample, "line")) == lines[i] && + dm_preview_metadata_matches(dm_preview_metadata(report, i, "raw"), + original[i], 1024, true, false) && + ties_correct; + } + fprintf(stderr, "doc preview full_source_order=%d\n", ties_correct); + correct = ties_correct && correct; + yyjson_doc_free(env); + cbm_store_close(store); + return !correct; +} + +TEST(doc_mentions_index_status_preview_order) { + int result = dm_preview_fixture(dm_preview_order_checks); + ASSERT_EQ(result, 0); + PASS(); +} + +static bool dm_preview_error_result(yyjson_doc *env, bool json) { + if (json) { + yyjson_val *report = dm_preview_report(env); + return dm_preview_status_is(env, "error") && dm_preview_json_consistent(env) && + yyjson_arr_size(yyjson_obj_get(report, "samples")) == 0 && + yyjson_arr_size(yyjson_obj_get(report, "samples_preview")) == 0 && + !yyjson_obj_get(report, "samples_preview_note"); + } + const char *text = dm_preview_text(env); + const char *report = text ? strstr(text, "doc_links:\n") : NULL; + return report && strstr(report, "\n status: error\n") && !strstr(report, "src/Failure") && + !strstr(report, "samples_preview"); +} + +static bool dm_preview_retry_result(yyjson_doc *env, bool json) { + if (json) { + yyjson_val *samples = yyjson_obj_get(dm_preview_report(env), "samples"); + if (!dm_preview_status_is(env, "ok") || !dm_preview_json_consistent(env) || + yyjson_arr_size(samples) != 2) { + return false; + } + for (size_t i = 0; i < 2; i++) { + yyjson_val *row = yyjson_arr_get(samples, i); + const char *raw = yyjson_get_str(yyjson_obj_get(row, "raw")); + const char *path = yyjson_get_str(yyjson_obj_get(row, "rel_path")); + if (yyjson_obj_size(row) != 5 || !raw || + strcmp(raw, i ? "Second" : "First\\u{000A}") != 0 || !path || + strcmp(path, i ? "src/FailureB.cs" : "src/FailureA.cs") != 0) { + return false; + } + } + yyjson_val *report = dm_preview_report(env); + return yyjson_arr_size(yyjson_obj_get(report, "samples_preview")) == 1 && + dm_preview_metadata_matches(dm_preview_metadata(report, 0, "raw"), 6, 6, false, + true) && + yyjson_get_str(yyjson_obj_get(report, "samples_preview_note")); + } + const char *text = dm_preview_text(env); + return text && dm_preview_compact_status(env, "ok") && + dm_preview_compact_metadata(env, 0, "raw", 6, 6, false, true) && + strstr(text, "\n samples_preview_note:") && + strstr(text, "samples: 2 (cols: rel_path line syntax raw reason)\n" + " src/FailureA.cs 3 see First\\u{000A} missing\n" + " src/FailureB.cs 4 see Second missing\n"); +} + +static int dm_preview_failure_checks(const char *db, const char *project) { + cbm_store_t *store = cbm_store_open_path(db); + if (!store) { + return 1; + } + cbm_doc_link_row_t rows[] = { + {.rel_path = "src/FailureA.cs", + .line = 3, + .syntax = "see", + .raw = "First\n", + .reason = "missing"}, + {.rel_path = "src/FailureB.cs", + .line = 4, + .syntax = "see", + .raw = "Second", + .reason = "missing"}, + }; + bool correct = cbm_store_doc_links_replace(store, project, rows, 2) == CBM_STORE_OK; + for (int format = 0; format < 2; format++) { + yyjson_doc *warm = dm_preview_status(project, format != 0); + correct = dm_preview_retry_result(warm, format != 0) && correct; + yyjson_doc_free(warm); + } + + /* Full getters and summary-only callers must not consume sample faults or + * record projected sample bytes. This also protects incremental callers. */ + cbm_store_doc_links_test_sample_stats_reset(); + cbm_store_doc_links_test_fail_sample_alloc_after(0); + cbm_mcp_doc_links_test_fail_sample_alloc_after(0); + cbm_doc_link_row_t *full = NULL, *samples = NULL; + cbm_doc_link_reason_count_t *reasons = NULL; + int count = 0, nsamples = 0, nreasons = 0; + bool present = false; + bool bypass = + cbm_store_doc_links_get(store, project, &full, &count, &present) == CBM_STORE_OK && + present && count == 2; + cbm_store_free_doc_links(full, count); + bypass = cbm_store_doc_links_summary(store, project, &reasons, &nreasons, &samples, &nsamples, + 0, &present) == CBM_STORE_OK && + present && nsamples == 0 && nreasons == 1 && bypass; + cbm_store_free_doc_links(samples, nsamples); + cbm_store_free_doc_link_reasons(reasons, nreasons); + char *summary = dm_index_status(project, false); + bypass = summary && !strstr(summary, "src/Failure") && bypass; + free(summary); + cbm_doc_links_sample_test_stats_t stats = {0}; + cbm_store_doc_links_test_sample_stats(&stats); + bypass = stats.field_copies == 0 && stats.copied_bytes == 0 && stats.requested_bytes == 0 && + !cbm_store_doc_links_test_sample_alloc_failed() && + !cbm_mcp_doc_links_test_sample_alloc_failed() && bypass; + cbm_store_doc_links_test_fail_sample_alloc_after(-1); + cbm_mcp_doc_links_test_fail_sample_alloc_after(-1); + fprintf(stderr, "doc preview sample_only=%d\n", bypass); + correct = bypass && correct; + + for (int stage = 0; stage < 2; stage++) { + for (int nth = 0; nth < 2; nth++) { + for (int format = 0; format < 2; format++) { + uint64_t before = cbm_mem_tracked_live_bytes(); + if (stage == 0) { + cbm_store_doc_links_test_fail_sample_alloc_after(nth ? 4 : 0); + } else { + cbm_mcp_doc_links_test_fail_sample_alloc_after(nth ? 4 : 0); + } + yyjson_doc *failed = dm_preview_status(project, format != 0); + bool consumed = stage == 0 ? cbm_store_doc_links_test_sample_alloc_failed() + : cbm_mcp_doc_links_test_sample_alloc_failed(); + bool error = dm_preview_error_result(failed, format != 0); + yyjson_doc_free(failed); + cbm_store_doc_links_test_fail_sample_alloc_after(-1); + cbm_mcp_doc_links_test_fail_sample_alloc_after(-1); + uint64_t after_failure = cbm_mem_tracked_live_bytes(); + yyjson_doc *retried = dm_preview_status(project, format != 0); + bool retry = dm_preview_retry_result(retried, format != 0); + yyjson_doc_free(retried); + uint64_t after_retry = cbm_mem_tracked_live_bytes(); + bool clean = before == after_failure && before == after_retry; + correct = consumed && error && retry && clean && correct; + fprintf(stderr, + "doc preview fault stage=%d nth=%d json=%d consumed=%d error=%d retry=%d " + "before=%llu failed=%llu retried=%llu clean=%d\n", + stage, nth ? 5 : 1, format, consumed, error, retry, + (unsigned long long)before, (unsigned long long)after_failure, + (unsigned long long)after_retry, clean); + } + } + } + cbm_store_close(store); + return !correct; +} + +TEST(doc_mentions_index_status_preview_failure) { + int result = dm_preview_fixture(dm_preview_failure_checks); + ASSERT_EQ(result, 0); + PASS(); +} + +TEST(doc_mentions_index_status_and_delete) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_st_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400]; + char cache_dir[400]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + snprintf(cache_dir, sizeof(cache_dir), "%s/cache", tmp); + dm_write_resolver_fixture(repo); + cbm_mkdir_p(cache_dir, 0700); + /* the tool reads the project's database from the cache directory: a + * private one for this test, restored whatever the checks say */ + const char *saved_cache = getenv("CBM_CACHE_DIR"); + char *saved_cache_copy = saved_cache ? strdup(saved_cache) : NULL; + cbm_setenv("CBM_CACHE_DIR", cache_dir, 1); + int rc = dm_index_status_checks(tmp, repo, cache_dir); + if (saved_cache_copy) { + cbm_setenv("CBM_CACHE_DIR", saved_cache_copy, 1); + free(saved_cache_copy); + } else { + cbm_unsetenv("CBM_CACHE_DIR"); + } + th_rmtree(tmp); + return rc; +} + +/* ── incremental == full ─────────────────────────────────────────── */ + +static const char DM_MAIN[] = + "using P;\n" + "using Q;\n" + "namespace N\n" + "{\n" + " /// See , , .\n" + " /// \n" + " /// \n" + " /// \n" + " public class Main { }\n" + "}\n"; +static const char DM_MAIN_EDITED[] = + "using P;\n" + "using Q;\n" + "namespace N\n" + "{\n" + " /// See , , .\n" + " /// \n" + " /// \n" + " /// Also .\n" + " public class Main { }\n" + "}\n"; +static const char DM_MAIN_USING[] = + "using P;\n" + "using Q;\n" + "using S;\n" + "namespace N\n" + "{\n" + " /// See , , .\n" + " /// \n" + " /// \n" + " /// Also .\n" + " public class Main { }\n" + "}\n"; +/* a file with a row and no edge into any file the steps change */ +static const char DM_SIDE[] = "using Q;\n" + "namespace N\n" + "{\n" + " /// \n" + " public class Side { }\n" + "}\n"; +static const char DM_P[] = "namespace P\n" + "{\n" + " public class Target { }\n" + " public class Dup { }\n" + " public class Keep { }\n" + "}\n"; +static const char DM_P_NO_DUP[] = "namespace P\n" + "{\n" + " public class Target { }\n" + " public class Keep { }\n" + "}\n"; +static const char DM_P_RENAMED[] = "namespace P\n" + "{\n" + " public class Target { }\n" + " public class Kept { }\n" + "}\n"; +static const char DM_P_NO_TARGET[] = "namespace P\n" + "{\n" + " public class Kept { }\n" + "}\n"; +static const char DM_Q[] = "namespace Q\n" + "{\n" + " public class Dup { }\n" + " public class Other { }\n" + "}\n"; +static const char DM_Q_TWIN[] = "namespace Q\n" + "{\n" + " public class Dup { }\n" + " public class Other { }\n" + " public class Target { }\n" + "}\n"; +/* the same declarations, every one on another line */ +static const char DM_Q_TWIN_MOVED[] = "// moved\n" + "\n" + "namespace Q\n" + "{\n" + " public class Dup { }\n" + "\n" + " public class Other { }\n" + " public class Target { }\n" + "}\n"; +static const char DM_SIG[] = "namespace Q\n" + "{\n" + " public class Sig\n" + " {\n" + " public void Go(int x) { }\n" + " }\n" + "}\n"; +static const char DM_SIG_BODY[] = "namespace Q\n" + "{\n" + " public class Sig\n" + " {\n" + " public void Go(int x) { x++; }\n" + " }\n" + "}\n"; +static const char DM_SIG_CHANGED[] = "namespace Q\n" + "{\n" + " public class Sig\n" + " {\n" + " public void Go(string x) { }\n" + " }\n" + "}\n"; +static const char DM_OV[] = "namespace Q\n" + "{\n" + " public class Ov\n" + " {\n" + " public void Run(int n) { }\n" + " public void Run(string s) { }\n" + " }\n" + "}\n"; +static const char DM_OV_ONE[] = "namespace Q\n" + "{\n" + " public class Ov\n" + " {\n" + " public void Run(int n) { }\n" + " }\n" + "}\n"; +/* `Tw` and `Tw` share one node; it belongs to the later declaration */ +static const char DM_TW[] = "namespace Q\n" + "{\n" + " public class Tw { }\n" + " public class Tw { }\n" + "}\n"; +static const char DM_TW_SWAPPED[] = "namespace Q\n" + "{\n" + " public class Tw { }\n" + " public class Tw { }\n" + "}\n"; +static const char DM_R[] = "namespace R\n" + "{\n" + " public class ViaProject { }\n" + "}\n"; +static const char DM_S[] = "namespace S\n" + "{\n" + " public class Extra { }\n" + "}\n"; +static const char DM_CSPROJ[] = "\n" + " \n" + " \n" + " \n" + "\n"; +static const char DM_CSPROJ_NO_USING[] = "\n" + "\n"; + +TEST(doc_mentions_incremental_equals_full) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_inc_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + th_write_file(TH_PATH(repo, "src/Main.cs"), DM_MAIN); + th_write_file(TH_PATH(repo, "src/Side.cs"), DM_SIDE); + th_write_file(TH_PATH(repo, "src/P.cs"), DM_P); + th_write_file(TH_PATH(repo, "src/Q.cs"), DM_Q); + th_write_file(TH_PATH(repo, "src/Sig.cs"), DM_SIG); + th_write_file(TH_PATH(repo, "src/Ov.cs"), DM_OV); + th_write_file(TH_PATH(repo, "src/Tw.cs"), DM_TW); + th_write_file(TH_PATH(repo, "src/R.cs"), DM_R); + th_write_file(TH_PATH(repo, "src/S.cs"), DM_S); + th_write_file(TH_PATH(repo, "src/App.csproj"), DM_CSPROJ); + char inc_db[512]; + char full_db[512]; + snprintf(inc_db, sizeof(inc_db), "%s/inc.db", tmp); + snprintf(full_db, sizeof(full_db), "%s/full.db", tmp); + ASSERT_EQ(dm_index(repo, inc_db, NULL), 0); + char props[512]; + char reason[64]; + int n = 0; + const char *main_cs = "src/Main.cs"; + const char *side_cs = "src/Side.cs"; + /* the preconditions the steps below move away from */ + dm_edge(inc_db, "Main.Main", "P.Target", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_edge(inc_db, "Main.Main", "R.ViaProject", props, sizeof(props), &n); + ASSERT_EQ(n, 1); /* through the project file's */ + dm_edge(inc_db, "Main.Main", "Sig.Sig.Go", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_edge(inc_db, "Main.Main", "Tw.Tw", props, sizeof(props), &n); + ASSERT_EQ(n, 1); /* Tw, the later declaration, owns the node */ + dm_row(inc_db, main_cs, "Dup", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "ambiguous"); /* P.Dup and Q.Dup through the usings */ + dm_row(inc_db, main_cs, "Extra", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); /* namespace S is not imported */ + dm_row(inc_db, main_cs, "Ov.Run", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "ambiguous"); /* an overload group */ + dm_row(inc_db, side_cs, "Tw", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "graph_gap"); /* the arity-0 twin has no node */ + + /* a doc-comment edit re-extracts the file alone */ + th_write_file(TH_PATH(repo, "src/Main.cs"), DM_MAIN_EDITED); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "doc edit", CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR), 0); + dm_edge(inc_db, "Main.Main", "Q.Other", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* a body edit in a TARGET file: its scope is unchanged, and the edge into + * it stands */ + th_write_file(TH_PATH(repo, "src/Sig.cs"), DM_SIG_BODY); + ASSERT_EQ( + dm_step(repo, inc_db, full_db, "target body edit", CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR), + 0); + dm_edge(inc_db, "Main.Main", "Sig.Sig.Go", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* the file's own using directives scope nothing but the file */ + th_write_file(TH_PATH(repo, "src/Main.cs"), DM_MAIN_USING); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "using added", CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR), + 0); + dm_edge(inc_db, "Main.Main", "S.Extra", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* a removed member (an overload group shrinks to one) re-resolves the + * files whose rows name it, although they have no edge into the changed + * file */ + th_write_file(TH_PATH(repo, "src/Ov.cs"), DM_OV_ONE); + ASSERT_EQ( + dm_step(repo, inc_db, full_db, "member removed", CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR), 0); + dm_edge(inc_db, "Main.Main", "Ov.Ov.Run", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* a removed TYPE is not repaired file by file (it may be another type's + * base): one of two ambiguous candidates goes, the other binds */ + th_write_file(TH_PATH(repo, "src/P.cs"), DM_P_NO_DUP); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "type removed", CBM_INCREMENTAL_ROUTE_FORCED_FULL), 0); + dm_edge(inc_db, "Main.Main", "Q.Dup", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* renaming a target */ + th_write_file(TH_PATH(repo, "src/P.cs"), DM_P_RENAMED); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "rename", CBM_INCREMENTAL_ROUTE_FORCED_FULL), 0); + dm_row(inc_db, main_cs, "Keep", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + + /* an ambiguous twin */ + th_write_file(TH_PATH(repo, "src/Q.cs"), DM_Q_TWIN); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "ambiguous twin", CBM_INCREMENTAL_ROUTE_FORCED_FULL), + 0); + dm_row(inc_db, main_cs, "Target", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "ambiguous"); + + /* deleting a target */ + th_write_file(TH_PATH(repo, "src/P.cs"), DM_P_NO_TARGET); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "delete target", CBM_INCREMENTAL_ROUTE_FORCED_FULL), + 0); + dm_edge(inc_db, "Main.Main", "Q.Target", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* moving every declaration of a target file to another line changes no + * scope: the stored scopes carry no line numbers */ + th_write_file(TH_PATH(repo, "src/Q.cs"), DM_Q_TWIN_MOVED); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "lines moved", CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR), + 0); + dm_edge(inc_db, "Main.Main", "Q.Target", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* a changed signature can re-route references in files with no edge into + * the changed one */ + th_write_file(TH_PATH(repo, "src/Sig.cs"), DM_SIG_CHANGED); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "signature", CBM_INCREMENTAL_ROUTE_FORCED_FULL), 0); + dm_row(inc_db, main_cs, "Sig.Go(int)", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + + /* declaration order decides which same-path declaration owns the node: + * swapping the twins moves it, for a file with no edge into this one too */ + th_write_file(TH_PATH(repo, "src/Tw.cs"), DM_TW_SWAPPED); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "twins swapped", CBM_INCREMENTAL_ROUTE_FORCED_FULL), + 0); + dm_edge(inc_db, "Side.Side", "Tw.Tw", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_row(inc_db, main_cs, "Tw{T}", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "graph_gap"); + + /* the project file sets the global usings of every file of the project */ + th_write_file(TH_PATH(repo, "src/App.csproj"), DM_CSPROJ_NO_USING); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "project file", CBM_INCREMENTAL_ROUTE_FORCED_FULL), 0); + dm_edge(inc_db, "Main.Main", "R.ViaProject", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + dm_row(inc_db, main_cs, "ViaProject", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + /* Q.Target, Q.Dup, Q.Other, S.Extra, Ov.Run */ + ASSERT_EQ(dm_mentions_from(inc_db, "Main.Main"), 5); + + /* a deleted file takes its namespace and types along */ + ASSERT_EQ(unlink(TH_PATH(repo, "src/Q.cs")), 0); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "delete file", CBM_INCREMENTAL_ROUTE_FORCED_FULL), 0); + ASSERT_EQ(dm_mentions_from(inc_db, "Main.Main"), 2); /* S.Extra, Ov.Run */ + + dm_unlink_db(inc_db); + dm_unlink_db(full_db); + th_rmtree(tmp); + PASS(); +} + +/* A project file is a file of the index like any other: the planner judges a + * change to one by its scope blob. An edit the blob does not hold repairs + * file by file; an edit of a , a new project file and a deleted one + * rebuild. Every step ends in what a full index of the same tree holds. */ +TEST(doc_mentions_incremental_project_files) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_ipf_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + th_write_file(TH_PATH(repo, "src/App/Main.cs"), + "namespace N\n" + "{\n" + " /// \n" + " public class Main { }\n" + "}\n"); + th_write_file(TH_PATH(repo, "src/R.cs"), "namespace R\n" + "{\n" + " public class ViaProject { }\n" + "}\n" + "namespace S\n" + "{\n" + " public class ViaProps { }\n" + "}\n"); + th_write_file(TH_PATH(repo, "src/App/App.csproj"), + "\n" + " \n" + " \n" + "\n"); + char inc_db[512]; + char full_db[512]; + snprintf(inc_db, sizeof(inc_db), "%s/inc.db", tmp); + snprintf(full_db, sizeof(full_db), "%s/full.db", tmp); + ASSERT_EQ(dm_index(repo, inc_db, NULL), 0); + char props[512]; + char reason[64]; + int n = 0; + dm_edge(inc_db, "Main.Main", "R.ViaProject", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + dm_row(inc_db, "src/App/Main.cs", "ViaProps", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + + /* an edit the scope blob does not hold: a package version */ + th_write_file(TH_PATH(repo, "src/App/App.csproj"), + "\n" + " \n" + " \n" + "\n"); + ASSERT_EQ( + dm_step(repo, inc_db, full_db, "package version", CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR), 0); + dm_edge(inc_db, "Main.Main", "R.ViaProject", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* a project file is added: Directory.Build.props brings namespace S in */ + th_write_file(TH_PATH(repo, "Directory.Build.props"), + "\n" + " \n" + "\n"); + ASSERT_EQ( + dm_step(repo, inc_db, full_db, "project file added", CBM_INCREMENTAL_ROUTE_FORCED_FULL), 0); + dm_edge(inc_db, "Main.Main", "R.ViaProps", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + + /* an edit of what the blob holds: the goes */ + th_write_file(TH_PATH(repo, "src/App/App.csproj"), + "\n" + " \n" + "\n"); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "using removed", CBM_INCREMENTAL_ROUTE_FORCED_FULL), + 0); + dm_edge(inc_db, "Main.Main", "R.ViaProject", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + + /* a project file is deleted */ + ASSERT_EQ(unlink(TH_PATH(repo, "Directory.Build.props")), 0); + ASSERT_EQ( + dm_step(repo, inc_db, full_db, "project file deleted", CBM_INCREMENTAL_ROUTE_FORCED_FULL), + 0); + dm_edge(inc_db, "Main.Main", "R.ViaProps", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + ASSERT_EQ(dm_mentions_from(inc_db, "Main.Main"), 0); + + dm_unlink_db(inc_db); + dm_unlink_db(full_db); + th_rmtree(tmp); + PASS(); +} + +/* ── which scope changes are repairable file by file ─────────────── */ + +/* The scope delta between two versions of one C# file (NULL: the file does + * not exist); the removed names joined by ','. */ +static int dm_delta(const char *before, const char *after, char *names, size_t cap) { + return dm_scope_delta(CBM_LANG_CSHARP, "S.cs", before, after, names, cap); +} + +#define DM_DELTA_HEAD "using Acme.Local;\n" +#define DM_DELTA_OPEN "namespace N\n{\n public class W : Base\n {\n" +#define DM_DELTA_RUN_INT " public void Run(int n) { }\n" +#define DM_DELTA_RUN_STR " public void Run(string s) { }\n" +#define DM_DELTA_SIZE " public int Size;\n" +#define DM_DELTA_CLOSE " }\n public class Other { }\n}\n" + +TEST(doc_mentions_scope_delta_rules) { + const char *base = + DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_RUN_STR DM_DELTA_SIZE DM_DELTA_CLOSE; + struct { + const char *what; + const char *after; + int want; + const char *names; + } cases[] = { + {"unchanged", base, CBM_DOCLINK_DELTA_LOCAL, ""}, + /* nothing of the persisted scope carries a line number or a body */ + {"lines moved", + "// moved\n\n" DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_RUN_STR DM_DELTA_SIZE + DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_LOCAL, ""}, + {"body edit", + DM_DELTA_HEAD DM_DELTA_OPEN + " public void Run(int n) { n++; }\n" DM_DELTA_RUN_STR DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_LOCAL, ""}, + /* the file declares a type with a base list (`W : Base`), and a base + * list is resolved through the file's usings: what W derives from + * decides how other files' references to W's members come out */ + {"using added", + DM_DELTA_HEAD "using Acme.More;\n" DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_RUN_STR + DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + {"using removed", + DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_RUN_STR DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + {"static using and alias added", + DM_DELTA_HEAD "using static Acme.S;\nusing A = Acme.B;\n" DM_DELTA_OPEN DM_DELTA_RUN_INT + DM_DELTA_RUN_STR DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + /* ... a global using scopes every file of the project */ + {"global using added", + "global using Acme.G;\n" DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_RUN_STR + DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + /* a removed member is reported by name */ + {"overload removed", + DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_LOCAL, "Run"}, + {"field removed", + DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_RUN_STR DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_LOCAL, "Size"}, + {"two members removed", DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_LOCAL, "Run,Size"}, + /* everything else can re-route references of files with no edge here */ + {"member added", + DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_RUN_STR DM_DELTA_SIZE + " public int More;\n" DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + {"signature changed", + DM_DELTA_HEAD DM_DELTA_OPEN + " public void Run(long n) { }\n" DM_DELTA_RUN_STR DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + {"members reordered", + DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_STR DM_DELTA_RUN_INT DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + {"type removed", + DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_RUN_STR DM_DELTA_SIZE " }\n}\n", + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + {"type added", + DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_RUN_STR DM_DELTA_SIZE + " }\n public class Other { }\n public class New { }\n}\n", + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + {"base changed", + DM_DELTA_HEAD + "namespace N\n{\n public class W : Base2\n {\n" DM_DELTA_RUN_INT DM_DELTA_RUN_STR + DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + {"namespace renamed", + DM_DELTA_HEAD + "namespace M\n{\n public class W : Base\n {\n" DM_DELTA_RUN_INT DM_DELTA_RUN_STR + DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + /* a member the parser can no longer show makes its type incomplete */ + {"member hidden", + DM_DELTA_HEAD DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_RUN_STR + " public safe extern int Size();\n" DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL, NULL}, + {"file deleted", NULL, CBM_DOCLINK_DELTA_GLOBAL, NULL}, + }; + char names[256]; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + int got = dm_delta(base, cases[i].after, names, sizeof(names)); + if (got != cases[i].want || (cases[i].names && strcmp(names, cases[i].names) != 0)) { + printf(" %s: delta %d [%s], want %d [%s]\n", cases[i].what, got, names, cases[i].want, + cases[i].names ? cases[i].names : "any"); + FAIL("scope delta"); + } + } + /* A file that declares no type with a base list: its own usings scope + * nothing but the file itself, which is re-extracted anyway. */ +#define DM_DELTA_PLAIN "namespace N\n{\n public class W\n {\n" + const char *plain = DM_DELTA_HEAD DM_DELTA_PLAIN DM_DELTA_RUN_INT DM_DELTA_SIZE DM_DELTA_CLOSE; + struct { + const char *what; + const char *after; + int want; + } usings[] = { + {"plain: using added", + DM_DELTA_HEAD + "using Acme.More;\n" DM_DELTA_PLAIN DM_DELTA_RUN_INT DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_LOCAL}, + {"plain: using removed", DM_DELTA_PLAIN DM_DELTA_RUN_INT DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_LOCAL}, + {"plain: using changed", + "using Acme.Other;\n" DM_DELTA_PLAIN DM_DELTA_RUN_INT DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_LOCAL}, + {"plain: static using and alias added", + DM_DELTA_HEAD "using static Acme.S;\nusing A = Acme.B;\n" DM_DELTA_PLAIN DM_DELTA_RUN_INT + DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_LOCAL}, + /* a using inside a namespace declaration is the file's own too */ + {"plain: block using added", + DM_DELTA_HEAD + "namespace N\n{\n using Acme.In;\n public class W\n {\n" DM_DELTA_RUN_INT + DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_LOCAL}, + /* ... and with the edit that gives W a base list, the usings count */ + {"plain: base list and using added together", + DM_DELTA_HEAD + "using Acme.More;\n" DM_DELTA_OPEN DM_DELTA_RUN_INT DM_DELTA_SIZE DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL}, + {"plain: global static using added", + "global using static Acme.S;\n" DM_DELTA_HEAD DM_DELTA_PLAIN DM_DELTA_RUN_INT DM_DELTA_SIZE + DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL}, + {"plain: global alias added", + "global using A = Acme.B;\n" DM_DELTA_HEAD DM_DELTA_PLAIN DM_DELTA_RUN_INT DM_DELTA_SIZE + DM_DELTA_CLOSE, + CBM_DOCLINK_DELTA_GLOBAL}, + }; + for (size_t i = 0; i < sizeof(usings) / sizeof(usings[0]); i++) { + int got = dm_delta(plain, usings[i].after, names, sizeof(names)); + if (got != usings[i].want || names[0]) { + printf(" %s: delta %d [%s], want %d\n", usings[i].what, got, names, usings[i].want); + FAIL("scope delta of a file's own usings"); + } + } + /* a scope that appears is as global as one that leaves */ + ASSERT_EQ(dm_delta(NULL, base, names, sizeof(names)), CBM_DOCLINK_DELTA_GLOBAL); + /* no scope before and after: a file of a language without one */ + ASSERT_EQ(cbm_doclinks_scope_delta(NULL, NULL, dm_name_put, NULL), CBM_DOCLINK_DELTA_LOCAL); + /* a blob no resolver claims is never repaired file by file */ + dm_names_t none = {{0}}; + ASSERT_EQ( + cbm_doclinks_scope_delta("zz9\nM\t0\tc\t0\tW.Run\t\tint\n", "zz9\n", dm_name_put, &none), + CBM_DOCLINK_DELTA_GLOBAL); + ASSERT_STR_EQ(none.names, ""); + + /* The MSBuild files that set a project's global usings are no scope + * inputs outside the index: each has a scope blob, and the planner judges + * a change to one by that blob like any other file's. */ + ASSERT_FALSE(cbm_doclinks_is_scope_input("src/App/App.csproj")); + ASSERT_FALSE(cbm_doclinks_is_scope_input("Directory.Build.props")); + ASSERT_FALSE(cbm_doclinks_is_scope_input("src/App/App.cs")); + ASSERT_FALSE(cbm_doclinks_is_scope_input(NULL)); +#define DM_PROJ_HEAD "\n" +#define DM_PROJ_PROPS " enable\n" +#define DM_PROJ_USING " \n" +#define DM_PROJ_PKG " \n" +#define DM_PROJ_TAIL "\n" + const char *proj = DM_PROJ_HEAD DM_PROJ_PROPS DM_PROJ_USING DM_PROJ_PKG DM_PROJ_TAIL; + struct { + const char *what; + const char *after; + int want; + } projects[] = { + {"project unchanged", proj, CBM_DOCLINK_DELTA_LOCAL}, + /* what the blob does not hold changes nobody's scope */ + {"package version", + DM_PROJ_HEAD DM_PROJ_PROPS DM_PROJ_USING " \n" DM_PROJ_TAIL, + CBM_DOCLINK_DELTA_LOCAL}, + {"target added", + DM_PROJ_HEAD DM_PROJ_PROPS DM_PROJ_USING DM_PROJ_PKG + " \n" DM_PROJ_TAIL, + CBM_DOCLINK_DELTA_LOCAL}, + /* what it holds changes every file's of the project */ + {"using removed", DM_PROJ_HEAD DM_PROJ_PROPS DM_PROJ_PKG DM_PROJ_TAIL, + CBM_DOCLINK_DELTA_GLOBAL}, + {"property changed", + DM_PROJ_HEAD + " disable\n" DM_PROJ_USING DM_PROJ_PKG + DM_PROJ_TAIL, + CBM_DOCLINK_DELTA_GLOBAL}, + {"condition added", + DM_PROJ_HEAD DM_PROJ_PROPS + " \n" DM_PROJ_PKG DM_PROJ_TAIL, + CBM_DOCLINK_DELTA_GLOBAL}, + {"import added", + DM_PROJ_HEAD + " \n" DM_PROJ_PROPS DM_PROJ_USING DM_PROJ_PKG DM_PROJ_TAIL, + CBM_DOCLINK_DELTA_GLOBAL}, + {"no longer readable", DM_PROJ_HEAD DM_PROJ_PROPS, CBM_DOCLINK_DELTA_GLOBAL}, + {"project file deleted", NULL, CBM_DOCLINK_DELTA_GLOBAL}, + }; + for (size_t i = 0; i < sizeof(projects) / sizeof(projects[0]); i++) { + int got = dm_scope_delta(CBM_LANG_XML, "src/App.csproj", proj, projects[i].after, names, + sizeof(names)); + if (got != projects[i].want || names[0]) { + printf(" %s: delta %d [%s], want %d\n", projects[i].what, got, names, + projects[i].want); + FAIL("project scope delta"); + } + } + PASS(); +} + +/* ── the parallel path == the sequential path ────────────────────── */ + +enum { DM_RING = 64 }; /* above MIN_FILES_FOR_PARALLEL */ + +/* Every class documents its successor, a shared target (by name, by overload + * group and by signature) and a name nothing declares. */ +static void dm_write_ring_fixture(const char *repo) { + for (int i = 0; i < DM_RING; i++) { + char path[512]; + char body[1024]; + snprintf(path, sizeof(path), "%s/src/C%02d.cs", repo, i); + snprintf(body, sizeof(body), + "using Ring.Shared;\n" + "namespace Ring\n" + "{\n" + " /// Next , shared ,\n" + " /// , .\n" + " /// x\n" + " public class C%02d { }\n" + "}\n", + (i + 1) % DM_RING, i, i); + th_write_file(path, body); + } + th_write_file(TH_PATH(repo, "src/Hub.cs"), "namespace Ring.Shared\n" + "{\n" + " public class Hub\n" + " {\n" + " public void Run(int n) { }\n" + " public void Run(string s) { }\n" + " }\n" + "}\n"); +} + +/* The worker pipeline (extract workers, resolve workers, per-file rows merged + * at the end) and the sequential passes publish the same edges and rows. */ +TEST(doc_mentions_parallel_equals_sequential) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_par_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + dm_write_ring_fixture(repo); + char par_db[512]; + char seq_db[512]; + snprintf(par_db, sizeof(par_db), "%s/par.db", tmp); + snprintf(seq_db, sizeof(seq_db), "%s/seq.db", tmp); + ASSERT_EQ(dm_workers_agree(repo, par_db, seq_db), 0); + /* per class: successor (see), Hub (seealso), Hub.Run(int) (exception) */ + ASSERT_EQ(dm_count(par_db, "SELECT COUNT(*) FROM edges WHERE type = 'MENTIONS'"), DM_RING * 3); + /* per class: the overload group (ambiguous) and the undeclared name */ + ASSERT_EQ(dm_count(par_db, "SELECT COUNT(*) FROM doc_link_unresolved"), DM_RING * 2); + ASSERT_EQ( + dm_count(par_db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'ambiguous'"), + DM_RING); + ASSERT_EQ(dm_count(par_db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'missing'"), + DM_RING); + + dm_unlink_db(par_db); + dm_unlink_db(seq_db); + th_rmtree(tmp); + PASS(); +} + +/* ── the scope scanner's limits and cost ─────────────────────────── */ + +/* `prefix`, then `unit` n times, then `suffix`; the caller frees it. */ +static char *dm_repeated(const char *prefix, const char *unit, int n, const char *suffix) { + size_t pl = strlen(prefix); + size_t ul = strlen(unit); + size_t sl = strlen(suffix); + char *out = malloc(pl + (ul * (size_t)n) + sl + 1); + if (!out) { + return NULL; + } + memcpy(out, prefix, pl); + for (int i = 0; i < n; i++) { + memcpy(out + pl + (ul * (size_t)i), unit, ul); + } + memcpy(out + pl + (ul * (size_t)n), suffix, sl + 1); + return out; +} + +typedef struct { + char *src; + const char *rel_path; + CBMFileResult *result; +} dm_extract_job_t; + +/* cbm_thread_create body: extract job->src as C#, then free this thread's + * parser and caches as a worker thread does when it ends (LeakSanitizer + * reports them otherwise). */ +static void *dm_extract_thread(void *arg) { + dm_extract_job_t *job = (dm_extract_job_t *)arg; + job->result = dm_extract(job->src, CBM_LANG_CSHARP, job->rel_path); + cbm_destroy_thread_parser(); + cbm_kind_in_set_free_cache(); + return NULL; +} + +/* Interpolated strings nest: a hole of code can hold the next string. The + * brace scan follows them to a fixed depth; a file that goes deeper is not + * placed, and scanning it costs no stack. */ +TEST(doc_mentions_cs_scan_nested_holes) { + /* three levels, in a file whose tree has a parse error (the unsafe + * dereference): the braces are read from the text, through the holes */ + const char *nested = "namespace N\n" /* 1 */ + "{\n" + " public class Deep\n" /* 3 */ + " {\n" + " unsafe void E(void* p) { _r = ref *(int*)p; }\n" + " void M() { s = $\"a{$\"b{$\"c{1}\"}\"}\"; }\n" + " public int Q;\n" /* 7 */ + " }\n" /* 8 */ + " public class After { }\n" + "}\n"; + CBMFileResult *r = dm_extract(nested, CBM_LANG_CSHARP, "Nested.cs"); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + ASSERT_NOT_NULL(strstr(r->doc_scope, "T\t1\t3\t8\tc!\t-\tDeep\t")); + ASSERT_NOT_NULL(strstr(r->doc_scope, "\tAfter\t")); + ASSERT_NULL(strstr(r->doc_scope, "\nX\t")); + cbm_free_result(r); + + /* a hundred levels, every one closed again: deeper than the scan follows. + * The file is not placed, and its scope says so */ + enum { DM_DEEP = 100 }; + char *closers = dm_repeated("1", "}\"", DM_DEEP, "; }\n }\n}\n"); + ASSERT_NOT_NULL(closers); + char *deep = dm_repeated("namespace N\n" + "{\n" + " public class Deep\n" + " {\n" + " unsafe void E(void* p) { _r = ref *(int*)p; }\n" + " void M() { s = ", + "$\"{", DM_DEEP, closers); + free(closers); + ASSERT_NOT_NULL(deep); + r = dm_extract(deep, CBM_LANG_CSHARP, "Hundred.cs"); + free(deep); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + ASSERT_NOT_NULL(strstr(r->doc_scope, "\nX\t")); + ASSERT_NOT_NULL(strstr(r->doc_scope, "Q\tDeep\n")); + ASSERT_NULL(strstr(r->doc_scope, "\nT\t")); + cbm_free_result(r); + PASS(); +} + +/* ... and following them costs no stack: 8,000 levels left open, extracted on + * a thread with an eighth of an index worker's stack. A scan that follows + * them all needs one recursion level per three bytes of source. */ +TEST(doc_mentions_cs_scan_holes_stack) { + enum { DM_HOLES = 8000, DM_SMALL_STACK = 1024 * 1024 }; + dm_extract_job_t job = {.src = dm_repeated("namespace N\n" + "{\n" + " public class Ok { }\n" + " public class C\n" + " {\n" + " string s = ", + "$\"{", DM_HOLES, "\n"), + .rel_path = "Holes.cs"}; + ASSERT_NOT_NULL(job.src); + cbm_thread_t thread; + ASSERT_EQ(cbm_thread_create(&thread, DM_SMALL_STACK, dm_extract_thread, &job), 0); + ASSERT_EQ(cbm_thread_join(&thread), 0); + free(job.src); + CBMFileResult *r = job.result; + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + /* nothing is placed: the namespace's brace never closes */ + ASSERT_NOT_NULL(strstr(r->doc_scope, "\nX\t")); + ASSERT_NULL(strstr(r->doc_scope, "\nT\t")); + ASSERT_NULL(strstr(r->doc_scope, "\nR\t")); + ASSERT_NOT_NULL(strstr(r->doc_scope, "Q\tOk\n")); + cbm_free_result(r); + PASS(); +} + +/* The cost of scanning `src` as C#: text positions, brace-stack entries and + * modifier children visited, bytes taken from the scratch arena. false when the file has no scope. + */ +static bool dm_scan_cost(const char *src, uint64_t *steps, uint64_t *bytes) { + cbm_doclink_cs_test_cost_reset(); + CBMFileResult *r = src ? dm_extract(src, CBM_LANG_CSHARP, "Cost.cs") : NULL; + bool ok = r && r->doc_scope; + cbm_doclink_cs_test_cost(steps, bytes); + if (r) { + cbm_free_result(r); + } + return ok; +} + +/* The scan's cost of an input twice as large, as a multiple of the smaller + * one's: about 2 for a scan that is linear, 4 for a quadratic one. -1 when a + * scan fails. `prefix` + `unit` x n + `mid` + `unit2` x n + `suffix`. */ +static double dm_cost_growth(const char *prefix, const char *unit, const char *mid, + const char *unit2, int n, bool bytes) { + uint64_t cost[2] = {0, 0}; + for (int k = 0; k < 2; k++) { + int reps = n * (k + 1); + char *tail = dm_repeated(mid, unit2, reps, "\n"); + char *src = tail ? dm_repeated(prefix, unit, reps, tail) : NULL; + uint64_t steps = 0; + uint64_t scratch = 0; + bool ok = dm_scan_cost(src, &steps, &scratch); + free(tail); + free(src); + if (!ok) { + return -1.0; + } + cost[k] = bytes ? scratch : steps; + } + return cost[0] ? (double)cost[1] / (double)cost[0] : -1.0; +} + +/* The scan's work grows with its input, not faster: no clock decides these, + * the scanner's own counters do. */ + +/* A run of `$` that starts no string is passed once, not once per `$`. */ +TEST(doc_mentions_cs_scan_dollar_run) { + double growth = dm_cost_growth("class C { int x = ", "$", "; }", "", 20000, false); + if (!(growth > 0 && growth < 3.0)) { + printf(" `$` run: twice the input costs %.2f times the steps\n", growth); + FAIL("the scan of a `$` run is not linear"); + } + PASS(); +} + +/* A conditional remembers where the open braces stood, not a copy of them. */ +TEST(doc_mentions_cs_scan_branch_memory) { + double growth = + dm_cost_growth("class C { void M() {\n", "{", "\n", "#if X\n#endif\n", 3000, true); + if (!(growth > 0 && growth < 3.0)) { + printf(" #if under open braces: twice the input takes %.2f times the memory\n", growth); + FAIL("the memory of the brace scan is not linear"); + } + PASS(); +} + +/* Empty sibling branches cannot revisit a deep stack that predates them. */ +TEST(doc_mentions_cs_scan_branch_merge_work) { + enum { DEPTH = 512, SIBLINGS = 512 }; + char *opened = dm_repeated("class C { void M() {\n", "{", DEPTH, "\n#if A\n"); + ASSERT_NOT_NULL(opened); + char *closed = dm_repeated(opened, "}", DEPTH, "\n"); + free(opened); + ASSERT_NOT_NULL(closed); + char *reopened = dm_repeated(closed, "{", DEPTH, "\n"); + free(closed); + ASSERT_NOT_NULL(reopened); + char *branches = dm_repeated(reopened, "#elif B\n", SIBLINGS, "#endif\n"); + free(reopened); + ASSERT_NOT_NULL(branches); + char *src = dm_repeated(branches, "}", DEPTH, "\n} }\n"); + free(branches); + ASSERT_NOT_NULL(src); + uint64_t steps = 0; + uint64_t scratch = 0; + bool ok = dm_scan_cost(src, &steps, &scratch); + size_t bytes = strlen(src); + free(src); + ASSERT_TRUE(ok); + ASSERT_LTE(steps, (uint64_t)bytes * 16); + PASS(); +} + +/* Attributes belong to the declaration, not to each variable it declares. */ +TEST(doc_mentions_cs_scan_declarator_modifier_work) { + enum { ATTRIBUTES = 512, DECLARATORS = 512 }; + char *head = dm_repeated("class C {\n", "[A]\n", ATTRIBUTES, "public int "); + ASSERT_NOT_NULL(head); + size_t cap = strlen(head) + DECLARATORS * 16 + 16; + char *src = malloc(cap); + ASSERT_NOT_NULL(src); + size_t pos = (size_t)snprintf(src, cap, "%s", head); + free(head); + for (int i = 0; i < DECLARATORS; i++) { + pos += (size_t)snprintf(src + pos, cap - pos, "%sv%d", i ? "," : "", i); + } + pos += (size_t)snprintf(src + pos, cap - pos, ";\n}\n"); + cbm_doclink_cs_test_cost_reset(); + CBMFileResult *r = dm_extract(src, CBM_LANG_CSHARP, "Fields.cs"); + free(src); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + /* Require the grammar to expose every declarator before assessing work. */ + int members = 0; + const char *p = r->doc_scope; + while ((p = strstr(p, "\nM\t")) != NULL) { + members++; + p++; + } + uint64_t steps = 0; + uint64_t scratch = 0; + cbm_doclink_cs_test_cost(&steps, &scratch); + cbm_free_result(r); + ASSERT_EQ(members, DECLARATORS); + ASSERT_LTE(steps, (uint64_t)pos * 16); + PASS(); +} + +/* Sharing a large comment must not copy or parse its prose per declarator, + * and its references are taken once, from the first declarator (S4). */ +TEST(doc_mentions_cs_shared_doc_work) { + bool bounded = true; + for (int n = 16; n <= 32; n *= 2) { + for (int prose = 2048; prose <= 4096; prose *= 2) { + char *head = + dm_repeated("class C {\n/// ", "x", prose, " \npublic int "); + ASSERT_NOT_NULL(head); + size_t cap = strlen(head) + (size_t)n * 16 + 16; + char *src = malloc(cap); + ASSERT_NOT_NULL(src); + size_t len = (size_t)snprintf(src, cap, "%s", head); + free(head); + for (int i = 0; i < n; i++) { + len += (size_t)snprintf(src + len, cap - len, "%sv%d", i ? "," : "", i); + } + len += (size_t)snprintf(src + len, cap - len, ";\n}\n"); + cbm_doclink_test_doc_work_reset(); + CBMFileResult *r = dm_extract(src, CBM_LANG_CSHARP, "Shared.cs"); + free(src); + ASSERT_NOT_NULL(r); + int tokens = r->doc_links.count; + uint64_t copied = 0, parse_input = 0, cleaned = 0; + cbm_doclink_test_doc_work(&copied, &parse_input, &cleaned); + cbm_free_result(r); + printf(" shared doc: n=%d prose=%d input=%zu copied=%llu parse_input=%llu " + "cleaned=%llu tokens=%d\n", + n, prose, len, (unsigned long long)copied, (unsigned long long)parse_input, + (unsigned long long)cleaned, tokens); + ASSERT_EQ(tokens, 1); + bounded = bounded && copied <= (uint64_t)len * 2 && parse_input <= (uint64_t)len * 2 && + cleaned <= 12; + } + } + ASSERT_TRUE(bounded); + PASS(); +} + +/* A failed allocation while a shared doc comment is read loses references: + * the file's doc links say so (S17), and the later declarators do not take + * the comment again -- neither from the shared text nor by a lookup of their + * own (S4). Without a failure the first declarator has every reference. */ +TEST(doc_mentions_cs_shared_doc_allocation_failure) { + static const struct { + int kind, nth, refs; + } cases[] = { + {CBM_DOCLINK_ALLOC_KINDS, 0, 2}, /* no failure */ + {CBM_DOCLINK_ALLOC_VALUE, 2, 2}, {CBM_DOCLINK_ALLOC_TOKENS, 2, 20}, + {CBM_DOCLINK_ALLOC_TEXT, 1, 2}, {CBM_DOCLINK_ALLOC_SPAN, 2, 12}, + }; + bool ok = true; + for (size_t k = 0; k < sizeof(cases) / sizeof(cases[0]); k++) { + char *src = dm_repeated("class C {\n", "/// \n", cases[k].refs, + "public int a,b,c;\n}\n"); + ASSERT_NOT_NULL(src); + bool fail = cases[k].kind < CBM_DOCLINK_ALLOC_KINDS; + if (fail) { + cbm_doclink_test_fail_alloc_after(cases[k].kind, cases[k].nth); + } + CBMFileResult *r = dm_extract(src, CBM_LANG_CSHARP, "Failure.cs"); + cbm_doclink_test_reset_alloc(); + free(src); + ASSERT_NOT_NULL(r); + int counts[3] = {0}; + for (int i = 0; i < r->doc_links.count; i++) { + const char *name = strrchr(r->doc_links.items[i].source_qn, '.'); + ASSERT_NOT_NULL(name); + name++; + ASSERT_TRUE(name[0] >= 'a' && name[0] <= 'c' && name[1] == '\0'); + counts[name[0] - 'a']++; + } + bool failed = r->doc_links.failed; + cbm_free_result(r); + printf(" shared doc allocation: stage=%d counts=%d,%d,%d failed=%d\n", cases[k].kind, + counts[0], counts[1], counts[2], failed); + bool first = fail ? counts[0] < cases[k].refs : counts[0] == cases[k].refs; + ok = ok && failed == fail && first && counts[1] == 0 && counts[2] == 0; + } + ASSERT_TRUE(ok); + PASS(); +} + +/* Each shared comment yields its references once, from the first declarator + * of its own declaration -- distinct comments and files stay distinct. */ +TEST(doc_mentions_cs_shared_doc_replay) { + enum { REFS = 80 }; + char *first = dm_repeated("class C {\n/// ", " ", REFS, + "\npublic int a,\nb,\nc;\n/// "); + ASSERT_NOT_NULL(first); + char *src = dm_repeated(first, " ", REFS, "\npublic int d,\ne,\nf;\n}\n"); + free(first); + ASSERT_NOT_NULL(src); + CBMFileResult *r = dm_extract(src, CBM_LANG_CSHARP, "Shared.cs"); + free(src); + ASSERT_NOT_NULL(r); + ASSERT_EQ(r->doc_links.count, 2 * REFS); + int counts[6] = {0}; + for (int i = 0; i < r->doc_links.count; i++) { + const CBMDocLink *link = &r->doc_links.items[i]; + const char *name = strrchr(link->source_qn, '.'); + ASSERT_NOT_NULL(name); + name++; + ASSERT_TRUE(name[0] >= 'a' && name[0] <= 'f' && name[1] == '\0'); + int n = name[0] - 'a'; + ASSERT_STR_EQ(link->raw, "First"); + ASSERT_EQ(link->line, n < 3 ? 2 : 6); + ASSERT_EQ(link->def_line, (uint32_t)(n < 3 ? n + 3 : n + 4)); + ASSERT_EQ(link->syntax, CBM_DOCLINK_CS_SEE); + ASSERT_EQ(link->flags, 0); + counts[n]++; + } + cbm_free_result(r); + for (int i = 0; i < 6; i++) { + ASSERT_EQ(counts[i], (i == 0 || i == 3) ? REFS : 0); + } + r = dm_extract("class C {\n/// \npublic int a,b,c;\n}\n", CBM_LANG_CSHARP, + "Shared.cs"); + ASSERT_NOT_NULL(r); + ASSERT_EQ(r->doc_links.count, 1); + ASSERT_STR_EQ(r->doc_links.items[0].raw, "Other"); + ASSERT_EQ(r->doc_links.items[0].line, 2); + cbm_free_result(r); + PASS(); +} + +/* A declaration keyword's header is read up to the next keyword, so every + * byte of the file is read a bounded number of times. */ +TEST(doc_mentions_cs_scan_header_reads) { + const char *heads[] = {"class a ", "class a<[ "}; + for (size_t i = 0; i < sizeof(heads) / sizeof(heads[0]); i++) { + char *src = dm_repeated("namespace N {\n", heads[i], 20000, "\n"); + ASSERT_NOT_NULL(src); + uint64_t steps = 0; + uint64_t scratch = 0; + bool ok = dm_scan_cost(src, &steps, &scratch); + size_t len = strlen(src); + free(src); + ASSERT_TRUE(ok); + if (steps > (uint64_t)len * 16) { + printf(" `%s` x 20000: %llu steps for %zu bytes\n", heads[i], + (unsigned long long)steps, len); + FAIL("declaration headers are read over and over"); + } + } + PASS(); +} + +/* The gate of the project scan is the file's name: an XML file that cannot be + * an MSBuild project file is not looked at, whatever its size, and a project + * file larger than a project file is not read and says so. No clock decides + * this: the scan counts the bytes it passes. */ +TEST(doc_mentions_msbuild_gate) { + enum { DM_BIG = 2 * 1024 * 1024 }; + char *filler = malloc(DM_BIG + 1); + ASSERT_NOT_NULL(filler); + memset(filler, 'x', DM_BIG); + filler[DM_BIG] = '\0'; + char *big = + dm_repeated("

", filler, 1, "

\n"); + free(filler); + ASSERT_NOT_NULL(big); + uint64_t steps = 0; + uint64_t bytes = 0; + /* not a project file's name: no scope, and not one byte passed */ + cbm_doclink_cs_test_cost_reset(); + CBMFileResult *r = dm_extract(big, CBM_LANG_XML, "data/huge.xml"); + ASSERT_NOT_NULL(r); + ASSERT_NULL(r->doc_scope); + cbm_free_result(r); + cbm_doclink_cs_test_cost(&steps, &bytes); + ASSERT_EQ(steps, 0); + /* a project file's name on a file larger than one: not read either, and + * the blob says that it was not */ + r = dm_extract(big, CBM_LANG_XML, "eng/Huge.props"); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + ASSERT_STR_EQ(r->doc_scope, "cs1\nP\t\t>\n"); + cbm_free_result(r); + cbm_doclink_cs_test_cost(&steps, &bytes); + ASSERT_EQ(steps, 0); + /* a project file: every byte is passed once */ + const char *small = "\n"; + r = dm_extract(small, CBM_LANG_XML, "eng/Small.targets"); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + cbm_free_result(r); + cbm_doclink_cs_test_cost(&steps, &bytes); + ASSERT_EQ(steps, strlen(small)); + /* XML of another kind under a project file's name: the scan stops at its + * root element */ + cbm_doclink_cs_test_cost_reset(); + char *other = dm_repeated("", "", 4000, ""); + ASSERT_NOT_NULL(other); + r = dm_extract(other, CBM_LANG_XML, "eng/Other.props"); + free(other); + ASSERT_NOT_NULL(r); + ASSERT_NULL(r->doc_scope); + cbm_free_result(r); + cbm_doclink_cs_test_cost(&steps, &bytes); + ASSERT_EQ(steps, strlen("")); + + /* a file that was not read is counted where it is evaluated, as the + * project file itself or as an import; what it holds is unknown, so the + * scope of the project that evaluates it is open (S16) */ + const dm_project_file_t files[] = { + {"eng/Huge.props", big}, + {"eng/Broken.props", ""}, + {"App.csproj", "\n" + " \n" + " \n" + " \n" + "\n"}, + }; + cbm_msb_result_t res; + ASSERT_TRUE(dm_msb_eval(files, 3, "App.csproj", &res)); + ASSERT_TRUE(dm_has_using(&res, 'n', "Still.Here")); + ASSERT_EQ(res.unevaluable, 2); + ASSERT_TRUE(res.open); + cbm_msb_result_free(&res); + ASSERT_TRUE(dm_msb_eval(files, 2, "eng/Broken.props", &res)); + ASSERT_EQ(res.count, 0); + ASSERT_EQ(res.unevaluable, 1); + cbm_msb_result_free(&res); + free(big); + PASS(); +} + +/* ── MSBuild inputs are files of the index, never of the disk ────── */ + +#ifndef _WIN32 +static const char DM_USING_EXTRA[] = "\n" + " \n" + " \n" + " \n" + "\n"; +static const char DM_USING_MORE[] = "\n" + " \n" + " \n" + " \n" + "\n"; + +/* A repository whose project files point out of it: `src/App/Evil.csproj` is a + * symbolic link to a project file outside, and `src/Lib/Lib.csproj` imports + * through `src/Lib/ext`, a link to a directory outside. Both outside files + * hold a that would bring a type of the repository into scope. */ +static void dm_write_linked_fixture(const char *repo, const char *outside) { + th_write_file(TH_PATH(outside, "Real.csproj"), DM_USING_EXTRA); + th_write_file(TH_PATH(outside, "dir/Linked.props"), DM_USING_MORE); + th_write_file(TH_PATH(repo, "src/Decl.cs"), "namespace Acme.Extra\n" + "{\n" + " public class Tool { }\n" + "}\n" + "namespace Acme.More\n" + "{\n" + " public class Gear { }\n" + "}\n"); + th_write_file(TH_PATH(repo, "src/App/Uses.cs"), + "namespace Acme.App\n" + "{\n" + " /// \n" + " public class Uses { }\n" + "}\n"); + th_write_file(TH_PATH(repo, "src/Lib/Lib.csproj"), "\n" + " \n" + "\n"); + th_write_file(TH_PATH(repo, "src/Lib/UsesLib.cs"), + "namespace Acme.Lib\n" + "{\n" + " /// \n" + " public class UsesLib { }\n" + "}\n"); +} + +/* Index the linked fixture under `tmp` into `db` (a buffer of 512 bytes). */ +static int dm_index_linked_fixture(const char *tmp, char *db) { + char repo[400]; + char outside[400]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + snprintf(outside, sizeof(outside), "%s/outside", tmp); + dm_write_linked_fixture(repo, outside); + char target[512]; + snprintf(target, sizeof(target), "%s/Real.csproj", outside); + if (symlink(target, TH_PATH(repo, "src/App/Evil.csproj")) != 0) { + return -1; + } + snprintf(target, sizeof(target), "%s/dir", outside); + if (symlink(target, TH_PATH(repo, "src/Lib/ext")) != 0) { + return -1; + } + /* a named pipe with a project file's name: opening it would block */ + if (mkfifo(TH_PATH(repo, "src/Lib/Pipe.csproj"), 0600) != 0) { + return -1; + } + snprintf(db, 512, "%s/lnk.db", tmp); + return dm_index(repo, db, NULL); +} + +/* What is not a file of the repository is not read: discovery skips symbolic + * links, and the resolver takes project files from the index alone. A project + * file that is a link to a file outside has no in effect. */ +TEST(doc_mentions_msbuild_linked_project_file) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_lnk_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char db[512]; + ASSERT_EQ(dm_index_linked_fixture(tmp, db), 0); + char props[512]; + char reason[64]; + int n = 0; + dm_edge(db, "Uses.Uses", "Decl.Tool", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + dm_row(db, "src/App/Uses.cs", "Tool", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + dm_unlink_db(db); + th_rmtree(tmp); + PASS(); +} + +/* ... and neither has a file imported through a linked directory. */ +TEST(doc_mentions_msbuild_linked_import_directory) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_lnd_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char db[512]; + ASSERT_EQ(dm_index_linked_fixture(tmp, db), 0); + char props[512]; + char reason[64]; + int n = 0; + dm_edge(db, "UsesLib.UsesLib", "Decl.Gear", props, sizeof(props), &n); + ASSERT_EQ(n, 0); + dm_row(db, "src/Lib/UsesLib.cs", "Gear", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + dm_unlink_db(db); + th_rmtree(tmp); + PASS(); +} +#endif /* !_WIN32 */ + +/* ── the C# lookup rules, one small repository per rule ──────────── */ + +typedef struct { + const char *path; + const char *text; +} dm_source_t; + +/* What one written reference must come to. `rel` and `raw` name the + * reference (a raw text is written once per file, or comes to the same row + * every time), `src` the documented definition. + * target an edge from `src` to the node whose qualified name ends so, and + * no row; `tier` (when set) is in the edge's properties + * reason a row with that reason + * neither no row: the reference is local, or only `never` is asked + * never a node `src` must not mention */ +typedef struct { + const char *rel; + const char *raw; + const char *src; + const char *target; + const char *reason; + const char *never; + const char *tier; +} dm_want_t; + +#define DM_EXACT "\"tier\":\"exact\"" +#define DM_UNIQUE "\"tier\":\"unique\"" + +/* 1 when the database does not hold what `w` asks for (and says what it + * holds instead), else 0. */ +static int dm_want_failed(const char *db, const dm_want_t *w) { + char props[512]; + char reason[64]; + int n = 0; + int bad = 0; + dm_row(db, w->rel, w->raw, reason, sizeof(reason), NULL, 0); + if (strcmp(reason, w->reason ? w->reason : "") != 0) { + printf(" `%s` in %s: row [%s], want [%s]\n", w->raw, w->rel, reason, + w->reason ? w->reason : ""); + bad = 1; + } + if (w->target) { + dm_edge(db, w->src, w->target, props, sizeof(props), &n); + if (n != 1) { + printf(" `%s` in %s: %d edges %s -> %s, want 1\n", w->raw, w->rel, n, w->src, + w->target); + bad = 1; + } else if (w->tier && !strstr(props, w->tier)) { + printf(" `%s` in %s: edge %s, want %s\n", w->raw, w->rel, props, w->tier); + bad = 1; + } + } + if (w->never) { + dm_edge(db, w->src, w->never, props, sizeof(props), &n); + if (n != 0) { + printf(" `%s` in %s: %s -> %s is bound, want no such edge\n", w->raw, w->rel, w->src, + w->never); + bad = 1; + } + } + return bad; +} + +/* Index `files` as a repository of their own and hold the result against + * `wants`. Returns how many of them failed (each is printed), -1 when the + * repository could not be indexed. */ +static int dm_check_repo(const char *tag, const dm_source_t *files, int nfiles, + const dm_want_t *wants, int nwants) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_%s_XXXXXX", tag); + if (!cbm_mkdtemp(tmp)) { + return -1; + } + for (int i = 0; i < nfiles; i++) { + th_write_file(TH_PATH(tmp, files[i].path), files[i].text); + } + char db[512]; + snprintf(db, sizeof(db), "%s/lookup.db", tmp); + int bad = dm_index(tmp, db, NULL) == 0 ? 0 : -1; + for (int i = 0; bad >= 0 && i < nwants; i++) { + bad += dm_want_failed(db, &wants[i]); + } + dm_unlink_db(db); + th_rmtree(tmp); + return bad; +} + +#define DM_COUNT(a) ((int)(sizeof(a) / sizeof((a)[0]))) + +/* Variables documented together by one comment: the comment's references + * are the first declarator's (S4); the other declarators mention nothing. */ +TEST(doc_mentions_cs_shared_doc_sources) { + static const char source[] = "namespace N {\n" + "public class First { } public class Second { }\n" + "public class C {\n" + "/// \n" + "public int a,\n" + "b,\n" + "c;\n" + "}\n}\n"; + CBMFileResult *r = dm_extract(source, CBM_LANG_CSHARP, "Shared.cs"); + ASSERT_NOT_NULL(r); + ASSERT_EQ(r->doc_links.count, 2); + int pairs[3][2] = {{0}}; + for (int i = 0; i < r->doc_links.count; i++) { + const CBMDocLink *link = &r->doc_links.items[i]; + const char *name = strrchr(link->source_qn, '.'); + ASSERT_NOT_NULL(name); + name++; + int src = strcmp(name, "a") == 0 ? 0 + : strcmp(name, "b") == 0 ? 1 + : strcmp(name, "c") == 0 ? 2 + : -1; + int dst = strcmp(link->raw, "First") == 0 ? 0 : strcmp(link->raw, "Second") == 0 ? 1 : -1; + ASSERT_TRUE(src >= 0 && dst >= 0); + ASSERT_EQ(link->line, 4); + ASSERT_EQ(link->def_line, (uint32_t)(5 + src)); + pairs[src][dst]++; + } + cbm_free_result(r); + for (int i = 0; i < 3; i++) { + for (int j = 0; j < 2; j++) { + ASSERT_EQ(pairs[i][j], i == 0 ? 1 : 0); + } + } + const dm_source_t files[] = {{"Shared.cs", source}}; + static const dm_want_t wants[] = { + {"Shared.cs", "First", "C.a", "First", NULL, NULL, NULL}, + {"Shared.cs", "Second", "C.a", "Second", NULL, NULL, NULL}, + {"Shared.cs", "First", "C.b", NULL, NULL, "First", NULL}, + {"Shared.cs", "Second", "C.b", NULL, NULL, "Second", NULL}, + {"Shared.cs", "First", "C.c", NULL, NULL, "First", NULL}, + {"Shared.cs", "Second", "C.c", NULL, NULL, "Second", NULL}, + }; + ASSERT_EQ(dm_check_repo("shared_doc", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* S4: tokens, resolutions and rows of a doc comment shared by the + * declarators of one field declaration grow with its references, not with + * references x declarators. Doubling both the declarators and the + * references doubles the tokens and the rows, no more. */ +static int dm_shared_rows(int declarators, int refs, int *tokens) { + size_t cap = (size_t)(declarators + refs) * 40 + 256; + char *src = malloc(cap); + if (!src) { + return -1; + } + size_t w = (size_t)snprintf(src, cap, "namespace N\n{\n public class C\n {\n ///"); + for (int i = 0; i < refs; i++) { + w += (size_t)snprintf(src + w, cap - w, " ", i); + } + w += (size_t)snprintf(src + w, cap - w, "\n public int v0"); + for (int i = 1; i < declarators; i++) { + w += (size_t)snprintf(src + w, cap - w, ", v%d", i); + } + snprintf(src + w, cap - w, ";\n }\n}\n"); + CBMFileResult *r = dm_extract(src, CBM_LANG_CSHARP, "Fields.cs"); + *tokens = r ? r->doc_links.count : -1; + cbm_free_result(r); + const dm_source_t files[] = {{"Fields.cs", src}}; + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_s4rows_XXXXXX"); + int rows = -1; + if (cbm_mkdtemp(tmp)) { + th_write_file(TH_PATH(tmp, files[0].path), files[0].text); + char db[512]; + snprintf(db, sizeof(db), "%s/rows.db", tmp); + if (dm_index(tmp, db, NULL) == 0) { + rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"); + } + dm_unlink_db(db); + th_rmtree(tmp); + } + free(src); + return rows; +} + +TEST(doc_mentions_cs_shared_doc_once) { + int tokens_small = 0; + int tokens_large = 0; + int rows_small = dm_shared_rows(8, 8, &tokens_small); + int rows_large = dm_shared_rows(16, 16, &tokens_large); + printf(" shared doc: 8 x 8 -> %d tokens, %d rows; 16 x 16 -> %d tokens, %d rows\n", + tokens_small, rows_small, tokens_large, rows_large); + ASSERT_EQ(tokens_small, 8); + ASSERT_EQ(rows_small, 8); + ASSERT_EQ(tokens_large, 16); + ASSERT_EQ(rows_large, 16); + PASS(); +} + +/* The order of the scope levels: an inner namespace declaration's own usings + * and aliases are asked before an outer namespace's types; a qualified name + * and a using's target are relative to the namespaces around them before + * they are absolute; and a nearer namespace that has the first segment is + * the end of the search. */ +TEST(doc_mentions_cs_lookup_order) { + static const dm_source_t files[] = { + {"src/AcmeLogger.cs", "namespace Acme\n{\n public class Logger { }\n}\n"}, + {"src/WidgetsLogger.cs", "namespace Acme.Widgets\n" + "{\n" + " public class Logger { }\n" + " public class OnlyWidgets { }\n" + "}\n"}, + {"src/UtilGlobal.cs", "namespace Util\n" + "{\n" + " public class Helper { }\n" + " public class OnlyGlobal { }\n" + "}\n"}, + {"src/UtilAcme.cs", "namespace Acme.Util\n{\n public class Helper { }\n}\n"}, + {"src/BlockUsing.cs", "namespace Acme.App\n" + "{\n" + " using Acme.Widgets;\n" + "\n" + " /// \n" + " public class BlockUsing { }\n" + "}\n"}, + {"src/BlockAlias.cs", "namespace Acme.App\n" + "{\n" + " using Logger = Acme.Widgets.OnlyWidgets;\n" + "\n" + " /// \n" + " public class BlockAlias { }\n" + "}\n"}, + {"src/FileUsing.cs", "using Acme.Widgets;\n" + "\n" + "namespace Acme.App\n" + "{\n" + " /// \n" + " public class FileUsing { }\n" + "}\n"}, + {"src/Relative.cs", + "namespace Acme.App\n" + "{\n" + " /// \n" + " public class Relative { }\n" + "\n" + " /// \n" + " public class Absolute { }\n" + "}\n"}, + {"src/RelativeUsing.cs", "namespace Acme.App\n" + "{\n" + " using Widgets;\n" + " using W = Widgets.OnlyWidgets;\n" + "\n" + " /// \n" + " public class RelativeUsing { }\n" + "\n" + " /// \n" + " public class RelativeAlias { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + /* the block's using comes before the outer namespace Acme */ + {"src/BlockUsing.cs", "Logger", "BlockUsing.BlockUsing", "WidgetsLogger.Logger", NULL, + "AcmeLogger.Logger", DM_UNIQUE}, + /* ... and so does its alias */ + {"src/BlockAlias.cs", "Logger", "BlockAlias.BlockAlias", "WidgetsLogger.OnlyWidgets", NULL, + "AcmeLogger.Logger", DM_EXACT}, + /* the same using at the top of the file comes after every namespace */ + {"src/FileUsing.cs", "Logger", "FileUsing.FileUsing", "AcmeLogger.Logger", NULL, + "WidgetsLogger.Logger", NULL}, + /* `Util` is Acme.Util from inside Acme.App, not the global Util */ + {"src/Relative.cs", "Util.Helper", "Relative.Relative", "UtilAcme.Helper", NULL, + "UtilGlobal.Helper", DM_EXACT}, + /* ... and Acme.Util is where the search ends: no second try further out */ + {"src/Relative.cs", "Util.OnlyGlobal", "Relative.Relative", NULL, "missing", + "UtilGlobal.OnlyGlobal", NULL}, + {"src/Relative.cs", "global::Util.Helper", "Relative.Absolute", "UtilGlobal.Helper", NULL, + "UtilAcme.Helper", DM_EXACT}, + /* a using's target and an alias's are relative to their block */ + {"src/RelativeUsing.cs", "OnlyWidgets", "RelativeUsing.RelativeUsing", + "WidgetsLogger.OnlyWidgets", NULL, NULL, DM_UNIQUE}, + {"src/RelativeUsing.cs", "W", "RelativeUsing.RelativeAlias", "WidgetsLogger.OnlyWidgets", + NULL, NULL, DM_EXACT}, + }; + ASSERT_EQ(dm_check_repo("order", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* A type parameter in scope shadows a type of its name: a reference to it + * names the definition's own parameter, and is neither an edge nor a row. */ +TEST(doc_mentions_cs_type_parameters) { + static const dm_source_t files[] = { + {"src/Generic.cs", + "namespace Acme.Gen\n" + "{\n" + " public class TItem { }\n" + " public class TKey { }\n" + "\n" + " /// \n" + " public class Bag\n" + " {\n" + " /// \n" + " public void Put(TKey key) { }\n" + "\n" + " /// \n" + " public void Other() { }\n" + " }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + /* on the type: its own parameter is local, the class TKey is a class */ + {"src/Generic.cs", "TItem", "Generic.Bag", NULL, NULL, "Generic.TItem", NULL}, + {"src/Generic.cs", "TKey", "Generic.Bag", "Generic.TKey", NULL, NULL, NULL}, + /* on the generic method: both names are parameters in scope */ + {"src/Generic.cs", "TItem", "Generic.Bag.Put", NULL, NULL, "Generic.TItem", NULL}, + {"src/Generic.cs", "TKey", "Generic.Bag.Put", NULL, NULL, "Generic.TKey", NULL}, + /* on a method without the parameter: the class again */ + {"src/Generic.cs", "TKey", "Generic.Bag.Other", "Generic.TKey", NULL, NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("tparam", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +#define DM_EMPTY_PROJECT "\n\n" + +/* `using static` brings in a type's nested types and its static members + * (no instance member, no extension method). A `global using` -- of a + * namespace, of a type's static members, of an alias alike -- and a project + * file's serve every file of their project, and no other project. */ +TEST(doc_mentions_cs_usings) { + static const dm_source_t files[] = { + {"src/Lib/Maths.cs", "namespace Acme.Calc\n" + "{\n" + " public static class Maths\n" + " {\n" + " public static int Max(int a, int b) { return a; }\n" + " public static int Extension(this string s) { return 0; }\n" + " public const int Limit = 1;\n" + " public class Nested { }\n" + " }\n" + " public class Shape\n" + " {\n" + " public int Instance() { return 0; }\n" + " public static int Area() { return 0; }\n" + " }\n" + " public enum Color { Red }\n" + "}\n"}, + {"src/App/UseStatic.cs", + "using static Acme.Calc.Maths;\n" + "using static Acme.Calc.Shape;\n" + "using static Acme.Calc.Color;\n" + "\n" + "namespace Acme.App\n" + "{\n" + " /// \n" + " /// \n" + " /// \n" + " public class UseStatic { }\n" + "}\n"}, + {"src/Proj/Proj.csproj", DM_EMPTY_PROJECT}, + {"src/Proj/Globals.cs", "global using Acme.Calc;\n" + "global using static Acme.Calc.Maths;\n" + "global using GC = Acme.Calc.Color;\n"}, + {"src/Proj/UseGlobal.cs", + "namespace Proj.App\n" + "{\n" + " /// \n" + " public class UseGlobal { }\n" + "}\n"}, + {"src/Other/Other.csproj", DM_EMPTY_PROJECT}, + {"src/Other/NoGlobal.cs", + "namespace Other.App\n" + "{\n" + " /// \n" + " public class NoGlobal { }\n" + "}\n"}, + {"src/Msb/Msb.csproj", "\n" + " \n" + " \n" + " \n" + " \n" + "\n"}, + {"src/Msb/UseMsb.cs", + "namespace Msb.App\n" + "{\n" + " /// \n" + " public class UseMsb { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"src/App/UseStatic.cs", "Max", "UseStatic.UseStatic", "Maths.Maths.Max", NULL, NULL, + DM_UNIQUE}, + {"src/App/UseStatic.cs", "Limit", "UseStatic.UseStatic", "Maths.Maths.Limit", NULL, NULL, + NULL}, + {"src/App/UseStatic.cs", "Nested", "UseStatic.UseStatic", "Maths.Maths.Nested", NULL, NULL, + NULL}, + {"src/App/UseStatic.cs", "Area", "UseStatic.UseStatic", "Maths.Shape.Area", NULL, NULL, + NULL}, + {"src/App/UseStatic.cs", "Red", "UseStatic.UseStatic", "Maths.Color.Red", NULL, NULL, NULL}, + /* an instance member and an extension method do not come with it */ + {"src/App/UseStatic.cs", "Instance", "UseStatic.UseStatic", NULL, "missing", + "Maths.Shape.Instance", NULL}, + {"src/App/UseStatic.cs", "Extension", "UseStatic.UseStatic", NULL, "missing", + "Maths.Maths.Extension", NULL}, + /* the three kinds of `global using`, from another file of the project */ + {"src/Proj/UseGlobal.cs", "Shape", "UseGlobal.UseGlobal", "Maths.Shape", NULL, NULL, NULL}, + {"src/Proj/UseGlobal.cs", "Max", "UseGlobal.UseGlobal", "Maths.Maths.Max", NULL, NULL, + NULL}, + {"src/Proj/UseGlobal.cs", "GC", "UseGlobal.UseGlobal", "Maths.Color", NULL, NULL, DM_EXACT}, + /* ... and not from another project */ + {"src/Other/NoGlobal.cs", "Shape", "NoGlobal.NoGlobal", NULL, "missing", "Maths.Shape", + NULL}, + {"src/Other/NoGlobal.cs", "Max", "NoGlobal.NoGlobal", NULL, "missing", "Maths.Maths.Max", + NULL}, + {"src/Other/NoGlobal.cs", "GC", "NoGlobal.NoGlobal", NULL, "missing", "Maths.Color", NULL}, + /* a project file's and */ + {"src/Msb/UseMsb.cs", "Max", "UseMsb.UseMsb", "Maths.Maths.Max", NULL, NULL, NULL}, + {"src/Msb/UseMsb.cs", "MC", "UseMsb.UseMsb", "Maths.Color", NULL, NULL, DM_EXACT}, + {"src/Msb/UseMsb.cs", "Shape", "UseMsb.UseMsb", NULL, "missing", "Maths.Shape", NULL}, + }; + ASSERT_EQ(dm_check_repo("usings", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* What a type is: its name, its arity, the arity of every type around it, and + * its assembly. `Outer.Inner` and `Outer.Inner` are two types; two + * projects that each declare `Acme.Shared.Config` declare two types; the + * parts of a partial type in one place are one type. */ +TEST(doc_mentions_cs_entities) { + static const dm_source_t files[] = { + {"src/Twins.cs", + "namespace Acme.Ent\n" + "{\n" + " public class Outer { public class Inner { public void A() { } } }\n" + " public class Outer { public class Inner { public void B() { } } }\n" + "\n" + " /// \n" + " /// \n" + " public class UsesPlain { }\n" + "\n" + " /// \n" + " /// \n" + " public class UsesGeneric { }\n" + "}\n"}, + {"src/Part1.cs", "namespace Acme.Parts\n" + "{\n" + " public partial class Split { public void First() { } }\n" + "}\n"}, + {"src/Part2.cs", "namespace Acme.Parts\n" + "{\n" + " public partial class Split\n" + " {\n" + " public void Second() { }\n" + " public class In { public class Deep { } }\n" + " }\n" + "\n" + " /// \n" + " /// \n" + " public class UsesSplit { }\n" + "}\n"}, + {"src/P1/P1.csproj", DM_EMPTY_PROJECT}, + {"src/P1/Config.cs", + "namespace Acme.Shared\n" + "{\n" + " public class Config { public void OnlyOne() { } }\n" + "\n" + " /// \n" + " /// \n" + " public class UsesOne { }\n" + "}\n"}, + {"src/P2/P2.csproj", DM_EMPTY_PROJECT}, + {"src/P2/Config.cs", + "namespace Acme.Shared\n" + "{\n" + " public class Config { public void OnlyTwo() { } }\n" + "\n" + " /// \n" + " /// \n" + " public class UsesTwo { }\n" + "}\n"}, + {"src/P3/P3.csproj", DM_EMPTY_PROJECT}, + {"src/P3/Uses.cs", "namespace Acme.Shared\n" + "{\n" + " /// \n" + " public class UsesThree { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + /* `Outer.Inner` is the type nested in the arity-0 Outer. The node of + * that path belongs to the later declaration, the one in Outer: a + * graph gap, never the twin's node. Its members are its own. */ + {"src/Twins.cs", "Outer.Inner", "Twins.UsesPlain", NULL, "graph_gap", "Twins.Outer.Inner", + NULL}, + {"src/Twins.cs", "Outer.Inner.A", "Twins.UsesPlain", "Twins.Outer.Inner.A", NULL, NULL, + NULL}, + {"src/Twins.cs", "Outer.Inner.B", "Twins.UsesPlain", NULL, "missing", "Twins.Outer.Inner.B", + NULL}, + {"src/Twins.cs", "Outer{T}.Inner", "Twins.UsesGeneric", "Twins.Outer.Inner", NULL, NULL, + NULL}, + {"src/Twins.cs", "Outer{T}.Inner.B", "Twins.UsesGeneric", "Twins.Outer.Inner.B", NULL, NULL, + NULL}, + {"src/Twins.cs", "Outer{T}.Inner.A", "Twins.UsesGeneric", NULL, "missing", + "Twins.Outer.Inner.A", NULL}, + /* a partial type over two files: one type, the members of both, and + * the node of the declaration nearest to the reference */ + {"src/Part2.cs", "Split", "Part2.UsesSplit", "Part2.Split", NULL, "Part1.Split", NULL}, + {"src/Part2.cs", "Split.First", "Part2.UsesSplit", "Part1.Split.First", NULL, NULL, NULL}, + {"src/Part2.cs", "Split.Second", "Part2.UsesSplit", "Part2.Split.Second", NULL, NULL, NULL}, + /* nested types: by their path, never by their simple name from outside */ + {"src/Part2.cs", "Split.In", "Part2.UsesSplit", "Part2.Split.In", NULL, NULL, NULL}, + {"src/Part2.cs", "Split.In.Deep", "Part2.UsesSplit", "Part2.Split.In.Deep", NULL, NULL, + NULL}, + {"src/Part2.cs", "In", "Part2.UsesSplit", NULL, "missing", NULL, NULL}, + /* one project's Config is not the other's: each sees its own type + * and its own type's members */ + {"src/P1/Config.cs", "Config", "P1.Config.UsesOne", "P1.Config.Config", NULL, + "P2.Config.Config", NULL}, + {"src/P1/Config.cs", "Config.OnlyOne", "P1.Config.UsesOne", "P1.Config.Config.OnlyOne", + NULL, NULL, NULL}, + {"src/P1/Config.cs", "Config.OnlyTwo", "P1.Config.UsesOne", NULL, "missing", + "P2.Config.Config.OnlyTwo", NULL}, + {"src/P2/Config.cs", "Config", "P2.Config.UsesTwo", "P2.Config.Config", NULL, + "P1.Config.Config", NULL}, + {"src/P2/Config.cs", "Config.OnlyOne", "P2.Config.UsesTwo", NULL, "missing", + "P1.Config.Config.OnlyOne", NULL}, + /* a third project sees two types of that name: neither is chosen */ + {"src/P3/Uses.cs", "Config", "Uses.UsesThree", NULL, "ambiguous", "P1.Config.Config", NULL}, + }; + ASSERT_EQ(dm_check_repo("ent", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* An assembly is named by its project file: projects of one name are one + * assembly. The parts they declare are one type; complete declarations in + * two of them are flavours -- the referencing file's own project's binds, and + * from anywhere else none is chosen. Decoys: a part in a project of another + * name is no part of it, and a directory that holds two project files is an + * assembly of its own, whatever they are called. */ +TEST(doc_mentions_cs_assemblies_by_name) { + static const dm_source_t files[] = { + {"kit/win/Acme.Kit.csproj", DM_EMPTY_PROJECT}, + {"kit/win/EngineWin.cs", + "namespace Acme.Kit\n" + "{\n" + " public partial class Engine { public void Start() { } }\n" + " public class Clock { public void Tick() { } }\n" + "\n" + " /// \n" + " /// \n" + " /// \n" + " public class UsesWin { }\n" + "}\n"}, + {"kit/unix/Acme.Kit.csproj", DM_EMPTY_PROJECT}, + {"kit/unix/EngineUnix.cs", "namespace Acme.Kit\n" + "{\n" + " public partial class Engine { public void Stop() { } }\n" + " public class Clock { public void Tick() { } }\n" + "}\n"}, + {"kit/tools/Acme.Kit.csproj", DM_EMPTY_PROJECT}, + {"kit/tools/Tools.cs", + "namespace Acme.Kit\n" + "{\n" + " /// \n" + " /// \n" + " public class UsesTools { }\n" + "}\n"}, + {"other/Acme.Other.csproj", DM_EMPTY_PROJECT}, + {"other/EngineOther.cs", "namespace Acme.Kit\n" + "{\n" + " public partial class Engine { public void Halt() { } }\n" + " public class Solo { }\n" + "}\n"}, + {"pair/Acme.Kit.csproj", DM_EMPTY_PROJECT}, + {"pair/Second.csproj", DM_EMPTY_PROJECT}, + {"pair/EnginePair.cs", "namespace Acme.Kit\n" + "{\n" + " public partial class Engine { public void Pair() { } }\n" + "}\n"}, + {"app/App.csproj", DM_EMPTY_PROJECT}, + {"app/UsesApp.cs", + "namespace Acme.App\n" + "{\n" + " /// \n" + " public class UsesApp { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + /* the parts of two projects of one name are one type */ + {"kit/win/EngineWin.cs", "Engine.Stop", "EngineWin.UsesWin", "unix.EngineUnix.Engine.Stop", + NULL, NULL, NULL}, + {"kit/tools/Tools.cs", "Engine.Start", "Tools.UsesTools", "win.EngineWin.Engine.Start", + NULL, NULL, NULL}, + /* decoys: another name, and a directory with two project files */ + {"kit/win/EngineWin.cs", "Engine.Halt", "EngineWin.UsesWin", NULL, "missing", + "EngineOther.Engine.Halt", NULL}, + {"kit/win/EngineWin.cs", "Engine.Pair", "EngineWin.UsesWin", NULL, "missing", + "EnginePair.Engine.Pair", NULL}, + /* flavours: the own project's declaration and its member */ + {"kit/win/EngineWin.cs", "Clock", "EngineWin.UsesWin", "win.EngineWin.Clock", NULL, + "unix.EngineUnix.Clock", NULL}, + {"kit/win/EngineWin.cs", "Clock.Tick", "EngineWin.UsesWin", "win.EngineWin.Clock.Tick", + NULL, "unix.EngineUnix.Clock.Tick", NULL}, + /* ... and from a project that has none of them, of the same assembly + * or of another, none is chosen */ + {"kit/tools/Tools.cs", "Clock", "Tools.UsesTools", NULL, "ambiguous", "win.EngineWin.Clock", + NULL}, + {"kit/tools/Tools.cs", "Clock.Tick", "Tools.UsesTools", NULL, "ambiguous", + "win.EngineWin.Clock.Tick", NULL}, + {"app/UsesApp.cs", "Acme.Kit.Clock", "UsesApp.UsesApp", NULL, "ambiguous", + "unix.EngineUnix.Clock", NULL}, + /* what one other assembly declares binds */ + {"app/UsesApp.cs", "Acme.Kit.Solo", "UsesApp.UsesApp", "EngineOther.Solo", NULL, NULL, + DM_EXACT}, + }; + ASSERT_EQ(dm_check_repo("asm", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* Inside one assembly a declaration in a project directory named `ref` -- a + * reference assembly's source -- stands behind the implementation: the + * implementation's node binds, the stub's only where the implementation has + * none (here: `Hidden` and `Hidden` share one node in the implementation's + * file, and it is `Hidden`'s). Decoy: a directory named `ref` inside a + * project is no reference assembly, and what stands in it is no stub. */ +TEST(doc_mentions_cs_reference_source) { + static const dm_source_t files[] = { + {"lib/ref/Acme.Lib.csproj", DM_EMPTY_PROJECT}, + {"lib/ref/Stubs.cs", + "namespace Acme.Lib\n" + "{\n" + " public partial class Widget { public void Spin() { } public void OnlyStub() { } }\n" + " public partial class Hidden { public void Peek() { } }\n" + "}\n"}, + {"lib/src/Acme.Lib.csproj", DM_EMPTY_PROJECT}, + {"lib/src/Widget.cs", "namespace Acme.Lib\n" + "{\n" + " public class Widget { public void Spin() { } }\n" + "}\n"}, + {"lib/src/Hidden.cs", "namespace Acme.Lib\n" + "{\n" + " public class Hidden { public void Peek() { } }\n" + " public class Hidden { }\n" + "}\n"}, + {"lib/src/ref/Helper.cs", "namespace Acme.Lib\n" + "{\n" + " public partial class Helper { public void Near() { } }\n" + "\n" + " /// \n" + " public class UsesHelper { }\n" + "}\n"}, + {"lib/src/HelperMore.cs", "namespace Acme.Lib\n" + "{\n" + " public partial class Helper { public void Far() { } }\n" + "}\n"}, + {"app/App.csproj", DM_EMPTY_PROJECT}, + {"app/UsesApp.cs", + "namespace Acme.App\n" + "{\n" + " /// \n" + " /// \n" + " /// \n" + " public class UsesApp { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + /* stub and implementation are one type: the implementation's nodes */ + {"app/UsesApp.cs", "Acme.Lib.Widget", "UsesApp.UsesApp", "src.Widget.Widget", NULL, + "ref.Stubs.Widget", DM_EXACT}, + {"app/UsesApp.cs", "Acme.Lib.Widget.Spin", "UsesApp.UsesApp", "src.Widget.Widget.Spin", + NULL, "ref.Stubs.Widget.Spin", NULL}, + /* the stub's, where the implementation has none */ + {"app/UsesApp.cs", "Acme.Lib.Widget.OnlyStub", "UsesApp.UsesApp", + "ref.Stubs.Widget.OnlyStub", NULL, NULL, NULL}, + {"app/UsesApp.cs", "Acme.Lib.Hidden", "UsesApp.UsesApp", "ref.Stubs.Hidden", NULL, + "src.Hidden.Hidden", NULL}, + {"app/UsesApp.cs", "Acme.Lib.Hidden.Peek", "UsesApp.UsesApp", "src.Hidden.Hidden.Peek", + NULL, "ref.Stubs.Hidden.Peek", NULL}, + /* decoy: both parts are implementation, the nearest is the one beside + * the reference */ + {"lib/src/ref/Helper.cs", "Helper", "ref.Helper.UsesHelper", "ref.Helper.Helper", NULL, + "HelperMore.Helper", NULL}, + }; + ASSERT_EQ(dm_check_repo("refsrc", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* Declarations of one full name in projects of different names are different + * types, and a project that declares none of them is not told which one it + * references: nothing chooses -- not the one that is a stub, not the one that + * is not, not the nearer directory. Decoys: the declaring assembly sees its + * own type, and a name only one other assembly declares binds. */ +TEST(doc_mentions_cs_assemblies_differ) { + static const dm_source_t files[] = { + {"impl/Acme.Impl.csproj", DM_EMPTY_PROJECT}, + {"impl/Gadget.cs", "namespace Acme.Things\n" + "{\n" + " public class Gadget { public void Run() { } }\n" + " public class OnlyImpl { }\n" + "\n" + " /// \n" + " public class UsesImpl { }\n" + "}\n"}, + {"facade/ref/Acme.Facade.csproj", DM_EMPTY_PROJECT}, + {"facade/ref/Stubs.cs", "namespace Acme.Things\n" + "{\n" + " public partial class Gadget { public void Run() { } }\n" + "}\n"}, + {"impl/near/Near.csproj", DM_EMPTY_PROJECT}, + {"impl/near/UsesNear.cs", "namespace Acme.Near\n" + "{\n" + " /// \n" + " /// \n" + " public class UsesNear { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"impl/near/UsesNear.cs", "Acme.Things.Gadget", "UsesNear.UsesNear", NULL, "ambiguous", + "impl.Gadget.Gadget", NULL}, + {"impl/near/UsesNear.cs", "Acme.Things.Gadget", "UsesNear.UsesNear", NULL, "ambiguous", + "ref.Stubs.Gadget", NULL}, + {"impl/near/UsesNear.cs", "Acme.Things.Gadget.Run", "UsesNear.UsesNear", NULL, "ambiguous", + "impl.Gadget.Gadget.Run", NULL}, + {"impl/near/UsesNear.cs", "Acme.Things.OnlyImpl", "UsesNear.UsesNear", + "impl.Gadget.OnlyImpl", NULL, NULL, DM_EXACT}, + {"impl/Gadget.cs", "Gadget", "Gadget.UsesImpl", "impl.Gadget.Gadget", NULL, + "ref.Stubs.Gadget", NULL}, + }; + ASSERT_EQ(dm_check_repo("differ", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* A file no project file stands above is of a shared tree, and the parts of + * a partial type there are parts of every assembly's partial type of that + * name: what a reference sees is its own assembly's parts and the shared + * trees'. A part in a project of another name is no part of it, and a name + * such a part declares is, seen without it, neither bound nor missing. + * Decoy: a complete type takes no shared parts. */ +TEST(doc_mentions_cs_shared_parts) { + static const dm_source_t files[] = { + {"shared/kit/ToolShared.cs", + "namespace Acme.Kit\n" + "{\n" + " public partial class Tool { public void Common() { } }\n" + " public partial class Gear\n" + " {\n" + " /// \n" + " public void Turn() { }\n" + " public void Mesh() { }\n" + " }\n" + " public class Spare { }\n" + "\n" + " /// \n" + " public class UsesShared { }\n" + "}\n"}, + {"one/One.csproj", DM_EMPTY_PROJECT}, + {"one/ToolOne.cs", "namespace Acme.Kit\n" + "{\n" + " public partial class Tool { public void OnlyOne() { } }\n" + " public partial class Gear\n" + " {\n" + " public void OnlyOne() { }\n" + " public void Mesh(int teeth) { }\n" + " public int Spare;\n" + " public class Inner { public void Deep() { } }\n" + " }\n" + "\n" + " /// \n" + " /// \n" + " public class UsesOne { }\n" + "}\n"}, + {"two/Two.csproj", DM_EMPTY_PROJECT}, + {"two/ToolTwo.cs", "namespace Acme.Kit\n" + "{\n" + " public partial class Tool { public void OnlyTwo() { } }\n" + "}\n"}, + {"three/Three.csproj", DM_EMPTY_PROJECT}, + {"three/UsesThree.cs", + "namespace Acme.Kit\n" + "{\n" + " /// \n" + " /// \n" + " /// \n" + " public class UsesThree { }\n" + "}\n"}, + {"four/Four.csproj", DM_EMPTY_PROJECT}, + {"four/ToolFour.cs", + "namespace Acme.Kit\n" + "{\n" + " public class Tool { public void OnlyFour() { } }\n" + "\n" + " /// \n" + " public class UsesFour { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + /* an assembly's partial type: its own parts and the shared trees' */ + {"one/ToolOne.cs", "Tool", "ToolOne.UsesOne", "one.ToolOne.Tool", NULL, "two.ToolTwo.Tool", + NULL}, + {"one/ToolOne.cs", "Tool.Common", "ToolOne.UsesOne", "kit.ToolShared.Tool.Common", NULL, + NULL, NULL}, + {"one/ToolOne.cs", "Tool.OnlyOne", "ToolOne.UsesOne", "one.ToolOne.Tool.OnlyOne", NULL, + NULL, NULL}, + /* ... and not another assembly's */ + {"one/ToolOne.cs", "Tool.OnlyTwo", "ToolOne.UsesOne", NULL, "missing", + "two.ToolTwo.Tool.OnlyTwo", NULL}, + /* an assembly without parts of its own sees the shared parts */ + {"three/UsesThree.cs", "Gear", "UsesThree.UsesThree", "kit.ToolShared.Gear", NULL, NULL, + NULL}, + {"three/UsesThree.cs", "Gear.Turn", "UsesThree.UsesThree", "kit.ToolShared.Gear.Turn", NULL, + NULL, NULL}, + /* a name another assembly's part declares: not bound, not missing */ + {"three/UsesThree.cs", "Gear.OnlyOne", "UsesThree.UsesThree", NULL, "ambiguous", + "one.ToolOne.Gear.OnlyOne", NULL}, + {"three/UsesThree.cs", "Gear.Gone", "UsesThree.UsesThree", NULL, "missing", NULL, NULL}, + /* ... also where the shared parts have a member of that name, and + * where something further out has: the name is the part's first */ + {"three/UsesThree.cs", "Gear.Mesh", "UsesThree.UsesThree", NULL, "ambiguous", + "kit.ToolShared.Gear.Mesh", NULL}, + {"shared/kit/ToolShared.cs", "Spare", "ToolShared.Gear.Turn", NULL, "ambiguous", + "kit.ToolShared.Spare", NULL}, + /* ... and a type nested in such a part, on the way to its member */ + {"three/UsesThree.cs", "Gear.Inner.Deep", "UsesThree.UsesThree", NULL, "ambiguous", + "one.ToolOne.Gear.Inner.Deep", NULL}, + /* ... and so from the shared tree itself */ + {"shared/kit/ToolShared.cs", "Gear.Turn", "ToolShared.UsesShared", + "kit.ToolShared.Gear.Turn", NULL, NULL, NULL}, + {"shared/kit/ToolShared.cs", "Gear.OnlyOne", "ToolShared.UsesShared", NULL, "ambiguous", + "one.ToolOne.Gear.OnlyOne", NULL}, + /* decoy: a complete type has no further parts */ + {"four/ToolFour.cs", "Tool.Common", "ToolFour.UsesFour", NULL, "missing", + "kit.ToolShared.Tool.Common", NULL}, + {"four/ToolFour.cs", "Tool.OnlyFour", "ToolFour.UsesFour", "four.ToolFour.Tool.OnlyFour", + NULL, NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("parts", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* A shared tree is the unit of the files no project file stands above: the + * largest directory around them with no project file in or below it. Its + * files see its declarations as their own, whatever directory of it they + * stand in; two trees' declarations of one name are not chosen between; one + * tree's partial and complete declarations of a name are one type. Decoy: a + * project is one unit whatever its directories. */ +TEST(doc_mentions_cs_shared_trees) { + static const dm_source_t files[] = { + {"core/sys/Thing.cs", "namespace Acme.Sys\n" + "{\n" + " public class Thing { public void InCore() { } }\n" + " public class OnlyCore { }\n" + " public partial class Part { public void Same() { } }\n" + "}\n"}, + {"core/sys/Wrap.cs", "namespace Acme.Sys\n" + "{\n" + " public partial class Wrap\n" + " {\n" + " /// \n" + " public void Real() { }\n" + " }\n" + "}\n"}, + {"core/sys/WrapOther.cs", "namespace Acme.Sys\n" + "{\n" + " public class Wrap { public void Real() { } }\n" + "}\n"}, + {"core/text/UsesCore.cs", + "namespace Acme.Sys.Text\n" + "{\n" + " /// \n" + " /// \n" + " public class UsesCore { }\n" + "}\n"}, + {"base/sys/Thing.cs", "namespace Acme.Sys\n" + "{\n" + " public class Thing { public void InBase() { } }\n" + " public partial class Part { public void Same() { } }\n" + "}\n"}, + {"base/UsesBase.cs", "namespace Acme.Sys\n" + "{\n" + " /// \n" + " public class UsesBase { }\n" + "}\n"}, + {"app/App.csproj", DM_EMPTY_PROJECT}, + {"app/UsesApp.cs", + "namespace Acme.App\n" + "{\n" + " /// \n" + " /// \n" + " public class UsesApp { }\n" + "}\n"}, + {"app/inner/Deep.cs", "namespace Acme.App\n{\n public class Deep { }\n}\n"}, + {"app/other/UsesDeep.cs", "namespace Acme.App\n" + "{\n" + " /// \n" + " public class UsesDeep { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + /* its own tree's declaration, from another directory of the tree */ + {"core/text/UsesCore.cs", "Thing", "UsesCore.UsesCore", "core.sys.Thing.Thing", NULL, + "base.sys.Thing.Thing", NULL}, + {"core/text/UsesCore.cs", "Thing.InCore", "UsesCore.UsesCore", + "core.sys.Thing.Thing.InCore", NULL, NULL, NULL}, + {"core/text/UsesCore.cs", "Thing.InBase", "UsesCore.UsesCore", NULL, "missing", + "base.sys.Thing.Thing.InBase", NULL}, + {"base/UsesBase.cs", "Thing", "UsesBase.UsesBase", "base.sys.Thing.Thing", NULL, + "core.sys.Thing.Thing", NULL}, + /* parts in two trees: the own tree's part and member */ + {"core/text/UsesCore.cs", "Part", "UsesCore.UsesCore", "core.sys.Thing.Part", NULL, + "base.sys.Thing.Part", NULL}, + {"core/text/UsesCore.cs", "Part.Same", "UsesCore.UsesCore", "core.sys.Thing.Part.Same", + NULL, "base.sys.Thing.Part.Same", NULL}, + /* one tree's partial and complete declarations of a name: one type */ + {"core/sys/Wrap.cs", "Wrap", "Wrap.Wrap.Real", "core.sys.Wrap.Wrap", NULL, "WrapOther.Wrap", + NULL}, + /* from outside the trees nothing is chosen between two of them */ + {"app/UsesApp.cs", "Acme.Sys.Thing", "UsesApp.UsesApp", NULL, "ambiguous", + "core.sys.Thing.Thing", NULL}, + {"app/UsesApp.cs", "Acme.Sys.Part", "UsesApp.UsesApp", NULL, "ambiguous", + "core.sys.Thing.Part", NULL}, + {"app/UsesApp.cs", "Acme.Sys.Part.Same", "UsesApp.UsesApp", NULL, "ambiguous", + "core.sys.Thing.Part.Same", NULL}, + /* ... and what one tree declares binds */ + {"app/UsesApp.cs", "Acme.Sys.OnlyCore", "UsesApp.UsesApp", "core.sys.Thing.OnlyCore", NULL, + NULL, DM_EXACT}, + /* decoy: the directories of a project are one unit */ + {"app/other/UsesDeep.cs", "Deep", "UsesDeep.UsesDeep", "inner.Deep.Deep", NULL, NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("trees", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* A repository without any project file is one unit: a `global using` in one + * directory serves the files of every other one. */ +TEST(doc_mentions_cs_no_project_files) { + static const dm_source_t files[] = { + {"far/Far.cs", "namespace Acme.Far\n{\n public class Remote { }\n}\n"}, + {"conf/Globals.cs", "global using Acme.Far;\n"}, + {"use/Uses.cs", "namespace Acme.Use\n" + "{\n" + " /// \n" + " public class Uses { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"use/Uses.cs", "Remote", "Uses.Uses", "far.Far.Remote", NULL, NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("noproj", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* A contract and its one implementation are one type: an assembly whose + * every declaration of a type stands in a project directory named `ref`, and + * the one declaration of that name and arity the shared trees hold. The + * implementation's node binds, the stub's only where the implementation has + * none, and no binding that exists only through the join is exact. Decoys: + * two shared trees declare the name; the assembly has a declaration that is + * no stub; the arity differs. */ +TEST(doc_mentions_cs_contract_join) { + static const dm_source_t files[] = { + {"shared/impl/Bag.cs", "namespace Acme.Coll\n" + "{\n" + " public class Bag { public void Add() { } }\n" + " public class Twice { }\n" + " public class Mix { public void Shared() { } }\n" + " public class Pair { }\n" + " public partial class Spread { public void A() { } }\n" + " public class Rivaled { }\n" + "}\n"}, + {"shared/impl/Ghost.cs", "namespace Acme.Coll\n" + "{\n" + " public class Ghost { public void Boo() { } }\n" + " public class Ghost { }\n" + "}\n"}, + {"second/Twice.cs", "namespace Acme.Coll\n" + "{\n" + " public class Twice { }\n" + " public partial class Spread { public void B() { } }\n" + "}\n"}, + {"facade/ref/Acme.Facade.csproj", DM_EMPTY_PROJECT}, + {"facade/ref/Stubs.cs", + "namespace Acme.Coll\n" + "{\n" + " public partial class Bag { public void Add() { } public void OnlyStub() { } }\n" + " public partial class Twice { }\n" + " public partial class Pair { }\n" + " public partial class Ghost { public void Boo() { } }\n" + " public partial class Spread { }\n" + " public partial class Rivaled { }\n" + "\n" + " /// \n" + " /// \n" + " public partial class UsesFacade { }\n" + "\n" + " /// \n" + " /// \n" + " public partial class UsesFacadeFull { }\n" + "}\n"}, + {"mixed/ref/Acme.Mixed.csproj", DM_EMPTY_PROJECT}, + {"mixed/ref/Stubs.cs", "namespace Acme.Coll\n" + "{\n" + " public partial class Mix { public void Own() { } }\n" + "}\n"}, + {"mixed/src/Acme.Mixed.csproj", DM_EMPTY_PROJECT}, + {"mixed/src/Mix.cs", + "namespace Acme.Coll\n" + "{\n" + " public class Mix { public void Own() { } }\n" + "\n" + " /// \n" + " public class UsesMixed { }\n" + "}\n"}, + {"rival/Acme.Rival.csproj", DM_EMPTY_PROJECT}, + {"rival/Rivaled.cs", "namespace Acme.Coll\n{\n public class Rivaled { }\n}\n"}, + {"app/App.csproj", DM_EMPTY_PROJECT}, + {"app/UsesApp.cs", + "namespace Acme.App\n" + "{\n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " public class UsesApp { }\n" + "}\n"}, + {"app/UsesAlias.cs", "using B = Acme.Coll.Bag;\n" + "\n" + "namespace Acme.App\n" + "{\n" + " /// \n" + " public class UsesAlias { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + /* the stub is no second type: the implementation binds, and never as + * an exact binding -- by its full name, by an alias */ + {"app/UsesApp.cs", "Acme.Coll.Bag", "UsesApp.UsesApp", "impl.Bag.Bag", NULL, + "ref.Stubs.Bag", DM_UNIQUE}, + {"app/UsesApp.cs", "Acme.Coll.Bag.Add", "UsesApp.UsesApp", "impl.Bag.Bag.Add", NULL, + "ref.Stubs.Bag.Add", DM_UNIQUE}, + {"app/UsesAlias.cs", "B", "UsesAlias.UsesAlias", "impl.Bag.Bag", NULL, NULL, DM_UNIQUE}, + /* from the contract's own assembly: the implementation's node, and + * the stub's own member where the implementation has none */ + {"facade/ref/Stubs.cs", "Bag", "Stubs.UsesFacade", "impl.Bag.Bag", NULL, "ref.Stubs.Bag", + DM_UNIQUE}, + {"facade/ref/Stubs.cs", "Bag.OnlyStub", "Stubs.UsesFacade", "ref.Stubs.Bag.OnlyStub", NULL, + NULL, NULL}, + {"facade/ref/Stubs.cs", "Acme.Coll.Bag", "Stubs.UsesFacadeFull", "impl.Bag.Bag", NULL, + "ref.Stubs.Bag", DM_UNIQUE}, + {"facade/ref/Stubs.cs", "Acme.Coll.Bag.Add", "Stubs.UsesFacadeFull", "impl.Bag.Bag.Add", + NULL, "ref.Stubs.Bag.Add", DM_UNIQUE}, + /* the implementation without a node: the stub's node stands in */ + {"app/UsesApp.cs", "Acme.Coll.Ghost", "UsesApp.UsesApp", "ref.Stubs.Ghost", NULL, + "impl.Ghost.Ghost", DM_UNIQUE}, + /* decoy: two shared trees declare the name -- complete in each, or a + * part in each: no join, and the contract's assembly sees its stub */ + {"app/UsesApp.cs", "Acme.Coll.Twice", "UsesApp.UsesApp", NULL, "ambiguous", + "impl.Bag.Twice", NULL}, + {"facade/ref/Stubs.cs", "Twice", "Stubs.UsesFacade", "ref.Stubs.Twice", NULL, + "impl.Bag.Twice", NULL}, + {"facade/ref/Stubs.cs", "Spread", "Stubs.UsesFacade", "ref.Stubs.Spread", NULL, + "impl.Bag.Spread", NULL}, + /* decoy: another assembly holds an implementation of the name too -- + * the contract is not joined to one of two implementations */ + {"facade/ref/Stubs.cs", "Rivaled", "Stubs.UsesFacade", "ref.Stubs.Rivaled", NULL, + "impl.Bag.Rivaled", NULL}, + /* decoy: the assembly has a declaration that is no stub */ + {"app/UsesApp.cs", "Acme.Coll.Mix", "UsesApp.UsesApp", NULL, "ambiguous", "impl.Bag.Mix", + NULL}, + {"mixed/src/Mix.cs", "Mix.Own", "Mix.UsesMixed", "src.Mix.Mix.Own", NULL, NULL, NULL}, + {"mixed/src/Mix.cs", "Mix.Shared", "Mix.UsesMixed", NULL, "missing", "impl.Bag.Mix.Shared", + NULL}, + /* decoy: another arity is another type -- each is the only one of its + * arity, and bound without the join */ + {"app/UsesApp.cs", "Acme.Coll.Pair", "UsesApp.UsesApp", "impl.Bag.Pair", NULL, NULL, + DM_EXACT}, + {"app/UsesApp.cs", "Acme.Coll.Pair{T}", "UsesApp.UsesApp", "ref.Stubs.Pair", NULL, NULL, + DM_EXACT}, + }; + ASSERT_EQ(dm_check_repo("join", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* A project file of an SDK that compiles nothing (NoTargets: it only runs + * build steps) is the project of no source file: the sources below it that + * have no nearer project are a shared tree, and a contract finds its one + * implementation there. Decoy: a project file that compiles keeps the files + * below it, and their types are that assembly's. */ +TEST(doc_mentions_cs_idle_project_file) { + static const dm_source_t files[] = { + {"libs/build.csproj", "\n\n"}, + {"libs/core/src/Impl.cs", "namespace Acme.Core\n" + "{\n" + " public class Thing { public void Go() { } }\n" + "}\n"}, + {"libs/core/ref/Acme.Core.csproj", DM_EMPTY_PROJECT}, + {"libs/core/ref/Stubs.cs", "namespace Acme.Core\n" + "{\n" + " public partial class Thing { public void Go() { } }\n" + "}\n"}, + {"libs/user/User.csproj", DM_EMPTY_PROJECT}, + {"libs/user/Uses.cs", + "namespace Acme.User\n" + "{\n" + " /// \n" + " /// \n" + " public class Uses { }\n" + "}\n"}, + {"apps/Apps.csproj", DM_EMPTY_PROJECT}, + {"apps/tool/src/Impl.cs", "namespace Acme.Apps\n{\n public class Gizmo { }\n}\n"}, + {"apps/tool/ref/Tool.csproj", DM_EMPTY_PROJECT}, + {"apps/tool/ref/Stubs.cs", "namespace Acme.Apps\n" + "{\n" + " public partial class Gizmo { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"libs/user/Uses.cs", "Acme.Core.Thing", "Uses.Uses", "core.src.Impl.Thing", NULL, + "core.ref.Stubs.Thing", DM_UNIQUE}, + {"libs/user/Uses.cs", "Acme.Apps.Gizmo", "Uses.Uses", NULL, "ambiguous", + "tool.src.Impl.Gizmo", NULL}, + }; + ASSERT_EQ(dm_check_repo("idle", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* For product code a namespace that only test code declares does not exist: + * it does not stand in the way of a type a using brings in, and a name that + * is not under it is judged without it -- while a name that is under it is + * test code's (test_only_target). Decoys: a namespace product code declares + * does stand there (the global namespace's own names come before the file's + * usings), and test code sees the test namespace. */ +TEST(doc_mentions_cs_test_namespaces) { + static const dm_source_t files[] = { + {"src/Lib/Widget.cs", "namespace Acme.Lib\n" + "{\n" + " public class Widget { }\n" + " public class Gadget { }\n" + "}\n"}, + {"src/App/Uses.cs", + "using Acme.Lib;\n" + "\n" + "namespace Acme.App\n" + "{\n" + " /// \n" + " public class Uses { }\n" + "\n" + " /// \n" + " /// \n" + " /// \n" + " public class UsesFull { }\n" + "}\n"}, + {"tests/Widget/WidgetTests.cs", "namespace Widget\n{\n public class Fixture { }\n}\n"}, + {"tests/Probe/Rig.cs", "namespace Probe\n{\n public class Rig { }\n}\n"}, + {"tests/Lib/Extra.cs", "namespace Acme.Lib.Extra\n{\n public class Bonus { }\n}\n"}, + {"src/Gadget/Kinds.cs", "namespace Gadget\n{\n public class Kind { }\n}\n"}, + {"tests/App/UsesTests.cs", "using Acme.Lib;\n" + "\n" + "namespace Acme.App.Tests\n" + "{\n" + " /// \n" + " public class UsesTests { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"src/App/Uses.cs", "Widget", "Uses.Uses", "Lib.Widget.Widget", NULL, NULL, NULL}, + {"src/App/Uses.cs", "Gadget", "Uses.Uses", NULL, "graph_gap", "Lib.Widget.Gadget", NULL}, + {"tests/App/UsesTests.cs", "Widget", "UsesTests.UsesTests", NULL, "graph_gap", + "Lib.Widget.Widget", NULL}, + /* through a test-only namespace: what is there is test code's ... */ + {"src/App/Uses.cs", "Probe.Rig", "Uses.UsesFull", NULL, "test_only_target", "Probe.Rig.Rig", + NULL}, + {"src/App/Uses.cs", "Acme.Lib.Extra.Bonus", "Uses.UsesFull", NULL, "test_only_target", + "Lib.Extra.Bonus", NULL}, + {"src/App/Uses.cs", "T:Probe.Rig", "Uses.UsesFull", NULL, "test_only_target", + "Probe.Rig.Rig", NULL}, + /* ... and what is not there is judged as if the namespace were not: + * no namespace of the repository, and a namespace without the name */ + {"src/App/Uses.cs", "Probe.Nothing", "Uses.UsesFull", NULL, "external", NULL, NULL}, + {"src/App/Uses.cs", "Acme.Lib.Extra.Nothing", "Uses.UsesFull", NULL, "missing", NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("testns", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* Members: the type that has the name decides, and the written parameter + * list is matched against ITS overloads -- exactly first, a type variable by + * its position, a type nothing is known about only when nothing matches + * exactly. Operators and indexers are declared or they are not. */ +TEST(doc_mentions_cs_members) { + static const dm_source_t files[] = { + {"src/Members.cs", + "namespace Acme.M\n" + "{\n" + " public class Value { public Value(int x) { } }\n" + " public class Two { public Two(long x) { } }\n" + "\n" + " public class Box\n" + " {\n" + " public void Put(T x) { }\n" + " public void Two(int a) { }\n" + " public void Two(string a) { }\n" + " public void Gen() { }\n" + " public void Gen() { }\n" + " public void Dup(int a) { }\n" + " public void Dup(int a) { }\n" + " public void OnlyGen(U u) { }\n" + " public void Map(K key, V value) { }\n" + " public void Loose(Unknown.Thing x) { }\n" + " public int Value { get; set; }\n" + " public int field;\n" + " public event System.Action Changed;\n" + " public int this[int i] { get { return i; } }\n" + " public static Box operator +(Box a, Box b) { return a; }\n" + " public void Item(string key) { }\n" + "\n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " public void Doc() { }\n" + " }\n" + "}\n"}, + {"src/Plain.cs", + "namespace Acme.M\n" + "{\n" + " public class Plain\n" + " {\n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " public void Doc() { }\n" + " }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"src/Members.cs", "Put(T)", "Members.Box.Doc", "Members.Box.Put", NULL, NULL, NULL}, + /* the type has `Put`, and no overload of it takes an int */ + {"src/Members.cs", "Put(int)", "Members.Box.Doc", NULL, "missing", NULL, NULL}, + {"src/Members.cs", "Two(int)", "Members.Box.Doc", "Members.Box.Two", NULL, NULL, NULL}, + /* ... and nothing further out is asked: not the class Two, whose + * constructor takes a long, and not the class Value for a property */ + {"src/Members.cs", "Two(long)", "Members.Box.Doc", NULL, "missing", "Members.Two.Two", + NULL}, + {"src/Members.cs", "Value(int)", "Members.Box.Doc", NULL, "missing", "Members.Value.Value", + NULL}, + /* without type arguments a non-generic method stands before a generic one */ + {"src/Members.cs", "Gen", "Members.Box.Doc", "Members.Box.Gen", NULL, NULL, NULL}, + {"src/Members.cs", "Gen{U}", "Members.Box.Doc", "Members.Box.Gen", NULL, NULL, NULL}, + {"src/Members.cs", "OnlyGen", "Members.Box.Doc", "Members.Box.OnlyGen", NULL, NULL, NULL}, + /* ... with a parameter list too: both take an int, and that is no ambiguity */ + {"src/Members.cs", "Dup(int)", "Members.Box.Doc", "Members.Box.Dup", NULL, NULL, NULL}, + {"src/Members.cs", "Dup{U}(int)", "Members.Box.Doc", "Members.Box.Dup", NULL, NULL, NULL}, + {"src/Members.cs", "Dup", "Members.Box.Doc", "Members.Box.Dup", NULL, NULL, NULL}, + /* a type variable is its position, whatever it is called */ + {"src/Members.cs", "Map{A, B}(A, B)", "Members.Box.Doc", "Members.Box.Map", NULL, NULL, + NULL}, + {"src/Members.cs", "Map{A, B}(B, A)", "Members.Box.Doc", NULL, "missing", NULL, NULL}, + /* the declared parameter type is the same written text */ + {"src/Members.cs", "Loose(Thing{int})", "Members.Box.Doc", "Members.Box.Loose", NULL, NULL, + NULL}, + {"src/Members.cs", "Loose(Other)", "Members.Box.Doc", NULL, "missing", NULL, NULL}, + /* an accessor names its property or its event, nothing else */ + {"src/Members.cs", "get_Value", "Members.Box.Doc", "Members.Box.Value", NULL, NULL, NULL}, + {"src/Members.cs", "get_field", "Members.Box.Doc", NULL, "missing", "Members.Box.field", + NULL}, + {"src/Members.cs", "add_Changed", "Members.Box.Doc", NULL, "graph_gap", NULL, NULL}, + /* a method named Item is a method; without one, `Item(int)` is how + * a doc ID spells the indexer */ + {"src/Members.cs", "Item(string)", "Members.Box.Doc", "Members.Box.Item", NULL, NULL, NULL}, + {"src/Members.cs", "Item(int)", "Members.Box.Doc", NULL, "missing", NULL, NULL}, + /* declared operators and indexers have no node: a gap */ + {"src/Members.cs", "this[int]", "Members.Box.Doc", NULL, "graph_gap", NULL, NULL}, + {"src/Members.cs", "operator +", "Members.Box.Doc", NULL, "graph_gap", NULL, NULL}, + /* ... and one the type does not declare is not there */ + {"src/Members.cs", "operator -", "Members.Box.Doc", NULL, "missing", NULL, NULL}, + {"src/Plain.cs", "this[int]", "Plain.Plain.Doc", NULL, "missing", NULL, NULL}, + {"src/Plain.cs", "operator +", "Plain.Plain.Doc", NULL, "missing", NULL, NULL}, + {"src/Plain.cs", "Plain.this[int]", "Plain.Plain.Doc", NULL, "missing", NULL, NULL}, + {"src/Plain.cs", "Box{T}.this[int]", "Plain.Plain.Doc", NULL, "graph_gap", NULL, NULL}, + {"src/Plain.cs", "Box{T}.operator +", "Plain.Plain.Doc", NULL, "graph_gap", NULL, NULL}, + /* through the type's name too: a property has accessors, a field has none */ + {"src/Plain.cs", "Box{T}.get_Value", "Plain.Plain.Doc", "Members.Box.Value", NULL, NULL, + NULL}, + {"src/Plain.cs", "Box{T}.get_field", "Plain.Plain.Doc", NULL, "missing", + "Members.Box.field", NULL}, + }; + ASSERT_EQ(dm_check_repo("members", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + /* `Item(int)` is written twice in Plain.cs with two outcomes, so it is + * asked per definition: nothing declared (missing) and an indexer (gap) */ + static const dm_source_t item_files[] = { + {"src/NoIndexer.cs", "namespace Acme.M\n" + "{\n" + " public class NoIndexer\n" + " {\n" + " /// \n" + " public void Doc() { }\n" + " }\n" + "}\n"}, + {"src/Indexer.cs", "namespace Acme.M\n" + "{\n" + " public class Indexer\n" + " {\n" + " public int this[int i] { get { return i; } }\n" + "\n" + " /// \n" + " public void Doc() { }\n" + " }\n" + "}\n"}, + }; + static const dm_want_t item_wants[] = { + {"src/NoIndexer.cs", "Item(int)", "NoIndexer.NoIndexer.Doc", NULL, "missing", NULL, NULL}, + {"src/Indexer.cs", "Item(int)", "Indexer.Indexer.Doc", NULL, "graph_gap", NULL, NULL}, + }; + ASSERT_EQ( + dm_check_repo("item", item_files, DM_COUNT(item_files), item_wants, DM_COUNT(item_wants)), + 0); + PASS(); +} + +/* Constructors: a parameter list on a type's name, however the type is + * named; `Foo.Foo`; and `Foo(int)` written inside a generic `Foo`. The + * constructors no source writes are declared and have no node. */ +TEST(doc_mentions_cs_constructors) { + static const dm_source_t files[] = { + {"src/Ctors.cs", + "namespace Acme.C\n" + "{\n" + " public class Widget { }\n" + " public class Gadget\n" + " {\n" + " public Gadget(int size) { }\n" + " public class Part { public Part(string s) { } }\n" + " }\n" + " public record Rec(int Width);\n" + " public class Prim(int seed) { }\n" + " public struct Point { public Point(int x) { } }\n" + " public interface IThing { }\n" + "\n" + " public class Gen\n" + " {\n" + " public Gen(T first) { }\n" + "\n" + " /// \n" + " /// \n" + " public void Doc() { }\n" + " }\n" + "\n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " public class Rows { }\n" + "\n" + " /// \n" + " public class Simple { }\n" + "\n" + " /// \n" + " public class Qualified { }\n" + "\n" + " /// \n" + " public class Nested { }\n" + "\n" + " /// \n" + " public class ByName { }\n" + "\n" + " /// \n" + " public class OfStruct { }\n" + "}\n"}, + {"src/Outside.cs", "namespace Acme.C\n" + "{\n" + " /// \n" + " public class Outside { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + /* a class that writes no constructor has one: declared, no node */ + {"src/Ctors.cs", "Widget()", "Ctors.Rows", NULL, "graph_gap", NULL, NULL}, + {"src/Ctors.cs", "Widget(int)", "Ctors.Rows", NULL, "missing", NULL, NULL}, + /* ... and a class that writes one has no other */ + {"src/Ctors.cs", "Gadget()", "Ctors.Rows", NULL, "missing", NULL, NULL}, + /* primary constructors and a record's copy constructor: declared, no node */ + {"src/Ctors.cs", "Rec(int)", "Ctors.Rows", NULL, "graph_gap", NULL, NULL}, + {"src/Ctors.cs", "Rec(Rec)", "Ctors.Rows", NULL, "graph_gap", NULL, NULL}, + {"src/Ctors.cs", "Prim(int)", "Ctors.Rows", NULL, "graph_gap", NULL, NULL}, + /* a struct keeps its parameterless constructor beside the one it writes */ + {"src/Ctors.cs", "Point()", "Ctors.Rows", NULL, "graph_gap", NULL, NULL}, + {"src/Ctors.cs", "Point.Point", "Ctors.Rows", NULL, "ambiguous", NULL, NULL}, + {"src/Ctors.cs", "IThing()", "Ctors.Rows", NULL, "missing", NULL, NULL}, + {"src/Ctors.cs", "Gadget(int)", "Ctors.Simple", "Ctors.Gadget.Gadget", NULL, NULL, NULL}, + /* the type may be named by any path */ + {"src/Ctors.cs", "Acme.C.Gadget(int)", "Ctors.Qualified", "Ctors.Gadget.Gadget", NULL, NULL, + DM_EXACT}, + {"src/Ctors.cs", "Gadget.Part(string)", "Ctors.Nested", "Ctors.Gadget.Part.Part", NULL, + NULL, NULL}, + {"src/Ctors.cs", "Gadget.Gadget", "Ctors.ByName", "Ctors.Gadget.Gadget", NULL, NULL, NULL}, + {"src/Ctors.cs", "Point(int)", "Ctors.OfStruct", "Ctors.Point.Point", NULL, NULL, NULL}, + /* inside Gen: `Gen(T)` is its constructor although no type `Gen` + * without type arguments is in scope; a bare `Gen` names nothing */ + {"src/Ctors.cs", "Gen(T)", "Ctors.Gen.Doc", "Ctors.Gen.Gen", NULL, NULL, NULL}, + {"src/Ctors.cs", "Gen", "Ctors.Gen.Doc", NULL, "missing", "Ctors.Gen", NULL}, + /* `Gen{T}.Gen` is the constructor from outside the type; written on the + * generic type itself, without parentheses, the compiler binds nothing */ + {"src/Outside.cs", "Gen{T}.Gen", "Outside.Outside", "Ctors.Gen.Gen", NULL, NULL, NULL}, + {"src/Ctors.cs", "Gen{T}.Gen", "Ctors.Gen.Doc", NULL, "missing", NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("ctors", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* The compiler looks up no inherited member in a cref: not through a class's + * base class, not through an interface's base interface, not through the + * roots every type has. A member is named through the type that declares it. */ +TEST(doc_mentions_cs_inherited_members) { + static const dm_source_t files[] = { + {"src/Sys.cs", + "namespace System\n" + "{\n" + " public class Object { public virtual string ToString() { return null; } }\n" + " public class ValueType { public string Kind() { return null; } }\n" + "}\n"}, + {"src/Inherit.cs", + "namespace Acme.I\n" + "{\n" + " public class Base\n" + " {\n" + " public void Run() { }\n" + " public void Put(int x) { }\n" + " public class Nested { }\n" + " }\n" + " public interface IBase { void Ping(); }\n" + " public interface IDerived : IBase { void Own(); }\n" + " public class Ping { }\n" + "\n" + " /// \n" + " public class Derived : Base, IBase\n" + " {\n" + " void IBase.Ping() { }\n" + " public void Put(string s) { }\n" + "\n" + " /// \n" + " /// \n" + " public void Doc() { }\n" + " }\n" + "\n" + " /// \n" + " public class Declared { }\n" + "\n" + " /// \n" + " public class ThroughInterface { }\n" + "\n" + " public record class RecC(int A);\n" + "\n" + " /// \n" + " /// \n" + " public class OfRecord { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + /* through the declaring type: bound */ + {"src/Inherit.cs", "Base.Run", "Inherit.Derived", "Inherit.Base.Run", NULL, NULL, NULL}, + /* a class finds no member of an interface it implements: `Ping` is + * the class of that name */ + {"src/Inherit.cs", "Ping", "Inherit.Derived", "Inherit.Ping", NULL, "Inherit.IBase.Ping", + NULL}, + /* by a simple name, and through the derived type's name: not bound */ + {"src/Inherit.cs", "Run", "Inherit.Derived.Doc", NULL, "missing", "Inherit.Base.Run", NULL}, + {"src/Inherit.cs", "Derived.Run", "Inherit.Derived.Doc", NULL, "missing", NULL, NULL}, + {"src/Inherit.cs", "Nested", "Inherit.Derived.Doc", NULL, "missing", "Inherit.Base.Nested", + NULL}, + /* the type has `Put`; the overload asked for is its base class's */ + {"src/Inherit.cs", "Put(int)", "Inherit.Derived.Doc", NULL, "missing", "Inherit.Base.Put", + NULL}, + {"src/Inherit.cs", "ToString", "Inherit.Derived.Doc", NULL, "missing", + "Sys.Object.ToString", NULL}, + /* an interface's own members, and not its base interface's */ + {"src/Inherit.cs", "IDerived.Own", "Inherit.Declared", "Inherit.IDerived.Own", NULL, NULL, + NULL}, + {"src/Inherit.cs", "IBase.Ping", "Inherit.Declared", "Inherit.IBase.Ping", NULL, NULL, + NULL}, + {"src/Inherit.cs", "IDerived.Ping", "Inherit.ThroughInterface", NULL, "missing", + "Inherit.IBase.Ping", NULL}, + /* a record class is no value type, and its roots are not searched */ + {"src/Inherit.cs", "RecC.ToString", "Inherit.OfRecord", NULL, "missing", + "Sys.Object.ToString", NULL}, + {"src/Inherit.cs", "RecC.Kind", "Inherit.OfRecord", NULL, "missing", "Sys.ValueType.Kind", + NULL}, + {"src/Inherit.cs", "RecC.A", "Inherit.OfRecord", "Inherit.RecC.A", NULL, NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("inherit", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* Doc IDs: a full name from the global namespace, the letter saying what it + * names. `M:` without parentheses is the overload without parameters; a type + * parameter is written as its position. */ +TEST(doc_mentions_cs_doc_ids) { + static const dm_source_t files[] = { + {"src/Ids.cs", + "namespace Acme.D\n" + "{\n" + " public class Outer\n" + " {\n" + " public class Inner { }\n" + " public void Go() { }\n" + " public void Go(int n) { }\n" + " public void Take(T item, int n) { }\n" + " public void Take(int n, T item) { }\n" + " public int Prop { get; set; }\n" + " public int Fld;\n" + " public event System.Action Evt;\n" + " public Outer() { }\n" + " public Outer(string s) { }\n" + " static Outer() { }\n" + " public int this[int i] { get { return i; } }\n" + " public static Outer operator +(Outer a, Outer b) { return a; }\n" + " }\n" + " public class Box\n" + " {\n" + " public void Put(T item) { }\n" + " public void Put(int n) { }\n" + " public class In { public void Both(T a, U b) { } }\n" + " }\n" + "\n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " public class Rows { }\n" + "\n" + " /// \n" + " public class OfType { }\n" + " /// \n" + " public class OfNested { }\n" + " /// \n" + " public class OfGeneric { }\n" + " /// \n" + " public class OfMethod { }\n" + " /// \n" + " public class OfOverload { }\n" + " /// \n" + " public class OfGenericMethod { }\n" + " /// \n" + " public class OfCtor { }\n" + " /// \n" + " public class OfStaticCtor { }\n" + " /// \n" + " public class OfProperty { }\n" + " /// \n" + " public class OfField { }\n" + " /// \n" + " public class OfSlot { }\n" + " /// \n" + " public class OfNestedSlots { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"src/Ids.cs", "T:Acme.D.Outer", "Ids.OfType", "Ids.Outer", NULL, NULL, DM_EXACT}, + {"src/Ids.cs", "T:Acme.D.Outer.Inner", "Ids.OfNested", "Ids.Outer.Inner", NULL, NULL, + DM_EXACT}, + {"src/Ids.cs", "T:Acme.D.Box`1", "Ids.OfGeneric", "Ids.Box", NULL, NULL, DM_EXACT}, + /* the namespace is the repository's: the type is not there */ + {"src/Ids.cs", "T:Acme.D.Nope", "Ids.Rows", NULL, "missing", NULL, NULL}, + {"src/Ids.cs", "T:Acme.D.Box", "Ids.Rows", NULL, "missing", "Ids.Box", NULL}, + {"src/Ids.cs", "T:Acme.D", "Ids.Rows", NULL, "missing", NULL, NULL}, + /* ... a namespace the repository does not declare is outside */ + {"src/Ids.cs", "T:Other.Thing", "Ids.Rows", NULL, "external", NULL, NULL}, + {"src/Ids.cs", "N:Acme.D", "Ids.Rows", NULL, "graph_gap", NULL, NULL}, + {"src/Ids.cs", "N:Acme.Nope", "Ids.Rows", NULL, "external", NULL, NULL}, + /* `M:` without parentheses: the overload without parameters, not the group */ + {"src/Ids.cs", "M:Acme.D.Outer.Go", "Ids.OfMethod", "Ids.Outer.Go", NULL, NULL, DM_EXACT}, + {"src/Ids.cs", "M:Acme.D.Outer.Go(System.Int32)", "Ids.OfOverload", "Ids.Outer.Go", NULL, + NULL, NULL}, + {"src/Ids.cs", "M:Acme.D.Outer.Go(System.String)", "Ids.Rows", NULL, "missing", NULL, NULL}, + /* a method's type parameter by its position: the overload it is in */ + {"src/Ids.cs", "M:Acme.D.Outer.Take``1(``0,System.Int32)", "Ids.OfGenericMethod", + "Ids.Outer.Take", NULL, NULL, NULL}, + {"src/Ids.cs", "M:Acme.D.Outer.#ctor(System.String)", "Ids.OfCtor", "Ids.Outer.Outer", NULL, + NULL, NULL}, + {"src/Ids.cs", "M:Acme.D.Outer.#cctor", "Ids.OfStaticCtor", "Ids.Outer.Outer", NULL, NULL, + NULL}, + /* the letter names the kind */ + {"src/Ids.cs", "P:Acme.D.Outer.Prop", "Ids.OfProperty", "Ids.Outer.Prop", NULL, NULL, NULL}, + {"src/Ids.cs", "F:Acme.D.Outer.Fld", "Ids.OfField", "Ids.Outer.Fld", NULL, NULL, NULL}, + {"src/Ids.cs", "F:Acme.D.Outer.Prop", "Ids.Rows", NULL, "missing", "Ids.Outer.Prop", NULL}, + {"src/Ids.cs", "E:Acme.D.Outer.Evt", "Ids.Rows", NULL, "graph_gap", NULL, NULL}, + {"src/Ids.cs", "P:Acme.D.Outer.Item(System.Int32)", "Ids.Rows", NULL, "graph_gap", NULL, + NULL}, + /* a type's parameter by its position, counted on from its outer types */ + {"src/Ids.cs", "M:Acme.D.Box`1.Put(`0)", "Ids.OfSlot", "Ids.Box.Put", NULL, NULL, NULL}, + {"src/Ids.cs", "M:Acme.D.Box`1.In`1.Both(`0,`1)", "Ids.OfNestedSlots", "Ids.Box.In.Both", + NULL, NULL, NULL}, + {"src/Ids.cs", "M:Acme.D.Box`1.In`1.Both(`1,`0)", "Ids.Rows", NULL, "missing", NULL, NULL}, + {"src/Ids.cs", "M:Acme.D.Outer.op_Addition(Acme.D.Outer,Acme.D.Outer)", "Ids.Rows", NULL, + "graph_gap", NULL, NULL}, + {"src/Ids.cs", "M:Acme.D.Outer.op_Subtraction(Acme.D.Outer,Acme.D.Outer)", "Ids.Rows", NULL, + "missing", NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("ids", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* The reasons. `external` is what the repository's own structure says is + * outside: a namespace it does not declare. No name is outside because it is + * on a list -- a repository that IS Xunit declares Xunit. A reference too + * long to be read is not judged by its beginning; a product reference never + * binds what only a test declares, and does not count it either. */ +TEST(doc_mentions_cs_reasons) { + /* `Known.M(TTT...T, int)`: cut at a buffer's end it would read as a call + * with ONE parameter of a type nothing is known about, which M(int) fits */ + char longref[1300]; + char long_cs[1600]; + memset(longref, 'T', sizeof(longref)); + memcpy(longref, "Known.M(", 8); + snprintf(longref + 1100, sizeof(longref) - 1100, ", int)"); + snprintf(long_cs, sizeof(long_cs), + "namespace Acme.R\n" + "{\n" + " public class Known { public void M(int a) { } }\n" + "\n" + " /// \n" + " public class TooLong { }\n" + "}\n", + longref); + const dm_source_t files[] = { + {"src/Declared.cs", "namespace Xunit.Sdk\n" + "{\n" + " public class Runner { }\n" + "}\n" + "namespace Acme.R\n" + "{\n" + " public class Here { }\n" + "}\n"}, + {"src/Structure.cs", + "using Xunit.Sdk;\n" + "\n" + "namespace Acme.R\n" + "{\n" + " /// \n" + " /// \n" + " /// \n" + " public class Structure { }\n" + "}\n"}, + {"src/OpenScope.cs", "using Moq;\n" + "\n" + "namespace Acme.R\n" + "{\n" + " /// \n" + " public class OpenScope { }\n" + "}\n"}, + /* the repository adds to System and to a namespace under it: polyfills */ + {"src/Polyfill.cs", "namespace System\n" + "{\n" + " public class Polyfill { }\n" + "}\n" + "namespace System.Text.Json\n" + "{\n" + " public class Extra { }\n" + "}\n"}, + {"src/Standard.cs", + "using System.Text.Json;\n" + "\n" + "namespace Acme.R\n" + "{\n" + " /// \n" + " /// \n" + " /// \n" + " public class Standard { }\n" + "}\n"}, + {"src/Unicode.cs", "namespace Acme.R\n" + "{\n" + " public class Gr\xC3\xB6\xC3\x9F" + "e { public void L\xC3\xA4nge() { } }\n" + "\n" + " /// \n" + " /// \n" + " public class Unicode { }\n" + "}\n"}, + {"src/Long.cs", long_cs}, + {"src/Mix.cs", + "namespace Acme.R\n" + "{\n" + " public partial class Mix\n" + " {\n" + " public void Prod() { }\n" + " public void Over(string s) { }\n" + " }\n" + "\n" + " /// \n" + " /// \n" + " public class UsesMix { }\n" + "}\n"}, + {"src/tests/MixTests.cs", "namespace Acme.R\n" + "{\n" + " public partial class Mix\n" + " {\n" + " public void OnlyTest() { }\n" + " public void Over(int a) { }\n" + " }\n" + "\n" + " /// \n" + " public class FromTest { }\n" + "}\n"}, + }; + const dm_want_t wants[] = { + /* a namespace the repository declares: its missing names are missing */ + {"src/Structure.cs", "Xunit.Sdk.Nope", "Structure.Structure", NULL, "missing", NULL, NULL}, + {"src/Structure.cs", "Acme.R.Nope", "Structure.Structure", NULL, "missing", NULL, NULL}, + {"src/Structure.cs", "Xunit.Sdk.Runner", "Structure.Structure", "Declared.Runner", NULL, + NULL, DM_EXACT}, + /* ... one it only has namespaces under, and one it does not have: outside */ + {"src/Structure.cs", "Xunit.Assert", "Structure.Structure", NULL, "external", NULL, NULL}, + {"src/Structure.cs", "Outside.Lib.Thing", "Structure.Structure", NULL, "external", NULL, + NULL}, + /* every using of the file names a declared namespace: the scope is closed */ + {"src/Structure.cs", "NotAnywhere", "Structure.Structure", NULL, "missing", NULL, NULL}, + /* ... a using of a namespace the repository does not declare opens it */ + {"src/OpenScope.cs", "NotAnywhere", "OpenScope.OpenScope", NULL, "external", NULL, NULL}, + /* System is the standard library's whoever adds to it: what the + * repository does not have there is outside, though it declares the + * namespace -- by a full name, by a doc ID, through a using */ + {"src/Standard.cs", "System.Nope", "Standard.Standard", NULL, "external", NULL, NULL}, + {"src/Standard.cs", "System.Text.Json.Nope", "Standard.Standard", NULL, "external", NULL, + NULL}, + {"src/Standard.cs", "System.Text.Nope", "Standard.Standard", NULL, "external", NULL, NULL}, + {"src/Standard.cs", "T:System.Nope2", "Standard.Standard", NULL, "external", NULL, NULL}, + {"src/Standard.cs", "NotInJson", "Standard.Standard", NULL, "external", NULL, NULL}, + /* ... and what it has there is its own */ + {"src/Standard.cs", "System.Polyfill", "Standard.Standard", "Polyfill.Polyfill", NULL, NULL, + DM_EXACT}, + /* identifiers are not ASCII only */ + {"src/Unicode.cs", + "Gr\xC3\xB6\xC3\x9F" + "e", + "Unicode.Unicode", + "Unicode.Gr\xC3\xB6\xC3\x9F" + "e", + NULL, NULL, NULL}, + {"src/Unicode.cs", + "Gr\xC3\xB6\xC3\x9F" + "e.L\xC3\xA4nge", + "Unicode.Unicode", + "Unicode.Gr\xC3\xB6\xC3\x9F" + "e.L\xC3\xA4nge", + NULL, NULL, NULL}, + {"src/Unicode.cs", + "Gr\xC3\xB6\xC3\x9F" + "e.Nicht", + "Unicode.Unicode", NULL, "missing", NULL, NULL}, + /* too long to be read: never resolved by what is left after a cut */ + {"src/Long.cs", longref, "Long.TooLong", NULL, "unparseable", "Long.Known.M", NULL}, + /* only the test part of the type declares OnlyTest */ + {"src/Mix.cs", "Mix.OnlyTest", "Mix.UsesMix", NULL, "test_only_target", + "MixTests.Mix.OnlyTest", NULL}, + /* ... and its Over(int) is no overload for product code: one Over */ + {"src/Mix.cs", "Mix.Over", "Mix.UsesMix", "Mix.Mix.Over", NULL, "MixTests.Mix.Over", NULL}, + {"src/Mix.cs", "Mix.Prod", "Mix.UsesMix", "Mix.Mix.Prod", NULL, NULL, NULL}, + /* test code sees both parts: an overload group */ + {"src/tests/MixTests.cs", "Mix.Over", "MixTests.FromTest", NULL, "ambiguous", NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("reasons", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* The text around a reference: CRLF line ends and a byte-order mark change + * nothing; a reference inside an XML comment or a CDATA section of the doc + * text is no reference. */ +TEST(doc_mentions_cs_text_forms) { + const char *crlf = "\xEF\xBB\xBFnamespace Acme.T\r\n" /* 1 */ + "{\r\n" /* 2 */ + " public class Target { }\r\n" /* 3 */ + "\r\n" /* 4 */ + " /// First \r\n" /* 5 */ + " /// and .\r\n" /* 7 */ + " public class Uses\r\n" /* 8 */ + " {\r\n" /* 9 */ + " public int Field;\r\n" /* 10 */ + " }\r\n" /* 11 */ + "}\r\n"; /* 12 */ + CBMFileResult *r = dm_extract(crlf, CBM_LANG_CSHARP, "Crlf.cs"); + ASSERT_NOT_NULL(r); + const CBMDocLink *first = dm_find_token(r, "Target"); + ASSERT_NOT_NULL(first); + ASSERT_EQ(first->line, 5); + ASSERT_EQ(first->def_line, 8); + const CBMDocLink *second = dm_find_token(r, "Target.Nope"); + ASSERT_NOT_NULL(second); + ASSERT_EQ(second->line, 6); + ASSERT_NOT_NULL(r->doc_scope); + ASSERT_NOT_NULL(strstr(r->doc_scope, "R\t1\t0\t1\t12\tAcme.T\n")); + ASSERT_NOT_NULL(strstr(r->doc_scope, "T\t1\t3\t3\tc\t-\tTarget\t\t\n")); + ASSERT_NOT_NULL(strstr(r->doc_scope, "T\t1\t8\t11\tc\t-\tUses\t\t\n")); + ASSERT_NOT_NULL(strstr(r->doc_scope, "M\t10\tv\t0\t1\tField\t\t-\n")); + ASSERT_NULL(strchr(r->doc_scope, '\r')); + cbm_free_result(r); + + const char *hidden = "namespace Acme.T\n" + "{\n" + " /// \n" + " /// \n" + " /// ]]>\n" + " /// \n" + " /// \n" + " /// \n" + " public class Hidden { }\n" + "}\n"; + r = dm_extract(hidden, CBM_LANG_CSHARP, "Hidden.cs"); + ASSERT_NOT_NULL(r); + ASSERT_NULL(dm_find_token(r, "InComment")); + ASSERT_NULL(dm_find_token(r, "InCdata")); + ASSERT_NULL(dm_find_token(r, "InLongComment")); + ASSERT_NOT_NULL(dm_find_token(r, "Shown")); + /* markup inside is markup: the compiler binds a cref there too */ + ASSERT_NOT_NULL(dm_find_token(r, "InCode")); + cbm_free_result(r); + + /* ... and end to end, through the pipeline */ + const dm_source_t files[] = {{"src/Crlf.cs", crlf}, {"src/Hidden.cs", hidden}}; + static const dm_want_t wants[] = { + {"src/Crlf.cs", "Target", "Crlf.Uses", "Crlf.Target", NULL, NULL, NULL}, + {"src/Crlf.cs", "Target.Nope", "Crlf.Uses", NULL, "missing", NULL, NULL}, + /* what a comment or a CDATA section holds is no reference: no row */ + {"src/Hidden.cs", "InComment", "Hidden.Hidden", NULL, NULL, NULL, NULL}, + {"src/Hidden.cs", "InCdata", "Hidden.Hidden", NULL, NULL, NULL, NULL}, + {"src/Hidden.cs", "InLongComment", "Hidden.Hidden", NULL, NULL, NULL, NULL}, + {"src/Hidden.cs", "Shown", "Hidden.Hidden", NULL, "missing", NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("crlf", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* ── cost and limits ─────────────────────────────────────────────── */ + +enum { DM_COST_REFS = 40, DM_COST_TEXT = 16384 }; + +/* What the C# resolver looks at for DM_COST_REFS simple names that are not in + * scope, documented `types` types deep in a namespace of `depth` segments, in + * a file with `usings` using directives. `declared`: the names are types of + * a namespace the file does not import (else no file declares them). 0 when + * the repository cannot be indexed. */ +static uint64_t dm_lookup_work(int depth, int usings, int types, bool declared) { + char *src = malloc(DM_COST_TEXT); + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_cost_XXXXXX"); + if (!src || !cbm_mkdtemp(tmp)) { + free(src); + return 0; + } + /* the namespaces the usings name, and (when declared) the names asked + * for, in a namespace nothing imports */ + size_t w = 0; + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "namespace Far\n{\n"); + for (int i = 0; declared && i < DM_COST_REFS; i++) { + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "public class Nowhere%d { }\n", i); + } + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "}\n"); + for (int i = 0; i < usings; i++) { + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, + "namespace In.U%d { public class Other%d { } }\n", i, i); + } + th_write_file(TH_PATH(tmp, "src/Far.cs"), src); + w = 0; + for (int i = 0; i < usings; i++) { + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "using In.U%d;\n", i); + } + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "namespace "); + for (int i = 0; i < depth; i++) { + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "%sN%d", i ? "." : "", i); + } + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "\n{\n"); + for (int i = 0; i + 1 < types; i++) { + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "public class T%d {\n", i); + } + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "/// "); + for (int i = 0; i < DM_COST_REFS; i++) { + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, " ", i); + } + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "\npublic class Leaf { }\n"); + for (int i = 0; i < types; i++) { + w += (size_t)snprintf(src + w, DM_COST_TEXT - w, "}\n"); + } + th_write_file(TH_PATH(tmp, "src/Deep.cs"), src); + free(src); + char db[512]; + snprintf(db, sizeof(db), "%s/cost.db", tmp); + cbm_doclink_cs_test_work_reset(); + uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_work() : 0; + int rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"); + dm_unlink_db(db); + th_rmtree(tmp); + return rows == DM_COST_REFS ? work : 0; /* every name was looked up, and found nowhere */ +} + +/* The same for DM_COST_REFS references, written in product code, to an + * operator that only test code declares -- `overloads` times over. */ +static uint64_t dm_operator_work(int overloads) { + char *src = malloc(DM_COST_TEXT * 4); + size_t cap = DM_COST_TEXT * 4; + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_opcost_XXXXXX"); + if (!src || !cbm_mkdtemp(tmp)) { + free(src); + return 0; + } + th_write_file(TH_PATH(tmp, "src/Vec.cs"), + "namespace Geo\n{\n public partial struct Vec { }\n}\n"); + size_t w = + (size_t)snprintf(src, cap, "namespace Geo\n{\n public partial struct Vec\n {\n"); + for (int i = 0; i < overloads; i++) { + w += (size_t)snprintf( + src + w, cap - w, + " public static Vec operator +(Vec a, P%03d b) { return a; }\n", i); + } + snprintf(src + w, cap - w, " }\n}\n"); + th_write_file(TH_PATH(tmp, "tests/VecOps.cs"), src); + w = (size_t)snprintf(src, cap, "namespace Geo\n{\n /// "); + for (int i = 0; i < DM_COST_REFS; i++) { + w += (size_t)snprintf(src + w, cap - w, " ", i); + } + snprintf(src + w, cap - w, "\n public class Uses { }\n}\n"); + th_write_file(TH_PATH(tmp, "src/Uses.cs"), src); + free(src); + char db[512]; + snprintf(db, sizeof(db), "%s/cost.db", tmp); + cbm_doclink_cs_test_work_reset(); + uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_work() : 0; + int rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'missing'"); + dm_unlink_db(db); + th_rmtree(tmp); + return rows == DM_COST_REFS ? work : 0; /* product code sees none of them */ +} + +/* The same for DM_COST_REFS references `Big.M(int)` to a method with 100 + * overloads of `params` parameters each: none takes one parameter. */ +static uint64_t dm_signature_work(int params) { + char *src = malloc(DM_COST_TEXT * 4); + size_t cap = DM_COST_TEXT * 4; + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_sigcost_XXXXXX"); + if (!src || !cbm_mkdtemp(tmp)) { + free(src); + return 0; + } + size_t w = (size_t)snprintf(src, cap, "namespace Sig\n{\n public class Big\n {\n"); + for (int i = 0; i < 100; i++) { + w += (size_t)snprintf(src + w, cap - w, " public void M(P%03d a0", i); + for (int p = 1; p < params; p++) { + w += (size_t)snprintf(src + w, cap - w, ", int a%d", p); + } + w += (size_t)snprintf(src + w, cap - w, ") { }\n"); + } + w += (size_t)snprintf(src + w, cap - w, " }\n\n /// "); + for (int i = 0; i < DM_COST_REFS; i++) { + w += (size_t)snprintf(src + w, cap - w, " ", i); + } + snprintf(src + w, cap - w, "\n public class Uses { }\n}\n"); + th_write_file(TH_PATH(tmp, "src/Big.cs"), src); + free(src); + char db[512]; + snprintf(db, sizeof(db), "%s/cost.db", tmp); + cbm_doclink_cs_test_work_reset(); + uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_work() : 0; + int rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'missing'"); + dm_unlink_db(db); + th_rmtree(tmp); + return rows == DM_COST_REFS ? work : 0; +} + +/* The same for DM_COST_REFS references in the documentation of a method that + * shares its line with `members` - 1 other methods. */ +static uint64_t dm_line_work(int members) { + char *src = malloc(DM_COST_TEXT * 2); + size_t cap = DM_COST_TEXT * 2; + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_linecost_XXXXXX"); + if (!src || !cbm_mkdtemp(tmp)) { + free(src); + return 0; + } + size_t w = (size_t)snprintf( + src, cap, "namespace Line\n{\n public class Wide\n {\n /// "); + for (int i = 0; i < DM_COST_REFS; i++) { + w += (size_t)snprintf(src + w, cap - w, " ", i); + } + w += (size_t)snprintf(src + w, cap - w, "\n "); + for (int i = 0; i < members; i++) { + w += (size_t)snprintf(src + w, cap - w, " public void M%d() { }", i); + } + snprintf(src + w, cap - w, "\n }\n}\n"); + th_write_file(TH_PATH(tmp, "src/Wide.cs"), src); + free(src); + char db[512]; + snprintf(db, sizeof(db), "%s/cost.db", tmp); + cbm_doclink_cs_test_work_reset(); + uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_work() : 0; + int rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"); + dm_unlink_db(db); + th_rmtree(tmp); + return rows == DM_COST_REFS ? work : 0; +} + +enum { DM_TREE_FILES = 120 }; + +/* What the resolver looks at to find the project of DM_TREE_FILES files that + * stand `depth` directories deep, in a repository without a project file. */ +static uint64_t dm_tree_work(int depth) { + char tmp[256]; + char dir[256]; + char rel[320]; + char text[192]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_tree_XXXXXX"); + if (depth * 2 >= (int)sizeof(dir) || !cbm_mkdtemp(tmp)) { + return 0; + } + size_t d = 0; + dir[0] = '\0'; + for (int i = 0; i < depth; i++) { + d += (size_t)snprintf(dir + d, sizeof(dir) - d, "d/"); + } + for (int i = 0; i < DM_TREE_FILES; i++) { + snprintf(rel, sizeof(rel), "%sF%03d.cs", dir, i); + snprintf(text, sizeof(text), "namespace Deep\n{\n%s public class F%03d { }\n}\n", + i == 0 ? " /// \n" : "", i); + th_write_file(TH_PATH(tmp, rel), text); + } + char db[512]; + snprintf(db, sizeof(db), "%s/cost.db", tmp); + cbm_doclink_cs_test_work_reset(); + uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_work() : 0; + int rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"); + dm_unlink_db(db); + th_rmtree(tmp); + return rows == 1 ? work : 0; +} + +/* Every import kind is queried with names absent everywhere or declared only + * outside the imported scopes. Count visits/probes, never elapsed time. */ +typedef struct { + uint64_t lookup, build; +} dm_import_work_t; + +static dm_import_work_t dm_import_lookup_work(int imports, char kind, bool global, bool declared) { + enum { CAP = DM_COST_TEXT * 4 }; + char *defs = malloc(CAP); + char *directives = malloc(CAP); + char *use = malloc(CAP); + char tmp[256] = "/tmp/cbm_dm_importcost_XXXXXX"; + if (!defs || !directives || !use || !cbm_mkdtemp(tmp)) { + free(defs); + free(directives); + free(use); + return (dm_import_work_t){0}; + } + size_t d = 0, u = 0; + for (int i = 0; i < imports; i++) { + d += (size_t)snprintf(defs + d, CAP - d, + "namespace In.U%d { public static class Other%d { " + "public static int Member%d; } }\n", + i, i, i); + if (kind == 'n') { + u += (size_t)snprintf(directives + u, CAP - u, "%susing In.U%d;\n", + global ? "global " : "", i); + } else if (kind == 's') { + u += (size_t)snprintf(directives + u, CAP - u, "%susing static In.U%d.Other%d;\n", + global ? "global " : "", i, i); + } else { + u += (size_t)snprintf(directives + u, CAP - u, "%susing Alias%d = In.U%d.Other%d;\n", + global ? "global " : "", i, i, i); + } + } + d += (size_t)snprintf(defs + d, CAP - d, "namespace Far {\n"); + for (int i = 0; declared && i < DM_COST_REFS; i++) { + d += (size_t)snprintf(defs + d, CAP - d, "public class Nowhere%d { }\n", i); + } + snprintf(defs + d, CAP - d, "}\n"); + size_t w = (size_t)snprintf(use, CAP, "%snamespace App {\n/// ", global ? "" : directives); + for (int i = 0; i < DM_COST_REFS; i++) { + w += (size_t)snprintf(use + w, CAP - w, " ", i); + } + snprintf(use + w, CAP - w, "\npublic class Use { }\n}\n"); + th_write_file(TH_PATH(tmp, "Defs.cs"), defs); + th_write_file(TH_PATH(tmp, "Use.cs"), use); + th_write_file(TH_PATH(tmp, "App.csproj"), DM_EMPTY_PROJECT); + if (global) { + th_write_file(TH_PATH(tmp, "Globals.cs"), directives); + } + free(defs); + free(directives); + free(use); + char db[512]; + snprintf(db, sizeof(db), "%s/cost.db", tmp); + cbm_doclink_cs_test_work_reset(); + dm_import_work_t work = {0}; + if (dm_index(tmp, db, NULL) == 0) { + work.lookup = cbm_doclink_cs_test_work(); + work.build = cbm_doclink_cs_test_index_work(); + } + int rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"); + dm_unlink_db(db); + th_rmtree(tmp); + return rows == DM_COST_REFS ? work : (dm_import_work_t){0}; +} + +/* Many repository declarations of a name must not burden a scope importing + * only one of them. Static members have distinct graph owners in this file. */ +static uint64_t dm_common_import_work(int declarations) { + enum { CAP = DM_COST_TEXT * 4 }; + char *defs = malloc(CAP); + char *use = malloc(CAP); + char tmp[256] = "/tmp/cbm_dm_commonimport_XXXXXX"; + if (!defs || !use || !cbm_mkdtemp(tmp)) { + free(defs); + free(use); + return 0; + } + size_t d = (size_t)snprintf(defs, CAP, "namespace In {\n"); + for (int i = 0; i < declarations; i++) { + d += (size_t)snprintf(defs + d, CAP - d, + "public static class C%d { public static int Common; }\n", i); + } + snprintf(defs + d, CAP - d, "}\n"); + size_t u = (size_t)snprintf(use, CAP, "using static In.C0;\nnamespace App {\n/// "); + for (int i = 0; i < DM_COST_REFS; i++) { + u += (size_t)snprintf(use + u, CAP - u, " "); + } + snprintf(use + u, CAP - u, "\npublic class Use { }\n}\n"); + th_write_file(TH_PATH(tmp, "Defs.cs"), defs); + th_write_file(TH_PATH(tmp, "Use.cs"), use); + free(defs); + free(use); + char db[512], props[512]; + snprintf(db, sizeof(db), "%s/cost.db", tmp); + cbm_doclink_cs_test_work_reset(); + uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_work() : 0; + int edges = 0; + dm_edge(db, "Use.Use", "Defs.C0.Common", props, sizeof(props), &edges); + int rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"); + dm_unlink_db(db); + th_rmtree(tmp); + return edges == 1 && rows == 0 ? work : 0; +} + +TEST(doc_mentions_cs_import_lookup_cost) { + bool bounded = true; + static const char kinds[] = {'n', 'a', 's'}; + for (int k = 0; k < 3; k++) { + for (int global = 0; global < 2; global++) { + for (int declared = 0; declared < 2; declared++) { + dm_import_work_t small = + dm_import_lookup_work(60, kinds[k], global != 0, declared != 0); + dm_import_work_t large = + dm_import_lookup_work(120, kinds[k], global != 0, declared != 0); + double ratio = small.lookup ? (double)large.lookup / (double)small.lookup : 0.0; + double build_ratio = small.build ? (double)large.build / (double)small.build : 0.0; + printf(" import lookup kind=%c global=%d declared=%d: %llu -> %llu, %.2f; " + "index build %llu -> %llu, %.2f\n", + kinds[k], global, declared, (unsigned long long)small.lookup, + (unsigned long long)large.lookup, ratio, (unsigned long long)small.build, + (unsigned long long)large.build, build_ratio); + bounded = bounded && small.lookup >= DM_COST_REFS && ratio >= 0.8 && ratio <= 1.4 && + small.build > 0 && build_ratio >= 0.8 && build_ratio <= 3.0; + } + } + } + uint64_t small = dm_common_import_work(64); + uint64_t large = dm_common_import_work(128); + double ratio = small ? (double)large / (double)small : 0.0; + printf(" common name, one import: %llu -> %llu, %.2f\n", (unsigned long long)small, + (unsigned long long)large, ratio); + bounded = bounded && small >= DM_COST_REFS && ratio >= 0.8 && ratio <= 1.4; + ASSERT_TRUE(bounded); + PASS(); +} + +/* S5: `Shared` is declared in k namespaces nobody imports; one file has k + * usings of other namespaces and r documented classes that each name it, so + * every one of its r lookups would walk the k directives. The resolver's + * lookup work, and the number of `missing` rows for `Shared` (r when all are + * resolved; -1 when the repository cannot be indexed). */ +static uint64_t dm_repeated_name_work(int k, int r, int *rows) { + enum { CAP = 256 * 1024 }; + char *src = malloc(CAP); + char tmp[256] = "/tmp/cbm_dm_repeated_XXXXXX"; + *rows = -1; + if (!src || !cbm_mkdtemp(tmp)) { + free(src); + return 0; + } + size_t w = 0; + for (int i = 0; i < k; i++) { + w += (size_t)snprintf(src + w, CAP - w, + "namespace P%d { public class Shared { } }\n" + "namespace In.U%d { public class Other%d { } }\n", + i, i, i); + } + th_write_file(TH_PATH(tmp, "src/Far.cs"), src); + w = 0; + for (int i = 0; i < k; i++) { + w += (size_t)snprintf(src + w, CAP - w, "using In.U%d;\n", i); + } + w += (size_t)snprintf(src + w, CAP - w, "namespace N0\n{\n"); + for (int j = 0; j < r; j++) { + w += (size_t)snprintf(src + w, CAP - w, + "/// \n" + "public class L%d { }\n", + j); + } + snprintf(src + w, CAP - w, "}\n"); + th_write_file(TH_PATH(tmp, "src/Uses.cs"), src); + free(src); + char db[512]; + snprintf(db, sizeof(db), "%s/repeated.db", tmp); + cbm_doclink_cs_test_work_reset(); + uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_work() : 0; + *rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved " + "WHERE raw = 'Shared' AND reason = 'missing'"); + dm_unlink_db(db); + th_rmtree(tmp); + return work; +} + +/* A name looked up again in the same region costs once: with k and r both + * four times as large the work grows with the input (about 4 x), not with + * r x k (16 x). */ +TEST(doc_mentions_cs_lookup_repeated_name_work) { + int rows[2] = {0}; + uint64_t small = dm_repeated_name_work(60, 40, &rows[0]); + uint64_t large = dm_repeated_name_work(240, 160, &rows[1]); + double ratio = small ? (double)large / (double)small : 0.0; + printf(" repeated name, k 60 -> 240 and r 40 -> 160: %llu -> %llu steps, %.2f times the " + "work, want at most 6\n", + (unsigned long long)small, (unsigned long long)large, ratio); + ASSERT_EQ(rows[0], 40); + ASSERT_EQ(rows[1], 160); + ASSERT_GT(small, 0); + ASSERT_LTE(large, 6 * small); + PASS(); +} + +/* R1: `Shared` is declared in k namespaces nobody imports; the project has k + * global usings of other namespaces, and f files each name `Shared` once, so + * each of them would walk the k directives of the unit. The resolver's work + * and the number of `missing` rows for `Shared` (f when all are resolved; -1 + * when the repository cannot be indexed). */ +static uint64_t dm_unit_usings_work(int k, int f, int *rows) { + enum { CAP = 256 * 1024 }; + char *src = malloc(CAP); + char tmp[256] = "/tmp/cbm_dm_unit_usings_XXXXXX"; + *rows = -1; + if (!src || !cbm_mkdtemp(tmp)) { + free(src); + return 0; + } + th_write_file(TH_PATH(tmp, "src/App.csproj"), DM_EMPTY_PROJECT); + size_t w = 0; + for (int i = 0; i < k; i++) { + w += (size_t)snprintf(src + w, CAP - w, + "namespace P%d { public class Shared { } }\n" + "namespace In.U%d { public class Other%d { } }\n", + i, i, i); + } + th_write_file(TH_PATH(tmp, "src/Far.cs"), src); + w = 0; + for (int i = 0; i < k; i++) { + w += (size_t)snprintf(src + w, CAP - w, "global using In.U%d;\n", i); + } + th_write_file(TH_PATH(tmp, "src/GlobalUsings.cs"), src); + for (int j = 0; j < f; j++) { + char path[512]; + snprintf(path, sizeof(path), "%s/src/Uses%d.cs", tmp, j); + snprintf(src, CAP, + "namespace N0\n{\n" + " /// \n" + " public class L%d { }\n" + "}\n", + j); + th_write_file(path, src); + } + free(src); + char db[512]; + snprintf(db, sizeof(db), "%s/unit_usings.db", tmp); + cbm_doclink_cs_test_work_reset(); + uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_work() : 0; + *rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved " + "WHERE raw = 'Shared' AND reason = 'missing'"); + dm_unlink_db(db); + th_rmtree(tmp); + return work; +} + +/* R1: what the unit's directives give a query is the same for every file of + * the unit that asks it: with k and f both twice as large the work about + * doubles (the input does), not f x k (4 x). */ +TEST(doc_mentions_cs_unit_usings_across_files_work) { + int rows[2] = {0}; + uint64_t small = dm_unit_usings_work(150, 100, &rows[0]); + uint64_t large = dm_unit_usings_work(300, 200, &rows[1]); + double ratio = small ? (double)large / (double)small : 0.0; + printf(" unit usings, k 150 -> 300 and f 100 -> 200: %llu -> %llu steps, %.2f times the " + "work, want at most 2.6\n", + (unsigned long long)small, (unsigned long long)large, ratio); + ASSERT_EQ(rows[0], 100); + ASSERT_EQ(rows[1], 200); + ASSERT_GT(small, 0); + ASSERT_LTE(large * 5, small * 13); + PASS(); +} + +/* R1: a file asks only the unit directives that give its query anything, and + * stops where the walk of all of them would: the edges and rows are those of + * a run without the unit memo. The unit imports A (X, Y), B (X), D (Z) and the + * test-only T (W). Cases: nothing from the file and two unit candidates (U1); + * one from the file, another from the unit (U2); the same one from both (U3); + * the same one, then another (U4); one from the unit only (U5); test code + * only (U6 product, UT test code); a unit without global usings (UO). */ +TEST(doc_mentions_cs_unit_usings_unchanged) { + static const dm_source_t files[] = { + {"src/App.csproj", DM_EMPTY_PROJECT}, + {"src/A.cs", "namespace A\n{\n public class X { }\n public class Y { }\n}\n"}, + {"src/B.cs", "namespace B\n{\n public class X { }\n}\n"}, + {"src/C.cs", "namespace C\n{\n public class X { }\n}\n"}, + {"src/D.cs", "namespace D\n{\n public class Z { }\n}\n"}, + {"src/tests/T.cs", "namespace T\n{\n public class W { }\n}\n"}, + {"src/GlobalUsings.cs", "global using A;\nglobal using B;\nglobal using D;\n" + "global using T;\n"}, + {"src/U1.cs", "namespace N\n{\n /// \n" + " public class U1 { }\n}\n"}, + {"src/U2.cs", "using C;\nnamespace N\n{\n /// \n" + " public class U2 { }\n}\n"}, + {"src/U3.cs", "using A;\nnamespace N\n{\n /// \n" + " public class U3 { }\n}\n"}, + {"src/U4.cs", "using A;\nnamespace N\n{\n /// \n" + " public class U4 { }\n}\n"}, + {"src/U5.cs", "namespace N\n{\n /// \n" + " public class U5 { }\n}\n"}, + {"src/U6.cs", "namespace N\n{\n /// \n" + " public class U6 { }\n}\n"}, + {"src/tests/UT.cs", "namespace N\n{\n /// " + "\n public class UT { }\n}\n"}, + {"other/Other.csproj", DM_EMPTY_PROJECT}, + {"other/UO.cs", "namespace N\n{\n /// " + "\n public class UO { }\n}\n"}, + }; + static const dm_want_t wants[] = { + {"src/U1.cs", "X", "U1.U1", NULL, "ambiguous", NULL, NULL}, + {"src/U2.cs", "X", "U2.U2", NULL, "ambiguous", NULL, NULL}, + {"src/U3.cs", "Y", "U3.U3", "A.Y", NULL, NULL, NULL}, + {"src/U4.cs", "X", "U4.U4", NULL, "ambiguous", NULL, NULL}, + {"src/U5.cs", "Z", "U5.U5", "D.Z", NULL, NULL, NULL}, + {"src/U6.cs", "W", "U6.U6", NULL, "test_only_target", NULL, NULL}, + {"src/tests/UT.cs", "W", "UT.UT", "T.W", NULL, NULL, NULL}, + {"src/tests/UT.cs", "X", "UT.UT", NULL, "ambiguous", NULL, NULL}, + {"other/UO.cs", "X", "UO.UO", NULL, "missing", NULL, NULL}, + {"other/UO.cs", "Z", "UO.UO", NULL, "missing", NULL, NULL}, + }; + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_unit_same_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + for (int i = 0; i < DM_COUNT(files); i++) { + th_write_file(TH_PATH(repo, files[i].path), files[i].text); + } + char with_db[512]; + char without_db[512]; + snprintf(with_db, sizeof(with_db), "%s/with.db", tmp); + snprintf(without_db, sizeof(without_db), "%s/without.db", tmp); + int with_rc = dm_index(repo, with_db, NULL); + cbm_doclink_cs_test_unit_memo(false); + int without_rc = dm_index(repo, without_db, NULL); + cbm_doclink_cs_test_unit_memo(true); + char *with = dm_doclink_state(with_db); + char *without = dm_doclink_state(without_db); + bool same = with && without && strcmp(with, without) == 0; + if (!same) { + printf(" with the unit memo\n%s without it\n%s", with ? with : "(null)", + without ? without : "(null)"); + } + int bad = 0; + for (int i = 0; i < DM_COUNT(wants); i++) { + bad += dm_want_failed(with_db, &wants[i]); + } + free(with); + free(without); + dm_unlink_db(with_db); + dm_unlink_db(without_db); + th_rmtree(tmp); + ASSERT_EQ(with_rc, 0); + ASSERT_EQ(without_rc, 0); + ASSERT_TRUE(same); + ASSERT_EQ(bad, 0); + PASS(); +} + +/* S5: k namespaces all declare `Shared` and one file imports all of them; + * `refs` documented classes name it (0 or 1). The resolver's lookup work and + * the reason of the row (empty when there is none). */ +static uint64_t dm_ambiguous_name_work(int k, int refs, char *reason, size_t cap) { + enum { SRC_CAP = 64 * 1024 }; + char *src = malloc(SRC_CAP); + char tmp[256] = "/tmp/cbm_dm_ambiguous_XXXXXX"; + reason[0] = '\0'; + if (!src || !cbm_mkdtemp(tmp)) { + free(src); + return 0; + } + size_t w = 0; + for (int i = 0; i < k; i++) { + w += (size_t)snprintf(src + w, SRC_CAP - w, "namespace P%d { public class Shared { } }\n", + i); + } + th_write_file(TH_PATH(tmp, "src/Far.cs"), src); + w = 0; + for (int i = 0; i < k; i++) { + w += (size_t)snprintf(src + w, SRC_CAP - w, "using P%d;\n", i); + } + w += (size_t)snprintf(src + w, SRC_CAP - w, "namespace N0\n{\n"); + for (int j = 0; j < refs; j++) { + w += (size_t)snprintf(src + w, SRC_CAP - w, + "/// \n" + "public class L%d { }\n", + j); + } + snprintf(src + w, SRC_CAP - w, "public class Last { }\n}\n"); + th_write_file(TH_PATH(tmp, "src/Uses.cs"), src); + free(src); + char db[512]; + snprintf(db, sizeof(db), "%s/ambiguous.db", tmp); + cbm_doclink_cs_test_work_reset(); + uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_work() : 0; + dm_row(db, "src/Uses.cs", "Shared", reason, cap, NULL, 0); + dm_unlink_db(db); + th_rmtree(tmp); + return work; +} + +/* One lookup stops at its second candidate: the name is ambiguous whatever + * the directives after it bring. The work of one ambiguous reference (the + * run with it less the run without it) does not grow with the k imports that + * all declare the name. */ +TEST(doc_mentions_cs_lookup_ambiguous_stops) { + static const int ks[2] = {50, 200}; + uint64_t one[2] = {0}; + bool ambiguous = true; + for (int i = 0; i < 2; i++) { + char reason[64]; + char none[64]; + uint64_t with = dm_ambiguous_name_work(ks[i], 1, reason, sizeof(reason)); + uint64_t without = dm_ambiguous_name_work(ks[i], 0, none, sizeof(none)); + one[i] = with > without ? with - without : 0; + ambiguous = ambiguous && strcmp(reason, "ambiguous") == 0 && !none[0]; + } + printf(" one ambiguous reference, k 50 -> 200 imports that declare it: %llu -> %llu steps, " + "want at most 1.5 times\n", + (unsigned long long)one[0], (unsigned long long)one[1]); + ASSERT_TRUE(ambiguous); + ASSERT_GT(one[0], 0); + ASSERT_LTE(2 * one[1], 3 * one[0]); + PASS(); +} + +/* Parent/global imports are ready for a child's directives. The child's own + * directives stay excluded while those targets are resolved. */ +TEST(doc_mentions_cs_import_stage_aliases) { + static const dm_source_t files[] = { + {"App.csproj", DM_EMPTY_PROJECT}, + {"Defs.cs", "namespace Lib { public class Target { } public class Other { } " + "public static class Statics { public static int Value; } }\n"}, + {"Globals.cs", "global using Base = Lib;\nglobal using static Lib.Statics;\n"}, + {"Use.cs", "namespace App {\nusing Parent = Lib;\nnamespace Child {\n" + "using Pick = Parent.Target;\nusing Again = Base.Target;\n" + "/// \n" + "public class Use { }\n}\nnamespace Sibling {\n" + "using Parent = Absent;\nusing Decoy = Parent.Target;\n" + "/// \n" + "public class Mask { }\n}\n}\n"}, + {"Aliases.cs", "using Clash = Lib.Target;\nusing Z = Absent;\n" + "using Same = Lib.Target;\nusing Noise = Lib;\nusing Same = Lib.Other;\n" + "public class Clash { }\n" + "/// \n" + "public class Aliases { }\n"}, + }; + static const dm_want_t wants[] = { + {"Use.cs", "Pick", "Use.Use", "Defs.Target", NULL, NULL, DM_EXACT}, + {"Use.cs", "Again", "Use.Use", "Defs.Target", NULL, NULL, DM_EXACT}, + {"Use.cs", "Value", "Use.Use", "Defs.Statics.Value", NULL, NULL, NULL}, + {"Use.cs", "Decoy", "Use.Mask", "Defs.Target", NULL, NULL, DM_EXACT}, + {"Use.cs", "Parent", "Use.Mask", NULL, "external", NULL, NULL}, + {"Aliases.cs", "Clash", "Aliases.Aliases", NULL, "ambiguous", "Defs.Target", NULL}, + {"Aliases.cs", "Same", "Aliases.Aliases", NULL, "ambiguous", "Defs.Target", NULL}, + {"Aliases.cs", "Z", "Aliases.Aliases", NULL, "external", NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("importstage", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* Static imports see both parts of a type; unseen named parts still make an + * outsider's reference ambiguous. No unrelated directive may hide that fact. */ +TEST(doc_mentions_cs_import_static_parts) { + enum { CAP = DM_COST_TEXT * 4 }; + char *noise = malloc(CAP); + char *own = malloc(CAP); + char *outside = malloc(CAP); + ASSERT_NOT_NULL(noise); + ASSERT_NOT_NULL(own); + ASSERT_NOT_NULL(outside); + size_t n = 0, u = 0; + for (int i = 0; i < 40; i++) { + n += (size_t)snprintf(noise + n, CAP - n, + "namespace Noise.N%d { public static class K%d { " + "public static int Other%d; } }\n", + i, i, i); + u += (size_t)snprintf(own + u, CAP - u, "using static Noise.N%d.K%d;\n", i, i); + } + u += (size_t)snprintf(own + u, CAP - u, "using static Lib.Tool;\nusing static Lib.TestOnly;\n"); + memcpy(outside, own, u); + snprintf(own + u, CAP - u, + "/// " + "\npublic class Use { }\n"); + snprintf(outside + u, CAP - u, + "/// " + "\npublic class Outside { }\n"); + const dm_source_t files[] = { + {"shared/Core.cs", "namespace Lib { public partial class Tool { " + "public static int Common; public class Shared { } } }\n"}, + {"shared/Noise.cs", noise}, + {"tests/TestOnly.cs", "namespace Lib { public static class TestOnly { " + "public static int Hidden; } }\n"}, + {"one/One.csproj", DM_EMPTY_PROJECT}, + {"one/Part.cs", "namespace Lib { public partial class Tool { " + "public static int Own; public class OnlyOne { } } }\n"}, + {"one/Use.cs", own}, + {"two/Two.csproj", DM_EMPTY_PROJECT}, + {"two/Outside.cs", outside}, + }; + static const dm_want_t wants[] = { + {"one/Use.cs", "Common", "Use.Use", "Core.Tool.Common", NULL, NULL, NULL}, + {"one/Use.cs", "Shared", "Use.Use", "Core.Tool.Shared", NULL, NULL, NULL}, + {"one/Use.cs", "Own", "Use.Use", "Part.Tool.Own", NULL, NULL, NULL}, + {"one/Use.cs", "Hidden", "Use.Use", NULL, "test_only_target", "TestOnly.TestOnly.Hidden", + NULL}, + {"two/Outside.cs", "Common", "Outside.Outside", "Core.Tool.Common", NULL, NULL, NULL}, + {"two/Outside.cs", "OnlyOne", "Outside.Outside", NULL, "ambiguous", "Part.Tool.OnlyOne", + NULL}, + {"two/Outside.cs", "Own", "Outside.Outside", NULL, "ambiguous", "Part.Tool.Own", NULL}, + {"two/Outside.cs", "Gone", "Outside.Outside", NULL, "missing", NULL, NULL}, + }; + int bad = dm_check_repo("staticparts", files, DM_COUNT(files), wants, DM_COUNT(wants)); + free(noise); + free(own); + free(outside); + ASSERT_EQ(bad, 0); + PASS(); +} + +/* A failed candidate buffer must use the complete existing scan. */ +TEST(doc_mentions_cs_import_candidate_allocation) { + enum { CAP = DM_COST_TEXT * 4, IMPORTS = 128, MATCHES = 40 }; + char *defs = malloc(CAP); + char *use = malloc(CAP); + ASSERT_NOT_NULL(defs); + ASSERT_NOT_NULL(use); + size_t d = 0, u = 0; + for (int i = 0; i < IMPORTS; i++) { + d += (size_t)snprintf(defs + d, CAP - d, "namespace N%d { public class %s { } }\n", i, + i < MATCHES ? "Common" : "Other"); + u += (size_t)snprintf(use + u, CAP - u, "using N%d;\n", i); + } + snprintf(use + u, CAP - u, "/// \npublic class Use { }\n"); + const dm_source_t files[] = {{"Defs.cs", defs}, {"Use.cs", use}}; + static const dm_want_t wants[] = { + {"Use.cs", "Common", "Use.Use", NULL, "ambiguous", NULL, NULL}, + }; + bool ok = true; + for (int fail = 0; fail < 2; fail++) { + cbm_doclink_cs_test_fail_candidate_alloc(fail != 0); + int bad = dm_check_repo("importalloc", files, DM_COUNT(files), wants, DM_COUNT(wants)); + bool hit = cbm_doclink_cs_test_candidate_alloc_failed(); + cbm_doclink_cs_test_fail_candidate_alloc(false); + printf(" import candidate allocation: injected=%d consumed=%d failures=%d\n", fail, hit, + bad); + ok = ok && bad == 0 && hit == (fail != 0); + } + free(defs); + free(use); + ASSERT_TRUE(ok); + PASS(); +} + +/* Enclosing scopes cost one step each. Import lookup uses the queried name; + * unrelated directives add only index probes. Other fixtures retain their + * original overload, source-line, signature and project-directory contracts. */ +TEST(doc_mentions_cs_lookup_cost) { + struct { + const char *what; + uint64_t small; + uint64_t large; + double low; + double high; + } runs[] = { + {"namespace depth 16 -> 32", dm_lookup_work(16, 0, 1, true), dm_lookup_work(32, 0, 1, true), + 1.5, 2.5}, + {"type nesting 16 -> 32", dm_lookup_work(1, 0, 16, true), dm_lookup_work(1, 0, 32, true), + 1.5, 2.5}, + {"usings 60 -> 120", dm_lookup_work(1, 60, 1, true), dm_lookup_work(1, 120, 1, true), 0.9, + 1.4}, + {"usings 60 -> 120, names no file declares", dm_lookup_work(1, 60, 1, false), + dm_lookup_work(1, 120, 1, false), 0.9, 1.1}, + {"test-only operator overloads 100 -> 200", dm_operator_work(100), dm_operator_work(200), + 0.9, 1.1}, + {"members on the documented line 100 -> 200", dm_line_work(100), dm_line_work(200), 0.9, + 1.1}, + {"parameters of 100 overloads none of which fits 4 -> 8", dm_signature_work(4), + dm_signature_work(8), 0.9, 1.1}, + {"directory depth 6 -> 48", dm_tree_work(6), dm_tree_work(48), 0.9, 2.5}, + }; + for (size_t i = 0; i < sizeof(runs) / sizeof(runs[0]); i++) { + double ratio = runs[i].small ? (double)runs[i].large / (double)runs[i].small : 0.0; + printf(" %s: %llu -> %llu steps, %.2f times the work, want %.1f to %.1f\n", runs[i].what, + (unsigned long long)runs[i].small, (unsigned long long)runs[i].large, ratio, + runs[i].low, runs[i].high); + if (runs[i].small < DM_COST_REFS || ratio < runs[i].low || ratio > runs[i].high) { + FAIL("lookup cost"); + } + } + PASS(); +} + +enum { DM_MANY = 300, DM_MANY_TEXT = 65536 }; + +/* No limit decides silently. More usings than any fixed list holds are all + * asked. A type with more overloads of one name than a lookup compares still + * binds the overload a reference writes out; what would need the comparison + * is `ambiguous`, never a guess and never `missing`. The same for a project + * that holds more complete declarations of one type than a lookup compares + * (beside another project of the assembly that holds one): none of them is + * picked. */ +TEST(doc_mentions_cs_no_silent_limits) { + char *decls = malloc(DM_MANY_TEXT); + char *uses = malloc(DM_MANY_TEXT); + char *big = malloc(DM_MANY_TEXT); + char *flav = malloc(DM_MANY_TEXT); + ASSERT_NOT_NULL(decls); + ASSERT_NOT_NULL(uses); + ASSERT_NOT_NULL(big); + ASSERT_NOT_NULL(flav); + size_t d = 0; + size_t u = 0; + size_t b = 0; + size_t f = (size_t)snprintf(flav, DM_MANY_TEXT, "namespace Lib\n{\n"); + for (int i = 0; i < DM_MANY; i++) { + f += (size_t)snprintf(flav + f, DM_MANY_TEXT - f, " public class Flav { }\n"); + } + snprintf(flav + f, DM_MANY_TEXT - f, "}\n"); + for (int i = 0; i < DM_MANY; i++) { + d += (size_t)snprintf(decls + d, DM_MANY_TEXT - d, + "namespace Many.N%03d { public class C%03d { } }\n", i, i); + u += (size_t)snprintf(uses + u, DM_MANY_TEXT - u, "using Many.N%03d;\n", i); + } + snprintf(uses + u, DM_MANY_TEXT - u, + "namespace App\n" + "{\n" + " /// \n" + " public class Uses { }\n" + "}\n"); + b += (size_t)snprintf(big + b, DM_MANY_TEXT - b, + "namespace App\n{\n public class Big\n {\n"); + for (int i = 0; i < DM_MANY; i++) { + b += (size_t)snprintf(big + b, DM_MANY_TEXT - b, " public void M(P%03d a) { }\n", i); + } + snprintf(big + b, DM_MANY_TEXT - b, + " }\n" + "\n" + " /// \n" + " public class Written { }\n" + "\n" + " /// \n" + " public class Unwritten { }\n" + "}\n"); + const dm_source_t files[] = { + {"src/Decls.cs", decls}, + {"src/Uses.cs", uses}, + {"src/Big.cs", big}, + {"a/Lib.csproj", DM_EMPTY_PROJECT}, + {"a/Many.cs", flav}, + {"a/UsesA.cs", "namespace Lib\n" + "{\n" + " /// \n" + " public class UsesA { }\n" + "}\n"}, + {"b/Lib.csproj", DM_EMPTY_PROJECT}, + {"b/Flav.cs", "namespace Lib\n{\n public class Flav { }\n}\n"}, + {"b/UsesB.cs", "namespace Lib\n" + "{\n" + " /// \n" + " public class UsesB { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"src/Uses.cs", "C000", "Uses.Uses", "Decls.C000", NULL, NULL, NULL}, + {"src/Uses.cs", "C299", "Uses.Uses", "Decls.C299", NULL, NULL, NULL}, + {"src/Big.cs", "Big.M(P299)", "Big.Written", "Big.Big.M", NULL, NULL, NULL}, + {"src/Big.cs", "Big.M", "Big.Unwritten", NULL, "ambiguous", "Big.Big.M", NULL}, + {"src/Big.cs", "Big.M(Other)", "Big.Unwritten", NULL, "ambiguous", "Big.Big.M", NULL}, + /* more declarations in the own project than a lookup compares: none + * is picked -- and the other project of the assembly binds its own */ + {"a/UsesA.cs", "Flav", "UsesA.UsesA", NULL, "ambiguous", "Many.Flav", NULL}, + {"b/UsesB.cs", "Flav", "UsesB.UsesB", "b.Flav.Flav", NULL, "Many.Flav", NULL}, + }; + int bad = dm_check_repo("limits", files, DM_COUNT(files), wants, DM_COUNT(wants)); + free(decls); + free(uses); + free(big); + free(flav); + ASSERT_EQ(bad, 0); + PASS(); +} + +enum { DM_WIDE = 66 }; /* more assemblies than one lookup compares */ + +/* The number of assemblies that hold a part or a stub of a type decides + * nothing. A contract is joined to its one implementation however many + * assemblies hold that contract; where the implementation has no node, the + * one stub-only assembly's stub stands in however many assemblies have parts + * of the type; and a name is told from what an unseen part declares by + * looking the name up, not by counting the assemblies: one that no part + * declares is `missing`, and an unrelated name written inside such a type is + * bound as anywhere else. */ +TEST(doc_mentions_cs_many_assemblies) { + static const char stub[] = "namespace Acme.Coll\n" + "{\n" + " public partial class Thing { public void Go() { } }\n" + "\n" + " /// \n" + " public partial class UsesStub { }\n" + "}\n"; + static const char part[] = "namespace Acme.Coll\n" + "{\n" + " public partial class Phantom { public void Own() { } }\n" + " public partial class Wide { public void Extra() { } }\n" + "}\n"; + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_wide_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + th_write_file(TH_PATH(tmp, "shared/impl/Things.cs"), + "namespace Acme.Coll\n" + "{\n" + " public class Thing { public void Go() { } }\n" + " public partial class Phantom { public void Boo() { } }\n" + " public partial class Phantom { }\n" + "\n" + " public partial class Wide\n" + " {\n" + " /// \n" + " public void Common() { }\n" + " }\n" + "\n" + " /// \n" + " /// \n" + " public class UsesShared { }\n" + "}\n"); + th_write_file(TH_PATH(tmp, "facade/ref/Facade.csproj"), DM_EMPTY_PROJECT); + th_write_file(TH_PATH(tmp, "facade/ref/Stubs.cs"), + "namespace Acme.Coll\n" + "{\n" + " public partial class Phantom { public void Boo() { } }\n" + "}\n"); + for (int i = 0; i < DM_WIDE; i++) { + char rel[64]; + snprintf(rel, sizeof(rel), "s%02d/ref/S%02d.csproj", i, i); + th_write_file(TH_PATH(tmp, rel), DM_EMPTY_PROJECT); + snprintf(rel, sizeof(rel), "s%02d/ref/Stub.cs", i); + th_write_file(TH_PATH(tmp, rel), stub); + snprintf(rel, sizeof(rel), "p%02d/P%02d.csproj", i, i); + th_write_file(TH_PATH(tmp, rel), DM_EMPTY_PROJECT); + snprintf(rel, sizeof(rel), "p%02d/Part.cs", i); + th_write_file(TH_PATH(tmp, rel), part); + } + static const dm_want_t wants[] = { + /* DM_WIDE assemblies hold the contract: each is joined */ + {"s00/ref/Stub.cs", "Thing", "s00.ref.Stub.UsesStub", "impl.Things.Thing", NULL, + "s00.ref.Stub.Thing", DM_UNIQUE}, + {"s65/ref/Stub.cs", "Thing", "s65.ref.Stub.UsesStub", "impl.Things.Thing", NULL, + "s65.ref.Stub.Thing", DM_UNIQUE}, + /* DM_WIDE assemblies have parts of the type, one has only its stub: + * that stub stands in for the implementation without a node */ + {"shared/impl/Things.cs", "Phantom", "Things.UsesShared", "facade.ref.Stubs.Phantom", NULL, + NULL, DM_UNIQUE}, + /* what the unseen parts declare, and what they do not */ + {"shared/impl/Things.cs", "Wide.Extra", "Things.UsesShared", NULL, "ambiguous", NULL, NULL}, + {"shared/impl/Things.cs", "Wide.Nothing", "Things.UsesShared", NULL, "missing", NULL, NULL}, + {"shared/impl/Things.cs", "Thing", "Things.Wide.Common", "impl.Things.Thing", NULL, NULL, + NULL}, + }; + char db[512]; + snprintf(db, sizeof(db), "%s/wide.db", tmp); + int bad = dm_index(tmp, db, NULL) == 0 ? 0 : -1; + for (int i = 0; bad >= 0 && i < DM_COUNT(wants); i++) { + bad += dm_want_failed(db, &wants[i]); + } + dm_unlink_db(db); + th_rmtree(tmp); + ASSERT_EQ(bad, 0); + PASS(); +} + +/* A using directive of a file whose type has a base list is no private + * matter of that file: the base list is resolved through it, and what a type + * derives from decides how another file's reference to its members comes + * out. Changing it must not be repaired as if only the file had changed. */ +TEST(doc_mentions_incremental_using_of_based_type) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_incu_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + th_write_file(TH_PATH(repo, "src/V1.cs"), + "namespace Lib.V1\n{\n public class Base { public void Run() { } }\n}\n"); + th_write_file(TH_PATH(repo, "src/F.cs"), "using Lib.V1;\n" + "\n" + "namespace App\n" + "{\n" + " public class Job : Base { }\n" + "}\n"); + th_write_file(TH_PATH(repo, "src/G.cs"), "namespace App\n" + "{\n" + " /// \n" + " public class Uses { }\n" + "}\n"); + char inc_db[512]; + char full_db[512]; + snprintf(inc_db, sizeof(inc_db), "%s/inc.db", tmp); + snprintf(full_db, sizeof(full_db), "%s/full.db", tmp); + ASSERT_EQ(dm_index(repo, inc_db, NULL), 0); + char reason[64]; + /* Job derives from a class of the repository: what it lacks is missing */ + dm_row(inc_db, "src/G.cs", "Job.Nope", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "missing"); + /* the using now names a namespace the repository does not declare: Job's + * base is outside, and so may be the member G.cs asks for */ + th_write_file(TH_PATH(repo, "src/F.cs"), "using Lib.V9;\n" + "\n" + "namespace App\n" + "{\n" + " public class Job : Base { }\n" + "}\n"); + ASSERT_EQ(dm_step(repo, inc_db, full_db, "using of a type with a base list changed", + CBM_INCREMENTAL_ROUTE_FORCED_FULL), + 0); + dm_row(inc_db, "src/G.cs", "Job.Nope", reason, sizeof(reason), NULL, 0); + ASSERT_STR_EQ(reason, "external"); + dm_unlink_db(inc_db); + dm_unlink_db(full_db); + th_rmtree(tmp); + PASS(); +} + +enum { DM_PROJECTS = 40 }; + +/* A directory may hold any number of project files, and every one of them + * sets global usings of the files there: none is left out, whatever order a + * directory listing would have had. */ +TEST(doc_mentions_msbuild_many_project_files) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_many_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + enum { SHARED_RECORDS = 512 }; + char *props = dm_repeated("", "G", SHARED_RECORDS, + ""); + ASSERT_NOT_NULL(props); + th_write_file(TH_PATH(tmp, "Directory.Build.props"), props); + free(props); + char *targets = + dm_repeated("", "$(Shared)", + SHARED_RECORDS, ""); + ASSERT_NOT_NULL(targets); + th_write_file(TH_PATH(tmp, "Directory.Build.targets"), targets); + free(targets); + char *decls = malloc(DM_MANY_TEXT); + char *uses = malloc(DM_MANY_TEXT); + ASSERT_NOT_NULL(decls); + ASSERT_NOT_NULL(uses); + size_t d = 0; + size_t u = (size_t)snprintf(uses, DM_MANY_TEXT, "namespace App\n{\n /// "); + for (int i = 0; i < DM_PROJECTS; i++) { + char rel[64]; + char xml[256]; + snprintf(rel, sizeof(rel), "src/Many/P%02d.csproj", i); + snprintf(xml, sizeof(xml), + "\n" + " \n" + "\n", + i); + th_write_file(TH_PATH(tmp, rel), xml); + d += (size_t)snprintf(decls + d, DM_MANY_TEXT - d, + "namespace G.N%02d { public class K%02d { } }\n", i, i); + u += (size_t)snprintf(uses + u, DM_MANY_TEXT - u, " ", i); + } + snprintf(uses + u, DM_MANY_TEXT - u, "\n public class Uses { }\n}\n"); + th_write_file(TH_PATH(tmp, "src/Decl.cs"), decls); + th_write_file(TH_PATH(tmp, "src/Many/Uses.cs"), uses); + free(decls); + free(uses); + char db[512]; + snprintf(db, sizeof(db), "%s/many.db", tmp); + cbm_msb_test_cost_reset(); + ASSERT_EQ(dm_index(tmp, db, NULL), 0); + uint64_t records = 0; + uint64_t peak = 0; + cbm_msb_test_cost(&records, &peak); + fprintf(stderr, + "msbuild pipeline projects=%d shared=%d interpreted=%llu work=%llu " + "peak=%llu live=%llu\n", + DM_PROJECTS, SHARED_RECORDS, (unsigned long long)records, + (unsigned long long)cbm_msb_test_work(), (unsigned long long)peak, + (unsigned long long)cbm_msb_test_value_live_bytes()); + ASSERT_EQ(dm_mentions_from(db, "Uses.Uses"), DM_PROJECTS); + ASSERT_EQ(dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"), 0); + bool shared_once = records <= 2 * SHARED_RECORDS + 16 * DM_PROJECTS; + bool released = cbm_msb_test_value_live_bytes() == 0; + dm_unlink_db(db); + th_rmtree(tmp); + ASSERT_TRUE(shared_once); + ASSERT_TRUE(released); + PASS(); +} + +enum { DM_NEST = 70, DM_NEST_KEPT = 64, DM_NEST_TEXT = 8192 }; + +static int dm_count_lines(const char *blob, const char *prefix) { + int n = 0; + size_t pl = strlen(prefix); + for (const char *p = blob; p && *p;) { + n += strncmp(p, prefix, pl) == 0; + p = strchr(p, '\n'); + p = p ? p + 1 : NULL; + } + return n; +} + +/* The scope records nesting to a depth no program has. What nests deeper is + * not placed, and says so: the type names are kept (to be resolved to + * nothing else), the lines have no scope. Nothing is cut silently. */ +TEST(doc_mentions_cs_scan_nesting_limits) { + char *src = malloc(DM_NEST_TEXT); + ASSERT_NOT_NULL(src); + size_t w = (size_t)snprintf(src, DM_NEST_TEXT, "namespace N\n{\n"); + for (int i = 0; i < DM_NEST; i++) { + w += (size_t)snprintf(src + w, DM_NEST_TEXT - w, "public class T%d\n{\n", i); + } + for (int i = 0; i <= DM_NEST; i++) { + w += (size_t)snprintf(src + w, DM_NEST_TEXT - w, "}\n"); + } + CBMFileResult *r = dm_scope(src); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + ASSERT_EQ(dm_count_lines(r->doc_scope, "T\t"), DM_NEST_KEPT); + ASSERT_NOT_NULL(strstr(r->doc_scope, "\tT63\t")); + ASSERT_NULL(strstr(r->doc_scope, "\tT64\t")); + ASSERT_EQ(dm_count_lines(r->doc_scope, "Q\t"), DM_NEST - DM_NEST_KEPT); + ASSERT_NOT_NULL(strstr(r->doc_scope, "\nQ\tT64\n")); + ASSERT_NOT_NULL(strstr(r->doc_scope, "\nQ\tT69\n")); + ASSERT_NOT_NULL(strstr(r->doc_scope, "\nX\t")); /* the lines of the 65th type */ + cbm_free_result(r); + + /* ... and a namespace of more segments than that */ + w = (size_t)snprintf(src, DM_NEST_TEXT, "namespace "); + for (int i = 0; i < DM_NEST; i++) { + w += (size_t)snprintf(src + w, DM_NEST_TEXT - w, "%sA%d", i ? "." : "", i); + } + snprintf(src + w, DM_NEST_TEXT - w, "\n{\n public class Deep { }\n}\n"); + r = dm_scope(src); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(r->doc_scope); + ASSERT_NULL(strstr(r->doc_scope, "\nR\t")); + ASSERT_NULL(strstr(r->doc_scope, "\nT\t")); + ASSERT_NOT_NULL(strstr(r->doc_scope, "\nQ\tDeep\n")); + ASSERT_NOT_NULL(strstr(r->doc_scope, "\nX\t")); + cbm_free_result(r); + free(src); + PASS(); +} + +/* Run one statement against the database; returns the rows it changed, -1 + * when it failed. */ +static int dm_exec(const char *db, const char *sql) { + sqlite3 *h = NULL; + int changed = -1; + if (sqlite3_open_v2(db, &h, SQLITE_OPEN_READWRITE, NULL) == SQLITE_OK && + sqlite3_exec(h, sql, NULL, NULL, NULL) == SQLITE_OK) { + changed = sqlite3_changes(h); + } + sqlite3_close(h); + return changed; +} + +/* One damaged stored scope (A's row, `find` replaced by `put`): 0 when the + * run that reads it fails visibly and the next one rebuilds; else the step + * that went wrong (1 setup, 2 the damaged run, 3 the rebuild), with the + * damaged run's error rows in *errors. */ +static int dm_damaged_scope_run(const char *find, const char *put, int *errors) { + *errors = -1; + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_dmg_XXXXXX"); + if (!cbm_mkdtemp(tmp)) { + return 1; + } + char repo[400]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + th_write_file(TH_PATH(repo, "src/A.cs"), "namespace N\n{\n public class Target { }\n}\n"); + th_write_file(TH_PATH(repo, "src/B.cs"), "namespace N\n" + "{\n" + " /// \n" + " public class Uses { }\n" + "}\n"); + char db[512]; + snprintf(db, sizeof(db), "%s/dmg.db", tmp); + char sql[512]; + snprintf(sql, sizeof(sql), + "UPDATE lsp_surface SET defs_json = replace(defs_json, '%s', '%s') " + "WHERE rel_path = 'src/A.cs' AND instr(defs_json, '%s') > 0", + find, put, find); + char props[512]; + int n = 0; + int step = 1; + bool ok = dm_index(repo, db, NULL) == 0; + if (ok) { + dm_edge(db, "B.Uses", "A.Target", props, sizeof(props), &n); + ok = n == 1 && dm_exec(db, sql) == 1; + } + if (ok) { + step = 2; + th_write_file(TH_PATH(repo, "src/B.cs"), + "namespace N\n" + "{\n" + " /// Again \n" + " public class Uses { }\n" + "}\n"); + cbm_pipeline_incremental_test_reset_faults(); + ok = dm_index(repo, db, NULL) == 0; + *errors = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'"); + /* never the row an index without A's declarations would write */ + ok = ok && *errors == 1 && + dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE raw = 'Target'") == 0; + } + if (ok) { + /* the next run rebuilds everything, and the edge is back */ + step = 3; + ok = dm_index(repo, db, NULL) == 0 && + cbm_pipeline_incremental_test_last_route() == CBM_INCREMENTAL_ROUTE_FORCED_FULL && + dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'") == 0; + if (ok) { + dm_edge(db, "B.Uses", "A.Target", props, sizeof(props), &n); + ok = n == 1; + } + } + dm_unlink_db(db); + th_rmtree(tmp); + return ok ? 0 : step; +} + +/* A stored scope this code did not write -- a damaged row -- is not read + * around: the file's declarations would silently be missing from every + * lookup, and references to them would be reported `missing`. The layer + * fails for that run, visibly, and the next index rebuilds from the files. + * R8: a region with an empty name is such a record (the scanner never writes + * one); read, it would nest a region without deepening the namespace. */ +TEST(doc_mentions_cs_damaged_stored_scope) { + static const struct { + const char *what, *find, *put; + } damages[] = { + {"one field too many in A's type record", "\\nT\\t", "\\nT\\tX\\t"}, + {"A's region without a name", "\\nR\\t1\\t0\\t0\\t0\\tN\\n", "\\nR\\t1\\t0\\t0\\t0\\t\\n"}, + }; + bool ok = true; + for (size_t i = 0; i < sizeof(damages) / sizeof(damages[0]); i++) { + int errors = -1; + int step = dm_damaged_scope_run(damages[i].find, damages[i].put, &errors); + if (step != 0) { + printf(" %s: step %d went wrong (%d error rows)\n", damages[i].what, step, errors); + ok = false; + } + } + ASSERT_TRUE(ok); + PASS(); +} + +/* ── security review 2 ───────────────────────────────────────────── */ + +/* The portable copy of a scope writes 0 for every line number and is never + * longer than the scope it is made from: an empty line field stays empty + * (writing a 0 for it made the copy one byte longer per such field, past the + * end of its buffer). */ +TEST(doc_mentions_cs_portable_scope_bound) { + static const char blob[] = "cs1\n" + "R\t1\t0\t\t\tN\n" + "T\t1\t\t\tc\t-\tC\t\t\n" + "M\t\tc\t0\t0\tGo\t\t\n" + "X\t\t\n" + "X\t\t\n" + "X\t\t\n"; + char *portable = cbm_doclink_cs_portable_scope(blob); + ASSERT_NOT_NULL(portable); + size_t n = strlen(portable); + bool same = strcmp(portable, blob) == 0; + cbm_free(CBM_MEM_CLASS_OTHER, portable); + ASSERT_LTE(n, strlen(blob)); + ASSERT_TRUE(same); + PASS(); +} + +/* A reference whose brackets do not pair, or that goes on after its + * parameter list, is unparseable: it is never resolved by the part that + * can be read (the segments before the open bracket, the parameters before + * it). Decoys: the same references written whole bind. */ +TEST(doc_mentions_cs_unbalanced_brackets) { + static const dm_source_t files[] = { + {"src/Outer.cs", + "namespace Acme\n" + "{\n" + " public class Outer\n" + " {\n" + " public class Inner { }\n" + " public void M(int a) { }\n" + " public void M(int a, System.Collections.Generic.List b) { }\n" + " }\n" + "\n" + " /// \n" + " /// \n" + " /// \n" + " /// \n" + " public class Broken { }\n" + "\n" + " /// \n" + " public class Whole { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"src/Outer.cs", "Outer.Inner{T", "Outer.Broken", NULL, "unparseable", "Outer.Outer", NULL}, + {"src/Outer.cs", "Outer.Inner}", "Outer.Broken", NULL, "unparseable", "Outer.Outer", NULL}, + {"src/Outer.cs", "Outer.M(int", "Outer.Broken", NULL, "unparseable", "Outer.Outer.M", NULL}, + {"src/Outer.cs", "Outer.M(int, List{string)", "Outer.Broken", NULL, "unparseable", + "Outer.Outer.M", NULL}, + {"src/Outer.cs", "Outer.M(int))", "Outer.Broken", NULL, "unparseable", "Outer.Outer.M", + NULL}, + {"src/Outer.cs", "Outer.M(int)x", "Outer.Broken", NULL, "unparseable", "Outer.Outer.M", + NULL}, + {"src/Outer.cs", "Outer.Inner{T}x", "Outer.Broken", NULL, "unparseable", + "Outer.Outer.Inner", NULL}, + {"src/Outer.cs", "Outer.Inner{T}", "Outer.Whole", "Outer.Outer.Inner", NULL, NULL, NULL}, + {"src/Outer.cs", "Outer.M(int)", "Outer.Whole", "Outer.Outer.M", NULL, NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("brackets", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +enum { DM_DEEP_SEGMENTS = 70 }; /* past the scanner's 64 namespace segments */ + +/* A scope read back from the store is held to the nesting its writer keeps: + * a region whose namespace has more segments than the scanner ever writes + * fails the build of the run (bad_scope, the error row), as any other stored + * scope this code did not write. The next run rebuilds everything. */ +TEST(doc_mentions_cs_stored_scope_nesting) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_deep_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + th_write_file(TH_PATH(repo, "src/A.cs"), "namespace N\n{\n public class Target { }\n}\n"); + th_write_file(TH_PATH(repo, "src/B.cs"), "namespace N\n" + "{\n" + " /// \n" + " public class Uses { }\n" + "}\n"); + char db[512]; + snprintf(db, sizeof(db), "%s/deep.db", tmp); + ASSERT_EQ(dm_index(repo, db, NULL), 0); + char props[512]; + int n = 0; + dm_edge(db, "B.Uses", "A.Target", props, sizeof(props), &n); + ASSERT_EQ(n, 1); + /* A's region record names a namespace DM_DEEP_SEGMENTS segments deep */ + char sql[1024]; + size_t w = (size_t)snprintf(sql, sizeof(sql), + "UPDATE lsp_surface SET defs_json = replace(defs_json, " + "'\\tN\\n', '\\tN"); + for (int i = 1; i < DM_DEEP_SEGMENTS; i++) { + w += (size_t)snprintf(sql + w, sizeof(sql) - w, ".a"); + } + snprintf(sql + w, sizeof(sql) - w, + "\\n') WHERE rel_path = 'src/A.cs' AND instr(defs_json, '\\tN\\n') > 0"); + ASSERT_EQ(dm_exec(db, sql), 1); + th_write_file(TH_PATH(repo, "src/B.cs"), + "namespace N\n" + "{\n" + " /// Again \n" + " public class Uses { }\n" + "}\n"); + cbm_pipeline_incremental_test_reset_faults(); + ASSERT_EQ(dm_index(repo, db, NULL), 0); + int errors = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'"); + /* the next run rebuilds everything, and the edge is back */ + ASSERT_EQ(dm_index(repo, db, NULL), 0); + cbm_incremental_route_t route = cbm_pipeline_incremental_test_last_route(); + dm_edge(db, "B.Uses", "A.Target", props, sizeof(props), &n); + dm_unlink_db(db); + th_rmtree(tmp); + ASSERT_EQ(errors, 1); + ASSERT_EQ(route, CBM_INCREMENTAL_ROUTE_FORCED_FULL); + ASSERT_EQ(n, 1); + PASS(); +} + +/* One row whose reason this layer never writes, then index_status in both + * forms: no key carries that text, the row is counted under one fixed key, + * and the status is error. 0 when all of that holds. */ +static int dm_unknown_reason_checks(const char *db, const char *project) { + if (dm_exec(db, "INSERT INTO doc_link_unresolved(project, rel_path, line, syntax, raw, reason) " + "SELECT project, 'src/Injected.cs', 1, 'see', 'X', 'injected_reason_text' " + "FROM doc_link_unresolved LIMIT 1") != 1) { + return 1; + } + int bad = 0; + char *text = dm_index_status(project, false); + const char *block = text ? strstr(text, "doc_links:\n") : NULL; + if (!block || strstr(text, "injected_reason_text") || + !strstr(block, "\n unrecognized_reason: 1\n") || !strstr(block, "\n status: error\n")) { + printf(" index_status text: %s\n", block ? block : "(no doc_links block)"); + bad = 1; + } + free(text); + yyjson_doc *env = dm_preview_status(project, true); + yyjson_val *unresolved = yyjson_obj_get(dm_preview_report(env), "unresolved"); + if (!unresolved || yyjson_obj_get(unresolved, "injected_reason_text") || + yyjson_get_int(yyjson_obj_get(unresolved, "unrecognized_reason")) != 1 || + !dm_preview_status_is(env, "error")) { + printf(" index_status json: unknown reason shown, or status not error\n"); + bad = 1; + } + yyjson_doc_free(env); + return bad; +} + +TEST(doc_mentions_index_status_unknown_reason) { + ASSERT_EQ(dm_preview_fixture(dm_unknown_reason_checks), 0); + PASS(); +} + +enum { DM_HUGE_PROJECT = 1100000 }; /* past CSX_MAX_PROJECT_BYTES */ + +/* A project file the scan does not read -- larger than a project file is, + * as the project itself or as a file it imports -- holds what nobody knows, + * not nothing: a simple name found nowhere in such a project is external. + * Decoy: a project whose files were read keeps a closed scope (missing). */ +TEST(doc_mentions_msbuild_unread_project_opens) { + char *filler = malloc(DM_HUGE_PROJECT + 1); + ASSERT_NOT_NULL(filler); + memset(filler, 'x', DM_HUGE_PROJECT); + filler[DM_HUGE_PROJECT] = '\0'; + char *huge_project = + dm_repeated("\n"); + char *huge_props = dm_repeated("\n"); + free(filler); + ASSERT_NOT_NULL(huge_project); + ASSERT_NOT_NULL(huge_props); + const dm_source_t files[] = { + {"a/A.csproj", huge_project}, + {"a/UsesA.cs", "namespace App\n" + "{\n" + " /// \n" + " public class UsesA { }\n" + "}\n"}, + {"b/Directory.Build.props", huge_props}, + {"b/B.csproj", DM_EMPTY_PROJECT}, + {"b/UsesB.cs", "namespace App\n" + "{\n" + " /// \n" + " public class UsesB { }\n" + "}\n"}, + {"c/C.csproj", DM_EMPTY_PROJECT}, + {"c/UsesC.cs", "namespace App\n" + "{\n" + " /// \n" + " public class UsesC { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"a/UsesA.cs", "NowhereA", "UsesA.UsesA", NULL, "external", NULL, NULL}, + {"b/UsesB.cs", "NowhereB", "UsesB.UsesB", NULL, "external", NULL, NULL}, + {"c/UsesC.cs", "NowhereC", "UsesC.UsesC", NULL, "missing", NULL, NULL}, + }; + int bad = dm_check_repo("unread", files, DM_COUNT(files), wants, DM_COUNT(wants)); + free(huge_project); + free(huge_props); + ASSERT_EQ(bad, 0); + PASS(); +} + +/* R6: a *.props or *.targets file that does not parse up to its root is not + * read either: what it holds is unknown. As the nearest Directory.Build.props + * it is the one MSBuild reads, so the scope of its project is open (a simple + * name found nowhere is external) and the usable file further up is not + * applied (its `Using Include="Lib"` imports nothing here). Decoy: a project + * with no nearer file reads the upper one (Gadget binds through it, a name + * found nowhere is missing). */ +TEST(doc_mentions_msbuild_malformed_props_opens) { + const dm_source_t files[] = { + {"Directory.Build.props", "" + "\n"}, + {"a/Directory.Build.props", "<\n" + "\n"}, + {"a/A.csproj", DM_EMPTY_PROJECT}, + {"a/Thing.cs", "namespace Lib\n{\n public class Thing { }\n}\n"}, + {"a/UsesA.cs", "namespace App\n" + "{\n" + " /// \n" + " public class UsesA { }\n" + "}\n"}, + {"c/C.csproj", DM_EMPTY_PROJECT}, + {"c/Gadget.cs", "namespace Lib\n{\n public class Gadget { }\n}\n"}, + {"c/UsesC.cs", + "namespace App\n" + "{\n" + " /// \n" + " public class UsesC { }\n" + "}\n"}, + }; + static const dm_want_t wants[] = { + {"a/UsesA.cs", "NowhereA", "UsesA.UsesA", NULL, "external", NULL, NULL}, + {"a/UsesA.cs", "Thing", "UsesA.UsesA", NULL, "external", "Thing.Thing", NULL}, + {"c/UsesC.cs", "Gadget", "UsesC.UsesC", "Gadget.Gadget", NULL, NULL, NULL}, + {"c/UsesC.cs", "NowhereC", "UsesC.UsesC", NULL, "missing", NULL, NULL}, + }; + ASSERT_EQ(dm_check_repo("malformed_props", files, DM_COUNT(files), wants, DM_COUNT(wants)), 0); + PASS(); +} + +/* Sized so that the large file parses far within the extraction budget on + * the slowest sanitizer leg, under load too (20,000 fields were cut there, + * and 4,000 beside five other suites, leaving the test without its large + * file). Every small file costs at least the scratch table's minimum size: + * these 20 and the large file come to 3,008 steps, within the bound, while + * the regression, DM_SMALL_FILES x 2 x DM_BIG_MEMBERS, is ten times it. */ +enum { DM_BIG_MEMBERS = 1000, DM_SMALL_FILES = 20 }; + +/* What the resolver looks at to build its index over one file of + * DM_BIG_MEMBERS fields followed (in path order) by DM_SMALL_FILES files of + * one type and one field each. 0 when the repository cannot be indexed or the + * large file did not reach the index whole. */ +static uint64_t dm_scratch_work(void) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_scratch_XXXXXX"); + if (!cbm_mkdtemp(tmp)) { + return 0; + } + size_t cap = (size_t)DM_BIG_MEMBERS * 32 + 64; + char *src = malloc(cap); + if (!src) { + th_rmtree(tmp); + return 0; + } + size_t w = (size_t)snprintf(src, cap, "namespace Big\n{\n public class Wide\n {\n"); + for (int i = 0; i < DM_BIG_MEMBERS; i++) { + w += (size_t)snprintf(src + w, cap - w, " public int f%d;\n", i); + } + snprintf(src + w, cap - w, " }\n}\n"); + th_write_file(TH_PATH(tmp, "a/Big.cs"), src); + free(src); + for (int i = 0; i < DM_SMALL_FILES; i++) { + char rel[64]; + char text[256]; + snprintf(rel, sizeof(rel), "b/S%03d.cs", i); + snprintf(text, sizeof(text), + "namespace Small\n{\n%s public class C%03d { public int x; }\n}\n", + i == 0 ? " /// \n" : "", i); + th_write_file(TH_PATH(tmp, rel), text); + } + char db[512]; + snprintf(db, sizeof(db), "%s/scratch.db", tmp); + cbm_doclink_cs_test_work_reset(); + uint64_t work = dm_index(tmp, db, NULL) == 0 ? cbm_doclink_cs_test_scratch_work() : 0; + int rows = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved"); + int fields = dm_count(db, "SELECT COUNT(*) FROM nodes WHERE label = 'Field' AND " + "file_path = 'a/Big.cs'"); + if (fields != DM_BIG_MEMBERS) { + printf(" scratch table: a/Big.cs reached the index with %d of %d fields\n", fields, + (int)DM_BIG_MEMBERS); + } + dm_unlink_db(db); + th_rmtree(tmp); + return rows == 1 && fields == DM_BIG_MEMBERS ? work : 0; +} + +/* The per-file scratch table of the index build is not emptied at the size + * one large file grew it to: after a file of DM_BIG_MEMBERS names, every one + * of DM_SMALL_FILES small files costs about the table's minimum size. The + * work stays within a small multiple of all the names the files declare + * (emptying the large table for every small file cost DM_SMALL_FILES times + * its size). */ +TEST(doc_mentions_cs_scratch_table_work) { + uint64_t names = (uint64_t)DM_BIG_MEMBERS + (uint64_t)DM_SMALL_FILES * 2; + uint64_t work = dm_scratch_work(); + printf(" scratch table: %llu steps for %llu names\n", (unsigned long long)work, + (unsigned long long)names); + ASSERT_TRUE(work > 0); + ASSERT_LTE(work, names * 4); + PASS(); +} + +/* An incremental repair finds the files whose unresolved rows name what a + * changed file no longer declares by the identifiers the resolver reads -- + * a name with a non-ASCII letter included. The parts of `Hub` in two shared + * trees both declare the member `Grüße` (the reference is ambiguous); one of + * them stops declaring it, and the repaired index equals a full one (the + * reference binds). */ +TEST(doc_mentions_incremental_non_ascii_name) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_utf8name_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400]; + char db[512]; + char full_db[512]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + snprintf(db, sizeof(db), "%s/inc.db", tmp); + snprintf(full_db, sizeof(full_db), "%s/full.db", tmp); + th_write_file(TH_PATH(repo, "t1/X.cs"), + "namespace N\n{\n public partial class Hub { public void Gr\xC3\xBC\xC3\x9F" + "e() { } }\n}\n"); + th_write_file(TH_PATH(repo, "t2/Y.cs"), + "namespace N\n{\n public partial class Hub { public void Gr\xC3\xBC\xC3\x9F" + "e() { } public void Keep() { } }\n}\n"); + th_write_file(TH_PATH(repo, "p/P.csproj"), DM_EMPTY_PROJECT); + th_write_file(TH_PATH(repo, "p/Uses.cs"), + "namespace App\n" + "{\n" + " /// \n" + " public class Uses { }\n" + "}\n"); + ASSERT_EQ(dm_index(repo, db, NULL), 0); + char reason[64]; + dm_row(db, "p/Uses.cs", + "N.Hub.Gr\xC3\xBC\xC3\x9F" + "e", + reason, sizeof(reason), NULL, 0); + bool ambiguous = strcmp(reason, "ambiguous") == 0; + th_write_file(TH_PATH(repo, "t2/Y.cs"), + "namespace N\n{\n public partial class Hub { public void Keep() { } }\n}\n"); + int step = dm_step(repo, db, full_db, "a name with a non-ASCII letter removed", + CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR); + char props[512]; + int n = 0; + dm_edge(db, "Uses.Uses", + "X.Hub.Gr\xC3\xBC\xC3\x9F" + "e", + props, sizeof(props), &n); + dm_unlink_db(db); + dm_unlink_db(full_db); + th_rmtree(tmp); + ASSERT_TRUE(ambiguous); + ASSERT_EQ(step, 0); + ASSERT_EQ(n, 1); + PASS(); +} + +/* S17: memory that runs out while doc-link data is extracted or published + * loses references, a scope or an edge -- the layer then says error, never + * ok over a silently thinner graph. One allocation fails at each point in + * turn (a doc span, a doc text, a reference value, a token, the scope scan, + * the project scan, the doc-line map, a MENTIONS edge); the run without a + * failure has no error row. */ +enum { DM_FAIL_EDGE = CBM_DOCLINK_ALLOC_KINDS }; + +static int dm_alloc_failure_errors(int point) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_s17_XXXXXX"); + if (!cbm_mkdtemp(tmp)) { + return -1; + } + th_write_file(TH_PATH(tmp, "src/App.csproj"), + "\n"); + th_write_file(TH_PATH(tmp, "src/A.cs"), "namespace N\n{\n public class Target { }\n}\n"); + th_write_file(TH_PATH(tmp, "src/B.cs"), "namespace N\n" + "{\n" + " /// \n" + " public class Uses\n" + " {\n" + " /// \n" + " public int a, b;\n" + " }\n" + "}\n"); + if (point == DM_FAIL_EDGE) { + cbm_doclinks_test_fail_edge_insert_after(1); + } else if (point >= 0) { + cbm_doclink_test_fail_alloc_after(point, 1); + } + char db[512]; + snprintf(db, sizeof(db), "%s/s17.db", tmp); + int errors = + dm_index(tmp, db, NULL) == 0 + ? dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'") + : -1; + cbm_doclinks_test_fail_edge_insert_after(0); + cbm_doclink_test_reset_alloc(); + dm_unlink_db(db); + th_rmtree(tmp); + return errors; +} + +/* True when s[0, n) is well-formed UTF-8 (no surrogate, nothing past + * U+10FFFF, no overlong form, nothing cut short). */ +static bool dm_utf8_ok(const char *s, size_t n) { + const unsigned char *u = (const unsigned char *)s; + for (size_t i = 0; i < n;) { + unsigned char c = u[i]; + size_t len = c < 0x80 ? 1 + : (c >= 0xC2 && c <= 0xDF) ? 2 + : (c >= 0xE0 && c <= 0xEF) ? 3 + : (c >= 0xF0 && c <= 0xF4) ? 4 + : 0; + if (len == 0 || i + len > n) { + return false; + } + if (len > 1) { + unsigned char c1 = u[i + 1]; + if ((c == 0xE0 && c1 < 0xA0) || (c == 0xED && c1 > 0x9F) || (c == 0xF0 && c1 < 0x90) || + (c == 0xF4 && c1 > 0x8F)) { + return false; + } + for (size_t k = 1; k < len; k++) { + if ((u[i + k] & 0xC0) != 0x80) { + return false; + } + } + } + i += len; + } + return true; +} + +static bool dm_scope_utf8_ok(const char *scope) { + return dm_utf8_ok(scope, strlen(scope)); +} + +/* S8, S9, S10: every scope blob the scanners write passes the reader's own + * record checks, is a C string of the length that was built, and is + * well-formed UTF-8 -- for namespace names with empty segments, and for + * control bytes and malformed UTF-8 in every place a C# scope keeps text. + * A project file's blob is well-formed UTF-8 for malformed bytes and for + * numeric references to surrogates. */ +TEST(doc_mentions_cs_scanner_output_reads_back) { + static const struct { + const char *label, *bytes; + size_t n; + } bytes[] = { + {"nul", "\x00", 1}, + {"soh", "\x01", 1}, + {"us", "\x1F", 1}, + {"del", "\x7F", 1}, + {"continuation", "\x80", 1}, + {"overlong", "\xC0\xAF", 2}, + {"surrogate", "\xED\xA0\x80", 3}, + {"above", "\xF4\x90\x80\x80", 4}, + {"truncated", "\xE2\x82", 2}, + {"valid", "\xC3\xA9", 2}, + }; + static const struct { + const char *label, *before, *after; + } places[] = { + {"using_target", "using Acme.", "Name;\nclass Local {}\n"}, + {"alias_target", "using Alias = Acme.", "Name;\nclass Local {}\n"}, + {"namespace_name", "namespace N", "Name { class Local {} }\n"}, + {"type_name", "class N", "Name {}\n"}, + {"method_name", "class Local { void N", "Name() {} }\n"}, + {"parameter_type", "class Local { void M(N", "Name arg) {} }\n"}, + {"type_parameter", "class Local {}\n"}, + {"method_type_parameter", "class Local { void M() {} }\n"}, + {"field_name", "class Local { int N", "Name; }\n"}, + {"base_type", "class Local : Acme.N", "Name {}\n"}, + {"broken_file", "class Local { void M( { N", "Name } }\n"}, + }; + bool ok = true; + for (size_t p = 0; p < sizeof(places) / sizeof(places[0]); p++) { + for (size_t b = 0; b < sizeof(bytes) / sizeof(bytes[0]); b++) { + char source[512]; + size_t a = strlen(places[p].before); + size_t c = strlen(places[p].after); + memcpy(source, places[p].before, a); + memcpy(source + a, bytes[b].bytes, bytes[b].n); + memcpy(source + a + bytes[b].n, places[p].after, c); + size_t len = a + bytes[b].n + c; + source[len] = '\0'; + cbm_doclink_cs_test_cost_reset(); + CBMFileResult *r = + cbm_extract_file(source, (int)len, CBM_LANG_CSHARP, "p", "Bad.cs", 0, NULL, NULL); + const char *scope = r ? r->doc_scope : NULL; + size_t built = (size_t)cbm_doclink_cs_test_scope_bytes(); + bool good = scope && strlen(scope) == built && dm_utf8_ok(scope, strlen(scope)) && + cbm_doclink_cs_test_scope_parses(scope); + if (!good) { + printf(" %s/%s: scope=%d length=%zu built=%zu utf8=%d reads=%d\n", places[p].label, + bytes[b].label, scope != NULL, scope ? strlen(scope) : (size_t)0, built, + scope ? dm_utf8_ok(scope, strlen(scope)) : 0, + scope ? cbm_doclink_cs_test_scope_parses(scope) : 0); + ok = false; + } + cbm_free_result(r); + } + } + for (size_t i = 0; i < sizeof(dm_namespace_cases) / sizeof(dm_namespace_cases[0]); i++) { + for (int file_scoped = 0; file_scoped < 2; file_scoped++) { + char source[2048]; + if (!dm_namespace_source(source, sizeof(source), i, file_scoped != 0)) { + ok = false; + continue; + } + CBMFileResult *r = dm_extract(source, CBM_LANG_CSHARP, "Bad.cs"); + const char *scope = r ? r->doc_scope : NULL; + if (!scope || !cbm_doclink_cs_test_scope_parses(scope)) { + printf(" namespace %s/%s: the reader refuses the scope\n", + dm_namespace_cases[i].id, file_scoped ? "file" : "block"); + ok = false; + } + cbm_free_result(r); + } + } + static const char *xml[] = { + "

N\x80Name

", + "Value", + "

N�Name�

", + "", + "

V

" + "
", + "V" + "", + }; + for (size_t i = 0; i < sizeof(xml) / sizeof(xml[0]); i++) { + CBMFileResult *r = cbm_extract_file(xml[i], (int)strlen(xml[i]), CBM_LANG_XML, "p", + "App.csproj", 0, NULL, NULL); + const char *scope = r ? r->doc_scope : NULL; + if (!scope || !dm_utf8_ok(scope, strlen(scope))) { + printf(" project file %zu: blob=%d utf8=%d\n", i, scope != NULL, + scope ? dm_utf8_ok(scope, strlen(scope)) : 0); + ok = false; + } + cbm_free_result(r); + } + ASSERT_TRUE(ok); + PASS(); +} + +/* S8: a scope written in this run that the reader refuses costs only its own + * file. The status stays ok and the other file's edges are there; the file's + * own reference is a graph gap; and nothing its parse set before the refusal + * stays -- a namespace it declares does not stand beside a type of that name + * (Shadow, Good.Lurker). + * R4: the type names its T and Q records carry are quarantined, as the names + * of declarations no scope is known for are: a reference to one, from any + * file, is a graph gap -- never a binding to another type of that name the + * full picture would not choose (Widget: Good.Widget of Bad.cs comes before + * the imported Lib.Widget), nor to one the unplaced declaration would make + * uncertain (Hidden2, as when the scope is taken). */ +/* S8: Bad.cs declares what would shadow and hide Healthy.cs's names; its + * scope is spoiled where it is written (cbm_doclink_cs_test_spoil_scope). */ +static const dm_source_t dm_rejected_files[] = { + {"Bad.cs", "namespace Shadow\n" + "{\n" + " public class Inner { }\n" + "}\n" + "namespace Good\n" + "{\n" + " namespace Lurker { public class Deep { } }\n" + "\n" + " /// \n" + " public class FromBad { }\n" + " public class Widget { }\n" + "}\n" + "namespace .Broken\n" + "{\n" + " public class Hidden2 { }\n" + "}\n"}, + {"Healthy.cs", "using Lib;\n" + "public class Shadow { }\n" + "namespace Good\n" + "{\n" + " public class Target { }\n" + " public class Lurker { }\n" + " public class Hidden2 { }\n" + "\n" + " /// \n" + " /// \n" + " /// \n" + " public class Uses { }\n" + "}\n"}, + {"Lib.cs", "namespace Lib\n" + "{\n" + " public class Widget { }\n" + "}\n"}, +}; + +static const dm_want_t dm_rejected_wants[] = { + {"Healthy.cs", "Target", "Healthy.Uses", "Healthy.Target", NULL, NULL, NULL}, + {"Healthy.cs", "Shadow", "Healthy.Uses", "Healthy.Shadow", NULL, NULL, NULL}, + {"Healthy.cs", "Lurker", "Healthy.Uses", "Healthy.Lurker", NULL, NULL, NULL}, + {"Healthy.cs", "Hidden2", "Healthy.Uses", NULL, "graph_gap", "Healthy.Hidden2", NULL}, + {"Healthy.cs", "Widget", "Healthy.Uses", NULL, "graph_gap", "Lib.Widget", NULL}, + {"Bad.cs", "Target", "Bad.FromBad", NULL, "graph_gap", "Healthy.Target", NULL}, +}; + +TEST(doc_mentions_cs_rejected_scope_contained) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_rejected_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + for (int i = 0; i < DM_COUNT(dm_rejected_files); i++) { + th_write_file(TH_PATH(tmp, dm_rejected_files[i].path), dm_rejected_files[i].text); + } + char db[512]; + snprintf(db, sizeof(db), "%s/rejected.db", tmp); + cbm_doclink_cs_test_spoil_scope("Bad.cs"); + int bad = dm_index(tmp, db, NULL) == 0 ? 0 : -1; + cbm_doclink_cs_test_spoil_scope(NULL); + int errors = bad == 0 ? dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved " + "WHERE reason = 'error'") + : -1; + for (int i = 0; bad >= 0 && i < DM_COUNT(dm_rejected_wants); i++) { + bad += dm_want_failed(db, &dm_rejected_wants[i]); + } + dm_unlink_db(db); + th_rmtree(tmp); + ASSERT_EQ(errors, 0); + ASSERT_EQ(bad, 0); + PASS(); +} + +/* The scope blob stored for `rel_path` (the surface row's `dl`), into out; + * "" when there is none. */ +static void dm_stored_scope(const char *db, const char *rel_path, char *out, size_t cap) { + out[0] = '\0'; + sqlite3 *h = NULL; + if (sqlite3_open_v2(db, &h, SQLITE_OPEN_READONLY, NULL) == SQLITE_OK) { + sqlite3_stmt *st = NULL; + if (sqlite3_prepare_v2(h, + "SELECT json_extract(defs_json, '$.dl') FROM lsp_surface " + "WHERE rel_path = ?1", + -1, &st, NULL) == SQLITE_OK) { + sqlite3_bind_text(st, 1, rel_path, -1, SQLITE_TRANSIENT); + if (sqlite3_step(st) == SQLITE_ROW && sqlite3_column_text(st, 0)) { + snprintf(out, cap, "%s", (const char *)sqlite3_column_text(st, 0)); + } + } + sqlite3_finalize(st); + } + sqlite3_close(h); +} + +/* S8 across runs: the scope its own run refused is stored as the rejected + * marker, so a later incremental run that reads it back treats the file as + * the full run does -- it declares nothing, its references are graph gaps -- + * instead of failing the layer over a stored scope the reader refuses. A + * body-only edit elsewhere reads the marker back; one in the rejected file + * compares its marker with the fresh one. Each step equals a full run. */ +TEST(doc_mentions_cs_rejected_scope_across_runs) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_rejected_runs_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char repo[400]; + char db[512]; + char full_db[512]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + snprintf(db, sizeof(db), "%s/inc.db", tmp); + snprintf(full_db, sizeof(full_db), "%s/full.db", tmp); + for (int i = 0; i < DM_COUNT(dm_rejected_files); i++) { + th_write_file(TH_PATH(repo, dm_rejected_files[i].path), dm_rejected_files[i].text); + } + cbm_doclink_cs_test_spoil_scope("Bad.cs"); + int indexed = dm_index(repo, db, NULL); + char stored[256]; + dm_stored_scope(db, "Bad.cs", stored, sizeof(stored)); + /* the marker, then one Q record per type name of the refused blob (R4) */ + static const char *const names[] = {"Inner", "Deep", "FromBad", "Widget", "Hidden2"}; + static const char marker[] = CBM_DOCLINK_CS_SCOPE_TAG "\n!\trejected\n"; + bool marked = strncmp(stored, marker, sizeof(marker) - 1) == 0; + size_t want_len = sizeof(marker) - 1; + for (size_t k = 0; k < sizeof(names) / sizeof(names[0]); k++) { + char q[64]; + snprintf(q, sizeof(q), "\nQ\t%s\n", names[k]); + marked = marked && strstr(stored, q) != NULL; + want_len += strlen(q) - 1; + } + if (!marked || strlen(stored) != want_len) { + printf(" stored for Bad.cs: %s\n", stored); + marked = false; + } + char edited[2048]; + snprintf(edited, sizeof(edited), "%s\n// body-only edit\n", dm_rejected_files[1].text); + th_write_file(TH_PATH(repo, "Healthy.cs"), edited); + int read_back = dm_step(repo, db, full_db, "the rejected scope read back", + CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR); + snprintf(edited, sizeof(edited), "%s\n// body-only edit\n", dm_rejected_files[0].text); + th_write_file(TH_PATH(repo, "Bad.cs"), edited); + int again = dm_step(repo, db, full_db, "the rejected file edited", + CBM_INCREMENTAL_ROUTE_CLOSURE_REPAIR); + cbm_doclink_cs_test_spoil_scope(NULL); + int errors = dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'"); + int bad = 0; + for (int i = 0; i < DM_COUNT(dm_rejected_wants); i++) { + bad += dm_want_failed(db, &dm_rejected_wants[i]); + } + dm_unlink_db(db); + dm_unlink_db(full_db); + th_rmtree(tmp); + ASSERT_EQ(indexed, 0); + ASSERT_TRUE(marked); + ASSERT_EQ(read_back, 0); + ASSERT_EQ(again, 0); + ASSERT_EQ(errors, 0); + ASSERT_EQ(bad, 0); + PASS(); +} + +enum { DM_DIRECTIVE_FILLERS = 60 }; /* with the C# file: above MIN_FILES_FOR_PARALLEL */ + +static const char DM_DIRECTIVES_ONLY[] = "global using System;\nglobal using System.IO;\n"; + +/* Error rows of one index of a repository whose only C# file holds nothing + * but global usings, its scope scan failing when `fail`. `workers` "4" runs + * the worker pipeline, "1" the sequential passes (CBM_WORKERS is restored). + * -1 when the repository cannot be indexed. */ +static int dm_directives_only_errors(const char *workers, bool fail) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_dm_r7_XXXXXX"); + if (!cbm_mkdtemp(tmp)) { + return -1; + } + char repo[400]; + snprintf(repo, sizeof(repo), "%s/repo", tmp); + th_write_file(TH_PATH(repo, "src/App.csproj"), DM_EMPTY_PROJECT); + th_write_file(TH_PATH(repo, "src/Usings.cs"), DM_DIRECTIVES_ONLY); + for (int i = 0; i < DM_DIRECTIVE_FILLERS; i++) { + char path[512]; + snprintf(path, sizeof(path), "%s/py/m%02d.py", repo, i); + th_write_file(path, "def f():\n return 1\n"); + } + const char *saved = getenv("CBM_WORKERS"); + char *saved_copy = saved ? strdup(saved) : NULL; + cbm_setenv("CBM_WORKERS", workers, 1); + if (fail) { + cbm_doclink_test_fail_alloc_after(CBM_DOCLINK_ALLOC_SCOPE, 1); /* the one C# file */ + } + char db[512]; + snprintf(db, sizeof(db), "%s/r7.db", tmp); + int errors = + dm_index(repo, db, NULL) == 0 + ? dm_count(db, "SELECT COUNT(*) FROM doc_link_unresolved WHERE reason = 'error'") + : -1; + cbm_doclink_test_reset_alloc(); + if (saved_copy) { + cbm_setenv("CBM_WORKERS", saved_copy, 1); + free(saved_copy); + } else { + cbm_unsetenv("CBM_WORKERS"); + } + dm_unlink_db(db); + th_rmtree(tmp); + return errors; +} + +/* R7: a C# file of nothing but global usings has no call, usage, throw, + * read/write or impl trait; its one definition is the Module every + * extraction pushes first. That keeps it out of the worker pipeline's skip + * of files with nothing to resolve, so its doc-link failure flag is read + * there: its scope scan running out of memory fails the layer in both + * routes. The fixture holds only while the file has nothing but the Module. */ +TEST(doc_mentions_directives_only_failure) { + CBMFileResult *r = dm_extract(DM_DIRECTIVES_ONLY, CBM_LANG_CSHARP, "src/Usings.cs"); + ASSERT_NOT_NULL(r); + bool module_only = r->calls.count == 0 && r->usages.count == 0 && r->throws.count == 0 && + r->rw.count == 0 && r->impl_traits.count == 0 && r->defs.count == 1 && + strcmp(r->defs.items[0].label, "Module") == 0; + cbm_free_result(r); + ASSERT_TRUE(module_only); + static const char *const routes[] = {"4", "1"}; + bool ok = true; + for (size_t k = 0; k < sizeof(routes) / sizeof(routes[0]); k++) { + int clean = dm_directives_only_errors(routes[k], false); + int failed = dm_directives_only_errors(routes[k], true); + if (clean != 0 || failed != 1) { + printf(" %s worker(s): %d error rows without a failure (want 0), %d with one (want " + "1)\n", + routes[k], clean, failed); + ok = false; + } + } + ASSERT_TRUE(ok); + PASS(); +} + +TEST(doc_mentions_alloc_failure_status) { + static const struct { + int point; + const char *what; + } points[] = { + {CBM_DOCLINK_ALLOC_SPAN, "doc span"}, {CBM_DOCLINK_ALLOC_TEXT, "doc text"}, + {CBM_DOCLINK_ALLOC_VALUE, "value"}, {CBM_DOCLINK_ALLOC_TOKENS, "token"}, + {CBM_DOCLINK_ALLOC_SCOPE, "scope scan"}, {CBM_DOCLINK_ALLOC_PROJECT, "project scan"}, + {CBM_DOCLINK_ALLOC_DOC_LINE, "doc-line map"}, {DM_FAIL_EDGE, "MENTIONS edge"}, + }; + int baseline = dm_alloc_failure_errors(-1); + bool ok = baseline == 0; + for (size_t i = 0; i < sizeof(points) / sizeof(points[0]); i++) { + int errors = dm_alloc_failure_errors(points[i].point); + if (errors != 1) { + printf(" a failed %s allocation: %d error rows, want 1\n", points[i].what, errors); + ok = false; + } + } + ASSERT_TRUE(ok); + PASS(); +} + +SUITE(doc_mentions) { + RUN_TEST(doc_mentions_extract_cs_tokens); + RUN_TEST(doc_mentions_cs_scope_blob); + RUN_TEST(doc_mentions_cs_declarator_modifiers); + RUN_TEST(doc_mentions_cs_namespace_scopes); + RUN_TEST(doc_mentions_cs_namespace_boundaries); + RUN_TEST(doc_mentions_cs_control_scopes); + RUN_TEST(doc_mentions_cs_control_publication); + RUN_TEST(doc_mentions_cs_utf8_surfaces); + RUN_TEST(doc_mentions_cs_namespace_publication); + RUN_TEST(doc_mentions_cs_scope_parse_errors); + RUN_TEST(doc_mentions_cs_norm_type); + RUN_TEST(doc_mentions_resolver_rules); + RUN_TEST(doc_mentions_resolver_arity_and_members); + RUN_TEST(doc_mentions_resolver_parse_errors); + RUN_TEST(doc_mentions_ship_gate); + RUN_TEST(doc_mentions_file_node_lookup); + RUN_TEST(doc_mentions_index_status_and_delete); + RUN_TEST(doc_mentions_index_status_preview_bounds); + RUN_TEST(doc_mentions_index_status_preview_text); + RUN_TEST(doc_mentions_index_status_preview_order); + RUN_TEST(doc_mentions_index_status_preview_failure); + RUN_TEST(doc_mentions_scope_delta_rules); + RUN_TEST(doc_mentions_incremental_equals_full); + RUN_TEST(doc_mentions_parallel_equals_sequential); + RUN_TEST(doc_mentions_cs_scan_dollar_run); + RUN_TEST(doc_mentions_cs_shared_doc_sources); + RUN_TEST(doc_mentions_cs_shared_doc_once); + RUN_TEST(doc_mentions_cs_shared_doc_replay); + RUN_TEST(doc_mentions_cs_shared_doc_allocation_failure); + RUN_TEST(doc_mentions_cs_scan_header_reads); + RUN_TEST(doc_mentions_cs_scan_nested_holes); + RUN_TEST(doc_mentions_cs_scan_holes_stack); + RUN_TEST(doc_mentions_incremental_project_files); + RUN_TEST(doc_mentions_cs_lookup_order); + RUN_TEST(doc_mentions_cs_type_parameters); + RUN_TEST(doc_mentions_cs_usings); + RUN_TEST(doc_mentions_cs_entities); + RUN_TEST(doc_mentions_cs_assemblies_by_name); + RUN_TEST(doc_mentions_cs_reference_source); + RUN_TEST(doc_mentions_cs_assemblies_differ); + RUN_TEST(doc_mentions_cs_shared_parts); + RUN_TEST(doc_mentions_cs_shared_trees); + RUN_TEST(doc_mentions_cs_no_project_files); + RUN_TEST(doc_mentions_cs_contract_join); + RUN_TEST(doc_mentions_cs_idle_project_file); + RUN_TEST(doc_mentions_cs_test_namespaces); + RUN_TEST(doc_mentions_cs_members); + RUN_TEST(doc_mentions_cs_constructors); + RUN_TEST(doc_mentions_cs_inherited_members); + RUN_TEST(doc_mentions_cs_doc_ids); + RUN_TEST(doc_mentions_cs_reasons); + RUN_TEST(doc_mentions_cs_text_forms); + RUN_TEST(doc_mentions_cs_unit_usings_unchanged); + RUN_TEST(doc_mentions_cs_lookup_ambiguous_stops); + RUN_TEST(doc_mentions_cs_import_stage_aliases); + RUN_TEST(doc_mentions_cs_import_static_parts); + RUN_TEST(doc_mentions_cs_import_candidate_allocation); + RUN_TEST(doc_mentions_cs_no_silent_limits); + RUN_TEST(doc_mentions_cs_many_assemblies); + RUN_TEST(doc_mentions_incremental_using_of_based_type); + RUN_TEST(doc_mentions_cs_scan_nesting_limits); + RUN_TEST(doc_mentions_cs_damaged_stored_scope); + RUN_TEST(doc_mentions_cs_portable_scope_bound); + RUN_TEST(doc_mentions_cs_unbalanced_brackets); + RUN_TEST(doc_mentions_cs_stored_scope_nesting); + RUN_TEST(doc_mentions_index_status_unknown_reason); + RUN_TEST(doc_mentions_incremental_non_ascii_name); + RUN_TEST(doc_mentions_alloc_failure_status); + RUN_TEST(doc_mentions_directives_only_failure); + RUN_TEST(doc_mentions_cs_scanner_output_reads_back); + RUN_TEST(doc_mentions_cs_rejected_scope_contained); + RUN_TEST(doc_mentions_cs_rejected_scope_across_runs); +} + +/* The cost tests (work counters held to a multiple of the input) and the MSBuild + * evaluator run as suites of their own, so that each stays within the per-suite + * wall clock of the harness on the slowest sanitizer legs. */ +SUITE(doc_mentions_cost) { + RUN_TEST(doc_mentions_cs_scan_branch_memory); + RUN_TEST(doc_mentions_cs_scan_branch_merge_work); + RUN_TEST(doc_mentions_cs_scan_declarator_modifier_work); + RUN_TEST(doc_mentions_cs_shared_doc_work); + RUN_TEST(doc_mentions_cs_lookup_cost); + RUN_TEST(doc_mentions_cs_import_lookup_cost); + RUN_TEST(doc_mentions_cs_lookup_repeated_name_work); + RUN_TEST(doc_mentions_cs_unit_usings_across_files_work); + RUN_TEST(doc_mentions_cs_scratch_table_work); +} + +SUITE(doc_mentions_msbuild) { + RUN_TEST(doc_mentions_msbuild_usings); + RUN_TEST(doc_mentions_msbuild_targets_isolation); + RUN_TEST(doc_mentions_msbuild_targets_failure); + RUN_TEST(doc_mentions_msbuild_items_isolation); + RUN_TEST(doc_mentions_msbuild_items_failure); + RUN_TEST(doc_mentions_msbuild_items_lifetime); + RUN_TEST(doc_mentions_msbuild_items_work); + RUN_TEST(doc_mentions_msbuild_targets_work); + RUN_TEST(doc_mentions_msbuild_nearest_work); + RUN_TEST(doc_mentions_msbuild_shared_closure_work); + RUN_TEST(doc_mentions_msbuild_shared_file_work); + RUN_TEST(doc_mentions_msbuild_shared_closure_isolation); + RUN_TEST(doc_mentions_msbuild_components_failure); + RUN_TEST(doc_mentions_msbuild_components_revision); + RUN_TEST(doc_mentions_msbuild_components_mixed_absence); + RUN_TEST(doc_mentions_msbuild_prefix_work); + RUN_TEST(doc_mentions_msbuild_prefix_isolation); + RUN_TEST(doc_mentions_msbuild_prefix_failure); + RUN_TEST(doc_mentions_msbuild_eval_storage); + RUN_TEST(doc_mentions_msbuild_value_lifetimes); + RUN_TEST(doc_mentions_msbuild_value_allocation); + RUN_TEST(doc_mentions_msbuild_blob); + RUN_TEST(doc_mentions_msbuild_import_group_blob_growth); + RUN_TEST(doc_mentions_msbuild_import_group_roundtrip); + RUN_TEST(doc_mentions_msbuild_import_group_condition_work); + RUN_TEST(doc_mentions_msbuild_item_group_condition_work); + RUN_TEST(doc_mentions_msbuild_import_group_condition_once); + RUN_TEST(doc_mentions_msbuild_legacy_import_blob); + RUN_TEST(doc_mentions_msbuild_bad_import_group_blob); + RUN_TEST(doc_mentions_msbuild_imports); + RUN_TEST(doc_mentions_msbuild_conditions); + RUN_TEST(doc_mentions_msbuild_unknown_spreads); + RUN_TEST(doc_mentions_msbuild_gate); +#ifndef _WIN32 + RUN_TEST(doc_mentions_msbuild_linked_project_file); + RUN_TEST(doc_mentions_msbuild_linked_import_directory); +#endif + RUN_TEST(doc_mentions_msbuild_many_project_files); + RUN_TEST(doc_mentions_msbuild_unread_project_opens); + RUN_TEST(doc_mentions_msbuild_malformed_props_opens); +} diff --git a/tests/test_doc_mentions_helpers.h b/tests/test_doc_mentions_helpers.h new file mode 100644 index 000000000..66fbcd52c --- /dev/null +++ b/tests/test_doc_mentions_helpers.h @@ -0,0 +1,414 @@ +/* + * test_doc_mentions_helpers.h — helpers shared by the doc-mentions suites + * (tests/test_doc_mentions.c and one test_doc_mentions_.c per language + * leg). + * + * Every pipeline helper indexes a real fixture through cbm_pipeline_run and + * reads the published database, so a test sees what a user's index holds. + * All functions are static inline, like test_helpers.h: including the header + * from several test files causes no linker issue, and a file that uses only + * some of them gets no unused-function warning. + */ +#ifndef TEST_DOC_MENTIONS_HELPERS_H +#define TEST_DOC_MENTIONS_HELPERS_H + +#include "../src/foundation/compat.h" +#include "test_helpers.h" + +#include "cbm.h" +#include "doclink.h" +#include "foundation/mem_core.h" +#include "mcp/mcp.h" +#include "pipeline/doc_links.h" +#include "pipeline/pipeline.h" +#include "pipeline/pipeline_internal.h" +#include "sqlite3.h" +#include + +#include +#include +#include +#include + +/* ── extraction ──────────────────────────────────────────────────── */ + +/* Extract one source text as language `lang` under project "p"; the result + * (tokens in doc_links, scope blob in doc_scope) is freed with + * cbm_free_result. */ +static inline CBMFileResult *dm_extract(const char *src, CBMLanguage lang, const char *rel_path) { + return cbm_extract_file(src, (int)strlen(src), lang, "p", rel_path, 0, NULL, NULL); +} + +/* The first token written as `raw`; NULL when there is none. */ +static inline const CBMDocLink *dm_find_token(const CBMFileResult *r, const char *raw) { + for (int i = 0; i < r->doc_links.count; i++) { + if (strcmp(r->doc_links.items[i].raw, raw) == 0) { + return &r->doc_links.items[i]; + } + } + return NULL; +} + +static inline int dm_count_tokens(const CBMFileResult *r, const char *raw) { + int n = 0; + for (int i = 0; i < r->doc_links.count; i++) { + n += strcmp(r->doc_links.items[i].raw, raw) == 0; + } + return n; +} + +/* ── the published database ──────────────────────────────────────── */ + +/* Index `repo` into `db` (full mode; a second call on the same database takes + * the incremental route). The project name is strdup'ed into *project_out + * when that is not NULL. Returns the pipeline's result. */ +static inline int dm_index(const char *repo, const char *db, char **project_out) { + cbm_pipeline_t *p = cbm_pipeline_new(repo, db, CBM_MODE_FULL); + if (!p) { + return -1; + } + int rc = cbm_pipeline_run(p); + if (project_out) { + *project_out = strdup(cbm_pipeline_project_name(p)); + } + cbm_pipeline_free(p); + return rc; +} + +/* Properties of the MENTIONS edge whose endpoints' qualified names END with + * the given local paths ("Widget", "Helper.Once"); "" when absent; count via + * *n. */ +static inline void dm_edge(const char *db, const char *src_suffix, const char *tgt_suffix, + char *props, size_t cap, int *n) { + props[0] = '\0'; + *n = 0; + sqlite3 *h = NULL; + if (sqlite3_open_v2(db, &h, SQLITE_OPEN_READONLY, NULL) != SQLITE_OK) { + sqlite3_close(h); + *n = -1; + return; + } + sqlite3_stmt *st = NULL; + const char *sql = + "SELECT e.properties FROM edges e JOIN nodes s ON s.id = e.source_id " + "JOIN nodes t ON t.id = e.target_id WHERE e.type = 'MENTIONS' " + "AND (s.qualified_name LIKE '%.' || ?1) AND (t.qualified_name LIKE '%.' || ?2)"; + if (sqlite3_prepare_v2(h, sql, -1, &st, NULL) == SQLITE_OK) { + sqlite3_bind_text(st, 1, src_suffix, -1, SQLITE_TRANSIENT); + sqlite3_bind_text(st, 2, tgt_suffix, -1, SQLITE_TRANSIENT); + while (sqlite3_step(st) == SQLITE_ROW) { + (*n)++; + snprintf(props, cap, "%s", (const char *)sqlite3_column_text(st, 0)); + } + } + sqlite3_finalize(st); + sqlite3_close(h); +} + +/* Number of MENTIONS edges leaving the definition whose qualified name ends + * with `src_suffix`; -1 when the database cannot be read. */ +static inline int dm_mentions_from(const char *db, const char *src_suffix) { + sqlite3 *h = NULL; + int n = -1; + if (sqlite3_open_v2(db, &h, SQLITE_OPEN_READONLY, NULL) == SQLITE_OK) { + sqlite3_stmt *st = NULL; + if (sqlite3_prepare_v2(h, + "SELECT COUNT(*) FROM edges e JOIN nodes s ON s.id = e.source_id " + "WHERE e.type = 'MENTIONS' AND s.qualified_name LIKE '%.' || ?1", + -1, &st, NULL) == SQLITE_OK) { + sqlite3_bind_text(st, 1, src_suffix, -1, SQLITE_TRANSIENT); + if (sqlite3_step(st) == SQLITE_ROW) { + n = sqlite3_column_int(st, 0); + } + } + sqlite3_finalize(st); + } + sqlite3_close(h); + return n; +} + +/* The single integer a query returns; -1 when it cannot be read. */ +static inline int dm_count(const char *db, const char *sql) { + sqlite3 *h = NULL; + int n = -1; + if (sqlite3_open_v2(db, &h, SQLITE_OPEN_READONLY, NULL) == SQLITE_OK) { + sqlite3_stmt *st = NULL; + if (sqlite3_prepare_v2(h, sql, -1, &st, NULL) == SQLITE_OK && + sqlite3_step(st) == SQLITE_ROW) { + n = sqlite3_column_int(st, 0); + } + sqlite3_finalize(st); + } + sqlite3_close(h); + return n; +} + +/* Reason (and, when `syntax` is not NULL, the family name) of the unresolved + * row with this raw text in this file; "" when there is none. */ +static inline void dm_row(const char *db, const char *rel, const char *raw, char *reason, + size_t cap, char *syntax, size_t scap) { + reason[0] = '\0'; + if (syntax) { + syntax[0] = '\0'; + } + sqlite3 *h = NULL; + if (sqlite3_open_v2(db, &h, SQLITE_OPEN_READONLY, NULL) == SQLITE_OK) { + sqlite3_stmt *st = NULL; + if (sqlite3_prepare_v2(h, + "SELECT reason, syntax FROM doc_link_unresolved WHERE rel_path = ?1 " + "AND raw = ?2", + -1, &st, NULL) == SQLITE_OK) { + sqlite3_bind_text(st, 1, rel, -1, SQLITE_TRANSIENT); + sqlite3_bind_text(st, 2, raw, -1, SQLITE_TRANSIENT); + if (sqlite3_step(st) == SQLITE_ROW) { + snprintf(reason, cap, "%s", (const char *)sqlite3_column_text(st, 0)); + if (syntax) { + snprintf(syntax, scap, "%s", (const char *)sqlite3_column_text(st, 1)); + } + } + } + sqlite3_finalize(st); + } + sqlite3_close(h); +} + +/* Canonical text of every MENTIONS edge and unresolved row: what two indexes + * of the same tree (full and incremental, one worker and several) must agree + * on byte for byte. The caller frees it; NULL when the database cannot be + * read. */ +static inline char *dm_doclink_state(const char *db) { + sqlite3 *h = NULL; + if (sqlite3_open_v2(db, &h, SQLITE_OPEN_READONLY, NULL) != SQLITE_OK) { + sqlite3_close(h); + return NULL; + } + size_t cap = 4096; + size_t len = 0; + char *buf = malloc(cap); + buf[0] = '\0'; + const char *queries[] = { + "SELECT 'E ' || s.qualified_name || ' -> ' || t.qualified_name || ' ' || e.properties " + "FROM edges e JOIN nodes s ON s.id = e.source_id JOIN nodes t ON t.id = e.target_id " + "WHERE e.type = 'MENTIONS' ORDER BY 1", + "SELECT 'R ' || rel_path || ':' || line || ' ' || syntax || ' [' || raw || '] ' || reason " + "FROM doc_link_unresolved ORDER BY 1", + }; + for (size_t q = 0; q < sizeof(queries) / sizeof(queries[0]); q++) { + sqlite3_stmt *st = NULL; + if (sqlite3_prepare_v2(h, queries[q], -1, &st, NULL) != SQLITE_OK) { + continue; + } + while (sqlite3_step(st) == SQLITE_ROW) { + const char *line = (const char *)sqlite3_column_text(st, 0); + size_t l = strlen(line); + if (len + l + 2 > cap) { + cap = (len + l + 2) * 2; + buf = realloc(buf, cap); + } + memcpy(buf + len, line, l); + len += l; + buf[len++] = '\n'; + buf[len] = '\0'; + } + sqlite3_finalize(st); + } + sqlite3_close(h); + return buf; +} + +/* Remove a database and its sidecars. */ +static inline void dm_unlink_db(const char *db) { + char side[600]; + unlink(db); + snprintf(side, sizeof(side), "%s-wal", db); + unlink(side); + snprintf(side, sizeof(side), "%s-shm", db); + unlink(side); +} + +/* ── incremental == full ─────────────────────────────────────────── */ + +/* One step of an incremental test, after the caller edited the tree: index + * `repo` incrementally into `inc_db`, then fully into a fresh `full_db`, and + * require identical MENTIONS edges and unresolved rows AND the expected + * route. Returns 0, or -1 after printing what differs (`what` names the + * step). */ +static inline int dm_step(const char *repo, const char *inc_db, const char *full_db, + const char *what, cbm_incremental_route_t want_route) { + cbm_pipeline_incremental_test_reset_faults(); + if (dm_index(repo, inc_db, NULL) != 0) { + printf(" %s: incremental index failed\n", what); + return -1; + } + cbm_incremental_route_t route = cbm_pipeline_incremental_test_last_route(); + dm_unlink_db(full_db); + if (dm_index(repo, full_db, NULL) != 0) { + printf(" %s: full index failed\n", what); + return -1; + } + char *inc = dm_doclink_state(inc_db); + char *full = dm_doclink_state(full_db); + int rc = 0; + if (!inc || !full || strcmp(inc, full) != 0) { + printf(" %s: incremental != full\n--- incremental (route %d)\n%s--- full\n%s", what, + (int)route, inc ? inc : "(null)", full ? full : "(null)"); + rc = -1; + } else if (route != want_route) { + printf(" %s: route %d, expected %d\n", what, (int)route, (int)want_route); + rc = -1; + } + free(inc); + free(full); + return rc; +} + +/* ── one worker == several workers ───────────────────────────────── */ + +/* Index `repo` with four workers into `par_db` and with one worker into + * `seq_db` (the fixture needs more than 50 files, or both take the sequential + * passes), and require identical MENTIONS edges and unresolved rows. + * CBM_WORKERS is restored. Returns 0, or -1 after printing what differs. */ +static inline int dm_workers_agree(const char *repo, const char *par_db, const char *seq_db) { + const char *saved_workers = getenv("CBM_WORKERS"); + char *saved_workers_copy = saved_workers ? strdup(saved_workers) : NULL; + cbm_setenv("CBM_WORKERS", "4", 1); + int par_rc = dm_index(repo, par_db, NULL); + cbm_setenv("CBM_WORKERS", "1", 1); /* one worker: the sequential passes */ + int seq_rc = dm_index(repo, seq_db, NULL); + if (saved_workers_copy) { + cbm_setenv("CBM_WORKERS", saved_workers_copy, 1); + free(saved_workers_copy); + } else { + cbm_unsetenv("CBM_WORKERS"); + } + if (par_rc != 0 || seq_rc != 0) { + printf(" index failed: parallel rc %d, sequential rc %d\n", par_rc, seq_rc); + return -1; + } + char *par = dm_doclink_state(par_db); + char *seq = dm_doclink_state(seq_db); + bool same = par && seq && strcmp(par, seq) == 0; + if (!same) { + printf(" parallel != sequential\n--- parallel\n%s--- sequential\n%s", par ? par : "(null)", + seq ? seq : "(null)"); + } + free(par); + free(seq); + return same ? 0 : -1; +} + +/* ── scope deltas ────────────────────────────────────────────────── */ + +/* The names a scope delta reported, joined by ','. */ +typedef struct { + char names[256]; +} dm_names_t; + +/* cbm_doclink_name_fn collecting into a dm_names_t. */ +static inline bool dm_name_put(void *ud, const char *name, size_t len) { + dm_names_t *n = (dm_names_t *)ud; + size_t used = strlen(n->names); + if (used + len + 2 > sizeof(n->names)) { + return false; + } + if (used > 0) { + n->names[used++] = ','; + } + memcpy(n->names + used, name, len); + n->names[used + len] = '\0'; + return true; +} + +enum { DM_DELTA_SCAN_FAILED = -2 }; + +/* The scope delta between two versions of one file of language `lang` (NULL: + * the file does not exist), through the language's real scanner and the + * persisted form of its scope; the removed names joined by ',' in `names`. + * Returns the cbm_doclink_delta_t, -1 as the hook does, or + * DM_DELTA_SCAN_FAILED when a version yields no scope. */ +static inline int dm_scope_delta(CBMLanguage lang, const char *rel_path, const char *before, + const char *after, char *names, size_t cap) { + names[0] = '\0'; + CBMFileResult *a = before ? dm_extract(before, lang, rel_path) : NULL; + CBMFileResult *b = after ? dm_extract(after, lang, rel_path) : NULL; + char *pa = (a && a->doc_scope) ? cbm_doclink_portable_scope(a->doc_scope) : NULL; + char *pb = (b && b->doc_scope) ? cbm_doclink_portable_scope(b->doc_scope) : NULL; + dm_names_t n = {{0}}; + int rc = ((before && !pa) || (after && !pb)) + ? DM_DELTA_SCAN_FAILED + : cbm_doclinks_scope_delta(pa, pb, dm_name_put, &n); + snprintf(names, cap, "%s", n.names); + cbm_free(CBM_MEM_CLASS_OTHER, pa); + cbm_free(CBM_MEM_CLASS_OTHER, pb); + if (a) { + cbm_free_result(a); + } + if (b) { + cbm_free_result(b); + } + return rc; +} + +/* ── index_status ────────────────────────────────────────────────── */ + +/* The text content of an MCP tool result (the report itself); the caller + * frees it. */ +static inline char *dm_tool_text(const char *mcp_result) { + yyjson_doc *doc = mcp_result ? yyjson_read(mcp_result, strlen(mcp_result), 0) : NULL; + yyjson_val *root = doc ? yyjson_doc_get_root(doc) : NULL; + yyjson_val *content = root ? yyjson_obj_get(root, "content") : NULL; + yyjson_val *item = content ? yyjson_arr_get(content, 0) : NULL; + const char *text = item ? yyjson_get_str(yyjson_obj_get(item, "text")) : NULL; + char *out = text ? strdup(text) : NULL; + yyjson_doc_free(doc); + return out; +} + +/* index_status of the project as text, from a server of its own (nothing + * cached from an earlier call). The project's database is looked up in + * CBM_CACHE_DIR: point that at a private directory first. */ +static inline char *dm_index_status(const char *project, bool full) { + cbm_mcp_server_t *srv = cbm_mcp_server_new(NULL); + if (!srv) { + return NULL; + } + char args[1200]; + snprintf(args, sizeof(args), "{\"project\":\"%s\"%s}", project, + full ? ",\"diagnostics\":\"full\"" : ""); + char *resp = cbm_mcp_handle_tool(srv, "index_status", args); + char *text = dm_tool_text(resp); + free(resp); + cbm_mcp_server_free(srv); + return text; +} + +/* Every reason the table holds is listed under doc_links.unresolved with its + * row count. The number of reasons checked; -1 when a line is missing. */ +static inline int dm_reason_lines(const char *db, const char *block) { + sqlite3 *h = NULL; + int n = -1; + if (sqlite3_open_v2(db, &h, SQLITE_OPEN_READONLY, NULL) == SQLITE_OK) { + sqlite3_stmt *st = NULL; + if (sqlite3_prepare_v2(h, + "SELECT reason, COUNT(*) FROM doc_link_unresolved GROUP BY reason", + -1, &st, NULL) == SQLITE_OK) { + n = 0; + while (n >= 0 && sqlite3_step(st) == SQLITE_ROW) { + char line[128]; + snprintf(line, sizeof(line), "\n %s: %d\n", + (const char *)sqlite3_column_text(st, 0), sqlite3_column_int(st, 1)); + if (strstr(block, line)) { + n++; + } else { + printf(" no line%s in\n%s\n", line, block); + n = -1; + } + } + } + sqlite3_finalize(st); + } + sqlite3_close(h); + return n; +} + +#endif /* TEST_DOC_MENTIONS_HELPERS_H */ diff --git a/tests/test_edge_structural.c b/tests/test_edge_structural.c index 798fef3d0..4b91ae05f 100644 --- a/tests/test_edge_structural.c +++ b/tests/test_edge_structural.c @@ -257,6 +257,7 @@ static const char *ES_ALL_EDGE_TYPES[] = {"CALLS", "IMPORTS", "INHERITS", "INFRA_MAPS", + "MENTIONS", "OVERRIDE", "REFERENCES_FILE", "SEMANTICALLY_RELATED", diff --git a/tests/test_lang_contract.c b/tests/test_lang_contract.c index 6eb993564..8d0192feb 100644 --- a/tests/test_lang_contract.c +++ b/tests/test_lang_contract.c @@ -1069,6 +1069,7 @@ static const char *ALL_EDGE_TYPES[] = {"CALLS", "IMPORTS", "INHERITS", "INFRA_MAPS", + "MENTIONS", "OVERRIDE", "REFERENCES_FILE", "SEMANTICALLY_RELATED", diff --git a/tests/test_main.c b/tests/test_main.c index b63f07035..8bd1a81d5 100644 --- a/tests/test_main.c +++ b/tests/test_main.c @@ -969,6 +969,9 @@ extern void suite_store_checkpoint(void); extern void suite_traces(void); extern void suite_configlink(void); extern void suite_doclinks(void); +extern void suite_doc_mentions(void); +extern void suite_doc_mentions_cost(void); +extern void suite_doc_mentions_msbuild(void); extern void suite_infrascan(void); extern void suite_cli(void); extern void suite_agent_clients(void); @@ -1336,6 +1339,11 @@ int main(int argc, char **argv) { /* Markdown file reference link */ RUN_SELECTED_SUITE(doclinks); + /* Doc-comment references -> MENTIONS */ + RUN_SELECTED_SUITE(doc_mentions); + RUN_SELECTED_SUITE(doc_mentions_cost); + RUN_SELECTED_SUITE(doc_mentions_msbuild); + /* Infrastructure scanning */ RUN_SELECTED_SUITE(infrascan);