diff --git a/internal/cbm/cbm.c b/internal/cbm/cbm.c index e54f4b272..35f7603b5 100644 --- a/internal/cbm/cbm.c +++ b/internal/cbm/cbm.c @@ -279,6 +279,8 @@ void cbm_channels_push(CBMChannelArray *arr, CBMArena *a, CBMChannel ch) { typedef struct { const char *string; uint32_t length; + /* #2078: serve one "\n" at offset `length` -- see cbm_parse_source. */ + bool virtual_newline; } CBMStringInput; static const char *cbm_string_read(void *payload, uint32_t byte, TSPoint point, @@ -286,6 +288,10 @@ static const char *cbm_string_read(void *payload, uint32_t byte, TSPoint point, (void)point; CBMStringInput *self = (CBMStringInput *)payload; if (byte >= self->length) { + if (self->virtual_newline && byte == self->length) { + *bytes_read = 1; + return "\n"; + } *bytes_read = 0; return ""; } @@ -293,6 +299,60 @@ static const char *cbm_string_read(void *payload, uint32_t byte, TSPoint point, return self->string + byte; } +/* Where the parser stands after reading all `len` real bytes. */ +static TSPoint cbm_source_end_point(const char *source, uint32_t len) { + TSPoint end = {0, 0}; + const char *p = source; + const char *stop = source + len; + const char *nl; + while (p < stop && (nl = memchr(p, '\n', (size_t)(stop - p))) != NULL) { + end.row++; + p = nl + 1; + } + end.column = (uint32_t)(stop - p); + return end; +} + +/* #2078: parse every file as if its last line were terminated. + * + * Many grammars treat the line terminator as part of the construct it ends -- + * a markdown fence, heading or list marker, a Makefile recipe, a Dockerfile + * instruction. A file whose last byte is not "\n" leaves that construct + * unterminated: at best a zero-width MISSING token (#1610/#1746 excuse those at + * EOF), at worst a WIDTH-BEARING error that costs the whole last line or, for a + * one-line file, the whole file, and sometimes a silent loss (an unterminated + * markdown heading produced no Section and no flag). The same bytes plus one + * "\n" parse clean, so the absent newline is supplied instead of excused. + * + * The parser reads one virtual "\n" past EOF; afterwards a single tree edit + * deletes it again. tree-sitter's own edit arithmetic then clamps every node + * that reached into the virtual byte back to the real end -- byte offsets AND + * row/column points -- so every consumer (def line ranges, node text, call + * byte spans, parse-coverage ranges, the retained tree the LSP passes reuse) + * sees only real bytes and real lines; nothing downstream needs to know. The + * caller's buffer is never copied or written. + * + * A file that is empty or already ends in "\n" is parsed exactly as before. */ +TSTree *cbm_parse_source(TSParser *parser, const char *source, uint32_t source_len, + TSParseOptions opts) { + CBMStringInput input = {source, source_len, source_len > 0 && source[source_len - 1] != '\n'}; + TSInput ts_input = {&input, cbm_string_read, TSInputEncodingUTF8, NULL}; + TSTree *tree = ts_parser_parse_with_options(parser, NULL, ts_input, opts); + if (tree && input.virtual_newline) { + TSPoint end = cbm_source_end_point(source, source_len); + TSInputEdit drop_virtual_newline = { + .start_byte = source_len, + .old_end_byte = source_len + 1, + .new_end_byte = source_len, + .start_point = end, + .old_end_point = {end.row + 1, 0}, + .new_end_point = end, + }; + ts_tree_edit(tree, &drop_virtual_newline); + } + return tree; +} + // --- Parse timeout callback --- /* Budget for the tree-sitter progress callback. The PRIMARY gate is per-thread @@ -1262,7 +1322,11 @@ static void cbm_error_regions_push(cbm_error_regions_t *acc, TSNode n) { * #1746: the Dockerfile grammar places that zero-width missing newline before * trailing whitespace rather than at raw EOF. Preserve the broad exact-EOF * rule above; only extend it past blanks when the missing token is specifically - * a newline. */ + * a newline. + * + * #2078: cbm_parse_source now supplies the absent final newline, so a MISSING + * newline at EOF no longer arises from it; the rule stays for the other + * zero-width terminators a grammar can leave at EOF. */ static bool cbm_is_blank_not_newline(char c) { return c == ' ' || c == '\t' || c == '\v' || c == '\f' || c == '\r'; } @@ -2206,8 +2270,6 @@ static void extract_cpp_branch_views(const CBMExtractCtx *raw, const TSLanguage return; } ts_parser_reset(parser); - CBMStringInput input = {view, (uint32_t)raw->source_len}; - TSInput ts_input = {&input, cbm_string_read, TSInputEncodingUTF8, NULL}; TSParseOptions opts = {0}; CBMParseBudget budget = {0}; // cppcheck-suppress unreadVariable if (timeout_micros > 0) { @@ -2217,7 +2279,9 @@ static void extract_cpp_branch_views(const CBMExtractCtx *raw, const TSLanguage opts.payload = &budget; opts.progress_callback = cbm_timeout_cb; } - TSTree *tree = ts_parser_parse_with_options(parser, NULL, ts_input, opts); + /* A view keeps the raw source's length and line structure, so it is + * parsed under the same terminated-last-line rule (#2078). */ + TSTree *tree = cbm_parse_source(parser, view, (uint32_t)raw->source_len, opts); if (!tree) { return; } @@ -2325,15 +2389,7 @@ static CBMFileResult *extract_file_ex_body(const char *source, int source_len, C uint64_t t0 = now_ns(); - // Build string input + timeout options for parse_with_options - CBMStringInput str_input = {source, (uint32_t)source_len}; - TSInput ts_input = { - &str_input, - cbm_string_read, - TSInputEncodingUTF8, - NULL, - }; - + // Timeout options for parse_with_options TSParseOptions opts = {0}; CBMParseBudget budget = {0}; // cppcheck-suppress unreadVariable uint64_t budget_ns = 0; @@ -2361,12 +2417,14 @@ static CBMFileResult *extract_file_ex_body(const char *source, int source_len, C /* #1735: a SQL data dump's literal-only INSERT rows carry no graph content * but dominate its parse. Keep them out through included ranges; offsets * and positions of everything kept are unchanged. The parser is - * thread-local and reused, so the ranges are cleared right after. */ + * thread-local and reused, so the ranges are cleared right after. The last + * range ends at the real EOF, so the virtual final newline cbm_parse_source + * supplies (#2078) stays outside a ranged parse. */ CBMSqlKeptRanges sql_kept = {NULL, 0}; bool sql_ranged = language == CBM_LANG_SQL && cbm_sql_values_exclusion_on(rel_path) && cbm_sql_values_kept_ranges(source, (uint32_t)source_len, &sql_kept) && ts_parser_set_included_ranges(parser, sql_kept.items, sql_kept.count); - TSTree *tree = ts_parser_parse_with_options(parser, NULL, ts_input, opts); + TSTree *tree = cbm_parse_source(parser, source, (uint32_t)source_len, opts); if (sql_ranged) { (void)ts_parser_set_included_ranges(parser, NULL, 0); } @@ -2587,7 +2645,7 @@ static CBMFileResult *extract_file_ex_body(const char *source, int source_len, C TSParser *pp_parser = get_thread_parser(ts_lang, language); if (pp_parser) { ts_parser_reset(pp_parser); - CBMStringInput pp_input = {expanded, (uint32_t)expanded_len}; + CBMStringInput pp_input = {expanded, (uint32_t)expanded_len, false}; TSInput pp_ts_input = { &pp_input, cbm_string_read, diff --git a/internal/cbm/cbm.h b/internal/cbm/cbm.h index 5d723c118..7982610a3 100644 --- a/internal/cbm/cbm.h +++ b/internal/cbm/cbm.h @@ -894,6 +894,14 @@ void cbm_work_arena_keep_begin(void); /* Free the compaction scratch this thread kept (cbm_work_arena_release calls it). */ void cbm_result_compact_release_thread(void); +/* Parse one whole file as if its last line ended with "\n" (#2078). The + * parser sees the source plus one virtual newline when the last byte is not + * already one; the returned tree is then clamped back to `source_len`, so no + * node range, point or text reaches past the real bytes. Every whole-file parse + * goes through here, so a retained tree and a fallback re-parse agree. */ +TSTree *cbm_parse_source(TSParser *parser, const char *source, uint32_t source_len, + TSParseOptions opts); + // Extract all data from one file. Caller must call cbm_free_result(). // source must remain valid for the duration of the call. // timeout_micros: per-file tree-sitter parse budget in microseconds of the diff --git a/internal/cbm/lsp/c_lsp.c b/internal/cbm/lsp/c_lsp.c index 29fd12284..a6760fd7d 100644 --- a/internal/cbm/lsp/c_lsp.c +++ b/internal/cbm/lsp/c_lsp.c @@ -6753,7 +6753,7 @@ bool cbm_run_c_lsp_cross_with_registry_with_test_owners(CBMArena *arena, const c return false; const TSLanguage *ts_lang = cpp_mode ? tree_sitter_cpp() : tree_sitter_c(); ts_parser_set_language(parser, ts_lang); - tree = ts_parser_parse_string(parser, NULL, source, source_len); + tree = cbm_parse_source(parser, source, (uint32_t)source_len, (TSParseOptions){0}); ts_parser_delete(parser); owns_tree = true; if (!tree) @@ -6823,7 +6823,7 @@ bool cbm_run_c_lsp_cross_with_test_owners(CBMArena *arena, const char *source, i return false; const TSLanguage *ts_lang = cpp_mode ? tree_sitter_cpp() : tree_sitter_c(); ts_parser_set_language(parser, ts_lang); - tree = ts_parser_parse_string(parser, NULL, source, source_len); + tree = cbm_parse_source(parser, source, (uint32_t)source_len, (TSParseOptions){0}); ts_parser_delete(parser); owns_tree = true; if (!tree) diff --git a/internal/cbm/lsp/cs_lsp.c b/internal/cbm/lsp/cs_lsp.c index fc3b548ef..749b126bf 100644 --- a/internal/cbm/lsp/cs_lsp.c +++ b/internal/cbm/lsp/cs_lsp.c @@ -3763,8 +3763,9 @@ void cbm_run_cs_lsp_cross_with_registry(CBMArena *arena, const char *source, int if (!parser) return; ts_parser_set_language(parser, tree_sitter_c_sharp()); - tree = ts_parser_parse_string( - parser, NULL, source, source_len > 0 ? (uint32_t)source_len : (uint32_t)strlen(source)); + tree = cbm_parse_source(parser, source, + source_len > 0 ? (uint32_t)source_len : (uint32_t)strlen(source), + (TSParseOptions){0}); ts_parser_delete(parser); owns = true; } @@ -3805,8 +3806,9 @@ void cbm_run_cs_lsp_cross(CBMArena *arena, const char *source, int source_len, if (!parser) return; ts_parser_set_language(parser, tree_sitter_c_sharp()); - tree = ts_parser_parse_string( - parser, NULL, source, source_len > 0 ? (uint32_t)source_len : (uint32_t)strlen(source)); + tree = cbm_parse_source(parser, source, + source_len > 0 ? (uint32_t)source_len : (uint32_t)strlen(source), + (TSParseOptions){0}); ts_parser_delete(parser); owns = true; } diff --git a/internal/cbm/lsp/go_lsp.c b/internal/cbm/lsp/go_lsp.c index ba7df163b..58c151a53 100644 --- a/internal/cbm/lsp/go_lsp.c +++ b/internal/cbm/lsp/go_lsp.c @@ -3224,7 +3224,7 @@ void cbm_run_go_lsp_cross(CBMArena *arena, const char *source, int source_len, if (!parser) return; ts_parser_set_language(parser, tree_sitter_go()); - tree = ts_parser_parse_string(parser, NULL, source, source_len); + tree = cbm_parse_source(parser, source, (uint32_t)source_len, (TSParseOptions){0}); owns_tree = true; if (!tree) { ts_parser_delete(parser); @@ -3630,7 +3630,7 @@ void cbm_run_go_lsp_cross_with_registry(CBMArena *arena, const char *source, int if (!parser) return; ts_parser_set_language(parser, tree_sitter_go()); - tree = ts_parser_parse_string(parser, NULL, source, source_len); + tree = cbm_parse_source(parser, source, (uint32_t)source_len, (TSParseOptions){0}); owns_tree = true; if (!tree) { ts_parser_delete(parser); diff --git a/internal/cbm/lsp/java_lsp.c b/internal/cbm/lsp/java_lsp.c index 9a79783ff..2b518d890 100644 --- a/internal/cbm/lsp/java_lsp.c +++ b/internal/cbm/lsp/java_lsp.c @@ -3931,7 +3931,7 @@ void cbm_run_java_lsp_cross_with_registry(CBMArena *arena, CBMFileResult *result return; } ts_parser_set_language(parser, tree_sitter_java()); - tree = ts_parser_parse_string(parser, NULL, source, (uint32_t)source_len); + tree = cbm_parse_source(parser, source, (uint32_t)source_len, (TSParseOptions){0}); ts_parser_delete(parser); owns_tree = true; } @@ -3985,7 +3985,7 @@ void cbm_run_java_lsp_cross(CBMArena *arena, const char *source, int source_len, if (!tree) { TSParser *parser = ts_parser_new(); ts_parser_set_language(parser, tree_sitter_java()); - tree = ts_parser_parse_string(parser, NULL, source, (uint32_t)source_len); + tree = cbm_parse_source(parser, source, (uint32_t)source_len, (TSParseOptions){0}); ts_parser_delete(parser); owns_tree = true; } diff --git a/internal/cbm/lsp/kotlin_lsp.c b/internal/cbm/lsp/kotlin_lsp.c index 0ca49afd5..cc6be1680 100644 --- a/internal/cbm/lsp/kotlin_lsp.c +++ b/internal/cbm/lsp/kotlin_lsp.c @@ -5415,7 +5415,7 @@ void cbm_run_kotlin_lsp_cross(CBMArena *arena, const char *source, int source_le return; } ts_parser_set_language(parser, tree_sitter_kotlin()); - tree = ts_parser_parse_string(parser, NULL, source, (uint32_t)source_len); + tree = cbm_parse_source(parser, source, (uint32_t)source_len, (TSParseOptions){0}); ts_parser_delete(parser); owns_tree = true; } diff --git a/internal/cbm/lsp/php_lsp.c b/internal/cbm/lsp/php_lsp.c index 4caf5ba0f..81347a94c 100644 --- a/internal/cbm/lsp/php_lsp.c +++ b/internal/cbm/lsp/php_lsp.c @@ -4498,7 +4498,7 @@ void cbm_run_php_lsp_cross(CBMArena *arena, const char *source, int source_len, if (!parser) return; ts_parser_set_language(parser, tree_sitter_php_only()); - tree = ts_parser_parse_string(parser, NULL, source, (uint32_t)source_len); + tree = cbm_parse_source(parser, source, (uint32_t)source_len, (TSParseOptions){0}); owns_tree = true; if (!tree) { ts_parser_delete(parser); diff --git a/internal/cbm/lsp/py_lsp.c b/internal/cbm/lsp/py_lsp.c index 70de5dce7..2b47fbf23 100644 --- a/internal/cbm/lsp/py_lsp.c +++ b/internal/cbm/lsp/py_lsp.c @@ -5110,7 +5110,7 @@ void cbm_run_py_lsp_cross(CBMArena *arena, const char *source, int source_len, if (!parser) return; ts_parser_set_language(parser, tree_sitter_python()); - tree = ts_parser_parse_string(parser, NULL, source, (uint32_t)source_len); + tree = cbm_parse_source(parser, source, (uint32_t)source_len, (TSParseOptions){0}); owns_tree = true; if (!tree) { ts_parser_delete(parser); @@ -5191,7 +5191,7 @@ void cbm_run_py_lsp_cross_with_registry(CBMArena *arena, const char *source, int if (!parser) return; ts_parser_set_language(parser, tree_sitter_python()); - tree = ts_parser_parse_string(parser, NULL, source, (uint32_t)source_len); + tree = cbm_parse_source(parser, source, (uint32_t)source_len, (TSParseOptions){0}); owns_tree = true; if (!tree) { ts_parser_delete(parser); diff --git a/internal/cbm/lsp/rust_lsp.c b/internal/cbm/lsp/rust_lsp.c index 68fb444a6..44680cb69 100644 --- a/internal/cbm/lsp/rust_lsp.c +++ b/internal/cbm/lsp/rust_lsp.c @@ -6377,7 +6377,7 @@ void cbm_run_rust_lsp_cross_with_registry(CBMArena *arena, const char *source, i if (!parser) return; ts_parser_set_language(parser, tree_sitter_rust()); - tree = ts_parser_parse_string(parser, NULL, source, source_len); + tree = cbm_parse_source(parser, source, (uint32_t)source_len, (TSParseOptions){0}); owns_tree = true; if (!tree) { ts_parser_delete(parser); @@ -6412,7 +6412,7 @@ void cbm_run_rust_lsp_cross_with_manifest(CBMArena *arena, const char *source, i if (!parser) return; ts_parser_set_language(parser, tree_sitter_rust()); - tree = ts_parser_parse_string(parser, NULL, source, source_len); + tree = cbm_parse_source(parser, source, (uint32_t)source_len, (TSParseOptions){0}); owns_tree = true; if (!tree) { ts_parser_delete(parser); diff --git a/internal/cbm/lsp/ts_lsp.c b/internal/cbm/lsp/ts_lsp.c index 967d06762..c25a58bbe 100644 --- a/internal/cbm/lsp/ts_lsp.c +++ b/internal/cbm/lsp/ts_lsp.c @@ -5899,7 +5899,7 @@ void cbm_run_ts_lsp_cross_with_registry(CBMArena *arena, const char *source, int jsx_mode ? (js_mode ? tree_sitter_javascript() : tree_sitter_tsx()) : (js_mode ? tree_sitter_javascript() : tree_sitter_typescript()); ts_parser_set_language(parser, lang); - tree = ts_parser_parse_string(parser, NULL, source, source_len); + tree = cbm_parse_source(parser, source, (uint32_t)source_len, (TSParseOptions){0}); ts_parser_delete(parser); if (!tree) return; @@ -6110,7 +6110,7 @@ void cbm_run_ts_lsp_cross(CBMArena *arena, const char *source, int source_len, jsx_mode ? (js_mode ? tree_sitter_javascript() : tree_sitter_tsx()) : (js_mode ? tree_sitter_javascript() : tree_sitter_typescript()); ts_parser_set_language(parser, lang); - tree = ts_parser_parse_string(parser, NULL, source, source_len); + tree = cbm_parse_source(parser, source, (uint32_t)source_len, (TSParseOptions){0}); ts_parser_delete(parser); if (!tree) return; diff --git a/tests/test_parse_coverage.c b/tests/test_parse_coverage.c index 5dbc38e4a..def6dffa9 100644 --- a/tests/test_parse_coverage.c +++ b/tests/test_parse_coverage.c @@ -507,11 +507,17 @@ TEST(real_error_before_eof_still_flagged_without_final_newline_issue1610) { } /* GUARD: a MISSING/ERROR node WITH WIDTH at EOF is a genuine loss and must - * still be flagged. A Makefile whose final recipe line lacks its newline really - * does drop the recipe from the tree — cbm's flag is honest there. */ + * still be flagged. The construct here is broken whether or not the line is + * terminated, so neither the zero-width rule nor the virtual final newline + * (#2078) may excuse it. + * + * This guard used a Makefile whose final recipe line lacks its newline, which + * really did drop the recipe from the tree. Since #2078 the parser sees that + * line terminated and the recipe parses, so there is no loss left to report -- + * see makefile_unterminated_recipe_is_parsed_issue2078. */ TEST(width_bearing_error_at_eof_still_flagged_issue1610) { - const char *src = "all:\n\techo hi"; /* no trailing newline; recipe is lost */ - CBMFileResult *r = do_extract(src, CBM_LANG_MAKEFILE, "Makefile"); + const char *src = "def ok():\n return 1\nx = (1,"; /* unclosed at EOF */ + CBMFileResult *r = do_extract(src, CBM_LANG_PYTHON, "eof.py"); ASSERT_NOT_NULL(r); bool flagged = r->parse_incomplete; cbm_free_result(r); @@ -521,6 +527,22 @@ TEST(width_bearing_error_at_eof_still_flagged_issue1610) { PASS(); } +/* #2078: the recipe the old #1610 guard lost is now parsed -- the file is not + * flagged, with or without a trailing blank. */ +TEST(makefile_unterminated_recipe_is_parsed_issue2078) { + const char *cases[] = {"all:\n\techo hi", "all:\n\techo hi "}; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + CBMFileResult *r = do_extract(cases[i], CBM_LANG_MAKEFILE, "Makefile"); + ASSERT_NOT_NULL(r); + bool flagged = r->parse_incomplete; + cbm_free_result(r); + if (flagged) { + FAIL("an unterminated final recipe line must parse like a terminated one"); + } + } + PASS(); +} + /* #1838: tree-sitter-perl v1.0.0 rejects a valid line-oriented format and * reports the file as partial. The supported upstream v1.2.1 grammar accepts * the format and preserves declaration extraction beyond its dot terminator. */ @@ -611,11 +633,12 @@ TEST(real_error_before_eof_still_flagged_with_trailing_blank_issue1746) { PASS(); } -/* GUARD: a WIDTH-BEARING loss at EOF stays honest with a blank tail too — the - * Makefile recipe really is dropped, and only zero-width nodes are excused. */ +/* GUARD: a WIDTH-BEARING loss at EOF stays honest with a blank tail too, and + * only zero-width nodes are excused. (Formerly the unterminated Makefile + * recipe, which #2078 now parses -- see the #1610 guard above.) */ TEST(width_bearing_error_at_eof_still_flagged_with_trailing_blank_issue1746) { - const char *src = "all:\n\techo hi "; - CBMFileResult *r = do_extract(src, CBM_LANG_MAKEFILE, "Makefile"); + const char *src = "def ok():\n return 1\nx = (1, "; + CBMFileResult *r = do_extract(src, CBM_LANG_PYTHON, "eof.py"); ASSERT_NOT_NULL(r); bool flagged = r->parse_incomplete; cbm_free_result(r); @@ -1615,6 +1638,246 @@ TEST(cs_malformed_conditional_remains_partial_issue1748) { PASS(); } +/* ── #2078: every file is parsed as if it ended with a newline ─────────────── + * + * Several grammars need a line terminator that a file's last line may simply + * not have. tree-sitter-markdown is the sharpest case: an opening code fence, + * an ATX heading or a bare list marker as the final bytes of a file leaves a + * WIDTH-BEARING ERROR (so the #1610 zero-width rule cannot excuse it), and a + * file that is only "```" came back as a whole-file error. Appending one "\n" + * made every one of these clean -- the reporter proved it on 21 real files. + * + * cbm now feeds the parser one virtual "\n" past EOF when the last byte is not + * already a newline, then clamps the tree back to the real length, so no + * consumer sees a byte, a line or a range that is not in the file. */ +TEST(markdown_unterminated_last_line_is_clean_issue2078) { + const char *cases[] = { + "text\n```", /* opening fence at EOF (upstream PR #262, closed unmerged) */ + "```", /* the whole file is one opening fence */ + "# Top", /* ATX heading */ + "- ", /* bare list marker */ + }; + int flagged = 0; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + CBMFileResult *r = do_extract(cases[i], CBM_LANG_MARKDOWN, "doc.md"); + ASSERT_NOT_NULL(r); + if (r->parse_incomplete || r->parse_unusable) { + fprintf(stderr, " case %zu flagged: partial=%d unusable=%d ranges=%s\n", i, + r->parse_incomplete, r->parse_unusable, + r->error_ranges ? r->error_ranges : "(none)"); + flagged++; + } + cbm_free_result(r); + } + if (flagged) { + FAIL("a markdown file must not be flagged only because its last line is unterminated"); + } + PASS(); +} + +/* Controls: the same bytes WITH the newline were already clean and stay so. */ +TEST(markdown_terminated_last_line_controls_unchanged_issue2078) { + const char *cases[] = {"text\n```\n", "```\n", "# Top\n", "- \n"}; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + CBMFileResult *r = do_extract(cases[i], CBM_LANG_MARKDOWN, "doc.md"); + ASSERT_NOT_NULL(r); + ASSERT_FALSE(r->parse_incomplete); + ASSERT_FALSE(r->parse_unusable); + cbm_free_result(r); + } + PASS(); +} + +/* Everything extraction reports, as one string, so two results compare + * byte-for-byte: defs (every text field and range), calls (exact byte spans), + * imports and the coverage verdict. */ +typedef struct { + char *buf; + size_t cap; + size_t len; +} dump_buf_t; + +static void dump_str(dump_buf_t *b, const char *tag, const char *s) { + int n = snprintf(b->buf + b->len, b->cap - b->len, " %s=%s", tag, s ? s : "~"); + if (n > 0 && (size_t)n < b->cap - b->len) { + b->len += (size_t)n; + } +} + +static void dump_num(dump_buf_t *b, const char *tag, long v) { + int n = snprintf(b->buf + b->len, b->cap - b->len, " %s=%ld", tag, v); + if (n > 0 && (size_t)n < b->cap - b->len) { + b->len += (size_t)n; + } +} + +/* The Module def spans the whole file, and its end line follows the line + * convention for a terminated last line (#1967), so it is compared on its own + * rather than through the dump. */ +static char *dump_extraction(const CBMFileResult *r) { + dump_buf_t b = {(char *)calloc(1u << 16, 1), 1u << 16, 0}; + if (!b.buf) { + return NULL; + } + for (int i = 0; i < r->defs.count; i++) { + const CBMDefinition *d = &r->defs.items[i]; + if (d->label && strcmp(d->label, "Module") == 0) { + continue; + } + dump_str(&b, "\nD", d->qualified_name); + dump_str(&b, "name", d->name); + dump_str(&b, "label", d->label); + dump_num(&b, "start", d->start_line); + dump_num(&b, "end", d->end_line); + dump_num(&b, "lines", d->lines); + dump_num(&b, "cx", d->complexity); + dump_num(&b, "params", d->param_count); + dump_str(&b, "sig", d->signature); + dump_str(&b, "ret", d->return_type); + dump_str(&b, "doc", d->docstring); + dump_str(&b, "prof", d->structural_profile); + dump_str(&b, "tok", d->body_tokens); + } + for (int i = 0; i < r->calls.count; i++) { + const CBMCall *c = &r->calls.items[i]; + dump_str(&b, "\nC", c->callee_name); + dump_str(&b, "in", c->enclosing_func_qn); + dump_num(&b, "line", c->start_line); + dump_num(&b, "from", (long)c->site_start_byte); + dump_num(&b, "to", (long)c->site_end_byte); + dump_num(&b, "args", c->arg_count); + } + for (int i = 0; i < r->imports.count; i++) { + dump_str(&b, "\nI", r->imports.items[i].local_name); + dump_str(&b, "path", r->imports.items[i].module_path); + } + dump_str(&b, "\nP", r->error_ranges); + dump_num(&b, "partial", r->parse_incomplete); + dump_num(&b, "unusable", r->parse_unusable); + dump_str(&b, "", "\n"); + return b.buf; +} + +static uint32_t module_end_line(const CBMFileResult *r) { + for (int i = 0; i < r->defs.count; i++) { + if (r->defs.items[i].label && strcmp(r->defs.items[i].label, "Module") == 0) { + return r->defs.items[i].end_line; + } + } + return 0; +} + +static char *extract_dump(const char *src, CBMLanguage lang, const char *path, + uint32_t *module_end) { + CBMFileResult *r = do_extract(src, lang, path); + if (!r) { + return NULL; + } + char *d = dump_extraction(r); + if (module_end) { + *module_end = module_end_line(r); + } + cbm_free_result(r); + return d; +} + +/* GUARD: for code grammars a trailing newline is insignificant, so a file + * without one must extract byte-identically to the same file with one -- every + * def, range, text field and call byte span. The virtual newline may never + * show up as phantom content or a phantom line. */ +TEST(code_without_final_newline_extracts_identically_issue2078) { + struct { + const char *src; /* no trailing newline */ + CBMLanguage lang; + const char *path; + uint32_t lines; + } cases[] = { + {"#include \n\nstatic int beta(int x) {\n return x + 1;\n}\n\n" + "void alpha(void) { printf(\"a\"); beta(2); }", + CBM_LANG_C, "a.c", 7}, + {"import os\n\n\nclass K:\n def m(self, a):\n return os.path.join(a)\n\n\n" + "def f(x):\n \"\"\"doc\"\"\"\n return K().m(x)", + CBM_LANG_PYTHON, "a.py", 11}, + }; + for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { + size_t n = strlen(cases[i].src); + char *terminated = (char *)malloc(n + 2); + ASSERT_NOT_NULL(terminated); + memcpy(terminated, cases[i].src, n); + terminated[n] = '\n'; + terminated[n + 1] = '\0'; + uint32_t bare_module_end = 0; + char *bare = extract_dump(cases[i].src, cases[i].lang, cases[i].path, &bare_module_end); + char *term = extract_dump(terminated, cases[i].lang, cases[i].path, NULL); + free(terminated); + ASSERT_NOT_NULL(bare); + ASSERT_NOT_NULL(term); + bool same = strcmp(bare, term) == 0; + if (!same) { + fprintf(stderr, " %s differs:\n--- without final newline%s--- with final newline%s", + cases[i].path, bare, term); + } + free(bare); + free(term); + if (!same) { + FAIL("extraction must not depend on whether the last line is terminated"); + } + if (bare_module_end != cases[i].lines) { + fprintf(stderr, " %s: module ends on line %u, file has %u\n", cases[i].path, + bare_module_end, cases[i].lines); + FAIL("the virtual newline must not add a line to the module"); + } + } + PASS(); +} + +/* GUARD against phantom content: a heading on an unterminated last line is a + * one-line file. No def may end past line 1, no text may carry the virtual + * newline, and the result must equal the terminated file's. */ +TEST(virtual_newline_adds_no_phantom_line_or_text_issue2078) { + CBMFileResult *r = do_extract("# Top", CBM_LANG_MARKDOWN, "top.md"); + ASSERT_NOT_NULL(r); + bool found = has_def(r, "Top"); + for (int i = 0; i < r->defs.count; i++) { + const CBMDefinition *d = &r->defs.items[i]; + if (d->end_line > 1 || d->start_line > 1) { + fprintf(stderr, " def %s spans %u-%u\n", d->name ? d->name : "?", d->start_line, + d->end_line); + cbm_free_result(r); + FAIL("no def may reach past the last real line"); + } + if ((d->name && strchr(d->name, '\n')) || (d->signature && strchr(d->signature, '\n'))) { + cbm_free_result(r); + FAIL("no extracted text may contain the virtual newline"); + } + } + cbm_free_result(r); + if (!found) { + /* Before #2078 the unterminated heading was silently dropped: no + * Section, and no parse_partial flag to say so. */ + FAIL("the heading on an unterminated last line must be extracted as a Section"); + } + PASS(); +} + +/* GUARD: the empty file and a file that is exactly "\n" get no virtual + * newline (nothing is unterminated) and stay clean. */ +TEST(empty_and_newline_only_files_unchanged_issue2078) { + const char *srcs[] = {"", "\n"}; + CBMLanguage langs[] = {CBM_LANG_MARKDOWN, CBM_LANG_PYTHON, CBM_LANG_C}; + const char *paths[] = {"e.md", "e.py", "e.c"}; + for (size_t s = 0; s < 2; s++) { + for (size_t l = 0; l < 3; l++) { + CBMFileResult *r = do_extract(srcs[s], langs[l], paths[l]); + ASSERT_NOT_NULL(r); + ASSERT_FALSE(r->parse_incomplete); + ASSERT_FALSE(r->parse_unusable); + cbm_free_result(r); + } + } + PASS(); +} + /* #1967: every line counter over one source buffer answers the same * question the same way. A trailing newline ends the last line; it does not * open a new one. */ @@ -1658,6 +1921,7 @@ SUITE(parse_coverage) { RUN_TEST(missing_final_newline_not_flagged_across_grammars_issue1610); RUN_TEST(real_error_before_eof_still_flagged_without_final_newline_issue1610); RUN_TEST(width_bearing_error_at_eof_still_flagged_issue1610); + RUN_TEST(makefile_unterminated_recipe_is_parsed_issue2078); RUN_TEST(c_ifdef_split_range_narrows_to_dropped_branch); RUN_TEST(c_ifdef_split_range_excludes_lines_the_preprocessor_explained); RUN_TEST(c_ifdef_split_range_never_starts_on_a_directive); @@ -1684,4 +1948,9 @@ SUITE(parse_coverage) { RUN_TEST(cs_collection_expression_in_conditional_is_complete_issue1748); RUN_TEST(cs_calls_inside_collection_expression_extracted_issue1748); RUN_TEST(cs_malformed_conditional_remains_partial_issue1748); + RUN_TEST(markdown_unterminated_last_line_is_clean_issue2078); + RUN_TEST(markdown_terminated_last_line_controls_unchanged_issue2078); + RUN_TEST(code_without_final_newline_extracts_identically_issue2078); + RUN_TEST(virtual_newline_adds_no_phantom_line_or_text_issue2078); + RUN_TEST(empty_and_newline_only_files_unchanged_issue2078); }