diff --git a/Makefile.cbm b/Makefile.cbm index 7a3cd31e16..368fce973e 100644 --- a/Makefile.cbm +++ b/Makefile.cbm @@ -348,6 +348,7 @@ EXTRACTION_SRCS = \ $(CBM_DIR)/extract_channels.c \ $(CBM_DIR)/extract_k8s.c \ $(CBM_DIR)/extract_dbt.c \ + $(CBM_DIR)/sql_values.c \ $(CBM_DIR)/helpers.c \ $(CBM_DIR)/result_compact.c \ $(CBM_DIR)/result_spill.c \ diff --git a/internal/cbm/cbm.c b/internal/cbm/cbm.c index b54fb54a7a..082bca1d6d 100644 --- a/internal/cbm/cbm.c +++ b/internal/cbm/cbm.c @@ -21,6 +21,7 @@ #include "lsp/kotlin_lsp.h" #include "lsp/rust_lsp.h" #include "preprocessor.h" +#include "sql_values.h" // #1735: literal INSERT rows kept out of the SQL parse #include "foundation/compat.h" #include "foundation/compat_fs.h" // cbm_fopen — crash-supervisor per-file marker write #include "foundation/hash_table.h" // CBMHashTable — crash-supervisor quarantine set @@ -1773,6 +1774,19 @@ static void cbm_subtract_recovered_regions(cbm_error_regions_t *regs, const CBMD *regs = out; } +#ifdef CBM_ENABLE_TEST_SEAMS +/* Source bytes this thread has read to locate lines for the #1071 macro check + * (walks and line-table builds). Lets a test pin the check's cost as a count + * instead of a clock (#1735). Compiled only into seam-enabled test artifacts. */ +static CBM_TLS uint64_t tl_macro_line_scan_bytes = 0; +uint64_t cbm_test_macro_line_scan_bytes(void) { + return tl_macro_line_scan_bytes; +} +#define CBM_MACRO_LINE_SCAN(n) (tl_macro_line_scan_bytes += (uint64_t)(n)) +#else +#define CBM_MACRO_LINE_SCAN(n) ((void)0) +#endif + /* #1071: a function-like macro invocation whose argument is a type token * (e.g. ALLOC(int, n)) makes tree-sitter's C/C++ grammar emit an ERROR node — it * parses `int` in expression position — which would be recorded as a parse_partial @@ -1862,6 +1876,7 @@ static bool cbm_span_is_macro_invocation(const char *src, int src_len, uint32_t } } if (line != start_line) { + CBM_MACRO_LINE_SCAN(span_start); return false; } int span_end = span_start; @@ -1870,6 +1885,7 @@ static bool cbm_span_is_macro_invocation(const char *src, int src_len, uint32_t line++; } } + CBM_MACRO_LINE_SCAN(span_end); return cbm_byte_span_is_macro_invocation(src, src_len, span_start, span_end, defs); } @@ -1896,21 +1912,59 @@ static bool cbm_region_inside_callable(uint32_t rs, uint32_t re, const CBMDefArr return false; } +/* cbm_span_is_macro_invocation over a cbm_line_offsets table: offs[k] is where + * 1-based line k+1 starts and offs[nlines] is the end of the source, so a span + * costs two reads instead of a walk from byte 0. Same answer as the walk; falls + * back to it when there is no table. */ +static bool cbm_lines_are_macro_invocation(const char *src, int src_len, const uint32_t *offs, + uint32_t nlines, uint32_t start_line, uint32_t end_line, + const CBMDefArray *defs) { + if (!offs) { + return cbm_span_is_macro_invocation(src, src_len, start_line, end_line, defs); + } + if (!src || src_len <= 0 || !defs || start_line == 0 || end_line < start_line || + start_line > nlines) { + return false; + } + uint32_t last = end_line < nlines ? end_line : nlines; + return cbm_byte_span_is_macro_invocation(src, src_len, (int)offs[start_line - 1], + (int)offs[last], defs); +} + +/* #1071 subtraction. Both questions are pure, so asking the cheap one first + * changes no answer: "is the region inside a function?" reads only the + * definition list, while "is it a macro call?" has to find the region's lines + * in the source. A region no function encloses — every region of a file with + * no functions, such as a SQL data dump — never touches the source, and the + * ones that do share one line table built on first use. The old order walked + * the source from byte 0 for every region: regions x file bytes (#1735). */ static void cbm_subtract_macro_invocation_regions(cbm_error_regions_t *regs, const CBMDefArray *defs, const char *src, int src_len) { + uint32_t *offs = NULL; + uint32_t nlines = 0; + bool offs_built = false; int kept = 0; for (int i = 0; i < regs->count; i++) { - bool benign = - cbm_span_is_macro_invocation(src, src_len, regs->starts[i], regs->ends[i], defs) && - cbm_region_inside_callable(regs->starts[i], regs->ends[i], defs); + uint32_t rs = regs->starts[i]; + uint32_t re = regs->ends[i]; + bool benign = false; + if (cbm_region_inside_callable(rs, re, defs)) { + if (!offs_built) { + offs = cbm_line_offsets(src, src_len, &nlines); + offs_built = true; + CBM_MACRO_LINE_SCAN(src_len); + } + benign = cbm_lines_are_macro_invocation(src, src_len, offs, nlines, rs, re, defs); + } if (!benign) { - regs->starts[kept] = regs->starts[i]; - regs->ends[kept] = regs->ends[i]; + regs->starts[kept] = rs; + regs->ends[kept] = re; kept++; } } regs->count = kept; + cbm_free(CBM_MEM_CLASS_EXTRACT, offs); } /* Push [start, end] after trimming no-code lines off both ends. A run made @@ -2072,6 +2126,21 @@ CBMFileResult *cbm_extract_file(const char *source, int source_len, CBMLanguage enum { CBM_EXTRACT_SCRATCH_BLOCK = CBM_SZ_512 * CBM_SZ_1K }; enum { CBM_EXTRACT_SCRATCH_KEEP_BYTES = 4 * CBM_SZ_1K * CBM_SZ_1K }; +/* The #1735 value-row exclusion is always on. Test builds can turn it off for + * one file (CBM_TEST_SQL_FULL_PARSE_ON=) so a test can + * compare the excluded parse against the full one on the same source. */ +static bool cbm_sql_values_exclusion_on(const char *rel_path) { +#ifdef CBM_ENABLE_TEST_SEAMS + const char *full_on = getenv("CBM_TEST_SQL_FULL_PARSE_ON"); + if (full_on && full_on[0] && rel_path && strstr(rel_path, full_on)) { + return false; + } +#else + (void)rel_path; +#endif + return true; +} + static CBMFileResult *extract_file_ex_body(const char *source, int source_len, CBMLanguage language, const char *project, const char *rel_path, int64_t timeout_micros, const char **extra_defines, @@ -2180,7 +2249,19 @@ static CBMFileResult *extract_file_ex_body(const char *source, int source_len, C #endif } + /* #1735: a SQL data dump's literal-only INSERT rows carry no graph content + * but dominate its parse. Keep them out through included ranges; offsets + * and positions of everything kept are unchanged. The parser is + * thread-local and reused, so the ranges are cleared right after. */ + CBMSqlKeptRanges sql_kept = {NULL, 0}; + bool sql_ranged = language == CBM_LANG_SQL && cbm_sql_values_exclusion_on(rel_path) && + cbm_sql_values_kept_ranges(source, (uint32_t)source_len, &sql_kept) && + ts_parser_set_included_ranges(parser, sql_kept.items, sql_kept.count); TSTree *tree = ts_parser_parse_with_options(parser, NULL, ts_input, opts); + if (sql_ranged) { + (void)ts_parser_set_included_ranges(parser, NULL, 0); + } + cbm_sql_kept_ranges_free(&sql_kept); uint64_t t1 = now_ns(); #ifdef CBM_ENABLE_TEST_SEAMS t1 += tl_parse_wall_seam_offset_ns; /* the stall seam inflates every wall reading */ diff --git a/internal/cbm/cbm.h b/internal/cbm/cbm.h index a00192a188..1b5d3f3ef9 100644 --- a/internal/cbm/cbm.h +++ b/internal/cbm/cbm.h @@ -800,6 +800,12 @@ void cbm_free_tree(CBMFileResult *result); // Free a standalone TSTree pointer (for Go layer cleanup). void cbm_free_tree_ptr(TSTree *tree); +#ifdef CBM_ENABLE_TEST_SEAMS +// Test-only: source bytes this thread has read to locate lines for the #1071 +// macro-invocation check, cumulative (#1735). +uint64_t cbm_test_macro_line_scan_bytes(void); +#endif + // Reset the thread-local parser's internal state, releasing slab-allocated // subtrees. Must be called BEFORE cbm_slab_reset_thread() so the slab rebuild // doesn't corrupt live parser state. diff --git a/internal/cbm/sql_values.c b/internal/cbm/sql_values.c new file mode 100644 index 0000000000..21d2866896 --- /dev/null +++ b/internal/cbm/sql_values.c @@ -0,0 +1,474 @@ +// sql_values.c — exclude literal-only INSERT value tuples from the SQL parse (#1735). +// +// See sql_values.h for why. The scanner is one forward pass over the file. It +// tracks enough of SQL's lexical structure to never mistake the inside of a +// string, quoted identifier or comment for syntax: +// - '...' strings with both the standard '' escape and MySQL's backslash escape +// - "..." and `...` quoted identifiers, $tag$...$tag$ dollar-quoted bodies +// - -- and # line comments, /* */ block comments +// and it knows only one statement shape: INSERT/REPLACE ... VALUES (..), (..). +// +// Everything it is unsure about stays in the parse. A tuple is excluded only +// when every value in it is a plain literal and the tuple closes cleanly; a +// string holding a raw newline also keeps its tuple, because a dump writes line +// breaks as \n and a raw one means the scan's idea of where strings end may not +// match the grammar's. + +#include "sql_values.h" +#include "foundation/mem_core.h" +#include +#include +#include +#include + +typedef struct { + const char *s; + uint32_t len; + uint32_t i; + uint32_t row; + uint32_t bol; /* byte offset where the current row starts */ +} sqlv_scan_t; + +typedef struct { + TSRange *items; + uint32_t count; + uint32_t cap; + bool failed; +} sqlv_ranges_t; + +enum { SQLV_INITIAL_CAP = 64, SQLV_GROWTH = 2 }; + +static char sqlv_at(const sqlv_scan_t *sc, uint32_t off) { + return sc->i + off < sc->len ? sc->s[sc->i + off] : '\0'; +} + +static void sqlv_step(sqlv_scan_t *sc) { + if (sc->s[sc->i] == '\n') { + sc->row++; + sc->bol = sc->i + 1; + } + sc->i++; +} + +static TSPoint sqlv_point(const sqlv_scan_t *sc) { + return (TSPoint){sc->row, sc->i - sc->bol}; +} + +static bool sqlv_word_char(char c) { + return isalnum((unsigned char)c) || c == '_' || c == '$' || (unsigned char)c >= 0x80; +} + +static bool sqlv_word_is(const char *w, uint32_t n, const char *kw) { + return strlen(kw) == n && strncasecmp(w, kw, n) == 0; +} + +/* Consume a '..', "..", or `..` run starting at the opening quote. Doubled + * quotes escape everywhere; a backslash escapes in strings and "..". Sets + * *raw_newline when the run crosses a line break. False if it never closes. */ +static bool sqlv_skip_quoted(sqlv_scan_t *sc, bool *raw_newline) { + char q = sc->s[sc->i]; + sqlv_step(sc); + while (sc->i < sc->len) { + char c = sc->s[sc->i]; + if (c == '\n') { + *raw_newline = true; + } + if (c == '\\' && q != '`' && sc->i + 1 < sc->len) { + sqlv_step(sc); + sqlv_step(sc); + continue; + } + sqlv_step(sc); + if (c == q) { + if (sqlv_at(sc, 0) != q) { + return true; + } + sqlv_step(sc); /* doubled quote: an escaped quote character */ + } + } + return false; +} + +/* $tag$ ... $tag$ (PostgreSQL). Returns false, consuming nothing, when the `$` + * does not open a dollar quote ($1 parameters, a lone $). */ +static bool sqlv_skip_dollar(sqlv_scan_t *sc) { + uint32_t t = sc->i + 1; + if (t < sc->len && isdigit((unsigned char)sc->s[t])) { + return false; + } + while (t < sc->len && sc->s[t] != '$' && sqlv_word_char(sc->s[t])) { + t++; + } + if (t >= sc->len || sc->s[t] != '$') { + return false; + } + uint32_t tag_len = t - sc->i + 1; /* the opener, both dollars included */ + const char *tag = sc->s + sc->i; + for (uint32_t k = 0; k < tag_len; k++) { + sqlv_step(sc); + } + while (sc->i < sc->len) { + if (sc->s[sc->i] == '$' && sc->len - sc->i >= tag_len && + memcmp(sc->s + sc->i, tag, tag_len) == 0) { + for (uint32_t k = 0; k < tag_len; k++) { + sqlv_step(sc); + } + return true; + } + sqlv_step(sc); + } + return true; +} + +/* Consume one comment starting at the cursor; false if there is none. */ +static bool sqlv_skip_comment(sqlv_scan_t *sc) { + char c = sqlv_at(sc, 0); + char n = sqlv_at(sc, 1); + if ((c == '-' && n == '-') || c == '#') { + while (sc->i < sc->len && sc->s[sc->i] != '\n') { + sqlv_step(sc); + } + return true; + } + if (c == '/' && n == '*') { + sqlv_step(sc); + sqlv_step(sc); + while (sc->i < sc->len && !(sc->s[sc->i] == '*' && sqlv_at(sc, 1) == '/')) { + sqlv_step(sc); + } + if (sc->i < sc->len) { + sqlv_step(sc); + sqlv_step(sc); + } + return true; + } + return false; +} + +/* Skip whitespace and comments. *ws_only (optional) is cleared if a comment was + * among them. */ +static void sqlv_skip_trivia(sqlv_scan_t *sc, bool *ws_only) { + while (sc->i < sc->len) { + if (isspace((unsigned char)sc->s[sc->i])) { + sqlv_step(sc); + } else if (sqlv_skip_comment(sc)) { + if (ws_only) { + *ws_only = false; + } + } else { + return; + } + } +} + +/* Consume one quoted or dollar-quoted run if one starts here. */ +static bool sqlv_skip_any_quoted(sqlv_scan_t *sc) { + char c = sc->s[sc->i]; + bool nl = false; + if (c == '\'' || c == '"' || c == '`') { + (void)sqlv_skip_quoted(sc, &nl); + return true; + } + return c == '$' && sqlv_skip_dollar(sc); +} + +/* From inside an open parenthesis, consume through its matching ')'. */ +static bool sqlv_skip_balanced(sqlv_scan_t *sc) { + int depth = 1; + while (sc->i < sc->len) { + if (sqlv_skip_comment(sc) || sqlv_skip_any_quoted(sc)) { + continue; + } + char c = sc->s[sc->i]; + sqlv_step(sc); + if (c == '(') { + depth++; + } else if (c == ')' && --depth == 0) { + return true; + } + } + return false; +} + +static bool sqlv_read_string(sqlv_scan_t *sc) { + bool nl = false; + return sqlv_skip_quoted(sc, &nl) && !nl; +} + +/* 12, -3.5, 1e-9, .5, 0x1F, 0b101 — followed by a non-word character. */ +static bool sqlv_read_number(sqlv_scan_t *sc) { + char n = sqlv_at(sc, 1); + if (sqlv_at(sc, 0) == '0' && (n == 'x' || n == 'X') && + isxdigit((unsigned char)sqlv_at(sc, 2))) { + sqlv_step(sc); + sqlv_step(sc); + while (isxdigit((unsigned char)sqlv_at(sc, 0))) { + sqlv_step(sc); + } + } else if (sqlv_at(sc, 0) == '0' && (n == 'b' || n == 'B') && + (sqlv_at(sc, 2) == '0' || sqlv_at(sc, 2) == '1')) { + sqlv_step(sc); + sqlv_step(sc); + while (sqlv_at(sc, 0) == '0' || sqlv_at(sc, 0) == '1') { + sqlv_step(sc); + } + } else { + while (isdigit((unsigned char)sqlv_at(sc, 0)) || sqlv_at(sc, 0) == '.') { + sqlv_step(sc); + } + char e = sqlv_at(sc, 0); + char s1 = sqlv_at(sc, 1); + if ((e == 'e' || e == 'E') && + (isdigit((unsigned char)s1) || + ((s1 == '+' || s1 == '-') && isdigit((unsigned char)sqlv_at(sc, 2))))) { + sqlv_step(sc); + sqlv_step(sc); + while (isdigit((unsigned char)sqlv_at(sc, 0))) { + sqlv_step(sc); + } + } + } + return !sqlv_word_char(sqlv_at(sc, 0)) && sqlv_at(sc, 0) != '.'; +} + +/* A word in value position: NULL/TRUE/FALSE/DEFAULT, an X'..' / B'..' / N'..' + * prefixed string, or a _charset introducer before a string or number. */ +static bool sqlv_read_word_literal(sqlv_scan_t *sc) { + const char *w = sc->s + sc->i; + uint32_t n = 0; + while (sqlv_word_char(sqlv_at(sc, 0))) { + sqlv_step(sc); + n++; + } + if (sqlv_word_is(w, n, "NULL") || sqlv_word_is(w, n, "TRUE") || sqlv_word_is(w, n, "FALSE") || + sqlv_word_is(w, n, "DEFAULT")) { + return true; + } + if (n == 1 && strchr("xXbBnN", w[0]) && sqlv_at(sc, 0) == '\'') { + return sqlv_read_string(sc); + } + if (n > 1 && w[0] == '_') { + sqlv_skip_trivia(sc, NULL); + if (sqlv_at(sc, 0) == '\'') { + return sqlv_read_string(sc); + } + return isdigit((unsigned char)sqlv_at(sc, 0)) && sqlv_read_number(sc); + } + return false; +} + +/* Consume one value if it is a plain literal. Never consumes a parenthesis, so + * on false the caller can still find the tuple's end from the cursor. */ +static bool sqlv_read_literal(sqlv_scan_t *sc) { + char c = sqlv_at(sc, 0); + char n = sqlv_at(sc, 1); + if (c == '\'') { + return sqlv_read_string(sc); + } + if ((c == '-' || c == '+') && + (isdigit((unsigned char)n) || (n == '.' && isdigit((unsigned char)sqlv_at(sc, 2))))) { + sqlv_step(sc); + return sqlv_read_number(sc); + } + if (isdigit((unsigned char)c) || (c == '.' && isdigit((unsigned char)n))) { + return sqlv_read_number(sc); + } + if (isalpha((unsigned char)c) || c == '_') { + return sqlv_read_word_literal(sc); + } + return false; +} + +/* Consume one tuple starting at its '('. *literal says whether every value in + * it was a plain literal. Returns false if the tuple never closes. */ +static bool sqlv_scan_tuple(sqlv_scan_t *sc, bool *literal) { + sqlv_step(sc); + sqlv_skip_trivia(sc, NULL); + if (sqlv_at(sc, 0) == ')') { + sqlv_step(sc); + *literal = true; + return true; + } + while (sqlv_read_literal(sc)) { + sqlv_skip_trivia(sc, NULL); + char c = sqlv_at(sc, 0); + if (c == ')') { + sqlv_step(sc); + *literal = true; + return true; + } + if (c != ',') { + break; + } + sqlv_step(sc); + sqlv_skip_trivia(sc, NULL); + } + *literal = false; + return sqlv_skip_balanced(sc); +} + +static void sqlv_push(sqlv_ranges_t *r, TSRange range) { + if (r->failed) { + return; + } + if (r->count == r->cap) { + uint32_t cap = r->cap ? r->cap * SQLV_GROWTH : SQLV_INITIAL_CAP; + TSRange *grown = + (TSRange *)cbm_realloc(CBM_MEM_CLASS_EXTRACT, r->items, (size_t)cap * sizeof(TSRange)); + if (!grown) { + r->failed = true; + return; + } + r->items = grown; + r->cap = cap; + } + r->items[r->count++] = range; +} + +/* The cursor sits just past VALUES. Keep the first tuple; exclude each later + * literal-only tuple together with the comma before it. Runs of excluded tuples + * separated only by whitespace become one exclusion. */ +static void sqlv_handle_values(sqlv_scan_t *sc, sqlv_ranges_t *ex) { + sqlv_skip_trivia(sc, NULL); + bool literal = false; + if (sqlv_at(sc, 0) != '(' || !sqlv_scan_tuple(sc, &literal)) { + return; + } + bool pending = false; + TSRange cur = {0}; + for (;;) { + bool ws_only = true; + sqlv_skip_trivia(sc, &ws_only); + if (sqlv_at(sc, 0) != ',') { + break; + } + uint32_t comma = sc->i; + TSPoint comma_pt = sqlv_point(sc); + sqlv_step(sc); + sqlv_skip_trivia(sc, NULL); + if (sqlv_at(sc, 0) != '(' || !sqlv_scan_tuple(sc, &literal)) { + break; + } + if (!literal) { + if (pending) { + sqlv_push(ex, cur); + } + pending = false; + continue; + } + if (!pending || !ws_only) { + if (pending) { + sqlv_push(ex, cur); + } + cur.start_byte = comma; + cur.start_point = comma_pt; + pending = true; + } + cur.end_byte = sc->i; + cur.end_point = sqlv_point(sc); + } + if (pending) { + sqlv_push(ex, cur); + } +} + +typedef struct { + bool started; /* a token of the current statement has been seen */ + bool insert; /* the statement began with INSERT or REPLACE */ + int depth; +} sqlv_stmt_t; + +static void sqlv_on_word(sqlv_scan_t *sc, sqlv_stmt_t *st, sqlv_ranges_t *ex) { + const char *w = sc->s + sc->i; + uint32_t n = 0; + while (sc->i < sc->len && sqlv_word_char(sc->s[sc->i])) { + sqlv_step(sc); + n++; + } + if (!st->started) { + st->started = true; + st->insert = sqlv_word_is(w, n, "INSERT") || sqlv_word_is(w, n, "REPLACE"); + } else if (st->insert && st->depth == 0 && + (sqlv_word_is(w, n, "VALUES") || sqlv_word_is(w, n, "VALUE"))) { + sqlv_handle_values(sc, ex); + } +} + +static void sqlv_scan_file(sqlv_scan_t *sc, sqlv_ranges_t *ex) { + sqlv_stmt_t st = {false, false, 0}; + while (sc->i < sc->len && !ex->failed) { + if (sqlv_skip_comment(sc)) { + continue; + } + char c = sc->s[sc->i]; + if (sqlv_skip_any_quoted(sc)) { + st.started = true; + continue; + } + if (sqlv_word_char(c)) { + sqlv_on_word(sc, &st, ex); + continue; + } + if (c == ';') { + st = (sqlv_stmt_t){false, false, 0}; + } else if (c == '(') { + st.depth++; + } else if (c == ')' && st.depth > 0) { + st.depth--; + } + if (!isspace((unsigned char)c) && c != ';') { + st.started = true; + } + sqlv_step(sc); + } +} + +bool cbm_sql_values_kept_ranges(const char *src, uint32_t len, CBMSqlKeptRanges *out) { + out->items = NULL; + out->count = 0; + if (!src || len == 0) { + return false; + } + sqlv_scan_t sc = {src, len, 0, 0, 0}; + sqlv_ranges_t ex = {NULL, 0, 0, false}; + sqlv_scan_file(&sc, &ex); + if (ex.failed || ex.count == 0) { + cbm_free(CBM_MEM_CLASS_EXTRACT, ex.items); + return false; + } + /* Kept = the complement of the exclusions. The scan ended at EOF, so its + * cursor point is the end of the file. */ + TSRange *kept = + (TSRange *)cbm_alloc(CBM_MEM_CLASS_EXTRACT, ((size_t)ex.count + 1) * sizeof(TSRange)); + if (!kept) { + cbm_free(CBM_MEM_CLASS_EXTRACT, ex.items); + return false; + } + uint32_t n = 0; + uint32_t from = 0; + TSPoint from_pt = {0, 0}; + for (uint32_t k = 0; k <= ex.count; k++) { + uint32_t to = k < ex.count ? ex.items[k].start_byte : len; + TSPoint to_pt = k < ex.count ? ex.items[k].start_point : sqlv_point(&sc); + if (to > from) { + kept[n++] = (TSRange){from_pt, to_pt, from, to}; + } + if (k < ex.count) { + from = ex.items[k].end_byte; + from_pt = ex.items[k].end_point; + } + } + cbm_free(CBM_MEM_CLASS_EXTRACT, ex.items); + out->items = kept; + out->count = n; + return true; +} + +void cbm_sql_kept_ranges_free(CBMSqlKeptRanges *ranges) { + if (!ranges) { + return; + } + cbm_free(CBM_MEM_CLASS_EXTRACT, ranges->items); + ranges->items = NULL; + ranges->count = 0; +} diff --git a/internal/cbm/sql_values.h b/internal/cbm/sql_values.h new file mode 100644 index 0000000000..5960ec3201 --- /dev/null +++ b/internal/cbm/sql_values.h @@ -0,0 +1,40 @@ +/* sql_values.h — keep a SQL data dump's literal rows out of the parse (#1735). + * + * A mysqldump-style file is almost entirely `INSERT ... VALUES (..),(..),...` + * rows of plain literals. Those rows contribute nothing to the graph (no + * definition, call or usage lives in a literal), yet tree-sitter builds a full + * tree for every one of them: ~70 bytes of tree per source byte and seconds of + * parse per ten megabytes, until the per-file budget gives up and the whole file + * — its CREATE TABLE statements included — is skipped as "parse timeout". + * + * cbm_sql_values_kept_ranges() finds, in one linear quote- and comment-aware + * scan, every value tuple AFTER THE FIRST of an INSERT/REPLACE ... VALUES list + * that consists only of literals (strings, numbers, NULL, TRUE, FALSE, DEFAULT), + * and returns the complement as tree-sitter included ranges. The parser then + * sees `INSERT INTO t VALUES (first row);` — a complete statement — while every + * byte offset and line/column of the kept text is unchanged, so every node the + * extractors read still points at the real source. + * + * A syntax rule, not a work cap: the same file always yields the same ranges, + * whatever its size. A tuple holding anything that could reference code — a + * subquery, a function call, an identifier, a quoted identifier — is kept. */ +#ifndef CBM_SQL_VALUES_H +#define CBM_SQL_VALUES_H + +#include "tree_sitter/api.h" +#include +#include + +typedef struct { + TSRange *items; /* ascending, non-overlapping; owned (cbm_sql_kept_ranges_free) */ + uint32_t count; +} CBMSqlKeptRanges; + +/* Compute the included ranges for `src`. Returns true and fills `out` only when + * at least one literal tuple was excluded; returns false (out zeroed) when the + * whole file must be parsed as-is or on allocation failure. */ +bool cbm_sql_values_kept_ranges(const char *src, uint32_t len, CBMSqlKeptRanges *out); + +void cbm_sql_kept_ranges_free(CBMSqlKeptRanges *ranges); + +#endif /* CBM_SQL_VALUES_H */ diff --git a/tests/test_extraction.c b/tests/test_extraction.c index 1e6506ed1a..77c6b47561 100644 --- a/tests/test_extraction.c +++ b/tests/test_extraction.c @@ -302,6 +302,58 @@ TEST(extract_cpp_real_in_body_error_still_flagged_issue1071) { PASS(); } +/* #1735: the #1071 check's cost. cbm_subtract_macro_invocation_regions asked + * "is this region a macro call?" before "is it inside a function?", and the + * first question walked the source from byte 0 for every region: regions x + * file bytes on any file with many error regions, whatever its language. + * Counted (test seam), never timed. */ +enum { MACRO_SCAN_SRC_CAP = 8192 }; + +TEST(macro_check_reads_no_source_for_regions_outside_functions_issue1735) { + /* 40 functions, each followed by a top-level junk line: 40 regions, none + * of them inside a function. */ + char src[MACRO_SCAN_SRC_CAP]; + int len = snprintf(src, sizeof(src), "#define ALLOC(T, n) ((T *)malloc(sizeof(T) * (n)))\n"); + for (int i = 0; i < 40; i++) { + len += snprintf(src + len, sizeof(src) - (size_t)len, + "int ok%d(void) {\n return %d;\n}\n} ] junk ( {\n", i, i); + } + ASSERT_LT(len, MACRO_SCAN_SRC_CAP); + uint64_t before = cbm_test_macro_line_scan_bytes(); + CBMFileResult *r = extract(src, CBM_LANG_C, "t", "junk.c"); + uint64_t scanned = cbm_test_macro_line_scan_bytes() - before; + ASSERT_NOT_NULL(r); + ASSERT_TRUE(r->parse_incomplete); /* top-level junk stays reported... */ + ASSERT_EQ(r->error_region_count, 40); /* ...one range per junk line, as before */ + ASSERT_EQ(scanned, 0u); + cbm_free_result(r); + PASS(); +} + +TEST(macro_check_locates_lines_with_one_table_issue1735) { + /* 60 functions whose bodies hold a real syntax error: every region sits + * inside a function, so the check has to look at each one's lines. */ + char src[MACRO_SCAN_SRC_CAP]; + int len = snprintf(src, sizeof(src), "#define ALLOC(T, n) ((T *)malloc(sizeof(T) * (n)))\n"); + for (int i = 0; i < 60; i++) { + len += snprintf(src + len, sizeof(src) - (size_t)len, + "int f%d(void) {\n int x = ;\n return x;\n}\n", i); + } + ASSERT_LT(len, MACRO_SCAN_SRC_CAP); + uint64_t before = cbm_test_macro_line_scan_bytes(); + CBMFileResult *r = extract(src, CBM_LANG_C, "t", "inbody.c"); + uint64_t scanned = cbm_test_macro_line_scan_bytes() - before; + ASSERT_NOT_NULL(r); + /* #1071 unchanged: a real in-body error is not a macro call and stays + * reported, one range per function. */ + ASSERT_TRUE(r->parse_incomplete); + ASSERT_EQ(r->error_region_count, 60); + /* One line table for the file, not one walk per region. */ + ASSERT_LTE(scanned, (uint64_t)len); + cbm_free_result(r); + PASS(); +} + /* --- GDScript: AST -> graph visitor (Godot, #186) --- */ TEST(extract_gdscript_issue186) { CBMFileResult *r = extract("extends Node\n" @@ -8325,6 +8377,8 @@ SUITE(extraction) { RUN_TEST(extract_cpp_macros_issue375); RUN_TEST(extract_cpp_functionlike_macro_type_arg_no_false_parse_partial_issue1071); RUN_TEST(extract_cpp_real_in_body_error_still_flagged_issue1071); + RUN_TEST(macro_check_reads_no_source_for_regions_outside_functions_issue1735); + RUN_TEST(macro_check_locates_lines_with_one_table_issue1735); RUN_TEST(extract_gdscript_issue186); RUN_TEST(extract_powershell_issue35); RUN_TEST(extract_luau_issue39); diff --git a/tests/test_parse_coverage.c b/tests/test_parse_coverage.c index fd0e64e478..c8ed901140 100644 --- a/tests/test_parse_coverage.c +++ b/tests/test_parse_coverage.c @@ -30,6 +30,9 @@ #include "test_framework.h" #include "cbm.h" +#include "sql_values.h" /* #1735 value-row scanner */ +#include "foundation/compat.h" /* cbm_setenv / cbm_unsetenv */ +#include #include #include #include @@ -983,6 +986,457 @@ TEST(coverage_gap_of_only_comments_is_not_a_miss) { PASS(); } +/* ── #1735: SQL data dumps ─────────────────────────────────────────────────── + * A mysqldump file is a few CREATE TABLEs followed by megabytes of literal + * INSERT rows. Tree-sitter built a full tree for every row until the parse + * budget ran out, and the whole file — tables included — was skipped as + * "parse timeout". Literal-only rows after the first of each VALUES list are + * now kept out of the parse (sql_values.c). These tests pin that nothing the + * full parse extracted outside those rows is lost (and, for standard '' escapes + * the grammar reads correctly, that nothing changes at all), that a row which + * can reference something is still parsed, and that the parse no longer grows + * with the row count. */ + +typedef struct { + char *s; + size_t len; + size_t cap; +} cov_buf_t; + +static void cov_put(cov_buf_t *b, const char *fmt, ...) { + va_list ap; + va_start(ap, fmt); + va_list ap2; + va_copy(ap2, ap); + int n = vsnprintf(NULL, 0, fmt, ap); + va_end(ap); + if (n < 0) { + va_end(ap2); + return; + } + if (b->len + (size_t)n + 1 > b->cap) { + size_t cap = b->cap ? b->cap : 4096; + while (b->len + (size_t)n + 1 > cap) { + cap *= 2; + } + char *grown = realloc(b->s, cap); + if (!grown) { + va_end(ap2); + return; + } + b->s = grown; + b->cap = cap; + } + vsnprintf(b->s + b->len, (size_t)n + 1, fmt, ap2); + va_end(ap2); + b->len += (size_t)n; +} + +static const char *const DUMP_WORDS[] = {"Kabul", "Herat", "Amsterdam", "O'Brien", + "Haag", "Rotterdam", "Zuid-Holland", "Utrecht"}; +enum { DUMP_WORD_COUNT = sizeof(DUMP_WORDS) / sizeof(DUMP_WORDS[0]) }; +static const char *const DUMP_ODD[] = {"NULL", "-12.5", "0x1F", "_binary 'ab'", "DEFAULT", + "TRUE", "1e-3", "X'0A'", "''", "'a''b'"}; +enum { DUMP_ODD_COUNT = sizeof(DUMP_ODD) / sizeof(DUMP_ODD[0]) }; + +/* Deterministic mysqldump-shaped SQL: two tables and a view, then `stmts` + * extended INSERTs of `rows` tuples. `mysql_esc` writes a quote inside a string + * as \' (MySQL) instead of '' (standard). `mixed` puts one tuple holding a + * subquery and one holding a function call in the middle of every INSERT. The + * first two INSERTs separate their tuples with a newline and with a comment, the + * way pretty-printed dumps do. */ +static char *sql_dump(int stmts, int rows, bool mysql_esc, bool mixed) { + cov_buf_t b = {NULL, 0, 0}; + cov_put(&b, "-- MySQL dump 10.13\n" + "/*!40101 SET NAMES utf8mb4 */;\n" + "CREATE TABLE `city` (\n" + " `ID` int NOT NULL,\n" + " `Name` char(35) NOT NULL DEFAULT '',\n" + " `CountryCode` char(3),\n" + " `Note` text,\n" + " `Population` int\n" + ");\n" + "CREATE TABLE country (Code char(3), Name char(52));\n" + "CREATE VIEW big_cities AS SELECT Name FROM city WHERE Population > 1000000;\n" + "LOCK TABLES `city` WRITE;\n"); + uint32_t seed = 1735; + int id = 1; + for (int s = 0; s < stmts; s++) { + cov_put(&b, "INSERT INTO `city` VALUES "); + const char *sep = s == 0 ? ",\n" : (s == 1 ? ", /* split */ " : ","); + for (int r = 0; r < rows; r++) { + if (r > 0) { + cov_put(&b, "%s", sep); + } + if (mixed && r == rows / 2) { + cov_put(&b, "((SELECT MAX(Code) FROM country),'x','AAA',NULL,1)"); + continue; + } + if (mixed && r == rows / 2 + 1) { + cov_put(&b, "(%d,UPPER('y'),'BBB',NULL,2)", id++); + continue; + } + seed = seed * 1103515245u + 12345u; + const char *w = DUMP_WORDS[(seed >> 8) % DUMP_WORD_COUNT]; + cov_put(&b, "(%d,'", id++); + for (const char *p = w; *p; p++) { + cov_put(&b, *p == '\'' ? (mysql_esc ? "\\'" : "''") : "%c", *p); + } + cov_put(&b, "','%c%c%c',%s,%u)", 'A' + (int)(seed % 26), 'A' + (int)((seed >> 5) % 26), + 'A' + (int)((seed >> 10) % 26), DUMP_ODD[(seed >> 12) % DUMP_ODD_COUNT], + (seed >> 3) % 9000000u); + } + cov_put(&b, ";\n"); + } + cov_put(&b, "UNLOCK TABLES;\n"); + return b.s; +} + +static void fp_s(cov_buf_t *b, const char *s) { + cov_put(b, "%s|", s ? s : "~"); +} + +/* Everything the graph is built from except calls and usages, in order (those + * two are compared site by site, sites_agree). */ +static char *result_fingerprint(const CBMFileResult *r) { + cov_buf_t b = {NULL, 0, 0}; + for (int i = 0; i < r->defs.count; i++) { + const CBMDefinition *d = &r->defs.items[i]; + fp_s(&b, d->label); + fp_s(&b, d->name); + fp_s(&b, d->qualified_name); + fp_s(&b, d->signature); + fp_s(&b, d->return_type); + fp_s(&b, d->structural_profile); + fp_s(&b, d->body_tokens); + cov_put(&b, "%u-%u c%d l%d\n", d->start_line, d->end_line, d->complexity, d->lines); + } + for (int i = 0; i < r->imports.count; i++) { + fp_s(&b, r->imports.items[i].local_name); + fp_s(&b, r->imports.items[i].module_path); + cov_put(&b, "imp\n"); + } + for (int i = 0; i < r->rw.count; i++) { + fp_s(&b, r->rw.items[i].var_name); + cov_put(&b, "rw%d\n", (int)r->rw.items[i].is_write); + } + for (int i = 0; i < r->type_refs.count; i++) { + fp_s(&b, r->type_refs.items[i].type_name); + cov_put(&b, "tref\n"); + } + for (int i = 0; i < r->env_accesses.count; i++) { + fp_s(&b, r->env_accesses.items[i].env_key); + cov_put(&b, "env\n"); + } + for (int i = 0; i < r->throws.count; i++) { + fp_s(&b, r->throws.items[i].exception_name); + cov_put(&b, "throw\n"); + } + for (int i = 0; i < r->string_refs.count; i++) { + fp_s(&b, r->string_refs.items[i].value); + cov_put(&b, "sref%d\n", (int)r->string_refs.items[i].kind); + } + cov_put(&b, "ta%d it%d rc%d ib%d ch%d\n", r->type_assigns.count, r->impl_traits.count, + r->resolved_calls.count, r->infra_bindings.count, r->channels.count); + return b.s; +} + +static int has_usage_named(const CBMFileResult *r, const char *name) { + for (int i = 0; i < r->usages.count; i++) { + if (r->usages.items[i].ref_name && strstr(r->usages.items[i].ref_name, name)) { + return 1; + } + } + return 0; +} + +static int has_call_named(const CBMFileResult *r, const char *name) { + for (int i = 0; i < r->calls.count; i++) { + if (r->calls.items[i].callee_name && strstr(r->calls.items[i].callee_name, name)) { + return 1; + } + } + return 0; +} + +typedef struct { + const char *name; + uint32_t start; + uint32_t end; +} cov_site_t; + +static int collect_sites(const CBMFileResult *r, bool calls, cov_site_t **out) { + int n = calls ? r->calls.count : r->usages.count; + *out = calloc((size_t)(n > 0 ? n : 1), sizeof(cov_site_t)); + for (int i = 0; i < n && *out; i++) { + (*out)[i] = + calls ? (cov_site_t){r->calls.items[i].callee_name, r->calls.items[i].site_start_byte, + r->calls.items[i].site_end_byte} + : (cov_site_t){r->usages.items[i].ref_name, r->usages.items[i].site_start_byte, + r->usages.items[i].site_end_byte}; + } + return *out ? n : 0; +} + +static bool site_in(cov_site_t s, const cov_site_t *arr, int n) { + for (int i = 0; i < n; i++) { + if (arr[i].start == s.start && arr[i].end == s.end && s.name && arr[i].name && + strcmp(arr[i].name, s.name) == 0) { + return true; + } + } + return false; +} + +/* Calls (calls=true) or usages, cut parse against full parse: + * - nothing the full parse found outside the dropped rows may be missing; + * - with `exact`, the cut parse may find nothing the full parse did not. + * The full parse does find things INSIDE dropped rows: the SQL grammar reads + * DEFAULT and a _binary introducer as identifiers, and a MySQL \' escape it + * does not know turns the rest of a string into one. None of those names + * anything. Without `exact` (MySQL escapes) the cut parse may find more: the + * full parse's error recovery around a misread escape swallows neighbouring + * rows, a subquery row among them, and the cut parse no longer misreads them. */ +static bool sites_agree(const char *src, const CBMFileResult *cut, const CBMFileResult *full, + bool calls, bool exact) { + cov_site_t *cs = NULL; + cov_site_t *fs = NULL; + int cn = collect_sites(cut, calls, &cs); + int fn = collect_sites(full, calls, &fs); + CBMSqlKeptRanges k = {NULL, 0}; + (void)cbm_sql_values_kept_ranges(src, (uint32_t)strlen(src), &k); + bool ok = cs && fs; + for (int i = 0; ok && exact && i < cn; i++) { + if (!site_in(cs[i], fs, fn)) { + fprintf(stderr, " %s only in the cut parse\n", cs[i].name); + ok = false; + } + } + for (int i = 0; ok && i < fn; i++) { + if (site_in(fs[i], cs, cn)) { + continue; + } + for (uint32_t j = 0; j < k.count; j++) { + if (fs[i].start < k.items[j].end_byte && fs[i].end > k.items[j].start_byte) { + fprintf(stderr, " %s lost outside the dropped rows\n", fs[i].name); + ok = false; + } + } + } + cbm_sql_kept_ranges_free(&k); + free(cs); + free(fs); + return ok; +} + +static int count_calls_named(const CBMFileResult *r, const char *name) { + int n = 0; + for (int i = 0; i < r->calls.count; i++) { + if (r->calls.items[i].callee_name && strcmp(r->calls.items[i].callee_name, name) == 0) { + n++; + } + } + return n; +} + +/* The same source parsed with the row exclusion and without it (test seam). */ +static CBMFileResult *extract_sql_full(const char *src, const char *path) { + cbm_setenv("CBM_TEST_SQL_FULL_PARSE_ON", path, 1); + CBMFileResult *r = do_extract(src, CBM_LANG_SQL, path); + cbm_unsetenv("CBM_TEST_SQL_FULL_PARSE_ON"); + return r; +} + +TEST(sql_dump_literal_rows_leave_the_graph_unchanged_issue1735) { + for (int esc = 0; esc < 2; esc++) { + char *src = sql_dump(4, 60, esc == 1, true); + ASSERT_NOT_NULL(src); + CBMFileResult *cut = do_extract(src, CBM_LANG_SQL, "dump.sql"); + CBMFileResult *full = extract_sql_full(src, "dump.sql"); + ASSERT_NOT_NULL(cut); + ASSERT_NOT_NULL(full); + ASSERT_FALSE(cut->has_error); + ASSERT_TRUE(has_def(cut, "big_cities")); + char *fp_cut = result_fingerprint(cut); + char *fp_full = result_fingerprint(full); + ASSERT_NOT_NULL(fp_cut); + ASSERT_NOT_NULL(fp_full); + ASSERT_STR_EQ(fp_cut, fp_full); + ASSERT_TRUE(sites_agree(src, cut, full, true, esc == 0)); + ASSERT_TRUE(sites_agree(src, cut, full, false, esc == 0)); + /* Every INSERT's subquery row is parsed. */ + ASSERT_EQ(count_calls_named(cut, "MAX"), 4); + /* ...and the rows really were left out: under a quarter of the file + * is parsed. */ + CBMSqlKeptRanges k = {NULL, 0}; + ASSERT_TRUE(cbm_sql_values_kept_ranges(src, (uint32_t)strlen(src), &k)); + size_t kept = 0; + for (uint32_t j = 0; j < k.count; j++) { + kept += k.items[j].end_byte - k.items[j].start_byte; + } + cbm_sql_kept_ranges_free(&k); + ASSERT_LT(kept * 4, strlen(src)); + free(fp_cut); + free(fp_full); + cbm_free_result(cut); + cbm_free_result(full); + free(src); + } + PASS(); +} + +TEST(sql_dump_tuple_with_subquery_or_call_is_still_parsed_issue1735) { + char *src = sql_dump(1, 20, true, true); + ASSERT_NOT_NULL(src); + CBMFileResult *r = do_extract(src, CBM_LANG_SQL, "dump.sql"); + ASSERT_NOT_NULL(r); + ASSERT_TRUE(has_usage_named(r, "country")); /* from the subquery row */ + ASSERT_TRUE(has_call_named(r, "UPPER")); /* from the function-call row */ + cbm_free_result(r); + free(src); + PASS(); +} + +TEST(sql_dump_parse_does_not_grow_with_the_row_count_issue1735) { + /* A count, not a clock: with literal rows kept out, doubling them must not + * add a single tree node. */ + char *small = sql_dump(1, 500, true, false); + char *big = sql_dump(1, 1000, true, false); + ASSERT_NOT_NULL(small); + ASSERT_NOT_NULL(big); + CBMFileResult *rs = do_extract(small, CBM_LANG_SQL, "small.sql"); + CBMFileResult *rb = do_extract(big, CBM_LANG_SQL, "big.sql"); + ASSERT_NOT_NULL(rs); + ASSERT_NOT_NULL(rb); + ASSERT_EQ(rs->tree_nodes, rb->tree_nodes); + ASSERT_EQ(rs->defs.count, rb->defs.count); + cbm_free_result(rs); + cbm_free_result(rb); + free(small); + free(big); + PASS(); +} + +TEST(sql_dump_of_many_megabytes_is_indexed_not_timed_out_issue1735) { + /* ~26 MB with MySQL escapes. The production parse budget is CPU time, so + * extracting under it made the verdict a function of runner speed (an + * ASan arm leg tripped it). The property behind "indexed, not timed out" + * is a count: the parse work is set by the statements, not by the rows. + * So: extract unbudgeted, assert the outcome, and bound the tree by the + * same 330 statements carrying only a handful of rows each (the full + * parse built a tree node for every row token). */ + char *src = sql_dump(330, 2000, true, true); + char *few = sql_dump(330, 4, true, true); + ASSERT_NOT_NULL(src); + ASSERT_NOT_NULL(few); + size_t len = strlen(src); + ASSERT_GT(len, (size_t)24 * 1024 * 1024); + CBMFileResult *r = + cbm_extract_file(src, (int)len, CBM_LANG_SQL, "covproj", "world.sql", 0, NULL, NULL); + CBMFileResult *rf = + cbm_extract_file(few, (int)strlen(few), CBM_LANG_SQL, "covproj", "few.sql", 0, NULL, NULL); + free(src); + free(few); + ASSERT_NOT_NULL(r); + ASSERT_NOT_NULL(rf); + bool indexed = !r->has_error && has_def(r, "big_cities") && has_usage_named(r, "country"); + uint32_t nodes = r->tree_nodes; + uint32_t few_nodes = rf->tree_nodes; + cbm_free_result(r); + cbm_free_result(rf); + ASSERT_TRUE(indexed); + ASSERT_GT(few_nodes, 0); + ASSERT_LTE(nodes, 2 * few_nodes); + PASS(); +} + +/* The text the parser sees: every kept range, concatenated. */ +static char *kept_text(const char *src, const CBMSqlKeptRanges *k) { + cov_buf_t b = {NULL, 0, 0}; + cov_put(&b, "%s", ""); + for (uint32_t i = 0; i < k->count; i++) { + cov_put(&b, "%.*s", (int)(k->items[i].end_byte - k->items[i].start_byte), + src + k->items[i].start_byte); + } + return b.s; +} + +TEST(sql_values_scanner_excludes_only_literal_rows_issue1735) { + /* {source, what the parser sees} — NULL means nothing is excluded. */ + static const char *const cases[][2] = { + {"INSERT INTO t VALUES (1,'a'),(2,'b\\'c'),(3,NULL);", "INSERT INTO t VALUES (1,'a');"}, + {"INSERT INTO t VALUES (1),(f(2)),(3);", "INSERT INTO t VALUES (1),(f(2));"}, + {"insert into t values (1),(-2.5e3),(0x1F),(_binary 'x'),(DEFAULT),(true),(X'0A')," + "('it''s'),(.5),();", + "insert into t values (1);"}, + {"REPLACE INTO t VALUES (1), /* c */ (2), (3);", "REPLACE INTO t VALUES (1);"}, + {"INSERT INTO t VALUES (1),(2) -- c\n,(3);", "INSERT INTO t VALUES (1) -- c\n;"}, + {"INSERT INTO t VALUES (1),((SELECT id FROM u)),(x),(\"q\"),(`b`),(a.b);", NULL}, + {"INSERT INTO t VALUES (1),('a\nb'),(2);", "INSERT INTO t VALUES (1),('a\nb');"}, + {"INSERT INTO t VALUES (1),(2) ON DUPLICATE KEY UPDATE a=VALUES(a);", + "INSERT INTO t VALUES (1) ON DUPLICATE KEY UPDATE a=VALUES(a);"}, + {"INSERT INTO t (a, b) VALUES (1, 2), (3, 4);", "INSERT INTO t (a, b) VALUES (1, 2);"}, + {"SELECT 'INSERT INTO t VALUES (1),(2)';", NULL}, + {"-- INSERT INTO t VALUES (1),(2)\nSELECT 1;", NULL}, + {"/* INSERT INTO t VALUES (1),(2) */ SELECT 1;", NULL}, + {"SELECT * FROM (VALUES (1),(2)) AS v(x);", NULL}, + {"CREATE FUNCTION f() AS $$ SELECT 1; INSERT INTO t VALUES (1),(2); $$;", NULL}, + {"INSERT INTO t VALUES (1),(2", NULL}, + {"INSERT INTO t VALUES (1),(1abc),(2);", "INSERT INTO t VALUES (1),(1abc);"}, + }; + for (size_t c = 0; c < sizeof(cases) / sizeof(cases[0]); c++) { + const char *src = cases[c][0]; + CBMSqlKeptRanges k = {NULL, 0}; + bool cut = cbm_sql_values_kept_ranges(src, (uint32_t)strlen(src), &k); + if (!cases[c][1]) { + if (cut) { + fprintf(stderr, " case %zu excluded rows unexpectedly: %s\n", c, src); + } + ASSERT_FALSE(cut); + continue; + } + ASSERT_TRUE(cut); + char *seen = kept_text(src, &k); + ASSERT_NOT_NULL(seen); + ASSERT_STR_EQ(seen, cases[c][1]); + free(seen); + cbm_sql_kept_ranges_free(&k); + } + PASS(); +} + +TEST(sql_values_scanner_keeps_positions_of_kept_text_issue1735) { + const char *src = "INSERT INTO t VALUES\n(1),\n(2);\nCREATE TABLE u (id int);\n"; + CBMSqlKeptRanges k = {NULL, 0}; + ASSERT_TRUE(cbm_sql_values_kept_ranges(src, (uint32_t)strlen(src), &k)); + ASSERT_EQ(k.count, 2u); + /* excluded: ",\n(2)" from byte 24 (row 1, col 3) to byte 29 (row 2, col 3) */ + ASSERT_EQ(k.items[0].start_byte, 0u); + ASSERT_EQ(k.items[0].end_byte, 24u); + ASSERT_EQ(k.items[0].end_point.row, 1u); + ASSERT_EQ(k.items[0].end_point.column, 3u); + ASSERT_EQ(k.items[1].start_byte, 29u); + ASSERT_EQ(k.items[1].start_point.row, 2u); + ASSERT_EQ(k.items[1].start_point.column, 3u); + ASSERT_EQ(k.items[1].end_byte, (uint32_t)strlen(src)); + ASSERT_EQ(k.items[1].end_point.row, 4u); + ASSERT_EQ(k.items[1].end_point.column, 0u); + cbm_sql_kept_ranges_free(&k); + /* The table after the cut keeps its real line. */ + CBMFileResult *r = do_extract(src, CBM_LANG_SQL, "pos.sql"); + ASSERT_NOT_NULL(r); + bool found = false; + for (int i = 0; i < r->defs.count; i++) { + if (r->defs.items[i].name && strcmp(r->defs.items[i].name, "u") == 0) { + ASSERT_EQ(r->defs.items[i].start_line, 4u); + found = true; + } + } + ASSERT_TRUE(found); + cbm_free_result(r); + PASS(); +} + SUITE(parse_coverage) { RUN_TEST(c_ifdef_split_brace_sets_parse_incomplete); RUN_TEST(c_ifdef_split_brace_neighbors_still_extracted); @@ -1023,4 +1477,10 @@ SUITE(parse_coverage) { RUN_TEST(coverage_range_never_ends_past_the_last_line_issue963); RUN_TEST(coverage_range_never_covers_an_extracted_definition); RUN_TEST(coverage_gap_of_only_comments_is_not_a_miss); + RUN_TEST(sql_values_scanner_excludes_only_literal_rows_issue1735); + RUN_TEST(sql_values_scanner_keeps_positions_of_kept_text_issue1735); + RUN_TEST(sql_dump_literal_rows_leave_the_graph_unchanged_issue1735); + RUN_TEST(sql_dump_tuple_with_subquery_or_call_is_still_parsed_issue1735); + RUN_TEST(sql_dump_parse_does_not_grow_with_the_row_count_issue1735); + RUN_TEST(sql_dump_of_many_megabytes_is_indexed_not_timed_out_issue1735); } diff --git a/tests/test_pipeline.c b/tests/test_pipeline.c index 6ef16a89be..eb83f50034 100644 --- a/tests/test_pipeline.c +++ b/tests/test_pipeline.c @@ -344,6 +344,165 @@ TEST(pipeline_doclinks_edge_lands_in_store) { PASS(); } +/* #1735: every node and edge of a stored index, one sorted line each. */ +static void graph_listing_append(sqlite3 *db, const char *sql, const char *project, char **buf, + size_t *len) { + sqlite3_stmt *stmt = NULL; + if (sqlite3_prepare_v2(db, sql, -1, &stmt, NULL) != SQLITE_OK) { + return; + } + sqlite3_bind_text(stmt, 1, project, -1, SQLITE_TRANSIENT); + while (sqlite3_step(stmt) == SQLITE_ROW) { + const char *row = (const char *)sqlite3_column_text(stmt, 0); + size_t n = row ? strlen(row) : 0; + char *grown = realloc(*buf, *len + n + 2); + if (!grown) { + break; + } + *buf = grown; + memcpy(*buf + *len, row ? row : "", n); + (*buf)[*len + n] = '\n'; + *len += n + 1; + (*buf)[*len] = '\0'; + } + sqlite3_finalize(stmt); +} + +static char *graph_listing(const char *db_path, const char *project) { + sqlite3 *db = NULL; + if (sqlite3_open_v2(db_path, &db, SQLITE_OPEN_READONLY, NULL) != SQLITE_OK) { + sqlite3_close(db); + return NULL; + } + char *buf = calloc(1, 1); + size_t len = 0; + graph_listing_append(db, + "SELECT label||'|'||name||'|'||qualified_name||'|'||file_path||'|'||" + "start_line||'|'||end_line||'|'||properties FROM nodes " + "WHERE project=?1 ORDER BY 1", + project, &buf, &len); + graph_listing_append(db, + "SELECT e.type||'|'||s.qualified_name||'|'||t.qualified_name||'|'||" + "e.properties FROM edges e JOIN nodes s ON s.id=e.source_id " + "JOIN nodes t ON t.id=e.target_id WHERE e.project=?1 ORDER BY 1", + project, &buf, &len); + sqlite3_close(db); + return buf; +} + +static char *index_and_list(const char *repo, const char *db_name) { + char db_path[512]; + snprintf(db_path, sizeof(db_path), "%s/%s", repo, db_name); + cbm_pipeline_t *p = cbm_pipeline_new(repo, db_path, CBM_MODE_FULL); + if (!p) { + return NULL; + } + char *listing = + cbm_pipeline_run(p) == 0 ? graph_listing(db_path, cbm_pipeline_project_name(p)) : NULL; + cbm_pipeline_free(p); + return listing; +} + +/* True if every line of `sub` is a line of `super`. */ +static bool listing_lines_within(const char *sub, const char *super) { + for (const char *line = sub; *line;) { + const char *nl = strchr(line, '\n'); + size_t n = nl ? (size_t)(nl - line) : strlen(line); + bool found = false; + for (const char *p = super; *p && !found;) { + const char *pnl = strchr(p, '\n'); + size_t pn = pnl ? (size_t)(pnl - p) : strlen(p); + found = pn == n && memcmp(p, line, n) == 0; + p = pnl ? pnl + 1 : p + pn; + } + if (!found) { + fprintf(stderr, " missing from the cut index: %.*s\n", (int)n, line); + return false; + } + line = nl ? nl + 1 : line + n; + } + return true; +} + +/* A small dump: two tables, a view, three INSERTs of 200 rows, each with one + * subquery row and one function-call row among literal rows. */ +static char *sql_dump_small(bool mysql_esc) { + size_t cap = 256 * 1024; + char *src = malloc(cap); + if (!src) { + return NULL; + } + const char *quoted = mysql_esc ? "(%d,'O\\'Brien',DEFAULT)" : "(%d,'O''Brien',DEFAULT)"; + size_t len = (size_t)snprintf(src, cap, + "CREATE TABLE city (ID int, Name char(35), Population int);\n" + "CREATE TABLE country (Code char(3), Name char(52));\n" + "CREATE VIEW big_cities AS SELECT Name FROM city " + "WHERE Population > 1000000;\n"); + for (int s = 0; s < 3; s++) { + len += (size_t)snprintf(src + len, cap - len, "INSERT INTO `city` VALUES "); + for (int r = 0; r < 200; r++) { + if (r) { + src[len++] = ','; + } + int id = s * 1000 + r; + if (r == 100) { + len += (size_t)snprintf(src + len, cap - len, + "((SELECT MAX(Code) FROM country),'x',1)"); + } else if (r == 101) { + len += (size_t)snprintf(src + len, cap - len, "(%d,UPPER('y'),2)", id); + } else if (r % 3) { + len += (size_t)snprintf(src + len, cap - len, quoted, id); + } else { + len += (size_t)snprintf(src + len, cap - len, "(%d,_binary 'ab',NULL)", id); + } + } + len += (size_t)snprintf(src + len, cap - len, ";\n"); + } + return src; +} + +/* #1735: leaving a dump's literal INSERT rows out of the parse must not change + * the graph. The same repository is indexed twice — once as shipped and once + * with the full parse (test seam) — and the stored nodes and edges compared. + * Standard '' escapes, which the SQL grammar reads correctly: identical. + * MySQL \' escapes, which it misreads: the full parse's error recovery swallows + * neighbouring rows, so the cut index must hold everything the full one has + * (and may hold more — the rows the full parse lost). */ +TEST(pipeline_sql_dump_graph_matches_the_full_parse_issue1735) { + for (int esc = 0; esc < 2; esc++) { + char tmp[256]; + snprintf(tmp, sizeof(tmp), "/tmp/cbm_sql_dump_XXXXXX"); + ASSERT_NOT_NULL(cbm_mkdtemp(tmp)); + char *src = sql_dump_small(esc == 1); + ASSERT_NOT_NULL(src); + write_temp_file(tmp, "dump.sql", src); + free(src); + + char *cut = index_and_list(tmp, "cut.db"); + cbm_setenv("CBM_TEST_SQL_FULL_PARSE_ON", "dump.sql", 1); + char *full = index_and_list(tmp, "full.db"); + cbm_unsetenv("CBM_TEST_SQL_FULL_PARSE_ON"); + th_rmtree(tmp); + ASSERT_NOT_NULL(cut); + ASSERT_NOT_NULL(full); + /* Not vacuous: the tables, the view and its lineage are there. */ + ASSERT_NOT_NULL(strstr(cut, "Table|city|")); + ASSERT_NOT_NULL(strstr(cut, "View|big_cities|")); + ASSERT_NOT_NULL(strstr(cut, "USAGE|")); + if (esc == 0 && strcmp(cut, full) != 0) { + fprintf(stderr, "--- cut ---\n%s--- full ---\n%s", cut, full); + } + if (esc == 0) { + ASSERT_STR_EQ(cut, full); + } else { + ASSERT_TRUE(listing_lines_within(full, cut)); + } + free(cut); + free(full); + } + PASS(); +} + /* Spilling must be invisible in the OUTPUT: the same repository indexed with * results parked on disk must produce the same graph as one indexed entirely in * memory. It did not. The namespace map that `use`/`using`/package imports @@ -15110,6 +15269,7 @@ SUITE(pipeline) { /* Integration: structure pass */ RUN_TEST(pipeline_grpc_routes_cover_every_service_past_the_old_cap); RUN_TEST(pipeline_doclinks_edge_lands_in_store); + RUN_TEST(pipeline_sql_dump_graph_matches_the_full_parse_issue1735); RUN_TEST(pipeline_spill_resolves_namespace_imports_like_memory); RUN_TEST(pipeline_structure_nodes); RUN_TEST(pipeline_committed_counts_match_persisted);