Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
54 changes: 39 additions & 15 deletions internal/cbm/extract_usages.c
Original file line number Diff line number Diff line change
Expand Up @@ -622,14 +622,38 @@ static bool is_elixir_def_binding(CBMExtractCtx *ctx, TSNode node) {
return false;
}

static bool is_first_named_part_of(TSNode node, const char *container_kind) {
for (TSNode parent = ts_node_parent(node); !ts_node_is_null(parent);
parent = ts_node_parent(parent)) {
/* Climbs from `node` to its nearest `container_kind` ancestor and reports
* whether `node` sits in that ancestor's first named child. It runs for every
* named leaf of ReScript, Julia, Typst, Elm, PureScript and Nickel files, so it
* climbs the unified walk's cursor (O(1) per hop) whenever the walk is on
* `node`. ts_node_parent answers by descending from the ROOT and scanning the
* children at every level: on an error-heavy parse (a binary Godot `.res`
* resource yields a root with ~170k children per MB) each call is
* O(root children), and one climb per leaf made the file quadratic (#2176:
* 1 MB spent 46 s in this walk, 4.7 MB never finished). The node-based climb
* remains only for callers outside the walk. */
static bool is_first_named_part_of(TSNode node, const char *container_kind, WalkState *state) {
TSTreeCursor *cursor = reset_occurrence_cursor(state, node);
TSNode current = node;
for (;;) {
TSNode parent;
if (cursor) {
if (!ts_tree_cursor_goto_parent(cursor)) {
return false;
}
parent = ts_tree_cursor_current_node(cursor);
} else {
usage_slow_parent_fallback_test_note(); /* one root descent per hop */
parent = ts_node_parent(current);
if (ts_node_is_null(parent)) {
return false;
}
}
if (strcmp(ts_node_type(parent), container_kind) == 0) {
return named_child_contains(parent, 0, node);
}
current = parent;
}
return false;
}

static bool is_wolfram_lhs(TSNode node) {
Expand Down Expand Up @@ -1032,8 +1056,8 @@ static bool is_pkl_declaration_binding(TSNode node) {
return false;
}

static bool is_policy_binding(CBMExtractCtx *ctx, TSNode node,
const CBMOccurrenceSpec *occurrence) {
static bool is_policy_binding(CBMExtractCtx *ctx, TSNode node, const CBMOccurrenceSpec *occurrence,
WalkState *state) {
switch (occurrence->policy) {
case CBM_OCCURRENCE_LISP_DEF:
return is_lisp_def_binding(ctx, node);
Expand All @@ -1044,11 +1068,11 @@ static bool is_policy_binding(CBMExtractCtx *ctx, TSNode node,
case CBM_OCCURRENCE_ELIXIR_DEF:
return is_elixir_def_binding(ctx, node);
case CBM_OCCURRENCE_JULIA_FUNCTION:
return is_first_named_part_of(node, "function_definition");
return is_first_named_part_of(node, "function_definition", state);
case CBM_OCCURRENCE_WOLFRAM_SET:
return is_wolfram_lhs(node);
case CBM_OCCURRENCE_TYPST_LET:
return is_first_named_part_of(node, "let");
return is_first_named_part_of(node, "let", state);
case CBM_OCCURRENCE_AGDA_FUNCTION:
for (TSNode parent = ts_node_parent(node); !ts_node_is_null(parent);
parent = ts_node_parent(parent)) {
Expand All @@ -1064,14 +1088,14 @@ static bool is_policy_binding(CBMExtractCtx *ctx, TSNode node,
case CBM_OCCURRENCE_HCL_ATTRIBUTE:
return is_hcl_attribute_binding(node);
case CBM_OCCURRENCE_ELM_VALUE:
return is_first_named_part_of(node, "value_declaration");
return is_first_named_part_of(node, "value_declaration", state);
case CBM_OCCURRENCE_RESCRIPT_LET:
return is_first_named_part_of(node, "let_binding");
return is_first_named_part_of(node, "let_binding", state);
case CBM_OCCURRENCE_PURESCRIPT_LHS:
return is_first_named_part_of(node, "function");
return is_first_named_part_of(node, "function", state);
case CBM_OCCURRENCE_NICKEL_LET:
return is_first_named_part_of(node, "let_binding") ||
is_first_named_part_of(node, "pattern_fun");
return is_first_named_part_of(node, "let_binding", state) ||
is_first_named_part_of(node, "pattern_fun", state);
case CBM_OCCURRENCE_ERLANG_CLAUSE:
return is_erlang_clause_binding(node);
case CBM_OCCURRENCE_NIX_FUNCTION:
Expand Down Expand Up @@ -1119,7 +1143,7 @@ static bool is_binding_occurrence(CBMExtractCtx *ctx, TSNode node, const CBMLang
if (is_exact_language_binding(ctx, node, state)) {
return true;
}
if (is_policy_binding(ctx, node, occurrence)) {
if (is_policy_binding(ctx, node, occurrence, state)) {
return true;
}

Expand Down Expand Up @@ -2629,7 +2653,7 @@ void handle_usages(CBMExtractCtx *ctx, TSNode node, const CBMLangSpec *spec, Wal
bool possible_binding_leaf =
ts_node_is_named(node) && ts_node_named_child_count(node) == 0 &&
(state->inside_import || is_exact_language_binding(ctx, node, state) ||
is_policy_binding(ctx, node, occurrence));
is_policy_binding(ctx, node, occurrence, state));
if (!reference_node && !possible_binding_leaf) {
return;
}
Expand Down
1 change: 1 addition & 0 deletions internal/cbm/vendored/grammars/MANIFEST.md
Original file line number Diff line number Diff line change
Expand Up @@ -85,6 +85,7 @@ row instead.
|---|---|---|---|
| crystal | `crystal/scanner.c`, serialize | guard `memcpy(&buffer[offset], state->literals.contents, literal_content_size)` with `if (literal_content_size > 0)` | UBSan: zero-length `memcpy` with a NULL/0-size source on the empty-state serialize round-trip (formal UB, harmless) |
| rescript | `rescript/scanner.c`, deserialize | guard `memcpy(state, buffer, n_bytes)` with `if (n_bytes > 0)` | UBSan: zero-length `memcpy` with a NULL `buffer` / `n_bytes == 0` on empty-state deserialize (formal UB, harmless). The sibling serialize copies a fixed `sizeof(ScannerState)` (always > 0, non-NULL src) and needs no guard. |
| rescript | `rescript/scanner.c`, scan, `TEMPLATE_CHARS` branch | `if (valid_symbols[TEMPLATE_CHARS])` → `if (valid_symbols[TEMPLATE_CHARS] && !valid_symbols[NEWLINE])` | **Quadratic lexing in error recovery (#2176).** The only parse state whose external lex state enables both tokens is tree-sitter's ERROR state (`ts_lex_modes[0]`, external state 1; the in-template state 7 has no `_newline`), where every external token is marked valid and the lexer retries at each byte. From every byte the template loop ran to the next `` ` ``, `$`, `\` or NUL — the end of file when there is none — and returned false, so an unparseable stretch cost O(n²) inside lexing, where the parse budget's progress callback never runs. 50 KB of `0xFF` took 5.3 s, 100 KB 30 s; after: 3 ms, linear (lexer bytes pulled for 4 KB → 8 KB of `0xFF`: 8.8 M → 35.3 M before, 8.7 K → 17.4 K after, `complexity_rescript_error_recovery_lexing_is_linear`). Valid files parse to byte-identical trees; broken files recover more (a `let` after an unterminated expression is kept instead of swallowed). Upstream pin `43c2f1f35024` carries the unguarded branch |
| purescript | `purescript/scanner.c`, serialize | guard `memcpy(buffer, indents->data, to_copy)` with `if (to_copy > 0)` | UBSan: zero-length `memcpy` with a NULL/0-size source when the indent vector is empty (formal UB, harmless) |
| plsql | `plsql/parser.c`, include | `#include <tree_sitter/parser.h>` → `#include "tree_sitter/parser.h"` | The older ABI-14 generator emits angle brackets; every other vendored grammar uses the quoted form, which resolves the per-grammar `tree_sitter/` header from the including file's directory |
| properties | `properties/scanner.c`, whole file | move the `reached_eof` flag out of a file-scope `static bool` and into a per-parser payload allocated by `..._external_scanner_create()` | **Data race, and the only one of its kind in 103 vendored scanners.** That flag is how the scanner refuses to emit a second end-of-input `FAKE_EOL`, and upstream keeps it in ONE process-wide object. With several worker threads indexing `.properties` files at once, whichever reached EOF first set the flag, so another thread's `!reached_eof` was false and its parse never received the `FAKE_EOL` it needed to finish — it sat at end-of-input asking for a token that would never arrive. Measured 2026-09-19: ~106 M parse operations on a 253-byte file before the per-file budget stopped it, after which the file was dropped from the graph and java/kotlin indexes differed run to run. Interleaved A/B under identical load, 40 runs each: unpatched 6 stalls, patched 0. The serialize/deserialize wire format is unchanged (the state is still the returned length). Upstream `6310671b24d4` still carries the static, so a re-vendor must re-apply this |
Expand Down
7 changes: 6 additions & 1 deletion internal/cbm/vendored/grammars/rescript/scanner.c

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

4 changes: 2 additions & 2 deletions scripts/vendored-checksums.txt
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,7 @@ c5cfb43042b6b72045f4ba997834d0a7786d2793d91680868b5815b39f14fc78 internal/cbm/v
b29c1c9fb7cc82f58c84b376df1297d6e2737a1d655fd356db0859e3c29c2fea internal/cbm/vendored/common/tree_sitter/alloc.h
31e60a1bff6f715afacce03b5b70efe42b58371b4f9595dd4af52a577ff9608c internal/cbm/vendored/common/tree_sitter/array.h
180b893c8734778fd32f372dfbc27bd6ad1cd2221f26150b31256ff6716320d2 internal/cbm/vendored/common/tree_sitter/parser.h
a66a04ed735fdb3d4a90ee4353b8f0490dc0360da5965b0d13ec5d6766f146fa internal/cbm/vendored/grammars/MANIFEST.md
9fe439eff24b2a474618f2cb75cbeb06074840f0396fbeb1d2503e668a5d056a internal/cbm/vendored/grammars/MANIFEST.md
ad8425038de519f8c4e3e9339feebf99dfad8a6002dcf79227d91402780a32cc internal/cbm/vendored/grammars/ada/LICENSE
02805ec13939b749c891567be36bf024b09034e04c80683a1ae667272458d549 internal/cbm/vendored/grammars/ada/parser.c
115a75d000bef9c70c4de7dfbb7f2a80b90cbb14ba265056dc7abb1ef3b9b6db internal/cbm/vendored/grammars/ada/tree_sitter/alloc.h
Expand Down Expand Up @@ -667,7 +667,7 @@ c4b9482069f61a2a26a590181baa34f08eb6dcc57a138def578e6f9720c35ea5 internal/cbm/v
c4b9482069f61a2a26a590181baa34f08eb6dcc57a138def578e6f9720c35ea5 internal/cbm/vendored/grammars/requirements/tree_sitter/parser.h
21da9b4e463af4b4501cbd4e0c5886a1f321a740b584002a9c9a5fe7b2d46a8a internal/cbm/vendored/grammars/rescript/LICENSE
d75538517e1024056c206f6d1b10cb4ed83670a5d6fb338dbad7b2bb4269403a internal/cbm/vendored/grammars/rescript/parser.c
4bf614fc0b599bf75414c1512e92c971d4bb04d0334e5f562ff34a0dbd745f8a internal/cbm/vendored/grammars/rescript/scanner.c
d550ab7d815666f14b938e805f106a2ce4d99f6f86377d7a301dcf076db5f138 internal/cbm/vendored/grammars/rescript/scanner.c
115a75d000bef9c70c4de7dfbb7f2a80b90cbb14ba265056dc7abb1ef3b9b6db internal/cbm/vendored/grammars/rescript/tree_sitter/alloc.h
02d139d44303b0b0b85a899839dbc1346c478e7318d1f2971fad2b38fd57cb73 internal/cbm/vendored/grammars/rescript/tree_sitter/array.h
c4b9482069f61a2a26a590181baa34f08eb6dcc57a138def578e6f9720c35ea5 internal/cbm/vendored/grammars/rescript/tree_sitter/parser.h
Expand Down
4 changes: 4 additions & 0 deletions src/discover/discover.c
Original file line number Diff line number Diff line change
Expand Up @@ -799,6 +799,10 @@ static CBMLanguage detect_file_language(const char *entry_name, const char *abs_
if (dot && strcmp(dot, ".frm") == 0) {
lang = cbm_disambiguate_frm(abs_path);
}
/* Special: .res is also a binary Godot / Windows resource (#2176) */
if (dot && strcmp(dot, ".res") == 0) {
lang = cbm_disambiguate_res(abs_path);
}
/* Special: ObjectScript Studio Export XML (<Export generator="...">) is
* detected by content; otherwise .xml stays XML. */
if (lang == CBM_LANG_XML) {
Expand Down
6 changes: 6 additions & 0 deletions src/discover/discover.h
Original file line number Diff line number Diff line change
Expand Up @@ -51,6 +51,12 @@ CBMLanguage cbm_disambiguate_cls(const char *path);
* CBM_LANG_FORM. On read failure, defaults to CBM_LANG_FORM. */
CBMLanguage cbm_disambiguate_frm(const char *path);

/* Disambiguate .res files by reading first 4KB of content (#2176).
* Returns CBM_LANG_COUNT (not indexed) for binary content -- a Godot resource
* or a Windows compiled resource file, recognised by a NUL byte -- otherwise
* CBM_LANG_RESCRIPT. On read failure, defaults to CBM_LANG_RESCRIPT. */
CBMLanguage cbm_disambiguate_res(const char *path);

/* Disambiguate .inc files by reading first 4KB of content.
* Returns CBM_LANG_OBJECTSCRIPT_ROUTINE if it looks like an ObjectScript
* include (a "ROUTINE <Uppercase>" header), otherwise CBM_LANG_BITBAKE.
Expand Down
21 changes: 21 additions & 0 deletions src/discover/language.c
Original file line number Diff line number Diff line change
Expand Up @@ -1309,6 +1309,27 @@ CBMLanguage cbm_disambiguate_frm(const char *path) {
return has_vb6_markers(buf) ? CBM_LANG_COUNT : CBM_LANG_FORM;
}

/* Disambiguate .res files (#2176): ReScript source shares the extension with
* two binary formats -- Godot resources (RSRC/RSCC magic, the default save
* format for imported meshes) and Windows compiled resource files. Binary
* content has no ReScript meaning, and a NUL byte never occurs in ReScript
* text while both binary formats carry NULs in their first bytes. */
CBMLanguage cbm_disambiguate_res(const char *path) {
if (!path) {
return CBM_LANG_RESCRIPT;
}

FILE *f = cbm_fopen(path, "rb");
if (!f) {
return CBM_LANG_RESCRIPT;
}

char buf[CBM_SZ_4K];
size_t n = fread(buf, SKIP_ONE, sizeof(buf), f);
(void)fclose(f);
return memchr(buf, '\0', n) ? CBM_LANG_COUNT : CBM_LANG_RESCRIPT;
}

/* Disambiguate .cls files: shared by InterSystems ObjectScript UDL, Salesforce
* Apex and Visual Basic 6 class modules (#721). ObjectScript class files begin
* with a line of the form "Class <UppercasePackage>..."; VB6 class modules
Expand Down
77 changes: 77 additions & 0 deletions tests/test_complexity.c
Original file line number Diff line number Diff line change
Expand Up @@ -49,6 +49,8 @@
#include "../src/foundation/profile.h"
#include "cbm.h"
#include "discover/discover.h"
#include "lang_specs.h"
#include "tree_sitter/api.h"
#include "pipeline/pass_lsp_cross.h"
#include "pipeline/pipeline.h"
#include "pipeline/pipeline_internal.h"
Expand Down Expand Up @@ -760,7 +762,82 @@ TEST(complexity_importance_scoring_is_linear) {
PASS();
}

/* ── Lexer work in error recovery (#2176) ──────────────────────────────
* tree-sitter lexes an unparseable stretch by retrying at every byte with
* every external token marked valid. The ReScript scanner then ran its
* template-string loop from each byte to the next '`', '$', '\\' or NUL — to
* the end of the file when there is none — and threw the result away. That
* is O(stretch) per byte, O(n^2) per file, all inside lexing where the parse
* budget's progress callback never runs; a binary Godot `.res` of high bytes
* (or plain text such as a run of '~') was dropped by the clock instead of
* parsed. Work is counted as the bytes the lexer pulls through a chunked
* TSInput: a pure function of (grammar, input), independent of speed. */
enum { CX_LEX_CHUNK = 64, CX_LEX_BASE_BYTES = 4096 };

typedef struct {
const char *src;
uint32_t len;
uint64_t bytes_pulled;
} CxLexInput;

static const char *cx_lex_read(void *payload, uint32_t byte_index, TSPoint position,
uint32_t *bytes_read) {
(void)position;
CxLexInput *in = (CxLexInput *)payload;
if (byte_index >= in->len) {
*bytes_read = 0;
return "";
}
uint32_t n = in->len - byte_index;
if (n > CX_LEX_CHUNK) {
n = CX_LEX_CHUNK;
}
in->bytes_pulled += n;
*bytes_read = n;
return in->src + byte_index;
}

/* Bytes pulled while parsing `len` copies of `fill` as ReScript; 0 on failure. */
static uint64_t cx_rescript_lex_work(unsigned char fill, uint32_t len) {
char *src = malloc(len);
TSParser *parser = ts_parser_new();
uint64_t work = 0;
if (src && parser && ts_parser_set_language(parser, cbm_ts_language(CBM_LANG_RESCRIPT))) {
memset(src, fill, len);
CxLexInput in = {src, len, 0};
TSInput input = {&in, cx_lex_read, TSInputEncodingUTF8, NULL};
TSTree *tree = ts_parser_parse(parser, NULL, input);
if (tree) {
work = in.bytes_pulled;
ts_tree_delete(tree);
}
}
if (parser) {
ts_parser_delete(parser);
}
free(src);
return work;
}

TEST(complexity_rescript_error_recovery_lexing_is_linear) {
/* 0xFF: the reporter's binary bytes (invalid UTF-8). '~': plain ASCII text
* that ReScript cannot parse either — the defect is not binary-only. */
static const unsigned char fills[] = {0xFF, '~'};
for (size_t i = 0; i < sizeof(fills); i++) {
uint64_t base = cx_rescript_lex_work(fills[i], CX_LEX_BASE_BYTES);
uint64_t doubled = cx_rescript_lex_work(fills[i], 2 * CX_LEX_BASE_BYTES);
double r = cx_ratio((double)doubled, (double)base);
printf(" fill 0x%02x: lexer bytes %llu -> %llu ratio %.2f (linear ~2, quadratic ~4)\n",
fills[i], (unsigned long long)base, (unsigned long long)doubled, r);
/* Non-vacuous: the lexer must at least have read the input once. */
ASSERT_GTE(base, (uint64_t)CX_LEX_BASE_BYTES);
ASSERT_TRUE(r >= CX_RATIO_LO && r <= CX_RATIO_HI);
}
PASS();
}

SUITE(complexity) {
RUN_TEST(complexity_rescript_error_recovery_lexing_is_linear);
RUN_TEST(complexity_replicated_modules_scale_linearly);
RUN_TEST(complexity_perfile_registry_work_is_linear);
RUN_TEST(complexity_importance_scoring_is_linear);
Expand Down
Loading
Loading