From 8093d10b812d9a7840ae99d8f9d24a04cb0f9559 Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Fri, 25 Sep 2026 22:24:01 +0200 Subject: [PATCH 1/2] fix(rescript): stop binary .res resources hanging the indexer (#2176) A 4.7 MB binary Godot resource (piko_walk_mesh.res) never finished indexing: `.res` maps to ReScript, so the resource was parsed as ReScript source and the definitions pass spun. Three defects, each fixed: 1. Quadratic usage walk (the hang). The ReScript `let` binding check (is_first_named_part_of, shared with Julia, Typst, Elm, PureScript and Nickel) climbed from every named leaf with ts_node_parent, which descends from the root and scans each level's children. A binary file parses into a flat error tree (~170k root children per MB), so each climb cost O(root children) and the file O(n^2): 1 MB spent 46 s in cbm_extract_unified, 4.7 MB never returned. The check now climbs the unified walk's own cursor (O(1) per hop), like the other occurrence classifiers; the node-based climb stays only for callers outside the walk. 1 MB: 46 s -> 0.03 s; 4.7 MB: 2.2 s end to end. 2. Quadratic lexing in error recovery. tree-sitter retries the lexer at every byte of an unparseable stretch with every external token valid; the ReScript scanner's template-string loop then ran to the next backtick/$/backslash/NUL (end of file if none) and discarded the result. That cost O(n^2) inside lexing, where the parse budget's progress callback never runs, so such files were dropped by the 5 s CPU budget instead of parsed -- plain text too (100 KB of non-ReScript text: 15 s -> 0.18 s). Local scanner patch: skip the template branch when NEWLINE is also valid, which only the ERROR state allows. Valid files parse to identical trees. Recorded in vendored/grammars/MANIFEST.md. 3. Binary content parsed as source. `.res` is also Godot's binary resource format and the Windows compiled-resource format. Following the .m/.cls/.frm/.inc/.cfc precedent, cbm_disambiguate_res reads the first 4 KB and reports a file holding a NUL byte as unsupported; ReScript text never contains one. Tests (deterministic work counters, no clocks): - extraction::extract_rescript_let_bindings_use_the_walk_cursor -- zero root-descending parent hops at 128 and 1024 statements (before: 515 / 4099). - complexity::complexity_rescript_error_recovery_lexing_is_linear -- lexer bytes pulled through a chunked TSInput, 4 KB vs 8 KB, ratio must be linear (before 3.99, after 2.00). - language::lang_res_binary_resource_unsupported and lang_res_rescript_stays_rescript. Fixes #2176 Signed-off-by: Martin Vogel --- internal/cbm/extract_usages.c | 54 +++++++++---- internal/cbm/vendored/grammars/MANIFEST.md | 1 + .../cbm/vendored/grammars/rescript/scanner.c | 7 +- src/discover/discover.c | 4 + src/discover/discover.h | 6 ++ src/discover/language.c | 21 +++++ tests/test_complexity.c | 77 +++++++++++++++++++ tests/test_extraction.c | 68 ++++++++++++++++ tests/test_language.c | 32 ++++++++ 9 files changed, 254 insertions(+), 16 deletions(-) diff --git a/internal/cbm/extract_usages.c b/internal/cbm/extract_usages.c index f6b3913151..bdad1c8c7c 100644 --- a/internal/cbm/extract_usages.c +++ b/internal/cbm/extract_usages.c @@ -622,14 +622,38 @@ static bool is_elixir_def_binding(CBMExtractCtx *ctx, TSNode node) { return false; } -static bool is_first_named_part_of(TSNode node, const char *container_kind) { - for (TSNode parent = ts_node_parent(node); !ts_node_is_null(parent); - parent = ts_node_parent(parent)) { +/* Climbs from `node` to its nearest `container_kind` ancestor and reports + * whether `node` sits in that ancestor's first named child. It runs for every + * named leaf of ReScript, Julia, Typst, Elm, PureScript and Nickel files, so it + * climbs the unified walk's cursor (O(1) per hop) whenever the walk is on + * `node`. ts_node_parent answers by descending from the ROOT and scanning the + * children at every level: on an error-heavy parse (a binary Godot `.res` + * resource yields a root with ~170k children per MB) each call is + * O(root children), and one climb per leaf made the file quadratic (#2176: + * 1 MB spent 46 s in this walk, 4.7 MB never finished). The node-based climb + * remains only for callers outside the walk. */ +static bool is_first_named_part_of(TSNode node, const char *container_kind, WalkState *state) { + TSTreeCursor *cursor = reset_occurrence_cursor(state, node); + TSNode current = node; + for (;;) { + TSNode parent; + if (cursor) { + if (!ts_tree_cursor_goto_parent(cursor)) { + return false; + } + parent = ts_tree_cursor_current_node(cursor); + } else { + usage_slow_parent_fallback_test_note(); /* one root descent per hop */ + parent = ts_node_parent(current); + if (ts_node_is_null(parent)) { + return false; + } + } if (strcmp(ts_node_type(parent), container_kind) == 0) { return named_child_contains(parent, 0, node); } + current = parent; } - return false; } static bool is_wolfram_lhs(TSNode node) { @@ -1032,8 +1056,8 @@ static bool is_pkl_declaration_binding(TSNode node) { return false; } -static bool is_policy_binding(CBMExtractCtx *ctx, TSNode node, - const CBMOccurrenceSpec *occurrence) { +static bool is_policy_binding(CBMExtractCtx *ctx, TSNode node, const CBMOccurrenceSpec *occurrence, + WalkState *state) { switch (occurrence->policy) { case CBM_OCCURRENCE_LISP_DEF: return is_lisp_def_binding(ctx, node); @@ -1044,11 +1068,11 @@ static bool is_policy_binding(CBMExtractCtx *ctx, TSNode node, case CBM_OCCURRENCE_ELIXIR_DEF: return is_elixir_def_binding(ctx, node); case CBM_OCCURRENCE_JULIA_FUNCTION: - return is_first_named_part_of(node, "function_definition"); + return is_first_named_part_of(node, "function_definition", state); case CBM_OCCURRENCE_WOLFRAM_SET: return is_wolfram_lhs(node); case CBM_OCCURRENCE_TYPST_LET: - return is_first_named_part_of(node, "let"); + return is_first_named_part_of(node, "let", state); case CBM_OCCURRENCE_AGDA_FUNCTION: for (TSNode parent = ts_node_parent(node); !ts_node_is_null(parent); parent = ts_node_parent(parent)) { @@ -1064,14 +1088,14 @@ static bool is_policy_binding(CBMExtractCtx *ctx, TSNode node, case CBM_OCCURRENCE_HCL_ATTRIBUTE: return is_hcl_attribute_binding(node); case CBM_OCCURRENCE_ELM_VALUE: - return is_first_named_part_of(node, "value_declaration"); + return is_first_named_part_of(node, "value_declaration", state); case CBM_OCCURRENCE_RESCRIPT_LET: - return is_first_named_part_of(node, "let_binding"); + return is_first_named_part_of(node, "let_binding", state); case CBM_OCCURRENCE_PURESCRIPT_LHS: - return is_first_named_part_of(node, "function"); + return is_first_named_part_of(node, "function", state); case CBM_OCCURRENCE_NICKEL_LET: - return is_first_named_part_of(node, "let_binding") || - is_first_named_part_of(node, "pattern_fun"); + return is_first_named_part_of(node, "let_binding", state) || + is_first_named_part_of(node, "pattern_fun", state); case CBM_OCCURRENCE_ERLANG_CLAUSE: return is_erlang_clause_binding(node); case CBM_OCCURRENCE_NIX_FUNCTION: @@ -1119,7 +1143,7 @@ static bool is_binding_occurrence(CBMExtractCtx *ctx, TSNode node, const CBMLang if (is_exact_language_binding(ctx, node, state)) { return true; } - if (is_policy_binding(ctx, node, occurrence)) { + if (is_policy_binding(ctx, node, occurrence, state)) { return true; } @@ -2629,7 +2653,7 @@ void handle_usages(CBMExtractCtx *ctx, TSNode node, const CBMLangSpec *spec, Wal bool possible_binding_leaf = ts_node_is_named(node) && ts_node_named_child_count(node) == 0 && (state->inside_import || is_exact_language_binding(ctx, node, state) || - is_policy_binding(ctx, node, occurrence)); + is_policy_binding(ctx, node, occurrence, state)); if (!reference_node && !possible_binding_leaf) { return; } diff --git a/internal/cbm/vendored/grammars/MANIFEST.md b/internal/cbm/vendored/grammars/MANIFEST.md index 74322763e8..9bedf192e2 100644 --- a/internal/cbm/vendored/grammars/MANIFEST.md +++ b/internal/cbm/vendored/grammars/MANIFEST.md @@ -85,6 +85,7 @@ row instead. |---|---|---|---| | crystal | `crystal/scanner.c`, serialize | guard `memcpy(&buffer[offset], state->literals.contents, literal_content_size)` with `if (literal_content_size > 0)` | UBSan: zero-length `memcpy` with a NULL/0-size source on the empty-state serialize round-trip (formal UB, harmless) | | rescript | `rescript/scanner.c`, deserialize | guard `memcpy(state, buffer, n_bytes)` with `if (n_bytes > 0)` | UBSan: zero-length `memcpy` with a NULL `buffer` / `n_bytes == 0` on empty-state deserialize (formal UB, harmless). The sibling serialize copies a fixed `sizeof(ScannerState)` (always > 0, non-NULL src) and needs no guard. | +| rescript | `rescript/scanner.c`, scan, `TEMPLATE_CHARS` branch | `if (valid_symbols[TEMPLATE_CHARS])` → `if (valid_symbols[TEMPLATE_CHARS] && !valid_symbols[NEWLINE])` | **Quadratic lexing in error recovery (#2176).** The only parse state whose external lex state enables both tokens is tree-sitter's ERROR state (`ts_lex_modes[0]`, external state 1; the in-template state 7 has no `_newline`), where every external token is marked valid and the lexer retries at each byte. From every byte the template loop ran to the next `` ` ``, `$`, `\` or NUL — the end of file when there is none — and returned false, so an unparseable stretch cost O(n²) inside lexing, where the parse budget's progress callback never runs. 50 KB of `0xFF` took 5.3 s, 100 KB 30 s; after: 3 ms, linear (lexer bytes pulled for 4 KB → 8 KB of `0xFF`: 8.8 M → 35.3 M before, 8.7 K → 17.4 K after, `complexity_rescript_error_recovery_lexing_is_linear`). Valid files parse to byte-identical trees; broken files recover more (a `let` after an unterminated expression is kept instead of swallowed). Upstream pin `43c2f1f35024` carries the unguarded branch | | purescript | `purescript/scanner.c`, serialize | guard `memcpy(buffer, indents->data, to_copy)` with `if (to_copy > 0)` | UBSan: zero-length `memcpy` with a NULL/0-size source when the indent vector is empty (formal UB, harmless) | | plsql | `plsql/parser.c`, include | `#include ` → `#include "tree_sitter/parser.h"` | The older ABI-14 generator emits angle brackets; every other vendored grammar uses the quoted form, which resolves the per-grammar `tree_sitter/` header from the including file's directory | | properties | `properties/scanner.c`, whole file | move the `reached_eof` flag out of a file-scope `static bool` and into a per-parser payload allocated by `..._external_scanner_create()` | **Data race, and the only one of its kind in 103 vendored scanners.** That flag is how the scanner refuses to emit a second end-of-input `FAKE_EOL`, and upstream keeps it in ONE process-wide object. With several worker threads indexing `.properties` files at once, whichever reached EOF first set the flag, so another thread's `!reached_eof` was false and its parse never received the `FAKE_EOL` it needed to finish — it sat at end-of-input asking for a token that would never arrive. Measured 2026-09-19: ~106 M parse operations on a 253-byte file before the per-file budget stopped it, after which the file was dropped from the graph and java/kotlin indexes differed run to run. Interleaved A/B under identical load, 40 runs each: unpatched 6 stalls, patched 0. The serialize/deserialize wire format is unchanged (the state is still the returned length). Upstream `6310671b24d4` still carries the static, so a re-vendor must re-apply this | diff --git a/internal/cbm/vendored/grammars/rescript/scanner.c b/internal/cbm/vendored/grammars/rescript/scanner.c index 171d4b6285..fb7aecfae6 100644 --- a/internal/cbm/vendored/grammars/rescript/scanner.c +++ b/internal/cbm/vendored/grammars/rescript/scanner.c @@ -159,7 +159,12 @@ bool tree_sitter_rescript_external_scanner_scan( skip(lexer); } - if (valid_symbols[TEMPLATE_CHARS]) { + // cbm local patch (#2176, see MANIFEST.md): NEWLINE is never valid inside + // a template string, so both valid at once means error recovery, where + // tree-sitter marks every external token valid and retries at each byte. + // Scanning template chars there ran to the next '`', '$', '\\' or NUL (the + // end of file when there is none) from every byte: O(n^2) per file. + if (valid_symbols[TEMPLATE_CHARS] && !valid_symbols[NEWLINE]) { lexer->result_symbol = TEMPLATE_CHARS; for (bool has_content = false;; has_content = true) { lexer->mark_end(lexer); diff --git a/src/discover/discover.c b/src/discover/discover.c index fd4cea2720..0e5ae5e4f9 100644 --- a/src/discover/discover.c +++ b/src/discover/discover.c @@ -799,6 +799,10 @@ static CBMLanguage detect_file_language(const char *entry_name, const char *abs_ if (dot && strcmp(dot, ".frm") == 0) { lang = cbm_disambiguate_frm(abs_path); } + /* Special: .res is also a binary Godot / Windows resource (#2176) */ + if (dot && strcmp(dot, ".res") == 0) { + lang = cbm_disambiguate_res(abs_path); + } /* Special: ObjectScript Studio Export XML () is * detected by content; otherwise .xml stays XML. */ if (lang == CBM_LANG_XML) { diff --git a/src/discover/discover.h b/src/discover/discover.h index fb8d7b5e6a..89ccca4f36 100644 --- a/src/discover/discover.h +++ b/src/discover/discover.h @@ -51,6 +51,12 @@ CBMLanguage cbm_disambiguate_cls(const char *path); * CBM_LANG_FORM. On read failure, defaults to CBM_LANG_FORM. */ CBMLanguage cbm_disambiguate_frm(const char *path); +/* Disambiguate .res files by reading first 4KB of content (#2176). + * Returns CBM_LANG_COUNT (not indexed) for binary content -- a Godot resource + * or a Windows compiled resource file, recognised by a NUL byte -- otherwise + * CBM_LANG_RESCRIPT. On read failure, defaults to CBM_LANG_RESCRIPT. */ +CBMLanguage cbm_disambiguate_res(const char *path); + /* Disambiguate .inc files by reading first 4KB of content. * Returns CBM_LANG_OBJECTSCRIPT_ROUTINE if it looks like an ObjectScript * include (a "ROUTINE " header), otherwise CBM_LANG_BITBAKE. diff --git a/src/discover/language.c b/src/discover/language.c index c9d92719aa..dcc41dafd0 100644 --- a/src/discover/language.c +++ b/src/discover/language.c @@ -1309,6 +1309,27 @@ CBMLanguage cbm_disambiguate_frm(const char *path) { return has_vb6_markers(buf) ? CBM_LANG_COUNT : CBM_LANG_FORM; } +/* Disambiguate .res files (#2176): ReScript source shares the extension with + * two binary formats -- Godot resources (RSRC/RSCC magic, the default save + * format for imported meshes) and Windows compiled resource files. Binary + * content has no ReScript meaning, and a NUL byte never occurs in ReScript + * text while both binary formats carry NULs in their first bytes. */ +CBMLanguage cbm_disambiguate_res(const char *path) { + if (!path) { + return CBM_LANG_RESCRIPT; + } + + FILE *f = cbm_fopen(path, "rb"); + if (!f) { + return CBM_LANG_RESCRIPT; + } + + char buf[CBM_SZ_4K]; + size_t n = fread(buf, SKIP_ONE, sizeof(buf), f); + (void)fclose(f); + return memchr(buf, '\0', n) ? CBM_LANG_COUNT : CBM_LANG_RESCRIPT; +} + /* Disambiguate .cls files: shared by InterSystems ObjectScript UDL, Salesforce * Apex and Visual Basic 6 class modules (#721). ObjectScript class files begin * with a line of the form "Class ..."; VB6 class modules diff --git a/tests/test_complexity.c b/tests/test_complexity.c index d9c651ff9e..e510f52eb7 100644 --- a/tests/test_complexity.c +++ b/tests/test_complexity.c @@ -49,6 +49,8 @@ #include "../src/foundation/profile.h" #include "cbm.h" #include "discover/discover.h" +#include "lang_specs.h" +#include "tree_sitter/api.h" #include "pipeline/pass_lsp_cross.h" #include "pipeline/pipeline.h" #include "pipeline/pipeline_internal.h" @@ -760,7 +762,82 @@ TEST(complexity_importance_scoring_is_linear) { PASS(); } +/* ── Lexer work in error recovery (#2176) ────────────────────────────── + * tree-sitter lexes an unparseable stretch by retrying at every byte with + * every external token marked valid. The ReScript scanner then ran its + * template-string loop from each byte to the next '`', '$', '\\' or NUL — to + * the end of the file when there is none — and threw the result away. That + * is O(stretch) per byte, O(n^2) per file, all inside lexing where the parse + * budget's progress callback never runs; a binary Godot `.res` of high bytes + * (or plain text such as a run of '~') was dropped by the clock instead of + * parsed. Work is counted as the bytes the lexer pulls through a chunked + * TSInput: a pure function of (grammar, input), independent of speed. */ +enum { CX_LEX_CHUNK = 64, CX_LEX_BASE_BYTES = 4096 }; + +typedef struct { + const char *src; + uint32_t len; + uint64_t bytes_pulled; +} CxLexInput; + +static const char *cx_lex_read(void *payload, uint32_t byte_index, TSPoint position, + uint32_t *bytes_read) { + (void)position; + CxLexInput *in = (CxLexInput *)payload; + if (byte_index >= in->len) { + *bytes_read = 0; + return ""; + } + uint32_t n = in->len - byte_index; + if (n > CX_LEX_CHUNK) { + n = CX_LEX_CHUNK; + } + in->bytes_pulled += n; + *bytes_read = n; + return in->src + byte_index; +} + +/* Bytes pulled while parsing `len` copies of `fill` as ReScript; 0 on failure. */ +static uint64_t cx_rescript_lex_work(unsigned char fill, uint32_t len) { + char *src = malloc(len); + TSParser *parser = ts_parser_new(); + uint64_t work = 0; + if (src && parser && ts_parser_set_language(parser, cbm_ts_language(CBM_LANG_RESCRIPT))) { + memset(src, fill, len); + CxLexInput in = {src, len, 0}; + TSInput input = {&in, cx_lex_read, TSInputEncodingUTF8, NULL}; + TSTree *tree = ts_parser_parse(parser, NULL, input); + if (tree) { + work = in.bytes_pulled; + ts_tree_delete(tree); + } + } + if (parser) { + ts_parser_delete(parser); + } + free(src); + return work; +} + +TEST(complexity_rescript_error_recovery_lexing_is_linear) { + /* 0xFF: the reporter's binary bytes (invalid UTF-8). '~': plain ASCII text + * that ReScript cannot parse either — the defect is not binary-only. */ + static const unsigned char fills[] = {0xFF, '~'}; + for (size_t i = 0; i < sizeof(fills); i++) { + uint64_t base = cx_rescript_lex_work(fills[i], CX_LEX_BASE_BYTES); + uint64_t doubled = cx_rescript_lex_work(fills[i], 2 * CX_LEX_BASE_BYTES); + double r = cx_ratio((double)doubled, (double)base); + printf(" fill 0x%02x: lexer bytes %llu -> %llu ratio %.2f (linear ~2, quadratic ~4)\n", + fills[i], (unsigned long long)base, (unsigned long long)doubled, r); + /* Non-vacuous: the lexer must at least have read the input once. */ + ASSERT_GTE(base, (uint64_t)CX_LEX_BASE_BYTES); + ASSERT_TRUE(r >= CX_RATIO_LO && r <= CX_RATIO_HI); + } + PASS(); +} + SUITE(complexity) { + RUN_TEST(complexity_rescript_error_recovery_lexing_is_linear); RUN_TEST(complexity_replicated_modules_scale_linearly); RUN_TEST(complexity_perfile_registry_work_is_linear); RUN_TEST(complexity_importance_scoring_is_linear); diff --git a/tests/test_extraction.c b/tests/test_extraction.c index 1e6506ed1a..c235910ac1 100644 --- a/tests/test_extraction.c +++ b/tests/test_extraction.c @@ -6920,6 +6920,73 @@ TEST(extract_csharp_argument_values_use_the_walk_cursor) { } PASS(); } + +/* #2176: a binary Godot `.res` resource (4.7 MB) never finished indexing. It + * parses as ReScript into an error-heavy tree whose root has ~170k children + * per MB, and the ReScript `let` binding check climbed from every named leaf + * with ts_node_parent — a descent from the root that scans the children at + * every level, so O(root children) per leaf and quadratic per file (1 MB spent + * 46 s in the unified walk). Valid ReScript with many top-level statements has + * the same flat root, so plain text drives it too. The walk now climbs its own + * cursor. The counter records every root-descending hop that check makes; the + * assertion is that the walk makes none, at any size. */ +static int extract_rescript_let_leaf_fallbacks(int statement_count, int *out_usages, + uint64_t *out_slow_parent_fallbacks) { + static const char prefix[] = "let target = 1\n"; + static const char statement[] = "let v = target\n"; + size_t capacity = sizeof(prefix) + (size_t)statement_count * sizeof(statement); + char *source = malloc(capacity); + if (!source) { + return -1; + } + size_t offset = 0; + memcpy(source + offset, prefix, sizeof(prefix) - 1U); + offset += sizeof(prefix) - 1U; + for (int i = 0; i < statement_count; i++) { + memcpy(source + offset, statement, sizeof(statement) - 1U); + offset += sizeof(statement) - 1U; + } + source[offset] = '\0'; + + cbm_usage_field_lookup_test_reset(); + CBMFileResult *result = + cbm_extract_file(source, (int)offset, CBM_LANG_RESCRIPT, "proj", "Flat.res", 0, NULL, NULL); + free(source); + if (!result) { + return -1; + } + int usages = 0; + for (int i = 0; i < result->usages.count; i++) { + if (result->usages.items[i].ref_name && + strcmp(result->usages.items[i].ref_name, "target") == 0) { + usages++; + } + } + *out_slow_parent_fallbacks = cbm_usage_slow_parent_fallback_test_count(); + cbm_free_result(result); + *out_usages = usages; + return 0; +} + +TEST(extract_rescript_let_bindings_use_the_walk_cursor) { + enum { SMALL = 128, BIG = 1024 }; + int small_usages = 0; + int big_usages = 0; + uint64_t small_fallbacks = 0; + uint64_t big_fallbacks = 0; + ASSERT_EQ(extract_rescript_let_leaf_fallbacks(SMALL, &small_usages, &small_fallbacks), 0); + ASSERT_EQ(extract_rescript_let_leaf_fallbacks(BIG, &big_usages, &big_fallbacks), 0); + fprintf(stderr, " [rescript-let-bindings] fallbacks(%d)=%llu fallbacks(%d)=%llu\n", SMALL, + (unsigned long long)small_fallbacks, BIG, (unsigned long long)big_fallbacks); + /* Anti-vacuous: every leaf really was classified (each `target` read is a + * usage, each `v` a binding), so zero fallbacks means "took the cursor", + * not "never looked". */ + ASSERT_EQ(small_usages, SMALL); + ASSERT_EQ(big_usages, BIG); + ASSERT_EQ(small_fallbacks, 0); + ASSERT_EQ(big_fallbacks, 0); + PASS(); +} #endif /* =================================================================== @@ -8266,6 +8333,7 @@ SUITE(extraction) { #if defined(CBM_CALL_REFERENCE_LOOKUP_TEST_API) && CBM_CALL_REFERENCE_LOOKUP_TEST_API RUN_TEST(extract_wide_flat_reference_fields_are_linear); RUN_TEST(extract_csharp_argument_values_use_the_walk_cursor); + RUN_TEST(extract_rescript_let_bindings_use_the_walk_cursor); #endif /* Perl call-graph noise (#459 follow-up) */ diff --git a/tests/test_language.c b/tests/test_language.c index 18ad096008..dc2bcf7d36 100644 --- a/tests/test_language.c +++ b/tests/test_language.c @@ -4,6 +4,7 @@ * RED phase: These tests define the expected behavior for registered languages. */ #include "../src/foundation/compat.h" +#include "../src/foundation/compat_fs.h" #include "test_framework.h" #include "discover/discover.h" @@ -736,6 +737,35 @@ TEST(lang_frm_form_stays_form) { PASS(); } +/* #2176: a Godot binary resource shares the .res extension with ReScript. + * Parsed as ReScript it produced an error tree that never finished indexing; + * binary content is not ReScript source and must not be indexed as such. */ +TEST(lang_res_binary_resource_unsupported) { + static const unsigned char godot_head[] = {'R', 'S', 'R', 'C', 0x00, 0x00, 0x00, 0x00, + 0x04, 0x00, 0x00, 0x00, 'A', 'r', 'r', 'a', + 'y', 'M', 'e', 's', 'h', 0x00, 0xff, 0x80}; + char path[256]; + snprintf(path, sizeof(path), "%s/test_lang_godot.res", cbm_tmpdir()); + FILE *f = cbm_fopen(path, "wb"); + ASSERT_NOT_NULL(f); + ASSERT_EQ(fwrite(godot_head, 1, sizeof(godot_head), f), sizeof(godot_head)); + fclose(f); + ASSERT_EQ(cbm_disambiguate_res(path), CBM_LANG_COUNT); + remove(path); + PASS(); +} + +TEST(lang_res_rescript_stays_rescript) { + char path[256]; + snprintf(path, sizeof(path), "%s/test_lang_rescript.res", cbm_tmpdir()); + ASSERT_TRUE(write_probe_file(path, "let greet = name => `Hello ${name}`\n" + "module M = {\n let x = 1\n}\n")); + ASSERT_EQ(cbm_disambiguate_res(path), CBM_LANG_RESCRIPT); + remove(path); + ASSERT_EQ(cbm_disambiguate_res("/tmp/nonexistent_file_12345.res"), CBM_LANG_RESCRIPT); + PASS(); +} + TEST(lang_cfc_tag_component) { /* wrapper ⇒ tag dialect. */ ASSERT_EQ(disambiguate_cfc_content("test_cfc_tag.cfc", @@ -1456,6 +1486,8 @@ SUITE(language) { RUN_TEST(lang_cls_objectscript_stays_objectscript); RUN_TEST(lang_frm_vb6_form_unsupported); RUN_TEST(lang_frm_form_stays_form); + RUN_TEST(lang_res_binary_resource_unsupported); + RUN_TEST(lang_res_rescript_stays_rescript); /* Go test ports */ /* New languages */ From ea6fa49fc2c209de69c1b0f393d9d2cc25b61419 Mon Sep 17 00:00:00 2001 From: Martin Vogel Date: Mon, 28 Sep 2026 00:42:58 +0200 Subject: [PATCH 2/2] chore(security): refresh vendored checksums for the #2176 ReScript scanner patch security-static (scripts/security-vendored.sh) failed on this branch: the #2176 fix patches internal/cbm/vendored/grammars/rescript/scanner.c (skip the template-string branch in the error-recovery state) and records that patch in vendored/grammars/MANIFEST.md, but the checked-in digests in scripts/vendored-checksums.txt still described the unpatched files. Both changes are intentional and reviewed in the parent commit; this records their SHA-256 via `scripts/security-vendored.sh --update`. Only the two affected lines change, and the integrity check passes again. Signed-off-by: Martin Vogel --- scripts/vendored-checksums.txt | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/scripts/vendored-checksums.txt b/scripts/vendored-checksums.txt index 176f96d1e4..b561de94e2 100644 --- a/scripts/vendored-checksums.txt +++ b/scripts/vendored-checksums.txt @@ -5,7 +5,7 @@ c5cfb43042b6b72045f4ba997834d0a7786d2793d91680868b5815b39f14fc78 internal/cbm/v b29c1c9fb7cc82f58c84b376df1297d6e2737a1d655fd356db0859e3c29c2fea internal/cbm/vendored/common/tree_sitter/alloc.h 31e60a1bff6f715afacce03b5b70efe42b58371b4f9595dd4af52a577ff9608c internal/cbm/vendored/common/tree_sitter/array.h 180b893c8734778fd32f372dfbc27bd6ad1cd2221f26150b31256ff6716320d2 internal/cbm/vendored/common/tree_sitter/parser.h -a66a04ed735fdb3d4a90ee4353b8f0490dc0360da5965b0d13ec5d6766f146fa internal/cbm/vendored/grammars/MANIFEST.md +9fe439eff24b2a474618f2cb75cbeb06074840f0396fbeb1d2503e668a5d056a internal/cbm/vendored/grammars/MANIFEST.md ad8425038de519f8c4e3e9339feebf99dfad8a6002dcf79227d91402780a32cc internal/cbm/vendored/grammars/ada/LICENSE 02805ec13939b749c891567be36bf024b09034e04c80683a1ae667272458d549 internal/cbm/vendored/grammars/ada/parser.c 115a75d000bef9c70c4de7dfbb7f2a80b90cbb14ba265056dc7abb1ef3b9b6db internal/cbm/vendored/grammars/ada/tree_sitter/alloc.h @@ -667,7 +667,7 @@ c4b9482069f61a2a26a590181baa34f08eb6dcc57a138def578e6f9720c35ea5 internal/cbm/v c4b9482069f61a2a26a590181baa34f08eb6dcc57a138def578e6f9720c35ea5 internal/cbm/vendored/grammars/requirements/tree_sitter/parser.h 21da9b4e463af4b4501cbd4e0c5886a1f321a740b584002a9c9a5fe7b2d46a8a internal/cbm/vendored/grammars/rescript/LICENSE d75538517e1024056c206f6d1b10cb4ed83670a5d6fb338dbad7b2bb4269403a internal/cbm/vendored/grammars/rescript/parser.c -4bf614fc0b599bf75414c1512e92c971d4bb04d0334e5f562ff34a0dbd745f8a internal/cbm/vendored/grammars/rescript/scanner.c +d550ab7d815666f14b938e805f106a2ce4d99f6f86377d7a301dcf076db5f138 internal/cbm/vendored/grammars/rescript/scanner.c 115a75d000bef9c70c4de7dfbb7f2a80b90cbb14ba265056dc7abb1ef3b9b6db internal/cbm/vendored/grammars/rescript/tree_sitter/alloc.h 02d139d44303b0b0b85a899839dbc1346c478e7318d1f2971fad2b38fd57cb73 internal/cbm/vendored/grammars/rescript/tree_sitter/array.h c4b9482069f61a2a26a590181baa34f08eb6dcc57a138def578e6f9720c35ea5 internal/cbm/vendored/grammars/rescript/tree_sitter/parser.h