From 4905ffcce6e08f9e455eab0819b3aef674aebb81 Mon Sep 17 00:00:00 2001 From: Edson Date: Thu, 1 Oct 2026 09:59:36 -0400 Subject: [PATCH 1/2] test(haskell): an escaped char literal must not swallow the next def (#2439) The vendored grammar lexes an escaped char literal with /'\\[^ ]*'/, whose [^ ]* also crosses newlines and quotes. With only newlines between '\n' and a primed name (g'), the char token runs through g', so f spans g' and g' gets no node. On one line, ['\n','\t'] lexes as a single char token with no parse error. Add regression tests for both forms. Signed-off-by: Edson --- tests/test_extraction.c | 77 +++++++++++++++++++++++++++++++++++++++++ 1 file changed, 77 insertions(+) diff --git a/tests/test_extraction.c b/tests/test_extraction.c index cde8aee45..1bb6291e6 100644 --- a/tests/test_extraction.c +++ b/tests/test_extraction.c @@ -1433,6 +1433,81 @@ TEST(haskell_function) { PASS(); } +/* #2439: the vendored grammar lexes an escaped char literal with + * /'\\[^ ]*'/, whose `[^ ]*` also matches newlines and quotes. With only + * newlines between `'\n'` and the next definition's primed name, the char + * token runs from `'\n` through `g'`, so `f` swallows `g'` and `g'` gets no + * node. Each top-level definition must keep its own node and span. */ +TEST(haskell_escaped_char_does_not_swallow_primed_def) { + CBMFileResult *r = extract("module M where\n" + "\n" + "f = '\\n'\n" + "\n" + "g' = 2\n" + "\n" + "h = 3\n", + CBM_LANG_HASKELL, "t", "M.hs"); + ASSERT_NOT_NULL(r); + const CBMDefinition *f = NULL; + for (int i = 0; i < r->defs.count; i++) { + if (strcmp(r->defs.items[i].name, "f") == 0) { + f = &r->defs.items[i]; + } + } + ASSERT_NOT_NULL(f); + ASSERT_EQ((int)f->end_line, 3); + ASSERT(has_def(r, "Function", "g'")); + ASSERT(has_def(r, "Function", "h")); + ASSERT_FALSE(r->has_error); + cbm_free_result(r); + PASS(); +} + +/* Append the text of every `char` node under root to out, space-separated. */ +static void haskell_char_texts(TSNode root, const char *src, char *out, size_t out_sz) { + TSTreeCursor c = ts_tree_cursor_new(root); + size_t len = 0; + for (;;) { + TSNode n = ts_tree_cursor_current_node(&c); + if (strcmp(ts_node_type(n), "char") == 0 && len < out_sz) { + uint32_t s = ts_node_start_byte(n); + uint32_t e = ts_node_end_byte(n); + len += (size_t)snprintf(out + len, out_sz - len, "%.*s ", (int)(e - s), src + s); + } + if (ts_tree_cursor_goto_first_child(&c)) { + continue; + } + while (!ts_tree_cursor_goto_next_sibling(&c)) { + if (!ts_tree_cursor_goto_parent(&c)) { + ts_tree_cursor_delete(&c); + return; + } + } + } +} + +/* #2439, same-line form: the old regex ran from the first `'\` to the last + * `'` before a space, so `['\n','\t']` lexed as ONE char token with no parse + * error. Each literal must be its own token, and the escapes that contain a + * quote or a backslash (`'\''`, `'\\'`) must still lex whole. */ +TEST(haskell_escaped_char_list_lexes_each_literal) { + const char *src = "x = ['\\n','\\t','\\'','\\\\','\\x41','\\SOH','a']\n"; + TSParser *parser = ts_parser_new(); + ASSERT_NOT_NULL(parser); + ASSERT_TRUE(ts_parser_set_language(parser, cbm_ts_language(CBM_LANG_HASKELL))); + TSTree *tree = ts_parser_parse_string(parser, NULL, src, (uint32_t)strlen(src)); + ASSERT_NOT_NULL(tree); + TSNode root = ts_tree_root_node(tree); + char got[128] = ""; + haskell_char_texts(root, src, got, sizeof(got)); + bool has_error = ts_node_has_error(root); + ts_tree_delete(tree); + ts_parser_delete(parser); + ASSERT_FALSE(has_error); + ASSERT_STR_EQ(got, "'\\n' '\\t' '\\'' '\\\\' '\\x41' '\\SOH' 'a' "); + PASS(); +} + /* --- OCaml --- */ TEST(ocaml_function) { CBMFileResult *r = @@ -9354,6 +9429,8 @@ SUITE(extraction) { RUN_TEST(elixir_function); RUN_TEST(elixir_call_string_argument); RUN_TEST(haskell_function); + RUN_TEST(haskell_escaped_char_does_not_swallow_primed_def); + RUN_TEST(haskell_escaped_char_list_lexes_each_literal); RUN_TEST(ocaml_function); RUN_TEST(erlang_function); From 340bb580077d577fc5415e72ee32f69a706688f7 Mon Sep 17 00:00:00 2001 From: Edson Date: Thu, 1 Oct 2026 11:18:55 -0400 Subject: [PATCH 2/2] fix(haskell): stop escaped char literals at the closing quote (#2439) tree-sitter-haskell lexes an escaped char literal with /'\\[^ ]*'/. [^ ]* also matches newlines and quotes, and the lexer keeps the longest match, so the token ran on to the last ' before the next space. In f = '\n' g' = 2 it ran from '\n through g': f spanned three lines, g' got no node and the valid module reported has_error. On one line ['\n','\t'] lexed as a single char token, silently. Patch the generated char lex states in haskell/parser.c: the escape body (state 28) also stops at whitespace, and after the closing quote states 112/114/116 no longer re-enter the body; 114 takes one more ' only to complete '\''. The token is now '\ [^'\s]* ' '? plus up to two #. '\'', '\\', '\x41', '\SOH', '\^A', '\1234' and '\n'# lex as before. The reporter's suspect, take_char_literal in scanner.c, is not involved: that path handles '\n' correctly. Upstream pin 7fa19f195803 and master both carry the regex, so the patch is recorded in vendored/grammars/MANIFEST.md. The MANIFEST.md and parser.c lines in scripts/vendored-checksums.txt are refreshed with scripts/security-vendored.sh --update. Fixes #2439 Signed-off-by: Edson --- internal/cbm/vendored/grammars/MANIFEST.md | 1 + internal/cbm/vendored/grammars/haskell/parser.c | 11 ++--------- scripts/vendored-checksums.txt | 4 ++-- 3 files changed, 5 insertions(+), 11 deletions(-) diff --git a/internal/cbm/vendored/grammars/MANIFEST.md b/internal/cbm/vendored/grammars/MANIFEST.md index 3ee1344a5..39f509492 100644 --- a/internal/cbm/vendored/grammars/MANIFEST.md +++ b/internal/cbm/vendored/grammars/MANIFEST.md @@ -112,6 +112,7 @@ row instead. | plsql | `plsql/parser.c`, include | `#include ` → `#include "tree_sitter/parser.h"` | The older ABI-14 generator emits angle brackets; every other vendored grammar uses the quoted form, which resolves the per-grammar `tree_sitter/` header from the including file's directory | | properties | `properties/scanner.c`, whole file | move the `reached_eof` flag out of a file-scope `static bool` and into a per-parser payload allocated by `..._external_scanner_create()` | **Data race, and the only one of its kind in 103 vendored scanners.** That flag is how the scanner refuses to emit a second end-of-input `FAKE_EOL`, and upstream keeps it in ONE process-wide object. With several worker threads indexing `.properties` files at once, whichever reached EOF first set the flag, so another thread's `!reached_eof` was false and its parse never received the `FAKE_EOL` it needed to finish — it sat at end-of-input asking for a token that would never arrive. Measured 2026-09-19: ~106 M parse operations on a 253-byte file before the per-file budget stopped it, after which the file was dropped from the graph and java/kotlin indexes differed run to run. Interleaved A/B under identical load, 40 runs each: unpatched 6 stalls, patched 0. The serialize/deserialize wire format is unchanged (the state is still the returned length). Upstream `6310671b24d4` still carries the static, so a re-vendor must re-apply this | | swift | `swift/scanner.c`, `OP_SYMBOL_SUPPRESSOR` + `eat_operators` | `1UL <<` / `1 <<` → `1ULL <<` | UBSan: `1 << suppressor` shifts an `int` by up to `TOKEN_COUNT` bits, undefined once the index reaches 31, while the mask it feeds is `uint64_t`. `1UL << FAKE_TRY_BANG` is the same defect on Windows, where `unsigned long` is 32 bits and `FAKE_TRY_BANG` is 32; the CLANGARM64 leg runs UBSan in trap mode, so there it is an illegal instruction rather than a log line. Upstream already carries both changes: `fb63a7004f07` (2026-04-06, upstream #558) for `eat_operators`, `6ab8d1d74ebd` (2026-08-10) for the `OP_SYMBOL_SUPPRESSOR` entry. Our pin `8abb3e8b3325` (2026-03-20) predates both, so this is a backport rather than a local invention — a re-vendor past 2026-08-10 should delete this row, not re-apply it | +| haskell | `haskell/parser.c`, `ts_lex` states 28, 112, 114, 116 (the `char` token) | state 28 (escape body) also stops at `\t`..`\r`; after the closing quote, states 112/114/116 no longer re-enter the body, and 114 takes one more `'` only to complete `'\''` (114 → 115). The token is now `'\` `[^'\s]*` `'` `'`? then up to two `#` | **Escaped char literals swallowed the following code (#2439).** Upstream `grammar/literal.js` lexes an escaped char as `/'\\[^ ]*'/`, and `[^ ]*` also crosses newlines and quotes, so the lexer's longest match ran on to the last `'` before the next space: in `f = '\n'` followed by a blank line and `g' = 2`, the token ran from `'\n` through `g'`, `f` spanned three lines, `g'` got no node and the valid module reported `has_error`. On one line `['\n','\t']` lexed as a single `char` silently. The generated lexer is patched directly; regenerating `parser.c` would instead need that regex narrowed in `literal.js`. `'\''`, `'\\'`, `'\x41'`, `'\SOH'`, `'\^A'`, `'\1234'` and MagicHash `'\n'#` lex as before (extraction tests `haskell_escaped_char_does_not_swallow_primed_def`, `haskell_escaped_char_list_lexes_each_literal`). Side effect: an escaped char immediately followed by a stray `'` (`'\n''`, not valid Haskell) is taken as one token. Upstream pin `7fa19f195803` and upstream master both still carry the regex, so a re-vendor must re-apply this | ## Vendored from verified upstream diff --git a/internal/cbm/vendored/grammars/haskell/parser.c b/internal/cbm/vendored/grammars/haskell/parser.c index e82fd1092..7beff686b 100644 --- a/internal/cbm/vendored/grammars/haskell/parser.c +++ b/internal/cbm/vendored/grammars/haskell/parser.c @@ -21498,6 +21498,7 @@ static bool ts_lex(TSLexer *lexer, TSStateId state) { case 28: if (lookahead == '\'') ADVANCE(114); if (lookahead != 0 && + (lookahead < '\t' || '\r' < lookahead) && lookahead != ' ') ADVANCE(28); END_STATE(); case 29: @@ -21860,9 +21861,6 @@ static bool ts_lex(TSLexer *lexer, TSStateId state) { case 112: ACCEPT_TOKEN(sym_char); if (lookahead == '#') ADVANCE(116); - if (lookahead == '\'') ADVANCE(114); - if (lookahead != 0 && - lookahead != ' ') ADVANCE(28); END_STATE(); case 113: ACCEPT_TOKEN(sym_char); @@ -21871,9 +21869,7 @@ static bool ts_lex(TSLexer *lexer, TSStateId state) { case 114: ACCEPT_TOKEN(sym_char); if (lookahead == '#') ADVANCE(112); - if (lookahead == '\'') ADVANCE(114); - if (lookahead != 0 && - lookahead != ' ') ADVANCE(28); + if (lookahead == '\'') ADVANCE(115); END_STATE(); case 115: ACCEPT_TOKEN(sym_char); @@ -21881,9 +21877,6 @@ static bool ts_lex(TSLexer *lexer, TSStateId state) { END_STATE(); case 116: ACCEPT_TOKEN(sym_char); - if (lookahead == '\'') ADVANCE(114); - if (lookahead != 0 && - lookahead != ' ') ADVANCE(28); END_STATE(); case 117: ACCEPT_TOKEN(sym_string); diff --git a/scripts/vendored-checksums.txt b/scripts/vendored-checksums.txt index ef424f4a1..84e0f13f8 100644 --- a/scripts/vendored-checksums.txt +++ b/scripts/vendored-checksums.txt @@ -5,7 +5,7 @@ c5cfb43042b6b72045f4ba997834d0a7786d2793d91680868b5815b39f14fc78 internal/cbm/v b29c1c9fb7cc82f58c84b376df1297d6e2737a1d655fd356db0859e3c29c2fea internal/cbm/vendored/common/tree_sitter/alloc.h 31e60a1bff6f715afacce03b5b70efe42b58371b4f9595dd4af52a577ff9608c internal/cbm/vendored/common/tree_sitter/array.h 180b893c8734778fd32f372dfbc27bd6ad1cd2221f26150b31256ff6716320d2 internal/cbm/vendored/common/tree_sitter/parser.h -44bade69b14ac382b08835965eeec5f0fd6e54ccb7a80da21c0161128a10503f internal/cbm/vendored/grammars/MANIFEST.md +b2e0bacd8ee44b6e26c35eec170b2f4b0821500d04c4d8735c994cd346a66d2a internal/cbm/vendored/grammars/MANIFEST.md ad8425038de519f8c4e3e9339feebf99dfad8a6002dcf79227d91402780a32cc internal/cbm/vendored/grammars/ada/LICENSE 02805ec13939b749c891567be36bf024b09034e04c80683a1ae667272458d549 internal/cbm/vendored/grammars/ada/parser.c 115a75d000bef9c70c4de7dfbb7f2a80b90cbb14ba265056dc7abb1ef3b9b6db internal/cbm/vendored/grammars/ada/tree_sitter/alloc.h @@ -320,7 +320,7 @@ bb6111e9f89cde51262d132b9d4ca57b012a20b990102744886e6fa0aaae742d internal/cbm/v 02d139d44303b0b0b85a899839dbc1346c478e7318d1f2971fad2b38fd57cb73 internal/cbm/vendored/grammars/hare/tree_sitter/array.h c4b9482069f61a2a26a590181baa34f08eb6dcc57a138def578e6f9720c35ea5 internal/cbm/vendored/grammars/hare/tree_sitter/parser.h 2e0110e07abef7c2548b26ec9d6969775617ca539a0dc8dbeeb14d6452c711d1 internal/cbm/vendored/grammars/haskell/LICENSE -a0f269377b61613f0f90b11126d03661fff08a1bbb808e0a3d5c477b6c9041fc internal/cbm/vendored/grammars/haskell/parser.c +a3d6165ba54c6469b22d7f7e66ac0a8eeeb46e9c41f1e9f3fa9ec36e02dc5a23 internal/cbm/vendored/grammars/haskell/parser.c 2895fbd5518cffa672d42dda7ba030add2ec4b4280de7befe61edf2cf6447dea internal/cbm/vendored/grammars/haskell/scanner.c 115a75d000bef9c70c4de7dfbb7f2a80b90cbb14ba265056dc7abb1ef3b9b6db internal/cbm/vendored/grammars/haskell/tree_sitter/alloc.h 02d139d44303b0b0b85a899839dbc1346c478e7318d1f2971fad2b38fd57cb73 internal/cbm/vendored/grammars/haskell/tree_sitter/array.h