From c76af4f338fb99a31d43bf43c8b53af37f11f567 Mon Sep 17 00:00:00 2001 From: Patrick Meredith Date: Wed, 23 Sep 2026 14:42:43 -0400 Subject: [PATCH] fix: accept bare ampersands in JSX text `

Shipping & Handling

` is valid JSX (React renders the ampersand), but scan_jsx_text stopped at every `&` and the grammar only accepts a full character reference there, so a bare `&` was a parse error. The scanner now looks past an `&`: if what follows is the rest of an HTML character reference as html_character_reference defines it (`&`, `&`, `&`), the text ends before it and the reference is its own token as before; otherwise the ampersand is text. Fixes #366 Co-Authored-By: Claude Fable 5.1 --- src/scanner.c | 47 ++++++++++++++++++++++++++++++++++++++-- test/corpus/literals.txt | 38 ++++++++++++++++++++++++++++++++ 2 files changed, 83 insertions(+), 2 deletions(-) diff --git a/src/scanner.c b/src/scanner.c index 795916dd..7a87c3e5 100644 --- a/src/scanner.c +++ b/src/scanner.c @@ -289,6 +289,35 @@ static bool scan_html_comment(TSLexer *lexer) { return true; } +// After an `&`: is what follows the rest of an HTML character reference +// (`&`, `&`, `&`), as the grammar's html_character_reference +// token defines it? Consumes the name or number either way. +static bool scan_character_reference_tail(TSLexer *lexer) { + unsigned length = 0; + if (lexer->lookahead == '#') { + advance(lexer); + if (lexer->lookahead == 'x' || lexer->lookahead == 'X') { + advance(lexer); + while (length < 6 && iswxdigit(lexer->lookahead)) { + advance(lexer); + length++; + } + } else { + while (length < 5 && iswdigit(lexer->lookahead)) { + advance(lexer); + length++; + } + } + } else { + while (length < 30 && ((lexer->lookahead >= 'a' && lexer->lookahead <= 'z') || + (lexer->lookahead >= 'A' && lexer->lookahead <= 'Z'))) { + advance(lexer); + length++; + } + } + return length > 0 && lexer->lookahead == ';'; +} + static bool scan_jsx_text(TSLexer *lexer) { // saw_text will be true if we see any non-whitespace content, or any whitespace content that is not a newline and // does not immediately follow a newline. @@ -297,8 +326,22 @@ static bool scan_jsx_text(TSLexer *lexer) { // immediately follows a newline. bool at_newline = false; + lexer->result_symbol = JSX_TEXT; + while (lexer->lookahead != 0 && lexer->lookahead != '<' && lexer->lookahead != '>' && lexer->lookahead != '{' && - lexer->lookahead != '}' && lexer->lookahead != '&') { + lexer->lookahead != '}') { + if (lexer->lookahead == '&') { + // A character reference is its own token: the text ends before it. + // A bare ampersand (`Shipping & Handling`) is text. + lexer->mark_end(lexer); + advance(lexer); + if (scan_character_reference_tail(lexer)) { + return saw_text; + } + saw_text = true; + at_newline = false; + continue; + } bool is_wspace = iswspace(lexer->lookahead); if (lexer->lookahead == '\n') { at_newline = true; @@ -326,7 +369,7 @@ static bool scan_jsx_text(TSLexer *lexer) { advance(lexer); } - lexer->result_symbol = JSX_TEXT; + lexer->mark_end(lexer); return saw_text; } diff --git a/test/corpus/literals.txt b/test/corpus/literals.txt index 4a617652..9df86da7 100644 --- a/test/corpus/literals.txt +++ b/test/corpus/literals.txt @@ -171,3 +171,41 @@ JSX with HTML character references (entities) (jsx_text) (jsx_closing_element (identifier))))) +============================================= +JSX with bare ampersands in text +============================================= + +

Shipping & Handling

; + +

a &b; c &#; d &#x; e &

; + +

&

; + +---- + +(program + (expression_statement + (jsx_element + (jsx_opening_element + (identifier)) + (jsx_text) + (jsx_closing_element + (identifier)))) + (expression_statement + (jsx_element + (jsx_opening_element + (identifier)) + (jsx_text) + (html_character_reference) + (jsx_text) + (html_character_reference) + (jsx_closing_element + (identifier)))) + (expression_statement + (jsx_element + (jsx_opening_element + (identifier)) + (jsx_text) + (jsx_closing_element + (identifier))))) +