diff --git a/src/scanner.c b/src/scanner.c index 795916dd..7a87c3e5 100644 --- a/src/scanner.c +++ b/src/scanner.c @@ -289,6 +289,35 @@ static bool scan_html_comment(TSLexer *lexer) { return true; } +// After an `&`: is what follows the rest of an HTML character reference +// (`&`, `&`, `&`), as the grammar's html_character_reference +// token defines it? Consumes the name or number either way. +static bool scan_character_reference_tail(TSLexer *lexer) { + unsigned length = 0; + if (lexer->lookahead == '#') { + advance(lexer); + if (lexer->lookahead == 'x' || lexer->lookahead == 'X') { + advance(lexer); + while (length < 6 && iswxdigit(lexer->lookahead)) { + advance(lexer); + length++; + } + } else { + while (length < 5 && iswdigit(lexer->lookahead)) { + advance(lexer); + length++; + } + } + } else { + while (length < 30 && ((lexer->lookahead >= 'a' && lexer->lookahead <= 'z') || + (lexer->lookahead >= 'A' && lexer->lookahead <= 'Z'))) { + advance(lexer); + length++; + } + } + return length > 0 && lexer->lookahead == ';'; +} + static bool scan_jsx_text(TSLexer *lexer) { // saw_text will be true if we see any non-whitespace content, or any whitespace content that is not a newline and // does not immediately follow a newline. @@ -297,8 +326,22 @@ static bool scan_jsx_text(TSLexer *lexer) { // immediately follows a newline. bool at_newline = false; + lexer->result_symbol = JSX_TEXT; + while (lexer->lookahead != 0 && lexer->lookahead != '<' && lexer->lookahead != '>' && lexer->lookahead != '{' && - lexer->lookahead != '}' && lexer->lookahead != '&') { + lexer->lookahead != '}') { + if (lexer->lookahead == '&') { + // A character reference is its own token: the text ends before it. + // A bare ampersand (`Shipping & Handling`) is text. + lexer->mark_end(lexer); + advance(lexer); + if (scan_character_reference_tail(lexer)) { + return saw_text; + } + saw_text = true; + at_newline = false; + continue; + } bool is_wspace = iswspace(lexer->lookahead); if (lexer->lookahead == '\n') { at_newline = true; @@ -326,7 +369,7 @@ static bool scan_jsx_text(TSLexer *lexer) { advance(lexer); } - lexer->result_symbol = JSX_TEXT; + lexer->mark_end(lexer); return saw_text; } diff --git a/test/corpus/literals.txt b/test/corpus/literals.txt index 4a617652..9df86da7 100644 --- a/test/corpus/literals.txt +++ b/test/corpus/literals.txt @@ -171,3 +171,41 @@ JSX with HTML character references (entities) (jsx_text) (jsx_closing_element (identifier))))) +============================================= +JSX with bare ampersands in text +============================================= + +

Shipping & Handling

; + +

a &b; c &#; d &#x; e &

; + +

&

; + +---- + +(program + (expression_statement + (jsx_element + (jsx_opening_element + (identifier)) + (jsx_text) + (jsx_closing_element + (identifier)))) + (expression_statement + (jsx_element + (jsx_opening_element + (identifier)) + (jsx_text) + (html_character_reference) + (jsx_text) + (html_character_reference) + (jsx_closing_element + (identifier)))) + (expression_statement + (jsx_element + (jsx_opening_element + (identifier)) + (jsx_text) + (jsx_closing_element + (identifier))))) +