feat(kernel): R7b Scala walker — scala module, vendored-grammar-C master@0aca5d0a6f, scala default-routed (#1385)
R7b batch 4 #3 (docs/design/scala-kernel-port-checklist.md is the authoritative quirk list). The third vendored-grammar-C language and the biggest grammar in the tree (35MB parser.c): the vendored wasm is tree-sitter/tree-sitter-scala master@0aca5d0a6f — a post-v0.26.0 generation sync that is not a release (the 0.26.0 crate is 30 states BEHIND, so a crate pin would be a silent downgrade). NO wasm change: production has parsed with this exact revision since #91 — the kernel-grammar-parity row (ABI 15, 26,650 states, 32 fields, id-by-id tables) is the whole alignment proof. Preserved bug-for-bug (all probe-pinned): the leak-through asymmetries — extension methods mint NO nodes (first def's body calls leak to the enclosing scope, later defs invisible, and the braced form resolves its body field to the `{` TOKEN via first-match-wins field lookup → whole extension invisible); anonymous `new T { … }` template_body members leak to the enclosing scope (findAnonymousClassBody misses template_body); the bodied-vs-bodiless class asymmetry (bodiless headers walk class_parameters → default-value calls emit FROM the class; bodied ones never see them) — plus first-segment import names (`import com.example.C` → `com`), the val/var hook keyed on the enclosing-definition NODE TYPE (object vals → constants/value-ref targets, class/trait/enum/given vals → fields) with consumed initializers, every def routed through extractMethod with the top-level function fallback, nested defs in bodies minting NOTHING (the inverse of kotlin) while body-local classes extract fully, curried signatures keeping only the FIRST parameter list (type params win the `parameters` field), enum cases positioned at the CASE node with invisible params/extends tails, extends with-chains via scalaBaseTypeName, `@deprecated(args)` decorates, the #750 capitalized-chain re-encode (`WidgetS.create().render`), literal-receiver silence, static-member reads AND writes, infix invisibility, `derives` silence, scaladoc retention with the CRLF `\r` pin, full value-reference machinery (shadow prune, last-wins same-name targets, `$X`/`${X}` interpolation reads), and SCALA_SPEC fn-refs (bare ids + postfix eta unwrap + varinit, var-init non-capture). Gates: parity sweeps first-run 0-diff on os-lib/cats/scala3-compiler-src/ scala3-library-src — 1,935 clean files byte-parity, deferrals 0/15/57/116 matching the survey's predictions exactly (scala-3's PHANTOM hasError files — flag-true, zero ERROR nodes, capture-checking `^` — defer on the FLAG); full-init dumps byte-identical ×3 (os-lib, cats, scala3 whole-repo 950,889 dump lines); kernel-scala-parity suite (9 fixtures + 9 in-memory CRLF variants incl. Scala-3 indentation through the external scanner + phantom/real-error defer pins + first-segment/namespace/value-ref pins); full suite 2,669 green ×3 with CODEGRAPH_KERNEL_EXPECT=1 (kernel-scaffold's stays-wasm example moved scala → pascal). DEFAULT_ROUTED += scala (19 langs). Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Fable 5
parent
e32135171e
commit
bdd687b49f
@@ -0,0 +1,21 @@
|
||||
The MIT License (MIT)
|
||||
|
||||
Copyright (c) 2018 Max Brunsfeld and GitHub
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,614 @@
|
||||
#include "tree_sitter/alloc.h"
|
||||
#include "tree_sitter/array.h"
|
||||
#include "tree_sitter/parser.h"
|
||||
|
||||
#include <wctype.h>
|
||||
|
||||
// #define DEBUG
|
||||
|
||||
#ifdef DEBUG
|
||||
#define LOG(...) fprintf(stderr, __VA_ARGS__)
|
||||
#else
|
||||
#define LOG(...)
|
||||
#endif
|
||||
|
||||
enum TokenType {
|
||||
AUTOMATIC_SEMICOLON,
|
||||
INDENT,
|
||||
OUTDENT,
|
||||
COMMA_OUTDENT,
|
||||
SIMPLE_STRING_START,
|
||||
SIMPLE_STRING_MIDDLE,
|
||||
SIMPLE_MULTILINE_STRING_START,
|
||||
INTERPOLATED_STRING_MIDDLE,
|
||||
INTERPOLATED_MULTILINE_STRING_MIDDLE,
|
||||
RAW_STRING_START,
|
||||
RAW_STRING_MIDDLE,
|
||||
RAW_STRING_MULTILINE_MIDDLE,
|
||||
SINGLE_LINE_STRING_END,
|
||||
MULTILINE_STRING_END,
|
||||
ELSE,
|
||||
CATCH,
|
||||
FINALLY,
|
||||
EXTENDS,
|
||||
DERIVES,
|
||||
WITH,
|
||||
ERROR_SENTINEL
|
||||
};
|
||||
|
||||
const char* token_name[] = {
|
||||
"AUTOMATIC_SEMICOLON",
|
||||
"INDENT",
|
||||
"OUTDENT",
|
||||
"COMMA_OUTDENT",
|
||||
"SIMPLE_STRING_START",
|
||||
"SIMPLE_STRING_MIDDLE",
|
||||
"SIMPLE_MULTILINE_STRING_START",
|
||||
"INTERPOLATED_STRING_MIDDLE",
|
||||
"INTERPOLATED_MULTILINE_STRING_MIDDLE",
|
||||
"RAW_STRING_MIDDLE",
|
||||
"RAW_STRING_MULTILINE_MIDDLE",
|
||||
"SINGLE_LINE_STRING_END",
|
||||
"MULTILINE_STRING_END",
|
||||
"ELSE",
|
||||
"CATCH",
|
||||
"FINALLY",
|
||||
"EXTENDS",
|
||||
"DERIVES",
|
||||
"WITH",
|
||||
"ERROR_SENTINEL"
|
||||
};
|
||||
|
||||
typedef struct {
|
||||
Array(int16_t) indents;
|
||||
int16_t last_indentation_size;
|
||||
int16_t last_newline_count;
|
||||
int16_t last_column;
|
||||
} Scanner;
|
||||
|
||||
void *tree_sitter_scala_external_scanner_create() {
|
||||
Scanner *scanner = ts_calloc(1, sizeof(Scanner));
|
||||
array_init(&scanner->indents);
|
||||
scanner->last_indentation_size = -1;
|
||||
scanner->last_column = -1;
|
||||
return scanner;
|
||||
}
|
||||
|
||||
void tree_sitter_scala_external_scanner_destroy(void *payload) {
|
||||
Scanner *scanner = payload;
|
||||
array_delete(&scanner->indents);
|
||||
ts_free(scanner);
|
||||
}
|
||||
|
||||
unsigned tree_sitter_scala_external_scanner_serialize(void *payload, char *buffer) {
|
||||
Scanner *scanner = (Scanner*)payload;
|
||||
|
||||
if ((scanner->indents.size + 3) * sizeof(int16_t) > TREE_SITTER_SERIALIZATION_BUFFER_SIZE) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
size_t size = 0;
|
||||
memcpy(buffer + size, &scanner->last_indentation_size, sizeof(int16_t));
|
||||
size += sizeof(int16_t);
|
||||
memcpy(buffer + size, &scanner->last_newline_count, sizeof(int16_t));
|
||||
size += sizeof(int16_t);
|
||||
memcpy(buffer + size, &scanner->last_column, sizeof(int16_t));
|
||||
size += sizeof(int16_t);
|
||||
|
||||
for (unsigned i = 0; i < scanner->indents.size; i++) {
|
||||
memcpy(buffer + size, &scanner->indents.contents[i], sizeof(int16_t));
|
||||
size += sizeof(int16_t);
|
||||
}
|
||||
|
||||
return size;
|
||||
}
|
||||
|
||||
void tree_sitter_scala_external_scanner_deserialize(void *payload, const char *buffer,
|
||||
unsigned length) {
|
||||
Scanner *scanner = (Scanner*)payload;
|
||||
array_clear(&scanner->indents);
|
||||
scanner->last_indentation_size = -1;
|
||||
scanner->last_column = -1;
|
||||
scanner->last_newline_count = 0;
|
||||
|
||||
if (length == 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
size_t size = 0;
|
||||
|
||||
scanner->last_indentation_size = *(int16_t *)&buffer[size];
|
||||
size += sizeof(int16_t);
|
||||
scanner->last_newline_count = *(int16_t *)&buffer[size];
|
||||
size += sizeof(int16_t);
|
||||
scanner->last_column = *(int16_t *)&buffer[size];
|
||||
size += sizeof(int16_t);
|
||||
|
||||
while (size < length) {
|
||||
array_push(&scanner->indents, *(int16_t *)&buffer[size]);
|
||||
size += sizeof(int16_t);
|
||||
}
|
||||
|
||||
assert(size == length);
|
||||
}
|
||||
|
||||
static inline void advance(TSLexer *lexer) { lexer->advance(lexer, false); }
|
||||
|
||||
static inline void skip(TSLexer *lexer) { lexer->advance(lexer, true); }
|
||||
|
||||
// Used to detect leading infix operators on continuation lines.
|
||||
// See: https://www.scala-lang.org/api/3.x/docs/changed-features/operators.html
|
||||
static bool is_op_char(int32_t c) {
|
||||
switch (c) {
|
||||
case '!': case '#': case '%': case '&':
|
||||
case '*': case '+': case '-': case '<':
|
||||
case '=': case '>': case '?': case '@':
|
||||
case '\\': case '^': case '|': case '~':
|
||||
case ':':
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// We enumerate 3 types of strings that we need to handle differently:
|
||||
// 1. Simple strings, `"..."` or `"""..."""`
|
||||
// 2. Interpolated strings, `s"..."` or `f"..."` or `foo"..."` or foo"""...""".
|
||||
// 3. Raw strings, `raw"..."`
|
||||
typedef enum {
|
||||
STRING_MODE_SIMPLE,
|
||||
STRING_MODE_INTERPOLATED,
|
||||
STRING_MODE_RAW
|
||||
} StringMode;
|
||||
|
||||
static bool scan_string_content(TSLexer *lexer, bool is_multiline, StringMode string_mode) {
|
||||
LOG("scan_string_content(%d, %d, %c)\n", is_multiline, string_mode, lexer->lookahead);
|
||||
unsigned closing_quote_count = 0;
|
||||
for (;;) {
|
||||
if (lexer->lookahead == '"') {
|
||||
advance(lexer);
|
||||
closing_quote_count++;
|
||||
if (!is_multiline) {
|
||||
lexer->result_symbol = SINGLE_LINE_STRING_END;
|
||||
lexer->mark_end(lexer);
|
||||
return true;
|
||||
}
|
||||
if (closing_quote_count >= 3 && lexer->lookahead != '"') {
|
||||
lexer->result_symbol = MULTILINE_STRING_END;
|
||||
lexer->mark_end(lexer);
|
||||
return true;
|
||||
}
|
||||
} else if (lexer->lookahead == '$' && string_mode != STRING_MODE_SIMPLE) {
|
||||
switch (string_mode) {
|
||||
case STRING_MODE_INTERPOLATED:
|
||||
lexer->result_symbol = is_multiline ? INTERPOLATED_MULTILINE_STRING_MIDDLE : INTERPOLATED_STRING_MIDDLE;
|
||||
break;
|
||||
case STRING_MODE_RAW:
|
||||
lexer->result_symbol = is_multiline ? RAW_STRING_MULTILINE_MIDDLE : RAW_STRING_MIDDLE;
|
||||
break;
|
||||
default:
|
||||
assert(false);
|
||||
}
|
||||
lexer->mark_end(lexer);
|
||||
return true;
|
||||
} else {
|
||||
closing_quote_count = 0;
|
||||
if (lexer->lookahead == '\\') {
|
||||
// Multiline strings ignore escape sequences
|
||||
if (is_multiline || string_mode == STRING_MODE_RAW) {
|
||||
// FIXME: In raw string mode, we have to jump over escaped quotes.
|
||||
advance(lexer);
|
||||
// In single-line raw strings, `\"` is not translated to `"`, but it also does
|
||||
// not close the string. Likewise, `\\` is not translated to `\`, but it does
|
||||
// stop the second `\` from stopping a double-quote from closing the string.
|
||||
if (!is_multiline && string_mode == STRING_MODE_RAW &&
|
||||
(lexer->lookahead == '"' || lexer->lookahead == '\\')) {
|
||||
advance(lexer);
|
||||
}
|
||||
} else {
|
||||
lexer->result_symbol = string_mode == STRING_MODE_SIMPLE ? SIMPLE_STRING_MIDDLE : INTERPOLATED_STRING_MIDDLE;
|
||||
lexer->mark_end(lexer);
|
||||
return true;
|
||||
}
|
||||
// During error recovery and dynamic precedence resolution, the external
|
||||
// scanner will be invoked with all valid_symbols set to true, which means
|
||||
// we will be asked to scan a string token when we are not actually in a
|
||||
// string context. Here we detect these cases and return false.
|
||||
} else if (lexer->lookahead == '\n' && !is_multiline) {
|
||||
return false;
|
||||
} else if (lexer->eof(lexer)) {
|
||||
return false;
|
||||
} else {
|
||||
advance(lexer);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static bool detect_comment_start(TSLexer *lexer) {
|
||||
lexer->mark_end(lexer);
|
||||
// Comments should not affect indentation
|
||||
if (lexer->lookahead == '/') {
|
||||
advance(lexer);
|
||||
if (lexer->lookahead == '/' || lexer -> lookahead == '*') {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool scan_word(TSLexer *lexer, const char* const word) {
|
||||
for (uint8_t i = 0; word[i] != '\0'; i++) {
|
||||
if (lexer->lookahead != word[i]) {
|
||||
return false;
|
||||
}
|
||||
advance(lexer);
|
||||
}
|
||||
return !iswalnum(lexer->lookahead);
|
||||
}
|
||||
|
||||
// Returns true if the lookahead starts a leading infix operator — a symbolic
|
||||
// operator or back-ticked identifier followed by whitespace and then a
|
||||
// non-whitespace operand on the same line. Such a line is a continuation of
|
||||
// the previous expression, so neither AUTOMATIC_SEMICOLON nor OUTDENT should
|
||||
// fire ahead of it. Advances the lexer; the caller must not rely on position.
|
||||
static bool is_leading_infix_continuation(TSLexer *lexer) {
|
||||
if (is_op_char(lexer->lookahead)) {
|
||||
advance(lexer);
|
||||
while (is_op_char(lexer->lookahead)) {
|
||||
advance(lexer);
|
||||
}
|
||||
bool found_space = false;
|
||||
while (lexer->lookahead == ' ' || lexer->lookahead == '\t') {
|
||||
advance(lexer);
|
||||
found_space = true;
|
||||
}
|
||||
return found_space && !iswspace(lexer->lookahead) && !lexer->eof(lexer);
|
||||
}
|
||||
if (lexer->lookahead == '`') {
|
||||
advance(lexer);
|
||||
while (lexer->lookahead != '`' && !lexer->eof(lexer)) {
|
||||
advance(lexer);
|
||||
}
|
||||
if (lexer->lookahead != '`') {
|
||||
return false;
|
||||
}
|
||||
advance(lexer);
|
||||
bool found_space = false;
|
||||
while (lexer->lookahead == ' ' || lexer->lookahead == '\t') {
|
||||
advance(lexer);
|
||||
found_space = true;
|
||||
}
|
||||
return found_space && !iswspace(lexer->lookahead) && !lexer->eof(lexer);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
static inline void debug_indents(Scanner *scanner) {
|
||||
LOG(" indents(%d): ", scanner->indents.size);
|
||||
for (unsigned i = 0; i < scanner->indents.size; i++) {
|
||||
LOG("%d ", scanner->indents.contents[i]);
|
||||
}
|
||||
LOG("\n");
|
||||
}
|
||||
|
||||
bool tree_sitter_scala_external_scanner_scan(void *payload, TSLexer *lexer,
|
||||
const bool *valid_symbols) {
|
||||
#ifdef DEBUG
|
||||
{
|
||||
if (valid_symbols[ERROR_SENTINEL]) {
|
||||
LOG("entering tree_sitter_scala_external_scanner_scan. ERROR_SENTINEL is valid\n");
|
||||
} else {
|
||||
char debug_str[1024] = "entering tree_sitter_scala_external_scanner_scan valid symbols: ";
|
||||
for (unsigned i = 0; i < ERROR_SENTINEL; i++) {
|
||||
if (valid_symbols[i]) {
|
||||
strcat(debug_str, token_name[i]);
|
||||
strcat(debug_str, ", ");
|
||||
}
|
||||
}
|
||||
strcat(debug_str, "\n");
|
||||
LOG("%s", debug_str);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
Scanner *scanner = (Scanner *)payload;
|
||||
int16_t prev = scanner->indents.size > 0 ? *array_back(&scanner->indents) : -1;
|
||||
int16_t newline_count = 0;
|
||||
int16_t indentation_size = 0;
|
||||
|
||||
while (iswspace(lexer->lookahead)) {
|
||||
if (lexer->lookahead == '\n') {
|
||||
newline_count++;
|
||||
indentation_size = 0;
|
||||
}
|
||||
else {
|
||||
indentation_size++;
|
||||
}
|
||||
skip(lexer);
|
||||
}
|
||||
|
||||
// Separate from OUTDENT because the scanner cannot distinguish a comma that
|
||||
// terminates an indented block (e.g. `map: x => f(x),`) from one that is
|
||||
// internal to it (e.g. `case EnumCase1, EnumCase2`). By using a distinct
|
||||
// token, tree-sitter only makes it valid in grammar contexts where comma
|
||||
// termination is expected (colon_argument, _indentable_expression).
|
||||
if (valid_symbols[COMMA_OUTDENT] && lexer->lookahead == ',' && prev != -1) {
|
||||
if (scanner->indents.size > 0) {
|
||||
array_pop(&scanner->indents);
|
||||
}
|
||||
lexer->mark_end(lexer);
|
||||
lexer->result_symbol = COMMA_OUTDENT;
|
||||
return true;
|
||||
}
|
||||
|
||||
// Before advancing the lexer, check if we can double outdent
|
||||
if (
|
||||
valid_symbols[OUTDENT] &&
|
||||
(
|
||||
lexer->lookahead == 0 ||
|
||||
(
|
||||
prev != -1 &&
|
||||
(
|
||||
lexer->lookahead == ')' ||
|
||||
lexer->lookahead == ']' ||
|
||||
lexer->lookahead == '}'
|
||||
)
|
||||
) ||
|
||||
(
|
||||
scanner->last_indentation_size != -1 &&
|
||||
prev != -1 &&
|
||||
scanner->last_indentation_size < prev
|
||||
)
|
||||
)
|
||||
) {
|
||||
if (scanner->indents.size > 0) {
|
||||
array_pop(&scanner->indents);
|
||||
}
|
||||
LOG(" pop\n");
|
||||
LOG(" OUTDENT\n");
|
||||
lexer->result_symbol = OUTDENT;
|
||||
return true;
|
||||
}
|
||||
scanner->last_indentation_size = -1;
|
||||
|
||||
if (
|
||||
valid_symbols[INDENT] &&
|
||||
newline_count > 0 &&
|
||||
(
|
||||
scanner->indents.size == 0 ||
|
||||
indentation_size > *array_back(&scanner->indents)
|
||||
)
|
||||
) {
|
||||
if (detect_comment_start(lexer)) {
|
||||
return false;
|
||||
}
|
||||
array_push(&scanner->indents, indentation_size);
|
||||
lexer->result_symbol = INDENT;
|
||||
LOG(" INDENT\n");
|
||||
return true;
|
||||
}
|
||||
|
||||
// This saves the indentation_size and newline_count so it can be used
|
||||
// in subsequent calls for multiple outdent or auto-semicolon.
|
||||
if (valid_symbols[OUTDENT] &&
|
||||
(lexer->lookahead == 0 ||
|
||||
(
|
||||
newline_count > 0 &&
|
||||
prev != -1 &&
|
||||
indentation_size < prev
|
||||
)
|
||||
)
|
||||
) {
|
||||
lexer->mark_end(lexer);
|
||||
if (detect_comment_start(lexer)) {
|
||||
return false;
|
||||
}
|
||||
scanner->last_indentation_size = indentation_size;
|
||||
scanner->last_newline_count = newline_count;
|
||||
if (lexer->eof(lexer)) {
|
||||
scanner->last_column = -1;
|
||||
} else {
|
||||
scanner->last_column = (int16_t)lexer->get_column(lexer);
|
||||
}
|
||||
// Don't close the indented block when the next line starts with a leading
|
||||
// infix operator: that operator continues the previous expression.
|
||||
if (lexer->lookahead != 0 && is_leading_infix_continuation(lexer)) {
|
||||
return false;
|
||||
}
|
||||
if (scanner->indents.size > 0) {
|
||||
array_pop(&scanner->indents);
|
||||
}
|
||||
LOG(" pop\n");
|
||||
LOG(" OUTDENT\n");
|
||||
lexer->result_symbol = OUTDENT;
|
||||
return true;
|
||||
}
|
||||
|
||||
// Recover newline_count from the outdent reset
|
||||
bool is_eof = lexer->eof(lexer);
|
||||
if (
|
||||
(
|
||||
scanner->last_newline_count > 0 &&
|
||||
(is_eof && scanner->last_column == -1)
|
||||
) ||
|
||||
(!is_eof && lexer->get_column(lexer) == (uint32_t)scanner->last_column)
|
||||
) {
|
||||
newline_count += scanner->last_newline_count;
|
||||
}
|
||||
scanner->last_newline_count = 0;
|
||||
|
||||
if (valid_symbols[AUTOMATIC_SEMICOLON] && newline_count > 0) {
|
||||
// AUTOMATIC_SEMICOLON should not be issued in the middle of expressions
|
||||
// Thus, we exit this branch when encountering comments, else/catch clauses, etc.
|
||||
|
||||
lexer->mark_end(lexer);
|
||||
lexer->result_symbol = AUTOMATIC_SEMICOLON;
|
||||
|
||||
// Probably, a multi-line field expression, e.g.
|
||||
// a
|
||||
// .b
|
||||
// .c
|
||||
if (lexer->lookahead == '.') {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Single-line and multi-line comments
|
||||
if (lexer->lookahead == '/') {
|
||||
advance(lexer);
|
||||
if (lexer->lookahead == '/') {
|
||||
return false;
|
||||
}
|
||||
if (lexer->lookahead == '*') {
|
||||
advance(lexer);
|
||||
while (!lexer->eof(lexer)) {
|
||||
if (lexer->lookahead == '*') {
|
||||
advance(lexer);
|
||||
if (lexer->lookahead == '/') {
|
||||
advance(lexer);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
advance(lexer);
|
||||
}
|
||||
}
|
||||
while (iswspace(lexer->lookahead)) {
|
||||
if (lexer->lookahead == '\n' || lexer->lookahead == '\r') {
|
||||
return false;
|
||||
}
|
||||
skip(lexer);
|
||||
}
|
||||
// If some code is present at the same line after comment end,
|
||||
// we should still produce AUTOMATIC_SEMICOLON, e.g. in
|
||||
// val a = 1
|
||||
// /* comment */ val b = 2
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
if (valid_symbols[ELSE]) {
|
||||
return !scan_word(lexer, "else");
|
||||
}
|
||||
|
||||
if (valid_symbols[CATCH]) {
|
||||
if (scan_word(lexer, "catch")) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
if (valid_symbols[FINALLY]) {
|
||||
if (scan_word(lexer, "finally")) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
if (valid_symbols[EXTENDS]) {
|
||||
if (scan_word(lexer, "extends")) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
if (valid_symbols[WITH]) {
|
||||
if (scan_word(lexer, "with")) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
if (valid_symbols[DERIVES]) {
|
||||
if (scan_word(lexer, "derives")) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
if (newline_count > 1) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Don't insert automatic semicolon before leading infix operators:
|
||||
// - symbolic, e.g. || or &&
|
||||
// - back-ticked, e.g. `in`
|
||||
// Only suppress if the operator is followed by horizontal whitespace
|
||||
// and then non-newline content on the same line, meaning it has an operand.
|
||||
if (is_leading_infix_continuation(lexer)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
while (iswspace(lexer->lookahead)) {
|
||||
if (lexer->lookahead == '\n') {
|
||||
newline_count++;
|
||||
}
|
||||
skip(lexer);
|
||||
}
|
||||
|
||||
if (valid_symbols[SIMPLE_STRING_START] && lexer->lookahead == '"') {
|
||||
advance(lexer);
|
||||
lexer->mark_end(lexer);
|
||||
|
||||
if (lexer->lookahead == '"') {
|
||||
advance(lexer);
|
||||
if (lexer->lookahead == '"') {
|
||||
advance(lexer);
|
||||
lexer->result_symbol = SIMPLE_MULTILINE_STRING_START;
|
||||
lexer->mark_end(lexer);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
lexer->result_symbol = SIMPLE_STRING_START;
|
||||
return true;
|
||||
}
|
||||
|
||||
// We need two tokens of lookahead to determine if we are parsing a raw string,
|
||||
// the `raw` and the `"`, which is why we need to do it in the external scanner.
|
||||
if (valid_symbols[RAW_STRING_START] && lexer->lookahead == 'r') {
|
||||
advance(lexer);
|
||||
if (lexer->lookahead == 'a') {
|
||||
advance(lexer);
|
||||
if (lexer->lookahead == 'w') {
|
||||
advance(lexer);
|
||||
if (lexer->lookahead == '"') {
|
||||
lexer->mark_end(lexer);
|
||||
lexer->result_symbol = RAW_STRING_START;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (valid_symbols[SIMPLE_STRING_MIDDLE]) {
|
||||
return scan_string_content(lexer, false, STRING_MODE_SIMPLE);
|
||||
}
|
||||
|
||||
if (valid_symbols[INTERPOLATED_STRING_MIDDLE]) {
|
||||
return scan_string_content(lexer, false, STRING_MODE_INTERPOLATED);
|
||||
}
|
||||
|
||||
if (valid_symbols[RAW_STRING_MIDDLE]) {
|
||||
return scan_string_content(lexer, false, STRING_MODE_RAW);
|
||||
}
|
||||
|
||||
if (valid_symbols[RAW_STRING_MULTILINE_MIDDLE]) {
|
||||
return scan_string_content(lexer, true, STRING_MODE_RAW);
|
||||
}
|
||||
|
||||
if (valid_symbols[INTERPOLATED_MULTILINE_STRING_MIDDLE]) {
|
||||
return scan_string_content(lexer, true, STRING_MODE_INTERPOLATED);
|
||||
}
|
||||
|
||||
// We still need to handle the simple multiline string case, but there is
|
||||
// no `MULTILINE_STRING_MIDDLE` token, and `MULTILINE_STRING_END` is used
|
||||
// by all three of simple raw, and interpolated multiline strings. So this
|
||||
// check needs to come after the `INTERPOLATED_MULTILINE_STRING_MIDDLE` and
|
||||
// `RAW_STRING_MULTILINE_MIDDLE` check, so that we can be sure we are in a
|
||||
// simple multiline string context.
|
||||
if (valid_symbols[MULTILINE_STRING_END]) {
|
||||
return scan_string_content(lexer, true, STRING_MODE_SIMPLE);
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
//
|
||||
@@ -0,0 +1,54 @@
|
||||
#ifndef TREE_SITTER_ALLOC_H_
|
||||
#define TREE_SITTER_ALLOC_H_
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
#include <stdbool.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
|
||||
// Allow clients to override allocation functions
|
||||
#ifdef TREE_SITTER_REUSE_ALLOCATOR
|
||||
|
||||
extern void *(*ts_current_malloc)(size_t size);
|
||||
extern void *(*ts_current_calloc)(size_t count, size_t size);
|
||||
extern void *(*ts_current_realloc)(void *ptr, size_t size);
|
||||
extern void (*ts_current_free)(void *ptr);
|
||||
|
||||
#ifndef ts_malloc
|
||||
#define ts_malloc ts_current_malloc
|
||||
#endif
|
||||
#ifndef ts_calloc
|
||||
#define ts_calloc ts_current_calloc
|
||||
#endif
|
||||
#ifndef ts_realloc
|
||||
#define ts_realloc ts_current_realloc
|
||||
#endif
|
||||
#ifndef ts_free
|
||||
#define ts_free ts_current_free
|
||||
#endif
|
||||
|
||||
#else
|
||||
|
||||
#ifndef ts_malloc
|
||||
#define ts_malloc malloc
|
||||
#endif
|
||||
#ifndef ts_calloc
|
||||
#define ts_calloc calloc
|
||||
#endif
|
||||
#ifndef ts_realloc
|
||||
#define ts_realloc realloc
|
||||
#endif
|
||||
#ifndef ts_free
|
||||
#define ts_free free
|
||||
#endif
|
||||
|
||||
#endif
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif // TREE_SITTER_ALLOC_H_
|
||||
@@ -0,0 +1,330 @@
|
||||
#ifndef TREE_SITTER_ARRAY_H_
|
||||
#define TREE_SITTER_ARRAY_H_
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
#include "./alloc.h"
|
||||
|
||||
#include <assert.h>
|
||||
#include <stdbool.h>
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
#ifdef _MSC_VER
|
||||
#pragma warning(push)
|
||||
#pragma warning(disable : 4101)
|
||||
#elif defined(__GNUC__) || defined(__clang__)
|
||||
#pragma GCC diagnostic push
|
||||
#pragma GCC diagnostic ignored "-Wunused-variable"
|
||||
#endif
|
||||
|
||||
#define Array(T) \
|
||||
struct { \
|
||||
T *contents; \
|
||||
uint32_t size; \
|
||||
uint32_t capacity; \
|
||||
}
|
||||
|
||||
/// Initialize an array.
|
||||
#define array_init(self) \
|
||||
((self)->size = 0, (self)->capacity = 0, (self)->contents = NULL)
|
||||
|
||||
/// Create an empty array.
|
||||
#define array_new() \
|
||||
{ NULL, 0, 0 }
|
||||
|
||||
/// Get a pointer to the element at a given `index` in the array.
|
||||
#define array_get(self, _index) \
|
||||
(assert((uint32_t)(_index) < (self)->size), &(self)->contents[_index])
|
||||
|
||||
/// Get a pointer to the first element in the array.
|
||||
#define array_front(self) array_get(self, 0)
|
||||
|
||||
/// Get a pointer to the last element in the array.
|
||||
#define array_back(self) array_get(self, (self)->size - 1)
|
||||
|
||||
/// Clear the array, setting its size to zero. Note that this does not free any
|
||||
/// memory allocated for the array's contents.
|
||||
#define array_clear(self) ((self)->size = 0)
|
||||
|
||||
/// Reserve `new_capacity` elements of space in the array. If `new_capacity` is
|
||||
/// less than the array's current capacity, this function has no effect.
|
||||
#define array_reserve(self, new_capacity) \
|
||||
((self)->contents = _array__reserve( \
|
||||
(void *)(self)->contents, &(self)->capacity, \
|
||||
array_elem_size(self), new_capacity) \
|
||||
)
|
||||
|
||||
/// Free any memory allocated for this array. Note that this does not free any
|
||||
/// memory allocated for the array's contents.
|
||||
#define array_delete(self) \
|
||||
do { \
|
||||
if ((self)->contents) ts_free((self)->contents); \
|
||||
(self)->contents = NULL; \
|
||||
(self)->size = 0; \
|
||||
(self)->capacity = 0; \
|
||||
} while (0)
|
||||
|
||||
/// Push a new `element` onto the end of the array.
|
||||
#define array_push(self, element) \
|
||||
do { \
|
||||
(self)->contents = _array__grow( \
|
||||
(void *)(self)->contents, (self)->size, &(self)->capacity, \
|
||||
1, array_elem_size(self) \
|
||||
); \
|
||||
(self)->contents[(self)->size++] = (element); \
|
||||
} while(0)
|
||||
|
||||
/// Increase the array's size by `count` elements.
|
||||
/// New elements are zero-initialized.
|
||||
#define array_grow_by(self, count) \
|
||||
do { \
|
||||
if ((count) == 0) break; \
|
||||
(self)->contents = _array__grow( \
|
||||
(self)->contents, (self)->size, &(self)->capacity, \
|
||||
count, array_elem_size(self) \
|
||||
); \
|
||||
memset((self)->contents + (self)->size, 0, (count) * array_elem_size(self)); \
|
||||
(self)->size += (count); \
|
||||
} while (0)
|
||||
|
||||
/// Append all elements from one array to the end of another.
|
||||
#define array_push_all(self, other) \
|
||||
array_extend((self), (other)->size, (other)->contents)
|
||||
|
||||
/// Append `count` elements to the end of the array, reading their values from the
|
||||
/// `contents` pointer.
|
||||
#define array_extend(self, count, other_contents) \
|
||||
(self)->contents = _array__splice( \
|
||||
(void*)(self)->contents, &(self)->size, &(self)->capacity, \
|
||||
array_elem_size(self), (self)->size, 0, count, other_contents \
|
||||
)
|
||||
|
||||
/// Remove `old_count` elements from the array starting at the given `index`. At
|
||||
/// the same index, insert `new_count` new elements, reading their values from the
|
||||
/// `new_contents` pointer.
|
||||
#define array_splice(self, _index, old_count, new_count, new_contents) \
|
||||
(self)->contents = _array__splice( \
|
||||
(void *)(self)->contents, &(self)->size, &(self)->capacity, \
|
||||
array_elem_size(self), _index, old_count, new_count, new_contents \
|
||||
)
|
||||
|
||||
/// Insert one `element` into the array at the given `index`.
|
||||
#define array_insert(self, _index, element) \
|
||||
(self)->contents = _array__splice( \
|
||||
(void *)(self)->contents, &(self)->size, &(self)->capacity, \
|
||||
array_elem_size(self), _index, 0, 1, &(element) \
|
||||
)
|
||||
|
||||
/// Remove one element from the array at the given `index`.
|
||||
#define array_erase(self, _index) \
|
||||
_array__erase((void *)(self)->contents, &(self)->size, array_elem_size(self), _index)
|
||||
|
||||
/// Pop the last element off the array, returning the element by value.
|
||||
#define array_pop(self) ((self)->contents[--(self)->size])
|
||||
|
||||
/// Assign the contents of one array to another, reallocating if necessary.
|
||||
#define array_assign(self, other) \
|
||||
(self)->contents = _array__assign( \
|
||||
(void *)(self)->contents, &(self)->size, &(self)->capacity, \
|
||||
(const void *)(other)->contents, (other)->size, array_elem_size(self) \
|
||||
)
|
||||
|
||||
/// Swap one array with another
|
||||
#define array_swap(self, other) \
|
||||
do { \
|
||||
void *_array_swap_tmp = (void *)(self)->contents; \
|
||||
(self)->contents = (other)->contents; \
|
||||
(other)->contents = _array_swap_tmp; \
|
||||
_array__swap(&(self)->size, &(self)->capacity, \
|
||||
&(other)->size, &(other)->capacity); \
|
||||
} while (0)
|
||||
|
||||
/// Get the size of the array contents
|
||||
#define array_elem_size(self) (sizeof *(self)->contents)
|
||||
|
||||
/// Search a sorted array for a given `needle` value, using the given `compare`
|
||||
/// callback to determine the order.
|
||||
///
|
||||
/// If an existing element is found to be equal to `needle`, then the `index`
|
||||
/// out-parameter is set to the existing value's index, and the `exists`
|
||||
/// out-parameter is set to true. Otherwise, `index` is set to an index where
|
||||
/// `needle` should be inserted in order to preserve the sorting, and `exists`
|
||||
/// is set to false.
|
||||
#define array_search_sorted_with(self, compare, needle, _index, _exists) \
|
||||
_array__search_sorted(self, 0, compare, , needle, _index, _exists)
|
||||
|
||||
/// Search a sorted array for a given `needle` value, using integer comparisons
|
||||
/// of a given struct field (specified with a leading dot) to determine the order.
|
||||
///
|
||||
/// See also `array_search_sorted_with`.
|
||||
#define array_search_sorted_by(self, field, needle, _index, _exists) \
|
||||
_array__search_sorted(self, 0, _compare_int, field, needle, _index, _exists)
|
||||
|
||||
/// Insert a given `value` into a sorted array, using the given `compare`
|
||||
/// callback to determine the order.
|
||||
#define array_insert_sorted_with(self, compare, value) \
|
||||
do { \
|
||||
unsigned _index, _exists; \
|
||||
array_search_sorted_with(self, compare, &(value), &_index, &_exists); \
|
||||
if (!_exists) array_insert(self, _index, value); \
|
||||
} while (0)
|
||||
|
||||
/// Insert a given `value` into a sorted array, using integer comparisons of
|
||||
/// a given struct field (specified with a leading dot) to determine the order.
|
||||
///
|
||||
/// See also `array_search_sorted_by`.
|
||||
#define array_insert_sorted_by(self, field, value) \
|
||||
do { \
|
||||
unsigned _index, _exists; \
|
||||
array_search_sorted_by(self, field, (value) field, &_index, &_exists); \
|
||||
if (!_exists) array_insert(self, _index, value); \
|
||||
} while (0)
|
||||
|
||||
// Private
|
||||
|
||||
// Pointers to individual `Array` fields (rather than the entire `Array` itself)
|
||||
// are passed to the various `_array__*` functions below to address strict aliasing
|
||||
// violations that arises when the _entire_ `Array` struct is passed as `Array(void)*`.
|
||||
//
|
||||
// The `Array` type itself was not altered as a solution in order to avoid breakage
|
||||
// with existing consumers (in particular, parsers with external scanners).
|
||||
|
||||
/// This is not what you're looking for, see `array_erase`.
|
||||
static inline void _array__erase(void* self_contents, uint32_t *size,
|
||||
size_t element_size, uint32_t index) {
|
||||
assert(index < *size);
|
||||
char *contents = (char *)self_contents;
|
||||
memmove(contents + index * element_size, contents + (index + 1) * element_size,
|
||||
(*size - index - 1) * element_size);
|
||||
(*size)--;
|
||||
}
|
||||
|
||||
/// This is not what you're looking for, see `array_reserve`.
|
||||
static inline void *_array__reserve(void *contents, uint32_t *capacity,
|
||||
size_t element_size, uint32_t new_capacity) {
|
||||
void *new_contents = contents;
|
||||
if (new_capacity > *capacity) {
|
||||
if (contents) {
|
||||
new_contents = ts_realloc(contents, new_capacity * element_size);
|
||||
} else {
|
||||
new_contents = ts_malloc(new_capacity * element_size);
|
||||
}
|
||||
*capacity = new_capacity;
|
||||
}
|
||||
return new_contents;
|
||||
}
|
||||
|
||||
/// This is not what you're looking for, see `array_assign`.
|
||||
static inline void *_array__assign(void* self_contents, uint32_t *self_size, uint32_t *self_capacity,
|
||||
const void *other_contents, uint32_t other_size, size_t element_size) {
|
||||
void *new_contents = _array__reserve(self_contents, self_capacity, element_size, other_size);
|
||||
*self_size = other_size;
|
||||
memcpy(new_contents, other_contents, *self_size * element_size);
|
||||
return new_contents;
|
||||
}
|
||||
|
||||
/// This is not what you're looking for, see `array_swap`.
|
||||
static inline void _array__swap(uint32_t *self_size, uint32_t *self_capacity,
|
||||
uint32_t *other_size, uint32_t *other_capacity) {
|
||||
uint32_t tmp_size = *self_size;
|
||||
uint32_t tmp_capacity = *self_capacity;
|
||||
*self_size = *other_size;
|
||||
*self_capacity = *other_capacity;
|
||||
*other_size = tmp_size;
|
||||
*other_capacity = tmp_capacity;
|
||||
}
|
||||
|
||||
/// This is not what you're looking for, see `array_push` or `array_grow_by`.
|
||||
static inline void *_array__grow(void *contents, uint32_t size, uint32_t *capacity,
|
||||
uint32_t count, size_t element_size) {
|
||||
void *new_contents = contents;
|
||||
uint32_t new_size = size + count;
|
||||
if (new_size > *capacity) {
|
||||
uint32_t new_capacity = *capacity * 2;
|
||||
if (new_capacity < 8) new_capacity = 8;
|
||||
if (new_capacity < new_size) new_capacity = new_size;
|
||||
new_contents = _array__reserve(contents, capacity, element_size, new_capacity);
|
||||
}
|
||||
return new_contents;
|
||||
}
|
||||
|
||||
/// This is not what you're looking for, see `array_splice`.
|
||||
static inline void *_array__splice(void *self_contents, uint32_t *size, uint32_t *capacity,
|
||||
size_t element_size,
|
||||
uint32_t index, uint32_t old_count,
|
||||
uint32_t new_count, const void *elements) {
|
||||
uint32_t new_size = *size + new_count - old_count;
|
||||
uint32_t old_end = index + old_count;
|
||||
uint32_t new_end = index + new_count;
|
||||
assert(old_end <= *size);
|
||||
|
||||
void *new_contents = _array__reserve(self_contents, capacity, element_size, new_size);
|
||||
|
||||
char *contents = (char *)new_contents;
|
||||
if (*size > old_end) {
|
||||
memmove(
|
||||
contents + new_end * element_size,
|
||||
contents + old_end * element_size,
|
||||
(*size - old_end) * element_size
|
||||
);
|
||||
}
|
||||
if (new_count > 0) {
|
||||
if (elements) {
|
||||
memcpy(
|
||||
(contents + index * element_size),
|
||||
elements,
|
||||
new_count * element_size
|
||||
);
|
||||
} else {
|
||||
memset(
|
||||
(contents + index * element_size),
|
||||
0,
|
||||
new_count * element_size
|
||||
);
|
||||
}
|
||||
}
|
||||
*size += new_count - old_count;
|
||||
|
||||
return new_contents;
|
||||
}
|
||||
|
||||
/// A binary search routine, based on Rust's `std::slice::binary_search_by`.
|
||||
/// This is not what you're looking for, see `array_search_sorted_with` or `array_search_sorted_by`.
|
||||
#define _array__search_sorted(self, start, compare, suffix, needle, _index, _exists) \
|
||||
do { \
|
||||
*(_index) = start; \
|
||||
*(_exists) = false; \
|
||||
uint32_t size = (self)->size - *(_index); \
|
||||
if (size == 0) break; \
|
||||
int comparison; \
|
||||
while (size > 1) { \
|
||||
uint32_t half_size = size / 2; \
|
||||
uint32_t mid_index = *(_index) + half_size; \
|
||||
comparison = compare(&((self)->contents[mid_index] suffix), (needle)); \
|
||||
if (comparison <= 0) *(_index) = mid_index; \
|
||||
size -= half_size; \
|
||||
} \
|
||||
comparison = compare(&((self)->contents[*(_index)] suffix), (needle)); \
|
||||
if (comparison == 0) *(_exists) = true; \
|
||||
else if (comparison < 0) *(_index) += 1; \
|
||||
} while (0)
|
||||
|
||||
/// Helper macro for the `_sorted_by` routines below. This takes the left (existing)
|
||||
/// parameter by reference in order to work with the generic sorting function above.
|
||||
#define _compare_int(a, b) ((int)*(a) - (int)(b))
|
||||
|
||||
#ifdef _MSC_VER
|
||||
#pragma warning(pop)
|
||||
#elif defined(__GNUC__) || defined(__clang__)
|
||||
#pragma GCC diagnostic pop
|
||||
#endif
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif // TREE_SITTER_ARRAY_H_
|
||||
@@ -0,0 +1,286 @@
|
||||
#ifndef TREE_SITTER_PARSER_H_
|
||||
#define TREE_SITTER_PARSER_H_
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
#include <stdbool.h>
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
|
||||
#define ts_builtin_sym_error ((TSSymbol)-1)
|
||||
#define ts_builtin_sym_end 0
|
||||
#define TREE_SITTER_SERIALIZATION_BUFFER_SIZE 1024
|
||||
|
||||
#ifndef TREE_SITTER_API_H_
|
||||
typedef uint16_t TSStateId;
|
||||
typedef uint16_t TSSymbol;
|
||||
typedef uint16_t TSFieldId;
|
||||
typedef struct TSLanguage TSLanguage;
|
||||
typedef struct TSLanguageMetadata {
|
||||
uint8_t major_version;
|
||||
uint8_t minor_version;
|
||||
uint8_t patch_version;
|
||||
} TSLanguageMetadata;
|
||||
#endif
|
||||
|
||||
typedef struct {
|
||||
TSFieldId field_id;
|
||||
uint8_t child_index;
|
||||
bool inherited;
|
||||
} TSFieldMapEntry;
|
||||
|
||||
// Used to index the field and supertype maps.
|
||||
typedef struct {
|
||||
uint16_t index;
|
||||
uint16_t length;
|
||||
} TSMapSlice;
|
||||
|
||||
typedef struct {
|
||||
bool visible;
|
||||
bool named;
|
||||
bool supertype;
|
||||
} TSSymbolMetadata;
|
||||
|
||||
typedef struct TSLexer TSLexer;
|
||||
|
||||
struct TSLexer {
|
||||
int32_t lookahead;
|
||||
TSSymbol result_symbol;
|
||||
void (*advance)(TSLexer *, bool);
|
||||
void (*mark_end)(TSLexer *);
|
||||
uint32_t (*get_column)(TSLexer *);
|
||||
bool (*is_at_included_range_start)(const TSLexer *);
|
||||
bool (*eof)(const TSLexer *);
|
||||
void (*log)(const TSLexer *, const char *, ...);
|
||||
};
|
||||
|
||||
typedef enum {
|
||||
TSParseActionTypeShift,
|
||||
TSParseActionTypeReduce,
|
||||
TSParseActionTypeAccept,
|
||||
TSParseActionTypeRecover,
|
||||
} TSParseActionType;
|
||||
|
||||
typedef union {
|
||||
struct {
|
||||
uint8_t type;
|
||||
TSStateId state;
|
||||
bool extra;
|
||||
bool repetition;
|
||||
} shift;
|
||||
struct {
|
||||
uint8_t type;
|
||||
uint8_t child_count;
|
||||
TSSymbol symbol;
|
||||
int16_t dynamic_precedence;
|
||||
uint16_t production_id;
|
||||
} reduce;
|
||||
uint8_t type;
|
||||
} TSParseAction;
|
||||
|
||||
typedef struct {
|
||||
uint16_t lex_state;
|
||||
uint16_t external_lex_state;
|
||||
} TSLexMode;
|
||||
|
||||
typedef struct {
|
||||
uint16_t lex_state;
|
||||
uint16_t external_lex_state;
|
||||
uint16_t reserved_word_set_id;
|
||||
} TSLexerMode;
|
||||
|
||||
typedef union {
|
||||
TSParseAction action;
|
||||
struct {
|
||||
uint8_t count;
|
||||
bool reusable;
|
||||
} entry;
|
||||
} TSParseActionEntry;
|
||||
|
||||
typedef struct {
|
||||
int32_t start;
|
||||
int32_t end;
|
||||
} TSCharacterRange;
|
||||
|
||||
struct TSLanguage {
|
||||
uint32_t abi_version;
|
||||
uint32_t symbol_count;
|
||||
uint32_t alias_count;
|
||||
uint32_t token_count;
|
||||
uint32_t external_token_count;
|
||||
uint32_t state_count;
|
||||
uint32_t large_state_count;
|
||||
uint32_t production_id_count;
|
||||
uint32_t field_count;
|
||||
uint16_t max_alias_sequence_length;
|
||||
const uint16_t *parse_table;
|
||||
const uint16_t *small_parse_table;
|
||||
const uint32_t *small_parse_table_map;
|
||||
const TSParseActionEntry *parse_actions;
|
||||
const char * const *symbol_names;
|
||||
const char * const *field_names;
|
||||
const TSMapSlice *field_map_slices;
|
||||
const TSFieldMapEntry *field_map_entries;
|
||||
const TSSymbolMetadata *symbol_metadata;
|
||||
const TSSymbol *public_symbol_map;
|
||||
const uint16_t *alias_map;
|
||||
const TSSymbol *alias_sequences;
|
||||
const TSLexerMode *lex_modes;
|
||||
bool (*lex_fn)(TSLexer *, TSStateId);
|
||||
bool (*keyword_lex_fn)(TSLexer *, TSStateId);
|
||||
TSSymbol keyword_capture_token;
|
||||
struct {
|
||||
const bool *states;
|
||||
const TSSymbol *symbol_map;
|
||||
void *(*create)(void);
|
||||
void (*destroy)(void *);
|
||||
bool (*scan)(void *, TSLexer *, const bool *symbol_whitelist);
|
||||
unsigned (*serialize)(void *, char *);
|
||||
void (*deserialize)(void *, const char *, unsigned);
|
||||
} external_scanner;
|
||||
const TSStateId *primary_state_ids;
|
||||
const char *name;
|
||||
const TSSymbol *reserved_words;
|
||||
uint16_t max_reserved_word_set_size;
|
||||
uint32_t supertype_count;
|
||||
const TSSymbol *supertype_symbols;
|
||||
const TSMapSlice *supertype_map_slices;
|
||||
const TSSymbol *supertype_map_entries;
|
||||
TSLanguageMetadata metadata;
|
||||
};
|
||||
|
||||
static inline bool set_contains(const TSCharacterRange *ranges, uint32_t len, int32_t lookahead) {
|
||||
uint32_t index = 0;
|
||||
uint32_t size = len - index;
|
||||
while (size > 1) {
|
||||
uint32_t half_size = size / 2;
|
||||
uint32_t mid_index = index + half_size;
|
||||
const TSCharacterRange *range = &ranges[mid_index];
|
||||
if (lookahead >= range->start && lookahead <= range->end) {
|
||||
return true;
|
||||
} else if (lookahead > range->end) {
|
||||
index = mid_index;
|
||||
}
|
||||
size -= half_size;
|
||||
}
|
||||
const TSCharacterRange *range = &ranges[index];
|
||||
return (lookahead >= range->start && lookahead <= range->end);
|
||||
}
|
||||
|
||||
/*
|
||||
* Lexer Macros
|
||||
*/
|
||||
|
||||
#ifdef _MSC_VER
|
||||
#define UNUSED __pragma(warning(suppress : 4101))
|
||||
#else
|
||||
#define UNUSED __attribute__((unused))
|
||||
#endif
|
||||
|
||||
#define START_LEXER() \
|
||||
bool result = false; \
|
||||
bool skip = false; \
|
||||
UNUSED \
|
||||
bool eof = false; \
|
||||
int32_t lookahead; \
|
||||
goto start; \
|
||||
next_state: \
|
||||
lexer->advance(lexer, skip); \
|
||||
start: \
|
||||
skip = false; \
|
||||
lookahead = lexer->lookahead;
|
||||
|
||||
#define ADVANCE(state_value) \
|
||||
{ \
|
||||
state = state_value; \
|
||||
goto next_state; \
|
||||
}
|
||||
|
||||
#define ADVANCE_MAP(...) \
|
||||
{ \
|
||||
static const uint16_t map[] = { __VA_ARGS__ }; \
|
||||
for (uint32_t i = 0; i < sizeof(map) / sizeof(map[0]); i += 2) { \
|
||||
if (map[i] == lookahead) { \
|
||||
state = map[i + 1]; \
|
||||
goto next_state; \
|
||||
} \
|
||||
} \
|
||||
}
|
||||
|
||||
#define SKIP(state_value) \
|
||||
{ \
|
||||
skip = true; \
|
||||
state = state_value; \
|
||||
goto next_state; \
|
||||
}
|
||||
|
||||
#define ACCEPT_TOKEN(symbol_value) \
|
||||
result = true; \
|
||||
lexer->result_symbol = symbol_value; \
|
||||
lexer->mark_end(lexer);
|
||||
|
||||
#define END_STATE() return result;
|
||||
|
||||
/*
|
||||
* Parse Table Macros
|
||||
*/
|
||||
|
||||
#define SMALL_STATE(id) ((id) - LARGE_STATE_COUNT)
|
||||
|
||||
#define STATE(id) id
|
||||
|
||||
#define ACTIONS(id) id
|
||||
|
||||
#define SHIFT(state_value) \
|
||||
{{ \
|
||||
.shift = { \
|
||||
.type = TSParseActionTypeShift, \
|
||||
.state = (state_value) \
|
||||
} \
|
||||
}}
|
||||
|
||||
#define SHIFT_REPEAT(state_value) \
|
||||
{{ \
|
||||
.shift = { \
|
||||
.type = TSParseActionTypeShift, \
|
||||
.state = (state_value), \
|
||||
.repetition = true \
|
||||
} \
|
||||
}}
|
||||
|
||||
#define SHIFT_EXTRA() \
|
||||
{{ \
|
||||
.shift = { \
|
||||
.type = TSParseActionTypeShift, \
|
||||
.extra = true \
|
||||
} \
|
||||
}}
|
||||
|
||||
#define REDUCE(symbol_name, children, precedence, prod_id) \
|
||||
{{ \
|
||||
.reduce = { \
|
||||
.type = TSParseActionTypeReduce, \
|
||||
.symbol = symbol_name, \
|
||||
.child_count = children, \
|
||||
.dynamic_precedence = precedence, \
|
||||
.production_id = prod_id \
|
||||
}, \
|
||||
}}
|
||||
|
||||
#define RECOVER() \
|
||||
{{ \
|
||||
.type = TSParseActionTypeRecover \
|
||||
}}
|
||||
|
||||
#define ACCEPT_INPUT() \
|
||||
{{ \
|
||||
.type = TSParseActionTypeAccept \
|
||||
}}
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif // TREE_SITTER_PARSER_H_
|
||||
Reference in New Issue
Block a user