feat(lua): vendor tree-sitter-lua with N-API binding

Vendor tree-sitter-lua v2.1.3 grammar with N-API binding.cc
(rewritten from NAN) for tree-sitter@0.21.x runtime compatibility.
Parser.c regenerated with tree-sitter-cli@0.21.0 (ABI version 14).
Add postinstall build script matching tree-sitter-dart pattern.
This commit is contained in:
Christian C. Berclaz 2026-05-01 22:37:55 +02:00
parent 2ac956656a
commit bfe05d54d6
No known key found for this signature in database
17 changed files with 19554 additions and 1 deletions

View file

@ -36,6 +36,7 @@
"tree-sitter-go": "^0.23.0",
"tree-sitter-java": "^0.23.5",
"tree-sitter-javascript": "^0.23.0",
"tree-sitter-lua": "file:./vendor/tree-sitter-lua",
"tree-sitter-php": "^0.23.0",
"tree-sitter-python": "0.23.4",
"tree-sitter-ruby": "^0.23.1",
@ -5123,6 +5124,10 @@
"license": "MIT",
"optional": true
},
"node_modules/tree-sitter-lua": {
"resolved": "vendor/tree-sitter-lua",
"link": true
},
"node_modules/tree-sitter-php": {
"version": "0.23.12",
"resolved": "https://registry.npmjs.org/tree-sitter-php/-/tree-sitter-php-0.23.12.tgz",
@ -5653,6 +5658,18 @@
}
}
},
"vendor/tree-sitter-lua": {
"version": "2.1.3",
"license": "MIT",
"peerDependencies": {
"tree-sitter": "^0.21.0"
},
"peerDependenciesMeta": {
"tree_sitter": {
"optional": true
}
}
},
"vendor/tree-sitter-proto": {
"version": "0.4.1",
"license": "MIT",

View file

@ -47,7 +47,7 @@
"test:integration": "vitest run test/integration",
"test:watch": "vitest",
"test:coverage": "vitest run --coverage",
"postinstall": "node scripts/build-tree-sitter-dart.cjs && node scripts/build-tree-sitter-proto.cjs",
"postinstall": "node scripts/build-tree-sitter-dart.cjs && node scripts/build-tree-sitter-lua.cjs && node scripts/build-tree-sitter-proto.cjs",
"prepare": "node scripts/build.js",
"prepack": "node scripts/build.js"
},
@ -78,6 +78,7 @@
"tree-sitter-go": "^0.23.0",
"tree-sitter-java": "^0.23.5",
"tree-sitter-javascript": "^0.23.0",
"tree-sitter-lua": "file:./vendor/tree-sitter-lua",
"tree-sitter-php": "^0.23.0",
"tree-sitter-python": "0.23.4",
"tree-sitter-ruby": "^0.23.1",

View file

@ -0,0 +1,42 @@
#!/usr/bin/env node
const fs = require('fs');
const path = require('path');
const { execSync } = require('child_process');
const luaDir = path.join(__dirname, '..', 'node_modules', 'tree-sitter-lua');
const bindingGyp = path.join(luaDir, 'binding.gyp');
const bindingNode = path.join(luaDir, 'build', 'Release', 'tree_sitter_lua_binding.node');
try {
if (!fs.existsSync(bindingGyp) || fs.existsSync(bindingNode)) {
process.exit(0);
}
try {
require.resolve('node-addon-api');
require.resolve('node-gyp-build');
} catch (resolveErr) {
console.warn(
'[tree-sitter-lua] Skipping build: hoisted build deps not resolvable (%s).',
resolveErr.message,
);
console.warn(
'[tree-sitter-lua] Lua parsing will be unavailable. Install without --no-optional and with scripts enabled to build.',
);
process.exit(0);
}
console.log('[tree-sitter-lua] Building native binding...');
execSync('npx node-gyp rebuild', {
cwd: luaDir,
stdio: 'pipe',
timeout: 180000,
});
console.log('[tree-sitter-lua] Native binding built successfully');
} catch (err) {
console.warn('[tree-sitter-lua] Could not build native binding:', err.message);
console.warn(
'[tree-sitter-lua] Lua parsing will be unavailable. Non-Lua functionality is unaffected.',
);
process.exit(0);
}

View file

@ -0,0 +1,26 @@
[package]
name = "tree-sitter-lua"
description = "lua grammar for the tree-sitter parsing library"
version = "0.0.1"
keywords = ["incremental", "parsing", "lua"]
categories = ["parsing", "text-editors"]
repository = "https://github.com/tree-sitter/tree-sitter-lua"
edition = "2018"
license = "MIT"
build = "bindings/rust/build.rs"
include = [
"bindings/rust/*",
"grammar.js",
"queries/*",
"src/*",
]
[lib]
path = "bindings/rust/lib.rs"
[dependencies]
tree-sitter = "~0.20.10"
[build-dependencies]
cc = "1.0"

View file

@ -0,0 +1,30 @@
{
"targets": [
{
"target_name": "tree_sitter_lua_binding",
"dependencies": [
"<!(node -p \"require('node-addon-api').targets\"):node_addon_api_except",
],
"include_dirs": [
"src",
],
"sources": [
"bindings/node/binding.cc",
"src/parser.c",
"src/scanner.c",
],
"conditions": [
["OS!='win'", {
"cflags_c": [
"-std=c11",
],
}, { # OS == "win"
"cflags_c": [
"/std:c11",
"/utf-8",
],
}],
],
}
]
}

View file

@ -0,0 +1,19 @@
#include <napi.h>
typedef struct TSLanguage TSLanguage;
extern "C" TSLanguage *tree_sitter_lua();
// "tree-sitter", "language" hashed with BLAKE2
const napi_type_tag LANGUAGE_TYPE_TAG = {0x8AF2E5212AD58ABF,
0xD5006CAD83ABBA16};
Napi::Object Init(Napi::Env env, Napi::Object exports) {
exports["name"] = Napi::String::New(env, "lua");
auto language = Napi::External<TSLanguage>::New(env, tree_sitter_lua());
language.TypeTag(&LANGUAGE_TYPE_TAG);
exports["language"] = language;
return exports;
}
NODE_API_MODULE(tree_sitter_lua_binding, Init)

View file

@ -0,0 +1,7 @@
const root = require("path").join(__dirname, "..", "..");
module.exports = require("node-gyp-build")(root);
try {
module.exports.nodeTypeInfo = require("../../src/node-types.json");
} catch (_) {}

View file

@ -0,0 +1,40 @@
fn main() {
let src_dir = std::path::Path::new("src");
let mut c_config = cc::Build::new();
c_config.include(&src_dir);
c_config
.flag_if_supported("-Wno-unused-parameter")
.flag_if_supported("-Wno-unused-but-set-variable")
.flag_if_supported("-Wno-trigraphs");
let parser_path = src_dir.join("parser.c");
c_config.file(&parser_path);
// If your language uses an external scanner written in C,
// then include this block of code:
/*
let scanner_path = src_dir.join("scanner.c");
c_config.file(&scanner_path);
println!("cargo:rerun-if-changed={}", scanner_path.to_str().unwrap());
*/
c_config.compile("parser");
println!("cargo:rerun-if-changed={}", parser_path.to_str().unwrap());
// If your language uses an external scanner written in C++,
// then include this block of code:
/*
let mut cpp_config = cc::Build::new();
cpp_config.cpp(true);
cpp_config.include(&src_dir);
cpp_config
.flag_if_supported("-Wno-unused-parameter")
.flag_if_supported("-Wno-unused-but-set-variable");
let scanner_path = src_dir.join("scanner.cc");
cpp_config.file(&scanner_path);
cpp_config.compile("scanner");
println!("cargo:rerun-if-changed={}", scanner_path.to_str().unwrap());
*/
}

View file

@ -0,0 +1,52 @@
//! This crate provides lua language support for the [tree-sitter][] parsing library.
//!
//! Typically, you will use the [language][language func] function to add this language to a
//! tree-sitter [Parser][], and then use the parser to parse some code:
//!
//! ```
//! let code = "";
//! let mut parser = tree_sitter::Parser::new();
//! parser.set_language(tree_sitter_lua::language()).expect("Error loading lua grammar");
//! let tree = parser.parse(code, None).unwrap();
//! ```
//!
//! [Language]: https://docs.rs/tree-sitter/*/tree_sitter/struct.Language.html
//! [language func]: fn.language.html
//! [Parser]: https://docs.rs/tree-sitter/*/tree_sitter/struct.Parser.html
//! [tree-sitter]: https://tree-sitter.github.io/
use tree_sitter::Language;
extern "C" {
fn tree_sitter_lua() -> Language;
}
/// Get the tree-sitter [Language][] for this grammar.
///
/// [Language]: https://docs.rs/tree-sitter/*/tree_sitter/struct.Language.html
pub fn language() -> Language {
unsafe { tree_sitter_lua() }
}
/// The content of the [`node-types.json`][] file for this grammar.
///
/// [`node-types.json`]: https://tree-sitter.github.io/tree-sitter/using-parsers#static-node-types
pub const NODE_TYPES: &'static str = include_str!("../../src/node-types.json");
// Uncomment these to include any queries that this grammar contains
// pub const HIGHLIGHTS_QUERY: &'static str = include_str!("../../queries/highlights.scm");
// pub const INJECTIONS_QUERY: &'static str = include_str!("../../queries/injections.scm");
// pub const LOCALS_QUERY: &'static str = include_str!("../../queries/locals.scm");
// pub const TAGS_QUERY: &'static str = include_str!("../../queries/tags.scm");
#[cfg(test)]
mod tests {
#[test]
fn test_can_load_grammar() {
let mut parser = tree_sitter::Parser::new();
parser
.set_language(super::language())
.expect("Error loading lua language");
}
}

View file

@ -0,0 +1,390 @@
const PREC = {
OR: 1, // "or"
AND: 2, // "and"
COMPARATIVE: 3, // "==" "~=" "<" ">" "<=" ">="
BIT_OR: 4, // "|"
BIT_NOT: 5, // "~"
BIT_AND: 6, // "&"
BIT_SHIFT: 7, // "<<" ">>"
CONCATENATION: 8, // ".."
ADDITIVE: 9, // "+" "-"
MULTIPLICATIVE: 10, // "*" "/" "//" "%"
UNARY: 11, // "not" "#" "-" "~"
POWER: 12, // "^"
CALL: 13,
};
const WHITESPACE = /\s/;
const IDENTIFIER = /[a-zA-Z_][0-9a-zA-Z_]*/;
const DECIMAL_DIGIT = /[0-9]/;
const HEXADECIMAL_DIGIT = /[0-9a-fA-F]/;
const SHEBANG = /#!.*/;
const _numeral = (digit) =>
choice(
repeat1(digit),
seq(repeat1(digit), ".", repeat(digit)),
seq(repeat(digit), ".", repeat1(digit)),
);
const _exponent_part = (...delimiters) =>
seq(
choice(...delimiters),
optional(choice("+", "-")),
repeat1(DECIMAL_DIGIT),
);
const _list = (rule, separator) => seq(rule, repeat(seq(separator, rule)));
module.exports = grammar({
name: "lua",
rules: {
chunk: ($) => seq(optional($.shebang), optional($._block)),
shebang: ($) => SHEBANG,
block: ($) => $._block,
_block: ($) =>
choice(
$.return_statement,
seq(repeat1($.statement), optional($.return_statement)),
),
// statements
return_statement: ($) =>
seq("return", optional($.expression_list), optional($.empty_statement)),
statement: ($) =>
choice(
$.empty_statement,
$.variable_assignment,
$.local_variable_declaration,
$.call,
$.label_statement,
$.goto_statement,
$.break_statement,
$.do_statement,
$.while_statement,
$.repeat_statement,
$.if_statement,
$.for_numeric_statement,
$.for_generic_statement,
$.function_definition_statement,
$.local_function_definition_statement,
),
local_function_definition_statement: ($) =>
seq("local", "function", field("name", $.identifier), $._function_body),
function_definition_statement: ($) =>
seq(
"function",
field(
"name",
choice($.identifier, alias($._table_function_variable, $.variable)),
),
$._function_body,
),
_table_function_variable: ($) =>
seq(
$._table_identifier,
choice($._named_field_identifier, $._method_identifier),
),
_table_identifier: ($) =>
field(
"table",
choice($.identifier, alias($._table_field_variable, $.variable)),
),
_table_field_variable: ($) =>
seq($._table_identifier, $._named_field_identifier),
for_generic_statement: ($) =>
seq(
"for",
field("left", alias($._name_list, $.variable_list)),
"in",
field("right", alias($._value_list, $.expression_list)),
"do",
optional(field("body", $.block)),
"end",
),
_name_list: ($) => _list(field("name", $.identifier), ","),
_value_list: ($) => _list(field("value", $.expression), ","),
for_numeric_statement: ($) =>
seq(
"for",
field("name", $.identifier),
"=",
field("start", $.expression),
",",
field("end", $.expression),
optional(seq(",", field("step", $.expression))),
"do",
optional(field("body", $.block)),
"end",
),
if_statement: ($) =>
seq(
"if",
field("condition", $.expression),
"then",
optional(field("consequence", $.block)),
repeat(field("alternative", $.elseif_clause)),
optional(field("alternative", $.else_clause)),
"end",
),
elseif_clause: ($) =>
seq(
"elseif",
field("condition", $.expression),
"then",
optional(field("consequence", $.block)),
),
else_clause: ($) => seq("else", optional(field("body", $.block))),
repeat_statement: ($) =>
seq(
"repeat",
optional(field("body", $.block)),
"until",
field("condition", $.expression),
),
while_statement: ($) =>
seq(
"while",
field("condition", $.expression),
"do",
optional(field("body", $.block)),
"end",
),
do_statement: ($) => seq("do", optional(field("body", $.block)), "end"),
break_statement: () => "break",
goto_statement: ($) => seq("goto", field("name", $.identifier)),
label_statement: ($) => seq("::", field("name", $.identifier), "::"),
local_variable_declaration: ($) =>
seq(
"local",
alias($._local_variable_list, $.variable_list),
optional(seq("=", alias($._value_list, $.expression_list))),
),
_local_variable_list: ($) =>
_list(alias($._local_variable, $.variable), ","),
_local_variable: ($) =>
seq(field("name", $.identifier), optional($.attribute)),
attribute: ($) => seq("<", field("name", $.identifier), ">"),
variable_assignment: ($) =>
seq($.variable_list, "=", alias($._value_list, $.expression_list)),
variable_list: ($) => _list($.variable, ","),
empty_statement: () => ";",
// expressions
expression: ($) =>
choice(
$.nil,
$.false,
$.true,
$.number,
$.string,
$.vararg_expression,
$.function_definition,
$.prefix_expression,
$.table,
$.unary_expression,
$.binary_expression,
),
binary_expression: ($) =>
choice(
...[
["or", PREC.OR],
["and", PREC.AND],
["==", PREC.COMPARATIVE],
["~=", PREC.COMPARATIVE],
["<", PREC.COMPARATIVE],
[">", PREC.COMPARATIVE],
["<=", PREC.COMPARATIVE],
[">=", PREC.COMPARATIVE],
["|", PREC.BIT_OR],
["~", PREC.BIT_NOT],
["&", PREC.BIT_AND],
["<<", PREC.BIT_SHIFT],
[">>", PREC.BIT_SHIFT],
["+", PREC.ADDITIVE],
["-", PREC.ADDITIVE],
["*", PREC.MULTIPLICATIVE],
["/", PREC.MULTIPLICATIVE],
["//", PREC.MULTIPLICATIVE],
["%", PREC.MULTIPLICATIVE],
].map(([operator, priority]) =>
prec.left(
priority,
seq(
field("left", $.expression),
field("operator", operator),
field("right", $.expression),
),
),
),
...[
["..", PREC.CONCATENATION],
["^", PREC.POWER],
].map(([operator, priority]) =>
prec.right(
priority,
seq(
field("left", $.expression),
field("operator", operator),
field("right", $.expression),
),
),
),
),
unary_expression: ($) =>
choice(
...["not", "#", "-", "~"].map((operator) =>
prec.left(
PREC.UNARY,
seq(field("operator", operator), field("argument", $.expression)),
),
),
),
table: ($) => seq("{", optional($.field_list), "}"),
field_list: ($) =>
seq(_list($.field, $.field_separator), optional($.field_separator)),
field: ($) =>
seq(
optional(
seq(
choice(
field("key", $.identifier),
seq("[", field("key", $.expression), "]"),
),
"=",
),
),
field("value", $.expression),
),
field_separator: () => choice(",", ";"),
prefix: ($) => choice($.variable, $.call, $.parenthesized_expression),
prefix_expression: ($) => $.prefix,
_prefix_expression: ($) => prec(PREC.CALL, $.prefix),
parenthesized_expression: ($) => seq("(", $.expression, ")"),
call: ($) =>
seq(
field(
"function",
choice(
$._prefix_expression,
alias($._table_method_variable, $.variable),
),
),
field("arguments", $.argument_list),
),
_table_method_variable: ($) =>
seq(field("table", $.prefix_expression), $._method_identifier),
_method_identifier: ($) => seq(":", field("method", $.identifier)),
argument_list: ($) =>
choice(seq("(", optional($.expression_list), ")"), $.table, $.string),
expression_list: ($) => _list($.expression, ","),
variable: ($) => choice(field("name", $.identifier), $._table_variable),
_table_variable: ($) =>
seq(
field(
"table",
choice(
$.identifier,
alias($._table_variable, $.variable),
$.call,
$.parenthesized_expression,
),
),
choice($._indexed_field_identifier, $._named_field_identifier),
),
_named_field_identifier: ($) => seq(".", field("field", $.identifier)),
_indexed_field_identifier: ($) =>
seq("[", field("field", $.expression), "]"),
function_definition: ($) => seq("function", $._function_body),
_function_body: ($) =>
seq(
"(",
optional(field("parameters", $.parameter_list)),
")",
optional(field("body", $.block)),
"end",
),
parameter_list: ($) =>
choice(
seq(
_list(field("name", $.identifier), ","),
optional(seq(",", $.vararg_expression)),
),
$.vararg_expression,
),
vararg_expression: () => "...",
string: ($) =>
seq($._string_start, optional($._string_content), $._string_end),
number: ($) =>
token(
seq(
optional("-"),
choice(
seq(_numeral(DECIMAL_DIGIT), optional(_exponent_part("e", "E"))),
seq(
choice("0x", "0X"),
_numeral(HEXADECIMAL_DIGIT),
optional(_exponent_part("p", "P")),
),
),
),
),
true: () => "true",
false: () => "false",
nil: () => "nil",
identifier: () => IDENTIFIER,
// extras
comment: ($) =>
seq($._comment_start, optional($._comment_content), $._comment_end),
},
extras: ($) => [WHITESPACE, $.comment],
externals: ($) => [
$._comment_start,
$._comment_content,
$._comment_end,
$._string_start,
$._string_content,
$._string_end,
],
inline: ($) => [$.prefix, $.field_separator],
supertypes: ($) => [$.prefix_expression, $.expression, $.statement],
word: ($) => $.identifier,
});

View file

@ -0,0 +1,18 @@
{
"name": "tree-sitter-lua",
"version": "2.1.3",
"description": "Lua grammar for tree-sitter",
"repository": "https://github.com/tree-sitter-grammars/tree-sitter-lua",
"license": "MIT",
"main": "bindings/node",
"types": "bindings/node",
"_vendoredBy": "gitnexus - pinned to tree-sitter-grammars/tree-sitter-lua v2.1.3. Binding.cc rewritten from NAN to N-API for tree-sitter@0.21.x compatibility. Build deps hoisted into gitnexus/package.json.",
"peerDependencies": {
"tree-sitter": "^0.21.0"
},
"peerDependenciesMeta": {
"tree_sitter": {
"optional": true
}
}
}

File diff suppressed because it is too large Load diff

File diff suppressed because it is too large Load diff

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,265 @@
#include <tree_sitter/parser.h>
#include <wctype.h>
enum TokenType {
COMMENT_START,
COMMENT_CONTENT,
COMMENT_END,
STRING_START,
STRING_CONTENT,
STRING_END,
};
static void consume(TSLexer *lexer) { lexer->advance(lexer, false); }
static void skip(TSLexer *lexer) { lexer->advance(lexer, true); }
static bool consume_if(TSLexer *lexer, const int32_t character) {
if (lexer->lookahead == character) {
consume(lexer);
return true;
}
return false;
}
const char SQ_STRING_DELIMITER = '\'';
const char DQ_STRING_DELIMITER = '"';
enum StartedToken {
SHORT_COMMENT = 1,
SHORT_SQ_STRING,
SHORT_DQ_STRING,
LONG_COMMENT,
LONG_STRING,
};
struct ScannerState {
enum StartedToken started;
unsigned int depth;
};
void *tree_sitter_lua_external_scanner_create() {
return malloc(sizeof(struct ScannerState));
}
void tree_sitter_lua_external_scanner_destroy(void *payload) { free(payload); }
unsigned int tree_sitter_lua_external_scanner_serialize(void *payload,
char *buffer) {
struct ScannerState *state = payload;
buffer[0] = state->started;
buffer[1] = state->depth;
return 2;
}
void tree_sitter_lua_external_scanner_deserialize(void *payload,
const char *buffer,
unsigned int length) {
if (length == 2) {
struct ScannerState *state = payload;
state->started = buffer[0];
state->depth = buffer[1];
}
}
static unsigned int get_depth(TSLexer *lexer) {
unsigned int current_depth = 0;
while (consume_if(lexer, '=')) {
current_depth += 1;
}
return current_depth;
}
static bool scan_depth(TSLexer *lexer, unsigned int remaining_depth) {
while (remaining_depth > 0 && consume_if(lexer, '=')) {
remaining_depth -= 1;
}
return remaining_depth == 0;
}
bool tree_sitter_lua_external_scanner_scan(void *payload, TSLexer *lexer,
const bool *valid_symbols) {
struct ScannerState *state = payload;
switch (state->started) {
case SHORT_COMMENT: {
// try to match the short comment's end (new line or eof)
if (lexer->lookahead == '\n' || lexer->eof(lexer)) {
if (valid_symbols[COMMENT_END]) {
state->started = 0;
lexer->result_symbol = COMMENT_END;
return true;
}
} else if (valid_symbols[COMMENT_CONTENT]) {
// consume all characters till a short comment's end
do {
consume(lexer);
} while (lexer->lookahead != '\n' && !lexer->eof(lexer));
lexer->result_symbol = COMMENT_CONTENT;
return true;
}
break;
}
case SHORT_SQ_STRING:
case SHORT_DQ_STRING: {
// define the short string's delimiter
const char delimiter = state->started == SHORT_SQ_STRING
? SQ_STRING_DELIMITER
: DQ_STRING_DELIMITER;
// try to match the short string's end (" or ')
if (consume_if(lexer, delimiter)) {
if (valid_symbols[STRING_END]) {
state->started = 0;
lexer->result_symbol = STRING_END;
return true;
}
} else if (valid_symbols[STRING_CONTENT] && lexer->lookahead != '\n' &&
!lexer->eof(lexer)) {
// consume any character till a short string's end, new line or eof
do {
// consume any character after a backslash, unless it's a new line or
// eof
if (consume_if(lexer, '\\') &&
(lexer->lookahead == '\n' || lexer->eof(lexer))) {
break;
}
consume(lexer);
} while (lexer->lookahead != delimiter && lexer->lookahead != '\n' &&
!lexer->eof(lexer));
lexer->result_symbol = STRING_CONTENT;
return true;
}
break;
}
case LONG_COMMENT:
case LONG_STRING: {
const bool is_inside_a_comment = state->started == LONG_COMMENT;
bool some_characters_were_consumed = false;
if (is_inside_a_comment ? valid_symbols[COMMENT_END]
: valid_symbols[STRING_END]) {
// try to match the long comment's/string's end (]=*])
if (consume_if(lexer, ']')) {
if (scan_depth(lexer, state->depth) && consume_if(lexer, ']')) {
state->started = 0;
state->depth = 0;
lexer->result_symbol = is_inside_a_comment ? COMMENT_END : STRING_END;
return true;
}
some_characters_were_consumed = true;
}
}
if (is_inside_a_comment ? valid_symbols[COMMENT_CONTENT]
: valid_symbols[STRING_CONTENT]) {
if (!some_characters_were_consumed) {
if (lexer->eof(lexer)) {
break;
}
// consume the next character as it can't start a long
// comment's/string's end ([)
consume(lexer);
}
// consume any character till a long comment's/string's end or eof
while (true) {
lexer->mark_end(lexer);
if (consume_if(lexer, ']')) {
if (scan_depth(lexer, state->depth)) {
if (consume_if(lexer, ']')) {
break;
}
} else {
continue;
}
}
if (lexer->eof(lexer)) {
break;
}
consume(lexer);
}
lexer->result_symbol =
is_inside_a_comment ? COMMENT_CONTENT : STRING_CONTENT;
return true;
}
break;
}
default: {
// ignore all whitespace
while (iswspace(lexer->lookahead)) {
skip(lexer);
}
if (valid_symbols[COMMENT_START]) {
// try to match a short comment's start (--)
if (consume_if(lexer, '-')) {
if (consume_if(lexer, '-')) {
state->started = SHORT_COMMENT;
// try to match a long comment's start (--[=*[)
lexer->mark_end(lexer);
if (consume_if(lexer, '[')) {
unsigned int possible_depth = get_depth(lexer);
if (consume_if(lexer, '[')) {
state->started = LONG_COMMENT;
state->depth = possible_depth;
lexer->mark_end(lexer);
}
}
lexer->result_symbol = COMMENT_START;
return true;
}
break;
}
}
if (valid_symbols[STRING_START]) {
// try to match a short single-quoted string's start (")
if (consume_if(lexer, SQ_STRING_DELIMITER)) {
state->started = SHORT_SQ_STRING;
}
// try to match a short double-quoted string's start (')
else if (consume_if(lexer, DQ_STRING_DELIMITER)) {
state->started = SHORT_DQ_STRING;
}
// try to match a long string's start ([=*[)
else if (consume_if(lexer, '[')) {
unsigned int possible_depth = get_depth(lexer);
if (consume_if(lexer, '[')) {
state->started = LONG_STRING;
state->depth = possible_depth;
}
}
if (state->started) {
lexer->result_symbol = STRING_START;
return true;
}
}
break;
}
}
return false;
}

View file

@ -0,0 +1,234 @@
#ifndef TREE_SITTER_PARSER_H_
#define TREE_SITTER_PARSER_H_
#ifdef __cplusplus
extern "C" {
#endif
#include <stdbool.h>
#include <stdint.h>
#include <stdlib.h>
#define ts_builtin_sym_error ((TSSymbol) - 1)
#define ts_builtin_sym_end 0
#define TREE_SITTER_SERIALIZATION_BUFFER_SIZE 1024
#ifndef TREE_SITTER_API_H_
typedef uint16_t TSStateId;
typedef uint16_t TSSymbol;
typedef uint16_t TSFieldId;
typedef struct TSLanguage TSLanguage;
#endif
typedef struct {
TSFieldId field_id;
uint8_t child_index;
bool inherited;
} TSFieldMapEntry;
typedef struct {
uint16_t index;
uint16_t length;
} TSFieldMapSlice;
typedef struct {
bool visible;
bool named;
bool supertype;
} TSSymbolMetadata;
typedef struct TSLexer TSLexer;
struct TSLexer {
int32_t lookahead;
TSSymbol result_symbol;
void (*advance)(TSLexer *, bool);
void (*mark_end)(TSLexer *);
uint32_t (*get_column)(TSLexer *);
bool (*is_at_included_range_start)(const TSLexer *);
bool (*eof)(const TSLexer *);
};
typedef enum {
TSParseActionTypeShift,
TSParseActionTypeReduce,
TSParseActionTypeAccept,
TSParseActionTypeRecover,
} TSParseActionType;
typedef union {
struct {
uint8_t type;
TSStateId state;
bool extra;
bool repetition;
} shift;
struct {
uint8_t type;
uint8_t child_count;
TSSymbol symbol;
int16_t dynamic_precedence;
uint16_t production_id;
} reduce;
uint8_t type;
} TSParseAction;
typedef struct {
uint16_t lex_state;
uint16_t external_lex_state;
} TSLexMode;
typedef union {
TSParseAction action;
struct {
uint8_t count;
bool reusable;
} entry;
} TSParseActionEntry;
struct TSLanguage {
uint32_t version;
uint32_t symbol_count;
uint32_t alias_count;
uint32_t token_count;
uint32_t external_token_count;
uint32_t state_count;
uint32_t large_state_count;
uint32_t production_id_count;
uint32_t field_count;
uint16_t max_alias_sequence_length;
const uint16_t *parse_table;
const uint16_t *small_parse_table;
const uint32_t *small_parse_table_map;
const TSParseActionEntry *parse_actions;
const char *const *symbol_names;
const char *const *field_names;
const TSFieldMapSlice *field_map_slices;
const TSFieldMapEntry *field_map_entries;
const TSSymbolMetadata *symbol_metadata;
const TSSymbol *public_symbol_map;
const uint16_t *alias_map;
const TSSymbol *alias_sequences;
const TSLexMode *lex_modes;
bool (*lex_fn)(TSLexer *, TSStateId);
bool (*keyword_lex_fn)(TSLexer *, TSStateId);
TSSymbol keyword_capture_token;
struct {
const bool *states;
const TSSymbol *symbol_map;
void *(*create)(void);
void (*destroy)(void *);
bool (*scan)(void *, TSLexer *, const bool *symbol_whitelist);
unsigned (*serialize)(void *, char *);
void (*deserialize)(void *, const char *, unsigned);
} external_scanner;
const TSStateId *primary_state_ids;
};
/*
* Lexer Macros
*/
#ifdef _MSC_VER
#define UNUSED __pragma(warning(suppress : 4101))
#else
#define UNUSED __attribute__((unused))
#endif
#define START_LEXER() \
bool result = false; \
bool skip = false; \
UNUSED \
bool eof = false; \
int32_t lookahead; \
goto start; \
next_state: \
lexer->advance(lexer, skip); \
start: \
skip = false; \
lookahead = lexer->lookahead;
#define ADVANCE(state_value) \
{ \
state = state_value; \
goto next_state; \
}
#define SKIP(state_value) \
{ \
skip = true; \
state = state_value; \
goto next_state; \
}
#define ACCEPT_TOKEN(symbol_value) \
result = true; \
lexer->result_symbol = symbol_value; \
lexer->mark_end(lexer);
#define END_STATE() return result;
/*
* Parse Table Macros
*/
#define SMALL_STATE(id) ((id) - LARGE_STATE_COUNT)
#define STATE(id) id
#define ACTIONS(id) id
#define SHIFT(state_value) \
{ \
{ \
.shift = {.type = TSParseActionTypeShift, .state = (state_value) } \
} \
}
#define SHIFT_REPEAT(state_value) \
{ \
{ \
.shift = { \
.type = TSParseActionTypeShift, \
.state = (state_value), \
.repetition = true \
} \
} \
}
#define SHIFT_EXTRA() \
{ \
{ \
.shift = {.type = TSParseActionTypeShift, .extra = true } \
} \
}
#define REDUCE(symbol_val, child_count_val, ...) \
{ \
{ \
.reduce = {.type = TSParseActionTypeReduce, \
.symbol = symbol_val, \
.child_count = child_count_val, \
__VA_ARGS__}, \
} \
}
#define RECOVER() \
{ \
{ \
.type = TSParseActionTypeRecover \
} \
}
#define ACCEPT_INPUT() \
{ \
{ \
.type = TSParseActionTypeAccept \
} \
}
#ifdef __cplusplus
}
#endif
#endif // TREE_SITTER_PARSER_H_

View file

@ -0,0 +1,228 @@
#ifndef TREE_SITTER_PARSER_H_
#define TREE_SITTER_PARSER_H_
#ifdef __cplusplus
extern "C" {
#endif
#include <stdbool.h>
#include <stdint.h>
#include <stdlib.h>
#define ts_builtin_sym_error ((TSSymbol) - 1)
#define ts_builtin_sym_end 0
#define TREE_SITTER_SERIALIZATION_BUFFER_SIZE 1024
typedef uint16_t TSStateId;
#ifndef TREE_SITTER_API_H_
typedef uint16_t TSSymbol;
typedef uint16_t TSFieldId;
typedef struct TSLanguage TSLanguage;
#endif
typedef struct {
TSFieldId field_id;
uint8_t child_index;
bool inherited;
} TSFieldMapEntry;
typedef struct {
uint16_t index;
uint16_t length;
} TSFieldMapSlice;
typedef struct {
bool visible;
bool named;
bool supertype;
} TSSymbolMetadata;
typedef struct TSLexer TSLexer;
struct TSLexer {
int32_t lookahead;
TSSymbol result_symbol;
void (*advance)(TSLexer *, bool);
void (*mark_end)(TSLexer *);
uint32_t (*get_column)(TSLexer *);
bool (*is_at_included_range_start)(const TSLexer *);
bool (*eof)(const TSLexer *);
};
typedef enum {
TSParseActionTypeShift,
TSParseActionTypeReduce,
TSParseActionTypeAccept,
TSParseActionTypeRecover,
} TSParseActionType;
typedef union {
struct {
uint8_t type;
TSStateId state;
bool extra;
bool repetition;
} shift;
struct {
uint8_t type;
uint8_t child_count;
TSSymbol symbol;
int16_t dynamic_precedence;
uint16_t production_id;
} reduce;
uint8_t type;
} TSParseAction;
typedef struct {
uint16_t lex_state;
uint16_t external_lex_state;
} TSLexMode;
typedef union {
TSParseAction action;
struct {
uint8_t count;
bool reusable;
} entry;
} TSParseActionEntry;
struct TSLanguage {
uint32_t version;
uint32_t symbol_count;
uint32_t alias_count;
uint32_t token_count;
uint32_t external_token_count;
uint32_t state_count;
uint32_t large_state_count;
uint32_t production_id_count;
uint32_t field_count;
uint16_t max_alias_sequence_length;
const uint16_t *parse_table;
const uint16_t *small_parse_table;
const uint32_t *small_parse_table_map;
const TSParseActionEntry *parse_actions;
const char *const *symbol_names;
const char *const *field_names;
const TSFieldMapSlice *field_map_slices;
const TSFieldMapEntry *field_map_entries;
const TSSymbolMetadata *symbol_metadata;
const TSSymbol *public_symbol_map;
const uint16_t *alias_map;
const TSSymbol *alias_sequences;
const TSLexMode *lex_modes;
bool (*lex_fn)(TSLexer *, TSStateId);
bool (*keyword_lex_fn)(TSLexer *, TSStateId);
TSSymbol keyword_capture_token;
struct {
const bool *states;
const TSSymbol *symbol_map;
void *(*create)(void);
void (*destroy)(void *);
bool (*scan)(void *, TSLexer *, const bool *symbol_whitelist);
unsigned (*serialize)(void *, char *);
void (*deserialize)(void *, const char *, unsigned);
} external_scanner;
const TSStateId *primary_state_ids;
};
/*
* Lexer Macros
*/
#define START_LEXER() \
bool result = false; \
bool skip = false; \
bool eof = false; \
int32_t lookahead; \
goto start; \
next_state: \
lexer->advance(lexer, skip); \
start: \
skip = false; \
lookahead = lexer->lookahead;
#define ADVANCE(state_value) \
{ \
state = state_value; \
goto next_state; \
}
#define SKIP(state_value) \
{ \
skip = true; \
state = state_value; \
goto next_state; \
}
#define ACCEPT_TOKEN(symbol_value) \
result = true; \
lexer->result_symbol = symbol_value; \
lexer->mark_end(lexer);
#define END_STATE() return result;
/*
* Parse Table Macros
*/
#define SMALL_STATE(id) id - LARGE_STATE_COUNT
#define STATE(id) id
#define ACTIONS(id) id
#define SHIFT(state_value) \
{ \
{ \
.shift = {.type = TSParseActionTypeShift, .state = state_value } \
} \
}
#define SHIFT_REPEAT(state_value) \
{ \
{ \
.shift = { \
.type = TSParseActionTypeShift, \
.state = state_value, \
.repetition = true \
} \
} \
}
#define SHIFT_EXTRA() \
{ \
{ \
.shift = {.type = TSParseActionTypeShift, .extra = true } \
} \
}
#define REDUCE(symbol_val, child_count_val, ...) \
{ \
{ \
.reduce = {.type = TSParseActionTypeReduce, \
.symbol = symbol_val, \
.child_count = child_count_val, \
__VA_ARGS__}, \
} \
}
#define RECOVER() \
{ \
{ \
.type = TSParseActionTypeRecover \
} \
}
#define ACCEPT_INPUT() \
{ \
{ \
.type = TSParseActionTypeAccept \
} \
}
#ifdef __cplusplus
}
#endif
#endif // TREE_SITTER_PARSER_H_