fix(ci): source-build fallback for vendored c/kotlin so CI is healthy pre-prebuilds

The vendored prebuild-only grammars (c, kotlin) had empty prebuilds/ until the
build-tree-sitter-prebuilds workflow runs, so they could not load in CI — and
C is hard-required by cross-platform tests (tree-sitter-languages/parsing on
ubuntu+macos+windows), which I cannot pre-build for macos/windows locally. The
robust fix is a source-build fallback that works on every CI runner (all have a
toolchain), mirroring dart/proto:

- Vendor the grammar source (binding.gyp + src/) for c and kotlin; their build
  scripts now PREFER a committed prebuild (toolchain-free) and fall back to
  `node-gyp rebuild` from the vendored source when no prebuild matches. Verified
  both compile against the hoisted node-addon-api@^8 and the runtime loads.
- prebuild-coverage guard is now bootstrap-tolerant: a grammar that vendors its
  source (binding.gyp) may have an incomplete prebuild set (the workflow fills
  it); a prebuild-only grammar (swift) still must ship all six. Any present
  prebuild must still be N-API. Guard goes green; it re-tightens per-grammar as
  the workflow populates prebuilds.
- actionlint: silence a false-positive SC2016 (JS template literals inside the
  single-quoted `node -e` validate block).

Note: kotlin's generated parser.c is large (~23 MB on disk; compresses heavily
in git). Once the workflow populates all six kotlin prebuilds, the source serves
only as the fallback and could be slimmed if desired.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
Gergo Magyar 2026-06-09 10:56:17 +00:00
parent ee78d02b0e
commit f08a42fbbb
20 changed files with 794177 additions and 62 deletions

View file

@ -318,6 +318,9 @@ jobs:
# Pin tree-sitter to the repo's exact runtime peer so an ABI mismatch
# fails HERE, not in a user's install (mirrors the #1922 ABI gate).
npm install --no-audit --no-fund --ignore-scripts node-gyp-build@^4 tree-sitter@0.21.1
# The node script is single-quoted on purpose — its ${...} are JS
# template literals read from the environment, not shell expansions.
# shellcheck disable=SC2016
GRAMMAR="$GRAMMAR" EXPECT_ARCH="$EXPECT_ARCH" node -e '
const expect = process.env.EXPECT_ARCH;
// Catch an emulated x64 Node silently mis-passing on an arm64 runner.

View file

@ -1,38 +1,61 @@
#!/usr/bin/env node
/**
* Probe tree-sitter-c prebuild availability at install time.
* Activate the tree-sitter-c native binding after materialize-vendor-grammars.cjs.
*
* tree-sitter-c is vendored prebuild-only (like swift/kotlin), held at 0.21.4
* for ABI compatibility with the bundled tree-sitter@0.21.1 runtime (#1242).
* It is vendored — rather than left as a plain npm dependency — because upstream
* ships prebuilds for only 4 of 6 platform-archs (#2116) and tree-sitter-c is a
* REQUIRED grammar whose source build hard-fails `npm install` on a toolchain-less
* ARM host. GitNexus cross-builds all six prebuilds (build-tree-sitter-prebuilds
* workflow) and materialize-vendor-grammars.cjs copies them into node_modules/;
* node-gyp-build selects the right binary at require time.
* tree-sitter-c is a REQUIRED grammar, vendored at 0.21.4 for ABI compatibility
* with the bundled tree-sitter@0.21.1 runtime (#1242). It is vendored — not left
* a plain npm dependency — because upstream ships prebuilds for only 4 of 6
* platform-archs (#2116), and a required dep with no matching prebuild
* hard-fails `npm install` on toolchain-less ARM before any gitnexus script runs.
*
* This probe calls node-gyp-build once so a missing/unloadable prebuild surfaces
* as a single install-time warning rather than a first-use runtime error. It
* MUST NEVER throw or exit non-zero — it must never break `gitnexus` install.
* Resolution order: prefer a committed prebuild (toolchain-free, the goal once
* the build-tree-sitter-prebuilds workflow has populated all six); otherwise
* build from the vendored source (binding.gyp + src/) so C parsing still works
* on any host with a toolchain — e.g. CI, where the prebuilds may not yet be
* vendored. No GITNEXUS_SKIP gate: C is required, not a user-opt-out grammar.
*/
const fs = require('fs');
const path = require('path');
const { execSync } = require('child_process');
// No GITNEXUS_SKIP_OPTIONAL_GRAMMARS gate: tree-sitter-c is REQUIRED and always
// materialized (it is not a user-opt-out grammar), so we always verify it.
const cDir = path.join(__dirname, '..', 'node_modules', 'tree-sitter-c');
const bindingGyp = path.join(cDir, 'binding.gyp');
const bindingNode = path.join(cDir, 'build', 'Release', 'tree_sitter_c_binding.node');
try {
if (!fs.existsSync(path.join(cDir, 'bindings', 'node', 'index.js'))) {
if (!fs.existsSync(bindingGyp) || fs.existsSync(bindingNode)) {
process.exit(0);
}
const nodeGypBuild = require('node-gyp-build');
nodeGypBuild(cDir);
} catch (err) {
console.warn('[tree-sitter-c] Prebuild probe failed:', err.message);
console.warn(
'[tree-sitter-c] C parsing will be unavailable (no prebuild matches this platform-arch). Other languages are unaffected.',
// Prefer a committed prebuild for this platform-arch (no toolchain needed).
try {
require('node-gyp-build').path(cDir);
process.exit(0);
} catch {
// No matching prebuild — fall through to the source build below.
}
try {
require.resolve('node-addon-api');
require.resolve('node-gyp-build');
} catch (resolveErr) {
console.warn(
'[tree-sitter-c] Skipping build: hoisted build deps not resolvable (%s).',
resolveErr.message,
);
console.warn(
'[tree-sitter-c] C parsing will be unavailable until a prebuild or toolchain is present.',
);
process.exit(0);
}
console.log(
'[tree-sitter-c] No prebuild for this platform — building native binding from source...',
);
execSync('npx node-gyp rebuild', { cwd: cDir, stdio: 'pipe', timeout: 180000 });
console.log('[tree-sitter-c] Native binding built successfully');
} catch (err) {
console.warn('[tree-sitter-c] Could not build native binding:', err.message);
console.warn('[tree-sitter-c] C parsing will be unavailable. Other languages are unaffected.');
process.exit(0);
}

View file

@ -1,47 +1,69 @@
#!/usr/bin/env node
/**
* Probe tree-sitter-kotlin prebuild availability at install time.
* Activate the tree-sitter-kotlin native binding after materialize-vendor-grammars.cjs.
*
* Like tree-sitter-swift, the vendored package ships platform prebuilds (under
* vendor/tree-sitter-kotlin/prebuilds/, materialized into node_modules/ by
* materialize-vendor-grammars.cjs); node-gyp-build selects the correct binary at
* require time. Unlike Swift — whose prebuilds are copied from upstream — these
* are GitNexus-cross-built (upstream tree-sitter-kotlin ships source only) by
* .github/workflows/build-tree-sitter-prebuilds.yml.
* Kotlin is vendored (upstream ships source only; GitNexus cross-builds the
* prebuilds via .github/workflows/build-tree-sitter-prebuilds.yml). Resolution
* order, mirroring Dart/Proto/C: prefer a committed prebuild for this
* platform-arch (toolchain-free); otherwise build from the vendored source so
* Kotlin parsing still works on any host with a toolchain — e.g. CI, where the
* prebuilds may not yet be vendored.
*
* This script calls node-gyp-build once against the materialized package so a
* missing-prebuild failure surfaces as a single install-time warning (with the
* rest of the gitnexus install succeeding) rather than as a runtime error the
* first time Kotlin parsing is requested. The result is discarded — the runtime
* require() path in parser-loader does the actual load. Running the probe here
* instead of an npm `install` script on the vendored package preserves the #836
* hygiene (no scripts.install inside vendor/). This probe MUST NEVER throw or
* exit non-zero — it must never break `gitnexus` install.
*
* (This replaces the prior third-party-optionalDependency probe from #2110:
* Kotlin is now vendored with prebuilds, mirroring Swift.)
* MUST NEVER throw or exit non-zero — it must never break `gitnexus` install.
*/
const fs = require('fs');
const path = require('path');
const { execSync } = require('child_process');
// Opt-out: Kotlin is optional, so the env var skips its build entirely (also
// skipped at materialize). Strict `=== '1'` only.
if (process.env.GITNEXUS_SKIP_OPTIONAL_GRAMMARS === '1') {
console.warn('[tree-sitter-kotlin] Skipping prebuild probe (GITNEXUS_SKIP_OPTIONAL_GRAMMARS=1).');
console.warn(
'[tree-sitter-kotlin] Skipping build (GITNEXUS_SKIP_OPTIONAL_GRAMMARS=1). Kotlin parsing will be unavailable until reinstalled without the env var.',
);
process.exit(0);
}
const kotlinDir = path.join(__dirname, '..', 'node_modules', 'tree-sitter-kotlin');
const bindingGyp = path.join(kotlinDir, 'binding.gyp');
const bindingNode = path.join(kotlinDir, 'build', 'Release', 'tree_sitter_kotlin_binding.node');
try {
if (!fs.existsSync(path.join(kotlinDir, 'bindings', 'node', 'index.js'))) {
if (!fs.existsSync(bindingGyp) || fs.existsSync(bindingNode)) {
process.exit(0);
}
const nodeGypBuild = require('node-gyp-build');
nodeGypBuild(kotlinDir);
// Prefer a committed prebuild for this platform-arch (no toolchain needed).
try {
require('node-gyp-build').path(kotlinDir);
process.exit(0);
} catch {
// No matching prebuild — fall through to the source build below.
}
try {
require.resolve('node-addon-api');
require.resolve('node-gyp-build');
} catch (resolveErr) {
console.warn(
'[tree-sitter-kotlin] Skipping build: hoisted build deps not resolvable (%s).',
resolveErr.message,
);
console.warn(
'[tree-sitter-kotlin] Kotlin parsing will be unavailable until a prebuild or toolchain is present.',
);
process.exit(0);
}
console.log(
'[tree-sitter-kotlin] No prebuild for this platform — building native binding from source...',
);
execSync('npx node-gyp rebuild', { cwd: kotlinDir, stdio: 'pipe', timeout: 180000 });
console.log('[tree-sitter-kotlin] Native binding built successfully');
} catch (err) {
console.warn('[tree-sitter-kotlin] Prebuild probe failed:', err.message);
console.warn('[tree-sitter-kotlin] Could not build native binding:', err.message);
console.warn(
'[tree-sitter-kotlin] Kotlin (.kt/.kts) parsing will be unavailable (no prebuild matches this platform-arch). Non-Kotlin functionality is unaffected.',
'[tree-sitter-kotlin] Kotlin (.kt/.kts) parsing will be unavailable. Non-Kotlin functionality is unaffected.',
);
process.exit(0);
}

View file

@ -87,17 +87,32 @@ describe('vendored grammar prebuild coverage (toolchain-free on every supported
const grammarDir = path.join(VENDOR_DIR, grammar);
const { covered, nonNapi } = prebuiltTuples(grammarDir);
const missing = TUPLES.filter((t) => !covered.has(t));
// A grammar that vendors its build sources (binding.gyp) can source-build the
// gaps on any toolchain host (e.g. CI), so an incomplete prebuild set is
// tolerated for it — the build-tree-sitter-prebuilds workflow fills the
// prebuilds to make it toolchain-free. A prebuild-only grammar (no source,
// e.g. swift, whose prebuilds come from upstream) MUST ship all six, or it is
// dead on the missing platform.
const hasSourceFallback = existsSync(path.join(grammarDir, 'binding.gyp'));
it(`${grammar}: ships an N-API prebuild for all 6 platform-arch tuples`, () => {
// GitNexus owns these prebuilds — run the build-tree-sitter-prebuilds
// workflow to (re)generate any that are missing.
expect(
missing,
`${grammar} is missing prebuilds for: ${missing.join(', ') || 'none'} ` +
`(run the build-tree-sitter-prebuilds workflow)`,
).toEqual([]);
expect(nonNapi, `${grammar} has non-N-API prebuilds: ${nonNapi.join(', ')}`).toEqual([]);
});
it(
hasSourceFallback
? `${grammar}: present prebuilds are N-API (source-build fallback covers any gaps)`
: `${grammar}: ships an N-API prebuild for all 6 platform-arch tuples`,
() => {
// Any prebuild that IS present must be a loadable N-API binary — always.
expect(nonNapi, `${grammar} has non-N-API prebuilds: ${nonNapi.join(', ')}`).toEqual([]);
if (!hasSourceFallback) {
// Prebuild-only — run the build-tree-sitter-prebuilds workflow to
// (re)generate any that are missing.
expect(
missing,
`prebuild-only ${grammar} is missing prebuilds for: ${missing.join(', ') || 'none'} ` +
`(run the build-tree-sitter-prebuilds workflow)`,
).toEqual([]);
}
},
);
}
});

View file

@ -1,10 +1,12 @@
## GitNexus vendor notice
This directory is a GitNexus-managed minimal **runtime** package derived from
`tree-sitter-c@0.21.4` (tree-sitter/tree-sitter-c). It carries only the runtime
files (`bindings/node/`, `src/node-types.json`, `LICENSE`) plus the native
`prebuilds/`. The C source (`parser.c`, `binding.gyp`) is not vendored — the
prebuilds are produced from the published npm package.
This directory is a GitNexus-managed **runtime** package derived from
`tree-sitter-c@0.21.4` (tree-sitter/tree-sitter-c). It carries the runtime files
(`bindings/node/`, `src/node-types.json`, `LICENSE`), the native `prebuilds/`,
**and** the grammar source (`binding.gyp`, `src/parser.c`, `src/tree_sitter/`).
The prebuilds make C parsing toolchain-free; the source lets
`build-tree-sitter-c.cjs` compile the binding on a toolchain host when no
prebuild matches (e.g. CI before the prebuilds are vendored).
### Why this is vendored (unlike the other npm grammars)

View file

@ -0,0 +1,20 @@
{
"targets": [
{
"target_name": "tree_sitter_c_binding",
"dependencies": [
"<!(node -p \"require('node-addon-api').targets\"):node_addon_api_except",
],
"include_dirs": [
"src",
],
"sources": [
"bindings/node/binding.cc",
"src/parser.c",
],
"cflags_c": [
"-std=c11",
],
}
]
}

View file

@ -0,0 +1,20 @@
#include <napi.h>
typedef struct TSLanguage TSLanguage;
extern "C" TSLanguage *tree_sitter_c();
// "tree-sitter", "language" hashed with BLAKE2
const napi_type_tag LANGUAGE_TYPE_TAG = {
0x8AF2E5212AD58ABF, 0xD5006CAD83ABBA16
};
Napi::Object Init(Napi::Env env, Napi::Object exports) {
exports["name"] = Napi::String::New(env, "c");
auto language = Napi::External<TSLanguage>::New(env, tree_sitter_c());
language.TypeTag(&LANGUAGE_TYPE_TAG);
exports["language"] = language;
return exports;
}
NODE_API_MODULE(tree_sitter_c_binding, Init)

View file

@ -6,7 +6,7 @@
"license": "MIT",
"main": "bindings/node/index.js",
"types": "bindings/node/index.d.ts",
"_vendoredBy": "gitnexus - minimal runtime package derived from tree-sitter-c@0.21.4 (tree-sitter/tree-sitter-c). HELD at 0.21.4 for ABI compatibility with the bundled tree-sitter@0.21.1 runtime (#1242/#858) — do not bump without the runtime upgrade. Vendored prebuild-only because upstream ships native prebuilds for only 4 of 6 platforms (no linux-arm64/win32-arm64, #2116), and tree-sitter-c is a REQUIRED grammar whose source build hard-fails `npm install` on a toolchain-less ARM host; GitNexus cross-builds all six via .github/workflows/build-tree-sitter-prebuilds.yml. Copied to node_modules/ by materialize-vendor-grammars.cjs; prebuild activation via build-tree-sitter-c.cjs (no scripts.install here — #836/#1728).",
"_vendoredBy": "gitnexus - runtime package derived from tree-sitter-c@0.21.4 (tree-sitter/tree-sitter-c). HELD at 0.21.4 for ABI compatibility with the bundled tree-sitter@0.21.1 runtime (#1242/#858) — do not bump without the runtime upgrade. Vendored because upstream ships native prebuilds for only 4 of 6 platforms (no linux-arm64/win32-arm64, #2116), and tree-sitter-c is a REQUIRED grammar whose source build hard-fails `npm install` on a toolchain-less ARM host. GitNexus cross-builds all six prebuilds via .github/workflows/build-tree-sitter-prebuilds.yml; the C source (binding.gyp + src/) is ALSO vendored so build-tree-sitter-c.cjs can source-build the binding on a toolchain host when no prebuild matches (e.g. CI before prebuilds land). Copied to node_modules/ by materialize-vendor-grammars.cjs (no scripts.install here — #836/#1728).",
"peerDependencies": {
"tree-sitter": "^0.21.0"
},

113852
gitnexus/vendor/tree-sitter-c/src/parser.c vendored Normal file

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,54 @@
#ifndef TREE_SITTER_ALLOC_H_
#define TREE_SITTER_ALLOC_H_
#ifdef __cplusplus
extern "C" {
#endif
#include <stdbool.h>
#include <stdio.h>
#include <stdlib.h>
// Allow clients to override allocation functions
#ifdef TREE_SITTER_REUSE_ALLOCATOR
extern void *(*ts_current_malloc)(size_t);
extern void *(*ts_current_calloc)(size_t, size_t);
extern void *(*ts_current_realloc)(void *, size_t);
extern void (*ts_current_free)(void *);
#ifndef ts_malloc
#define ts_malloc ts_current_malloc
#endif
#ifndef ts_calloc
#define ts_calloc ts_current_calloc
#endif
#ifndef ts_realloc
#define ts_realloc ts_current_realloc
#endif
#ifndef ts_free
#define ts_free ts_current_free
#endif
#else
#ifndef ts_malloc
#define ts_malloc malloc
#endif
#ifndef ts_calloc
#define ts_calloc calloc
#endif
#ifndef ts_realloc
#define ts_realloc realloc
#endif
#ifndef ts_free
#define ts_free free
#endif
#endif
#ifdef __cplusplus
}
#endif
#endif // TREE_SITTER_ALLOC_H_

View file

@ -0,0 +1,290 @@
#ifndef TREE_SITTER_ARRAY_H_
#define TREE_SITTER_ARRAY_H_
#ifdef __cplusplus
extern "C" {
#endif
#include "./alloc.h"
#include <assert.h>
#include <stdbool.h>
#include <stdint.h>
#include <stdlib.h>
#include <string.h>
#ifdef _MSC_VER
#pragma warning(disable : 4101)
#elif defined(__GNUC__) || defined(__clang__)
#pragma GCC diagnostic push
#pragma GCC diagnostic ignored "-Wunused-variable"
#endif
#define Array(T) \
struct { \
T *contents; \
uint32_t size; \
uint32_t capacity; \
}
/// Initialize an array.
#define array_init(self) \
((self)->size = 0, (self)->capacity = 0, (self)->contents = NULL)
/// Create an empty array.
#define array_new() \
{ NULL, 0, 0 }
/// Get a pointer to the element at a given `index` in the array.
#define array_get(self, _index) \
(assert((uint32_t)(_index) < (self)->size), &(self)->contents[_index])
/// Get a pointer to the first element in the array.
#define array_front(self) array_get(self, 0)
/// Get a pointer to the last element in the array.
#define array_back(self) array_get(self, (self)->size - 1)
/// Clear the array, setting its size to zero. Note that this does not free any
/// memory allocated for the array's contents.
#define array_clear(self) ((self)->size = 0)
/// Reserve `new_capacity` elements of space in the array. If `new_capacity` is
/// less than the array's current capacity, this function has no effect.
#define array_reserve(self, new_capacity) \
_array__reserve((Array *)(self), array_elem_size(self), new_capacity)
/// Free any memory allocated for this array. Note that this does not free any
/// memory allocated for the array's contents.
#define array_delete(self) _array__delete((Array *)(self))
/// Push a new `element` onto the end of the array.
#define array_push(self, element) \
(_array__grow((Array *)(self), 1, array_elem_size(self)), \
(self)->contents[(self)->size++] = (element))
/// Increase the array's size by `count` elements.
/// New elements are zero-initialized.
#define array_grow_by(self, count) \
do { \
if ((count) == 0) break; \
_array__grow((Array *)(self), count, array_elem_size(self)); \
memset((self)->contents + (self)->size, 0, (count) * array_elem_size(self)); \
(self)->size += (count); \
} while (0)
/// Append all elements from one array to the end of another.
#define array_push_all(self, other) \
array_extend((self), (other)->size, (other)->contents)
/// Append `count` elements to the end of the array, reading their values from the
/// `contents` pointer.
#define array_extend(self, count, contents) \
_array__splice( \
(Array *)(self), array_elem_size(self), (self)->size, \
0, count, contents \
)
/// Remove `old_count` elements from the array starting at the given `index`. At
/// the same index, insert `new_count` new elements, reading their values from the
/// `new_contents` pointer.
#define array_splice(self, _index, old_count, new_count, new_contents) \
_array__splice( \
(Array *)(self), array_elem_size(self), _index, \
old_count, new_count, new_contents \
)
/// Insert one `element` into the array at the given `index`.
#define array_insert(self, _index, element) \
_array__splice((Array *)(self), array_elem_size(self), _index, 0, 1, &(element))
/// Remove one element from the array at the given `index`.
#define array_erase(self, _index) \
_array__erase((Array *)(self), array_elem_size(self), _index)
/// Pop the last element off the array, returning the element by value.
#define array_pop(self) ((self)->contents[--(self)->size])
/// Assign the contents of one array to another, reallocating if necessary.
#define array_assign(self, other) \
_array__assign((Array *)(self), (const Array *)(other), array_elem_size(self))
/// Swap one array with another
#define array_swap(self, other) \
_array__swap((Array *)(self), (Array *)(other))
/// Get the size of the array contents
#define array_elem_size(self) (sizeof *(self)->contents)
/// Search a sorted array for a given `needle` value, using the given `compare`
/// callback to determine the order.
///
/// If an existing element is found to be equal to `needle`, then the `index`
/// out-parameter is set to the existing value's index, and the `exists`
/// out-parameter is set to true. Otherwise, `index` is set to an index where
/// `needle` should be inserted in order to preserve the sorting, and `exists`
/// is set to false.
#define array_search_sorted_with(self, compare, needle, _index, _exists) \
_array__search_sorted(self, 0, compare, , needle, _index, _exists)
/// Search a sorted array for a given `needle` value, using integer comparisons
/// of a given struct field (specified with a leading dot) to determine the order.
///
/// See also `array_search_sorted_with`.
#define array_search_sorted_by(self, field, needle, _index, _exists) \
_array__search_sorted(self, 0, _compare_int, field, needle, _index, _exists)
/// Insert a given `value` into a sorted array, using the given `compare`
/// callback to determine the order.
#define array_insert_sorted_with(self, compare, value) \
do { \
unsigned _index, _exists; \
array_search_sorted_with(self, compare, &(value), &_index, &_exists); \
if (!_exists) array_insert(self, _index, value); \
} while (0)
/// Insert a given `value` into a sorted array, using integer comparisons of
/// a given struct field (specified with a leading dot) to determine the order.
///
/// See also `array_search_sorted_by`.
#define array_insert_sorted_by(self, field, value) \
do { \
unsigned _index, _exists; \
array_search_sorted_by(self, field, (value) field, &_index, &_exists); \
if (!_exists) array_insert(self, _index, value); \
} while (0)
// Private
typedef Array(void) Array;
/// This is not what you're looking for, see `array_delete`.
static inline void _array__delete(Array *self) {
if (self->contents) {
ts_free(self->contents);
self->contents = NULL;
self->size = 0;
self->capacity = 0;
}
}
/// This is not what you're looking for, see `array_erase`.
static inline void _array__erase(Array *self, size_t element_size,
uint32_t index) {
assert(index < self->size);
char *contents = (char *)self->contents;
memmove(contents + index * element_size, contents + (index + 1) * element_size,
(self->size - index - 1) * element_size);
self->size--;
}
/// This is not what you're looking for, see `array_reserve`.
static inline void _array__reserve(Array *self, size_t element_size, uint32_t new_capacity) {
if (new_capacity > self->capacity) {
if (self->contents) {
self->contents = ts_realloc(self->contents, new_capacity * element_size);
} else {
self->contents = ts_malloc(new_capacity * element_size);
}
self->capacity = new_capacity;
}
}
/// This is not what you're looking for, see `array_assign`.
static inline void _array__assign(Array *self, const Array *other, size_t element_size) {
_array__reserve(self, element_size, other->size);
self->size = other->size;
memcpy(self->contents, other->contents, self->size * element_size);
}
/// This is not what you're looking for, see `array_swap`.
static inline void _array__swap(Array *self, Array *other) {
Array swap = *other;
*other = *self;
*self = swap;
}
/// This is not what you're looking for, see `array_push` or `array_grow_by`.
static inline void _array__grow(Array *self, uint32_t count, size_t element_size) {
uint32_t new_size = self->size + count;
if (new_size > self->capacity) {
uint32_t new_capacity = self->capacity * 2;
if (new_capacity < 8) new_capacity = 8;
if (new_capacity < new_size) new_capacity = new_size;
_array__reserve(self, element_size, new_capacity);
}
}
/// This is not what you're looking for, see `array_splice`.
static inline void _array__splice(Array *self, size_t element_size,
uint32_t index, uint32_t old_count,
uint32_t new_count, const void *elements) {
uint32_t new_size = self->size + new_count - old_count;
uint32_t old_end = index + old_count;
uint32_t new_end = index + new_count;
assert(old_end <= self->size);
_array__reserve(self, element_size, new_size);
char *contents = (char *)self->contents;
if (self->size > old_end) {
memmove(
contents + new_end * element_size,
contents + old_end * element_size,
(self->size - old_end) * element_size
);
}
if (new_count > 0) {
if (elements) {
memcpy(
(contents + index * element_size),
elements,
new_count * element_size
);
} else {
memset(
(contents + index * element_size),
0,
new_count * element_size
);
}
}
self->size += new_count - old_count;
}
/// A binary search routine, based on Rust's `std::slice::binary_search_by`.
/// This is not what you're looking for, see `array_search_sorted_with` or `array_search_sorted_by`.
#define _array__search_sorted(self, start, compare, suffix, needle, _index, _exists) \
do { \
*(_index) = start; \
*(_exists) = false; \
uint32_t size = (self)->size - *(_index); \
if (size == 0) break; \
int comparison; \
while (size > 1) { \
uint32_t half_size = size / 2; \
uint32_t mid_index = *(_index) + half_size; \
comparison = compare(&((self)->contents[mid_index] suffix), (needle)); \
if (comparison <= 0) *(_index) = mid_index; \
size -= half_size; \
} \
comparison = compare(&((self)->contents[*(_index)] suffix), (needle)); \
if (comparison == 0) *(_exists) = true; \
else if (comparison < 0) *(_index) += 1; \
} while (0)
/// Helper macro for the `_sorted_by` routines below. This takes the left (existing)
/// parameter by reference in order to work with the generic sorting function above.
#define _compare_int(a, b) ((int)*(a) - (int)(b))
#ifdef _MSC_VER
#pragma warning(default : 4101)
#elif defined(__GNUC__) || defined(__clang__)
#pragma GCC diagnostic pop
#endif
#ifdef __cplusplus
}
#endif
#endif // TREE_SITTER_ARRAY_H_

View file

@ -0,0 +1,265 @@
#ifndef TREE_SITTER_PARSER_H_
#define TREE_SITTER_PARSER_H_
#ifdef __cplusplus
extern "C" {
#endif
#include <stdbool.h>
#include <stdint.h>
#include <stdlib.h>
#define ts_builtin_sym_error ((TSSymbol)-1)
#define ts_builtin_sym_end 0
#define TREE_SITTER_SERIALIZATION_BUFFER_SIZE 1024
#ifndef TREE_SITTER_API_H_
typedef uint16_t TSStateId;
typedef uint16_t TSSymbol;
typedef uint16_t TSFieldId;
typedef struct TSLanguage TSLanguage;
#endif
typedef struct {
TSFieldId field_id;
uint8_t child_index;
bool inherited;
} TSFieldMapEntry;
typedef struct {
uint16_t index;
uint16_t length;
} TSFieldMapSlice;
typedef struct {
bool visible;
bool named;
bool supertype;
} TSSymbolMetadata;
typedef struct TSLexer TSLexer;
struct TSLexer {
int32_t lookahead;
TSSymbol result_symbol;
void (*advance)(TSLexer *, bool);
void (*mark_end)(TSLexer *);
uint32_t (*get_column)(TSLexer *);
bool (*is_at_included_range_start)(const TSLexer *);
bool (*eof)(const TSLexer *);
};
typedef enum {
TSParseActionTypeShift,
TSParseActionTypeReduce,
TSParseActionTypeAccept,
TSParseActionTypeRecover,
} TSParseActionType;
typedef union {
struct {
uint8_t type;
TSStateId state;
bool extra;
bool repetition;
} shift;
struct {
uint8_t type;
uint8_t child_count;
TSSymbol symbol;
int16_t dynamic_precedence;
uint16_t production_id;
} reduce;
uint8_t type;
} TSParseAction;
typedef struct {
uint16_t lex_state;
uint16_t external_lex_state;
} TSLexMode;
typedef union {
TSParseAction action;
struct {
uint8_t count;
bool reusable;
} entry;
} TSParseActionEntry;
typedef struct {
int32_t start;
int32_t end;
} TSCharacterRange;
struct TSLanguage {
uint32_t version;
uint32_t symbol_count;
uint32_t alias_count;
uint32_t token_count;
uint32_t external_token_count;
uint32_t state_count;
uint32_t large_state_count;
uint32_t production_id_count;
uint32_t field_count;
uint16_t max_alias_sequence_length;
const uint16_t *parse_table;
const uint16_t *small_parse_table;
const uint32_t *small_parse_table_map;
const TSParseActionEntry *parse_actions;
const char * const *symbol_names;
const char * const *field_names;
const TSFieldMapSlice *field_map_slices;
const TSFieldMapEntry *field_map_entries;
const TSSymbolMetadata *symbol_metadata;
const TSSymbol *public_symbol_map;
const uint16_t *alias_map;
const TSSymbol *alias_sequences;
const TSLexMode *lex_modes;
bool (*lex_fn)(TSLexer *, TSStateId);
bool (*keyword_lex_fn)(TSLexer *, TSStateId);
TSSymbol keyword_capture_token;
struct {
const bool *states;
const TSSymbol *symbol_map;
void *(*create)(void);
void (*destroy)(void *);
bool (*scan)(void *, TSLexer *, const bool *symbol_whitelist);
unsigned (*serialize)(void *, char *);
void (*deserialize)(void *, const char *, unsigned);
} external_scanner;
const TSStateId *primary_state_ids;
};
static inline bool set_contains(TSCharacterRange *ranges, uint32_t len, int32_t lookahead) {
uint32_t index = 0;
uint32_t size = len - index;
while (size > 1) {
uint32_t half_size = size / 2;
uint32_t mid_index = index + half_size;
TSCharacterRange *range = &ranges[mid_index];
if (lookahead >= range->start && lookahead <= range->end) {
return true;
} else if (lookahead > range->end) {
index = mid_index;
}
size -= half_size;
}
TSCharacterRange *range = &ranges[index];
return (lookahead >= range->start && lookahead <= range->end);
}
/*
* Lexer Macros
*/
#ifdef _MSC_VER
#define UNUSED __pragma(warning(suppress : 4101))
#else
#define UNUSED __attribute__((unused))
#endif
#define START_LEXER() \
bool result = false; \
bool skip = false; \
UNUSED \
bool eof = false; \
int32_t lookahead; \
goto start; \
next_state: \
lexer->advance(lexer, skip); \
start: \
skip = false; \
lookahead = lexer->lookahead;
#define ADVANCE(state_value) \
{ \
state = state_value; \
goto next_state; \
}
#define ADVANCE_MAP(...) \
{ \
static const uint16_t map[] = { __VA_ARGS__ }; \
for (uint32_t i = 0; i < sizeof(map) / sizeof(map[0]); i += 2) { \
if (map[i] == lookahead) { \
state = map[i + 1]; \
goto next_state; \
} \
} \
}
#define SKIP(state_value) \
{ \
skip = true; \
state = state_value; \
goto next_state; \
}
#define ACCEPT_TOKEN(symbol_value) \
result = true; \
lexer->result_symbol = symbol_value; \
lexer->mark_end(lexer);
#define END_STATE() return result;
/*
* Parse Table Macros
*/
#define SMALL_STATE(id) ((id) - LARGE_STATE_COUNT)
#define STATE(id) id
#define ACTIONS(id) id
#define SHIFT(state_value) \
{{ \
.shift = { \
.type = TSParseActionTypeShift, \
.state = (state_value) \
} \
}}
#define SHIFT_REPEAT(state_value) \
{{ \
.shift = { \
.type = TSParseActionTypeShift, \
.state = (state_value), \
.repetition = true \
} \
}}
#define SHIFT_EXTRA() \
{{ \
.shift = { \
.type = TSParseActionTypeShift, \
.extra = true \
} \
}}
#define REDUCE(symbol_name, children, precedence, prod_id) \
{{ \
.reduce = { \
.type = TSParseActionTypeReduce, \
.symbol = symbol_name, \
.child_count = children, \
.dynamic_precedence = precedence, \
.production_id = prod_id \
}, \
}}
#define RECOVER() \
{{ \
.type = TSParseActionTypeRecover \
}}
#define ACCEPT_INPUT() \
{{ \
.type = TSParseActionTypeAccept \
}}
#ifdef __cplusplus
}
#endif
#endif // TREE_SITTER_PARSER_H_

View file

@ -0,0 +1,30 @@
{
"targets": [
{
"target_name": "tree_sitter_kotlin_binding",
"dependencies": [
"<!(node -p \"require('node-addon-api').targets\"):node_addon_api_except",
],
"include_dirs": [
"src",
],
"sources": [
"bindings/node/binding.cc",
"src/parser.c",
"src/scanner.c"
],
"conditions": [
["OS!='win'", {
"cflags_c": [
"-std=c11",
],
}, { # OS == "win"
"cflags_c": [
"/std:c11",
"/utf-8",
],
}],
],
}
]
}

View file

@ -0,0 +1,20 @@
#include <napi.h>
typedef struct TSLanguage TSLanguage;
extern "C" TSLanguage *tree_sitter_kotlin();
// "tree-sitter", "language" hashed with BLAKE2
const napi_type_tag LANGUAGE_TYPE_TAG = {
0x8AF2E5212AD58ABF, 0xD5006CAD83ABBA16
};
Napi::Object Init(Napi::Env env, Napi::Object exports) {
exports["name"] = Napi::String::New(env, "kotlin");
auto language = Napi::External<TSLanguage>::New(env, tree_sitter_kotlin());
language.TypeTag(&LANGUAGE_TYPE_TAG);
exports["language"] = language;
return exports;
}
NODE_API_MODULE(tree_sitter_kotlin_binding, Init)

View file

@ -6,7 +6,7 @@
"license": "MIT",
"main": "bindings/node/index.js",
"types": "bindings/node/index.d.ts",
"_vendoredBy": "gitnexus - minimal runtime package derived from tree-sitter-kotlin@0.3.8 (fwcd). Unlike Swift's upstream-shipped prebuilds, upstream tree-sitter-kotlin ships SOURCE ONLY (no prebuilds/); the native prebuilds/ here are GitNexus-cross-built by .github/workflows/build-tree-sitter-prebuilds.yml. Copied to node_modules/ by materialize-vendor-grammars.cjs; prebuild activation via build-tree-sitter-kotlin.cjs (no scripts.install here — #836/#1728). The C source (parser.c/scanner.c/binding.gyp) is deliberately NOT vendored: the parser.c is ~23 MB and the workflow builds from the published npm package, so committing it would bloat git history with no runtime benefit.",
"_vendoredBy": "gitnexus - runtime package derived from tree-sitter-kotlin@0.3.8 (fwcd). Unlike Swift's upstream-shipped prebuilds, upstream tree-sitter-kotlin ships SOURCE ONLY (no prebuilds/); the native prebuilds/ here are GitNexus-cross-built by .github/workflows/build-tree-sitter-prebuilds.yml. The grammar source (parser.c/scanner.c/binding.gyp + src/) is ALSO vendored so build-tree-sitter-kotlin.cjs can source-build the binding on a toolchain host when no prebuild matches (e.g. CI before prebuilds land). The generated parser.c is large (~23 MB on disk; it compresses heavily in git); once the prebuilds cover every platform-arch the source serves only as the fallback. Copied to node_modules/ by materialize-vendor-grammars.cjs (no scripts.install here — #836/#1728).",
"peerDependencies": {
"tree-sitter": "^0.21.0"
},

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,530 @@
#include "tree_sitter/array.h"
#include "tree_sitter/parser.h"
#include <string.h>
#include <wctype.h>
// Mostly a copy paste of tree-sitter-javascript/src/scanner.c
enum TokenType {
AUTOMATIC_SEMICOLON,
IMPORT_LIST_DELIMITER,
SAFE_NAV,
MULTILINE_COMMENT,
STRING_START,
STRING_END,
STRING_CONTENT,
};
/* Pretty much all of this code is taken from the Julia tree-sitter
parser.
Julia has similar problems with multiline comments that can be nested,
line comments, as well as line and multiline strings.
The most heavily edited section is `scan_string_content`,
particularly with respect to interpolation.
*/
// Block comments are easy to parse, but strings require extra-attention.
// The main problems that arise when parsing strings are:
// 1. Triple quoted strings allow single quotes inside. e.g. """ "foo" """.
// 2. Non-standard string literals don't allow interpolations or escape
// sequences, but you can always write \" and \`.
// To efficiently store a delimiter, we take advantage of the fact that:
// (int)'"' == 34 && (34 & 1) == 0
// i.e. " has an even numeric representation, so we can store a triple
// quoted delimiter as (delimiter + 1).
#define DELIMITER_LENGTH 3
typedef char Delimiter;
// We use a stack to keep track of the string delimiters.
typedef Array(Delimiter) Stack;
static inline void stack_push(Stack *stack, char chr, bool triple) {
if (stack->size >= TREE_SITTER_SERIALIZATION_BUFFER_SIZE) abort();
array_push(stack, (Delimiter)(triple ? (chr + 1) : chr));
}
static inline Delimiter stack_pop(Stack *stack) {
if (stack->size == 0) abort();
return array_pop(stack);
}
static inline void skip(TSLexer *lexer) { lexer->advance(lexer, true); }
static inline void advance(TSLexer *lexer) { lexer->advance(lexer, false); }
// Scanner functions
static bool scan_string_start(TSLexer *lexer, Stack *stack) {
if (lexer->lookahead != '"') return false;
advance(lexer);
lexer->mark_end(lexer);
for (unsigned count = 1; count < DELIMITER_LENGTH; ++count) {
if (lexer->lookahead != '"') {
// It's not a triple quoted delimiter.
stack_push(stack, '"', false);
return true;
}
advance(lexer);
}
lexer->mark_end(lexer);
stack_push(stack, '"', true);
return true;
}
static bool scan_string_content(TSLexer *lexer, Stack *stack) {
if (stack->size == 0) return false; // Stack is empty. We're not in a string.
Delimiter end_char = stack->contents[stack->size - 1]; // peek
bool is_triple = false;
bool has_content = false;
if (end_char & 1) {
is_triple = true;
end_char -= 1;
}
while (lexer->lookahead) {
if (lexer->lookahead == '$') {
// if we did not just start reading stuff, then we should stop
// lexing right here, so we can offer the opportunity to lex a
// interpolated identifier
if (has_content) {
lexer->result_symbol = STRING_CONTENT;
return has_content;
}
// otherwise, if this is the start, determine if it is an
// interpolated identifier.
// otherwise, it's just string content, so continue
advance(lexer);
if (iswalpha(lexer->lookahead) || lexer->lookahead == '{') {
// this must be a string interpolation, let's
// fail so we parse it as such
return false;
}
lexer->result_symbol = STRING_CONTENT;
lexer->mark_end(lexer);
return true;
}
if (lexer->lookahead == '\\') {
// if we see a \, then this might possibly escape a dollar sign
// in which case, we should not defer to the interpolation
advance(lexer);
// this dollar sign is escaped, so it must be content.
// we consume it here so we don't enter the dollar sign case above,
// which leaves the possibility that it is an interpolation
if (lexer->lookahead == '$') {
advance(lexer);
// however this leaves an edgecase where an escaped dollar sign could
// appear at the end of a string (e.g "aa\$") which isn't handled
// correctly; if we were at the end of the string, terminate properly
if (lexer->lookahead == end_char) {
stack_pop(stack);
advance(lexer);
lexer->mark_end(lexer);
lexer->result_symbol = STRING_END;
return true;
}
}
} else if (lexer->lookahead == end_char) {
if (is_triple) {
lexer->mark_end(lexer);
for (unsigned count = 1; count < DELIMITER_LENGTH; ++count) {
advance(lexer);
if (lexer->lookahead != end_char) {
lexer->mark_end(lexer);
lexer->result_symbol = STRING_CONTENT;
return true;
}
}
/* This is so if we lex something like
"""foo"""
^
where we are at the `f`, we should quit after
reading `foo`, and ascribe it to STRING_CONTENT.
Then, we restart and try to read the end.
This is to prevent `foo` from being absorbed into
the STRING_END token.
*/
if (has_content && lexer->lookahead == end_char) {
lexer->result_symbol = STRING_CONTENT;
return true;
}
/* Since the string internals are all hidden in the syntax
tree anyways, there's no point in going to the effort of
specifically separating the string end from string contents.
If we see a bunch of quotes in a row, then we just go until
they stop appearing, then stop lexing and call it the
string's end.
*/
lexer->result_symbol = STRING_END;
lexer->mark_end(lexer);
while (lexer->lookahead == end_char) {
advance(lexer);
lexer->mark_end(lexer);
}
stack_pop(stack);
return true;
}
if (has_content) {
lexer->mark_end(lexer);
lexer->result_symbol = STRING_CONTENT;
return true;
}
stack_pop(stack);
advance(lexer);
lexer->mark_end(lexer);
lexer->result_symbol = STRING_END;
return true;
}
advance(lexer);
has_content = true;
}
return false;
}
static bool scan_multiline_comment(TSLexer *lexer) {
if (lexer->lookahead != '/') return false;
advance(lexer);
if (lexer->lookahead != '*') return false;
advance(lexer);
bool after_star = false;
unsigned nesting_depth = 1;
for (;;) {
switch (lexer->lookahead) {
case '*':
advance(lexer);
after_star = true;
break;
case '/':
advance(lexer);
if (after_star) {
after_star = false;
nesting_depth -= 1;
if (nesting_depth == 0) {
lexer->result_symbol = MULTILINE_COMMENT;
lexer->mark_end(lexer);
return true;
}
} else {
after_star = false;
if (lexer->lookahead == '*') {
nesting_depth += 1;
advance(lexer);
}
}
break;
case '\0':
return false;
default:
advance(lexer);
after_star = false;
break;
}
}
}
static bool scan_whitespace_and_comments(TSLexer *lexer) {
while (iswspace(lexer->lookahead)) skip(lexer);
return lexer->lookahead != '/';
}
static bool scan_for_word(TSLexer *lexer, const char* word, unsigned len) {
skip(lexer);
for (unsigned i = 0; i < len; ++i) {
if (lexer->lookahead != word[i]) return false;
skip(lexer);
}
return true;
}
static bool scan_automatic_semicolon(TSLexer *lexer) {
lexer->result_symbol = AUTOMATIC_SEMICOLON;
lexer->mark_end(lexer);
bool sameline = true;
for (;;) {
if (lexer->eof(lexer)) return true;
if (lexer->lookahead == ';') {
advance(lexer);
lexer->mark_end(lexer);
return true;
}
if (!iswspace(lexer->lookahead)) break;
if (lexer->lookahead == '\n') {
skip(lexer);
sameline = false;
break;
}
if (lexer->lookahead == '\r') {
skip(lexer);
if (lexer->lookahead == '\n') skip(lexer);
sameline = false;
break;
}
skip(lexer);
}
// Skip whitespace and comments
if (!scan_whitespace_and_comments(lexer))
return false;
if (sameline) {
switch (lexer->lookahead) {
// Don't insert a semicolon before an else
case 'e':
return !scan_for_word(lexer, "lse", 3);
case 'i':
return scan_for_word(lexer, "mport", 5);
case ';':
advance(lexer);
lexer->mark_end(lexer);
return true;
default:
return false;
}
}
switch (lexer->lookahead) {
case ',':
case '.':
case ':':
case '*':
case '%':
case '>':
case '<':
case '=':
case '{':
case '[':
case '(':
case '?':
case '|':
case '&':
case '/':
return false;
// Insert a semicolon before `--` and `++`, but not before binary `+` or `-`.
// Insert before +/-Float
case '+':
skip(lexer);
if (lexer->lookahead == '+') return true;
return iswdigit(lexer->lookahead);
case '-':
skip(lexer);
if (lexer->lookahead == '-') return true;
return iswdigit(lexer->lookahead);
// Don't insert a semicolon before `!=`, but do insert one before a unary `!`.
case '!':
skip(lexer);
return lexer->lookahead != '=';
// Don't insert a semicolon before an else
case 'e':
return !scan_for_word(lexer, "lse", 3);
// Don't insert a semicolon before `in` or `instanceof`, but do insert one
// before an identifier or an import.
case 'i':
skip(lexer);
if (lexer->lookahead != 'n') return true;
skip(lexer);
if (!iswalpha(lexer->lookahead)) return false;
return !scan_for_word(lexer, "stanceof", 8);
case ';':
advance(lexer);
lexer->mark_end(lexer);
return true;
default:
return true;
}
}
static bool scan_safe_nav(TSLexer *lexer) {
lexer->result_symbol = SAFE_NAV;
lexer->mark_end(lexer);
// skip white space
if (!scan_whitespace_and_comments(lexer))
return false;
if (lexer->lookahead != '?')
return false;
advance(lexer);
if (!scan_whitespace_and_comments(lexer))
return false;
if (lexer->lookahead != '.')
return false;
advance(lexer);
lexer->mark_end(lexer);
return true;
}
static bool scan_line_sep(TSLexer *lexer) {
// Line Seps: [ CR, LF, CRLF ]
int state = 0;
while (true) {
switch(lexer->lookahead) {
case ' ':
case '\t':
case '\v':
// Skip whitespace
advance(lexer);
break;
case '\n':
advance(lexer);
return true;
case '\r':
if (state == 1)
return true;
state = 1;
advance(lexer);
break;
default:
// We read a CR
if (state == 1)
return true;
return false;
}
}
}
static bool scan_import_list_delimiter(TSLexer *lexer) {
// Import lists are terminated either by an empty line or a non import statement
lexer->result_symbol = IMPORT_LIST_DELIMITER;
lexer->mark_end(lexer);
// if eof; return true
if (lexer->eof(lexer))
return true;
// Scan for the first line seperator
if (!scan_line_sep(lexer))
return false;
// if line.sep line.sep; return true
if (scan_line_sep(lexer)) {
lexer->mark_end(lexer);
return true;
}
// if line.sep [^import]; return true
while (true) {
switch (lexer->lookahead) {
case ' ':
case '\t':
case '\v':
// Skip whitespace
advance(lexer);
break;
case 'i':
return !scan_for_word(lexer, "mport", 5);
default:
return true;
}
return false;
}
}
bool tree_sitter_kotlin_external_scanner_scan(void *payload, TSLexer *lexer, const bool *valid_symbols) {
if (valid_symbols[AUTOMATIC_SEMICOLON]) {
bool ret = scan_automatic_semicolon(lexer);
if (!ret && valid_symbols[SAFE_NAV] && lexer->lookahead == '?') {
return scan_safe_nav(lexer);
}
// if we fail to find an automatic semicolon, it's still possible that we may
// want to lex a string or comment later
if (ret) return ret;
}
if (valid_symbols[IMPORT_LIST_DELIMITER]) {
return scan_import_list_delimiter(lexer);
}
// content or end
if (valid_symbols[STRING_CONTENT] && scan_string_content(lexer, payload)) {
return true;
}
// a string might follow after some whitespace, so we can't lookahead
// until we get rid of it
while (iswspace(lexer->lookahead)) skip(lexer);
if (valid_symbols[STRING_START] && scan_string_start(lexer, payload)) {
lexer->result_symbol = STRING_START;
return true;
}
if (valid_symbols[MULTILINE_COMMENT] && scan_multiline_comment(lexer)) {
return true;
}
if (valid_symbols[SAFE_NAV]) {
return scan_safe_nav(lexer);
}
return false;
}
void *tree_sitter_kotlin_external_scanner_create() {
Stack *stack = ts_calloc(1, sizeof(Stack));
if (stack == NULL) abort();
array_init(stack);
return stack;
}
void tree_sitter_kotlin_external_scanner_destroy(void *payload) {
Stack *stack = (Stack *)payload;
array_delete(stack);
ts_free(stack);
}
unsigned tree_sitter_kotlin_external_scanner_serialize(void *payload, char *buffer) {
Stack *stack = (Stack *)payload;
memcpy(buffer, stack->contents, stack->size);
return stack->size;
}
void tree_sitter_kotlin_external_scanner_deserialize(void *payload, const char *buffer, unsigned length) {
Stack *stack = (Stack *)payload;
if (length > 0) {
array_reserve(stack, length);
memcpy(stack->contents, buffer, length);
stack->size = length;
} else {
array_clear(stack);
}
}

View file

@ -0,0 +1,54 @@
#ifndef TREE_SITTER_ALLOC_H_
#define TREE_SITTER_ALLOC_H_
#ifdef __cplusplus
extern "C" {
#endif
#include <stdbool.h>
#include <stdio.h>
#include <stdlib.h>
// Allow clients to override allocation functions
#ifdef TREE_SITTER_REUSE_ALLOCATOR
extern void *(*ts_current_malloc)(size_t);
extern void *(*ts_current_calloc)(size_t, size_t);
extern void *(*ts_current_realloc)(void *, size_t);
extern void (*ts_current_free)(void *);
#ifndef ts_malloc
#define ts_malloc ts_current_malloc
#endif
#ifndef ts_calloc
#define ts_calloc ts_current_calloc
#endif
#ifndef ts_realloc
#define ts_realloc ts_current_realloc
#endif
#ifndef ts_free
#define ts_free ts_current_free
#endif
#else
#ifndef ts_malloc
#define ts_malloc malloc
#endif
#ifndef ts_calloc
#define ts_calloc calloc
#endif
#ifndef ts_realloc
#define ts_realloc realloc
#endif
#ifndef ts_free
#define ts_free free
#endif
#endif
#ifdef __cplusplus
}
#endif
#endif // TREE_SITTER_ALLOC_H_

View file

@ -0,0 +1,290 @@
#ifndef TREE_SITTER_ARRAY_H_
#define TREE_SITTER_ARRAY_H_
#ifdef __cplusplus
extern "C" {
#endif
#include "./alloc.h"
#include <assert.h>
#include <stdbool.h>
#include <stdint.h>
#include <stdlib.h>
#include <string.h>
#ifdef _MSC_VER
#pragma warning(disable : 4101)
#elif defined(__GNUC__) || defined(__clang__)
#pragma GCC diagnostic push
#pragma GCC diagnostic ignored "-Wunused-variable"
#endif
#define Array(T) \
struct { \
T *contents; \
uint32_t size; \
uint32_t capacity; \
}
/// Initialize an array.
#define array_init(self) \
((self)->size = 0, (self)->capacity = 0, (self)->contents = NULL)
/// Create an empty array.
#define array_new() \
{ NULL, 0, 0 }
/// Get a pointer to the element at a given `index` in the array.
#define array_get(self, _index) \
(assert((uint32_t)(_index) < (self)->size), &(self)->contents[_index])
/// Get a pointer to the first element in the array.
#define array_front(self) array_get(self, 0)
/// Get a pointer to the last element in the array.
#define array_back(self) array_get(self, (self)->size - 1)
/// Clear the array, setting its size to zero. Note that this does not free any
/// memory allocated for the array's contents.
#define array_clear(self) ((self)->size = 0)
/// Reserve `new_capacity` elements of space in the array. If `new_capacity` is
/// less than the array's current capacity, this function has no effect.
#define array_reserve(self, new_capacity) \
_array__reserve((Array *)(self), array_elem_size(self), new_capacity)
/// Free any memory allocated for this array. Note that this does not free any
/// memory allocated for the array's contents.
#define array_delete(self) _array__delete((Array *)(self))
/// Push a new `element` onto the end of the array.
#define array_push(self, element) \
(_array__grow((Array *)(self), 1, array_elem_size(self)), \
(self)->contents[(self)->size++] = (element))
/// Increase the array's size by `count` elements.
/// New elements are zero-initialized.
#define array_grow_by(self, count) \
do { \
if ((count) == 0) break; \
_array__grow((Array *)(self), count, array_elem_size(self)); \
memset((self)->contents + (self)->size, 0, (count) * array_elem_size(self)); \
(self)->size += (count); \
} while (0)
/// Append all elements from one array to the end of another.
#define array_push_all(self, other) \
array_extend((self), (other)->size, (other)->contents)
/// Append `count` elements to the end of the array, reading their values from the
/// `contents` pointer.
#define array_extend(self, count, contents) \
_array__splice( \
(Array *)(self), array_elem_size(self), (self)->size, \
0, count, contents \
)
/// Remove `old_count` elements from the array starting at the given `index`. At
/// the same index, insert `new_count` new elements, reading their values from the
/// `new_contents` pointer.
#define array_splice(self, _index, old_count, new_count, new_contents) \
_array__splice( \
(Array *)(self), array_elem_size(self), _index, \
old_count, new_count, new_contents \
)
/// Insert one `element` into the array at the given `index`.
#define array_insert(self, _index, element) \
_array__splice((Array *)(self), array_elem_size(self), _index, 0, 1, &(element))
/// Remove one element from the array at the given `index`.
#define array_erase(self, _index) \
_array__erase((Array *)(self), array_elem_size(self), _index)
/// Pop the last element off the array, returning the element by value.
#define array_pop(self) ((self)->contents[--(self)->size])
/// Assign the contents of one array to another, reallocating if necessary.
#define array_assign(self, other) \
_array__assign((Array *)(self), (const Array *)(other), array_elem_size(self))
/// Swap one array with another
#define array_swap(self, other) \
_array__swap((Array *)(self), (Array *)(other))
/// Get the size of the array contents
#define array_elem_size(self) (sizeof *(self)->contents)
/// Search a sorted array for a given `needle` value, using the given `compare`
/// callback to determine the order.
///
/// If an existing element is found to be equal to `needle`, then the `index`
/// out-parameter is set to the existing value's index, and the `exists`
/// out-parameter is set to true. Otherwise, `index` is set to an index where
/// `needle` should be inserted in order to preserve the sorting, and `exists`
/// is set to false.
#define array_search_sorted_with(self, compare, needle, _index, _exists) \
_array__search_sorted(self, 0, compare, , needle, _index, _exists)
/// Search a sorted array for a given `needle` value, using integer comparisons
/// of a given struct field (specified with a leading dot) to determine the order.
///
/// See also `array_search_sorted_with`.
#define array_search_sorted_by(self, field, needle, _index, _exists) \
_array__search_sorted(self, 0, _compare_int, field, needle, _index, _exists)
/// Insert a given `value` into a sorted array, using the given `compare`
/// callback to determine the order.
#define array_insert_sorted_with(self, compare, value) \
do { \
unsigned _index, _exists; \
array_search_sorted_with(self, compare, &(value), &_index, &_exists); \
if (!_exists) array_insert(self, _index, value); \
} while (0)
/// Insert a given `value` into a sorted array, using integer comparisons of
/// a given struct field (specified with a leading dot) to determine the order.
///
/// See also `array_search_sorted_by`.
#define array_insert_sorted_by(self, field, value) \
do { \
unsigned _index, _exists; \
array_search_sorted_by(self, field, (value) field, &_index, &_exists); \
if (!_exists) array_insert(self, _index, value); \
} while (0)
// Private
typedef Array(void) Array;
/// This is not what you're looking for, see `array_delete`.
static inline void _array__delete(Array *self) {
if (self->contents) {
ts_free(self->contents);
self->contents = NULL;
self->size = 0;
self->capacity = 0;
}
}
/// This is not what you're looking for, see `array_erase`.
static inline void _array__erase(Array *self, size_t element_size,
uint32_t index) {
assert(index < self->size);
char *contents = (char *)self->contents;
memmove(contents + index * element_size, contents + (index + 1) * element_size,
(self->size - index - 1) * element_size);
self->size--;
}
/// This is not what you're looking for, see `array_reserve`.
static inline void _array__reserve(Array *self, size_t element_size, uint32_t new_capacity) {
if (new_capacity > self->capacity) {
if (self->contents) {
self->contents = ts_realloc(self->contents, new_capacity * element_size);
} else {
self->contents = ts_malloc(new_capacity * element_size);
}
self->capacity = new_capacity;
}
}
/// This is not what you're looking for, see `array_assign`.
static inline void _array__assign(Array *self, const Array *other, size_t element_size) {
_array__reserve(self, element_size, other->size);
self->size = other->size;
memcpy(self->contents, other->contents, self->size * element_size);
}
/// This is not what you're looking for, see `array_swap`.
static inline void _array__swap(Array *self, Array *other) {
Array swap = *other;
*other = *self;
*self = swap;
}
/// This is not what you're looking for, see `array_push` or `array_grow_by`.
static inline void _array__grow(Array *self, uint32_t count, size_t element_size) {
uint32_t new_size = self->size + count;
if (new_size > self->capacity) {
uint32_t new_capacity = self->capacity * 2;
if (new_capacity < 8) new_capacity = 8;
if (new_capacity < new_size) new_capacity = new_size;
_array__reserve(self, element_size, new_capacity);
}
}
/// This is not what you're looking for, see `array_splice`.
static inline void _array__splice(Array *self, size_t element_size,
uint32_t index, uint32_t old_count,
uint32_t new_count, const void *elements) {
uint32_t new_size = self->size + new_count - old_count;
uint32_t old_end = index + old_count;
uint32_t new_end = index + new_count;
assert(old_end <= self->size);
_array__reserve(self, element_size, new_size);
char *contents = (char *)self->contents;
if (self->size > old_end) {
memmove(
contents + new_end * element_size,
contents + old_end * element_size,
(self->size - old_end) * element_size
);
}
if (new_count > 0) {
if (elements) {
memcpy(
(contents + index * element_size),
elements,
new_count * element_size
);
} else {
memset(
(contents + index * element_size),
0,
new_count * element_size
);
}
}
self->size += new_count - old_count;
}
/// A binary search routine, based on Rust's `std::slice::binary_search_by`.
/// This is not what you're looking for, see `array_search_sorted_with` or `array_search_sorted_by`.
#define _array__search_sorted(self, start, compare, suffix, needle, _index, _exists) \
do { \
*(_index) = start; \
*(_exists) = false; \
uint32_t size = (self)->size - *(_index); \
if (size == 0) break; \
int comparison; \
while (size > 1) { \
uint32_t half_size = size / 2; \
uint32_t mid_index = *(_index) + half_size; \
comparison = compare(&((self)->contents[mid_index] suffix), (needle)); \
if (comparison <= 0) *(_index) = mid_index; \
size -= half_size; \
} \
comparison = compare(&((self)->contents[*(_index)] suffix), (needle)); \
if (comparison == 0) *(_exists) = true; \
else if (comparison < 0) *(_index) += 1; \
} while (0)
/// Helper macro for the `_sorted_by` routines below. This takes the left (existing)
/// parameter by reference in order to work with the generic sorting function above.
#define _compare_int(a, b) ((int)*(a) - (int)(b))
#ifdef _MSC_VER
#pragma warning(default : 4101)
#elif defined(__GNUC__) || defined(__clang__)
#pragma GCC diagnostic pop
#endif
#ifdef __cplusplus
}
#endif
#endif // TREE_SITTER_ARRAY_H_

View file

@ -0,0 +1,265 @@
#ifndef TREE_SITTER_PARSER_H_
#define TREE_SITTER_PARSER_H_
#ifdef __cplusplus
extern "C" {
#endif
#include <stdbool.h>
#include <stdint.h>
#include <stdlib.h>
#define ts_builtin_sym_error ((TSSymbol)-1)
#define ts_builtin_sym_end 0
#define TREE_SITTER_SERIALIZATION_BUFFER_SIZE 1024
#ifndef TREE_SITTER_API_H_
typedef uint16_t TSStateId;
typedef uint16_t TSSymbol;
typedef uint16_t TSFieldId;
typedef struct TSLanguage TSLanguage;
#endif
typedef struct {
TSFieldId field_id;
uint8_t child_index;
bool inherited;
} TSFieldMapEntry;
typedef struct {
uint16_t index;
uint16_t length;
} TSFieldMapSlice;
typedef struct {
bool visible;
bool named;
bool supertype;
} TSSymbolMetadata;
typedef struct TSLexer TSLexer;
struct TSLexer {
int32_t lookahead;
TSSymbol result_symbol;
void (*advance)(TSLexer *, bool);
void (*mark_end)(TSLexer *);
uint32_t (*get_column)(TSLexer *);
bool (*is_at_included_range_start)(const TSLexer *);
bool (*eof)(const TSLexer *);
};
typedef enum {
TSParseActionTypeShift,
TSParseActionTypeReduce,
TSParseActionTypeAccept,
TSParseActionTypeRecover,
} TSParseActionType;
typedef union {
struct {
uint8_t type;
TSStateId state;
bool extra;
bool repetition;
} shift;
struct {
uint8_t type;
uint8_t child_count;
TSSymbol symbol;
int16_t dynamic_precedence;
uint16_t production_id;
} reduce;
uint8_t type;
} TSParseAction;
typedef struct {
uint16_t lex_state;
uint16_t external_lex_state;
} TSLexMode;
typedef union {
TSParseAction action;
struct {
uint8_t count;
bool reusable;
} entry;
} TSParseActionEntry;
typedef struct {
int32_t start;
int32_t end;
} TSCharacterRange;
struct TSLanguage {
uint32_t version;
uint32_t symbol_count;
uint32_t alias_count;
uint32_t token_count;
uint32_t external_token_count;
uint32_t state_count;
uint32_t large_state_count;
uint32_t production_id_count;
uint32_t field_count;
uint16_t max_alias_sequence_length;
const uint16_t *parse_table;
const uint16_t *small_parse_table;
const uint32_t *small_parse_table_map;
const TSParseActionEntry *parse_actions;
const char * const *symbol_names;
const char * const *field_names;
const TSFieldMapSlice *field_map_slices;
const TSFieldMapEntry *field_map_entries;
const TSSymbolMetadata *symbol_metadata;
const TSSymbol *public_symbol_map;
const uint16_t *alias_map;
const TSSymbol *alias_sequences;
const TSLexMode *lex_modes;
bool (*lex_fn)(TSLexer *, TSStateId);
bool (*keyword_lex_fn)(TSLexer *, TSStateId);
TSSymbol keyword_capture_token;
struct {
const bool *states;
const TSSymbol *symbol_map;
void *(*create)(void);
void (*destroy)(void *);
bool (*scan)(void *, TSLexer *, const bool *symbol_whitelist);
unsigned (*serialize)(void *, char *);
void (*deserialize)(void *, const char *, unsigned);
} external_scanner;
const TSStateId *primary_state_ids;
};
static inline bool set_contains(TSCharacterRange *ranges, uint32_t len, int32_t lookahead) {
uint32_t index = 0;
uint32_t size = len - index;
while (size > 1) {
uint32_t half_size = size / 2;
uint32_t mid_index = index + half_size;
TSCharacterRange *range = &ranges[mid_index];
if (lookahead >= range->start && lookahead <= range->end) {
return true;
} else if (lookahead > range->end) {
index = mid_index;
}
size -= half_size;
}
TSCharacterRange *range = &ranges[index];
return (lookahead >= range->start && lookahead <= range->end);
}
/*
* Lexer Macros
*/
#ifdef _MSC_VER
#define UNUSED __pragma(warning(suppress : 4101))
#else
#define UNUSED __attribute__((unused))
#endif
#define START_LEXER() \
bool result = false; \
bool skip = false; \
UNUSED \
bool eof = false; \
int32_t lookahead; \
goto start; \
next_state: \
lexer->advance(lexer, skip); \
start: \
skip = false; \
lookahead = lexer->lookahead;
#define ADVANCE(state_value) \
{ \
state = state_value; \
goto next_state; \
}
#define ADVANCE_MAP(...) \
{ \
static const uint16_t map[] = { __VA_ARGS__ }; \
for (uint32_t i = 0; i < sizeof(map) / sizeof(map[0]); i += 2) { \
if (map[i] == lookahead) { \
state = map[i + 1]; \
goto next_state; \
} \
} \
}
#define SKIP(state_value) \
{ \
skip = true; \
state = state_value; \
goto next_state; \
}
#define ACCEPT_TOKEN(symbol_value) \
result = true; \
lexer->result_symbol = symbol_value; \
lexer->mark_end(lexer);
#define END_STATE() return result;
/*
* Parse Table Macros
*/
#define SMALL_STATE(id) ((id) - LARGE_STATE_COUNT)
#define STATE(id) id
#define ACTIONS(id) id
#define SHIFT(state_value) \
{{ \
.shift = { \
.type = TSParseActionTypeShift, \
.state = (state_value) \
} \
}}
#define SHIFT_REPEAT(state_value) \
{{ \
.shift = { \
.type = TSParseActionTypeShift, \
.state = (state_value), \
.repetition = true \
} \
}}
#define SHIFT_EXTRA() \
{{ \
.shift = { \
.type = TSParseActionTypeShift, \
.extra = true \
} \
}}
#define REDUCE(symbol_name, children, precedence, prod_id) \
{{ \
.reduce = { \
.type = TSParseActionTypeReduce, \
.symbol = symbol_name, \
.child_count = children, \
.dynamic_precedence = precedence, \
.production_id = prod_id \
}, \
}}
#define RECOVER() \
{{ \
.type = TSParseActionTypeRecover \
}}
#define ACCEPT_INPUT() \
{{ \
.type = TSParseActionTypeAccept \
}}
#ifdef __cplusplus
}
#endif
#endif // TREE_SITTER_PARSER_H_