feat: HTTP embedding backend for self-hosted/remote endpoints

Adds support for OpenAI-compatible embedding endpoints as an alternative
to the local transformers.js pipeline. Enables using self-hosted servers
(Infinity, vLLM, TEI, llama.cpp) over Tailscale/VPN, or any cloud
endpoint — with higher-quality models like bge-large-en-v1.5 (1024d).

Configuration via environment variables:
  GITNEXUS_EMBEDDING_URL=http://your-server:8080/v1
  GITNEXUS_EMBEDDING_MODEL=BAAI/bge-large-en-v1.5
  GITNEXUS_EMBEDDING_API_KEY=your-key  (default: 'unused')
  GITNEXUS_EMBEDDING_DIMS=1024         (auto-detected if omitted)

When env vars are set:
- initEmbedder() skips local model download entirely
- embedText() and embedBatch() call the HTTP endpoint
- Dimensions auto-detected from first response
- Batches in groups of 64

When env vars are NOT set:
- Existing local transformers.js behavior is completely unchanged

Build: tsc clean
Tests: 1776 passed, 0 failed
This commit is contained in:
zm2231 2026-03-19 22:16:52 -04:00
parent c14a78a341
commit 9dc00e47a7
2 changed files with 127 additions and 4 deletions

View file

@ -18,7 +18,79 @@ import { pipeline, env, type FeatureExtractionPipeline } from '@huggingface/tran
import { existsSync } from 'fs';
import { execFileSync } from 'child_process';
import { join } from 'path';
import { DEFAULT_EMBEDDING_CONFIG, type EmbeddingConfig, type ModelProgress } from './types.js';
import { DEFAULT_EMBEDDING_CONFIG, type EmbeddingConfig, type HttpEmbeddingConfig, type ModelProgress } from './types.js';
// ─── HTTP Embedding Backend ───────────────────────────────────────────────────
// When GITNEXUS_EMBEDDING_URL is set, all embedding calls go to the HTTP
// endpoint instead of loading a local transformers.js model. This enables:
// - Self-hosted servers (Infinity, vLLM, TEI) over Tailscale/VPN
// - Higher-quality models (bge-large 1024d vs arctic-xs 384d)
// - Shared embedding infrastructure across tools
function getHttpConfig(): HttpEmbeddingConfig | null {
const baseUrl = process.env.GITNEXUS_EMBEDDING_URL;
const model = process.env.GITNEXUS_EMBEDDING_MODEL;
if (!baseUrl || !model) return null;
return {
baseUrl: baseUrl.replace(/\/+$/, ''),
model,
apiKey: process.env.GITNEXUS_EMBEDDING_API_KEY ?? 'unused',
dimensions: process.env.GITNEXUS_EMBEDDING_DIMS
? parseInt(process.env.GITNEXUS_EMBEDDING_DIMS, 10)
: undefined,
};
}
let httpConfig: HttpEmbeddingConfig | null | undefined;
let httpDimensions: number | null = null;
async function httpEmbed(texts: string[]): Promise<Float32Array[]> {
if (httpConfig === undefined) httpConfig = getHttpConfig();
if (!httpConfig) throw new Error('HTTP embedding not configured');
const url = `${httpConfig.baseUrl}/embeddings`;
const batchSize = 64;
const allVectors: Float32Array[] = [];
for (let i = 0; i < texts.length; i += batchSize) {
const batch = texts.slice(i, i + batchSize);
const resp = await fetch(url, {
method: 'POST',
headers: {
'Content-Type': 'application/json',
'Authorization': `Bearer ${httpConfig.apiKey}`,
},
body: JSON.stringify({ input: batch, model: httpConfig.model }),
});
if (!resp.ok) {
const body = await resp.text();
throw new Error(`Embedding endpoint ${resp.status}: ${body}`);
}
const data = (await resp.json()) as {
data: Array<{ embedding: number[] }>;
};
for (const item of data.data) {
allVectors.push(new Float32Array(item.embedding));
}
// Auto-detect dimensions from first response
if (httpDimensions === null && data.data.length > 0) {
httpDimensions = data.data[0].embedding.length;
}
}
return allVectors;
}
function isHttpMode(): boolean {
if (httpConfig === undefined) httpConfig = getHttpConfig();
return httpConfig !== null;
}
// ─── End HTTP Backend ─────────────────────────────────────────────────────────
/**
* Check whether CUDA libraries are actually available on this system.
@ -83,6 +155,12 @@ export const initEmbedder = async (
config: Partial<EmbeddingConfig> = {},
forceDevice?: 'dml' | 'cuda' | 'cpu' | 'wasm'
): Promise<FeatureExtractionPipeline> => {
// HTTP mode: skip local model loading entirely
if (isHttpMode()) {
// Return a dummy pipeline — embedText/embedBatch bypass it via isHttpMode()
return null as unknown as FeatureExtractionPipeline;
}
// Return existing instance if available
if (embedderInstance) {
return embedderInstance;
@ -195,7 +273,19 @@ export const initEmbedder = async (
* Check if the embedder is initialized and ready
*/
export const isEmbedderReady = (): boolean => {
return embedderInstance !== null;
return isHttpMode() || embedderInstance !== null;
};
/**
* Get the effective embedding dimensions.
* HTTP mode may use different dimensions than the local default.
*/
export const getEmbeddingDimensions = (): number => {
if (isHttpMode()) {
const cfg = getHttpConfig();
return cfg?.dimensions ?? httpDimensions ?? DEFAULT_EMBEDDING_CONFIG.dimensions;
}
return DEFAULT_EMBEDDING_CONFIG.dimensions;
};
/**
@ -212,9 +302,14 @@ export const getEmbedder = (): FeatureExtractionPipeline => {
* Embed a single text string
*
* @param text - Text to embed
* @returns Float32Array of embedding vector (384 dimensions)
* @returns Float32Array of embedding vector
*/
export const embedText = async (text: string): Promise<Float32Array> => {
if (isHttpMode()) {
const [vec] = await httpEmbed([text]);
return vec;
}
const embedder = getEmbedder();
const result = await embedder(text, {
@ -238,6 +333,10 @@ export const embedBatch = async (texts: string[]): Promise<Float32Array[]> => {
return [];
}
if (isHttpMode()) {
return httpEmbed(texts);
}
const embedder = getEmbedder();
// Process batch

View file

@ -53,7 +53,7 @@ export interface EmbeddingProgress {
* Configuration for the embedding pipeline
*/
export interface EmbeddingConfig {
/** Model identifier for transformers.js */
/** Model identifier for transformers.js (local) or the HTTP endpoint model name */
modelId: string;
/** Number of nodes to embed in each batch */
batchSize: number;
@ -65,6 +65,30 @@ export interface EmbeddingConfig {
maxSnippetLength: number;
}
/**
* Configuration for HTTP embedding endpoint (OpenAI-compatible).
* Set via environment variables:
* GITNEXUS_EMBEDDING_URL - Base URL (e.g. http://localhost:8080/v1)
* GITNEXUS_EMBEDDING_MODEL - Model name (e.g. BAAI/bge-large-en-v1.5)
* GITNEXUS_EMBEDDING_API_KEY - API key (default: "unused")
* GITNEXUS_EMBEDDING_DIMS - Dimensions (default: auto-detected from first response)
*
* Supports any OpenAI-compatible /v1/embeddings endpoint:
* - Self-hosted: Infinity, vLLM, TEI, llama.cpp
* - Cloud: OpenAI, Ollama (remote), LM Studio
* - VPS/Tailscale: any endpoint reachable over the network
*/
export interface HttpEmbeddingConfig {
/** Base URL for the embedding API (must include /v1) */
baseUrl: string;
/** Model name to send in the request */
model: string;
/** API key for authentication */
apiKey: string;
/** Override dimensions (auto-detected if not set) */
dimensions?: number;
}
/**
* Default embedding configuration
* Uses snowflake-arctic-embed-xs for browser efficiency