Stand-in import of elffuss/code
One parentless commit with only the files that the open seed tasks touch and the modules they import. The full squashed import (SPEC 8.6) is a later step: this tree is not the cutoff tree. T2T-StandIn-Source-Commit: 43f95b7333efe73aeaebf806023f17bffea27aba
This commit is contained in:
122
web/js/providers/api.js
Normal file
122
web/js/providers/api.js
Normal file
@@ -0,0 +1,122 @@
|
||||
// Proveedor genérico para APIs externas (configuración avanzada):
|
||||
// - kind 'openai' → /chat/completions (OpenAI, Ollama, llama-server…)
|
||||
// - kind 'anthropic' → /v1/messages (Claude)
|
||||
// Las llamadas salen DIRECTAS del navegador del usuario al proveedor; la clave
|
||||
// no pasa por ningún servidor nuestro. Streaming SSE en ambos dialectos.
|
||||
import { packHistoryAsync } from '../context.js';
|
||||
|
||||
let cfg = null;
|
||||
export let name = 'API';
|
||||
|
||||
export function configure(c) { cfg = c; name = c.label; }
|
||||
|
||||
export async function load() {
|
||||
if (!cfg) throw new Error('proveedor sin configurar');
|
||||
if (cfg.kind !== 'anthropic' && !cfg.apiKey && !cfg.baseURL.includes('localhost') && cfg.baseURL !== '/v1')
|
||||
throw new Error('falta la clave de API (config avanzada)');
|
||||
}
|
||||
|
||||
export async function chat(history, system, onToken = () => {}, signal = null) {
|
||||
return cfg.kind === 'anthropic'
|
||||
? anthropicChat(history, system, onToken)
|
||||
: openaiChat(history, system, onToken);
|
||||
}
|
||||
|
||||
// ---- OpenAI-compatible ----
|
||||
async function openaiChat(history, system, onToken) {
|
||||
const headers = { 'Content-Type': 'application/json' };
|
||||
if (cfg.apiKey) headers.Authorization = 'Bearer ' + cfg.apiKey;
|
||||
// packHistoryAsync: el empaquetado puede tener que codificar embeddings, así
|
||||
// que se resuelve ANTES de armar el cuerpo. Con el lado semántico apagado
|
||||
// devuelve exactamente lo mismo que la versión síncrona de siempre.
|
||||
const packed = await packHistoryAsync(history, 3000);
|
||||
const body = {
|
||||
model: cfg.model,
|
||||
messages: [{ role: 'system', content: system }, ...packed],
|
||||
stream: true,
|
||||
max_tokens: cfg.maxTokens || 1024,
|
||||
};
|
||||
if (cfg.temperature != null) body.temperature = cfg.temperature;
|
||||
if (cfg.top_p != null) body.top_p = cfg.top_p;
|
||||
if (cfg.thinking != null) body.chat_template_kwargs = { enable_thinking: cfg.thinking };
|
||||
|
||||
const res = await fetch(cfg.baseURL.replace(/\/$/, '') + '/chat/completions', {
|
||||
signal,
|
||||
method: 'POST', headers, body: JSON.stringify(body),
|
||||
});
|
||||
if (!res.ok || !res.body) throw new Error('HTTP ' + res.status + ' ' + (await res.text().catch(() => '')).slice(0, 120));
|
||||
|
||||
let out = '';
|
||||
await readSSE(res.body, payload => {
|
||||
if (payload === '[DONE]') return;
|
||||
const d = JSON.parse(payload).choices?.[0]?.delta || {};
|
||||
if (d.reasoning_content) onToken(d.reasoning_content);
|
||||
if (d.content) { out += d.content; onToken(d.content); }
|
||||
});
|
||||
return out.trim();
|
||||
}
|
||||
|
||||
// ---- Anthropic Messages ----
|
||||
async function anthropicChat(history, system, onToken) {
|
||||
const packed = await packHistoryAsync(history, 3000);
|
||||
const res = await fetch(cfg.baseURL.replace(/\/$/, '') + '/v1/messages', {
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
'x-api-key': cfg.apiKey,
|
||||
'anthropic-version': '2023-06-01',
|
||||
'anthropic-dangerous-direct-browser-access': 'true',
|
||||
},
|
||||
body: JSON.stringify({
|
||||
model: cfg.model,
|
||||
max_tokens: cfg.maxTokens || 1024,
|
||||
system,
|
||||
stream: true,
|
||||
messages: forAnthropic(packed),
|
||||
}),
|
||||
});
|
||||
if (!res.ok || !res.body) throw new Error('HTTP ' + res.status + ' ' + (await res.text().catch(() => '')).slice(0, 120));
|
||||
|
||||
let out = '';
|
||||
await readSSE(res.body, payload => {
|
||||
const evt = JSON.parse(payload);
|
||||
if (evt.type === 'content_block_delta' && evt.delta?.text) {
|
||||
out += evt.delta.text; onToken(evt.delta.text);
|
||||
}
|
||||
});
|
||||
return out.trim();
|
||||
}
|
||||
|
||||
// Anthropic exige roles alternos empezando por user; fusiona consecutivos.
|
||||
function forAnthropic(msgs) {
|
||||
const merged = [];
|
||||
for (const m of msgs) {
|
||||
const role = m.role === 'assistant' ? 'assistant' : 'user';
|
||||
const last = merged[merged.length - 1];
|
||||
if (last && last.role === role) last.content += '\n' + m.content;
|
||||
else merged.push({ role, content: m.content });
|
||||
}
|
||||
if (merged[0]?.role === 'assistant') merged.unshift({ role: 'user', content: '(continúa)' });
|
||||
return merged;
|
||||
}
|
||||
|
||||
// Lector SSE común: invoca fn con el texto tras cada 'data:'.
|
||||
async function readSSE(stream, fn) {
|
||||
const reader = stream.getReader();
|
||||
const dec = new TextDecoder();
|
||||
let buf = '';
|
||||
for (;;) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
buf += dec.decode(value, { stream: true });
|
||||
let nl;
|
||||
while ((nl = buf.indexOf('\n')) >= 0) {
|
||||
const line = buf.slice(0, nl).trim();
|
||||
buf = buf.slice(nl + 1);
|
||||
if (!line.startsWith('data:')) continue;
|
||||
const payload = line.slice(5).trim();
|
||||
if (!payload) continue;
|
||||
try { fn(payload); } catch { /* chunk parcial */ }
|
||||
}
|
||||
}
|
||||
}
|
||||
332
web/js/providers/litert.js
Normal file
332
web/js/providers/litert.js
Normal file
@@ -0,0 +1,332 @@
|
||||
// Gemma-4 vía LiteRT-LM de Google (early preview, solo WebGPU).
|
||||
// Patrón copiado de la demo verificada en agentic-install
|
||||
// (lab/bitacora/posts/08-jspace-live.html).
|
||||
// DECISIÓN 2026-07-14: cerebro = Gemma BASE (builds oficiales litert-community,
|
||||
// formato artisan). NO fine-tune propio: `@litert-lm/core` exige empaquetado
|
||||
// artisan y nuestras conversiones no lo producen (E-010). La agéntica va por el
|
||||
// system prompt (agent.js), no por pesos.
|
||||
export let name = 'Gemma · LiteRT-LM';
|
||||
|
||||
// Builds .litertlm elegibles. Los «-web» OFICIALES de Google (litert-community)
|
||||
// están exportados en formato artisan → SÍ cargan en el navegador (son los que
|
||||
// usaba la demo original). El healed de Elffuss es prefill_decode → hoy no carga
|
||||
// (E-010), por eso está gateado en el selector.
|
||||
export const MODELS = {
|
||||
'gemma-e2b': { url: 'https://huggingface.co/litert-community/gemma-4-E2B-it-litert-lm/resolve/main/gemma-4-E2B-it-web.litertlm', label: 'Gemma-4 E2B', tag: '~2 GB · ligero' },
|
||||
'gemma-e4b': { url: 'https://huggingface.co/litert-community/gemma-4-E4B-it-litert-lm/resolve/main/gemma-4-E4B-it-web.litertlm', label: 'Gemma-4 E4B', tag: '~4 GB · el mejor' },
|
||||
'elffuss-e4b': { url: 'https://huggingface.co/KikoCis/Elffuss-Gemma4-E4B-litert/resolve/main/model.litertlm', label: 'Elffuss E4B (healed)', tag: 'modelo propio' },
|
||||
};
|
||||
|
||||
let MODEL_URL = MODELS['gemma-e2b'].url;
|
||||
let curLabel = MODELS['gemma-e2b'].label;
|
||||
export function configure(key) {
|
||||
const m = MODELS[key] || MODELS['gemma-e2b'];
|
||||
MODEL_URL = m.url; curLabel = m.label; name = 'Gemma · LiteRT-LM (' + m.label + ')';
|
||||
}
|
||||
|
||||
let engine = null, conversation = null, sentCount = 0, sys = '';
|
||||
|
||||
// Muestreo. Por defecto VORAZ, que es como venía: mismo prompt, misma salida.
|
||||
// Eso hace que generar N veces cueste N y devuelva una sola respuesta distinta,
|
||||
// así que cualquier técnica de «genera varias y quédate con la mejor» era gasto
|
||||
// puro. Con temperatura y semilla se puede pedir variedad de verdad.
|
||||
const TOP_P = 2;
|
||||
let muestreo = null; // null = voraz
|
||||
export function setSampling(opts) { // {temperature, seed, p} o null
|
||||
const antes = JSON.stringify(muestreo);
|
||||
// «k» es OBLIGATORIO aunque el tipo sea TOP_P: sin él el wasm aborta en seco
|
||||
// con «Aborted()» y se lleva por delante el motor. Comprobado probando formas.
|
||||
muestreo = opts && opts.temperature > 0
|
||||
? { type: TOP_P, k: opts.k ?? 40, p: opts.p ?? 0.95, temperature: opts.temperature, seed: opts.seed ?? 0 }
|
||||
: null;
|
||||
// cambiar el muestreo exige rehacer la conversación: la sesión ya está creada
|
||||
// con los parámetros de antes y no los relee.
|
||||
if (JSON.stringify(muestreo) !== antes) { conversation = null; sentCount = 0; }
|
||||
return muestreo;
|
||||
}
|
||||
export function getSampling() { return muestreo; }
|
||||
export function __engine() { return engine; } // solo para el banco de pruebas
|
||||
|
||||
// Contexto: probamos de mayor a menor hasta el máximo que acepten el bundle y
|
||||
// la memoria GPU — así el contexto queda al tope permitido de serie.
|
||||
const CTX_LADDER = [32768, 16384, 8192, 4096];
|
||||
export let ctxTokens = 4096; // efectivo tras load() (la UI puede leerlo)
|
||||
|
||||
export async function load(onProgress = () => {}) {
|
||||
if (!navigator.gpu) throw new Error('LiteRT-LM necesita WebGPU (Chrome/Edge modernos)');
|
||||
// navigator.gpu puede existir como API sin adaptador real (algunos Linux/
|
||||
// drivers, entornos sandboxed…) — comprobarlo YA evita bajar 2-4 GB para
|
||||
// descubrir el fallo solo al crear el motor, al final de todo.
|
||||
let adapter = null;
|
||||
try { adapter = await navigator.gpu.requestAdapter(); } catch { /* sin adaptador */ }
|
||||
if (!adapter) throw new Error('No hay un adaptador WebGPU real disponible (la API existe pero no hay GPU accesible) — prueba con Elffuss LM, que corre en CPU/wasm.');
|
||||
// VERSIÓN FIJADA a propósito. Sin fijarla, la URL apunta siempre a la última
|
||||
// publicada: el 2026-08-11 salió 0.16.0, jsdelivr NO consigue construirle el
|
||||
// bundle `+esm` (404) y el cerebro Gemma dejó de cargar en producción sin que
|
||||
// nosotros tocáramos una línea. Al subir de versión hay que COMPROBAR que
|
||||
// `https://cdn.jsdelivr.net/npm/@litert-lm/core@<v>/+esm` responde 200.
|
||||
const litertlm = await import('https://cdn.jsdelivr.net/npm/@litert-lm/core@0.15.0/+esm');
|
||||
// El .litertlm lo descargamos NOSOTROS (cache-first en Cache Storage) y se lo
|
||||
// pasamos a Engine.create como Blob (la API acepta string|Blob|ReadableStream).
|
||||
// Motivo: el fetch interno de LiteRT baja el peso con XHR+Range desde un WORKER
|
||||
// que el service worker no intercepta → antes se re-descargaba SIEMPRE. Bajándolo
|
||||
// aquí queda cacheado de verdad y damos progreso real en MB.
|
||||
const model = await cachedModelBlob(MODEL_URL, onProgress);
|
||||
onProgress('Preparando el modelo IA en la GPU…');
|
||||
let lastErr = null;
|
||||
for (const n of CTX_LADDER) {
|
||||
try {
|
||||
engine = await litertlm.Engine.create({ model, mainExecutorSettings: { maxNumTokens: n } });
|
||||
ctxTokens = n;
|
||||
lastErr = null;
|
||||
break;
|
||||
} catch (e) {
|
||||
lastErr = e;
|
||||
// Errores de formato/carga no dependen del contexto: no insistir con la escalera.
|
||||
if (/not supported|tokenizer|format/i.test(String(e?.message))) throw e;
|
||||
onProgress(`Contexto ${n} no cabe, probando ${n / 2}…`);
|
||||
}
|
||||
}
|
||||
if (lastErr) throw lastErr;
|
||||
}
|
||||
|
||||
const MODEL_CACHE = 'elffuss-models-v1';
|
||||
// Devuelve el .litertlm como Blob desde Cache Storage; si no está, lo descarga
|
||||
// con progreso real y lo cachea (persistente). Ante cualquier fallo, devuelve la
|
||||
// URL para que LiteRT lo baje por su cuenta (nunca bloquea la carga del modelo).
|
||||
export async function cachedModelBlob(url, onProgress = () => {}) {
|
||||
// Caché COMPARTIDA + OPFS (runtime propio): un modelo bajado en cualquier web
|
||||
// de Elffuss se reutiliza sin re-descargar. Si no está disponible, caemos al
|
||||
// Cache Storage de siempre.
|
||||
try {
|
||||
const { getModelFile } = await import('../runtime/model-store.js');
|
||||
const f = await getModelFile(url, onProgress);
|
||||
if (f) return f;
|
||||
} catch (e) { console.warn('[elffuss] OPFS/broker no disponible, uso Cache Storage:', e); }
|
||||
|
||||
if (!self.caches) return url;
|
||||
try {
|
||||
const cache = await caches.open(MODEL_CACHE);
|
||||
const hit = await cache.match(url);
|
||||
if (hit) { onProgress('Cargando el modelo IA desde caché (sin descargar)…'); return await hit.blob(); }
|
||||
const net = await fetch(url);
|
||||
if (!net.ok || !net.body) return url;
|
||||
const total = +net.headers.get('content-length') || 0;
|
||||
const t0 = performance.now();
|
||||
// Progreso SIN tee(): con un modelo de gigabytes, tee() crea dos ramas que
|
||||
// se consumen a ritmos distintos y el navegador tiene que bufferizar la
|
||||
// diferencia en memoria → el cache.put acababa reventando y el modelo NO se
|
||||
// cacheaba NUNCA (medido con E4B: 2832 MB bajados y cero guardados). Peor
|
||||
// aún: al fallar se devolvía la URL suelta y `Engine.create({model: <URL>})`
|
||||
// monta un motor que CARGA pero no genera («Aborted()»). Con un
|
||||
// TransformStream hay un solo consumidor: contamos al vuelo y el mismo flujo
|
||||
// va a la caché.
|
||||
let loaded = 0;
|
||||
const counted = net.body.pipeThrough(new TransformStream({
|
||||
transform(chunk, ctrl) {
|
||||
loaded += chunk.byteLength ?? chunk.length;
|
||||
onProgress(fmtBytes(loaded, total, t0));
|
||||
ctrl.enqueue(chunk);
|
||||
},
|
||||
}));
|
||||
const headers = { 'Content-Type': 'application/octet-stream' };
|
||||
if (total) headers['Content-Length'] = String(total);
|
||||
// Cachear GIGABYTES puede fallar de verdad: ventana privada (Cache Storage
|
||||
// en memoria), disco lleno, cuota del origen. Si falla hay que DECIRLO: el
|
||||
// progreso ya ha prometido «se cachea para la próxima vez» y, callándolo,
|
||||
// el usuario se re-baja el modelo entero cada sesión sin saber por qué.
|
||||
try {
|
||||
await cache.put(url, new Response(counted, { headers }));
|
||||
} catch (e) {
|
||||
onProgress(`No se pudo guardar el modelo en caché (${e.name || 'error'}): habrá que descargarlo otra vez la próxima. ` +
|
||||
`Suele ser ventana privada o falta de espacio.`);
|
||||
console.warn('[elffuss] modelo NO cacheado:', e);
|
||||
return url;
|
||||
}
|
||||
const cached = await cache.match(url);
|
||||
if (!cached) { onProgress('No se pudo guardar el modelo en caché: habrá que descargarlo otra vez la próxima.'); return url; }
|
||||
return await cached.blob();
|
||||
} catch (e) {
|
||||
console.warn('[elffuss] caché de modelo no disponible:', e);
|
||||
return url;
|
||||
}
|
||||
}
|
||||
function fmtBytes(loaded, total, t0) {
|
||||
const mb = n => (n / 1048576).toFixed(0);
|
||||
const secs = (performance.now() - t0) / 1000;
|
||||
const spd = secs > 0 ? (loaded / 1048576 / secs).toFixed(1) : '0';
|
||||
return total
|
||||
? `Descargando el modelo IA… ${mb(loaded)}/${mb(total)} MB (${spd} MB/s) · se cachea para la próxima vez`
|
||||
: `Descargando el modelo IA… ${mb(loaded)} MB (${spd} MB/s)`;
|
||||
}
|
||||
|
||||
// Liberar el modelo (vigilante de RAM).
|
||||
export async function unload() {
|
||||
try { engine?.close?.(); } catch { /* mejor esfuerzo */ }
|
||||
engine = null; conversation = null; sentCount = 0;
|
||||
}
|
||||
|
||||
export async function chat(history, system, onToken = () => {}, signal = null) {
|
||||
if (!engine) throw new Error('Modelo no cargado');
|
||||
// Comparar solo la parte estática del prompt: el CONTEXTO AHORA va al final
|
||||
// y cambia cada turno — recrear la conversación tiraría el KV-cache.
|
||||
const sysKey = system.slice(0, 200);
|
||||
// Una conversación NUEVA llega con el historial reiniciado (más corto que lo
|
||||
// ya enviado). Sin esta comprobación, el KV-cache conservaba el chat anterior
|
||||
// entero: el modelo seguía «viendo» los ficheros y respuestas del chat de
|
||||
// antes y contestaba sobre ellos. Se notaba como alucinación («ese fichero no
|
||||
// existe») cuando en realidad era memoria de la conversación previa — y es
|
||||
// además una fuga: un chat nuevo no debe ver el contenido del anterior.
|
||||
const reiniciada = history.length <= sentCount;
|
||||
if (!conversation || sysKey !== sys || reiniciada) {
|
||||
sys = sysKey;
|
||||
conversation = await crearConversacion(system);
|
||||
sentCount = 0;
|
||||
}
|
||||
// La conversación LiteRT mantiene su propio KV-cache: enviamos solo lo nuevo.
|
||||
const fresh = history.slice(sentCount).filter(m => m.role === 'user');
|
||||
const nuevos = fresh.length ? fresh : [history.at(-1)];
|
||||
let text = nuevos.map(m => m.content).join('\n');
|
||||
|
||||
// ¿Cabe? Si no: compactar lo anterior y, si ni así, recortar (ver CONTEXTO).
|
||||
let antes = await tokensUsados();
|
||||
if (antes != null && estimaTokens(text) > ctxTokens - antes - RESERVA_SALIDA) {
|
||||
const ocupaba = estimaTokens(text);
|
||||
if (sentCount > 0) { await rehacerCompactada(history, sentCount, system, nuevos); antes = await tokensUsados(); }
|
||||
const largo = text.length;
|
||||
text = ajustarAlContexto(nuevos, libreEnCaracteres(antes, 0.9));
|
||||
console.warn(`[litert] el mensaje nuevo (~${ocupaba} tokens) no cabía en un contexto de ${ctxTokens}: ` +
|
||||
`${sentCount > 0 ? 'conversación compactada' : 'conversación vacía'}, mensaje ${text.length < largo ? 'recortado' : 'entero'} ` +
|
||||
`· usados ${antes} · se mandan ${text.length} caracteres (${largo} tenía) · ${charsPorToken} caracteres/token`);
|
||||
}
|
||||
|
||||
let out = '';
|
||||
const enviar = async t => {
|
||||
for await (const chunk of conversation.sendMessageStreaming(t)) {
|
||||
if (signal?.aborted) break; // parar: se devuelve lo generado hasta aquí
|
||||
for (const item of (chunk.content || []))
|
||||
if (item.type === 'text') { out += item.text; onToken(item.text); }
|
||||
}
|
||||
};
|
||||
try {
|
||||
await enviar(text);
|
||||
} catch (e) {
|
||||
// Si falla ANTES de escribir nada, lo normal es que no cupiera (la cuenta de
|
||||
// caracteres por token se quedó corta): se rehace compactada, se recorta a la
|
||||
// mitad de lo libre y se intenta UNA vez más. Si ya había escrito algo,
|
||||
// reintentar lo duplicaría: se deja subir el error.
|
||||
if (out || signal?.aborted) throw e;
|
||||
console.warn(`[litert] el envío falló sin respuesta (usados ${antes} · ${text.length} caracteres); ` +
|
||||
`rehago la conversación compactada y reintento una vez: ${e?.message || e}`);
|
||||
await rehacerCompactada(history, sentCount, system, nuevos);
|
||||
antes = await tokensUsados();
|
||||
text = ajustarAlContexto(nuevos, libreEnCaracteres(antes, 0.5));
|
||||
console.warn(`[litert] reintento: usados ${antes} · se mandan ${text.length} caracteres`);
|
||||
await enviar(text);
|
||||
}
|
||||
// Se marca como enviado DESPUÉS de enviarlo. Antes se marcaba al principio, y
|
||||
// un envío fallido dejaba el mensaje por enviado sin haber llegado nunca.
|
||||
sentCount = history.length;
|
||||
const despues = await tokensUsados();
|
||||
if (antes != null && despues != null && despues - antes > 64)
|
||||
charsPorToken = Math.min(2, Math.max(1.2, (text.length + out.length) / (despues - antes)));
|
||||
return out.trim();
|
||||
}
|
||||
|
||||
function crearConversacion(system) {
|
||||
return engine.createConversation({
|
||||
preface: { messages: [{ role: 'system', content: system }] },
|
||||
// Exprimir el navegador: no persistir los tokens de canal (tool-call/thinking)
|
||||
// del modelo en el KV-cache → libera KV → más contexto útil. Y prefill del
|
||||
// system prompt al crear la conversación → primera respuesta más rápida.
|
||||
filterChannelContentFromKvCache: true,
|
||||
prefillPrefaceOnInit: true,
|
||||
...(muestreo ? { sessionConfig: { samplerParams: muestreo } } : {}),
|
||||
});
|
||||
}
|
||||
|
||||
// ── CONTEXTO: que un resultado grande no tumbe la conversación ───────────────
|
||||
// Los proveedores sin estado pasan el historial por acer-core en cada llamada.
|
||||
// Este no: LiteRT guarda la conversación en su caché KV y aquí solo se le manda
|
||||
// lo nuevo, así que NADA la empaquetaba. Y lo nuevo puede ser enorme: fs.read
|
||||
// devuelve hasta 200.000 caracteres, decenas de miles de tokens, con un
|
||||
// contexto de 4.096 a 32.768. Antes de mandar se mira lo que queda
|
||||
// (getTokenCount) y, si no cabe:
|
||||
// 1. se rehace la conversación con lo anterior EMPAQUETADO por acer-core
|
||||
// —puntuado con la pregunta de ahora— dentro del prompt de sistema. Va como
|
||||
// texto y no como mensajes con rol a propósito: qué roles acepta el
|
||||
// preámbulo de Gemma no está documentado, y el texto no depende de eso;
|
||||
// 2. si ni así cabe, se recortan por el medio los resultados de herramienta
|
||||
// del mensaje nuevo (cabeza y cola, que es donde suele estar lo que
|
||||
// importa), avisando al modelo de que puede pedir un trozo concreto.
|
||||
// Si cabe, el mensaje sale exactamente igual que antes.
|
||||
//
|
||||
// Los caracteres por token no se saben de antemano (el tokenizador vive dentro
|
||||
// del wasm): se empieza en 2 y lo medido con getTokenCount solo puede BAJARLO.
|
||||
// Medido con Gemma E4B: la prosa sale a ~2,8 y un informe cargado de cifras a
|
||||
// 2,35. Dejando que subiera, el saludo calibraba a 2,81 y el informe de después
|
||||
// se mandaba contado con esa cifra: 77.112 caracteres que eran ~32.800 tokens, y
|
||||
// el envío fallaba («Too many tokens requested») hasta el reintento, pagando dos
|
||||
// veces la espera. Una proporción medida con un contenido no vale para otro, y
|
||||
// equivocarse por arriba cuesta un fallo; por abajo, solo recortar algo de más.
|
||||
const RESERVA_SALIDA = 1024; // tokens que se dejan para que conteste
|
||||
const FRAC_HISTORIAL = 0.35; // techo del historial empaquetado al rehacer
|
||||
let charsPorToken = 2;
|
||||
const estimaTokens = s => Math.ceil(s.length / charsPorToken);
|
||||
|
||||
async function tokensUsados() {
|
||||
try { const n = await conversation.getTokenCount(); return Number.isFinite(n) ? n : null; }
|
||||
catch { return null; }
|
||||
}
|
||||
const libreEnCaracteres = (usados, fraccion) =>
|
||||
Math.floor(Math.max(256, ctxTokens - (usados ?? 0) - RESERVA_SALIDA) * charsPorToken * fraccion);
|
||||
|
||||
async function rehacerCompactada(history, hasta, system, nuevos) {
|
||||
let bloque = '';
|
||||
const previos = history.slice(0, hasta);
|
||||
if (previos.length) {
|
||||
const { packHistoryAsync } = await import('../context.js');
|
||||
const { estimateTokens } = await import('../acer-core.js');
|
||||
// El presupuesto de acer-core va en SUS tokens estimados, no en los del modelo.
|
||||
const caracteres = previos.reduce((a, m) => a + m.content.length, 0) || 1;
|
||||
const suyos = previos.reduce((a, m) => a + estimateTokens(m.content), 0);
|
||||
const presupuesto = Math.max(100, Math.round(ctxTokens * FRAC_HISTORIAL * charsPorToken * suyos / caracteres));
|
||||
// La pregunta de AHORA va al final para que acer-core puntúe con ella; luego se quita.
|
||||
const noEsResultado = m => m.role === 'user' && !m.content.startsWith('[resultado');
|
||||
const pregunta = ([...nuevos].reverse().find(noEsResultado) || [...history].reverse().find(noEsResultado))?.content || '';
|
||||
const empaquetado = (await packHistoryAsync([...previos, { role: 'user', content: pregunta }], presupuesto)).slice(0, -1);
|
||||
bloque = '\n\nCONVERSACIÓN HASTA AHORA (no cabía entera: va lo más relevante para lo que se pide ahora):\n' +
|
||||
empaquetado.map(m => `${m.role === 'assistant' ? 'Elffuss' : 'Usuario'}: ${m.content}`).join('\n');
|
||||
}
|
||||
try { await conversation?.delete?.(); } catch { /* ya no estaba */ }
|
||||
conversation = await crearConversacion(system + bloque);
|
||||
}
|
||||
|
||||
function ajustarAlContexto(nuevos, maxCaracteres) {
|
||||
const entero = nuevos.map(m => m.content).join('\n');
|
||||
if (entero.length <= maxCaracteres) return entero;
|
||||
// Un mensaje puede traer VARIOS resultados seguidos (Elffuss Code junta los de
|
||||
// un mismo paso): se recorta cada uno por su lado, o el primero se comería el
|
||||
// sitio de los demás.
|
||||
const mensajes = nuevos.map(m => m.content.split(/\n\n(?=\[resultado )/));
|
||||
const esResultado = b => b.startsWith('[resultado');
|
||||
const deResultados = mensajes.flat().reduce((a, b) => a + (esResultado(b) ? b.length : 0), 0);
|
||||
const hueco = Math.max(0, maxCaracteres - (entero.length - deResultados));
|
||||
const t = deResultados
|
||||
? mensajes.map(bs => bs.map(b => (esResultado(b) ? recortarPorElMedio(b, Math.floor(b.length * hueco / deResultados)) : b)).join('\n\n')).join('\n')
|
||||
: entero;
|
||||
return t.length <= maxCaracteres ? t : recortarPorElMedio(t, maxCaracteres);
|
||||
}
|
||||
|
||||
function recortarPorElMedio(s, max) {
|
||||
if (s.length <= max) return s;
|
||||
const aviso = `\n… [recortado: ${s.length - max} caracteres no caben en el contexto; si hace falta, pide un fragmento concreto] …\n`;
|
||||
const util = Math.max(0, max - aviso.length);
|
||||
const cabeza = Math.ceil(util * 0.7);
|
||||
return s.slice(0, cabeza) + aviso + s.slice(s.length - (util - cabeza));
|
||||
}
|
||||
|
||||
// Solo para tests/litert-contexto.mjs: probar chat() en node con un motor de mentira.
|
||||
export function __usarMotor(motor, contexto) {
|
||||
engine = motor; ctxTokens = contexto; conversation = null; sentCount = 0; sys = ''; charsPorToken = 2;
|
||||
}
|
||||
84
web/js/providers/onnx.js
Normal file
84
web/js/providers/onnx.js
Normal file
@@ -0,0 +1,84 @@
|
||||
// Modelo vía ONNX Runtime Web (transformers.js, WebGPU con fallback a wasm).
|
||||
// Patrón copiado de la demo verificada en agentic-install
|
||||
// (lab/bitacora/posts/08-jspace-live.html): dtype 'q4' obligatorio — q4f16
|
||||
// genera basura vía WebGPU incluso con shader-f16.
|
||||
import { MODEL, ONNX_MODELS, setOnnxModel } from '../model-config.js';
|
||||
import { packHistoryAsync } from '../context.js';
|
||||
|
||||
export let name = MODEL.label;
|
||||
|
||||
// Elegir qué ONNX cargar. Si cambia respecto al ya cargado, se descarta la
|
||||
// sesión para que load() cree la nueva (un modelo distinto = otra sesión).
|
||||
let generator = null, TextStreamer = null, cargadoKey = null;
|
||||
export function configure(key) {
|
||||
const antes = MODEL.key;
|
||||
setOnnxModel(key);
|
||||
name = MODEL.label;
|
||||
if (MODEL.key !== cargadoKey && generator) { try { generator?.dispose?.(); } catch {} generator = null; }
|
||||
return MODEL.key !== antes;
|
||||
}
|
||||
export function models() { return Object.values(ONNX_MODELS); }
|
||||
|
||||
export async function load(onProgress = () => {}) {
|
||||
if (generator) return; // un solo modelo: nunca recargar/duplicar la sesión
|
||||
const tf = await import('https://cdn.jsdelivr.net/npm/@huggingface/transformers@4');
|
||||
TextStreamer = tf.TextStreamer;
|
||||
if (MODEL.selfHosted) {
|
||||
tf.env.allowRemoteModels = false;
|
||||
tf.env.localModelPath = MODEL.basePath;
|
||||
}
|
||||
// navigator.gpu puede EXISTIR como API sin que haya un adaptador real
|
||||
// (ciertos Linux/drivers, entornos sandboxed, navegadores headless…) — usar
|
||||
// solo la presencia del objeto como señal hace que pipeline() intente
|
||||
// WebGPU, falle con "Failed to get GPU adapter" DESPUÉS de descargar el
|
||||
// modelo entero, y en preloadModel() eso encadena a probar Gemma (varios
|
||||
// GB) para nada. Comprobar el adaptador de verdad antes de elegir.
|
||||
let device = 'wasm';
|
||||
if (navigator.gpu) {
|
||||
try { device = (await navigator.gpu.requestAdapter()) ? 'webgpu' : 'wasm'; }
|
||||
catch { device = 'wasm'; }
|
||||
}
|
||||
generator = await tf.pipeline('text-generation', MODEL.id, {
|
||||
device,
|
||||
dtype: MODEL.dtype,
|
||||
progress_callback: onProgress,
|
||||
});
|
||||
cargadoKey = MODEL.key;
|
||||
}
|
||||
|
||||
// Liberar el modelo (vigilante de RAM): suelta los buffers wasm/WebGPU.
|
||||
export async function unload() {
|
||||
try { await generator?.dispose?.(); } catch { /* mejor esfuerzo */ }
|
||||
generator = null; cargadoKey = null;
|
||||
}
|
||||
|
||||
export async function chat(history, system, onToken = () => {}, signal = null) {
|
||||
if (!generator) throw new Error('Modelo no cargado');
|
||||
// ACE-lite: eviction por relevancia. Presupuesto amplio (LFM2.5 aguanta
|
||||
// contexto largo); el tope POR MENSAJE (context.js) evita que un README
|
||||
// gigante dispare «Too many tokens requested».
|
||||
const messages = [{ role: 'system', content: system }, ...(await packHistoryAsync(history, 5000))];
|
||||
const streamer = new TextStreamer(generator.tokenizer, {
|
||||
skip_prompt: true,
|
||||
skip_special_tokens: true,
|
||||
callback_function: t => { if (!signal?.aborted) onToken(t); },
|
||||
});
|
||||
const out = await generator(messages, {
|
||||
max_new_tokens: 1024,
|
||||
do_sample: false, // determinista: los tool calls JSON lo agradecen
|
||||
repetition_penalty: 1.1,
|
||||
return_full_text: false,
|
||||
streamer,
|
||||
});
|
||||
const gen = out[0].generated_text;
|
||||
let txt = (typeof gen === 'string' ? gen : gen.at(-1).content);
|
||||
// Modelos de razonamiento (Qwen3): fuera el <think>. Si quedó abierto por el
|
||||
// tope de tokens, nos quedamos con lo de después de la última apertura.
|
||||
if (txt.includes('<think>')) {
|
||||
txt = txt.replace(/<think>[\s\S]*?<\/think>/g, '');
|
||||
const i = txt.lastIndexOf('<think>');
|
||||
if (i !== -1) txt = txt.slice(i + 7);
|
||||
txt = txt.replace(/<\/?think>/g, '');
|
||||
}
|
||||
return txt.trim();
|
||||
}
|
||||
48
web/js/providers/rules.js
Normal file
48
web/js/providers/rules.js
Normal file
@@ -0,0 +1,48 @@
|
||||
// Modo básico de Elffuss Code: órdenes deterministas sin modelo.
|
||||
export const name = 'Básico (sin modelo)';
|
||||
export async function load() {}
|
||||
|
||||
const call = obj => '```tool\n' + JSON.stringify(obj) + '\n```';
|
||||
|
||||
const HELP = `Estoy en modo básico (sin modelo). Entiendo:
|
||||
• «árbol» / «qué hay en el proyecto»
|
||||
• «lee <archivo>» · «abre <archivo>»
|
||||
• «busca <texto>»
|
||||
• «escribe <archivo>: <contenido>»
|
||||
Para programar de verdad, carga el modelo local (selector 🧠) o configura uno en ⚙️.`;
|
||||
|
||||
export async function chat(history, systemPrompt) {
|
||||
const last = history[history.length - 1];
|
||||
const text = (last?.content || '').trim();
|
||||
|
||||
// 🎯 Modo Objetivo (goal.js) también funciona en modo básico: un plan fijo
|
||||
// de 2 tareas (explorar + escribir), para poder probar/usar el planificador
|
||||
// sin depender de un modelo real cargado.
|
||||
if (/^ROL: PLANIFICADOR/.test(systemPrompt || '')) {
|
||||
return JSON.stringify({
|
||||
plan: 'Plan básico (sin modelo) para: ' + text.slice(0, 60),
|
||||
tasks: [
|
||||
{ title: 'Explorar el proyecto', description: 'Mira el árbol del proyecto para entender qué hay.' },
|
||||
{ title: 'Escribir el resultado', description: 'escribe objetivo.txt: ' + text },
|
||||
],
|
||||
});
|
||||
}
|
||||
|
||||
if (text.startsWith('[resultado')) {
|
||||
const body = text.slice(text.indexOf('\n') + 1).trim();
|
||||
return body.startsWith('ERROR:') ? 'No pude: ' + body.slice(6).trim() : body;
|
||||
}
|
||||
|
||||
const t = text.toLowerCase();
|
||||
if (/(árbol|arbol|estructura|qué hay|que hay)/.test(t))
|
||||
return call({ tool: 'code.tree', args: {} });
|
||||
const mRead = text.match(/(?:lee|abre|muestra|cat)\s+([\w./-]+\.\w+)/i);
|
||||
if (mRead) return call({ tool: 'code.read', args: { path: mRead[1] } });
|
||||
const mSearch = text.match(/busca(?:r)?\s+(?:d[oó]nde\s+)?(?:se\s+\w+\s+)?["«']?([^"»']{2,60})["»']?$/i);
|
||||
if (mSearch) return call({ tool: 'code.search', args: { query: mSearch[1].trim() } });
|
||||
const mWrite = text.match(/escribe\s+([\w./-]+\.\w+)\s*[:=]\s*([\s\S]+)/i);
|
||||
if (mWrite) return call({ tool: 'code.write', args: { path: mWrite[1], content: mWrite[2] } });
|
||||
|
||||
if (/^(hola|buenas|hey|hi)\b/.test(t)) return '¡Hola! ' + HELP;
|
||||
return HELP;
|
||||
}
|
||||
Reference in New Issue
Block a user