581 lines
20 KiB
JavaScript
581 lines
20 KiB
JavaScript
/**
|
|
* Copilot provider layer.
|
|
*
|
|
* Resolution order: OpenRouter if a key is set, else Ollama if it answers, else
|
|
* the deterministic rule-based fallback. The fallback is not an error path - it
|
|
* is a supported mode, and COPILOT_PROVIDER=fallback selects it deliberately.
|
|
*
|
|
* The hard requirement is that a question ALWAYS gets an answer. A missing key,
|
|
* an unreachable host, a model that 404s, a stream that dies halfway: every one
|
|
* of those degrades to the fallback rather than surfacing an error on stage.
|
|
*/
|
|
|
|
import { buildContext, SYSTEM_PROMPT } from './context.js';
|
|
import { answerFromRules } from './rules.js';
|
|
import { loadSettings, saveSettings, validatePatch, resetSettings } from './settings.js';
|
|
|
|
const PROBE_TIMEOUT_MS = 2000;
|
|
const REQUEST_TIMEOUT_MS = 90000;
|
|
|
|
/**
|
|
* Turn whatever OLLAMA_HOST happens to contain into a dialable origin.
|
|
*
|
|
* Ollama itself commonly sets OLLAMA_HOST=0.0.0.0 machine-wide to bind all
|
|
* interfaces. That is a BIND address, not a destination - you cannot connect to
|
|
* it - and because it is a real environment variable it silently overrides any
|
|
* default we set here. Also accepts a bare host, a host:port, or a full URL.
|
|
*/
|
|
export function normalizeOllamaHost(raw) {
|
|
let h = String(raw || '').trim();
|
|
if (!h) return 'http://127.0.0.1:11434';
|
|
if (!/^https?:\/\//i.test(h)) h = `http://${h}`;
|
|
try {
|
|
const u = new URL(h);
|
|
if (u.hostname === '0.0.0.0' || u.hostname === '::' || u.hostname === '[::]') {
|
|
u.hostname = '127.0.0.1';
|
|
}
|
|
if (!u.port) u.port = '11434';
|
|
return u.origin;
|
|
} catch {
|
|
return 'http://127.0.0.1:11434';
|
|
}
|
|
}
|
|
|
|
export class Copilot {
|
|
constructor(env = process.env) {
|
|
this.env = env;
|
|
// The key stays in the environment only. It is never part of the settings
|
|
// that the settings page can read or write.
|
|
this.openRouterKey = (env.OPENROUTER_API_KEY || '').trim();
|
|
this.applySettings(loadSettings(env));
|
|
|
|
this.provider = 'fallback';
|
|
this.model = null;
|
|
this.detail = 'not yet detected';
|
|
this.modelCache = new Map();
|
|
}
|
|
|
|
/** Adopt a settings object. Does not persist; see configure(). */
|
|
applySettings(settings) {
|
|
this.settings = { ...settings };
|
|
this.forced = settings.provider === 'auto' ? null : settings.provider;
|
|
this.openRouterModel = settings.openRouterModel;
|
|
this.ollamaModel = settings.ollamaModel;
|
|
this.ollamaHost = normalizeOllamaHost(settings.ollamaHost);
|
|
this.temperature = settings.temperature;
|
|
this.maxTokens = settings.maxTokens;
|
|
this.reasoningEffort = settings.reasoningEffort;
|
|
}
|
|
|
|
/**
|
|
* Validate, apply, persist and re-detect.
|
|
*
|
|
* Returns { ok, errors, settings, status }. A rejected patch changes nothing.
|
|
*/
|
|
async configure(patch) {
|
|
const { ok, errors, clean } = validatePatch(patch);
|
|
if (!ok) return { ok: false, errors, settings: this.publicSettings(), status: this.status() };
|
|
|
|
this.applySettings({ ...this.settings, ...clean });
|
|
saveSettings(this.settings);
|
|
// The chosen host may have changed, so availability has to be re-probed.
|
|
const status = await this.detect();
|
|
return { ok: true, errors: [], settings: this.publicSettings(), status };
|
|
}
|
|
|
|
/** Restore environment defaults, discarding the saved settings file. */
|
|
async reset() {
|
|
resetSettings();
|
|
this.applySettings(loadSettings(this.env));
|
|
const status = await this.detect();
|
|
return { ok: true, errors: [], settings: this.publicSettings(), status };
|
|
}
|
|
|
|
/** Settings safe to hand to a browser: never includes the API key. */
|
|
publicSettings() {
|
|
return {
|
|
...this.settings,
|
|
// Report the normalized host, since that is what actually gets dialled.
|
|
resolvedOllamaHost: this.ollamaHost,
|
|
openRouterKeyPresent: Boolean(this.openRouterKey),
|
|
};
|
|
}
|
|
|
|
/** Probe available providers. Safe to call repeatedly. */
|
|
async detect() {
|
|
if (this.forced === 'fallback') {
|
|
this.provider = 'fallback';
|
|
this.model = null;
|
|
this.detail = 'forced by COPILOT_PROVIDER=fallback';
|
|
return this.status();
|
|
}
|
|
|
|
if (this.openRouterKey && this.forced !== 'ollama') {
|
|
this.provider = 'openrouter';
|
|
this.model = this.openRouterModel;
|
|
this.detail = 'OpenRouter API key present';
|
|
return this.status();
|
|
}
|
|
|
|
if (this.forced !== 'openrouter') {
|
|
const reachable = await this.probeOllama();
|
|
if (reachable) {
|
|
this.provider = 'ollama';
|
|
this.model = this.ollamaModel;
|
|
this.detail = `Ollama at ${this.ollamaHost}${reachable.hasModel ? '' : ` (warning: model "${this.ollamaModel}" not in the local list)`}`;
|
|
return this.status();
|
|
}
|
|
}
|
|
|
|
this.provider = 'fallback';
|
|
this.model = null;
|
|
this.detail = this.openRouterKey
|
|
? 'no provider reachable'
|
|
: `no OPENROUTER_API_KEY and Ollama not reachable at ${this.ollamaHost}`;
|
|
return this.status();
|
|
}
|
|
|
|
async probeOllama() {
|
|
const ctl = new AbortController();
|
|
const timer = setTimeout(() => ctl.abort(), PROBE_TIMEOUT_MS);
|
|
try {
|
|
const res = await fetch(`${this.ollamaHost}/api/tags`, { signal: ctl.signal });
|
|
if (!res.ok) return null;
|
|
const body = await res.json();
|
|
const names = (body.models || []).map((m) => m.name);
|
|
return { hasModel: names.includes(this.ollamaModel), names };
|
|
} catch {
|
|
return null;
|
|
} finally {
|
|
clearTimeout(timer);
|
|
}
|
|
}
|
|
|
|
status() {
|
|
return {
|
|
provider: this.provider,
|
|
model: this.model,
|
|
detail: this.detail,
|
|
// The UI shows this so you always know on stage what is answering.
|
|
label: this.provider === 'fallback'
|
|
? 'Rule-based (offline)'
|
|
: `${this.provider === 'ollama' ? 'Ollama' : 'OpenRouter'} · ${this.model}`,
|
|
};
|
|
}
|
|
|
|
/**
|
|
* List the models a provider actually offers.
|
|
*
|
|
* Cached briefly: OpenRouter returns 400+ models and the settings page may be
|
|
* opened repeatedly while someone makes up their mind.
|
|
*/
|
|
async listModels(provider, { force = false } = {}) {
|
|
const key = provider;
|
|
const cached = this.modelCache.get(key);
|
|
if (!force && cached && Date.now() - cached.at < 5 * 60 * 1000) {
|
|
return { ...cached.value, cached: true };
|
|
}
|
|
|
|
let value;
|
|
try {
|
|
value = provider === 'openrouter'
|
|
? await this.listOpenRouterModels()
|
|
: await this.listOllamaModels();
|
|
} catch (err) {
|
|
return { provider, models: [], error: err.message, cached: false };
|
|
}
|
|
this.modelCache.set(key, { at: Date.now(), value });
|
|
return { ...value, cached: false };
|
|
}
|
|
|
|
async listOpenRouterModels() {
|
|
if (!this.openRouterKey) {
|
|
throw new Error('No OPENROUTER_API_KEY is set, so the model list cannot be fetched.');
|
|
}
|
|
const ctl = new AbortController();
|
|
const timer = setTimeout(() => ctl.abort(), 15000);
|
|
try {
|
|
const res = await fetch('https://openrouter.ai/api/v1/models', {
|
|
signal: ctl.signal,
|
|
headers: { Authorization: `Bearer ${this.openRouterKey}` },
|
|
});
|
|
if (!res.ok) throw new Error(`OpenRouter returned HTTP ${res.status}`);
|
|
const body = await res.json();
|
|
const models = (body.data || []).map((m) => ({
|
|
id: m.id,
|
|
name: m.name || m.id,
|
|
contextLength: m.context_length || null,
|
|
// Prices come back as per-token strings; per-million is what people read.
|
|
promptPerM: m.pricing && m.pricing.prompt ? Number(m.pricing.prompt) * 1e6 : null,
|
|
completionPerM: m.pricing && m.pricing.completion ? Number(m.pricing.completion) * 1e6 : null,
|
|
})).sort((a, b) => a.id.localeCompare(b.id));
|
|
return { provider: 'openrouter', models };
|
|
} finally {
|
|
clearTimeout(timer);
|
|
}
|
|
}
|
|
|
|
async listOllamaModels() {
|
|
const ctl = new AbortController();
|
|
const timer = setTimeout(() => ctl.abort(), PROBE_TIMEOUT_MS);
|
|
try {
|
|
const res = await fetch(`${this.ollamaHost}/api/tags`, { signal: ctl.signal });
|
|
if (!res.ok) throw new Error(`Ollama returned HTTP ${res.status}`);
|
|
const body = await res.json();
|
|
const models = (body.models || []).map((m) => ({
|
|
id: m.name,
|
|
name: m.name,
|
|
sizeBytes: m.size || null,
|
|
parameterSize: m.details ? m.details.parameter_size : null,
|
|
quantization: m.details ? m.details.quantization_level : null,
|
|
family: m.details ? m.details.family : null,
|
|
})).sort((a, b) => a.id.localeCompare(b.id));
|
|
return { provider: 'ollama', models };
|
|
} catch (err) {
|
|
if (err.name === 'AbortError') {
|
|
throw new Error(`Ollama did not respond at ${this.ollamaHost}. Is "ollama serve" running?`);
|
|
}
|
|
throw new Error(`Cannot reach Ollama at ${this.ollamaHost}: ${err.message}`);
|
|
} finally {
|
|
clearTimeout(timer);
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Measure a provider/model with a trivial prompt.
|
|
*
|
|
* Time-to-first-token is the number that matters here, not total time: a model
|
|
* that takes 30 s to start talking is unusable in front of a customer even if
|
|
* the answer is excellent. This exists so that judgement can be made from a
|
|
* measurement rather than a guess, before the meeting rather than during it.
|
|
*/
|
|
async testProvider({ provider, model } = {}) {
|
|
const resolved = this.resolve(provider);
|
|
const started = Date.now();
|
|
|
|
if (resolved === 'fallback') {
|
|
// Nothing to probe: the rule engine is in-process and cannot be unavailable.
|
|
return {
|
|
ok: true, provider: resolved, model: null,
|
|
firstTokenMs: Date.now() - started, totalMs: Date.now() - started,
|
|
chars: 0,
|
|
sample: 'Rule engine ready. No model, no network, no failure mode.',
|
|
};
|
|
}
|
|
|
|
const messages = [
|
|
{ role: 'system', content: 'You are a terse test endpoint. Reply with exactly: READY' },
|
|
{ role: 'user', content: 'Reply with exactly: READY' },
|
|
];
|
|
|
|
let firstTokenMs = null;
|
|
let text = '';
|
|
try {
|
|
const iter = resolved === 'openrouter'
|
|
? this.streamOpenRouter(messages, model)
|
|
: this.streamOllama(messages, model);
|
|
for await (const chunk of iter) {
|
|
if (!chunk) continue;
|
|
if (firstTokenMs === null) firstTokenMs = Date.now() - started;
|
|
text += chunk;
|
|
}
|
|
} catch (err) {
|
|
return {
|
|
ok: false, provider: resolved,
|
|
model: model || (resolved === 'openrouter' ? this.openRouterModel : this.ollamaModel),
|
|
error: err.message, totalMs: Date.now() - started,
|
|
};
|
|
}
|
|
|
|
const trimmed = text.trim();
|
|
return {
|
|
ok: trimmed.length > 0,
|
|
provider: resolved,
|
|
model: model || (resolved === 'openrouter' ? this.openRouterModel : this.ollamaModel),
|
|
firstTokenMs,
|
|
totalMs: Date.now() - started,
|
|
chars: trimmed.length,
|
|
sample: trimmed.slice(0, 120),
|
|
error: trimmed.length === 0
|
|
? 'The model returned an empty response. It may be spending its whole token budget on reasoning — try a higher max tokens or a lower reasoning effort.'
|
|
: undefined,
|
|
};
|
|
}
|
|
|
|
buildMessages(question, history, frame) {
|
|
const context = buildContext(frame);
|
|
const msgs = [{ role: 'system', content: `${SYSTEM_PROMPT}\n\n---\n\n${context}` }];
|
|
// Keep only the last few turns: a 4B model with a long context degrades fast,
|
|
// and the plant context is refreshed every turn anyway.
|
|
for (const m of (history || []).slice(-6)) {
|
|
if (m && (m.role === 'user' || m.role === 'assistant') && m.content) {
|
|
msgs.push({ role: m.role, content: String(m.content).slice(0, 4000) });
|
|
}
|
|
}
|
|
msgs.push({ role: 'user', content: question });
|
|
return msgs;
|
|
}
|
|
|
|
/**
|
|
* Stream an answer. Yields { type: 'meta'|'token'|'done'|'note' } objects.
|
|
*
|
|
* Any provider failure yields a 'note' explaining the degradation and then
|
|
* streams the fallback answer, so the caller never has to handle an error.
|
|
*/
|
|
/**
|
|
* Which provider to actually use for one request.
|
|
*
|
|
* A per-request override exists for a practical reason: a strong reasoning
|
|
* model can take 15-20 s to first token, which is dead air in front of a
|
|
* customer, while a local 4B model answers in about 2 s with shallower
|
|
* analysis. Being able to pick per question - fast for the live walkthrough,
|
|
* deep for the follow-up discussion - is worth the small amount of plumbing.
|
|
*/
|
|
resolve(override) {
|
|
const want = String(override || '').trim().toLowerCase();
|
|
if (!want || want === 'auto') return this.provider;
|
|
if (want === 'fallback') return 'fallback';
|
|
if (want === 'openrouter' && this.openRouterKey) return 'openrouter';
|
|
if (want === 'ollama') return 'ollama';
|
|
return this.provider;
|
|
}
|
|
|
|
statusFor(provider) {
|
|
if (provider === 'fallback') {
|
|
return { provider, model: null, detail: 'deterministic rule engine', label: 'Rule-based (offline)' };
|
|
}
|
|
if (provider === 'openrouter') {
|
|
return { provider, model: this.openRouterModel, detail: 'OpenRouter', label: `OpenRouter · ${this.openRouterModel}` };
|
|
}
|
|
return { provider, model: this.ollamaModel, detail: `Ollama at ${this.ollamaHost}`, label: `Ollama · ${this.ollamaModel}` };
|
|
}
|
|
|
|
async *stream(question, history, frame, override) {
|
|
const q = String(question || '').trim();
|
|
const provider = this.resolve(override);
|
|
|
|
if (!q) {
|
|
yield { type: 'meta', ...this.statusFor(provider) };
|
|
yield { type: 'token', text: 'Ask me something about the line.' };
|
|
yield { type: 'done' };
|
|
return;
|
|
}
|
|
|
|
if (provider === 'fallback') {
|
|
yield { type: 'meta', ...this.statusFor(provider) };
|
|
const { text } = answerFromRules(q, frame);
|
|
yield { type: 'token', text };
|
|
yield { type: 'done' };
|
|
return;
|
|
}
|
|
|
|
yield { type: 'meta', ...this.statusFor(provider) };
|
|
const messages = this.buildMessages(q, history, frame);
|
|
|
|
let produced = '';
|
|
let failure = null;
|
|
try {
|
|
const iter = provider === 'openrouter'
|
|
? this.streamOpenRouter(messages)
|
|
: this.streamOllama(messages);
|
|
for await (const text of iter) {
|
|
if (!text) continue;
|
|
produced += text;
|
|
yield { type: 'token', text };
|
|
}
|
|
} catch (err) {
|
|
failure = err && err.message ? err.message : String(err);
|
|
console.error('[copilot] provider failed:', failure);
|
|
}
|
|
|
|
// An empty answer is a failure even when nothing threw. A thinking model that
|
|
// exhausts its token budget mid-reasoning returns a clean 200 with no content,
|
|
// and a blank panel is the worst possible outcome in front of a customer.
|
|
if (!produced.trim()) {
|
|
yield {
|
|
type: 'note',
|
|
text: failure
|
|
? `${provider} failed (${failure}). Answering from the built-in rule engine instead.`
|
|
: `${provider} returned an empty answer. Answering from the built-in rule engine instead.`,
|
|
};
|
|
const { text } = answerFromRules(q, frame);
|
|
yield { type: 'token', text };
|
|
} else if (failure) {
|
|
yield { type: 'note', text: `Stream ended early: ${failure}` };
|
|
}
|
|
yield { type: 'done' };
|
|
}
|
|
|
|
// --- providers -----------------------------------------------------------
|
|
|
|
async *streamOpenRouter(messages, modelOverride) {
|
|
const ctl = new AbortController();
|
|
const timer = setTimeout(() => ctl.abort(), REQUEST_TIMEOUT_MS);
|
|
try {
|
|
const body = {
|
|
model: modelOverride || this.openRouterModel,
|
|
messages,
|
|
stream: true,
|
|
temperature: this.temperature,
|
|
// Reasoning tokens count against max_tokens, so a reasoning model on a
|
|
// tight budget burns the lot thinking and streams back nothing at all.
|
|
// Keeping the budget generous and reasoning minimal stops the
|
|
// intermittent empty answers and cuts time-to-first-token, which is the
|
|
// difference between a usable and an awkward live demo.
|
|
max_tokens: this.maxTokens,
|
|
};
|
|
if (this.reasoningEffort !== 'none') {
|
|
body.reasoning = { effort: this.reasoningEffort };
|
|
}
|
|
|
|
const res = await fetch('https://openrouter.ai/api/v1/chat/completions', {
|
|
method: 'POST',
|
|
signal: ctl.signal,
|
|
headers: {
|
|
'Content-Type': 'application/json',
|
|
Authorization: `Bearer ${this.openRouterKey}`,
|
|
'X-Title': 'Digital Twin Demo',
|
|
},
|
|
body: JSON.stringify(body),
|
|
});
|
|
if (!res.ok) {
|
|
throw new Error(`HTTP ${res.status} ${(await res.text()).slice(0, 200)}`);
|
|
}
|
|
const strip = makeThinkFilter();
|
|
for await (const line of readLines(res.body)) {
|
|
if (!line.startsWith('data:')) continue;
|
|
const payload = line.slice(5).trim();
|
|
if (!payload || payload === '[DONE]') continue;
|
|
let json;
|
|
try { json = JSON.parse(payload); } catch { continue; }
|
|
const delta = json.choices && json.choices[0] && json.choices[0].delta;
|
|
if (delta && delta.content) yield strip(delta.content);
|
|
}
|
|
yield strip(null); // flush
|
|
} finally {
|
|
clearTimeout(timer);
|
|
}
|
|
}
|
|
|
|
/**
|
|
* POST to Ollama. `think` of null omits the field entirely.
|
|
*/
|
|
postOllama(messages, think, signal, modelOverride) {
|
|
const body = {
|
|
model: modelOverride || this.ollamaModel,
|
|
messages,
|
|
stream: true,
|
|
options: { temperature: this.temperature, num_predict: this.maxTokens },
|
|
};
|
|
if (think !== null) body.think = think;
|
|
return fetch(`${this.ollamaHost}/api/chat`, {
|
|
method: 'POST',
|
|
signal,
|
|
headers: { 'Content-Type': 'application/json' },
|
|
body: JSON.stringify(body),
|
|
});
|
|
}
|
|
|
|
async *streamOllama(messages, modelOverride) {
|
|
const ctl = new AbortController();
|
|
const timer = setTimeout(() => ctl.abort(), REQUEST_TIMEOUT_MS);
|
|
try {
|
|
// Ollama's `think` is a boolean, so the four-level effort setting maps onto
|
|
// it: none/low disable reasoning, medium/high enable it. The Qwen3 family
|
|
// and other thinking models will otherwise spend the entire token budget
|
|
// inside a reasoning block and return EMPTY content (done_reason:
|
|
// "length") - a blank copilot panel on stage - and cost ~10x the latency.
|
|
const think = this.reasoningEffort === 'medium' || this.reasoningEffort === 'high';
|
|
let res = await this.postOllama(messages, think, ctl.signal, modelOverride);
|
|
|
|
if (!res.ok) {
|
|
const errBody = await res.text();
|
|
// Models with no reasoning mode reject the flag; retry without it.
|
|
if (/think/i.test(errBody)) {
|
|
res = await this.postOllama(messages, null, ctl.signal, modelOverride);
|
|
if (!res.ok) throw new Error(`HTTP ${res.status} ${(await res.text()).slice(0, 200)}`);
|
|
} else {
|
|
throw new Error(`HTTP ${res.status} ${errBody.slice(0, 200)}`);
|
|
}
|
|
}
|
|
|
|
const strip = makeThinkFilter();
|
|
for await (const line of readLines(res.body)) {
|
|
if (!line.trim()) continue;
|
|
let json;
|
|
try { json = JSON.parse(line); } catch { continue; }
|
|
if (json.error) throw new Error(json.error);
|
|
if (json.message && json.message.content) yield strip(json.message.content);
|
|
if (json.done) break;
|
|
}
|
|
yield strip(null); // flush
|
|
} finally {
|
|
clearTimeout(timer);
|
|
}
|
|
}
|
|
}
|
|
|
|
/** Async line reader over a fetch response body stream. */
|
|
async function* readLines(body) {
|
|
const decoder = new TextDecoder();
|
|
let buf = '';
|
|
for await (const chunk of body) {
|
|
buf += decoder.decode(chunk, { stream: true });
|
|
let idx;
|
|
while ((idx = buf.indexOf('\n')) >= 0) {
|
|
yield buf.slice(0, idx);
|
|
buf = buf.slice(idx + 1);
|
|
}
|
|
}
|
|
if (buf) yield buf;
|
|
}
|
|
|
|
/**
|
|
* Suppress <think>...</think> reasoning blocks.
|
|
*
|
|
* Several strong local models (the Qwen3 family among them) emit a reasoning
|
|
* block before the answer. Streamed verbatim onto a dashboard it looks like the
|
|
* product is malfunctioning, so it is filtered out. Call with null to flush.
|
|
*/
|
|
function makeThinkFilter() {
|
|
const OPEN = '<think>';
|
|
const CLOSE = '</think>';
|
|
let inThink = false;
|
|
let buf = '';
|
|
|
|
return (chunk) => {
|
|
if (chunk === null) {
|
|
const tail = inThink ? '' : buf;
|
|
buf = '';
|
|
return tail;
|
|
}
|
|
buf += chunk;
|
|
let out = '';
|
|
|
|
for (;;) {
|
|
if (!inThink) {
|
|
const i = buf.indexOf(OPEN);
|
|
if (i === -1) {
|
|
// Hold back a few characters in case a tag straddles two chunks.
|
|
const keep = Math.max(0, buf.length - (OPEN.length - 1));
|
|
out += buf.slice(0, keep);
|
|
buf = buf.slice(keep);
|
|
break;
|
|
}
|
|
out += buf.slice(0, i);
|
|
buf = buf.slice(i + OPEN.length);
|
|
inThink = true;
|
|
} else {
|
|
const j = buf.indexOf(CLOSE);
|
|
if (j === -1) {
|
|
buf = buf.slice(Math.max(0, buf.length - (CLOSE.length - 1)));
|
|
break;
|
|
}
|
|
buf = buf.slice(j + CLOSE.length);
|
|
inThink = false;
|
|
}
|
|
}
|
|
return out;
|
|
};
|
|
}
|