/**
* Copilot provider layer.
*
* Resolution order: OpenRouter if a key is set, else Ollama if it answers, else
* the deterministic rule-based fallback. The fallback is not an error path - it
* is a supported mode, and COPILOT_PROVIDER=fallback selects it deliberately.
*
* The hard requirement is that a question ALWAYS gets an answer. A missing key,
* an unreachable host, a model that 404s, a stream that dies halfway: every one
* of those degrades to the fallback rather than surfacing an error on stage.
*/
import { buildContext, SYSTEM_PROMPT } from './context.js';
import { answerFromRules } from './rules.js';
import { loadSettings, saveSettings, validatePatch, resetSettings } from './settings.js';
const PROBE_TIMEOUT_MS = 2000;
const REQUEST_TIMEOUT_MS = 90000;
/**
* Turn whatever OLLAMA_HOST happens to contain into a dialable origin.
*
* Ollama itself commonly sets OLLAMA_HOST=0.0.0.0 machine-wide to bind all
* interfaces. That is a BIND address, not a destination - you cannot connect to
* it - and because it is a real environment variable it silently overrides any
* default we set here. Also accepts a bare host, a host:port, or a full URL.
*/
export function normalizeOllamaHost(raw) {
let h = String(raw || '').trim();
if (!h) return 'http://127.0.0.1:11434';
if (!/^https?:\/\//i.test(h)) h = `http://${h}`;
try {
const u = new URL(h);
if (u.hostname === '0.0.0.0' || u.hostname === '::' || u.hostname === '[::]') {
u.hostname = '127.0.0.1';
}
if (!u.port) u.port = '11434';
return u.origin;
} catch {
return 'http://127.0.0.1:11434';
}
}
export class Copilot {
constructor(env = process.env) {
this.env = env;
// The key stays in the environment only. It is never part of the settings
// that the settings page can read or write.
this.openRouterKey = (env.OPENROUTER_API_KEY || '').trim();
this.applySettings(loadSettings(env));
this.provider = 'fallback';
this.model = null;
this.detail = 'not yet detected';
this.modelCache = new Map();
}
/** Adopt a settings object. Does not persist; see configure(). */
applySettings(settings) {
this.settings = { ...settings };
this.forced = settings.provider === 'auto' ? null : settings.provider;
this.openRouterModel = settings.openRouterModel;
this.ollamaModel = settings.ollamaModel;
this.ollamaHost = normalizeOllamaHost(settings.ollamaHost);
this.temperature = settings.temperature;
this.maxTokens = settings.maxTokens;
this.reasoningEffort = settings.reasoningEffort;
}
/**
* Validate, apply, persist and re-detect.
*
* Returns { ok, errors, settings, status }. A rejected patch changes nothing.
*/
async configure(patch) {
const { ok, errors, clean } = validatePatch(patch);
if (!ok) return { ok: false, errors, settings: this.publicSettings(), status: this.status() };
this.applySettings({ ...this.settings, ...clean });
saveSettings(this.settings);
// The chosen host may have changed, so availability has to be re-probed.
const status = await this.detect();
return { ok: true, errors: [], settings: this.publicSettings(), status };
}
/** Restore environment defaults, discarding the saved settings file. */
async reset() {
resetSettings();
this.applySettings(loadSettings(this.env));
const status = await this.detect();
return { ok: true, errors: [], settings: this.publicSettings(), status };
}
/** Settings safe to hand to a browser: never includes the API key. */
publicSettings() {
return {
...this.settings,
// Report the normalized host, since that is what actually gets dialled.
resolvedOllamaHost: this.ollamaHost,
openRouterKeyPresent: Boolean(this.openRouterKey),
};
}
/** Probe available providers. Safe to call repeatedly. */
async detect() {
if (this.forced === 'fallback') {
this.provider = 'fallback';
this.model = null;
this.detail = 'forced by COPILOT_PROVIDER=fallback';
return this.status();
}
if (this.openRouterKey && this.forced !== 'ollama') {
this.provider = 'openrouter';
this.model = this.openRouterModel;
this.detail = 'OpenRouter API key present';
return this.status();
}
if (this.forced !== 'openrouter') {
const reachable = await this.probeOllama();
if (reachable) {
this.provider = 'ollama';
this.model = this.ollamaModel;
this.detail = `Ollama at ${this.ollamaHost}${reachable.hasModel ? '' : ` (warning: model "${this.ollamaModel}" not in the local list)`}`;
return this.status();
}
}
this.provider = 'fallback';
this.model = null;
this.detail = this.openRouterKey
? 'no provider reachable'
: `no OPENROUTER_API_KEY and Ollama not reachable at ${this.ollamaHost}`;
return this.status();
}
async probeOllama() {
const ctl = new AbortController();
const timer = setTimeout(() => ctl.abort(), PROBE_TIMEOUT_MS);
try {
const res = await fetch(`${this.ollamaHost}/api/tags`, { signal: ctl.signal });
if (!res.ok) return null;
const body = await res.json();
const names = (body.models || []).map((m) => m.name);
return { hasModel: names.includes(this.ollamaModel), names };
} catch {
return null;
} finally {
clearTimeout(timer);
}
}
status() {
return {
provider: this.provider,
model: this.model,
detail: this.detail,
// The UI shows this so you always know on stage what is answering.
label: this.provider === 'fallback'
? 'Rule-based (offline)'
: `${this.provider === 'ollama' ? 'Ollama' : 'OpenRouter'} · ${this.model}`,
};
}
/**
* List the models a provider actually offers.
*
* Cached briefly: OpenRouter returns 400+ models and the settings page may be
* opened repeatedly while someone makes up their mind.
*/
async listModels(provider, { force = false } = {}) {
const key = provider;
const cached = this.modelCache.get(key);
if (!force && cached && Date.now() - cached.at < 5 * 60 * 1000) {
return { ...cached.value, cached: true };
}
let value;
try {
value = provider === 'openrouter'
? await this.listOpenRouterModels()
: await this.listOllamaModels();
} catch (err) {
return { provider, models: [], error: err.message, cached: false };
}
this.modelCache.set(key, { at: Date.now(), value });
return { ...value, cached: false };
}
async listOpenRouterModels() {
if (!this.openRouterKey) {
throw new Error('No OPENROUTER_API_KEY is set, so the model list cannot be fetched.');
}
const ctl = new AbortController();
const timer = setTimeout(() => ctl.abort(), 15000);
try {
const res = await fetch('https://openrouter.ai/api/v1/models', {
signal: ctl.signal,
headers: { Authorization: `Bearer ${this.openRouterKey}` },
});
if (!res.ok) throw new Error(`OpenRouter returned HTTP ${res.status}`);
const body = await res.json();
const models = (body.data || []).map((m) => ({
id: m.id,
name: m.name || m.id,
contextLength: m.context_length || null,
// Prices come back as per-token strings; per-million is what people read.
promptPerM: m.pricing && m.pricing.prompt ? Number(m.pricing.prompt) * 1e6 : null,
completionPerM: m.pricing && m.pricing.completion ? Number(m.pricing.completion) * 1e6 : null,
})).sort((a, b) => a.id.localeCompare(b.id));
return { provider: 'openrouter', models };
} finally {
clearTimeout(timer);
}
}
async listOllamaModels() {
const ctl = new AbortController();
const timer = setTimeout(() => ctl.abort(), PROBE_TIMEOUT_MS);
try {
const res = await fetch(`${this.ollamaHost}/api/tags`, { signal: ctl.signal });
if (!res.ok) throw new Error(`Ollama returned HTTP ${res.status}`);
const body = await res.json();
const models = (body.models || []).map((m) => ({
id: m.name,
name: m.name,
sizeBytes: m.size || null,
parameterSize: m.details ? m.details.parameter_size : null,
quantization: m.details ? m.details.quantization_level : null,
family: m.details ? m.details.family : null,
})).sort((a, b) => a.id.localeCompare(b.id));
return { provider: 'ollama', models };
} catch (err) {
if (err.name === 'AbortError') {
throw new Error(`Ollama did not respond at ${this.ollamaHost}. Is "ollama serve" running?`);
}
throw new Error(`Cannot reach Ollama at ${this.ollamaHost}: ${err.message}`);
} finally {
clearTimeout(timer);
}
}
/**
* Measure a provider/model with a trivial prompt.
*
* Time-to-first-token is the number that matters here, not total time: a model
* that takes 30 s to start talking is unusable in front of a customer even if
* the answer is excellent. This exists so that judgement can be made from a
* measurement rather than a guess, before the meeting rather than during it.
*/
async testProvider({ provider, model } = {}) {
const resolved = this.resolve(provider);
const started = Date.now();
if (resolved === 'fallback') {
// Nothing to probe: the rule engine is in-process and cannot be unavailable.
return {
ok: true, provider: resolved, model: null,
firstTokenMs: Date.now() - started, totalMs: Date.now() - started,
chars: 0,
sample: 'Rule engine ready. No model, no network, no failure mode.',
};
}
const messages = [
{ role: 'system', content: 'You are a terse test endpoint. Reply with exactly: READY' },
{ role: 'user', content: 'Reply with exactly: READY' },
];
let firstTokenMs = null;
let text = '';
try {
const iter = resolved === 'openrouter'
? this.streamOpenRouter(messages, model)
: this.streamOllama(messages, model);
for await (const chunk of iter) {
if (!chunk) continue;
if (firstTokenMs === null) firstTokenMs = Date.now() - started;
text += chunk;
}
} catch (err) {
return {
ok: false, provider: resolved,
model: model || (resolved === 'openrouter' ? this.openRouterModel : this.ollamaModel),
error: err.message, totalMs: Date.now() - started,
};
}
const trimmed = text.trim();
return {
ok: trimmed.length > 0,
provider: resolved,
model: model || (resolved === 'openrouter' ? this.openRouterModel : this.ollamaModel),
firstTokenMs,
totalMs: Date.now() - started,
chars: trimmed.length,
sample: trimmed.slice(0, 120),
error: trimmed.length === 0
? 'The model returned an empty response. It may be spending its whole token budget on reasoning — try a higher max tokens or a lower reasoning effort.'
: undefined,
};
}
buildMessages(question, history, frame) {
const context = buildContext(frame);
const msgs = [{ role: 'system', content: `${SYSTEM_PROMPT}\n\n---\n\n${context}` }];
// Keep only the last few turns: a 4B model with a long context degrades fast,
// and the plant context is refreshed every turn anyway.
for (const m of (history || []).slice(-6)) {
if (m && (m.role === 'user' || m.role === 'assistant') && m.content) {
msgs.push({ role: m.role, content: String(m.content).slice(0, 4000) });
}
}
msgs.push({ role: 'user', content: question });
return msgs;
}
/**
* Stream an answer. Yields { type: 'meta'|'token'|'done'|'note' } objects.
*
* Any provider failure yields a 'note' explaining the degradation and then
* streams the fallback answer, so the caller never has to handle an error.
*/
/**
* Which provider to actually use for one request.
*
* A per-request override exists for a practical reason: a strong reasoning
* model can take 15-20 s to first token, which is dead air in front of a
* customer, while a local 4B model answers in about 2 s with shallower
* analysis. Being able to pick per question - fast for the live walkthrough,
* deep for the follow-up discussion - is worth the small amount of plumbing.
*/
resolve(override) {
const want = String(override || '').trim().toLowerCase();
if (!want || want === 'auto') return this.provider;
if (want === 'fallback') return 'fallback';
if (want === 'openrouter' && this.openRouterKey) return 'openrouter';
if (want === 'ollama') return 'ollama';
return this.provider;
}
statusFor(provider) {
if (provider === 'fallback') {
return { provider, model: null, detail: 'deterministic rule engine', label: 'Rule-based (offline)' };
}
if (provider === 'openrouter') {
return { provider, model: this.openRouterModel, detail: 'OpenRouter', label: `OpenRouter · ${this.openRouterModel}` };
}
return { provider, model: this.ollamaModel, detail: `Ollama at ${this.ollamaHost}`, label: `Ollama · ${this.ollamaModel}` };
}
async *stream(question, history, frame, override) {
const q = String(question || '').trim();
const provider = this.resolve(override);
if (!q) {
yield { type: 'meta', ...this.statusFor(provider) };
yield { type: 'token', text: 'Ask me something about the line.' };
yield { type: 'done' };
return;
}
if (provider === 'fallback') {
yield { type: 'meta', ...this.statusFor(provider) };
const { text } = answerFromRules(q, frame);
yield { type: 'token', text };
yield { type: 'done' };
return;
}
yield { type: 'meta', ...this.statusFor(provider) };
const messages = this.buildMessages(q, history, frame);
let produced = '';
let failure = null;
try {
const iter = provider === 'openrouter'
? this.streamOpenRouter(messages)
: this.streamOllama(messages);
for await (const text of iter) {
if (!text) continue;
produced += text;
yield { type: 'token', text };
}
} catch (err) {
failure = err && err.message ? err.message : String(err);
console.error('[copilot] provider failed:', failure);
}
// An empty answer is a failure even when nothing threw. A thinking model that
// exhausts its token budget mid-reasoning returns a clean 200 with no content,
// and a blank panel is the worst possible outcome in front of a customer.
if (!produced.trim()) {
yield {
type: 'note',
text: failure
? `${provider} failed (${failure}). Answering from the built-in rule engine instead.`
: `${provider} returned an empty answer. Answering from the built-in rule engine instead.`,
};
const { text } = answerFromRules(q, frame);
yield { type: 'token', text };
} else if (failure) {
yield { type: 'note', text: `Stream ended early: ${failure}` };
}
yield { type: 'done' };
}
// --- providers -----------------------------------------------------------
async *streamOpenRouter(messages, modelOverride) {
const ctl = new AbortController();
const timer = setTimeout(() => ctl.abort(), REQUEST_TIMEOUT_MS);
try {
const body = {
model: modelOverride || this.openRouterModel,
messages,
stream: true,
temperature: this.temperature,
// Reasoning tokens count against max_tokens, so a reasoning model on a
// tight budget burns the lot thinking and streams back nothing at all.
// Keeping the budget generous and reasoning minimal stops the
// intermittent empty answers and cuts time-to-first-token, which is the
// difference between a usable and an awkward live demo.
max_tokens: this.maxTokens,
};
if (this.reasoningEffort !== 'none') {
body.reasoning = { effort: this.reasoningEffort };
}
const res = await fetch('https://openrouter.ai/api/v1/chat/completions', {
method: 'POST',
signal: ctl.signal,
headers: {
'Content-Type': 'application/json',
Authorization: `Bearer ${this.openRouterKey}`,
'X-Title': 'Digital Twin Demo',
},
body: JSON.stringify(body),
});
if (!res.ok) {
throw new Error(`HTTP ${res.status} ${(await res.text()).slice(0, 200)}`);
}
const strip = makeThinkFilter();
for await (const line of readLines(res.body)) {
if (!line.startsWith('data:')) continue;
const payload = line.slice(5).trim();
if (!payload || payload === '[DONE]') continue;
let json;
try { json = JSON.parse(payload); } catch { continue; }
const delta = json.choices && json.choices[0] && json.choices[0].delta;
if (delta && delta.content) yield strip(delta.content);
}
yield strip(null); // flush
} finally {
clearTimeout(timer);
}
}
/**
* POST to Ollama. `think` of null omits the field entirely.
*/
postOllama(messages, think, signal, modelOverride) {
const body = {
model: modelOverride || this.ollamaModel,
messages,
stream: true,
options: { temperature: this.temperature, num_predict: this.maxTokens },
};
if (think !== null) body.think = think;
return fetch(`${this.ollamaHost}/api/chat`, {
method: 'POST',
signal,
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify(body),
});
}
async *streamOllama(messages, modelOverride) {
const ctl = new AbortController();
const timer = setTimeout(() => ctl.abort(), REQUEST_TIMEOUT_MS);
try {
// Ollama's `think` is a boolean, so the four-level effort setting maps onto
// it: none/low disable reasoning, medium/high enable it. The Qwen3 family
// and other thinking models will otherwise spend the entire token budget
// inside a reasoning block and return EMPTY content (done_reason:
// "length") - a blank copilot panel on stage - and cost ~10x the latency.
const think = this.reasoningEffort === 'medium' || this.reasoningEffort === 'high';
let res = await this.postOllama(messages, think, ctl.signal, modelOverride);
if (!res.ok) {
const errBody = await res.text();
// Models with no reasoning mode reject the flag; retry without it.
if (/think/i.test(errBody)) {
res = await this.postOllama(messages, null, ctl.signal, modelOverride);
if (!res.ok) throw new Error(`HTTP ${res.status} ${(await res.text()).slice(0, 200)}`);
} else {
throw new Error(`HTTP ${res.status} ${errBody.slice(0, 200)}`);
}
}
const strip = makeThinkFilter();
for await (const line of readLines(res.body)) {
if (!line.trim()) continue;
let json;
try { json = JSON.parse(line); } catch { continue; }
if (json.error) throw new Error(json.error);
if (json.message && json.message.content) yield strip(json.message.content);
if (json.done) break;
}
yield strip(null); // flush
} finally {
clearTimeout(timer);
}
}
}
/** Async line reader over a fetch response body stream. */
async function* readLines(body) {
const decoder = new TextDecoder();
let buf = '';
for await (const chunk of body) {
buf += decoder.decode(chunk, { stream: true });
let idx;
while ((idx = buf.indexOf('\n')) >= 0) {
yield buf.slice(0, idx);
buf = buf.slice(idx + 1);
}
}
if (buf) yield buf;
}
/**
* Suppress ... reasoning blocks.
*
* Several strong local models (the Qwen3 family among them) emit a reasoning
* block before the answer. Streamed verbatim onto a dashboard it looks like the
* product is malfunctioning, so it is filtered out. Call with null to flush.
*/
function makeThinkFilter() {
const OPEN = '';
const CLOSE = '';
let inThink = false;
let buf = '';
return (chunk) => {
if (chunk === null) {
const tail = inThink ? '' : buf;
buf = '';
return tail;
}
buf += chunk;
let out = '';
for (;;) {
if (!inThink) {
const i = buf.indexOf(OPEN);
if (i === -1) {
// Hold back a few characters in case a tag straddles two chunks.
const keep = Math.max(0, buf.length - (OPEN.length - 1));
out += buf.slice(0, keep);
buf = buf.slice(keep);
break;
}
out += buf.slice(0, i);
buf = buf.slice(i + OPEN.length);
inThink = true;
} else {
const j = buf.indexOf(CLOSE);
if (j === -1) {
buf = buf.slice(Math.max(0, buf.length - (CLOSE.length - 1)));
break;
}
buf = buf.slice(j + CLOSE.length);
inThink = false;
}
}
return out;
};
}