/** * Copilot provider layer. * * Resolution order: OpenRouter if a key is set, else Ollama if it answers, else * the deterministic rule-based fallback. The fallback is not an error path - it * is a supported mode, and COPILOT_PROVIDER=fallback selects it deliberately. * * The hard requirement is that a question ALWAYS gets an answer. A missing key, * an unreachable host, a model that 404s, a stream that dies halfway: every one * of those degrades to the fallback rather than surfacing an error on stage. */ import { buildContext, SYSTEM_PROMPT } from './context.js'; import { answerFromRules } from './rules.js'; import { loadSettings, saveSettings, validatePatch, resetSettings } from './settings.js'; const PROBE_TIMEOUT_MS = 2000; const REQUEST_TIMEOUT_MS = 90000; /** * Turn whatever OLLAMA_HOST happens to contain into a dialable origin. * * Ollama itself commonly sets OLLAMA_HOST=0.0.0.0 machine-wide to bind all * interfaces. That is a BIND address, not a destination - you cannot connect to * it - and because it is a real environment variable it silently overrides any * default we set here. Also accepts a bare host, a host:port, or a full URL. */ export function normalizeOllamaHost(raw) { let h = String(raw || '').trim(); if (!h) return 'http://127.0.0.1:11434'; if (!/^https?:\/\//i.test(h)) h = `http://${h}`; try { const u = new URL(h); if (u.hostname === '0.0.0.0' || u.hostname === '::' || u.hostname === '[::]') { u.hostname = '127.0.0.1'; } if (!u.port) u.port = '11434'; return u.origin; } catch { return 'http://127.0.0.1:11434'; } } export class Copilot { constructor(env = process.env) { this.env = env; // The key stays in the environment only. It is never part of the settings // that the settings page can read or write. this.openRouterKey = (env.OPENROUTER_API_KEY || '').trim(); this.applySettings(loadSettings(env)); this.provider = 'fallback'; this.model = null; this.detail = 'not yet detected'; this.modelCache = new Map(); } /** Adopt a settings object. Does not persist; see configure(). */ applySettings(settings) { this.settings = { ...settings }; this.forced = settings.provider === 'auto' ? null : settings.provider; this.openRouterModel = settings.openRouterModel; this.ollamaModel = settings.ollamaModel; this.ollamaHost = normalizeOllamaHost(settings.ollamaHost); this.temperature = settings.temperature; this.maxTokens = settings.maxTokens; this.reasoningEffort = settings.reasoningEffort; } /** * Validate, apply, persist and re-detect. * * Returns { ok, errors, settings, status }. A rejected patch changes nothing. */ async configure(patch) { const { ok, errors, clean } = validatePatch(patch); if (!ok) return { ok: false, errors, settings: this.publicSettings(), status: this.status() }; this.applySettings({ ...this.settings, ...clean }); saveSettings(this.settings); // The chosen host may have changed, so availability has to be re-probed. const status = await this.detect(); return { ok: true, errors: [], settings: this.publicSettings(), status }; } /** Restore environment defaults, discarding the saved settings file. */ async reset() { resetSettings(); this.applySettings(loadSettings(this.env)); const status = await this.detect(); return { ok: true, errors: [], settings: this.publicSettings(), status }; } /** Settings safe to hand to a browser: never includes the API key. */ publicSettings() { return { ...this.settings, // Report the normalized host, since that is what actually gets dialled. resolvedOllamaHost: this.ollamaHost, openRouterKeyPresent: Boolean(this.openRouterKey), }; } /** Probe available providers. Safe to call repeatedly. */ async detect() { if (this.forced === 'fallback') { this.provider = 'fallback'; this.model = null; this.detail = 'forced by COPILOT_PROVIDER=fallback'; return this.status(); } if (this.openRouterKey && this.forced !== 'ollama') { this.provider = 'openrouter'; this.model = this.openRouterModel; this.detail = 'OpenRouter API key present'; return this.status(); } if (this.forced !== 'openrouter') { const reachable = await this.probeOllama(); if (reachable) { this.provider = 'ollama'; this.model = this.ollamaModel; this.detail = `Ollama at ${this.ollamaHost}${reachable.hasModel ? '' : ` (warning: model "${this.ollamaModel}" not in the local list)`}`; return this.status(); } } this.provider = 'fallback'; this.model = null; this.detail = this.openRouterKey ? 'no provider reachable' : `no OPENROUTER_API_KEY and Ollama not reachable at ${this.ollamaHost}`; return this.status(); } async probeOllama() { const ctl = new AbortController(); const timer = setTimeout(() => ctl.abort(), PROBE_TIMEOUT_MS); try { const res = await fetch(`${this.ollamaHost}/api/tags`, { signal: ctl.signal }); if (!res.ok) return null; const body = await res.json(); const names = (body.models || []).map((m) => m.name); return { hasModel: names.includes(this.ollamaModel), names }; } catch { return null; } finally { clearTimeout(timer); } } status() { return { provider: this.provider, model: this.model, detail: this.detail, // The UI shows this so you always know on stage what is answering. label: this.provider === 'fallback' ? 'Rule-based (offline)' : `${this.provider === 'ollama' ? 'Ollama' : 'OpenRouter'} · ${this.model}`, }; } /** * List the models a provider actually offers. * * Cached briefly: OpenRouter returns 400+ models and the settings page may be * opened repeatedly while someone makes up their mind. */ async listModels(provider, { force = false } = {}) { const key = provider; const cached = this.modelCache.get(key); if (!force && cached && Date.now() - cached.at < 5 * 60 * 1000) { return { ...cached.value, cached: true }; } let value; try { value = provider === 'openrouter' ? await this.listOpenRouterModels() : await this.listOllamaModels(); } catch (err) { return { provider, models: [], error: err.message, cached: false }; } this.modelCache.set(key, { at: Date.now(), value }); return { ...value, cached: false }; } async listOpenRouterModels() { if (!this.openRouterKey) { throw new Error('No OPENROUTER_API_KEY is set, so the model list cannot be fetched.'); } const ctl = new AbortController(); const timer = setTimeout(() => ctl.abort(), 15000); try { const res = await fetch('https://openrouter.ai/api/v1/models', { signal: ctl.signal, headers: { Authorization: `Bearer ${this.openRouterKey}` }, }); if (!res.ok) throw new Error(`OpenRouter returned HTTP ${res.status}`); const body = await res.json(); const models = (body.data || []).map((m) => ({ id: m.id, name: m.name || m.id, contextLength: m.context_length || null, // Prices come back as per-token strings; per-million is what people read. promptPerM: m.pricing && m.pricing.prompt ? Number(m.pricing.prompt) * 1e6 : null, completionPerM: m.pricing && m.pricing.completion ? Number(m.pricing.completion) * 1e6 : null, })).sort((a, b) => a.id.localeCompare(b.id)); return { provider: 'openrouter', models }; } finally { clearTimeout(timer); } } async listOllamaModels() { const ctl = new AbortController(); const timer = setTimeout(() => ctl.abort(), PROBE_TIMEOUT_MS); try { const res = await fetch(`${this.ollamaHost}/api/tags`, { signal: ctl.signal }); if (!res.ok) throw new Error(`Ollama returned HTTP ${res.status}`); const body = await res.json(); const models = (body.models || []).map((m) => ({ id: m.name, name: m.name, sizeBytes: m.size || null, parameterSize: m.details ? m.details.parameter_size : null, quantization: m.details ? m.details.quantization_level : null, family: m.details ? m.details.family : null, })).sort((a, b) => a.id.localeCompare(b.id)); return { provider: 'ollama', models }; } catch (err) { if (err.name === 'AbortError') { throw new Error(`Ollama did not respond at ${this.ollamaHost}. Is "ollama serve" running?`); } throw new Error(`Cannot reach Ollama at ${this.ollamaHost}: ${err.message}`); } finally { clearTimeout(timer); } } /** * Measure a provider/model with a trivial prompt. * * Time-to-first-token is the number that matters here, not total time: a model * that takes 30 s to start talking is unusable in front of a customer even if * the answer is excellent. This exists so that judgement can be made from a * measurement rather than a guess, before the meeting rather than during it. */ async testProvider({ provider, model } = {}) { const resolved = this.resolve(provider); const started = Date.now(); if (resolved === 'fallback') { // Nothing to probe: the rule engine is in-process and cannot be unavailable. return { ok: true, provider: resolved, model: null, firstTokenMs: Date.now() - started, totalMs: Date.now() - started, chars: 0, sample: 'Rule engine ready. No model, no network, no failure mode.', }; } const messages = [ { role: 'system', content: 'You are a terse test endpoint. Reply with exactly: READY' }, { role: 'user', content: 'Reply with exactly: READY' }, ]; let firstTokenMs = null; let text = ''; try { const iter = resolved === 'openrouter' ? this.streamOpenRouter(messages, model) : this.streamOllama(messages, model); for await (const chunk of iter) { if (!chunk) continue; if (firstTokenMs === null) firstTokenMs = Date.now() - started; text += chunk; } } catch (err) { return { ok: false, provider: resolved, model: model || (resolved === 'openrouter' ? this.openRouterModel : this.ollamaModel), error: err.message, totalMs: Date.now() - started, }; } const trimmed = text.trim(); return { ok: trimmed.length > 0, provider: resolved, model: model || (resolved === 'openrouter' ? this.openRouterModel : this.ollamaModel), firstTokenMs, totalMs: Date.now() - started, chars: trimmed.length, sample: trimmed.slice(0, 120), error: trimmed.length === 0 ? 'The model returned an empty response. It may be spending its whole token budget on reasoning — try a higher max tokens or a lower reasoning effort.' : undefined, }; } buildMessages(question, history, frame) { const context = buildContext(frame); const msgs = [{ role: 'system', content: `${SYSTEM_PROMPT}\n\n---\n\n${context}` }]; // Keep only the last few turns: a 4B model with a long context degrades fast, // and the plant context is refreshed every turn anyway. for (const m of (history || []).slice(-6)) { if (m && (m.role === 'user' || m.role === 'assistant') && m.content) { msgs.push({ role: m.role, content: String(m.content).slice(0, 4000) }); } } msgs.push({ role: 'user', content: question }); return msgs; } /** * Stream an answer. Yields { type: 'meta'|'token'|'done'|'note' } objects. * * Any provider failure yields a 'note' explaining the degradation and then * streams the fallback answer, so the caller never has to handle an error. */ /** * Which provider to actually use for one request. * * A per-request override exists for a practical reason: a strong reasoning * model can take 15-20 s to first token, which is dead air in front of a * customer, while a local 4B model answers in about 2 s with shallower * analysis. Being able to pick per question - fast for the live walkthrough, * deep for the follow-up discussion - is worth the small amount of plumbing. */ resolve(override) { const want = String(override || '').trim().toLowerCase(); if (!want || want === 'auto') return this.provider; if (want === 'fallback') return 'fallback'; if (want === 'openrouter' && this.openRouterKey) return 'openrouter'; if (want === 'ollama') return 'ollama'; return this.provider; } statusFor(provider) { if (provider === 'fallback') { return { provider, model: null, detail: 'deterministic rule engine', label: 'Rule-based (offline)' }; } if (provider === 'openrouter') { return { provider, model: this.openRouterModel, detail: 'OpenRouter', label: `OpenRouter · ${this.openRouterModel}` }; } return { provider, model: this.ollamaModel, detail: `Ollama at ${this.ollamaHost}`, label: `Ollama · ${this.ollamaModel}` }; } async *stream(question, history, frame, override) { const q = String(question || '').trim(); const provider = this.resolve(override); if (!q) { yield { type: 'meta', ...this.statusFor(provider) }; yield { type: 'token', text: 'Ask me something about the line.' }; yield { type: 'done' }; return; } if (provider === 'fallback') { yield { type: 'meta', ...this.statusFor(provider) }; const { text } = answerFromRules(q, frame); yield { type: 'token', text }; yield { type: 'done' }; return; } yield { type: 'meta', ...this.statusFor(provider) }; const messages = this.buildMessages(q, history, frame); let produced = ''; let failure = null; try { const iter = provider === 'openrouter' ? this.streamOpenRouter(messages) : this.streamOllama(messages); for await (const text of iter) { if (!text) continue; produced += text; yield { type: 'token', text }; } } catch (err) { failure = err && err.message ? err.message : String(err); console.error('[copilot] provider failed:', failure); } // An empty answer is a failure even when nothing threw. A thinking model that // exhausts its token budget mid-reasoning returns a clean 200 with no content, // and a blank panel is the worst possible outcome in front of a customer. if (!produced.trim()) { yield { type: 'note', text: failure ? `${provider} failed (${failure}). Answering from the built-in rule engine instead.` : `${provider} returned an empty answer. Answering from the built-in rule engine instead.`, }; const { text } = answerFromRules(q, frame); yield { type: 'token', text }; } else if (failure) { yield { type: 'note', text: `Stream ended early: ${failure}` }; } yield { type: 'done' }; } // --- providers ----------------------------------------------------------- async *streamOpenRouter(messages, modelOverride) { const ctl = new AbortController(); const timer = setTimeout(() => ctl.abort(), REQUEST_TIMEOUT_MS); try { const body = { model: modelOverride || this.openRouterModel, messages, stream: true, temperature: this.temperature, // Reasoning tokens count against max_tokens, so a reasoning model on a // tight budget burns the lot thinking and streams back nothing at all. // Keeping the budget generous and reasoning minimal stops the // intermittent empty answers and cuts time-to-first-token, which is the // difference between a usable and an awkward live demo. max_tokens: this.maxTokens, }; if (this.reasoningEffort !== 'none') { body.reasoning = { effort: this.reasoningEffort }; } const res = await fetch('https://openrouter.ai/api/v1/chat/completions', { method: 'POST', signal: ctl.signal, headers: { 'Content-Type': 'application/json', Authorization: `Bearer ${this.openRouterKey}`, 'X-Title': 'Digital Twin Demo', }, body: JSON.stringify(body), }); if (!res.ok) { throw new Error(`HTTP ${res.status} ${(await res.text()).slice(0, 200)}`); } const strip = makeThinkFilter(); for await (const line of readLines(res.body)) { if (!line.startsWith('data:')) continue; const payload = line.slice(5).trim(); if (!payload || payload === '[DONE]') continue; let json; try { json = JSON.parse(payload); } catch { continue; } const delta = json.choices && json.choices[0] && json.choices[0].delta; if (delta && delta.content) yield strip(delta.content); } yield strip(null); // flush } finally { clearTimeout(timer); } } /** * POST to Ollama. `think` of null omits the field entirely. */ postOllama(messages, think, signal, modelOverride) { const body = { model: modelOverride || this.ollamaModel, messages, stream: true, options: { temperature: this.temperature, num_predict: this.maxTokens }, }; if (think !== null) body.think = think; return fetch(`${this.ollamaHost}/api/chat`, { method: 'POST', signal, headers: { 'Content-Type': 'application/json' }, body: JSON.stringify(body), }); } async *streamOllama(messages, modelOverride) { const ctl = new AbortController(); const timer = setTimeout(() => ctl.abort(), REQUEST_TIMEOUT_MS); try { // Ollama's `think` is a boolean, so the four-level effort setting maps onto // it: none/low disable reasoning, medium/high enable it. The Qwen3 family // and other thinking models will otherwise spend the entire token budget // inside a reasoning block and return EMPTY content (done_reason: // "length") - a blank copilot panel on stage - and cost ~10x the latency. const think = this.reasoningEffort === 'medium' || this.reasoningEffort === 'high'; let res = await this.postOllama(messages, think, ctl.signal, modelOverride); if (!res.ok) { const errBody = await res.text(); // Models with no reasoning mode reject the flag; retry without it. if (/think/i.test(errBody)) { res = await this.postOllama(messages, null, ctl.signal, modelOverride); if (!res.ok) throw new Error(`HTTP ${res.status} ${(await res.text()).slice(0, 200)}`); } else { throw new Error(`HTTP ${res.status} ${errBody.slice(0, 200)}`); } } const strip = makeThinkFilter(); for await (const line of readLines(res.body)) { if (!line.trim()) continue; let json; try { json = JSON.parse(line); } catch { continue; } if (json.error) throw new Error(json.error); if (json.message && json.message.content) yield strip(json.message.content); if (json.done) break; } yield strip(null); // flush } finally { clearTimeout(timer); } } } /** Async line reader over a fetch response body stream. */ async function* readLines(body) { const decoder = new TextDecoder(); let buf = ''; for await (const chunk of body) { buf += decoder.decode(chunk, { stream: true }); let idx; while ((idx = buf.indexOf('\n')) >= 0) { yield buf.slice(0, idx); buf = buf.slice(idx + 1); } } if (buf) yield buf; } /** * Suppress ... reasoning blocks. * * Several strong local models (the Qwen3 family among them) emit a reasoning * block before the answer. Streamed verbatim onto a dashboard it looks like the * product is malfunctioning, so it is filtered out. Call with null to flush. */ function makeThinkFilter() { const OPEN = ''; const CLOSE = ''; let inThink = false; let buf = ''; return (chunk) => { if (chunk === null) { const tail = inThink ? '' : buf; buf = ''; return tail; } buf += chunk; let out = ''; for (;;) { if (!inThink) { const i = buf.indexOf(OPEN); if (i === -1) { // Hold back a few characters in case a tag straddles two chunks. const keep = Math.max(0, buf.length - (OPEN.length - 1)); out += buf.slice(0, keep); buf = buf.slice(keep); break; } out += buf.slice(0, i); buf = buf.slice(i + OPEN.length); inThink = true; } else { const j = buf.indexOf(CLOSE); if (j === -1) { buf = buf.slice(Math.max(0, buf.length - (CLOSE.length - 1))); break; } buf = buf.slice(j + CLOSE.length); inThink = false; } } return out; }; }