Initial commit

This commit is contained in:
2026-08-24 15:35:32 +05:30
commit fcade251a6
51 changed files with 11565 additions and 0 deletions
+174
View File
@@ -0,0 +1,174 @@
/**
* Builds the plant context handed to the copilot.
*
* Two rules govern what goes in here.
*
* First: send CONCLUSIONS, not raw floats. A 4B local model reasons well over
* "vibration is rising 0.04 mm/s per minute and reaches its 4.5 limit in about
* 7 minutes" and very badly over six hundred numbers. The analytics layer has
* already done the arithmetic; the model's job is explanation and judgement.
*
* Second: the copilot is NOT told which fault was injected. Operator injection
* log lines are filtered out, so the model has to diagnose from telemetry the
* way it would in a real plant. It is a weaker demo if the model can read the
* answer off a label, and a dishonest one.
*/
import { STATION_SPECS } from '../sim/stations.js';
import { formatDuration } from '../analytics/trend.js';
function fmt(v, precision = 1) {
return Number.isFinite(v) ? v.toFixed(precision) : 'n/a';
}
function pct(v) {
return Number.isFinite(v) ? `${(v * 100).toFixed(1)}%` : 'n/a';
}
/** Where a value sits relative to its declared thresholds. */
function thresholdNote(g, v) {
if (!Number.isFinite(v)) return '';
if (g.alarmHigh !== undefined && v >= g.alarmHigh) return ` [ALARM, limit ${g.alarmHigh}]`;
if (g.warnHigh !== undefined && v >= g.warnHigh) return ` [WARN, limit ${g.warnHigh}]`;
if (g.alarmLow !== undefined && v <= g.alarmLow) return ` [ALARM, limit ${g.alarmLow}]`;
if (g.warnLow !== undefined && v <= g.warnLow) return ` [WARN, limit ${g.warnLow}]`;
return '';
}
/**
* Assemble the context block. Returns plain text, kept deliberately compact.
*/
export function buildContext(frame) {
if (!frame) return 'No telemetry available yet.';
const L = [];
const a = frame.analytics || { alarms: [], predictions: [], trends: [] };
L.push(`PLANT CONTEXT - line ${frame.lineId}`);
L.push(`Simulated run time: ${formatDuration(frame.t)} (clock running at ${frame.sim ? frame.sim.speed : 1}x)`);
L.push('');
// --- KPIs, with the factor breakdown so the model can attribute the loss ---
const k = frame.kpi;
L.push('OEE (rolling 20 simulated minutes):');
L.push(` OEE ${pct(k.oee)} = Availability ${pct(k.availability)} x Performance ${pct(k.performance)} x Quality ${pct(k.quality)}`);
L.push(` Throughput ${fmt(k.throughputPerHour, 0)} good units/hour, scrap ${pct(k.scrapRate)}`);
L.push(` Energy ${fmt(k.energyKw)} kW, ${fmt(k.energyPerUnit, 3)} kWh per unit`);
L.push(` Totals since reset: ${k.produced} produced, ${k.good} good, ${k.rejected} rejected`);
L.push('');
// --- State vocabulary ---
// Without this the model guesses. Asked why OEE was down it invented "safety
// logic that stalls the spindle" and attributed micro-stops to Availability
// instead of Performance. These are definitions, not hints.
L.push('STATION STATE MEANINGS:');
L.push(' running - producing normally.');
L.push(' starved - idle because its input buffer is empty (the constraint is UPSTREAM).');
L.push(' blocked - idle because its output buffer is full (the constraint is DOWNSTREAM).');
L.push(' microstop - a brief stall of a few seconds. Normal line behaviour, counts against PERFORMANCE, not Availability.');
L.push(' down - an unplanned stop of tens of seconds. Counts against AVAILABILITY.');
L.push(' fault - stopped on a specific fault condition. Counts against AVAILABILITY.');
L.push('');
L.push('HOW OEE LOSS IS ATTRIBUTED ON THIS LINE:');
L.push(' Availability loss = time in "down" or "fault" states only.');
L.push(' Performance loss = micro-stops and running below the ideal 4.4 s bottleneck cycle.');
L.push(' Quality loss = parts rejected at INS-04.');
L.push(' "starved" and "blocked" are not losses in their own right; they are the consequence of a stoppage somewhere else on the line.');
L.push('');
// --- Station states and signals ---
L.push('STATIONS (in process order):');
for (const spec of STATION_SPECS) {
const st = frame.stations.find((s) => s.id === spec.id);
if (!st) continue;
const stale = st.online ? '' : ' (NOT REPORTING - values are last known)';
L.push(` ${st.id} ${st.name} - state: ${st.state}${stale}`);
const parts = [];
for (const g of spec.signals) {
if (g.key === 'partsInspected' || g.key === 'downtime') continue;
const v = st.signals[g.key];
parts.push(`${g.label} ${fmt(v, g.precision)} ${g.unit}${thresholdNote(g, v)}`);
}
L.push(` ${parts.join('; ')}`);
}
L.push('');
// --- WIP, which is how blocking and starvation become legible ---
const bufs = frame.buffers
.map((b, i) => `${STATION_SPECS[i].id}->${STATION_SPECS[i + 1].id}: ${b}/${frame.bufferCapacity}`)
.join(', ');
L.push(`WIP BUFFERS: ${bufs}`);
L.push(`Operator setpoints: oven ${frame.controls.setpoint} C, line speed ${frame.controls.lineSpeedPct}%`);
L.push('');
// --- Alarms ---
if (a.alarms.length) {
L.push('ACTIVE ALARMS (most severe first):');
for (const al of a.alarms.slice(0, 12)) {
L.push(` [${al.severity.toUpperCase()}] ${al.station}${al.signal ? '.' + al.signal : ''}: ${al.message}`);
}
} else {
L.push('ACTIVE ALARMS: none.');
}
L.push('');
// --- Trend projections ---
if (a.predictions.length) {
L.push('TREND PROJECTIONS (least-squares fit over recent run time, extrapolated):');
for (const p of a.predictions) {
L.push(` ${p.station}.${p.signal} (${p.label}): now ${fmt(p.current, 2)} ${p.unit}, rising ${p.slopePerMin.toFixed(4)} ${p.unit}/min, reaches limit ${p.threshold} ${p.unit} in about ${p.eta} of run time (fit quality r2=${p.r2.toFixed(2)})`);
}
L.push('');
}
// --- Notable slopes, even where no threshold crossing is projected ---
const notable = (a.trends || [])
.filter((t) => Math.abs(t.slopePerMin) > 1e-4 && t.r2 > 0.5)
.sort((x, y) => y.r2 - x.r2)
.slice(0, 6);
if (notable.length) {
L.push('OTHER MEASURED TRENDS:');
for (const t of notable) {
const dir = t.slopePerMin > 0 ? 'rising' : 'falling';
L.push(` ${t.station}.${t.signal} (${t.label}): ${dir} ${Math.abs(t.slopePerMin).toFixed(4)} ${t.unit}/min (r2=${t.r2.toFixed(2)})`);
}
L.push('');
}
// --- Recent events, with operator fault injections filtered out ---
const events = (frame.events || []).filter((e) => e.kind !== 'inject').slice(0, 8);
if (events.length) {
L.push('RECENT EVENTS (newest first):');
for (const e of events) {
L.push(` t=${formatDuration(e.t)} [${e.kind}] ${e.station}: ${e.message}`);
}
L.push('');
}
// --- Model relationships, so the copilot can reason causally ---
L.push('KNOWN PROCESS RELATIONSHIPS on this line:');
L.push(' - CNC-02 bearing vibration accelerates tool wear, and both push parts out of tolerance, raising INS-04 reject rate.');
L.push(' - High CNC-02 vibration also causes spindle chatter, which stalls the cut and costs Performance.');
L.push(' - OVN-03 zone 2 is the control zone. If actual temperature deviates from setpoint the cure is out of spec and INS-04 reject rate rises.');
L.push(' - If OVN-03 burner duty is saturated at 100% and zone 2 is still below setpoint, the oven has lost heating capacity.');
L.push(' - Stations are linked by finite WIP buffers. A stopped station fills the buffers behind it, so upstream stations become "blocked"; downstream stations become "starved".');
L.push(' - CNC-02 has the longest cycle time (4.4 s), so it is the line bottleneck.');
return L.join('\n');
}
export const SYSTEM_PROMPT = `You are the plant copilot for a manufacturing digital twin of production line LINE-1.
You are given a live telemetry context: station states, sensor readings with their alarm limits, OEE with its Availability/Performance/Quality breakdown, WIP buffer levels, active alarms, and measured trend projections.
How to answer:
- Be concise and concrete. Two to five short sentences for most questions, or a short bullet list. This is read on a factory dashboard, not in a report.
- Always ground claims in the specific numbers from the context. Cite the station and the value.
- Reason causally using the stated process relationships. Explain WHY, not just WHAT.
- Distinguish a measurement from an inference. "Vibration is 4.2 mm/s" is a measurement; "the spindle bearing is degrading" is an inference from it.
- The trend projections are straight-line extrapolations of recent data, not certainties. Present them as "at the current rate", never as a guarantee.
- If the context does not contain what you need, say so plainly. Do not invent readings, part numbers, timestamps, or history you were not given.
- Do NOT invent mechanisms. Explain causes only using the process relationships and state definitions given below. Never assert control logic, safety interlocks, PLC behaviour or physical mechanisms that are not stated there — a plausible-sounding invented mechanism is the single worst failure mode here, because a plant engineer will catch it.
- Attribute OEE losses strictly according to the attribution rules given below. Do not guess which factor a loss belongs to.
- Times are in simulated run time, because the twin can run faster than real time. Say "of run time" when quoting a projection.
- Never claim to have taken an action. You are advisory; the operator drives the line.`;
+580
View File
@@ -0,0 +1,580 @@
/**
* Copilot provider layer.
*
* Resolution order: OpenRouter if a key is set, else Ollama if it answers, else
* the deterministic rule-based fallback. The fallback is not an error path - it
* is a supported mode, and COPILOT_PROVIDER=fallback selects it deliberately.
*
* The hard requirement is that a question ALWAYS gets an answer. A missing key,
* an unreachable host, a model that 404s, a stream that dies halfway: every one
* of those degrades to the fallback rather than surfacing an error on stage.
*/
import { buildContext, SYSTEM_PROMPT } from './context.js';
import { answerFromRules } from './rules.js';
import { loadSettings, saveSettings, validatePatch, resetSettings } from './settings.js';
const PROBE_TIMEOUT_MS = 2000;
const REQUEST_TIMEOUT_MS = 90000;
/**
* Turn whatever OLLAMA_HOST happens to contain into a dialable origin.
*
* Ollama itself commonly sets OLLAMA_HOST=0.0.0.0 machine-wide to bind all
* interfaces. That is a BIND address, not a destination - you cannot connect to
* it - and because it is a real environment variable it silently overrides any
* default we set here. Also accepts a bare host, a host:port, or a full URL.
*/
export function normalizeOllamaHost(raw) {
let h = String(raw || '').trim();
if (!h) return 'http://127.0.0.1:11434';
if (!/^https?:\/\//i.test(h)) h = `http://${h}`;
try {
const u = new URL(h);
if (u.hostname === '0.0.0.0' || u.hostname === '::' || u.hostname === '[::]') {
u.hostname = '127.0.0.1';
}
if (!u.port) u.port = '11434';
return u.origin;
} catch {
return 'http://127.0.0.1:11434';
}
}
export class Copilot {
constructor(env = process.env) {
this.env = env;
// The key stays in the environment only. It is never part of the settings
// that the settings page can read or write.
this.openRouterKey = (env.OPENROUTER_API_KEY || '').trim();
this.applySettings(loadSettings(env));
this.provider = 'fallback';
this.model = null;
this.detail = 'not yet detected';
this.modelCache = new Map();
}
/** Adopt a settings object. Does not persist; see configure(). */
applySettings(settings) {
this.settings = { ...settings };
this.forced = settings.provider === 'auto' ? null : settings.provider;
this.openRouterModel = settings.openRouterModel;
this.ollamaModel = settings.ollamaModel;
this.ollamaHost = normalizeOllamaHost(settings.ollamaHost);
this.temperature = settings.temperature;
this.maxTokens = settings.maxTokens;
this.reasoningEffort = settings.reasoningEffort;
}
/**
* Validate, apply, persist and re-detect.
*
* Returns { ok, errors, settings, status }. A rejected patch changes nothing.
*/
async configure(patch) {
const { ok, errors, clean } = validatePatch(patch);
if (!ok) return { ok: false, errors, settings: this.publicSettings(), status: this.status() };
this.applySettings({ ...this.settings, ...clean });
saveSettings(this.settings);
// The chosen host may have changed, so availability has to be re-probed.
const status = await this.detect();
return { ok: true, errors: [], settings: this.publicSettings(), status };
}
/** Restore environment defaults, discarding the saved settings file. */
async reset() {
resetSettings();
this.applySettings(loadSettings(this.env));
const status = await this.detect();
return { ok: true, errors: [], settings: this.publicSettings(), status };
}
/** Settings safe to hand to a browser: never includes the API key. */
publicSettings() {
return {
...this.settings,
// Report the normalized host, since that is what actually gets dialled.
resolvedOllamaHost: this.ollamaHost,
openRouterKeyPresent: Boolean(this.openRouterKey),
};
}
/** Probe available providers. Safe to call repeatedly. */
async detect() {
if (this.forced === 'fallback') {
this.provider = 'fallback';
this.model = null;
this.detail = 'forced by COPILOT_PROVIDER=fallback';
return this.status();
}
if (this.openRouterKey && this.forced !== 'ollama') {
this.provider = 'openrouter';
this.model = this.openRouterModel;
this.detail = 'OpenRouter API key present';
return this.status();
}
if (this.forced !== 'openrouter') {
const reachable = await this.probeOllama();
if (reachable) {
this.provider = 'ollama';
this.model = this.ollamaModel;
this.detail = `Ollama at ${this.ollamaHost}${reachable.hasModel ? '' : ` (warning: model "${this.ollamaModel}" not in the local list)`}`;
return this.status();
}
}
this.provider = 'fallback';
this.model = null;
this.detail = this.openRouterKey
? 'no provider reachable'
: `no OPENROUTER_API_KEY and Ollama not reachable at ${this.ollamaHost}`;
return this.status();
}
async probeOllama() {
const ctl = new AbortController();
const timer = setTimeout(() => ctl.abort(), PROBE_TIMEOUT_MS);
try {
const res = await fetch(`${this.ollamaHost}/api/tags`, { signal: ctl.signal });
if (!res.ok) return null;
const body = await res.json();
const names = (body.models || []).map((m) => m.name);
return { hasModel: names.includes(this.ollamaModel), names };
} catch {
return null;
} finally {
clearTimeout(timer);
}
}
status() {
return {
provider: this.provider,
model: this.model,
detail: this.detail,
// The UI shows this so you always know on stage what is answering.
label: this.provider === 'fallback'
? 'Rule-based (offline)'
: `${this.provider === 'ollama' ? 'Ollama' : 'OpenRouter'} · ${this.model}`,
};
}
/**
* List the models a provider actually offers.
*
* Cached briefly: OpenRouter returns 400+ models and the settings page may be
* opened repeatedly while someone makes up their mind.
*/
async listModels(provider, { force = false } = {}) {
const key = provider;
const cached = this.modelCache.get(key);
if (!force && cached && Date.now() - cached.at < 5 * 60 * 1000) {
return { ...cached.value, cached: true };
}
let value;
try {
value = provider === 'openrouter'
? await this.listOpenRouterModels()
: await this.listOllamaModels();
} catch (err) {
return { provider, models: [], error: err.message, cached: false };
}
this.modelCache.set(key, { at: Date.now(), value });
return { ...value, cached: false };
}
async listOpenRouterModels() {
if (!this.openRouterKey) {
throw new Error('No OPENROUTER_API_KEY is set, so the model list cannot be fetched.');
}
const ctl = new AbortController();
const timer = setTimeout(() => ctl.abort(), 15000);
try {
const res = await fetch('https://openrouter.ai/api/v1/models', {
signal: ctl.signal,
headers: { Authorization: `Bearer ${this.openRouterKey}` },
});
if (!res.ok) throw new Error(`OpenRouter returned HTTP ${res.status}`);
const body = await res.json();
const models = (body.data || []).map((m) => ({
id: m.id,
name: m.name || m.id,
contextLength: m.context_length || null,
// Prices come back as per-token strings; per-million is what people read.
promptPerM: m.pricing && m.pricing.prompt ? Number(m.pricing.prompt) * 1e6 : null,
completionPerM: m.pricing && m.pricing.completion ? Number(m.pricing.completion) * 1e6 : null,
})).sort((a, b) => a.id.localeCompare(b.id));
return { provider: 'openrouter', models };
} finally {
clearTimeout(timer);
}
}
async listOllamaModels() {
const ctl = new AbortController();
const timer = setTimeout(() => ctl.abort(), PROBE_TIMEOUT_MS);
try {
const res = await fetch(`${this.ollamaHost}/api/tags`, { signal: ctl.signal });
if (!res.ok) throw new Error(`Ollama returned HTTP ${res.status}`);
const body = await res.json();
const models = (body.models || []).map((m) => ({
id: m.name,
name: m.name,
sizeBytes: m.size || null,
parameterSize: m.details ? m.details.parameter_size : null,
quantization: m.details ? m.details.quantization_level : null,
family: m.details ? m.details.family : null,
})).sort((a, b) => a.id.localeCompare(b.id));
return { provider: 'ollama', models };
} catch (err) {
if (err.name === 'AbortError') {
throw new Error(`Ollama did not respond at ${this.ollamaHost}. Is "ollama serve" running?`);
}
throw new Error(`Cannot reach Ollama at ${this.ollamaHost}: ${err.message}`);
} finally {
clearTimeout(timer);
}
}
/**
* Measure a provider/model with a trivial prompt.
*
* Time-to-first-token is the number that matters here, not total time: a model
* that takes 30 s to start talking is unusable in front of a customer even if
* the answer is excellent. This exists so that judgement can be made from a
* measurement rather than a guess, before the meeting rather than during it.
*/
async testProvider({ provider, model } = {}) {
const resolved = this.resolve(provider);
const started = Date.now();
if (resolved === 'fallback') {
// Nothing to probe: the rule engine is in-process and cannot be unavailable.
return {
ok: true, provider: resolved, model: null,
firstTokenMs: Date.now() - started, totalMs: Date.now() - started,
chars: 0,
sample: 'Rule engine ready. No model, no network, no failure mode.',
};
}
const messages = [
{ role: 'system', content: 'You are a terse test endpoint. Reply with exactly: READY' },
{ role: 'user', content: 'Reply with exactly: READY' },
];
let firstTokenMs = null;
let text = '';
try {
const iter = resolved === 'openrouter'
? this.streamOpenRouter(messages, model)
: this.streamOllama(messages, model);
for await (const chunk of iter) {
if (!chunk) continue;
if (firstTokenMs === null) firstTokenMs = Date.now() - started;
text += chunk;
}
} catch (err) {
return {
ok: false, provider: resolved,
model: model || (resolved === 'openrouter' ? this.openRouterModel : this.ollamaModel),
error: err.message, totalMs: Date.now() - started,
};
}
const trimmed = text.trim();
return {
ok: trimmed.length > 0,
provider: resolved,
model: model || (resolved === 'openrouter' ? this.openRouterModel : this.ollamaModel),
firstTokenMs,
totalMs: Date.now() - started,
chars: trimmed.length,
sample: trimmed.slice(0, 120),
error: trimmed.length === 0
? 'The model returned an empty response. It may be spending its whole token budget on reasoning — try a higher max tokens or a lower reasoning effort.'
: undefined,
};
}
buildMessages(question, history, frame) {
const context = buildContext(frame);
const msgs = [{ role: 'system', content: `${SYSTEM_PROMPT}\n\n---\n\n${context}` }];
// Keep only the last few turns: a 4B model with a long context degrades fast,
// and the plant context is refreshed every turn anyway.
for (const m of (history || []).slice(-6)) {
if (m && (m.role === 'user' || m.role === 'assistant') && m.content) {
msgs.push({ role: m.role, content: String(m.content).slice(0, 4000) });
}
}
msgs.push({ role: 'user', content: question });
return msgs;
}
/**
* Stream an answer. Yields { type: 'meta'|'token'|'done'|'note' } objects.
*
* Any provider failure yields a 'note' explaining the degradation and then
* streams the fallback answer, so the caller never has to handle an error.
*/
/**
* Which provider to actually use for one request.
*
* A per-request override exists for a practical reason: a strong reasoning
* model can take 15-20 s to first token, which is dead air in front of a
* customer, while a local 4B model answers in about 2 s with shallower
* analysis. Being able to pick per question - fast for the live walkthrough,
* deep for the follow-up discussion - is worth the small amount of plumbing.
*/
resolve(override) {
const want = String(override || '').trim().toLowerCase();
if (!want || want === 'auto') return this.provider;
if (want === 'fallback') return 'fallback';
if (want === 'openrouter' && this.openRouterKey) return 'openrouter';
if (want === 'ollama') return 'ollama';
return this.provider;
}
statusFor(provider) {
if (provider === 'fallback') {
return { provider, model: null, detail: 'deterministic rule engine', label: 'Rule-based (offline)' };
}
if (provider === 'openrouter') {
return { provider, model: this.openRouterModel, detail: 'OpenRouter', label: `OpenRouter · ${this.openRouterModel}` };
}
return { provider, model: this.ollamaModel, detail: `Ollama at ${this.ollamaHost}`, label: `Ollama · ${this.ollamaModel}` };
}
async *stream(question, history, frame, override) {
const q = String(question || '').trim();
const provider = this.resolve(override);
if (!q) {
yield { type: 'meta', ...this.statusFor(provider) };
yield { type: 'token', text: 'Ask me something about the line.' };
yield { type: 'done' };
return;
}
if (provider === 'fallback') {
yield { type: 'meta', ...this.statusFor(provider) };
const { text } = answerFromRules(q, frame);
yield { type: 'token', text };
yield { type: 'done' };
return;
}
yield { type: 'meta', ...this.statusFor(provider) };
const messages = this.buildMessages(q, history, frame);
let produced = '';
let failure = null;
try {
const iter = provider === 'openrouter'
? this.streamOpenRouter(messages)
: this.streamOllama(messages);
for await (const text of iter) {
if (!text) continue;
produced += text;
yield { type: 'token', text };
}
} catch (err) {
failure = err && err.message ? err.message : String(err);
console.error('[copilot] provider failed:', failure);
}
// An empty answer is a failure even when nothing threw. A thinking model that
// exhausts its token budget mid-reasoning returns a clean 200 with no content,
// and a blank panel is the worst possible outcome in front of a customer.
if (!produced.trim()) {
yield {
type: 'note',
text: failure
? `${provider} failed (${failure}). Answering from the built-in rule engine instead.`
: `${provider} returned an empty answer. Answering from the built-in rule engine instead.`,
};
const { text } = answerFromRules(q, frame);
yield { type: 'token', text };
} else if (failure) {
yield { type: 'note', text: `Stream ended early: ${failure}` };
}
yield { type: 'done' };
}
// --- providers -----------------------------------------------------------
async *streamOpenRouter(messages, modelOverride) {
const ctl = new AbortController();
const timer = setTimeout(() => ctl.abort(), REQUEST_TIMEOUT_MS);
try {
const body = {
model: modelOverride || this.openRouterModel,
messages,
stream: true,
temperature: this.temperature,
// Reasoning tokens count against max_tokens, so a reasoning model on a
// tight budget burns the lot thinking and streams back nothing at all.
// Keeping the budget generous and reasoning minimal stops the
// intermittent empty answers and cuts time-to-first-token, which is the
// difference between a usable and an awkward live demo.
max_tokens: this.maxTokens,
};
if (this.reasoningEffort !== 'none') {
body.reasoning = { effort: this.reasoningEffort };
}
const res = await fetch('https://openrouter.ai/api/v1/chat/completions', {
method: 'POST',
signal: ctl.signal,
headers: {
'Content-Type': 'application/json',
Authorization: `Bearer ${this.openRouterKey}`,
'X-Title': 'Digital Twin Demo',
},
body: JSON.stringify(body),
});
if (!res.ok) {
throw new Error(`HTTP ${res.status} ${(await res.text()).slice(0, 200)}`);
}
const strip = makeThinkFilter();
for await (const line of readLines(res.body)) {
if (!line.startsWith('data:')) continue;
const payload = line.slice(5).trim();
if (!payload || payload === '[DONE]') continue;
let json;
try { json = JSON.parse(payload); } catch { continue; }
const delta = json.choices && json.choices[0] && json.choices[0].delta;
if (delta && delta.content) yield strip(delta.content);
}
yield strip(null); // flush
} finally {
clearTimeout(timer);
}
}
/**
* POST to Ollama. `think` of null omits the field entirely.
*/
postOllama(messages, think, signal, modelOverride) {
const body = {
model: modelOverride || this.ollamaModel,
messages,
stream: true,
options: { temperature: this.temperature, num_predict: this.maxTokens },
};
if (think !== null) body.think = think;
return fetch(`${this.ollamaHost}/api/chat`, {
method: 'POST',
signal,
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify(body),
});
}
async *streamOllama(messages, modelOverride) {
const ctl = new AbortController();
const timer = setTimeout(() => ctl.abort(), REQUEST_TIMEOUT_MS);
try {
// Ollama's `think` is a boolean, so the four-level effort setting maps onto
// it: none/low disable reasoning, medium/high enable it. The Qwen3 family
// and other thinking models will otherwise spend the entire token budget
// inside a reasoning block and return EMPTY content (done_reason:
// "length") - a blank copilot panel on stage - and cost ~10x the latency.
const think = this.reasoningEffort === 'medium' || this.reasoningEffort === 'high';
let res = await this.postOllama(messages, think, ctl.signal, modelOverride);
if (!res.ok) {
const errBody = await res.text();
// Models with no reasoning mode reject the flag; retry without it.
if (/think/i.test(errBody)) {
res = await this.postOllama(messages, null, ctl.signal, modelOverride);
if (!res.ok) throw new Error(`HTTP ${res.status} ${(await res.text()).slice(0, 200)}`);
} else {
throw new Error(`HTTP ${res.status} ${errBody.slice(0, 200)}`);
}
}
const strip = makeThinkFilter();
for await (const line of readLines(res.body)) {
if (!line.trim()) continue;
let json;
try { json = JSON.parse(line); } catch { continue; }
if (json.error) throw new Error(json.error);
if (json.message && json.message.content) yield strip(json.message.content);
if (json.done) break;
}
yield strip(null); // flush
} finally {
clearTimeout(timer);
}
}
}
/** Async line reader over a fetch response body stream. */
async function* readLines(body) {
const decoder = new TextDecoder();
let buf = '';
for await (const chunk of body) {
buf += decoder.decode(chunk, { stream: true });
let idx;
while ((idx = buf.indexOf('\n')) >= 0) {
yield buf.slice(0, idx);
buf = buf.slice(idx + 1);
}
}
if (buf) yield buf;
}
/**
* Suppress <think>...</think> reasoning blocks.
*
* Several strong local models (the Qwen3 family among them) emit a reasoning
* block before the answer. Streamed verbatim onto a dashboard it looks like the
* product is malfunctioning, so it is filtered out. Call with null to flush.
*/
function makeThinkFilter() {
const OPEN = '<think>';
const CLOSE = '</think>';
let inThink = false;
let buf = '';
return (chunk) => {
if (chunk === null) {
const tail = inThink ? '' : buf;
buf = '';
return tail;
}
buf += chunk;
let out = '';
for (;;) {
if (!inThink) {
const i = buf.indexOf(OPEN);
if (i === -1) {
// Hold back a few characters in case a tag straddles two chunks.
const keep = Math.max(0, buf.length - (OPEN.length - 1));
out += buf.slice(0, keep);
buf = buf.slice(keep);
break;
}
out += buf.slice(0, i);
buf = buf.slice(i + OPEN.length);
inThink = true;
} else {
const j = buf.indexOf(CLOSE);
if (j === -1) {
buf = buf.slice(Math.max(0, buf.length - (CLOSE.length - 1)));
break;
}
buf = buf.slice(j + CLOSE.length);
inThink = false;
}
}
return out;
};
}
+290
View File
@@ -0,0 +1,290 @@
/**
* Deterministic rule-based copilot.
*
* This is the offline fallback, and it is not a stub. On stage, a dead API key or
* bad conference wifi must never turn the copilot panel into an error message, so
* this produces a real, grounded, useful answer from the same analytics the LLM
* would have seen. It is less fluent and cannot handle open-ended questions, but
* it never fails and never invents a number.
*
* It is also the honesty backstop: everything here is traceable to a threshold or
* a regression, so if you are ever asked "is the AI making this up?", you can run
* with COPILOT_PROVIDER=fallback and show the same conclusions without a model.
*/
import { STATION_SPECS } from '../sim/stations.js';
import { formatDuration } from '../analytics/trend.js';
const STATION_ALIASES = {
'CONV-01': ['conv', 'conveyor', 'infeed', 'belt'],
'CNC-02': ['cnc', 'spindle', 'machining', 'mill', 'bearing'],
'OVN-03': ['ovn', 'oven', 'cure', 'curing', 'burner', 'zone'],
'INS-04': ['ins', 'inspection', 'vision', 'camera', 'reject', 'quality station'],
'PKG-05': ['pkg', 'packer', 'packing', 'film', 'wrap'],
};
function pct(v) {
return Number.isFinite(v) ? `${(v * 100).toFixed(1)}%` : 'n/a';
}
function resolveStation(q) {
const lower = q.toLowerCase();
for (const [id, words] of Object.entries(STATION_ALIASES)) {
if (lower.includes(id.toLowerCase())) return id;
if (words.some((w) => lower.includes(w))) return id;
}
return null;
}
function signalLabel(stationId, key) {
const spec = STATION_SPECS.find((s) => s.id === stationId);
const g = spec && spec.signals.find((x) => x.key === key);
return g ? g.label : key;
}
/** The causal story behind the alarms currently active, where we can name one. */
function causalNarrative(frame) {
const a = frame.analytics;
const cnc = frame.stations.find((s) => s.id === 'CNC-02');
const ovn = frame.stations.find((s) => s.id === 'OVN-03');
const ins = frame.stations.find((s) => s.id === 'INS-04');
const out = [];
if (cnc && cnc.signals.vibration > 3.0) {
out.push(
`CNC-02 bearing vibration is ${cnc.signals.vibration.toFixed(2)} mm/s RMS against a 3.5 warn and 4.5 alarm limit, and spindle load has risen to ${cnc.signals.spindleLoad.toFixed(1)}%. That pattern is bearing degradation: it both stalls the cut, costing Performance, and pushes parts out of tolerance, which is why INS-04 reject rate is ${ins ? ins.signals.rejectRate.toFixed(2) : '?'}%.`,
);
}
if (ovn && Math.abs(ovn.signals.tempDeviation) > 6) {
const sat = ovn.signals.burnerDuty > 95;
out.push(
`OVN-03 zone 2 is ${ovn.signals.zone2Temp.toFixed(1)} °C against a ${ovn.signals.setpoint} °C setpoint, a deviation of ${ovn.signals.tempDeviation.toFixed(1)} °C, with burner duty at ${ovn.signals.burnerDuty.toFixed(0)}%.` +
(sat
? ' Duty is saturated and the zone still cannot reach setpoint, which means the oven has lost heating capacity rather than being mistuned. The cure is out of spec, so reject rate downstream is rising.'
: ' The controller is still recovering toward setpoint.'),
);
}
const stopped = frame.stations.filter((s) => s.state === 'fault' || s.state === 'down');
if (stopped.length) {
const blocked = frame.stations.filter((s) => s.state === 'blocked').map((s) => s.id);
out.push(
`${stopped.map((s) => s.id).join(', ')} ${stopped.length > 1 ? 'are' : 'is'} stopped.` +
(blocked.length
? ` WIP has backed up behind it and ${blocked.join(', ')} ${blocked.length > 1 ? 'are' : 'is'} now blocked, so the whole line has stopped producing.`
: ''),
);
}
const offline = frame.stations.filter((s) => !s.online);
if (offline.length) {
out.push(
`${offline.map((s) => s.id).join(', ')} ${offline.length > 1 ? 'are' : 'is'} not reporting. The values shown are the last known readings and should not be trusted as live.`,
);
}
if (!out.length && a.predictions.length) {
const p = a.predictions[0];
out.push(
`Nothing is over limit yet, but ${p.station} ${p.label} is rising ${p.slopePerMin.toFixed(4)} ${p.unit}/min and reaches its ${p.threshold} ${p.unit} limit in about ${p.eta} of run time at the current rate.`,
);
}
return out;
}
function oeeAnswer(frame) {
const k = frame.kpi;
const factors = [
{ name: 'Availability', v: k.availability, why: 'time lost to stoppages' },
{ name: 'Performance', v: k.performance, why: 'the line running slower than its ideal cycle, usually micro-stops' },
{ name: 'Quality', v: k.quality, why: 'parts rejected at inspection' },
].sort((a, b) => a.v - b.v);
const worst = factors[0];
const lines = [
`OEE is ${pct(k.oee)} over the last 20 simulated minutes: Availability ${pct(k.availability)} x Performance ${pct(k.performance)} x Quality ${pct(k.quality)}.`,
`The biggest loss is ${worst.name} at ${pct(worst.v)} — ${worst.why}.`,
];
const narrative = causalNarrative(frame);
if (narrative.length) lines.push(narrative[0]);
lines.push(`Current output is ${k.throughputPerHour.toFixed(0)} good units/hour with ${pct(k.scrapRate)} scrap.`);
return lines.join('\n\n');
}
function stationAnswer(frame, stationId) {
const spec = STATION_SPECS.find((s) => s.id === stationId);
const st = frame.stations.find((s) => s.id === stationId);
if (!st) return `I have no telemetry for ${stationId}.`;
const lines = [`${st.id} ${st.name} is currently ${st.state}${st.online ? '' : ' and NOT REPORTING (values below are stale)'}.`];
const issues = [];
for (const g of spec.signals) {
const v = st.signals[g.key];
if (!Number.isFinite(v)) continue;
if (g.alarmHigh !== undefined && v >= g.alarmHigh) issues.push(`${g.label} ${v.toFixed(g.precision)} ${g.unit} is over its ${g.alarmHigh} alarm limit`);
else if (g.warnHigh !== undefined && v >= g.warnHigh) issues.push(`${g.label} ${v.toFixed(g.precision)} ${g.unit} is over its ${g.warnHigh} warning limit`);
else if (g.alarmLow !== undefined && v <= g.alarmLow) issues.push(`${g.label} ${v.toFixed(g.precision)} ${g.unit} is under its ${g.alarmLow} alarm limit`);
else if (g.warnLow !== undefined && v <= g.warnLow) issues.push(`${g.label} ${v.toFixed(g.precision)} ${g.unit} is under its ${g.warnLow} warning limit`);
}
lines.push(issues.length ? `Out of limits: ${issues.join('; ')}.` : 'All signals are within limits.');
const preds = frame.analytics.predictions.filter((p) => p.station === stationId);
for (const p of preds) {
lines.push(`${p.label} is rising ${p.slopePerMin.toFixed(4)} ${p.unit}/min and would reach its ${p.threshold} ${p.unit} limit in about ${p.eta} of run time if the trend holds.`);
}
const narrative = causalNarrative(frame).filter((n) => n.includes(stationId));
lines.push(...narrative);
const readings = spec.signals
.filter((g) => g.key !== 'partsInspected' && g.key !== 'downtime')
.map((g) => `${g.label} ${st.signals[g.key].toFixed(g.precision)} ${g.unit}`)
.join(', ');
lines.push(`Current readings: ${readings}.`);
return lines.join('\n\n');
}
function predictionAnswer(frame) {
const preds = frame.analytics.predictions;
if (!preds.length) {
return 'No threshold crossings are projected right now. Either nothing is trending toward a limit, or the recent data does not fit a straight line well enough to extrapolate honestly.';
}
const lines = ['Projected threshold crossings, from a least-squares fit over recent run time:'];
for (const p of preds) {
lines.push(`- ${p.station} ${p.label}: now ${p.current.toFixed(2)} ${p.unit}, rising ${p.slopePerMin.toFixed(4)} ${p.unit}/min, reaches ${p.threshold} ${p.unit} in about ${p.eta} of run time (fit quality r²=${p.r2.toFixed(2)}).`);
}
lines.push('These are straight-line extrapolations of the current trend, not guarantees. A change in load or a maintenance action will change them.');
return lines.join('\n');
}
function workOrderAnswer(frame) {
const a = frame.analytics;
const worst = a.alarms[0];
const pred = a.predictions[0];
const cnc = frame.stations.find((s) => s.id === 'CNC-02');
const target = worst ? worst.station : pred ? pred.station : 'LINE-1';
const priority = worst
? worst.severity === 'critical' ? 'P1 - immediate' : worst.severity === 'major' ? 'P2 - same shift' : 'P3 - planned'
: 'P3 - planned';
const lines = [
'DRAFT MAINTENANCE WORK ORDER',
`Asset: ${target}${target === 'CNC-02' ? ' (CNC Machining Centre)' : ''}`,
`Priority: ${priority}`,
`Raised at: simulated run time ${formatDuration(frame.t)}`,
'',
'Observed condition:',
];
if (worst) lines.push(`- ${worst.message}`);
if (cnc && cnc.signals.vibration > 3) {
lines.push(`- Bearing vibration ${cnc.signals.vibration.toFixed(2)} mm/s RMS (baseline approximately 1.6), spindle load ${cnc.signals.spindleLoad.toFixed(1)}%, coolant ${cnc.signals.coolantTemp.toFixed(1)} °C.`);
}
if (pred) {
lines.push(`- ${pred.label} projected to reach ${pred.threshold} ${pred.unit} in about ${pred.eta} of run time at the current rate.`);
}
lines.push('', 'Recommended action:');
if (target === 'CNC-02') {
lines.push('- Inspect spindle bearing; take a vibration spectrum to confirm the fault frequency before replacing.');
lines.push('- Check tooling condition and replace if wear is contributing.');
lines.push('- Verify coolant flow and temperature.');
} else if (target === 'OVN-03') {
lines.push('- Inspect zone 2 burner, igniter and gas train; burner duty is saturated, which points at lost heating capacity.');
lines.push('- Verify zone 2 thermocouple against a reference before condemning the burner.');
} else if (target === 'PKG-05') {
lines.push('- Clear the film path and inspect the web for tearing; check tension control calibration.');
} else {
lines.push('- Investigate the alarm above and confirm against local instrumentation.');
}
lines.push('', 'Production impact:');
lines.push(`- OEE ${pct(frame.kpi.oee)} (A ${pct(frame.kpi.availability)} / P ${pct(frame.kpi.performance)} / Q ${pct(frame.kpi.quality)}), scrap ${pct(frame.kpi.scrapRate)}, output ${frame.kpi.throughputPerHour.toFixed(0)} units/h.`);
lines.push('', 'Drafted by the plant copilot from live telemetry. Review before issuing.');
return lines.join('\n');
}
function whatIfAnswer(frame, q) {
const lower = q.toLowerCase();
const k = frame.kpi;
const lines = [];
if (lower.includes('speed') || lower.includes('faster') || lower.includes('rate')) {
const cnc = frame.stations.find((s) => s.id === 'CNC-02');
lines.push(
`CNC-02 is the bottleneck at a 4.4 s cycle, so line output tracks it directly. Raising line speed shortens every cycle, but it also raises spindle load — currently ${cnc.signals.spindleLoad.toFixed(1)}% — and load rises roughly 22 points per 100% of added speed.`,
);
if (cnc.signals.vibration > 3) {
lines.push(`With bearing vibration already at ${cnc.signals.vibration.toFixed(2)} mm/s, running faster would accelerate the degradation and increase rejects. I would not raise speed until the bearing is addressed.`);
} else {
lines.push(`Vibration is ${cnc.signals.vibration.toFixed(2)} mm/s and load has headroom, so a modest increase is likely to hold. Watch spindle load against its 85% warning limit and reject rate against 4%.`);
}
lines.push('Use the line speed control in the what-if panel to try it — the twin will show the actual response.');
return lines.join('\n\n');
}
if (lower.includes('setpoint') || lower.includes('temperature') || lower.includes('oven') || lower.includes('hotter') || lower.includes('cooler')) {
const ovn = frame.stations.find((s) => s.id === 'OVN-03');
lines.push(
`Zone 2 is at ${ovn.signals.zone2Temp.toFixed(1)} °C against a ${ovn.signals.setpoint} °C setpoint with burner duty ${ovn.signals.burnerDuty.toFixed(0)}%. The oven responds with a first-order lag of roughly 50 s per zone, so a setpoint change takes a few minutes of run time to settle, not seconds.`,
);
lines.push('Reject rate rises once zone 2 deviates more than about 6 °C from setpoint in either direction, so the cure window is the constraint, not the absolute temperature.');
lines.push('Change the setpoint in the what-if panel and watch the zone 2 trend chart to see the real response.');
return lines.join('\n\n');
}
lines.push(`I can reason about two levers directly: oven setpoint and line speed. Current state is OEE ${pct(k.oee)}, output ${k.throughputPerHour.toFixed(0)} units/h, scrap ${pct(k.scrapRate)}.`);
lines.push('Ask about raising line speed or changing the oven setpoint, or make the change in the what-if panel and watch the twin respond.');
return lines.join('\n\n');
}
function overviewAnswer(frame) {
const a = frame.analytics;
const lines = [];
const narrative = causalNarrative(frame);
if (a.alarms.length === 0 && narrative.length === 0) {
lines.push(`The line is running clean. OEE ${pct(frame.kpi.oee)}, output ${frame.kpi.throughputPerHour.toFixed(0)} good units/hour, scrap ${pct(frame.kpi.scrapRate)}, no active alarms.`);
lines.push('All five stations are within limits and no threshold crossings are projected.');
return lines.join('\n\n');
}
if (a.alarms.length) {
const top = a.alarms.slice(0, 3);
lines.push(`${a.alarms.length} active alarm${a.alarms.length > 1 ? 's' : ''}. Most severe: ${top.map((x) => `[${x.severity}] ${x.message}`).join(' ')}`);
}
lines.push(...narrative);
lines.push(`OEE is ${pct(frame.kpi.oee)} (A ${pct(frame.kpi.availability)} / P ${pct(frame.kpi.performance)} / Q ${pct(frame.kpi.quality)}), output ${frame.kpi.throughputPerHour.toFixed(0)} units/h.`);
return lines.join('\n\n');
}
/**
* Route a question to a templated answer.
*
* Returns { text, provider: 'fallback' }.
*/
export function answerFromRules(question, frame) {
if (!frame) {
return { text: 'No telemetry has arrived yet. Start the simulation and ask again.', provider: 'fallback' };
}
const q = (question || '').toLowerCase();
if (/work order|maintenance order|raise a ticket|wo\b|cmms/.test(q)) return { text: workOrderAnswer(frame), provider: 'fallback' };
if (/what happens if|what if|should i|raise|increase|decrease|lower|try /.test(q)) return { text: whatIfAnswer(frame, question), provider: 'fallback' };
if (/oee|availability|performance|quality|efficiency|throughput|scrap/.test(q)) return { text: oeeAnswer(frame), provider: 'fallback' };
if (/predict|forecast|when will|how long|fail|remaining life|rul/.test(q)) return { text: predictionAnswer(frame), provider: 'fallback' };
const station = resolveStation(q);
if (station) return { text: stationAnswer(frame, station), provider: 'fallback' };
return { text: overviewAnswer(frame), provider: 'fallback' };
}
+113
View File
@@ -0,0 +1,113 @@
/**
* Runtime copilot settings.
*
* Precedence: saved settings file > environment > built-in defaults.
*
* Settings are persisted to a small JSON file rather than held only in memory, so
* a model chosen while preparing a demo survives a server restart. They are
* deliberately NOT written back into .env - that file holds the API key, and a
* process that rewrites its own secrets file is a bad habit to build.
*/
import fs from 'node:fs';
import path from 'node:path';
import { fileURLToPath } from 'node:url';
const HERE = path.dirname(fileURLToPath(import.meta.url));
export const SETTINGS_FILE = path.join(HERE, '..', '..', '.copilot-settings.json');
/** Fields a client is allowed to change, with validation for each. */
const FIELDS = {
provider: {
validate: (v) => ['auto', 'openrouter', 'ollama', 'fallback'].includes(v),
error: 'provider must be auto, openrouter, ollama or fallback',
},
openRouterModel: {
validate: (v) => typeof v === 'string' && v.length > 0 && v.length < 200,
error: 'openRouterModel must be a non-empty model id',
},
ollamaModel: {
validate: (v) => typeof v === 'string' && v.length > 0 && v.length < 200,
error: 'ollamaModel must be a non-empty model id',
},
ollamaHost: {
validate: (v) => typeof v === 'string' && v.length > 0 && v.length < 300,
error: 'ollamaHost must be a URL or host:port',
},
temperature: {
validate: (v) => Number.isFinite(v) && v >= 0 && v <= 2,
error: 'temperature must be between 0 and 2',
},
maxTokens: {
validate: (v) => Number.isInteger(v) && v >= 128 && v <= 8000,
error: 'maxTokens must be an integer between 128 and 8000',
},
reasoningEffort: {
validate: (v) => ['none', 'low', 'medium', 'high'].includes(v),
error: 'reasoningEffort must be none, low, medium or high',
},
};
export function defaultsFromEnv(env = process.env) {
return {
// 'auto' resolves by availability at detect() time.
provider: (env.COPILOT_PROVIDER || 'auto').trim().toLowerCase(),
openRouterModel: (env.OPENROUTER_MODEL || 'meta-llama/llama-3.3-70b-instruct').trim(),
ollamaModel: (env.OLLAMA_MODEL || 'qwen3.5-4b-32k:latest').trim(),
ollamaHost: (env.OLLAMA_HOST || 'http://127.0.0.1:11434').trim(),
temperature: 0.3,
maxTokens: 2000,
// Reasoning tokens count against the output budget on most providers, so
// 'low' keeps answers from being swallowed by a reasoning block and keeps
// time-to-first-token usable in a live demo.
reasoningEffort: 'low',
};
}
export function loadSettings(env = process.env) {
const base = defaultsFromEnv(env);
try {
const raw = JSON.parse(fs.readFileSync(SETTINGS_FILE, 'utf8'));
for (const key of Object.keys(FIELDS)) {
if (raw[key] !== undefined && FIELDS[key].validate(raw[key])) base[key] = raw[key];
}
} catch {
/* no saved settings yet, or unreadable - environment defaults stand */
}
return base;
}
/**
* Validate a patch. Returns { ok, errors, clean }.
*
* Unknown keys are rejected rather than ignored, so a typo in a field name fails
* loudly instead of silently doing nothing.
*/
export function validatePatch(patch) {
const errors = [];
const clean = {};
for (const [key, value] of Object.entries(patch || {})) {
const field = FIELDS[key];
if (!field) { errors.push(`unknown setting "${key}"`); continue; }
if (!field.validate(value)) { errors.push(field.error); continue; }
clean[key] = value;
}
return { ok: errors.length === 0, errors, clean };
}
export function saveSettings(settings) {
const out = {};
for (const key of Object.keys(FIELDS)) {
if (settings[key] !== undefined) out[key] = settings[key];
}
fs.writeFileSync(SETTINGS_FILE, `${JSON.stringify(out, null, 2)}\n`, 'utf8');
return out;
}
export function resetSettings() {
try {
fs.unlinkSync(SETTINGS_FILE);
} catch {
/* nothing saved - already at defaults */
}
}