mo commited on
Commit
9b04a00
·
1 Parent(s): 854b98d

feat(reasoning): deep reasoning engine — plan, act, reflect, verify

Browse files

- reasoning protocol: brain replies with explicit reasoning JSON
({reasoning, tool, args} | {reasoning, answer} | plain text)
- parseBrainReply parser: tolerant to prose/fences/arrays, malformed JSON
triggers the corrective nudge instead of leaking raw JSON to users
- reflect phase: after every tool result the brain must decide whether the
goal is satisfied before acting again
- self-verification pass: the final answer is re-checked against the tool
evidence before it reaches the user (caught 'Hi RULESOK' = two words live)
- owner-tunable brain config from the cloud: maxSteps, temperature,
extraRules, verify toggle — delivered with the task queue
- agent system prompt now injected into device tasks
- temperature is forwarded to the LLM call (all 5 providers)
- 9 new parser tests (43/43 passing)

apps/desktop/src/agent.js CHANGED
@@ -26,7 +26,7 @@ function truncate(s, n) {
26
  }
27
 
28
  /** Extract a {tool, args} JSON call from a model reply (plain, fenced, or with prose around it). */
29
- function extractBalancedJson(s, start) {
30
  let depth = 0, inStr = false, esc = false;
31
  for (let i = start; i < s.length; i++) {
32
  const c = s[i];
@@ -43,7 +43,8 @@ function extractBalancedJson(s, start) {
43
  return null;
44
  }
45
 
46
- function parseToolCall(text) {
 
47
  if (!text) return null;
48
  const bodies = [text];
49
  for (const m of text.matchAll(/```(?:json)?\s*([\s\S]*?)```/g)) bodies.push(m[1]);
@@ -74,6 +75,78 @@ function parseToolCall(text) {
74
  return null;
75
  }
76
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
77
  export class AgentDaemon extends EventEmitter {
78
  #creds;
79
  #control;
@@ -82,6 +155,8 @@ export class AgentDaemon extends EventEmitter {
82
  #currentTask = null;
83
  #taskPoll = null;
84
  #polling = false;
 
 
85
 
86
  constructor(creds) {
87
  super();
@@ -160,7 +235,9 @@ export class AgentDaemon extends EventEmitter {
160
  if (this.#polling) return;
161
  this.#polling = true;
162
  try {
163
- const tasks = await pollTasks(this.#creds.apiKey);
 
 
164
  for (const t of tasks) {
165
  if (t.status !== 'pending') continue;
166
  try { await claimTask(this.#creds.apiKey, t.id); } catch { /* already claimed */ }
@@ -173,9 +250,18 @@ export class AgentDaemon extends EventEmitter {
173
  }
174
  }
175
 
 
 
 
 
 
 
 
 
 
176
  // ── Task execution (cloud reasoning + local tools) ──────────────
177
  /** think() with auto-retry for transient failures — fail is never allowed. */
178
- async #thinkWithRetry(messages, runId) {
179
  let lastErr = null;
180
  for (let attempt = 1; attempt <= MAX_RETRIES; attempt++) {
181
  try {
@@ -183,6 +269,7 @@ export class AgentDaemon extends EventEmitter {
183
  apiKey: this.#creds.apiKey,
184
  messages,
185
  tools: tools.list(),
 
186
  onChunk: (delta) => {
187
  this.#control.token(delta, runId);
188
  this.emit('task:token', delta, runId);
@@ -287,15 +374,16 @@ export class AgentDaemon extends EventEmitter {
287
  final = res.content || '';
288
  this.#messages.push({ role: 'assistant', content: final });
289
  } else {
290
- // Sngine platform: agentic loop — brain reasons, device executes.
291
- const systemPrompt = this.#toolsPrompt();
 
292
  const messages = [
293
  { role: 'system', content: systemPrompt },
294
  { role: 'user', content: task },
295
  ];
296
 
297
  let corrections = 0;
298
- for (let i = 0; i < MAX_ITERS; i++) {
299
  const tThink = Date.now();
300
  const thinkRes = await this.#thinkWithRetry(messages, runId);
301
  const answer = thinkRes.text ?? thinkRes ?? '';
@@ -303,90 +391,119 @@ export class AgentDaemon extends EventEmitter {
303
  if (thinkRes.usage) addUsage(thinkRes.usage);
304
  if (thinkRes.model) lastModel = thinkRes.model;
305
  if (thinkRes.provider) lastProvider = thinkRes.provider;
306
- const toolCalls = parseToolCall(answer);
307
-
308
- // Trace: the brain's raw reasoning step
309
- await trace('think', {
310
- summary: truncate(answer.trim().split('\n')[0] || answer, 500),
311
- detail: truncate(answer, 6000),
312
- model: thinkRes.model || '',
313
- provider: thinkRes.provider || '',
314
- usage: thinkRes.usage || null,
315
- durationMs: thinkMs,
316
- });
317
-
318
- if (!toolCalls || !toolCalls.length) {
319
- const hasText = answer && answer.trim();
320
- const looksLikeAttempt = hasText && /"tool"\s*:/.test(answer);
321
- if (hasText && !looksLikeAttempt) { final = answer; break; }
322
- // empty or malformed corrective nudge (auto-reasoning)
323
- if (corrections >= MAX_CORRECTIONS) {
324
- final = hasText ? answer : 'The brain produced no usable reply. Check the activity feed for the trace.';
325
- break;
326
- }
327
- corrections++;
328
- const hint = looksLikeAttempt
329
- ? 'Your last message was not valid JSON. Reply with ONLY one JSON object: {"tool":"<tool name>","args":{...}} — or plain text if you are done.'
330
- : 'Your reply was empty. Either answer in plain text or emit ONE JSON tool call.';
331
- messages.push({ role: 'assistant', content: answer });
332
- messages.push({ role: 'user', content: hint });
333
- await trace('correct', { summary: looksLikeAttempt ? 'malformed reply — corrective nudge' : 'empty reply — corrective nudge' });
334
- await postActivity(this.#creds.apiKey, 'auto.correct', { reason: looksLikeAttempt ? 'malformed' : 'empty', attempt: corrections }, runId, this.#creds.agentId).catch(() => {});
335
- continue;
336
  }
337
- corrections = 0;
338
 
339
- messages.push({ role: 'assistant', content: answer });
 
 
 
 
 
 
 
 
 
 
 
340
 
341
- // Execute every requested tool (JSON arrays = multi-tool steps)
342
- const resultLines = [];
343
- for (const toolCall of toolCalls) {
344
- steps.push({ type: 'tool.call', tool: toolCall.tool, args: toolCall.args });
345
- this.#control.step('tool.call', { tool: toolCall.tool });
346
- this.emit('tool:start', toolCall.tool, toolCall.args);
347
- log.info(`Tool call: ${toolCall.tool}`);
348
- await postActivity(this.#creds.apiKey, 'tool.call', { tool: toolCall.tool, args: toolCall.args }, runId, this.#creds.agentId).catch(() => {});
349
- await trace('tool.call', { summary: toolCall.tool, detail: JSON.stringify(toolCall.args || {}, null, 2) });
350
-
351
- const tTool = Date.now();
352
- let result;
353
- try {
354
- result = await tools.run(toolCall.tool, toolCall.args || {});
355
- } catch (err) {
356
- result = { error: err.message };
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
357
  }
358
- this.#stats.toolCalls++;
359
-
360
- const out = truncate(JSON.stringify(result), TOOL_OUT_MAX);
361
- steps.push({ type: 'tool.result', tool: toolCall.tool, output: truncate(out, 400) });
362
- await postActivity(this.#creds.apiKey, 'tool.result', { tool: toolCall.tool, output: truncate(out, 200) }, runId, this.#creds.agentId).catch(() => {});
363
- await trace('tool.result', {
364
- summary: truncate(out, 500),
365
- detail: truncate(out, 4000),
366
- durationMs: Date.now() - tTool,
367
- });
368
- log.info(`Tool result (${toolCall.tool}): ${truncate(out, 120)}`);
369
 
370
- const failed = result && (result.error || result.exitCode);
371
- if (failed) await this.#debugSnapshot(runId, `tool ${toolCall.tool} failed`);
372
- resultLines.push(`TOOL RESULT (${toolCall.tool}):\n${out}`);
 
 
 
 
 
373
  }
374
 
375
- // Feed all results back together, with a diagnostic nudge on error
376
- const anyFailed = resultLines.some((l) => /"error"|exitCode":[1-9]/.test(l));
377
- const debugHint = anyFailed
378
- ? `\n\nAt least one tool returned an error. Diagnose it, then either fix the call or use a different tool/approach. Do not give up — verify your fix by running it again.`
379
- : '';
380
- messages.push({ role: 'user', content: resultLines.join('\n\n') + debugHint });
 
 
 
 
 
 
 
 
 
381
  }
382
 
383
- // Never give up: one forced plain-text conclusion when steps run out
384
  if (!final) {
385
- messages.push({ role: 'user', content: 'You must now answer in plain text. Summarize what you did, what the last tool result showed, and the current state. Do not call tools.' });
386
  try {
387
  const tThink = Date.now();
388
  const thinkRes = await this.#thinkWithRetry(messages, runId);
389
- final = thinkRes.text ?? thinkRes ?? '';
 
 
 
390
  if (thinkRes.usage) addUsage(thinkRes.usage);
391
  if (thinkRes.model) lastModel = thinkRes.model;
392
  if (thinkRes.provider) lastProvider = thinkRes.provider;
@@ -402,6 +519,33 @@ export class AgentDaemon extends EventEmitter {
402
  final = 'The agent hit its step limit. See the activity feed for the full execution trace.';
403
  }
404
  }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
405
  }
406
  } catch (err) {
407
  this.#stats.errors++;
@@ -479,23 +623,45 @@ export class AgentDaemon extends EventEmitter {
479
  }
480
  }
481
 
482
- #toolsPrompt() {
483
  const rows = tools.list()
484
  .map((t) => `- ${t.name}: ${t.description}${t.args ? ` (args: ${JSON.stringify(t.args)})` : ''}`)
485
  .join('\n');
486
- return `You are mona-agent — the AI agent controlling this device (${process.platform}). You reason and act: use the local tools below to get real information and take real actions, then give the user a direct, concise answer.
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
487
 
488
  Available tools:
489
  ${rows}
490
 
491
- How to use a tool — reply with ONLY one JSON object, nothing else:
492
- {"tool":"<tool name>","args":{...}}
493
-
494
  Rules:
495
  - GUI apps, servers, and long-running programs (e.g. a Python tkinter window) MUST use the shell tool with "background":true so they keep running.
496
  - To create a Python GUI window, generate a tkinter script and run it with "python3 -c '...'" in the background.
497
  - Never invent data you can read with a tool. Keep answers short and direct.
498
  - If a command fails, diagnose and retry differently — never give up.`;
 
 
 
 
 
 
 
499
  }
500
 
501
  // ── Lifecycle ───────────────────────────────────────────────────
 
26
  }
27
 
28
  /** Extract a {tool, args} JSON call from a model reply (plain, fenced, or with prose around it). */
29
+ export function extractBalancedJson(s, start) {
30
  let depth = 0, inStr = false, esc = false;
31
  for (let i = start; i < s.length; i++) {
32
  const c = s[i];
 
43
  return null;
44
  }
45
 
46
+ /** Legacy parser: extract tool calls from a model reply. */
47
+ export function parseToolCall(text) {
48
  if (!text) return null;
49
  const bodies = [text];
50
  for (const m of text.matchAll(/```(?:json)?\s*([\s\S]*?)```/g)) bodies.push(m[1]);
 
75
  return null;
76
  }
77
 
78
+ /**
79
+ * Reasoning-protocol parser: the brain answers in one of three shapes:
80
+ * {reasoning, tool, args} → {kind:'tools', calls:[...]}
81
+ * {reasoning, answer} → {kind:'answer', answer, reasoning}
82
+ * plain text → {kind:'text', text}
83
+ * valid JSON, wrong shape → null (malformed → corrective nudge)
84
+ * prose wrapping any of those → detected via balanced-brace extraction
85
+ */
86
+ export function parseBrainReply(text) {
87
+ if (!text) return null;
88
+ const bodies = [text];
89
+ for (const m of text.matchAll(/```(?:json)?\s*([\s\S]*?)```/g)) bodies.push(m[1]);
90
+
91
+ const classify = (o) => {
92
+ if (!o || typeof o !== 'object' || Array.isArray(o)) return null;
93
+ if (Array.isArray(o.tool) && o.tool.length && o.tool.every((x) => x && typeof x.tool === 'string')) {
94
+ return { kind: 'tools', calls: o.tool };
95
+ }
96
+ if (typeof o.tool === 'string') return { kind: 'tools', calls: [o] };
97
+ if (typeof o.answer === 'string') {
98
+ return { kind: 'answer', answer: o.answer, reasoning: typeof o.reasoning === 'string' ? o.reasoning : '' };
99
+ }
100
+ return null;
101
+ };
102
+
103
+ for (const b of bodies) {
104
+ const t = b.trim();
105
+ // Whole body is JSON
106
+ try {
107
+ const o = JSON.parse(t);
108
+ if (o && typeof o === 'object' && !Array.isArray(o)) {
109
+ const c = classify(o);
110
+ if (c) return c;
111
+ return null; // valid JSON but not tool/answer shaped → malformed
112
+ }
113
+ if (Array.isArray(o)) return null; // a bare JSON array is never a valid reply
114
+ } catch { /* scan for embedded JSON */ }
115
+
116
+ // Scan for embedded tool objects
117
+ let idx = 0;
118
+ while ((idx = t.indexOf('"tool"', idx)) !== -1) {
119
+ const start = t.lastIndexOf('{', idx);
120
+ if (start !== -1) {
121
+ const json = extractBalancedJson(t, start);
122
+ if (json) {
123
+ try {
124
+ const c = classify(JSON.parse(json));
125
+ if (c) return c;
126
+ } catch { /* keep scanning */ }
127
+ }
128
+ }
129
+ idx += 5;
130
+ }
131
+ // Scan for embedded answer objects
132
+ let ai = 0;
133
+ while ((ai = t.indexOf('"answer"', ai)) !== -1) {
134
+ const start = t.lastIndexOf('{', ai);
135
+ if (start !== -1) {
136
+ const json = extractBalancedJson(t, start);
137
+ if (json) {
138
+ try {
139
+ const c = classify(JSON.parse(json));
140
+ if (c) return c;
141
+ } catch { /* keep scanning */ }
142
+ }
143
+ }
144
+ ai += 6;
145
+ }
146
+ }
147
+ return { kind: 'text', text };
148
+ }
149
+
150
  export class AgentDaemon extends EventEmitter {
151
  #creds;
152
  #control;
 
155
  #currentTask = null;
156
  #taskPoll = null;
157
  #polling = false;
158
+ // Owner-configurable reasoning profile (set from the cloud per poll).
159
+ #brain = { maxSteps: 8, temperature: 0.4, extraRules: '', verify: true };
160
 
161
  constructor(creds) {
162
  super();
 
235
  if (this.#polling) return;
236
  this.#polling = true;
237
  try {
238
+ const data = await pollTasks(this.#creds.apiKey);
239
+ this.#mergeBrain(data.brain);
240
+ const tasks = data.tasks || [];
241
  for (const t of tasks) {
242
  if (t.status !== 'pending') continue;
243
  try { await claimTask(this.#creds.apiKey, t.id); } catch { /* already claimed */ }
 
250
  }
251
  }
252
 
253
+ /** Merge the owner's brain config from the cloud (clamped, safe defaults). */
254
+ #mergeBrain(brain) {
255
+ if (!brain || typeof brain !== 'object') return;
256
+ if (Number.isFinite(+brain.maxSteps)) this.#brain.maxSteps = Math.min(16, Math.max(2, +brain.maxSteps));
257
+ if (Number.isFinite(+brain.temperature)) this.#brain.temperature = Math.min(1, Math.max(0, +brain.temperature));
258
+ if (typeof brain.extraRules === 'string') this.#brain.extraRules = brain.extraRules.slice(0, 2000);
259
+ if (typeof brain.verify === 'boolean') this.#brain.verify = brain.verify;
260
+ }
261
+
262
  // ── Task execution (cloud reasoning + local tools) ──────────────
263
  /** think() with auto-retry for transient failures — fail is never allowed. */
264
+ async #thinkWithRetry(messages, runId, opts = {}) {
265
  let lastErr = null;
266
  for (let attempt = 1; attempt <= MAX_RETRIES; attempt++) {
267
  try {
 
269
  apiKey: this.#creds.apiKey,
270
  messages,
271
  tools: tools.list(),
272
+ temperature: opts.temperature ?? this.#brain.temperature,
273
  onChunk: (delta) => {
274
  this.#control.token(delta, runId);
275
  this.emit('task:token', delta, runId);
 
374
  final = res.content || '';
375
  this.#messages.push({ role: 'assistant', content: final });
376
  } else {
377
+ // Sngine platform: agentic loop — brain reasons deeply:
378
+ // plan → act (tools) → observe → reflect → answer → verify.
379
+ const systemPrompt = this.#toolsPrompt(cloudTask);
380
  const messages = [
381
  { role: 'system', content: systemPrompt },
382
  { role: 'user', content: task },
383
  ];
384
 
385
  let corrections = 0;
386
+ for (let i = 0; i < this.#brain.maxSteps; i++) {
387
  const tThink = Date.now();
388
  const thinkRes = await this.#thinkWithRetry(messages, runId);
389
  const answer = thinkRes.text ?? thinkRes ?? '';
 
391
  if (thinkRes.usage) addUsage(thinkRes.usage);
392
  if (thinkRes.model) lastModel = thinkRes.model;
393
  if (thinkRes.provider) lastProvider = thinkRes.provider;
394
+ const reply = parseBrainReply(answer);
395
+
396
+ // Direct final answer (protocol shape)
397
+ if (reply && reply.kind === 'answer') {
398
+ final = reply.answer;
399
+ await trace('answer', {
400
+ summary: truncate(reply.reasoning || reply.answer, 500),
401
+ detail: truncate(answer, 6000),
402
+ model: thinkRes.model || '',
403
+ provider: thinkRes.provider || '',
404
+ usage: thinkRes.usage || null,
405
+ durationMs: thinkMs,
406
+ });
407
+ break;
408
+ }
409
+ // Plain-text final answer
410
+ if (reply && reply.kind === 'text') {
411
+ final = reply.text;
412
+ await trace('answer', {
413
+ summary: truncate(reply.text, 500),
414
+ detail: truncate(reply.text, 6000),
415
+ model: thinkRes.model || '',
416
+ provider: thinkRes.provider || '',
417
+ usage: thinkRes.usage || null,
418
+ durationMs: thinkMs,
419
+ });
420
+ break;
 
 
 
421
  }
 
422
 
423
+ // Tool calls
424
+ if (reply && reply.kind === 'tools') {
425
+ corrections = 0;
426
+ await trace('think', {
427
+ summary: truncate(reply.calls[0]?.reasoning || `${reply.calls.length} tool call(s)`, 500),
428
+ detail: truncate(answer, 6000),
429
+ model: thinkRes.model || '',
430
+ provider: thinkRes.provider || '',
431
+ usage: thinkRes.usage || null,
432
+ durationMs: thinkMs,
433
+ });
434
+ messages.push({ role: 'assistant', content: answer });
435
 
436
+ // Execute every requested tool (JSON arrays = multi-tool steps)
437
+ const resultLines = [];
438
+ for (const toolCall of reply.calls) {
439
+ steps.push({ type: 'tool.call', tool: toolCall.tool, args: toolCall.args });
440
+ this.#control.step('tool.call', { tool: toolCall.tool });
441
+ this.emit('tool:start', toolCall.tool, toolCall.args);
442
+ log.info(`Tool call: ${toolCall.tool}`);
443
+ await postActivity(this.#creds.apiKey, 'tool.call', { tool: toolCall.tool, args: toolCall.args }, runId, this.#creds.agentId).catch(() => {});
444
+ await trace('tool.call', { summary: toolCall.tool, detail: JSON.stringify(toolCall.args || {}, null, 2) });
445
+
446
+ const tTool = Date.now();
447
+ let result;
448
+ try {
449
+ result = await tools.run(toolCall.tool, toolCall.args || {});
450
+ } catch (err) {
451
+ result = { error: err.message };
452
+ }
453
+ this.#stats.toolCalls++;
454
+
455
+ const out = truncate(JSON.stringify(result), TOOL_OUT_MAX);
456
+ steps.push({ type: 'tool.result', tool: toolCall.tool, output: truncate(out, 400) });
457
+ await postActivity(this.#creds.apiKey, 'tool.result', { tool: toolCall.tool, output: truncate(out, 200) }, runId, this.#creds.agentId).catch(() => {});
458
+ await trace('tool.result', {
459
+ summary: truncate(out, 500),
460
+ detail: truncate(out, 4000),
461
+ durationMs: Date.now() - tTool,
462
+ });
463
+ log.info(`Tool result (${toolCall.tool}): ${truncate(out, 120)}`);
464
+
465
+ const failed = result && (result.error || result.exitCode);
466
+ if (failed) await this.#debugSnapshot(runId, `tool ${toolCall.tool} failed`);
467
+ resultLines.push(`TOOL RESULT (${toolCall.tool}):\n${out}`);
468
  }
 
 
 
 
 
 
 
 
 
 
 
469
 
470
+ // Reflect phase: force a deliberate decision — done, or next action?
471
+ const anyFailed = resultLines.some((l) => /"error"|exitCode":[1-9]/.test(l));
472
+ const debugHint = anyFailed
473
+ ? `\n\nAt least one tool returned an error. Diagnose it, then either fix the call or use a different tool/approach. Do not give up — verify your fix by running it again.`
474
+ : '';
475
+ messages.push({ role: 'user', content: resultLines.join('\n\n') + debugHint +
476
+ `\n\nREFLECT: check the tool results against the user's goal. If the goal is now satisfied, answer immediately. If something is missing or failed, state your reasoning and take the next action.` });
477
+ continue;
478
  }
479
 
480
+ // Empty or malformed reply → corrective nudge (auto-reasoning)
481
+ const hasText = answer && answer.trim();
482
+ const looksLikeAttempt = hasText && /"(tool|answer)"\s*:/.test(answer);
483
+ if (corrections >= MAX_CORRECTIONS) {
484
+ final = hasText ? answer : 'The brain produced no usable reply. Check the activity feed for the trace.';
485
+ break;
486
+ }
487
+ corrections++;
488
+ const hint = looksLikeAttempt
489
+ ? 'Your last message was not valid JSON. Reply with ONLY one JSON object: {"reasoning":"...","tool":"<tool name>","args":{...}} — or {"reasoning":"...","answer":"..."} when done, or plain text.'
490
+ : 'Your reply was empty or not actionable. Either give the final answer in plain text, or emit ONE JSON object with "tool" or "answer".';
491
+ messages.push({ role: 'assistant', content: answer });
492
+ messages.push({ role: 'user', content: hint });
493
+ await trace('correct', { summary: looksLikeAttempt ? 'malformed reply — corrective nudge' : 'empty reply — corrective nudge' });
494
+ await postActivity(this.#creds.apiKey, 'auto.correct', { reason: looksLikeAttempt ? 'malformed' : 'empty', attempt: corrections }, runId, this.#creds.agentId).catch(() => {});
495
  }
496
 
497
+ // Never give up: one forced conclusion when steps run out
498
  if (!final) {
499
+ messages.push({ role: 'user', content: 'Step limit reached. Reply {"reasoning":"brief summary of what you did","answer":"..."} — or plain text. No more tools.' });
500
  try {
501
  const tThink = Date.now();
502
  const thinkRes = await this.#thinkWithRetry(messages, runId);
503
+ const r2 = parseBrainReply(thinkRes.text ?? '');
504
+ if (r2 && r2.kind === 'answer') final = r2.answer;
505
+ else if (r2 && r2.kind === 'text') final = r2.text;
506
+ else final = (thinkRes.text ?? '').trim() || 'The agent hit its step limit. See the activity feed for the full execution trace.';
507
  if (thinkRes.usage) addUsage(thinkRes.usage);
508
  if (thinkRes.model) lastModel = thinkRes.model;
509
  if (thinkRes.provider) lastProvider = thinkRes.provider;
 
519
  final = 'The agent hit its step limit. See the activity feed for the full execution trace.';
520
  }
521
  }
522
+
523
+ // Self-verification pass: the brain re-checks its own answer against
524
+ // the evidence before it reaches the user (fixes premature or sloppy answers).
525
+ if (final && this.#brain.verify) {
526
+ try {
527
+ messages.push({ role: 'assistant', content: final });
528
+ messages.push({ role: 'user', content: 'VERIFY: You are about to send this answer to the user. Check it against the tool results above: is every claim factual, complete and direct? If something is wrong or missing, fix it. Reply {"reasoning":"what you checked","answer":"<corrected or unchanged answer>"}.' });
529
+ const tThink = Date.now();
530
+ const vRes = await this.#thinkWithRetry(messages, runId);
531
+ const vr = parseBrainReply(vRes.text ?? '');
532
+ if (vr && vr.kind === 'answer' && vr.answer.trim()) final = vr.answer;
533
+ else if (vr && vr.kind === 'text' && (vRes.text ?? '').trim()) final = vRes.text;
534
+ if (vRes.usage) addUsage(vRes.usage);
535
+ if (vRes.model) lastModel = vRes.model;
536
+ if (vRes.provider) lastProvider = vRes.provider;
537
+ await trace('verify', {
538
+ summary: truncate((vr && vr.reasoning) || 'answer re-checked against tool results', 500),
539
+ detail: truncate(vRes.text ?? '', 6000),
540
+ model: vRes.model || '',
541
+ provider: vRes.provider || '',
542
+ usage: vRes.usage || null,
543
+ durationMs: Date.now() - tThink,
544
+ });
545
+ } catch {
546
+ // verification is best-effort — keep the original answer
547
+ }
548
+ }
549
  }
550
  } catch (err) {
551
  this.#stats.errors++;
 
623
  }
624
  }
625
 
626
+ #toolsPrompt(taskRow) {
627
  const rows = tools.list()
628
  .map((t) => `- ${t.name}: ${t.description}${t.args ? ` (args: ${JSON.stringify(t.args)})` : ''}`)
629
  .join('\n');
630
+ const brain = this.#brain;
631
+ let p = `You are mona-agent — the AI agent controlling this device (${process.platform}). You reason deeply and act precisely: plan, act, observe, reflect, then answer.
632
+
633
+ ## Reasoning protocol
634
+ Think before you act: what does the user actually want, what do you already know, what do you still need, and what is the safest way to get it.
635
+ - When you need information or need to change something on this device, reply with ONLY one JSON object:
636
+ {"reasoning":"<your concise thinking: goal, what you know, what you plan and why>","tool":"<tool name>","args":{...}}
637
+ - When you have everything you need, reply with ONLY:
638
+ {"reasoning":"<why the goal is now satisfied>","answer":"<the final answer for the user>"}
639
+ - Plain text is also accepted as a final answer. Never mix prose with JSON.
640
+
641
+ ## Answer quality
642
+ - Base every claim on actual tool results — never invent data you could have read.
643
+ - Be direct and concise. State what you did and what you found.
644
+ - If something failed, say what failed and what you tried instead.
645
+ - If the goal is already satisfied, stop and answer instead of calling more tools.
646
+
647
+ ## Memory
648
+ You have a persistent memory tool. Read it at the start of relevant tasks, and save user preferences and important facts so they survive across tasks.
649
 
650
  Available tools:
651
  ${rows}
652
 
 
 
 
653
  Rules:
654
  - GUI apps, servers, and long-running programs (e.g. a Python tkinter window) MUST use the shell tool with "background":true so they keep running.
655
  - To create a Python GUI window, generate a tkinter script and run it with "python3 -c '...'" in the background.
656
  - Never invent data you can read with a tool. Keep answers short and direct.
657
  - If a command fails, diagnose and retry differently — never give up.`;
658
+ if (taskRow?.system_prompt) {
659
+ p += `\n\n## Your role (set by the owner)\n${taskRow.system_prompt}`;
660
+ }
661
+ if (brain.extraRules) {
662
+ p += `\n\n## Owner's rules (always follow)\n${brain.extraRules}`;
663
+ }
664
+ return p;
665
  }
666
 
667
  // ── Lifecycle ───────────────────────────────────────────────────
apps/desktop/src/cloud.js CHANGED
@@ -57,10 +57,10 @@ function b64Body(obj) {
57
  // onUsage(usage) — called with final token counts (if provided)
58
  // Returns { text, usage, model, provider } — usage is null when the
59
  // cloud did not report it (older server or plain JSON without usage).
60
- export async function think({ apiKey, messages, tools, onChunk, onUsage, signal }) {
61
  const res = await apiFetch(P.think, {
62
  apiKey,
63
- body: b64Body({ messages, tools, stream: true }),
64
  signal,
65
  });
66
 
@@ -131,10 +131,12 @@ export async function reportToolResult(apiKey, agentId, tool, result) {
131
  }
132
 
133
  // ── Cloud task queue (sngine platform — device polls for work) ────
 
 
134
  export async function pollTasks(apiKey) {
135
  const res = await apiFetch('/api/v1/agent/tasks', { apiKey, method: 'GET' });
136
  const data = await res.json();
137
- return data?.tasks || [];
138
  }
139
 
140
  export async function claimTask(apiKey, id) {
 
57
  // onUsage(usage) — called with final token counts (if provided)
58
  // Returns { text, usage, model, provider } — usage is null when the
59
  // cloud did not report it (older server or plain JSON without usage).
60
+ export async function think({ apiKey, messages, tools, onChunk, onUsage, signal, temperature }) {
61
  const res = await apiFetch(P.think, {
62
  apiKey,
63
+ body: b64Body({ messages, tools, stream: true, temperature }),
64
  signal,
65
  });
66
 
 
131
  }
132
 
133
  // ── Cloud task queue (sngine platform — device polls for work) ────
134
+ // The response carries the task rows plus the owner's brain config
135
+ // (step budget, temperature, extra rules) so the loop can tune itself.
136
  export async function pollTasks(apiKey) {
137
  const res = await apiFetch('/api/v1/agent/tasks', { apiKey, method: 'GET' });
138
  const data = await res.json();
139
+ return data || { tasks: [] };
140
  }
141
 
142
  export async function claimTask(apiKey, id) {
apps/desktop/test/agent.test.mjs CHANGED
@@ -129,6 +129,66 @@ describe('tools/registry', () => {
129
  });
130
  });
131
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
132
  // ── Config tests ──────────────────────────────────────────────────
133
 
134
  describe('config', () => {
 
129
  });
130
  });
131
 
132
+ // ── Brain reply parser (reasoning protocol) ──────────────────────
133
+
134
+ describe('brain reply parser', () => {
135
+ let parseBrainReply, parseToolCall;
136
+ before(async () => {
137
+ ({ parseBrainReply, parseToolCall } = await import('../src/agent.js'));
138
+ });
139
+
140
+ it('parses a plain tool call (legacy)', () => {
141
+ const calls = parseToolCall('{"tool":"shell","args":{"cmd":"uptime"}}');
142
+ assert.equal(calls.length, 1);
143
+ assert.equal(calls[0].tool, 'shell');
144
+ });
145
+
146
+ it('parses fenced + prose-wrapped tool calls (legacy)', () => {
147
+ assert.equal(parseToolCall('Sure!\n```json\n{"tool":"sysinfo","args":{}}\n```').length, 1);
148
+ assert.equal(parseToolCall('I will run: {"tool":"shell","args":{"cmd":"df -h"}} now').length, 1);
149
+ });
150
+
151
+ it('parses multi-tool arrays (legacy)', () => {
152
+ const calls = parseToolCall('[{"tool":"sysinfo","args":{}},{"tool":"shell","args":{"cmd":"uptime"}}]');
153
+ assert.equal(calls.length, 2);
154
+ });
155
+
156
+ it('parses the reasoning protocol: tool with reasoning', () => {
157
+ const r = parseBrainReply('{"reasoning":"Need disk state first","tool":"shell","args":{"cmd":"df -h"}}');
158
+ assert.equal(r.kind, 'tools');
159
+ assert.equal(r.calls[0].tool, 'shell');
160
+ assert.equal(r.calls[0].reasoning, 'Need disk state first');
161
+ });
162
+
163
+ it('parses the reasoning protocol: final answer', () => {
164
+ const r = parseBrainReply('{"reasoning":"All facts collected","answer":"Disk is 40% full."}');
165
+ assert.equal(r.kind, 'answer');
166
+ assert.equal(r.answer, 'Disk is 40% full.');
167
+ });
168
+
169
+ it('treats plain text as a final answer', () => {
170
+ const r = parseBrainReply('Everything looks good.');
171
+ assert.equal(r.kind, 'text');
172
+ assert.equal(r.text, 'Everything looks good.');
173
+ });
174
+
175
+ it('rejects valid JSON with the wrong shape (→ corrective nudge)', () => {
176
+ assert.equal(parseBrainReply('{"foo":123}'), null);
177
+ assert.equal(parseBrainReply('[]'), null);
178
+ });
179
+
180
+ it('finds embedded answer objects in prose', () => {
181
+ const r = parseBrainReply('Done. Here you go: {"reasoning":"verified","answer":"Hi Mona"}.');
182
+ assert.equal(r.kind, 'answer');
183
+ assert.equal(r.answer, 'Hi Mona');
184
+ });
185
+
186
+ it('handles empty input', () => {
187
+ assert.equal(parseBrainReply(''), null);
188
+ assert.equal(parseBrainReply(null), null);
189
+ });
190
+ });
191
+
192
  // ── Config tests ──────────────────────────────────────────────────
193
 
194
  describe('config', () => {