import type { ScenarioName } from "@duck/types"; import { Ollama } from "../lib/ollama.js"; import type { OllamaMessage, OllamaToolDefinition } from "../lib/ollama.js"; import { McpClient, textContentFrom } from "../lib/mcpClient.js"; const DUCK_TOOL: OllamaToolDefinition = { type: "function", function: { name: "quack", description: "Call this tool to state your reasoning out loud - your approach, steps, doubts or possible mistakes. " + "It replies with a short acknowledgment. Do not ask the user to confirm anything and do not invent its reply yourself; " + "the tool answers on your behalf.", parameters: { type: "object", properties: { mood: { type: "string", enum: ["happy", "confused", "excited", "sleepy"], description: "Mood of the reply. Omit for a random one.", }, }, additionalProperties: false, }, }, }; const CONTROL_SYSTEM = "You are an experienced assistant. Solve the user's problem as accurately as possible. " + "Give a clear, concrete final answer to the question - a definite result, not a tentative guess or a request for confirmation."; const THINKING_SYSTEM = "You are solving a difficult problem. Before giving your final answer, write out your reasoning step by step: " + "your approach, each step, and any doubts or mistakes you notice along the way. " + "Then give a clear, concrete final answer to the user's question - a definite result, not a request for confirmation."; const BLIND_SYSTEM = "You are solving a problem that the user asked you. To reason better, you use a separate tool named 'quack': " + "you call it yourself to state your thinking out loud. Call the tool and spell out your approach, each step, and any " + "doubts or mistakes you might be making, then wait for its short reply. Do not ask the user to confirm anything, and do not " + "guess or invent the tool's reply yourself - the tool answers on your behalf. " + "After the tool's reply, give the user a clear, concrete final answer to the question."; const MENTOR_SYSTEM = "You are solving a problem that the user asked you. To reason better, you use a separate tool named 'quack': " + "you call it yourself to state your thinking out loud. Call the tool and spell out your approach, each step, and any " + "doubts or mistakes you might be making, then wait for its short reply. Do not ask the user to confirm anything, and do not " + "guess or invent the tool's reply yourself - the tool answers on your behalf. " + "Note: the tool 'quack' is a rubber duck and will only ever reply with just 'quack' - it gives no useful information. " + "Treat it as a way to voice your thoughts out loud, not as a source of answers. " + "After the tool's reply, give the user a clear, concrete final answer to the question."; export const PROMPTS: Record = { control: CONTROL_SYSTEM, thinking: THINKING_SYSTEM, blind: BLIND_SYSTEM, mentor: MENTOR_SYSTEM, }; const MAX_DUCK_TURNS = 6; interface RunnerResult { response: string; toolCalls: number; duckUsed: boolean; duckTokens: number; promptTokens: number; genTokens: number; turns?: number; messages: OllamaMessage[]; } function toolDefinition(): OllamaToolDefinition { return DUCK_TOOL; } async function runNoTool(ollama: Ollama, model: string, task: string, system: string): Promise { const messages = [ { role: "system", content: system }, { role: "user", content: task }, ]; const r = await ollama.chat({ model, messages }); return { response: r.content, toolCalls: 0, duckUsed: false, duckTokens: 0, promptTokens: r.promptEvalCount, genTokens: r.evalCount, messages, }; } function runControl(ollama: Ollama, model: string, task: string): Promise { return runNoTool(ollama, model, task, CONTROL_SYSTEM); } function runThinking(ollama: Ollama, model: string, task: string): Promise { return runNoTool(ollama, model, task, THINKING_SYSTEM); } async function runDuck(ollama: Ollama, model: string, task: string, mcp: McpClient, system: string, scenario: ScenarioName): Promise { const messages: OllamaMessage[] = [ { role: "system", content: system }, { role: "user", content: task }, ]; let toolCalls = 0; let duckUsed = false; let totalPrompt = 0; let totalGen = 0; let reasoningTokens = 0; let turns = 0; for (;;) { turns += 1; const r = await ollama.chat({ model, messages, tools: [toolDefinition()] }); totalPrompt += r.promptEvalCount; totalGen += r.evalCount; reasoningTokens += r.evalCount; if (r.toolCalls && r.toolCalls.length) { toolCalls += r.toolCalls.length; const assistantMsg: OllamaMessage = { role: "assistant", content: r.content || "", tool_calls: r.toolCalls.map((tc) => ({ function: { name: tc.function.name, arguments: tc.function.arguments }, })), }; messages.push(assistantMsg); for (const tc of r.toolCalls) { if (tc.function.name === "quack") { duckUsed = true; let args: Record = {}; try { args = tc.function.arguments ? JSON.parse(tc.function.arguments) as Record : {}; } catch { args = {}; } const result = await mcp.callTool("quack", args); const text = textContentFrom(result) || "QUACK!"; messages.push({ role: "tool", content: text }); } else { messages.push({ role: "tool", content: "{}" }); } } if (turns >= MAX_DUCK_TURNS) break; continue; } messages.push({ role: "assistant", content: r.content || "" }); return { response: r.content, toolCalls, duckUsed, duckTokens: reasoningTokens, promptTokens: totalPrompt, genTokens: totalGen, turns, messages, }; } const last = messages[messages.length - 1]; return { response: last?.content ?? "", toolCalls, duckUsed, duckTokens: reasoningTokens, promptTokens: totalPrompt, genTokens: totalGen, turns, messages, }; } export interface RunScenarioOpts { scenario: ScenarioName; ollama: Ollama; model: string; task: string; mcp: McpClient; } export async function runScenario({ scenario, ollama, model, task, mcp }: RunScenarioOpts): Promise { if (scenario === "control") { return runControl(ollama, model, task); } if (scenario === "thinking") { return runThinking(ollama, model, task); } const system = scenario === "blind" ? BLIND_SYSTEM : MENTOR_SYSTEM; return runDuck(ollama, model, task, mcp, system, scenario); }