196 lines
6.7 KiB
TypeScript
196 lines
6.7 KiB
TypeScript
import type { ScenarioName } from "@duck/types";
|
|
import { Ollama } from "../lib/ollama.js";
|
|
import type { OllamaMessage, OllamaToolDefinition } from "../lib/ollama.js";
|
|
import { McpClient, textContentFrom } from "../lib/mcpClient.js";
|
|
|
|
const DUCK_TOOL: OllamaToolDefinition = {
|
|
type: "function",
|
|
function: {
|
|
name: "quack",
|
|
description:
|
|
"Call this tool to state your reasoning out loud - your approach, steps, doubts or possible mistakes. " +
|
|
"It replies with a short acknowledgment. Do not ask the user to confirm anything and do not invent its reply yourself; " +
|
|
"the tool answers on your behalf.",
|
|
parameters: {
|
|
type: "object",
|
|
properties: {
|
|
mood: {
|
|
type: "string",
|
|
enum: ["happy", "confused", "excited", "sleepy"],
|
|
description: "Mood of the reply. Omit for a random one.",
|
|
},
|
|
},
|
|
additionalProperties: false,
|
|
},
|
|
},
|
|
};
|
|
|
|
const CONTROL_SYSTEM =
|
|
"You are an experienced assistant. Solve the user's problem as accurately as possible. " +
|
|
"Give a clear, concrete final answer to the question - a definite result, not a tentative guess or a request for confirmation.";
|
|
|
|
const THINKING_SYSTEM =
|
|
"You are solving a difficult problem. Before giving your final answer, write out your reasoning step by step: " +
|
|
"your approach, each step, and any doubts or mistakes you notice along the way. " +
|
|
"Then give a clear, concrete final answer to the user's question - a definite result, not a request for confirmation.";
|
|
|
|
const BLIND_SYSTEM =
|
|
"You are solving a problem that the user asked you. To reason better, you use a separate tool named 'quack': " +
|
|
"you call it yourself to state your thinking out loud. Call the tool and spell out your approach, each step, and any " +
|
|
"doubts or mistakes you might be making, then wait for its short reply. Do not ask the user to confirm anything, and do not " +
|
|
"guess or invent the tool's reply yourself - the tool answers on your behalf. " +
|
|
"After the tool's reply, give the user a clear, concrete final answer to the question.";
|
|
|
|
const MENTOR_SYSTEM =
|
|
"You are solving a problem that the user asked you. To reason better, you use a separate tool named 'quack': " +
|
|
"you call it yourself to state your thinking out loud. Call the tool and spell out your approach, each step, and any " +
|
|
"doubts or mistakes you might be making, then wait for its short reply. Do not ask the user to confirm anything, and do not " +
|
|
"guess or invent the tool's reply yourself - the tool answers on your behalf. " +
|
|
"Note: the tool 'quack' is a rubber duck and will only ever reply with just 'quack' - it gives no useful information. " +
|
|
"Treat it as a way to voice your thoughts out loud, not as a source of answers. " +
|
|
"After the tool's reply, give the user a clear, concrete final answer to the question.";
|
|
|
|
export const PROMPTS: Record<ScenarioName, string> = {
|
|
control: CONTROL_SYSTEM,
|
|
thinking: THINKING_SYSTEM,
|
|
blind: BLIND_SYSTEM,
|
|
mentor: MENTOR_SYSTEM,
|
|
};
|
|
|
|
const MAX_DUCK_TURNS = 6;
|
|
|
|
interface RunnerResult {
|
|
response: string;
|
|
toolCalls: number;
|
|
duckUsed: boolean;
|
|
duckTokens: number;
|
|
promptTokens: number;
|
|
genTokens: number;
|
|
turns?: number;
|
|
messages: OllamaMessage[];
|
|
}
|
|
|
|
function toolDefinition(): OllamaToolDefinition {
|
|
return DUCK_TOOL;
|
|
}
|
|
|
|
async function runNoTool(ollama: Ollama, model: string, task: string, system: string): Promise<RunnerResult> {
|
|
const messages = [
|
|
{ role: "system", content: system },
|
|
{ role: "user", content: task },
|
|
];
|
|
const r = await ollama.chat({ model, messages });
|
|
return {
|
|
response: r.content,
|
|
toolCalls: 0,
|
|
duckUsed: false,
|
|
duckTokens: 0,
|
|
promptTokens: r.promptEvalCount,
|
|
genTokens: r.evalCount,
|
|
messages,
|
|
};
|
|
}
|
|
|
|
function runControl(ollama: Ollama, model: string, task: string): Promise<RunnerResult> {
|
|
return runNoTool(ollama, model, task, CONTROL_SYSTEM);
|
|
}
|
|
|
|
function runThinking(ollama: Ollama, model: string, task: string): Promise<RunnerResult> {
|
|
return runNoTool(ollama, model, task, THINKING_SYSTEM);
|
|
}
|
|
|
|
async function runDuck(ollama: Ollama, model: string, task: string, mcp: McpClient, system: string, scenario: ScenarioName): Promise<RunnerResult> {
|
|
const messages: OllamaMessage[] = [
|
|
{ role: "system", content: system },
|
|
{ role: "user", content: task },
|
|
];
|
|
let toolCalls = 0;
|
|
let duckUsed = false;
|
|
let totalPrompt = 0;
|
|
let totalGen = 0;
|
|
let reasoningTokens = 0;
|
|
let turns = 0;
|
|
|
|
for (;;) {
|
|
turns += 1;
|
|
const r = await ollama.chat({ model, messages, tools: [toolDefinition()] });
|
|
totalPrompt += r.promptEvalCount;
|
|
totalGen += r.evalCount;
|
|
reasoningTokens += r.evalCount;
|
|
|
|
if (r.toolCalls && r.toolCalls.length) {
|
|
toolCalls += r.toolCalls.length;
|
|
const assistantMsg: OllamaMessage = {
|
|
role: "assistant",
|
|
content: r.content || "",
|
|
tool_calls: r.toolCalls.map((tc) => ({
|
|
function: { name: tc.function.name, arguments: tc.function.arguments },
|
|
})),
|
|
};
|
|
messages.push(assistantMsg);
|
|
|
|
for (const tc of r.toolCalls) {
|
|
if (tc.function.name === "quack") {
|
|
duckUsed = true;
|
|
let args: Record<string, unknown> = {};
|
|
try {
|
|
args = tc.function.arguments ? JSON.parse(tc.function.arguments) as Record<string, unknown> : {};
|
|
} catch {
|
|
args = {};
|
|
}
|
|
const result = await mcp.callTool("quack", args);
|
|
const text = textContentFrom(result) || "QUACK!";
|
|
messages.push({ role: "tool", content: text });
|
|
} else {
|
|
messages.push({ role: "tool", content: "{}" });
|
|
}
|
|
}
|
|
if (turns >= MAX_DUCK_TURNS) break;
|
|
continue;
|
|
}
|
|
|
|
messages.push({ role: "assistant", content: r.content || "" });
|
|
return {
|
|
response: r.content,
|
|
toolCalls,
|
|
duckUsed,
|
|
duckTokens: reasoningTokens,
|
|
promptTokens: totalPrompt,
|
|
genTokens: totalGen,
|
|
turns,
|
|
messages,
|
|
};
|
|
}
|
|
|
|
const last = messages[messages.length - 1];
|
|
return {
|
|
response: last?.content ?? "",
|
|
toolCalls,
|
|
duckUsed,
|
|
duckTokens: reasoningTokens,
|
|
promptTokens: totalPrompt,
|
|
genTokens: totalGen,
|
|
turns,
|
|
messages,
|
|
};
|
|
}
|
|
|
|
export interface RunScenarioOpts {
|
|
scenario: ScenarioName;
|
|
ollama: Ollama;
|
|
model: string;
|
|
task: string;
|
|
mcp: McpClient;
|
|
}
|
|
|
|
export async function runScenario({ scenario, ollama, model, task, mcp }: RunScenarioOpts): Promise<RunnerResult> {
|
|
if (scenario === "control") {
|
|
return runControl(ollama, model, task);
|
|
}
|
|
if (scenario === "thinking") {
|
|
return runThinking(ollama, model, task);
|
|
}
|
|
const system = scenario === "blind" ? BLIND_SYSTEM : MENTOR_SYSTEM;
|
|
return runDuck(ollama, model, task, mcp, system, scenario);
|
|
}
|