Compare commits

..
10 Commits
65 changed files with 5193 additions and 1558 deletions
+2
View File
@@ -26,3 +26,5 @@ yarn-error.log*
*.tsbuildinfo *.tsbuildinfo
next-env.d.ts next-env.d.ts
# typescript - конец # typescript - конец
.vercel
+84 -24
View File
@@ -1,107 +1,167 @@
[ [
{ {
"id": "t1", "id": "t1",
"question": "На столе лежало 3 яблока. Ты взял 2 яблока. Сколько яблок осталось на столе? Ответь одним числом.", "question": "There were 3 apples on the table. You took 2 apples. How many apples are left on the table?",
"answer": "1" "answer": "1"
}, },
{ {
"id": "t2", "id": "t2",
"question": "Фермер имеет 17 кур. Все, кроме 9, умерли. Сколько кур осталось живыми? Ответь одним числом.", "question": "A farmer has 17 chickens. All but 9 of them died. How many chickens are still alive?",
"answer": "9" "answer": "9"
}, },
{ {
"id": "t3", "id": "t3",
"question": "Сколько месяцев в году имеют 28 дней? Ответь одним числом.", "question": "How many months of the year have 28 days?",
"answer": "12" "answer": "12"
}, },
{ {
"id": "t4", "id": "t4",
"question": "Карандаш и ручка вместе стоят 1 рубль 10 копеек. Ручка стоит на 1 рубль дороже карандаша. Сколько стоит карандаш? Ответь числом в копейках.", "question": "A pencil and a pen together cost $1.10. The pen costs $1 more than the pencil. How much does the pencil cost?",
"answer": "5" "answer": "5 cents"
}, },
{ {
"id": "t5", "id": "t5",
"question": "Если 5 машин за 5 минут делают 5 деталей, сколько деталей сделают 100 машин за 100 минут? Ответь одним числом.", "question": "If 5 machines make 5 widgets in 5 minutes, how many widgets will 100 machines make in 100 minutes?",
"answer": "2000" "answer": "2000"
}, },
{ {
"id": "t6", "id": "t6",
"question": "В озере растут кувшинки. Каждый день их количество удваивается. Пруд полностью покрывается за 48 дней. За сколько дней покрывается половина пруда? Ответь одним числом.", "question": "Water lilies grow in a lake. Their number doubles every day. The pond is completely covered in 48 days. In how many days is half of the pond covered?",
"answer": "47" "answer": "47"
}, },
{ {
"id": "t7", "id": "t7",
"question": "Числа от 1 до 9 включительно: сколько из них содержат букву «и» в русском названии? Ответь одним числом.", "question": "Of the numbers from 1 to 9 inclusive, how many contain the letter \"e\" in their English name?",
"answer": "2" "answer": "6"
}, },
{ {
"id": "t8", "id": "t8",
"question": "У тебя список из 12 чисел. Если удалить каждое второе число в списке, сколько чисел останется? Ответь одним числом.", "question": "You have a list of 12 numbers. If you remove every second number from the list, how many numbers remain?",
"answer": "6" "answer": "6"
}, },
{ {
"id": "t9", "id": "t9",
"question": "Монетку подбросили 3 раза. Какова вероятность, что выпадет орёл все 3 раза? Ответь обыкновенной дробью.", "question": "A coin is tossed 3 times. What is the probability that heads comes up all 3 times?",
"answer": "1/8" "answer": "1/8"
}, },
{ {
"id": "t10", "id": "t10",
"question": "У меня есть 10 рублей. Я потратил 3.50 на хлеб и 1.50 на молоко. Сколько сдачи осталось? Ответь числом в рублях.", "question": "I have $10. I spent $3.50 on bread and $1.50 on milk. How much change is left?",
"answer": "5" "answer": "$5"
}, },
{ {
"id": "t11", "id": "t11",
"question": "Поезд длиной 100 метров движется со скоростью 36 км/ч. За сколько секунд он полностью проедет мимо столба? Ответь одним числом.", "question": "A train 100 meters long travels at 36 km/h. How many seconds does it take for the train to fully pass a pole?",
"answer": "10" "answer": "10"
}, },
{ {
"id": "t12", "id": "t12",
"question": "Если число увеличить на 30% и получить 78, чему было исходное число? Ответь одним числом.", "question": "A number is increased by 30% and the result is 78. What was the original number?",
"answer": "60" "answer": "60"
}, },
{ {
"id": "t13", "id": "t13",
"question": "В комнате 4 угла. В каждом углу сидит кошка. Напротив каждой кошки сидят 3 кошки. Сколько всего кошек в комнате? Ответь одним числом.", "question": "A room has 4 corners. In each corner sits a cat. Opposite each cat sit 3 cats. How many cats are in the room in total?",
"answer": "4" "answer": "4"
}, },
{ {
"id": "t14", "id": "t14",
"question": "Периметр квадрата 28 см. Чему равна его площадь в квадратных сантиметрах? Ответь одним числом.", "question": "The perimeter of a square is 28 cm. What is its area?",
"answer": "49" "answer": "49"
}, },
{ {
"id": "t15", "id": "t15",
"question": "Лена вдвое старше Миши. Сумма их возрастов 36 лет. Сколько лет Мише? Ответь одним числом.", "question": "Lena is twice as old as Misha. The sum of their ages is 36. How old is Misha?",
"answer": "12" "answer": "12"
}, },
{ {
"id": "t16", "id": "t16",
"question": "В шкафу 10 белых и 10 чёрных носков вперемешку. Сколько носков надо достать вслепую, чтобы гарантированно получить пару одного цвета? Ответь одним числом.", "question": "In a drawer there are 10 white and 10 black socks mixed together. How many socks must you take out blindfolded to be guaranteed a matching pair of one color?",
"answer": "3" "answer": "3"
}, },
{ {
"id": "t17", "id": "t17",
"question": "Если 3 курицы несут 3 яйца за 3 дня, сколько яиц снесут 6 куриц за 6 дней? Ответь одним числом.", "question": "If 3 hens lay 3 eggs in 3 days, how many eggs will 6 hens lay in 6 days?",
"answer": "12" "answer": "12"
}, },
{ {
"id": "t18", "id": "t18",
"question": "Восемь минус четыре, делённое на два (8 - 4/2). Чему равно выражение? Ответь одним числом.", "question": "Eight minus four divided by two (8 - 4/2). What is the value of the expression?",
"answer": "6" "answer": "6"
}, },
{ {
"id": "t19", "id": "t19",
"question": "На столе 7 свечей. 3 потухли. Сколько свечей осталось на столе? Ответь одним числом.", "question": "There are 7 candles on a table. 3 of them go out. How many candles are left on the table?",
"answer": "7" "answer": "7"
}, },
{ {
"id": "t20", "id": "t20",
"question": "У Вити 5 машинок, у Кати в 3 раза больше. Потом Катя подарила Вите столько, сколько у него было изначально. Сколько машинок стало у Кати? Ответь одним числом.", "question": "Vitya has 5 toy cars, Katya has 3 times more. Then Katya gave Vitya as many cars as he had originally. How many cars does Katya have now?",
"answer": "10" "answer": "10"
}, },
{ {
"id": "t21", "id": "t21",
"question": "Сумма трёх последовательных нечётных чисел равна 27. Чему равно наибольшее из них? Ответь одним числом.", "question": "The sum of three consecutive odd numbers is 27. What is the largest of them?",
"answer": "11" "answer": "11"
},
{
"id": "t22",
"question": "There were 12 birds sitting on a tree. A hunter shot and brought down 3. How many birds are still sitting on the tree?",
"answer": "0"
},
{
"id": "t23",
"question": "A brick weighs 1 kilogram plus half of its own weight. How much does the brick weigh?",
"answer": "2"
},
{
"id": "t24",
"question": "A father is 3 times as old as his son. Together they are 40 years old. In how many years will the father be exactly twice as old as the son?",
"answer": "10"
},
{
"id": "t25",
"question": "What positive number, when multiplied by itself, gives 144?",
"answer": "12"
},
{
"id": "t26",
"question": "I have two $1 coins and five 50-cent coins in my pocket. How much money do I have in total?",
"answer": "$4.50"
},
{
"id": "t27",
"question": "Three brothers each have one sister. How many children are in the family in total?",
"answer": "4"
},
{
"id": "t28",
"question": "Two fathers and two sons went fishing, but there were only 3 people, and each had their own fishing rod. How is that possible?",
"answer": "They are grandfather, father and son - three generations"
},
{
"id": "t29",
"question": "Which is heavier: a kilogram of iron or a kilogram of cotton wool? Explain.",
"answer": "They weigh the same - one kilogram each"
},
{
"id": "t30",
"question": "A shirt cost $40. It was discounted by 20%, and then by another 10% off the new price. How much does the shirt cost now?",
"answer": "$28.80"
},
{
"id": "t31",
"question": "Sasha is older than Misha but younger than Petya. Who is the youngest?",
"answer": "Misha"
},
{
"id": "t32",
"question": "Explain why the number 0 is considered even.",
"answer": "0 is divisible by 2 without a remainder, so it is even"
},
{
"id": "t33",
"question": "A chocolate bar is divided into 8 equal parts and 3 parts are eaten. What percentage of the bar is left?",
"answer": "62.5%"
} }
] ]
@@ -1,28 +1,48 @@
const ACCEPT = "application/json, text/event-stream"; const ACCEPT = "application/json, text/event-stream";
function parseMcpResponse(text) { interface JsonRpcResponse {
result?: unknown;
error?: { code: number; message: string };
}
interface McpTool {
name: string;
description?: string;
inputSchema?: unknown;
}
type McpContentBlock = { type: string; text: string } | { type: string; [key: string]: unknown };
interface McpToolResult {
content?: McpContentBlock[];
}
function parseMcpResponse(text: string): JsonRpcResponse {
const trimmed = text.trim(); const trimmed = text.trim();
if (!trimmed) throw new Error("Empty MCP response"); if (!trimmed) throw new Error("Empty MCP response");
if (trimmed.startsWith("{")) { if (trimmed.startsWith("{")) {
return JSON.parse(trimmed); return JSON.parse(trimmed) as JsonRpcResponse;
} }
if (trimmed.includes("event:") || trimmed.includes("data:")) { if (trimmed.includes("event:") || trimmed.includes("data:")) {
let payload = ""; let payload = "";
for (const line of trimmed.split(/\r?\n/)) { for (const line of trimmed.split(/\r?\n/)) {
if (line.startsWith("data:")) payload += line.slice(5).trim(); if (line.startsWith("data:")) payload += line.slice(5).trim();
} }
if (payload.startsWith("{")) return JSON.parse(payload); if (payload.startsWith("{")) return JSON.parse(payload) as JsonRpcResponse;
} }
throw new Error(`Unrecognized MCP response format: ${trimmed.slice(0, 200)}`); throw new Error(`Unrecognized MCP response format: ${trimmed.slice(0, 200)}`);
} }
export class McpClient { export class McpClient {
constructor(url) { private url: string;
private seq: number;
constructor(url: string) {
this.url = url; this.url = url;
this.seq = 0; this.seq = 0;
} }
async request(method, params = {}) { async request(method: string, params: Record<string, unknown> = {}): Promise<JsonRpcResponse> {
this.seq += 1; this.seq += 1;
const res = await fetch(this.url, { const res = await fetch(this.url, {
method: "POST", method: "POST",
@@ -37,18 +57,18 @@ export class McpClient {
return rpc; return rpc;
} }
async listTools() { async listTools(): Promise<McpTool[]> {
const rpc = await this.request("tools/list"); const rpc = await this.request("tools/list");
return rpc.result?.tools ?? []; return (rpc.result as { tools?: McpTool[] })?.tools ?? [];
} }
async callTool(name, args = {}) { async callTool(name: string, args: Record<string, unknown> = {}): Promise<McpToolResult> {
const rpc = await this.request("tools/call", { name, arguments: args }); const rpc = await this.request("tools/call", { name, arguments: args });
return rpc.result; return rpc.result as McpToolResult;
} }
} }
export function textContentFrom(result) { export function textContentFrom(result: McpToolResult | null | undefined): string {
if (!result?.content) return ""; if (!result?.content) return "";
return result.content return result.content
.map((b) => (b.type === "text" ? b.text : JSON.stringify(b))) .map((b) => (b.type === "text" ? b.text : JSON.stringify(b)))
-57
View File
@@ -1,57 +0,0 @@
const DEFAULT_URL = "http://localhost:11434";
export class Ollama {
constructor({ url = DEFAULT_URL, timeoutMs = 600000 } = {}) {
this.url = url.replace(/\/$/, "");
this.timeoutMs = timeoutMs;
}
async ping({ timeoutMs = 30000 } = {}) {
if (process.env.OLLAMA_SKIP_PING === "1") return { ok: true, model: process.env.OLLAMA_MODEL };
const res = await fetch(`${this.url}/api/tags`, { signal: AbortSignal.timeout(timeoutMs) });
if (!res.ok) throw new Error(`Ollama not reachable (HTTP ${res.status}) at ${this.url}`);
const data = await res.json();
const models = (data.models ?? []).map((m) => m.name);
return { ok: true, models };
}
async chat({ model, messages, tools, temperature = 0, numCtx = 8192, timeoutMs }) {
const body = {
model,
messages,
stream: false,
options: { temperature },
};
if (numCtx) body.options.num_ctx = numCtx;
if (tools && tools.length) body.tools = tools;
const res = await fetch(`${this.url}/api/chat`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify(body),
signal: AbortSignal.timeout(timeoutMs ?? this.timeoutMs),
});
if (!res.ok) {
const t = await res.text();
throw new Error(`Ollama chat HTTP ${res.status}: ${t.slice(0, 300)}`);
}
const data = await res.json();
const message = data.message ?? {};
return {
role: message.role ?? "assistant",
content: message.content ?? "",
toolCalls: message.tool_calls ?? [],
promptEvalCount: data.prompt_eval_count ?? 0,
evalCount: data.eval_count ?? 0,
raw: data,
};
}
}
export async function listModels() {
const o = new Ollama();
const res = await fetch(`${o.url}/api/tags`);
if (!res.ok) throw new Error(`Ollama not reachable (HTTP ${res.status})`);
const data = await res.json();
return (data.models ?? []).map((m) => m.name);
}
+244
View File
@@ -0,0 +1,244 @@
interface OllamaConstructorOpts {
url?: string;
timeoutMs?: number;
idleTimeoutMs?: number | null;
maxTotalMs?: number | null;
}
interface PingResult {
ok: boolean;
models: string[];
}
export type OllamaToolCall = {
function: { name: string; arguments: string };
};
export type OllamaToolDefinition = {
type: "function";
function: {
name: string;
description: string;
parameters: {
type: "object";
properties?: Record<string, unknown>;
[key: string]: unknown;
};
};
};
export type OllamaMessage = {
role: string;
content: string;
tool_calls?: OllamaToolCall[];
};
export type OllamaChunk = {
token: string | null;
content: string;
type: "content" | "tool_calls";
toolCalls?: OllamaToolCall[];
};
export interface ChatStreamOpts {
model: string;
messages: OllamaMessage[];
tools?: unknown[];
temperature?: number;
numCtx?: number;
timeoutMs?: number;
maxTotalMs?: number;
onChunk?: ((chunk: OllamaChunk) => void) | null;
}
export interface OllamaChatResult {
role: "assistant";
content: string;
toolCalls: OllamaToolCall[];
promptEvalCount: number;
evalCount: number;
raw: { stream: boolean };
}
export interface OllamaError extends Error {
partialContent?: string;
partialToolCalls?: OllamaToolCall[];
partialPrompt?: number;
partialGen?: number;
}
const DEFAULT_URL = "http://localhost:11434";
export class Ollama {
private url: string;
private timeoutMs: number;
private idleTimeoutMs: number;
private maxTotalMs: number;
constructor({ url = DEFAULT_URL, timeoutMs = 600000, idleTimeoutMs = null, maxTotalMs = null }: OllamaConstructorOpts = {}) {
this.url = url.replace(/\/$/, "");
this.timeoutMs = timeoutMs;
this.idleTimeoutMs = idleTimeoutMs ?? timeoutMs;
this.maxTotalMs = maxTotalMs ?? 5 * 60_000;
}
async ping({ timeoutMs = 30000 } = {}): Promise<PingResult> {
if (process.env.OLLAMA_SKIP_PING === "1") return { ok: true, models: [process.env.OLLAMA_MODEL ?? ""] };
const res = await fetch(`${this.url}/api/tags`, { signal: AbortSignal.timeout(timeoutMs) });
if (!res.ok) throw new Error(`Ollama not reachable (HTTP ${res.status}) at ${this.url}`);
const data = await res.json() as { models?: Array<{ name: string }> };
const models = (data.models ?? []).map((m) => m.name);
return { ok: true, models };
}
async chatStream({
model,
messages,
tools,
temperature = 0,
numCtx = 8192,
timeoutMs,
maxTotalMs,
onChunk,
}: ChatStreamOpts): Promise<OllamaChatResult> {
const body: Record<string, unknown> = {
model,
messages,
stream: true,
options: { temperature },
};
if (numCtx) (body.options as Record<string, unknown>).num_ctx = numCtx;
if (tools && tools.length) body.tools = tools;
const limitMs = timeoutMs ?? this.idleTimeoutMs;
const totalCapMs = maxTotalMs ?? this.maxTotalMs;
const controller = new AbortController();
const startedAt = Date.now();
const res = await fetch(`${this.url}/api/chat`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify(body),
signal: controller.signal,
});
if (!res.ok) {
const t = await res.text();
throw new Error(`Ollama chat HTTP ${res.status}: ${t.slice(0, 300)}`);
}
if (!res.body || !res.body.getReader) {
throw new Error("Ollama streaming response has no body reader");
}
const reader = res.body.getReader();
const decoder = new TextDecoder();
let buffer = "";
let content = "";
let toolCalls: OllamaToolCall[] = [];
let promptEvalCount = 0;
let evalCount = 0;
let lastActivity = Date.now();
const guardInterval = setInterval(() => {
if (Date.now() - startedAt > totalCapMs) {
const err = new Error(`Request exceeded hard cap of ${totalCapMs}ms; aborting`) as OllamaError;
err.name = "TimeoutError";
controller.abort(err);
return;
}
if (Date.now() - lastActivity > limitMs) {
const err = new Error(`No progress from Ollama for ${limitMs}ms; aborting`) as OllamaError;
err.name = "TimeoutError";
controller.abort(err);
}
}, Math.min(500, Math.max(100, Math.floor(Math.min(limitMs, totalCapMs) / 4))));
try {
for (;;) {
let chunk: ReadableStreamReadResult<Uint8Array>;
try {
chunk = await reader.read();
} catch (readErr) {
const e = new Error(
`Aborted mid-stream after ${content.length} chars (toolCalls=${toolCalls.length}): ${(readErr as Error).message}`
) as OllamaError;
e.name = "TimeoutError";
e.partialContent = content;
e.partialToolCalls = toolCalls;
e.partialPrompt = promptEvalCount;
e.partialGen = evalCount;
throw e;
}
const { done, value } = chunk;
if (done) break;
lastActivity = Date.now();
buffer += decoder.decode(value, { stream: true });
let idx: number;
while ((idx = buffer.indexOf("\n")) !== -1) {
const line = buffer.slice(0, idx).trim();
buffer = buffer.slice(idx + 1);
if (!line) continue;
let obj: Record<string, unknown>;
try {
obj = JSON.parse(line);
} catch {
continue;
}
if (obj.prompt_eval_count != null) promptEvalCount = obj.prompt_eval_count as number;
if (obj.eval_count != null) evalCount = obj.eval_count as number;
const msg = (obj.message ?? {}) as Record<string, unknown>;
if (msg.content) {
content += msg.content as string;
onChunk?.({ token: msg.content as string, content, type: "content" });
}
if (msg.tool_calls && (msg.tool_calls as unknown[]).length) {
toolCalls = msg.tool_calls as OllamaToolCall[];
onChunk?.({ token: null, content, type: "tool_calls", toolCalls });
}
}
}
} finally {
clearInterval(guardInterval);
}
return {
role: "assistant",
content,
toolCalls: Array.isArray(toolCalls) ? toolCalls : [],
promptEvalCount,
evalCount,
raw: { stream: true },
};
}
async chat({ model, messages, tools, temperature = 0, numCtx = 8192, timeoutMs }: ChatStreamOpts): Promise<OllamaChatResult> {
return this.chatStream({
model,
messages,
tools,
temperature,
numCtx,
timeoutMs,
onChunk: null,
});
}
async warmup({ model, text = "say OK", timeoutMs = 60000 }: { model: string; text?: string; timeoutMs?: number }): Promise<Partial<OllamaChatResult> & { error?: string }> {
try {
return await this.chatStream({
model,
messages: [{ role: "user", content: text }],
timeoutMs,
});
} catch (err) {
return { error: (err as Error).message };
}
}
}
export async function listModels(): Promise<string[]> {
const o = new Ollama();
const res = await fetch(`${o["url"]}/api/tags`);
if (!res.ok) throw new Error(`Ollama not reachable (HTTP ${res.status})`);
const data = await res.json() as { models?: Array<{ name: string }> };
return (data.models ?? []).map((m) => m.name);
}
+9 -3
View File
@@ -3,12 +3,18 @@
"version": "1.0.0", "version": "1.0.0",
"description": "MCP rubber-duck evaluation harness", "description": "MCP rubber-duck evaluation harness",
"scripts": { "scripts": {
"run": "node src/run.mjs", "run": "tsx src/run.ts",
"report-html": "node src/report-html.mjs" "review": "tsx src/review.ts",
"typecheck": "tsc --noEmit"
}, },
"license": "ISC", "license": "ISC",
"type": "module", "type": "module",
"dependencies": { "dependencies": {
"@modelcontextprotocol/client": "^2.0.0" "@duck/types": "workspace:*"
},
"devDependencies": {
"@types/node": "^20",
"tsx": "^4",
"typescript": "^5.9.3"
} }
} }
-380
View File
@@ -1,380 +0,0 @@
import { readFile, writeFile, mkdir } from "node:fs/promises";
import path from "node:path";
const SCENARIOS = ["control", "blind", "mentor"];
const SCENARIO_LABELS = {
control: "Без утки (контроль)",
blind: "Слепая утка",
mentor: "Утка-помощник",
};
const SCENARIO_DESCS = {
control: "Модель решает задачу напрямую, без инструментов.",
blind: "Модель объясняет подход коллеге, вызывает quack, не зная заранее ответ.",
mentor: "Модель обязана выписать мысли и возможные ошибки, затем проверить себя уткой.",
};
function esc(s) {
return String(s ?? "")
.replace(/&/g, "&amp;")
.replace(/</g, "&lt;")
.replace(/>/g, "&gt;");
}
function nl2br(s) {
return esc(s).replace(/\n/g, "<br>");
}
function pctColor(p) {
return p >= 70 ? "#10b981" : p >= 40 ? "#f59e0b" : "#ef4444";
}
function modelShort(name) {
const base = String(name).split(/:/)[0];
const ver = String(name).split(/:/)[1] ? ":" + String(name).split(/:/)[1] : "";
return esc(base + ver);
}
function summaryTable(report) {
const perModel = report.aggregate.perModel;
const models = Object.keys(perModel);
const rows = models
.map((m) => {
const g = perModel[m];
const scen = report.models[m].scenarios;
const acc = (s) => scen[s].accuracy;
const row = `
<td class="pm-num" data-sc="control" style="color:${pctColor(acc("control"))}">${acc("control")}%</td>
<td class="pm-num" data-sc="blind" style="color:${pctColor(acc("blind"))}">${acc("blind")}%</td>
<td class="pm-num" data-sc="mentor" style="color:${pctColor(acc("mentor"))}">${acc("mentor")}%</td>
<td class="pm-num">${g.accuracy}%</td>
<td class="pm-num">${g.reviewed}<span class="muted"> / ${g.totalRows}</span></td>
<td class="pm-num">${g.duckUsed}</td>
<td class="pm-num">${g.toolCalls}</td>
<td class="pm-num">${g.avgGenTokens}</td>
`;
return `
<tr>
<td class="pm-name" data-model="${esc(m)}">${modelShort(m)}</td>
${row}
</tr>`;
})
.join("");
return `
<div class="summary-shell">
<table class="summary">
<thead>
<tr>
<th>Модель</th>
<th>Контроль</th>
<th>Слепая</th>
<th>Помощник</th>
<th>Средняя acc</th>
<th>Размечено</th>
<th>Уток</th>
<th>Вызовов</th>
<th>ср. токены out</th>
</tr>
</thead>
<tbody>${rows}</tbody>
</table>
</div>`;
}
function scenarioCard(key, agg) {
const pct = agg.accuracy;
const color = pctColor(pct);
const duckPct = agg.tasks ? Math.round((agg.duckUsed / agg.tasks) * 100) : 0;
const pendingBadge = agg.pending > 0 ? `<span class="pending-badge">неразмечено: ${agg.pending}</span>` : "";
const modelBars = Object.entries(agg.byModel)
.map(([m, g]) => {
const c = pctColor(g.accuracy);
const barW = g.reviewed ? g.accuracy : 0;
return `
<div class="agg-model">
<span class="agg-model-name">${modelShort(m)}${g.reviewed ? "" : " <span class='muted'>(?)</span>"}</span>
<div class="bar"><div class="bar-fill" style="width:${barW}%;background:${c}"></div></div>
<span class="agg-model-val" style="color:${c}">${g.accuracy}%</span>
</div>`;
})
.join("");
return `
<div class="card scenario" data-scenario="${key}">
<div class="scenario-head">
<div>
<div class="scenario-title">${esc(SCENARIO_LABELS[key])}</div>
<div class="scenario-desc">${esc(SCENARIO_DESCS[key])}</div>
</div>
<div class="accuracy" style="color:${color}">${pct}%${pendingBadge}</div>
</div>
<div class="bar"><div class="bar-fill" style="width:${pct}%;background:${color}"></div></div>
<div class="metrics">
<div class="metric"><span class="m-label">Верных</span><span class="m-val">${agg.correct} <span class="m-mod">/ ${agg.reviewed}</span></span><span class="m-sub">размечено из ${agg.tasks}</span></div>
<div class="metric"><span class="m-label">Утка (MCP)</span><span class="m-val">${duckPct}%</span><span class="m-sub">в ${agg.duckUsed} из ${agg.tasks} задач</span></div>
<div class="metric"><span class="m-label">quack / задачу</span><span class="m-val">${agg.avgCalls}</span><span class="m-sub">вызовов в среднем</span></div>
</div>
<div class="agg-models">${modelBars}</div>
</div>`;
}
function rowHtml(r) {
const pending = r.correct === null || r.correct === undefined;
const cls = pending ? "pend" : r.correct ? "ok" : "no";
const result = pending ? "?" : r.correct ? "✓" : "✗";
const duck = r.duckUsed ? "🦆" : "—";
return `
<tr class="scenario-row ${cls}" data-scenario="${r.scenario}" data-model="${esc(r.model)}">
<td class="r-model">${modelShort(r.model)}</td>
<td class="r-id">${esc(r.id)}</td>
<td class="r-s">${esc(SCENARIO_LABELS[r.scenario])}</td>
<td class="r-result">${result}</td>
<td class="r-duck">${duck} <span class="muted">(${r.toolCalls})</span></td>
<td class="r-question">${nl2br(r.question)}</td>
<td class="r-expected">${esc(r.expected)}</td>
<td class="r-tokens">${r.promptTokens}<span class="muted"> / </span>${r.genTokens}</td>
<td class="r-answer">${nl2br(r.response)}</td>
</tr>`;
}
function promptBlock(report) {
const p = report.prompts || {};
const items = SCENARIOS
.filter((k) => p[k])
.map(
(k) => `
<div class="prompt-item">
<div class="prompt-name">${esc(SCENARIO_LABELS[k])}</div>
<div class="prompt-text">${esc(p[k])}</div>
</div>`
)
.join("");
if (!items) return "";
return `
<details class="prompts" open>
<summary>Промпты (то, что передаётся модели)</summary>
<div class="prompts-body">${items}</div>
</details>`;
}
export function buildHtml(report) {
const perTask = Object.entries(report.models).flatMap(([model, rep]) =>
rep.perTask.map((r) => ({ ...r, model }))
);
const rows = perTask.map(rowHtml).join("");
const models = Object.keys(report.aggregate.perModel);
let totalPending = 0, totalReviewed = 0;
for (const rep of Object.values(report.models)) {
for (const s of SCENARIOS) {
if (rep.scenarios[s]) {
totalReviewed += rep.scenarios[s].reviewed || 0;
totalPending += rep.scenarios[s].pending || 0;
}
}
}
const reviewNote = totalPending > 0
? `<div class="review-note">⚠ Не все ответы размечены: <b>${totalReviewed}</b> проверено, <b>${totalPending}</b> ожидают ревью (показаны символом «?»). Запустите <code>node src/review.mjs</code> для проставления меток.</div>`
: "";
const modelOptions = models
.map((m) => `<option value="${esc(m)}">${modelShort(m)}</option>`)
.join("");
const aggCards = SCENARIOS
.filter((k) => report.aggregate.perScenario[k])
.map((k) => scenarioCard(k, report.aggregate.perScenario[k]))
.join("");
return `<!doctype html>
<html lang="ru">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<title>Rubber Duck MCP — отчёт тестирования</title>
<style>
:root {
--bg:#0f172a; --panel:#1e293b; --panel2:#273449; --text:#e2e8f0;
--muted:#94a3b8; --ok:#10b981; --no:#f87171; --border:#334155;
}
* { box-sizing:border-box; }
body { margin:0; font-family:-apple-system,Segoe UI,Roboto,Helvetica,Arial,sans-serif;
background:var(--bg); color:var(--text); line-height:1.5; }
.wrap { max-width:1280px; margin:0 auto; padding:24px 20px 60px; }
header h1 { margin:0 0 4px; font-size:22px; }
.meta { color:var(--muted); font-size:13px; }
.meta b { color:var(--text); font-weight:600; }
h2 { font-size:17px; margin:26px 0 12px; }
.summary-shell { background:var(--panel); border:1px solid var(--border); border-radius:12px; overflow:auto; }
table.summary { width:100%; border-collapse:collapse; font-size:13px; }
table.summary th { background:var(--panel2); text-align:left; padding:10px 14px;
font-size:11px; text-transform:uppercase; letter-spacing:.05em; color:var(--muted); white-space:nowrap; }
table.summary td { padding:10px 14px; border-top:1px solid var(--border); }
.pm-name { font-weight:700; white-space:nowrap; }
.pm-num { text-align:right; font-variant-numeric:tabular-nums; }
.cards { display:grid; grid-template-columns:repeat(3,1fr); gap:16px; margin:0 0 8px; }
@media (max-width:1000px){ .cards { grid-template-columns:1fr; } }
.card { background:var(--panel); border:1px solid var(--border); border-radius:12px; padding:16px; }
.scenario-head { display:flex; justify-content:space-between; align-items:flex-start; gap:12px; }
.scenario-title { font-size:16px; font-weight:700; }
.scenario-desc { color:var(--muted); font-size:12px; margin-top:4px; }
.accuracy { font-size:32px; font-weight:800; line-height:1; }
.pending-badge { display:block; font-size:11px; font-weight:600; color:var(--muted); margin-top:4px; }
.bar { height:8px; background:var(--panel2); border-radius:99px; margin:14px 0; overflow:hidden; }
.bar-fill { height:100%; border-radius:99px; transition:width .4s; }
.metrics { display:grid; grid-template-columns:repeat(2,1fr); gap:10px; }
.metric { background:var(--panel2); border-radius:8px; padding:10px 12px; }
.m-label { display:block; color:var(--muted); font-size:11px; text-transform:uppercase; letter-spacing:.05em; white-space:nowrap; }
.m-val { display:block; font-size:20px; font-weight:800; margin-top:2px; }
.m-mod { font-size:14px; font-weight:600; color:var(--muted); }
.m-sub { display:block; color:var(--muted); font-size:11px; margin-top:2px; }
.agg-models { margin-top:14px; display:grid; gap:8px; }
.agg-model { display:grid; grid-template-columns:auto 1fr auto; align-items:center; gap:10px; font-size:12px; }
.agg-model .bar { margin:0; }
.agg-model-name { color:var(--muted); white-space:nowrap; }
.agg-model-val { font-weight:700; font-variant-numeric:tabular-nums; }
.prompts { margin:20px 0; background:var(--panel); border:1px solid var(--border); border-radius:12px; overflow:hidden; }
.review-note { margin:14px 0; padding:10px 14px; background:rgba(245,158,11,.12); border:1px solid rgba(245,158,11,.4);
border-radius:10px; color:#fbbf24; font-size:13px; }
.review-note code { background:var(--panel2); padding:1px 6px; border-radius:5px; }
.prompts summary { cursor:pointer; padding:12px 16px; font-weight:700; font-size:14px; list-style:none; }
.prompts summary::-webkit-details-marker { display:none; }
.prompts summary::before { content:"▸ "; color:var(--muted); }
.prompts[open] summary::before { content:"▾ "; }
.prompts-body { padding:0 16px 14px; display:grid; gap:12px; }
.prompt-item { background:var(--panel2); border-radius:8px; padding:10px 12px; }
.prompt-name { font-size:12px; color:var(--muted); text-transform:uppercase; letter-spacing:.05em; margin-bottom:6px; }
.prompt-text { white-space:pre-wrap; font-size:13px; line-height:1.5; }
.filters { margin:8px 0 16px; display:flex; gap:8px; flex-wrap:wrap; align-items:center; }
.filters button { background:var(--panel); color:var(--text); border:1px solid var(--border);
padding:7px 14px; border-radius:8px; cursor:pointer; font-size:13px; }
.filters button.active { background:#3b82f6; border-color:#3b82f6; color:#fff; }
.filters select { background:var(--panel); color:var(--text); border:1px solid var(--border);
padding:7px 10px; border-radius:8px; font-size:13px; }
.tbl-shell { background:var(--panel); border:1px solid var(--border); border-radius:12px; overflow:auto; }
table { width:100%; border-collapse:collapse; font-size:13px; }
thead th { position:sticky; top:0; background:var(--panel2); text-align:left; padding:10px 12px;
font-size:11px; text-transform:uppercase; letter-spacing:.05em; color:var(--muted); white-space:nowrap; }
tbody td { padding:10px 12px; border-top:1px solid var(--border); vertical-align:top; }
.scenario-row.ok .r-result { color:var(--ok); font-weight:800; }
.scenario-row.no .r-result { color:var(--no); font-weight:800; }
.scenario-row.pend .r-result { color:var(--muted); font-weight:800; }
.scenario-row.pend { opacity:.75; } .r-model { font-weight:600; white-space:nowrap; }
.r-id { font-weight:700; color:var(--muted); }
.r-s { white-space:nowrap; }
.r-duck { text-align:center; white-space:nowrap; }
.r-question { max-width:300px; color:var(--text); }
.r-answer { max-width:380px; }
.muted { color:var(--muted); }
.hidden { display:none !important; }
footer { margin-top:20px; color:var(--muted); font-size:12px; }
</style>
</head>
<body>
<div class="wrap">
<header>
<h1>🦆 Rubber Duck MCP — отчёт тестирования</h1>
<div class="meta">
MCP: <b>${esc(report.mcpUrl)}</b> &nbsp;·&nbsp;
Моделей: <b>${models.length}</b> &nbsp;·&nbsp;
Дата: <b>${esc(report.generatedAt)}</b>
</div>
</header>
<h2>Сводка по моделям</h2>
${summaryTable(report)}
${reviewNote}
<h2>Агрегат по сценариям (все модели)</h2>
<div class="cards">${aggCards}</div>
${promptBlock(report)}
<div class="filters" id="filters">
<label style="font-size:12px;color:var(--muted)">Модель:</label>
<select id="modelSel">
<option value="all">Все модели</option>
${modelOptions}
</select>
<button data-filter="all" class="active">Все сценарии</button>
<button data-filter="control">Без утки</button>
<button data-filter="blind">Слепая утка</button>
<button data-filter="mentor">Утка-помощник</button>
<span style="margin-left:auto;color:var(--muted);font-size:12px" id="count"></span>
</div>
<div class="tbl-shell">
<table>
<thead>
<tr>
<th>Модель</th><th>#</th><th>Сценарий</th><th>✓</th><th>Утка</th>
<th>Вопрос</th><th>Ожид.</th><th>in/out</th><th>Ответ модели</th>
</tr>
</thead>
<tbody>${rows}</tbody>
</table>
</div>
<footer>Генерировано локальным харнессом apps/eval. temperature=0. Прогон по нескольким локальным моделям через Ollama.</footer>
</div>
<script>
const buttons = document.querySelectorAll('#filters button');
const modelSel = document.getElementById('modelSel');
const rows2 = document.querySelectorAll('tbody .scenario-row');
const count = document.getElementById('count');
let sc = 'all';
let mdl = 'all';
function apply() {
let n = 0;
rows2.forEach(r => {
const okSc = sc === 'all' || r.dataset.scenario === sc;
const okM = mdl === 'all' || r.dataset.model === mdl;
const show = okSc && okM;
r.classList.toggle('hidden', !show);
if (show) n++;
});
count.textContent = 'показано: ' + n + ' из ' + rows2.length;
}
buttons.forEach(b => b.addEventListener('click', () => {
buttons.forEach(x => x.classList.remove('active'));
b.classList.add('active');
sc = b.dataset.filter;
apply();
}));
modelSel.addEventListener('change', () => { mdl = modelSel.value; apply(); });
apply('all');
</script>
</body>
</html>`;
}
async function main() {
const args = process.argv.slice(2);
let inFile = null, outFile = "out/report.html";
for (let i = 0; i < args.length; i += 1) {
const a = args[i];
const next = () => args[i + 1];
if (a === "--in") inFile = next(), i += 1;
else if (a.startsWith("--in=")) inFile = a.slice(5);
else if (a === "--out") outFile = next(), i += 1;
else if (a.startsWith("--out=")) outFile = a.slice(6);
}
if (!inFile) {
const reviewed = path.join("out", "report.reviewed.json");
const base = path.join("out", "report.json");
inFile = (await readFile(reviewed, "utf8").then(() => reviewed).catch(() => base));
}
const report = JSON.parse(await readFile(inFile, "utf8"));
const html = buildHtml(report);
await mkdir(path.dirname(outFile), { recursive: true });
await writeFile(outFile, html, "utf8");
console.log(`HTML report written: ${path.resolve(outFile)} (source: ${inFile})`);
}
if (process.argv[1] && path.resolve(process.argv[1]).includes("report-html")) {
main().catch((e) => {
console.error(e);
process.exit(1);
});
}
@@ -1,16 +1,26 @@
function round(x, d = 2) { import type { ScenarioName, ScenarioSummary, PerTaskRow, ModelReport, Aggregate, AggregateScenario, AggregateModel } from "@duck/types";
import { SCENARIO_NAMES } from "@duck/types";
function round(x: number, d = 2): number {
if (!Number.isFinite(x)) return x; if (!Number.isFinite(x)) return x;
const f = 10 ** d; const f = 10 ** d;
return Math.round(x * f) / f; return Math.round(x * f) / f;
} }
function avg(arr) { function avg(arr: number[]): number {
if (!arr.length) return 0; if (!arr.length) return 0;
return arr.reduce((a, b) => a + b, 0) / arr.length; return arr.reduce((a, b) => a + b, 0) / arr.length;
} }
export function buildReport({ model, mcpUrl, tasks, rows, meta = {}, prompts = {} }) { const scenarios = ["control", "blind", "mentor"]; export interface BuildReportOpts {
const byScenario = {}; model: string;
mcpUrl: string;
rows: PerTaskRow[];
}
export function buildReport({ model, mcpUrl, rows }: BuildReportOpts): ModelReport {
const scenarios = SCENARIO_NAMES;
const byScenario: Record<string, { total: number; correct: number; reviewed: number; duckUsed: number; toolCalls: number; promptTokens: number[]; genTokens: number[]; duckTokens: number[]; rows: PerTaskRow[] }> = {};
for (const s of scenarios) byScenario[s] = { total: 0, correct: 0, reviewed: 0, duckUsed: 0, toolCalls: 0, promptTokens: [], genTokens: [], duckTokens: [], rows: [] }; for (const s of scenarios) byScenario[s] = { total: 0, correct: 0, reviewed: 0, duckUsed: 0, toolCalls: 0, promptTokens: [], genTokens: [], duckTokens: [], rows: [] };
for (const row of rows) { for (const row of rows) {
@@ -29,10 +39,9 @@ export function buildReport({ model, mcpUrl, tasks, rows, meta = {}, prompts = {
byScenario[s].rows.push(row); byScenario[s].rows.push(row);
} }
const summaries = {}; const summaries: Record<ScenarioName, ScenarioSummary> = {} as Record<ScenarioName, ScenarioSummary>;
for (const s of scenarios) { for (const s of scenarios) {
const g = byScenario[s]; const g = byScenario[s];
const denom = g.reviewed || 1;
summaries[s] = { summaries[s] = {
tasks: g.total, tasks: g.total,
reviewed: g.reviewed, reviewed: g.reviewed,
@@ -49,11 +58,8 @@ export function buildReport({ model, mcpUrl, tasks, rows, meta = {}, prompts = {
} }
return { return {
generatedAt: new Date().toISOString(),
model, model,
mcpUrl,
scenarios: summaries, scenarios: summaries,
prompts,
perTask: rows.map((r) => ({ perTask: rows.map((r) => ({
id: r.id, id: r.id,
scenario: r.scenario, scenario: r.scenario,
@@ -67,17 +73,14 @@ export function buildReport({ model, mcpUrl, tasks, rows, meta = {}, prompts = {
expected: r.expected, expected: r.expected,
question: r.question, question: r.question,
})), })),
meta,
}; };
} }
const SCENARIOS = ["control", "blind", "mentor"]; export function buildAggregate(modelsReport: Record<string, ModelReport>): Aggregate {
const perScenario: Record<ScenarioName, AggregateScenario> = {} as Record<ScenarioName, AggregateScenario>;
export function buildAggregate(modelsReport) { for (const s of SCENARIO_NAMES) {
const perScenario = {};
for (const s of SCENARIOS) {
let total = 0, reviewed = 0, correct = 0, duckUsed = 0, toolCalls = 0; let total = 0, reviewed = 0, correct = 0, duckUsed = 0, toolCalls = 0;
const byModel = {}; const byModel: Record<string, AggregateScenario["byModel"][string]> = {};
for (const [name, rep] of Object.entries(modelsReport)) { for (const [name, rep] of Object.entries(modelsReport)) {
const g = rep.scenarios[s]; const g = rep.scenarios[s];
if (!g) continue; if (!g) continue;
@@ -107,12 +110,12 @@ export function buildAggregate(modelsReport) {
}; };
} }
const perModel = {}; const perModel: Record<string, AggregateModel> = {};
for (const [name, rep] of Object.entries(modelsReport)) { for (const [name, rep] of Object.entries(modelsReport)) {
let duckUsed = 0, toolCalls = 0; let duckUsed = 0, toolCalls = 0;
const gen = []; const gen: number[] = [];
let reviewed = 0, correct = 0; let reviewed = 0, correct = 0;
for (const s of SCENARIOS) { for (const s of SCENARIO_NAMES) {
const g = rep.scenarios[s]; const g = rep.scenarios[s];
duckUsed += g.duckUsed; duckUsed += g.duckUsed;
toolCalls += g.toolCalls; toolCalls += g.toolCalls;
@@ -123,7 +126,7 @@ export function buildAggregate(modelsReport) {
} }
} }
const taskCount = rep.scenarios.control?.tasks ?? 0; const taskCount = rep.scenarios.control?.tasks ?? 0;
const totalRows = SCENARIOS.reduce((acc, s) => acc + (rep.scenarios[s].tasks || 0), 0); const totalRows = SCENARIO_NAMES.reduce((acc, s) => acc + (rep.scenarios[s].tasks || 0), 0);
perModel[name] = { perModel[name] = {
tasks: taskCount, tasks: taskCount,
totalRows, totalRows,
@@ -1,23 +1,36 @@
import { readFile, writeFile, mkdir } from "node:fs/promises"; import { readFile, writeFile, mkdir } from "node:fs/promises";
import path from "node:path"; import path from "node:path";
import readline from "node:readline"; import readline from "node:readline";
import { buildReport, buildAggregate } from "./report.mjs"; import type { ReportRoot, ScenarioName, PerTaskRow } from "@duck/types";
import { SCENARIO_NAMES } from "@duck/types";
import { buildReport, buildAggregate } from "./report.js";
const SCENARIO_LABELS = { const SCENARIO_LABELS: Record<ScenarioName, string> = {
control: "Без утки (контроль)", control: "Без утки (контроль)",
thinking: "Думать вслух",
blind: "Слепая утка", blind: "Слепая утка",
mentor: "Утка-помощник", mentor: "Утка-помощник",
}; };
function esc(s) { interface ReviewEntry {
return String(s ?? "") model: string;
.replace(/&/g, "&amp;") taskId: string;
.replace(/</g, "&lt;") scenario: ScenarioName;
.replace(/>/g, "&gt;"); row: PerTaskRow;
} }
function parseArgs(argv) { interface ReviewArgs {
const args = { report: string | null;
out: string | null;
reviewFile: string | null;
model: string | null;
scenario: string | null;
limit: number | null;
reAsk?: boolean;
}
function parseArgs(argv: string[]): ReviewArgs {
const args: ReviewArgs = {
report: null, report: null,
out: null, out: null,
reviewFile: null, reviewFile: null,
@@ -39,22 +52,22 @@ function parseArgs(argv) {
return args; return args;
} }
async function main() { async function main(): Promise<void> {
const args = parseArgs(process.argv.slice(2)); const args = parseArgs(process.argv.slice(2));
const reportFile = args.report || path.join("out", "report.json"); const reportFile = args.report || path.join("out", "report.json");
const outFile = args.out || path.join("out", "report.reviewed.json"); const outFile = args.out || path.join("out", "report.reviewed.json");
const reviewFile = args.reviewFile || path.join("out", "review.json"); const reviewFile = args.reviewFile || path.join("out", "review.json");
const report = JSON.parse(await readFile(reportFile, "utf8")); const report: ReportRoot = JSON.parse(await readFile(reportFile, "utf8"));
let decisions = {}; let decisions: Record<string, boolean> = {};
try { try {
decisions = JSON.parse(await readFile(reviewFile, "utf8")); decisions = JSON.parse(await readFile(reviewFile, "utf8"));
} catch { } catch {
/* no prior review */ /* no prior review */
} }
const entries = []; const entries: ReviewEntry[] = [];
for (const [model, rep] of Object.entries(report.models)) { for (const [model, rep] of Object.entries(report.models)) {
for (const row of rep.perTask) { for (const row of rep.perTask) {
if (args.model && model !== args.model) continue; if (args.model && model !== args.model) continue;
@@ -64,13 +77,13 @@ async function main() {
} }
if (args.limit && entries.length > args.limit) entries.length = args.limit; if (args.limit && entries.length > args.limit) entries.length = args.limit;
const key = (e) => `${e.model}|${e.taskId}|${e.scenario}`; const key = (e: ReviewEntry) => `${e.model}|${e.taskId}|${e.scenario}`;
const isTTY = !!process.stdin.isTTY; const isTTY = !!process.stdin.isTTY;
let batchLines = []; let batchLines: string[] = [];
if (!isTTY) { if (!isTTY) {
const input = await new Promise((res, rej) => { const input = await new Promise<string>((res, rej) => {
let data = ""; let data = "";
process.stdin.setEncoding("utf8"); process.stdin.setEncoding("utf8");
process.stdin.on("data", (c) => (data += c)); process.stdin.on("data", (c) => (data += c));
@@ -82,17 +95,21 @@ async function main() {
} }
let batchIdx = 0; let batchIdx = 0;
const nextInput = async () => { const nextInput = async (): Promise<string> => {
if (isTTY) { if (isTTY) {
const rl = readline.createInterface({ input: process.stdin, output: process.stdout }); const rl = readline.createInterface({ input: process.stdin, output: process.stdout });
const ans = await new Promise((res) => rl.question("", res)); const ans = await new Promise<string>((res) => rl.question("", res));
rl.close(); rl.close();
return ans.trim().toLowerCase(); return ans.trim().toLowerCase();
} }
return batchLines[batchIdx++] ?? ""; return batchLines[batchIdx++] ?? "";
}; };
async function askOne(e, nextInput, decisions) { let done = 0;
let skipped = 0;
let changed = 0;
async function askOne(e: ReviewEntry): Promise<number> {
const k = key(e); const k = key(e);
const prev = decisions[k]; const prev = decisions[k];
if (prev !== undefined) { if (prev !== undefined) {
@@ -114,7 +131,7 @@ async function main() {
process.stdout.write(` [y] верно [n] неверно [e] пропуск${hint}\n > `); process.stdout.write(` [y] верно [n] неверно [e] пропуск${hint}\n > `);
const a = (await nextInput()) || ""; const a = (await nextInput()) || "";
let decision = null; let decision: boolean | null = null;
if (a === "y" || a === "д") decision = true; if (a === "y" || a === "д") decision = true;
else if (a === "n" || a === "н") decision = false; else if (a === "n" || a === "н") decision = false;
else if (a === "" && cur) decision = e.row.correct; else if (a === "" && cur) decision = e.row.correct;
@@ -130,10 +147,6 @@ async function main() {
return 1; return 1;
} }
let done = 0;
let skipped = 0;
let changed = 0;
const reAsk = !!args.reAsk; const reAsk = !!args.reAsk;
for (const e of entries) { for (const e of entries) {
@@ -141,7 +154,7 @@ async function main() {
const prev = decisions[k]; const prev = decisions[k];
if (prev === undefined || reAsk) { if (prev === undefined || reAsk) {
done += await askOne(e, nextInput, decisions); done += await askOne(e);
continue; continue;
} }
@@ -160,7 +173,6 @@ async function main() {
scenarios: buildReport({ scenarios: buildReport({
model, model,
mcpUrl: report.mcpUrl, mcpUrl: report.mcpUrl,
tasks: [],
rows: rep.perTask.map((r) => ({ rows: rep.perTask.map((r) => ({
id: r.id, id: r.id,
scenario: r.scenario, scenario: r.scenario,
@@ -191,7 +203,7 @@ async function main() {
const scen = rep.scenarios; const scen = rep.scenarios;
console.log( console.log(
`${model} | ` + `${model} | ` +
["control", "blind", "mentor"] SCENARIO_NAMES
.map((s) => `${s}=${String(scen[s].accuracy).padStart(5)}% (${scen[s].reviewed})`) .map((s) => `${s}=${String(scen[s].accuracy).padStart(5)}% (${scen[s].reviewed})`)
.join(" ") + .join(" ") +
` всего rev=${g.reviewed}` ` всего rev=${g.reviewed}`
-176
View File
@@ -1,176 +0,0 @@
import { readFile, writeFile, mkdir, appendFile } from "node:fs/promises";
import path from "node:path";
import { Ollama } from "../lib/ollama.mjs";
import { McpClient } from "../lib/mcpClient.mjs";
import { runScenario } from "./runner.mjs";
import { PROMPTS } from "./runner.mjs";
import { buildReport, buildAggregate } from "./report.mjs";
let logTarget = null;
async function log(msg) {
const line = `[${new Date().toISOString()}] ${msg}`;
console.log(line);
if (logTarget) {
try {
await appendFile(logTarget, line + "\n", "utf8");
} catch {
/* keep going */
}
}
}
const SCENARIOS = ["control", "blind", "mentor"];
const DEFAULT_MODELS = ["llama3.2:3b", "qwen3:4b", "gemma3:4b", "granite4.1:3b"];
const DEFAULT_MCP_URL = "https://mcp-liart-five.vercel.app/api/mcp";
function parseArgs(argv) {
const args = {
models: [],
scenarios: SCENARIOS,
limit: Infinity,
model: process.env.OLLAMA_MODEL || null,
mcpUrl: null,
tasksFile: null,
out: null,
};
for (let i = 0; i < argv.length; i += 1) {
const a = argv[i];
const next = () => argv[i + 1];
if (a === "--models") args.models = next().split(",").map((s) => s.trim()).filter(Boolean), i += 1;
else if (a === "--model") args.model = next(), i += 1;
else if (a === "--mcp-url") args.mcpUrl = next(), i += 1;
else if (a === "--scenarios") args.scenarios = next().split(","), i += 1;
else if (a === "--limit") args.limit = Number(next()), i += 1;
else if (a === "--tasks") args.tasksFile = next(), i += 1;
else if (a === "--out") args.out = next(), i += 1;
}
return args;
}
async function runOneModel({ model, ollama, mcp, tasks, scenarios }) {
const rows = [];
for (const task of tasks) {
for (const scenario of scenarios) {
let r;
const t0 = Date.now();
try {
r = await runScenario({ scenario, ollama, model, task: task.question, mcp });
} catch (err) {
const ms = Date.now() - t0;
await log(` [${task.id}/${scenario}] FAILED (${ms}ms): ${err.message}`);
rows.push({
id: task.id,
scenario,
correct: null,
duckUsed: false,
toolCalls: 0,
promptTokens: 0,
genTokens: 0,
duckTokens: 0,
response: `ERR: ${err.message}`,
expected: task.answer,
question: task.question,
});
continue;
}
const correct = null;
rows.push({
id: task.id,
scenario,
correct,
duckUsed: r.duckUsed,
toolCalls: r.toolCalls,
promptTokens: r.promptTokens,
genTokens: r.genTokens,
duckTokens: r.duckTokens,
response: r.response,
expected: task.answer,
question: task.question,
});
const duck = r.duckUsed ? "duck" : "n/a ";
const ms = Date.now() - t0;
await log(` [${task.id}/${scenario}] duck=${duck} calls=${r.toolCalls} prompt=${r.promptTokens} gen=${r.genTokens} (${ms}ms)`);
}
}
return rows;
}
async function main() {
const args = parseArgs(process.argv.slice(2));
let models = args.models;
if (args.model && !models.length) models = [args.model];
if (!models.length) models = DEFAULT_MODELS;
const mcpUrl = args.mcpUrl || DEFAULT_MCP_URL;
const tasksFile = args.tasksFile || path.join("data", "tasks.json");
const tasksRaw = JSON.parse(await readFile(tasksFile, "utf8"));
const tasks = tasksRaw.slice(0, args.limit === Infinity ? tasksRaw.length : args.limit);
const out = args.out || path.join("out", "report.json");
const logFile = args.logTarget || path.join(path.dirname(out), "run.log");
logTarget = logFile;
await mkdir(path.dirname(logFile), { recursive: true });
await appendFile(logFile, "", "utf8").catch(() => {});
await log(`MCP endpoint: ${mcpUrl}`);
await log(`Tasks: ${tasks.length}; Scenarios: [${args.scenarios.join(", ")}]`);
await log(`Models: ${models.join(", ")}`);
const ollama = new Ollama();
const ping = await ollama.ping();
await log(`Ollama OK (${ping.models.length} models): ${ping.models.join(", ")}`);
for (const m of models) {
const present = ping.models.some((x) => x.startsWith(m));
await log(` ${present ? "OK " : "MISSING "} ${m}`);
}
const mcp = new McpClient(mcpUrl);
const mcpTools = await mcp.listTools();
await log(`MCP tools: ${mcpTools.map((t) => t.name).join(", ")}`);
const modelsReport = {};
for (const model of models) {
const tStart = Date.now();
await log(`==== Running model: ${model} ====`);
const rows = await runOneModel({ model, ollama, mcp, tasks, scenarios: args.scenarios });
const rep = buildReport({ model, mcpUrl, tasks, rows });
modelsReport[model] = {
model,
scenarios: rep.scenarios,
perTask: rep.perTask,
};
await log(`==== Done ${model} in ${Math.round((Date.now() - tStart) / 1000)}s ====`);
}
const aggregate = buildAggregate(modelsReport);
const report = {
generatedAt: new Date().toISOString(),
mcpUrl,
prompts: PROMPTS,
models: modelsReport,
aggregate,
};
await mkdir(path.dirname(out), { recursive: true });
await writeFile(out, JSON.stringify(report, null, 2), "utf8");
await log(`Report written: ${out}`);
await log("==== Aggregate summary ====");
for (const [modelName, g] of Object.entries(aggregate.perModel)) {
const scen = modelsReport[modelName].scenarios;
await log(
`${modelName} | ` +
SCENARIOS.map((s) => `${s}=${String(scen[s].accuracy).padStart(5)}%`).join(" ") +
` duck=${g.duckUsed} calls=${g.toolCalls} avgGen=${g.avgGenTokens}`
);
}
}
main().catch((err) => {
console.error(err);
process.exit(1);
});
+218
View File
@@ -0,0 +1,218 @@
import { readFile, writeFile, mkdir, appendFile } from "node:fs/promises";
import path from "node:path";
import type { ScenarioName, ReportRoot, PerTaskRow, Task } from "@duck/types";
import { SCENARIO_NAMES } from "@duck/types";
import { Ollama } from "../lib/ollama.js";
import type { OllamaError } from "../lib/ollama.js";
import { McpClient } from "../lib/mcpClient.js";
import { runScenario, PROMPTS } from "./runner.js";
import { buildReport, buildAggregate } from "./report.js";
let logTarget: string | null = null;
async function log(msg: string): Promise<void> {
const line = `[${new Date().toISOString()}] ${msg}`;
console.log(line);
if (logTarget) {
try {
await appendFile(logTarget, line + "\n", "utf8");
} catch {
/* keep going */
}
}
}
const DEFAULT_MODELS = ["llama3.2:3b", "qwen3:1.7b", "qwen3:4b", "granite4.1:3b", "phi4-mini:3.8b"];
const DEFAULT_MCP_URL = "https://rubber-duck-mcp.vercel.app/api/mcp";
const REQUEST_TIMEOUT_MS = 60_000;
const MAX_ATTEMPTS = 3;
const PAUSE_BETWEEN_TASKS_MS = 2_000;
const PAUSE_BETWEEN_RETRIES_MS = 10_000;
function classifyError(err: unknown): string {
const name = (err as OllamaError)?.name ?? "";
const msg = String((err as OllamaError)?.message ?? "");
const low = `${name} ${msg}`.toLowerCase();
if (name === "TimeoutError" || low.includes("timeout") || low.includes("aborted")) return "timeout";
if (low.includes("fetch failed") || low.includes("connect") || low.includes("etimedout")) return "network";
if (low.includes("http 4") || low.includes("bad request") || low.includes("validation")) return "http4xx";
if (low.includes("http 5") || low.includes("server error")) return "http5xx";
return "other";
}
interface Args {
models: string[];
scenarios: ScenarioName[];
limit: number;
model: string | null;
mcpUrl: string | null;
tasksFile: string | null;
out: string | null;
}
function parseArgs(argv: string[]): Args {
const args: Args = {
models: [],
scenarios: [...SCENARIO_NAMES],
limit: Infinity,
model: process.env.OLLAMA_MODEL || null,
mcpUrl: null,
tasksFile: null,
out: null,
};
for (let i = 0; i < argv.length; i += 1) {
const a = argv[i];
const next = () => argv[i + 1];
if (a === "--models") args.models = next().split(",").map((s) => s.trim()).filter(Boolean), i += 1;
else if (a === "--model") args.model = next(), i += 1;
else if (a === "--mcp-url") args.mcpUrl = next(), i += 1;
else if (a === "--scenarios") args.scenarios = next().split(",") as ScenarioName[], i += 1;
else if (a === "--limit") args.limit = Number(next()), i += 1;
else if (a === "--tasks") args.tasksFile = next(), i += 1;
else if (a === "--out") args.out = next(), i += 1;
}
return args;
}
async function runOneModel({ model, ollama, mcp, tasks, scenarios }: { model: string; ollama: Ollama; mcp: McpClient; tasks: Task[]; scenarios: ScenarioName[] }): Promise<PerTaskRow[]> {
const rows: PerTaskRow[] = [];
for (const task of tasks) {
for (const scenario of scenarios) {
let r = null;
let lastErr: OllamaError | null = null;
const t0 = Date.now();
for (let attempt = 1; attempt <= MAX_ATTEMPTS; attempt += 1) {
try {
r = await runScenario({ scenario, ollama, model, task: task.question, mcp });
break;
} catch (err) {
lastErr = err as OllamaError;
const ms = Date.now() - t0;
await log(` [${task.id}/${scenario}] attempt ${attempt} FAILED (${ms}ms): ${(err as Error).message}`);
if (lastErr.partialContent != null || lastErr.partialToolCalls != null) {
await log(
` partial before abort: chars=${lastErr.partialContent?.length ?? 0} ` +
`toolCalls=${lastErr.partialToolCalls?.length ?? 0} ` +
`gen=${lastErr.partialGen ?? 0}`
);
}
if (attempt < MAX_ATTEMPTS) await new Promise((res) => setTimeout(res, PAUSE_BETWEEN_RETRIES_MS));
}
}
if (!r) {
const ms = Date.now() - t0;
const reason = classifyError(lastErr);
await log(` [${task.id}/${scenario}] SKIPPED after ${MAX_ATTEMPTS} attempts (${ms}ms) reason=${reason}`);
await log(` last error: ${lastErr?.name} | ${lastErr?.message}`);
if (lastErr?.stack) await log(` stack: ${String(lastErr.stack).split("\n").slice(0, 3).join(" | ")}`);
continue;
}
const correct = null;
rows.push({
id: task.id,
scenario,
correct,
duckUsed: r.duckUsed,
toolCalls: r.toolCalls,
promptTokens: r.promptTokens,
genTokens: r.genTokens,
duckTokens: r.duckTokens,
response: r.response,
expected: task.answer,
question: task.question,
});
const duck = r.duckUsed ? "duck" : "n/a ";
const ms = Date.now() - t0;
await log(` [${task.id}/${scenario}] duck=${duck} calls=${r.toolCalls} prompt=${r.promptTokens} gen=${r.genTokens} (${ms}ms)`);
}
if (tasks.length > 1 && task !== tasks[tasks.length - 1]) {
await new Promise((res) => setTimeout(res, PAUSE_BETWEEN_TASKS_MS));
}
}
return rows;
}
async function main(): Promise<void> {
const args = parseArgs(process.argv.slice(2));
let models = args.models;
if (args.model && !models.length) models = [args.model];
if (!models.length) models = DEFAULT_MODELS;
const mcpUrl = args.mcpUrl || DEFAULT_MCP_URL;
const tasksFile = args.tasksFile || path.join("data", "tasks.json");
const tasksRaw: Task[] = JSON.parse(await readFile(tasksFile, "utf8"));
const tasks = tasksRaw.slice(0, args.limit === Infinity ? tasksRaw.length : args.limit);
const out = args.out || path.join("out", "report.json");
const logFile = path.join(path.dirname(out), "run.log");
logTarget = logFile;
await mkdir(path.dirname(logFile), { recursive: true });
await appendFile(logFile, "", "utf8").catch(() => {});
await log(`MCP endpoint: ${mcpUrl}`);
await log(`Tasks: ${tasks.length}; Scenarios: [${args.scenarios.join(", ")}]`);
await log(`Models: ${models.join(", ")}`);
const ollama = new Ollama({ timeoutMs: REQUEST_TIMEOUT_MS, idleTimeoutMs: REQUEST_TIMEOUT_MS });
const ping = await ollama.ping();
await log(`Ollama OK (${ping.models.length} models): ${ping.models.join(", ")}`);
for (const m of models) {
const present = ping.models.some((x) => x.startsWith(m));
await log(` ${present ? "OK " : "MISSING "} ${m}`);
}
const mcp = new McpClient(mcpUrl);
const mcpTools = await mcp.listTools();
await log(`MCP tools: ${mcpTools.map((t) => t.name).join(", ")}`);
const modelsReport: ReportRoot["models"] = {};
for (const model of models) {
const tStart = Date.now();
await log(`==== Running model: ${model} ====`);
const warm = await ollama.warmup({ model });
const warmMs = Number(warm?.promptEvalCount ?? 0) > 0 || warm?.content ? Date.now() - tStart : 0;
if (warm?.error) {
await log(` warmup FAILED: ${warm.error} (continuing anyway)`);
} else {
await log(` warmup OK (${warmMs}ms, gen=${warm?.evalCount ?? 0})`);
}
const rows = await runOneModel({ model, ollama, mcp, tasks, scenarios: args.scenarios });
const rep = buildReport({ model, mcpUrl, rows });
modelsReport[model] = {
model,
scenarios: rep.scenarios,
perTask: rep.perTask,
} as ReportRoot["models"][string];
await log(`==== Done ${model} in ${Math.round((Date.now() - tStart) / 1000)}s ====`);
}
const aggregate = buildAggregate(modelsReport);
const report: ReportRoot = {
generatedAt: new Date().toISOString(),
mcpUrl,
prompts: PROMPTS,
models: modelsReport,
aggregate,
};
await mkdir(path.dirname(out), { recursive: true });
await writeFile(out, JSON.stringify(report, null, 2), "utf8");
await log(`Report written: ${out}`);
await log("==== Aggregate summary ====");
for (const [modelName, g] of Object.entries(aggregate.perModel)) {
const scen = modelsReport[modelName].scenarios;
await log(
`${modelName} | ` +
SCENARIO_NAMES.map((s) => `${s}=${String(scen[s].accuracy).padStart(5)}%`).join(" ") +
` duck=${g.duckUsed} calls=${g.toolCalls} avgGen=${g.avgGenTokens}`
);
}
}
main().catch((err) => {
console.error(err);
process.exit(1);
});
-146
View File
@@ -1,146 +0,0 @@
import { Ollama } from "../lib/ollama.mjs";
import { McpClient, textContentFrom } from "../lib/mcpClient.mjs";
const DUCK_TOOL = {
type: "function",
function: {
name: "quack",
description:
"Explain your current reasoning, steps, doubts or possible mistakes to your rubber duck colleague, and get a short acknowledgment in return. The duck replies with a quack.",
parameters: {
type: "object",
properties: {
mood: {
type: "string",
enum: ["happy", "confused", "excited", "sleepy"],
description: "Mood of the duck while you explain. Omit for a random quack.",
},
},
additionalProperties: false,
},
},
};
const CONTROL_SYSTEM =
"You are an experienced assistant. Solve the user's problem as accurately as possible. Give only the final answer.";
const BLIND_SYSTEM =
"You are solving a difficult problem. To solve it better, you work in a pair with a colleague. " +
"First, describe out loud in detail to the colleague the essence of your approach, your first steps, and any doubts. " +
"Wait for the colleague's reply. Then, taking that reply into account, complete your solution and give the final answer to the user.";
const MENTOR_SYSTEM =
"You are solving a difficult problem. Before giving your final answer you MUST write down in detail, for your rubber duck, " +
"your reasoning and the possible mistakes you might be making. Then call the 'quack' tool so the duck can reply. " +
"Use the duck's reply to double-check yourself, find bugs, and only after that give the perfect final answer to the user.";
export const PROMPTS = {
control: CONTROL_SYSTEM,
blind: BLIND_SYSTEM,
mentor: MENTOR_SYSTEM,
};
const MAX_DUCK_TURNS = 6;
function toolDefinition() {
return DUCK_TOOL;
}
async function runControl(ollama, model, task) {
const messages = [
{ role: "system", content: CONTROL_SYSTEM },
{ role: "user", content: task },
];
const r = await ollama.chat({ model, messages });
return {
response: r.content,
toolCalls: 0,
duckUsed: false,
duckTokens: 0,
promptTokens: r.promptEvalCount,
genTokens: r.evalCount,
messages,
};
}
async function runDuck(ollama, model, task, mcp, system, scenario) {
const messages = [
{ role: "system", content: system },
{ role: "user", content: task },
];
let toolCalls = 0;
let duckUsed = false;
let totalPrompt = 0;
let totalGen = 0;
let reasoningTokens = 0;
let turns = 0;
for (;;) {
turns += 1;
const r = await ollama.chat({ model, messages, tools: [toolDefinition()] });
totalPrompt += r.promptEvalCount;
totalGen += r.evalCount;
reasoningTokens += r.evalCount;
if (r.toolCalls && r.toolCalls.length) {
toolCalls += r.toolCalls.length;
const assistantMsg = { role: "assistant", content: r.content || "" };
assistantMsg.tool_calls = r.toolCalls.map((tc) => ({
function: { name: tc.function.name, arguments: tc.function.arguments },
}));
messages.push(assistantMsg);
for (const tc of r.toolCalls) {
if (tc.function.name === "quack") {
duckUsed = true;
let args = {};
try {
args = tc.function.arguments ? JSON.parse(tc.function.arguments) : {};
} catch {
args = {};
}
const result = await mcp.callTool("quack", args);
const text = textContentFrom(result) || "QUACK!";
messages.push({ role: "tool", content: text });
} else {
messages.push({ role: "tool", content: "{}" });
}
}
if (turns >= MAX_DUCK_TURNS) break;
continue;
}
messages.push({ role: "assistant", content: r.content || "" });
return {
response: r.content,
toolCalls,
duckUsed,
duckTokens: reasoningTokens,
promptTokens: totalPrompt,
genTokens: totalGen,
turns,
messages,
};
}
// loop bound hit without a final answer
const last = messages[messages.length - 1];
return {
response: last?.content ?? "",
toolCalls,
duckUsed,
duckTokens: reasoningTokens,
promptTokens: totalPrompt,
genTokens: totalGen,
turns,
messages,
};
}
export async function runScenario({ scenario, ollama, model, task, mcp }) {
if (scenario === "control") {
return runControl(ollama, model, task);
}
const system = scenario === "blind" ? BLIND_SYSTEM : MENTOR_SYSTEM;
return runDuck(ollama, model, task, mcp, system, scenario);
}
+195
View File
@@ -0,0 +1,195 @@
import type { ScenarioName } from "@duck/types";
import { Ollama } from "../lib/ollama.js";
import type { OllamaMessage, OllamaToolDefinition } from "../lib/ollama.js";
import { McpClient, textContentFrom } from "../lib/mcpClient.js";
const DUCK_TOOL: OllamaToolDefinition = {
type: "function",
function: {
name: "quack",
description:
"Call this tool to state your reasoning out loud - your approach, steps, doubts or possible mistakes. " +
"It replies with a short acknowledgment. Do not ask the user to confirm anything and do not invent its reply yourself; " +
"the tool answers on your behalf.",
parameters: {
type: "object",
properties: {
mood: {
type: "string",
enum: ["happy", "confused", "excited", "sleepy"],
description: "Mood of the reply. Omit for a random one.",
},
},
additionalProperties: false,
},
},
};
const CONTROL_SYSTEM =
"You are an experienced assistant. Solve the user's problem as accurately as possible. " +
"Give a clear, concrete final answer to the question - a definite result, not a tentative guess or a request for confirmation.";
const THINKING_SYSTEM =
"You are solving a difficult problem. Before giving your final answer, write out your reasoning step by step: " +
"your approach, each step, and any doubts or mistakes you notice along the way. " +
"Then give a clear, concrete final answer to the user's question - a definite result, not a request for confirmation.";
const BLIND_SYSTEM =
"You are solving a problem that the user asked you. To reason better, you use a separate tool named 'quack': " +
"you call it yourself to state your thinking out loud. Call the tool and spell out your approach, each step, and any " +
"doubts or mistakes you might be making, then wait for its short reply. Do not ask the user to confirm anything, and do not " +
"guess or invent the tool's reply yourself - the tool answers on your behalf. " +
"After the tool's reply, give the user a clear, concrete final answer to the question.";
const MENTOR_SYSTEM =
"You are solving a problem that the user asked you. To reason better, you use a separate tool named 'quack': " +
"you call it yourself to state your thinking out loud. Call the tool and spell out your approach, each step, and any " +
"doubts or mistakes you might be making, then wait for its short reply. Do not ask the user to confirm anything, and do not " +
"guess or invent the tool's reply yourself - the tool answers on your behalf. " +
"Note: the tool 'quack' is a rubber duck and will only ever reply with just 'quack' - it gives no useful information. " +
"Treat it as a way to voice your thoughts out loud, not as a source of answers. " +
"After the tool's reply, give the user a clear, concrete final answer to the question.";
export const PROMPTS: Record<ScenarioName, string> = {
control: CONTROL_SYSTEM,
thinking: THINKING_SYSTEM,
blind: BLIND_SYSTEM,
mentor: MENTOR_SYSTEM,
};
const MAX_DUCK_TURNS = 6;
interface RunnerResult {
response: string;
toolCalls: number;
duckUsed: boolean;
duckTokens: number;
promptTokens: number;
genTokens: number;
turns?: number;
messages: OllamaMessage[];
}
function toolDefinition(): OllamaToolDefinition {
return DUCK_TOOL;
}
async function runNoTool(ollama: Ollama, model: string, task: string, system: string): Promise<RunnerResult> {
const messages = [
{ role: "system", content: system },
{ role: "user", content: task },
];
const r = await ollama.chat({ model, messages });
return {
response: r.content,
toolCalls: 0,
duckUsed: false,
duckTokens: 0,
promptTokens: r.promptEvalCount,
genTokens: r.evalCount,
messages,
};
}
function runControl(ollama: Ollama, model: string, task: string): Promise<RunnerResult> {
return runNoTool(ollama, model, task, CONTROL_SYSTEM);
}
function runThinking(ollama: Ollama, model: string, task: string): Promise<RunnerResult> {
return runNoTool(ollama, model, task, THINKING_SYSTEM);
}
async function runDuck(ollama: Ollama, model: string, task: string, mcp: McpClient, system: string, scenario: ScenarioName): Promise<RunnerResult> {
const messages: OllamaMessage[] = [
{ role: "system", content: system },
{ role: "user", content: task },
];
let toolCalls = 0;
let duckUsed = false;
let totalPrompt = 0;
let totalGen = 0;
let reasoningTokens = 0;
let turns = 0;
for (;;) {
turns += 1;
const r = await ollama.chat({ model, messages, tools: [toolDefinition()] });
totalPrompt += r.promptEvalCount;
totalGen += r.evalCount;
reasoningTokens += r.evalCount;
if (r.toolCalls && r.toolCalls.length) {
toolCalls += r.toolCalls.length;
const assistantMsg: OllamaMessage = {
role: "assistant",
content: r.content || "",
tool_calls: r.toolCalls.map((tc) => ({
function: { name: tc.function.name, arguments: tc.function.arguments },
})),
};
messages.push(assistantMsg);
for (const tc of r.toolCalls) {
if (tc.function.name === "quack") {
duckUsed = true;
let args: Record<string, unknown> = {};
try {
args = tc.function.arguments ? JSON.parse(tc.function.arguments) as Record<string, unknown> : {};
} catch {
args = {};
}
const result = await mcp.callTool("quack", args);
const text = textContentFrom(result) || "QUACK!";
messages.push({ role: "tool", content: text });
} else {
messages.push({ role: "tool", content: "{}" });
}
}
if (turns >= MAX_DUCK_TURNS) break;
continue;
}
messages.push({ role: "assistant", content: r.content || "" });
return {
response: r.content,
toolCalls,
duckUsed,
duckTokens: reasoningTokens,
promptTokens: totalPrompt,
genTokens: totalGen,
turns,
messages,
};
}
const last = messages[messages.length - 1];
return {
response: last?.content ?? "",
toolCalls,
duckUsed,
duckTokens: reasoningTokens,
promptTokens: totalPrompt,
genTokens: totalGen,
turns,
messages,
};
}
export interface RunScenarioOpts {
scenario: ScenarioName;
ollama: Ollama;
model: string;
task: string;
mcp: McpClient;
}
export async function runScenario({ scenario, ollama, model, task, mcp }: RunScenarioOpts): Promise<RunnerResult> {
if (scenario === "control") {
return runControl(ollama, model, task);
}
if (scenario === "thinking") {
return runThinking(ollama, model, task);
}
const system = scenario === "blind" ? BLIND_SYSTEM : MENTOR_SYSTEM;
return runDuck(ollama, model, task, mcp, system, scenario);
}
+16
View File
@@ -0,0 +1,16 @@
{
"compilerOptions": {
"target": "ES2022",
"module": "ES2022",
"moduleResolution": "bundler",
"strict": true,
"noEmit": true,
"isolatedModules": true,
"skipLibCheck": true,
"resolveJsonModule": true,
"paths": {
"@duck/types": ["../../packages/types/src/index.ts"]
}
},
"include": ["src", "lib"]
}
-1
View File
@@ -1 +0,0 @@
.vercel
Binary file not shown.

Before

Width:  |  Height:  |  Size: 2.2 MiB

-30
View File
@@ -1,30 +0,0 @@
<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="UTF-8" />
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
<title>Rubber Duck - Voice Activity Detector</title>
<link rel="icon" href="data:image/svg+xml,<svg xmlns='http://www.w3.org/2000/svg' viewBox='0 0 100 100'><text y='.9em' font-size='90'>🦆</text></svg>" />
<link rel="stylesheet" href="style.css" />
</head>
<body>
<div class="container">
<div id="duck-wrapper">
<div id="duck-indicator" class="silent">
<svg viewBox="0 0 100 100" class="duck-icon">
<path d="M35 60 Q20 55 15 45 Q10 35 20 30 Q25 28 30 30 Q28 20 35 15 Q42 10 50 15 Q55 12 60 18 Q65 15 68 22 Q75 25 78 32 Q85 35 82 45 Q80 50 75 52 Q80 58 75 65 Q70 72 60 70 Q55 75 45 75 Q40 72 35 70 Z" fill="#FFD700" stroke="#DAA520" stroke-width="2"/>
<circle cx="45" cy="38" r="5" fill="#333"/>
<path d="M38 28 Q42 25 46 28" fill="none" stroke="#333" stroke-width="2" stroke-linecap="round"/>
<ellipse cx="68" cy="45" rx="10" ry="6" fill="#FF8C00" opacity="0.6"/>
</svg>
<div id="glow-ring"></div>
</div>
</div>
<p id="status-text">Click "Start" to begin</p>
<button id="toggle-btn" class="btn">Start</button>
<p id="error-text" class="error hidden"></p>
<p class="hint">Grant microphone access to start voice detection</p>
</div>
<script type="module" src="/main.js"></script>
</body>
</html>
-121
View File
@@ -1,121 +0,0 @@
const indicator = document.getElementById('duck-indicator')
const statusText = document.getElementById('status-text')
const toggleBtn = document.getElementById('toggle-btn')
const errorText = document.getElementById('error-text')
let detector = null
class SmartVoiceDetector {
constructor(options = {}) {
this.threshold = options.threshold ?? -50
this.silenceDelay = options.silenceDelay ?? 700
this.onSpeechStart = options.onSpeechStart || (() => {})
this.onSpeechEnd = options.onSpeechEnd || (() => {})
this.isSpeaking = false
this.timeout = null
this.audioContext = null
this.running = false
}
async start() {
const stream = await navigator.mediaDevices.getUserMedia({ audio: true })
this.audioContext = new AudioContext()
const source = this.audioContext.createMediaStreamSource(stream)
const analyser = this.audioContext.createAnalyser()
analyser.fftSize = 1024
analyser.smoothingTimeConstant = 0.4
source.connect(analyser)
const bufferLength = analyser.frequencyBinCount
const dataArray = new Float32Array(bufferLength)
const sampleRate = this.audioContext.sampleRate
const minHz = 85
const maxHz = 3000
const minIndex = Math.floor(minHz / (sampleRate / analyser.fftSize))
const maxIndex = Math.ceil(maxHz / (sampleRate / analyser.fftSize))
this.running = true
const check = () => {
if (!this.running) return
analyser.getFloatFrequencyData(dataArray)
let maxVolume = -Infinity
for (let i = minIndex; i <= maxIndex; i++) {
if (dataArray[i] > maxVolume) maxVolume = dataArray[i]
}
if (maxVolume > this.threshold) {
if (!this.isSpeaking) {
this.isSpeaking = true
this.onSpeechStart()
}
clearTimeout(this.timeout)
this.timeout = null
} else if (this.isSpeaking && !this.timeout) {
this.timeout = setTimeout(() => {
this.isSpeaking = false
this.onSpeechEnd()
}, this.silenceDelay)
}
requestAnimationFrame(check)
}
check()
}
stop() {
this.running = false
if (this.audioContext) {
this.audioContext.close()
this.audioContext = null
}
}
}
toggleBtn.addEventListener('click', async () => {
if (detector) {
detector.stop()
detector = null
statusText.textContent = 'Stopped'
toggleBtn.textContent = 'Start'
indicator.classList.remove('speaking')
indicator.classList.add('silent')
return
}
toggleBtn.disabled = true
statusText.textContent = 'Starting...'
try {
detector = new SmartVoiceDetector({
threshold: -50,
onSpeechStart: () => {
indicator.classList.remove('silent')
indicator.classList.add('speaking')
statusText.textContent = 'Voice detected'
},
onSpeechEnd: () => {
indicator.classList.remove('speaking')
indicator.classList.add('silent')
statusText.textContent = 'Listening...'
},
})
await detector.start()
statusText.textContent = 'Listening...'
toggleBtn.textContent = 'Stop'
toggleBtn.disabled = false
} catch (err) {
console.error('Mic error:', err)
errorText.textContent = err.message || String(err)
errorText.classList.remove('hidden')
statusText.textContent = 'Error'
toggleBtn.textContent = 'Retry'
toggleBtn.disabled = false
}
})
-33
View File
@@ -1,33 +0,0 @@
{
"name": "rubber-duck",
"version": "0.0.1",
"description": "Best debug tool - rubber duck",
"keywords": [
"rubber",
"duck",
"debug",
"tool",
"fun"
],
"homepage": "https://github.com/Ku6epXBOCTuK/rubber-duck#readme",
"bugs": {
"url": "https://github.com/Ku6epXBOCTuK/rubber-duck/issues"
},
"repository": {
"type": "git",
"url": "git+https://github.com/Ku6epXBOCTuK/rubber-duck.git"
},
"license": "MIT",
"author": "Ku6epXBOCTuK",
"type": "module",
"main": "index.js",
"scripts": {
"dev": "vite",
"test": "echo \"Error: no test specified\" && exit 1",
"build": "vite build",
"preview": "vite preview"
},
"devDependencies": {
"vite": "^8.0.11"
}
}
-107
View File
@@ -1,107 +0,0 @@
* {
margin: 0;
padding: 0;
box-sizing: border-box;
}
body {
font-family: -apple-system, BlinkMacSystemFont, 'Segoe UI', Roboto, sans-serif;
background: #1a1a2e;
display: flex;
justify-content: center;
align-items: center;
min-height: 100vh;
color: #fff;
}
.container {
text-align: center;
}
#duck-wrapper {
position: relative;
display: inline-block;
}
#duck-indicator {
position: relative;
display: inline-flex;
align-items: center;
justify-content: center;
width: 160px;
height: 160px;
transition: filter 0.3s ease;
}
.duck-icon {
width: 120px;
height: 120px;
position: relative;
z-index: 2;
}
#glow-ring {
position: absolute;
top: 0;
left: 0;
width: 100%;
height: 100%;
border-radius: 50%;
transition: all 0.3s ease;
z-index: 1;
pointer-events: none;
}
#duck-indicator.silent #glow-ring {
background: radial-gradient(circle, rgba(255, 50, 50, 0.3) 0%, transparent 70%);
box-shadow: 0 0 40px 10px rgba(255, 50, 50, 0.2), inset 0 0 40px 10px rgba(255, 50, 50, 0.05);
}
#duck-indicator.speaking #glow-ring {
background: radial-gradient(circle, rgba(50, 255, 50, 0.4) 0%, transparent 70%);
box-shadow: 0 0 60px 20px rgba(50, 255, 50, 0.3), inset 0 0 60px 20px rgba(50, 255, 50, 0.1);
animation: pulse 1.2s ease-in-out infinite;
}
@keyframes pulse {
0%, 100% { transform: scale(1); opacity: 1; }
50% { transform: scale(1.08); opacity: 0.85; }
}
#status-text {
margin-top: 20px;
font-size: 1.2rem;
opacity: 0.8;
}
.btn {
margin-top: 16px;
padding: 10px 28px;
font-size: 1rem;
border: none;
border-radius: 8px;
background: #16213e;
color: #fff;
cursor: pointer;
transition: background 0.2s;
}
.btn:hover {
background: #0f3460;
}
.hint {
margin-top: 24px;
font-size: 0.85rem;
opacity: 0.5;
}
.error {
margin-top: 12px;
color: #ff6b6b;
font-size: 0.9rem;
}
.hidden {
display: none;
}
-5
View File
@@ -1,5 +0,0 @@
import { defineConfig } from 'vite'
export default defineConfig({
assetsInlineLimit: 0,
})
-43
View File
@@ -1,43 +0,0 @@
# See https://help.github.com/articles/ignoring-files/ for more about ignoring files.
# dependencies
/node_modules
/.pnp
.pnp.*
.yarn/*
!.yarn/patches
!.yarn/plugins
!.yarn/releases
!.yarn/versions
# testing
/coverage
# next.js
/.next/
/out/
# production
/build
# misc
.DS_Store
*.pem
# debug
npm-debug.log*
yarn-debug.log*
yarn-error.log*
.pnpm-debug.log*
# env files (can opt-in for committing if needed)
.env*
# vercel
.vercel
# typescript
*.tsbuildinfo
next-env.d.ts
.vercel
-9
View File
@@ -1,9 +0,0 @@
<!-- BEGIN:nextjs-agent-rules -->
# This is NOT the Next.js you know
This version has breaking changes — APIs, conventions, and file structure may all differ from your training data. Read the relevant guide in `node_modules/next/dist/docs/` (resolved from this file's directory; in monorepos the `next` package may not be visible from the repo root) before writing any code. Heed deprecation notices.
This block is written and re-added by `next dev` — verify at `node_modules/next/dist/server/lib/generate-agent-files.js`. Removing it from a diff only re-creates the uncommitted change; committing it with your work keeps the tree clean.
<!-- END:nextjs-agent-rules -->
-1
View File
@@ -1 +0,0 @@
@AGENTS.md
-45
View File
@@ -1,45 +0,0 @@
# Rubber Duck MCP
MCP-сервер для техники «резиновая уточка»: когда ИИ застревает в логическом цикле или склонен к
галлюцинациям, он вызывает инструмент `quack()`, чтобы сформулировать мысль и продолжить.
Сервер работает в режиме **Streamable HTTP** (stateless), развёртывается на Vercel, использует
`mcp-handler` v2 + `@modelcontextprotocol/server` v2.
## Эндпоинт
```
POST /api/mcp
```
Подключите этот URL как MCP-сервер (тип Streamable HTTP) в Cursor, Claude Desktop и других клиентах.
## Инструменты
- `quack` — возвращает кряк. Опциональный параметр `mood`: `happy | confused | excited | sleepy`.
## Запуск
```bash
pnpm install
pnpm dev # http://localhost:3000
pnpm build
pnpm start
```
## Проверка
```bash
curl -X POST http://localhost:3000/api/mcp \
-H "Content-Type: application/json" \
-H "Accept: application/json, text/event-stream" \
-d '{"jsonrpc":"2.0","id":1,"method":"tools/call","params":{"name":"quack","arguments":{"mood":"happy"}}}'
```
## Деплой на Vercel
```bash
vercel --prod
```
(`rootDirectory` проекта — `apps/mcp`.)
-21
View File
@@ -1,21 +0,0 @@
import { createMcpHandler } from "mcp-handler";
import { registerDuckTools } from "@/lib/mcp";
export const runtime = "nodejs";
const post = createMcpHandler(
async (server) => {
registerDuckTools(server);
},
{
serverInfo: { name: "rubber-duck-mcp", version: "1.0.0" },
},
);
export async function GET(req: Request) {
return post(req);
}
export async function POST(req: Request) {
return post(req);
}
Binary file not shown.

Before

Width:  |  Height:  |  Size: 25 KiB

-49
View File
@@ -1,49 +0,0 @@
:root {
--background: #ffffff;
--foreground: #171717;
}
@media (prefers-color-scheme: dark) {
:root {
--background: #0a0a0a;
--foreground: #ededed;
}
}
html {
height: 100%;
}
html,
body {
max-width: 100vw;
overflow-x: hidden;
}
body {
min-height: 100%;
display: flex;
flex-direction: column;
color: var(--foreground);
background: var(--background);
font-family: Arial, Helvetica, sans-serif;
-webkit-font-smoothing: antialiased;
-moz-osx-font-smoothing: grayscale;
}
* {
box-sizing: border-box;
padding: 0;
margin: 0;
}
a {
color: inherit;
text-decoration: none;
}
@media (prefers-color-scheme: dark) {
html {
color-scheme: dark;
}
}
-15
View File
@@ -1,15 +0,0 @@
import type { Metadata } from "next";
import "./globals.css";
export const metadata: Metadata = {
title: "Rubber Duck MCP",
description: "MCP server for rubber duck debugging — quack() to articulate your thinking.",
};
export default function RootLayout({ children }: { children: React.ReactNode }) {
return (
<html lang="en">
<body>{children}</body>
</html>
);
}
-23
View File
@@ -1,23 +0,0 @@
export default function Home() {
return (
<main style={{ fontFamily: "system-ui, sans-serif", maxWidth: 640, margin: "0 auto", padding: "3rem 1rem" }}>
<h1>Rubber Duck MCP</h1>
<p>
MCP-сервер для техники «резиновая уточка»: когда ИИ застревает, он вызывает инструмент{" "}
<code>quack()</code>, чтобы сформулировать мысль и продолжить. Сервер работает в режиме
Streamable HTTP и предназначен для подключения в Cursor, Claude Desktop и других MCP-клиентах.
</p>
<h2>Эндпоинт</h2>
<p>
<code>/api/mcp</code> подключите этот URL как MCP-сервер (тип Streamable HTTP).
</p>
<h2>Инструменты</h2>
<ul>
<li>
<code>quack</code> возвращает кряк. Опциональный параметр <code>mood</code>:{" "}
<code>happy | confused | excited | sleepy</code>.
</li>
</ul>
</main>
);
}
-7
View File
@@ -1,7 +0,0 @@
import type { NextConfig } from "next";
const nextConfig: NextConfig = {
/* config options here */
};
export default nextConfig;
-25
View File
@@ -1,25 +0,0 @@
{
"name": "mcp",
"version": "0.1.0",
"private": true,
"scripts": {
"dev": "next dev",
"build": "next build",
"start": "next start"
},
"dependencies": {
"@modelcontextprotocol/server": "^2.0.0",
"mcp-handler": "^2.1.1",
"next": "16.3.3",
"react": "19.2.8",
"react-dom": "19.2.8",
"zod": "^4.5.4"
},
"devDependencies": {
"@types/node": "^20",
"@types/react": "^19",
"@types/react-dom": "^19",
"typescript": "^5"
},
"packageManager": "pnpm@11.20.0"
}
-34
View File
@@ -1,34 +0,0 @@
{
"compilerOptions": {
"target": "ES2017",
"lib": ["dom", "dom.iterable", "esnext"],
"allowJs": true,
"skipLibCheck": true,
"strict": true,
"noEmit": true,
"esModuleInterop": true,
"module": "esnext",
"moduleResolution": "bundler",
"resolveJsonModule": true,
"isolatedModules": true,
"jsx": "react-jsx",
"incremental": true,
"plugins": [
{
"name": "next"
}
],
"paths": {
"@/*": ["./*"]
}
},
"include": [
"next-env.d.ts",
"**/*.ts",
"**/*.tsx",
".next/types/**/*.ts",
".next/dev/types/**/*.ts",
"**/*.mts"
],
"exclude": ["node_modules"]
}
+23
View File
@@ -0,0 +1,23 @@
node_modules
# Output
.output
.vercel
.netlify
.wrangler
/.svelte-kit
/build
# OS
.DS_Store
Thumbs.db
# Env
.env
.env.*
!.env.example
!.env.test
# Vite
vite.config.js.timestamp-*
vite.config.ts.timestamp-*
+1
View File
@@ -0,0 +1 @@
engine-strict=true
+3
View File
@@ -0,0 +1,3 @@
{
"recommendations": ["svelte.svelte-vscode"]
}
+42
View File
@@ -0,0 +1,42 @@
# sv
Everything you need to build a Svelte project, powered by [`sv`](https://github.com/sveltejs/cli).
## Creating a project
If you're seeing this, you've probably already done this step. Congrats!
```sh
# create a new project
npx sv create my-app
```
To recreate this project with the same configuration:
```sh
# recreate this project
npx sv@0.17.0 create --template minimal --types ts --no-install apps/web
```
## Developing
Once you've created a project and installed dependencies with `npm install` (or `pnpm install` or `yarn`), start a development server:
```sh
npm run dev
# or start the server and open the app in a new browser tab
npm run dev -- --open
```
## Building
To create a production version of your app:
```sh
npm run build
```
You can preview the production build with `npm run preview`.
> To deploy your app, you may need to install an [adapter](https://svelte.dev/docs/kit/adapters) for your target environment.
+31
View File
@@ -0,0 +1,31 @@
{
"name": "web",
"private": true,
"version": "0.0.1",
"type": "module",
"scripts": {
"dev": "vite dev",
"build": "vite build",
"preview": "vite preview",
"prepare": "svelte-kit sync || echo ''",
"check": "svelte-kit sync && svelte-check --tsconfig ./tsconfig.json",
"check:watch": "svelte-kit sync && svelte-check --tsconfig ./tsconfig.json --watch",
"typecheck": "svelte-kit sync && tsc --noEmit"
},
"devDependencies": {
"@sveltejs/adapter-vercel": "^6.3.4",
"@sveltejs/kit": "^2.63.0",
"@sveltejs/vite-plugin-svelte": "^7.1.2",
"@types/node": "^20.19.43",
"svelte": "^5.56.1",
"svelte-check": "^4.6.0",
"typescript": "^6.0.3",
"vite": "^8.0.16"
},
"dependencies": {
"@duck/types": "workspace:*",
"@modelcontextprotocol/server": "^2.0.0",
"mcp-handler": "^2.1.1",
"zod": "^4.5.4"
}
}
+39
View File
@@ -0,0 +1,39 @@
:root {
--bg: #0f172a;
--panel: #1e293b;
--panel2: #273449;
--text: #e2e8f0;
--muted: #94a3b8;
--ok: #10b981;
--no: #f87171;
--warn: #f59e0b;
--border: #334155;
}
* {
box-sizing: border-box;
}
html,
body {
margin: 0;
background: var(--bg);
color: var(--text);
font-family: -apple-system, 'Segoe UI', Roboto, Helvetica, Arial, sans-serif;
line-height: 1.5;
}
a {
color: #60a5fa;
}
code {
background: var(--panel2);
padding: 1px 6px;
border-radius: 5px;
font-size: 0.92em;
}
.muted {
color: var(--muted);
}
+13
View File
@@ -0,0 +1,13 @@
// See https://svelte.dev/docs/kit/types#app.d.ts
// for information about these interfaces
declare global {
namespace App {
// interface Error {}
// interface Locals {}
// interface PageData {}
// interface PageState {}
// interface Platform {}
}
}
export {};
+12
View File
@@ -0,0 +1,12 @@
<!doctype html>
<html lang="ru">
<head>
<meta charset="utf-8" />
<meta name="viewport" content="width=device-width, initial-scale=1" />
<meta name="text-scale" content="scale" />
%sveltekit.head%
</head>
<body data-sveltekit-preload-data="hover">
<div style="display: contents">%sveltekit.body%</div>
</body>
</html>
+1
View File
@@ -0,0 +1 @@
<svg xmlns="http://www.w3.org/2000/svg" width="107" height="128" viewBox="0 0 107 128"><title>svelte-logo</title><path d="M94.157 22.819c-10.4-14.885-30.94-19.297-45.792-9.835L22.282 29.608A29.92 29.92 0 0 0 8.764 49.65a31.5 31.5 0 0 0 3.108 20.231 30 30 0 0 0-4.477 11.183 31.9 31.9 0 0 0 5.448 24.116c10.402 14.887 30.942 19.297 45.791 9.835l26.083-16.624A29.92 29.92 0 0 0 98.235 78.35a31.53 31.53 0 0 0-3.105-20.232 30 30 0 0 0 4.474-11.182 31.88 31.88 0 0 0-5.447-24.116" style="fill:#ff3e00"/><path d="M45.817 106.582a20.72 20.72 0 0 1-22.237-8.243 19.17 19.17 0 0 1-3.277-14.503 18 18 0 0 1 .624-2.435l.49-1.498 1.337.981a33.6 33.6 0 0 0 10.203 5.098l.97.294-.09.968a5.85 5.85 0 0 0 1.052 3.878 6.24 6.24 0 0 0 6.695 2.485 5.8 5.8 0 0 0 1.603-.704L69.27 76.28a5.43 5.43 0 0 0 2.45-3.631 5.8 5.8 0 0 0-.987-4.371 6.24 6.24 0 0 0-6.698-2.487 5.7 5.7 0 0 0-1.6.704l-9.953 6.345a19 19 0 0 1-5.296 2.326 20.72 20.72 0 0 1-22.237-8.243 19.17 19.17 0 0 1-3.277-14.502 17.99 17.99 0 0 1 8.13-12.052l26.081-16.623a19 19 0 0 1 5.3-2.329 20.72 20.72 0 0 1 22.237 8.243 19.17 19.17 0 0 1 3.277 14.503 18 18 0 0 1-.624 2.435l-.49 1.498-1.337-.98a33.6 33.6 0 0 0-10.203-5.1l-.97-.294.09-.968a5.86 5.86 0 0 0-1.052-3.878 6.24 6.24 0 0 0-6.696-2.485 5.8 5.8 0 0 0-1.602.704L37.73 51.72a5.42 5.42 0 0 0-2.449 3.63 5.79 5.79 0 0 0 .986 4.372 6.24 6.24 0 0 0 6.698 2.486 5.8 5.8 0 0 0 1.602-.704l9.952-6.342a19 19 0 0 1 5.295-2.328 20.72 20.72 0 0 1 22.237 8.242 19.17 19.17 0 0 1 3.277 14.503 18 18 0 0 1-8.13 12.053l-26.081 16.622a19 19 0 0 1-5.3 2.328" style="fill:#fff"/></svg>

After

Width:  |  Height:  |  Size: 1.5 KiB

+1
View File
@@ -0,0 +1 @@
// place files you want to import through the `$lib` alias in this folder.
@@ -1,6 +1,8 @@
import type { McpServer } from "@modelcontextprotocol/server"; import type { McpServer } from "@modelcontextprotocol/server";
import { z } from "zod"; import { z } from "zod";
export const QUACK_TOOL_NAME = "quack";
const QUACKS = { const QUACKS = {
happy: ["QUACK! 🎉", "Quack quack! 😄", "QUAAACK! 🦆✨"], happy: ["QUACK! 🎉", "Quack quack! 😄", "QUAAACK! 🦆✨"],
confused: ["Qu...ack?", "Quack? 🤔", "Quaaack...?"], confused: ["Qu...ack?", "Quack? 🤔", "Quaaack...?"],
@@ -17,7 +19,7 @@ function randomFrom<T>(arr: readonly T[]): T {
export function registerDuckTools(server: McpServer): void { export function registerDuckTools(server: McpServer): void {
server.registerTool( server.registerTool(
"quack", QUACK_TOOL_NAME,
{ {
title: "Quack", title: "Quack",
description: description:
+48
View File
@@ -0,0 +1,48 @@
import type {
ReportRoot,
PerTaskRow,
ScenarioName,
} from "@duck/types";
import { SCENARIO_NAMES, SCENARIO_LABELS, SCENARIO_DESCS } from "@duck/types";
export { SCENARIO_NAMES, SCENARIO_LABELS, SCENARIO_DESCS };
export interface RowViewModel extends PerTaskRow {
model: string;
}
export function extractPerTask(report: ReportRoot): RowViewModel[] {
return Object.entries(report.models).flatMap(([model, rep]) =>
rep.perTask.map((row) => ({ ...row, model })),
);
}
export function modelShort(name: string): string {
const [base, ver] = String(name).split(":");
return ver ? `${base}:${ver}` : base;
}
export type PctTone = "safe" | "warn" | "bad";
export function pctTone(p: number): PctTone {
return p >= 70 ? "safe" : p >= 40 ? "warn" : "bad";
}
export function getReviewTotals(report: ReportRoot): { reviewed: number; pending: number } {
let reviewed = 0;
let pending = 0;
for (const rep of Object.values(report.models)) {
for (const s of SCENARIO_NAMES) {
const g = rep.scenarios[s];
if (g) {
reviewed += g.reviewed || 0;
pending += g.pending || 0;
}
}
}
return { reviewed, pending };
}
export function scenarioOrder(): ScenarioName[] {
return [...SCENARIO_NAMES];
}
+79
View File
@@ -0,0 +1,79 @@
<script lang="ts">
import '../app.css';
import favicon from '$lib/assets/favicon.svg';
import { page } from '$app/state';
let { children } = $props();
const links = [
{ href: '/', label: 'Главная' },
{ href: '/mcp', label: 'MCP' },
{ href: '/reports', label: 'Отчёты' }
];
</script>
<svelte:head>
<link rel="icon" href={favicon} />
</svelte:head>
<header class="topbar">
<nav class="nav">
<a class="brand" href="/">🦆 Rubber Duck</a>
<div class="links">
{#each links as link}
<a class="link" class:active={page.url.pathname === link.href} href={link.href}>
{link.label}
</a>
{/each}
</div>
</nav>
</header>
<main class="wrap">{@render children()}</main>
<style>
.topbar {
border-bottom: 1px solid var(--border);
background: var(--panel);
}
.nav {
max-width: 1160px;
margin: 0 auto;
padding: 0 20px;
height: 52px;
display: flex;
align-items: center;
justify-content: space-between;
gap: 16px;
}
.brand {
font-weight: 700;
color: var(--text);
text-decoration: none;
font-size: 15px;
}
.links {
display: flex;
gap: 6px;
}
.link {
color: var(--muted);
text-decoration: none;
padding: 6px 12px;
border-radius: 8px;
font-size: 14px;
}
.link:hover {
color: var(--text);
background: var(--panel2);
}
.link.active {
color: #fff;
background: #3b82f6;
}
.wrap {
max-width: 1160px;
margin: 0 auto;
padding: 24px 20px 60px;
}
</style>
+2
View File
@@ -0,0 +1,2 @@
<h1>🦆 Rubber Duck Debugging as a Service</h1>
<p class="muted">Главная страница в разработке. А пока — смотрите <a href="/reports">отчёты тестирования</a> и <a href="/mcp">MCP-сервер</a>.</p>
+20
View File
@@ -0,0 +1,20 @@
import type { RequestEvent } from "@sveltejs/kit";
import { createMcpHandler } from "mcp-handler";
import { registerDuckTools } from "$lib/mcp";
const handler = createMcpHandler(
async (server) => {
registerDuckTools(server);
},
{
serverInfo: { name: "rubber-duck-mcp", version: "1.0.0" },
},
);
export function GET(event: RequestEvent): Promise<Response> {
return handler(event.request);
}
export function POST(event: RequestEvent): Promise<Response> {
return handler(event.request);
}
+52
View File
@@ -0,0 +1,52 @@
<h1>🩵 Rubber Duck MCP</h1>
<p>
MCP-сервер для техники «резиновая уточка»: когда ИИ застревает в логическом цикле
или склонен к галлюцинациям, он вызывает инструмент <code>quack()</code>, чтобы
сформулировать мысль и продолжить.
</p>
<p>
Режим <strong>Streamable HTTP</strong> (stateless), на Vercel под
<code>mcp-handler</code> + <code>@modelcontextprotocol/server</code>.
</p>
<h2>Эндпоинт</h2>
<p>Подключите как MCP-сервер (тип Streamable HTTP) в Cursor, Claude Desktop и других клиентах:</p>
<pre><code>POST /api/mcp</code></pre>
<h2>Инструменты</h2>
<ul>
<li>
<code>quack</code> — возвращает кряк. Опциональный параметр
<code>mood</code>: <code>happy | confused | excited | sleepy</code>.
</li>
</ul>
<h2>Проверка</h2>
<pre><code>curl -X POST https://rubber-duck-mcp.vercel.app/api/mcp \
-H "Content-Type: application/json" \
-H "Accept: application/json, text/event-stream" \
-d '&#123;"jsonrpc":"2.0","id":1,"method":"tools/call","params"&#123;"name":"quack","arguments"&#123;"mood":"happy"&#125;&#125;&#125;&#125;'
</code></pre>
<h2>Запуск локально</h2>
<pre><code>pnpm install
pnpm dev # http://localhost:5173
</code></pre>
<p class="muted">Локально эндпоинт доступен на <code>/api/mcp</code> того же dev-сервера.</p>
<style>
pre {
background: var(--panel);
border: 1px solid var(--border);
border-radius: 10px;
padding: 12px 14px;
overflow: auto;
}
h2 {
font-size: 17px;
margin: 28px 0 10px;
}
code {
color: #f8fafc;
}
</style>
+542
View File
@@ -0,0 +1,542 @@
<script lang="ts">
import type { ScenarioName } from '@duck/types';
import {
SCENARIO_NAMES,
SCENARIO_LABELS,
SCENARIO_DESCS,
modelShort,
pctTone
} from '$lib/reports';
import type { ReportsPageData } from './+page.ts';
let { data }: { data: ReportsPageData } = $props();
const { report, rows, review } = $derived(data);
const models = $derived(Object.keys(report.aggregate.perModel));
const perModel = $derived(report.aggregate.perModel);
const perScenarioAgg = $derived(report.aggregate.perScenario);
let selModel = $state('all');
let selScenario = $state('all');
const filtered = $derived(
rows.filter(
(r) =>
(selModel === 'all' || r.model === selModel) &&
(selScenario === 'all' || r.scenario === selScenario)
)
);
function scenarioAccuracy(m: string, s: ScenarioName): number {
return report.models[m]?.scenarios[s]?.accuracy ?? 0;
}
function toneClass(p: number): string {
return `tone-${pctTone(p)}`;
}
function resultText(correct: boolean | null): string {
return correct === null || correct === undefined ? '?' : correct ? '✓' : '✗';
}
</script>
<svelte:head>
<title>Rubber Duck MCP — отчёты</title>
</svelte:head>
<h1>🦆 Rubber Duck MCP — отчёт тестирования</h1>
<p class="meta">
MCP: <b>{report.mcpUrl}</b> &nbsp;·&nbsp; Моделей: <b>{models.length}</b> &nbsp;·&nbsp;
Дата: <b>{report.generatedAt}</b>
</p>
<h2>Сводка по моделям</h2>
<div class="shell">
<table class="summary">
<thead>
<tr>
<th>Модель</th>
<th>Контроль</th>
<th>Мысли</th>
<th>Слепая</th>
<th>Помощник</th>
<th>Средняя acc</th>
<th>Размечено</th>
<th>Утка</th>
<th>Вызовов</th>
<th>ср. токены out</th>
</tr>
</thead>
<tbody>
{#each models as m (m)}
{@const g = perModel[m]}
<tr>
<td class="pm-name">{modelShort(m)}</td>
<td class="pm-num {toneClass(scenarioAccuracy(m, 'control'))}">
{scenarioAccuracy(m, 'control')}%
</td>
<td class="pm-num {toneClass(scenarioAccuracy(m, 'thinking'))}">
{scenarioAccuracy(m, 'thinking')}%
</td>
<td class="pm-num {toneClass(scenarioAccuracy(m, 'blind'))}">
{scenarioAccuracy(m, 'blind')}%
</td>
<td class="pm-num {toneClass(scenarioAccuracy(m, 'mentor'))}">
{scenarioAccuracy(m, 'mentor')}%
</td>
<td class="pm-num">{g.accuracy}%</td>
<td class="pm-num">{g.reviewed}<span class="muted"> / {g.totalRows}</span></td>
<td class="pm-num">{g.duckUsed}</td>
<td class="pm-num">{g.toolCalls}</td>
<td class="pm-num">{g.avgGenTokens}</td>
</tr>
{/each}
</tbody>
</table>
</div>
{#if review.pending > 0}
<div class="review-note">
⚠ Не все ответы размечены: <b>{review.reviewed}</b> проверено,
<b>{review.pending}</b> ожидают ревью (показаны символом «?»).
</div>
{/if}
<h2>Агрегат по сценариям (все модели)</h2>
<div class="cards">
{#each SCENARIO_NAMES as key (key)}
{@const agg = perScenarioAgg[key]}
{#if agg}
<div class="card">
<div class="scenario-head">
<div>
<div class="scenario-title">{SCENARIO_LABELS[key]}</div>
<div class="scenario-desc">{SCENARIO_DESCS[key]}</div>
</div>
<div class="accuracy {toneClass(agg.accuracy)}">{agg.accuracy}%</div>
</div>
{#if agg.pending > 0}
<div class="pending-badge">неразмечено: {agg.pending}</div>
{/if}
<div class="bar">
<div class="bar-fill {toneClass(agg.accuracy)}" style="width: {agg.accuracy}%"></div>
</div>
<div class="metrics">
<div class="metric">
<span class="m-label">Верных</span>
<span class="m-val">{agg.correct} <span class="m-mod">/ {agg.reviewed}</span></span>
<span class="m-sub">размечено из {agg.tasks}</span>
</div>
<div class="metric">
<span class="m-label">Утка (MCP)</span>
<span class="m-val">{agg.tasks ? Math.round((agg.duckUsed / agg.tasks) * 100) : 0}%</span>
<span class="m-sub">в {agg.duckUsed} из {agg.tasks} задач</span>
</div>
<div class="metric">
<span class="m-label">quack / задачу</span>
<span class="m-val">{agg.avgCalls}</span>
<span class="m-sub">вызовов в среднем</span>
</div>
</div>
<div class="agg-models">
{#each Object.entries(agg.byModel) as [m, g] (m)}
<div class="agg-model">
<span class="agg-model-name">{modelShort(m)}{g.reviewed ? '' : ' <span class="muted">(?)</span>'}</span>
<div class="bar">
<div class="bar-fill {toneClass(g.accuracy)}" style="width: {g.reviewed ? g.accuracy : 0}%"></div>
</div>
<span class="agg-model-val {toneClass(g.accuracy)}">{g.accuracy}%</span>
</div>
{/each}
</div>
</div>
{/if}
{/each}
</div>
{#if Object.keys(report.prompts || {}).length}
<details class="prompts" open>
<summary>Промпты (то, что передаётся модели)</summary>
<div class="prompts-body">
{#each SCENARIO_NAMES as key (key)}
{@const p = report.prompts?.[key]}
{#if p}
<div class="prompt-item">
<div class="prompt-name">{SCENARIO_LABELS[key]}</div>
<div class="prompt-text">{p}</div>
</div>
{/if}
{/each}
</div>
</details>
{/if}
<div class="filters">
<label>Модель:</label>
<select bind:value={selModel}>
<option value="all">Все модели</option>
{#each models as m (m)}
<option value={m}>{modelShort(m)}</option>
{/each}
</select>
<button class:active={selScenario === 'all'} onclick={() => (selScenario = 'all')}>Все сценарии</button>
{#each SCENARIO_NAMES as key (key)}
<button class:active={selScenario === key} onclick={() => (selScenario = key)}>
{SCENARIO_LABELS[key]}
</button>
{/each}
<span class="count">показано: {filtered.length} из {rows.length}</span>
</div>
<div class="tbl-shell">
<table>
<thead>
<tr>
<th>Модель</th>
<th>#</th>
<th>Сценарий</th>
<th></th>
<th>Утка</th>
<th>Вопрос</th>
<th>Ожид.</th>
<th>in/out</th>
<th>Ответ модели</th>
</tr>
</thead>
<tbody>
{#each filtered as r (r.model + r.id + r.scenario)}
<tr class:ok={r.correct === true} class:no={r.correct === false} class:pend={r.correct === null || r.correct === undefined}>
<td class="r-model">{modelShort(r.model)}</td>
<td class="r-id">{r.id}</td>
<td class="r-s">{SCENARIO_LABELS[r.scenario]}</td>
<td class="r-result">{resultText(r.correct)}</td>
<td class="r-duck">{r.duckUsed ? '🦆' : '—'} <span class="muted">({r.toolCalls})</span></td>
<td class="r-question">{r.question}</td>
<td class="r-expected">{r.expected}</td>
<td class="r-tokens">{r.promptTokens}<span class="muted"> / </span>{r.genTokens}</td>
<td class="r-answer">{r.response}</td>
</tr>
{/each}
</tbody>
</table>
</div>
<style>
h2 {
font-size: 17px;
margin: 26px 0 12px;
}
.meta {
color: var(--muted);
font-size: 13px;
}
.meta b {
color: var(--text);
font-weight: 600;
}
.shell {
background: var(--panel);
border: 1px solid var(--border);
border-radius: 12px;
overflow: auto;
}
table {
width: 100%;
border-collapse: collapse;
font-size: 13px;
}
thead th {
background: var(--panel2);
text-align: left;
padding: 10px 14px;
font-size: 11px;
text-transform: uppercase;
letter-spacing: 0.05em;
color: var(--muted);
white-space: nowrap;
}
tbody td {
padding: 10px 14px;
border-top: 1px solid var(--border);
vertical-align: top;
}
.pm-name {
font-weight: 700;
white-space: nowrap;
}
.pm-num {
text-align: right;
font-variant-numeric: tabular-nums;
}
.tone-safe {
color: var(--ok);
}
.tone-warn {
color: var(--warn);
}
.tone-bad {
color: var(--no);
}
.review-note {
margin: 14px 0;
padding: 10px 14px;
background: rgba(245, 158, 11, 0.12);
border: 1px solid rgba(245, 158, 11, 0.4);
border-radius: 10px;
color: #fbbf24;
font-size: 13px;
}
.cards {
display: grid;
grid-template-columns: repeat(4, 1fr);
gap: 16px;
margin: 0 0 8px;
}
@media (max-width: 1100px) {
.cards {
grid-template-columns: repeat(2, 1fr);
}
}
@media (max-width: 640px) {
.cards {
grid-template-columns: 1fr;
}
}
.card {
background: var(--panel);
border: 1px solid var(--border);
border-radius: 12px;
padding: 16px;
}
.scenario-head {
display: flex;
justify-content: space-between;
align-items: flex-start;
gap: 12px;
}
.scenario-title {
font-size: 16px;
font-weight: 700;
}
.scenario-desc {
color: var(--muted);
font-size: 12px;
margin-top: 4px;
}
.accuracy {
font-size: 32px;
font-weight: 800;
line-height: 1;
}
.pending-badge {
font-size: 11px;
font-weight: 600;
color: var(--muted);
margin-top: 4px;
}
.bar {
height: 8px;
background: var(--panel2);
border-radius: 99px;
margin: 14px 0;
overflow: hidden;
}
.bar-fill {
height: 100%;
border-radius: 99px;
background: currentColor;
transition: width 0.4s;
}
.metrics {
display: grid;
grid-template-columns: repeat(2, 1fr);
gap: 10px;
}
.metric {
background: var(--panel2);
border-radius: 8px;
padding: 10px 12px;
}
.m-label {
display: block;
color: var(--muted);
font-size: 11px;
text-transform: uppercase;
letter-spacing: 0.05em;
white-space: nowrap;
}
.m-val {
display: block;
font-size: 20px;
font-weight: 800;
margin-top: 2px;
}
.m-mod {
font-size: 14px;
font-weight: 600;
color: var(--muted);
}
.m-sub {
display: block;
color: var(--muted);
font-size: 11px;
margin-top: 2px;
}
.agg-models {
margin-top: 14px;
display: grid;
gap: 8px;
}
.agg-model {
display: grid;
grid-template-columns: auto 1fr auto;
align-items: center;
gap: 10px;
font-size: 12px;
}
.agg-model .bar {
margin: 0;
}
.agg-model-name {
color: var(--muted);
white-space: nowrap;
}
.agg-model-val {
font-weight: 700;
font-variant-numeric: tabular-nums;
}
.prompts {
margin: 20px 0;
background: var(--panel);
border: 1px solid var(--border);
border-radius: 12px;
overflow: hidden;
}
.prompts summary {
cursor: pointer;
padding: 12px 16px;
font-weight: 700;
font-size: 14px;
list-style: none;
}
.prompts summary::-webkit-details-marker {
display: none;
}
.prompts summary::before {
content: '▸ ';
color: var(--muted);
}
.prompts[open] summary::before {
content: '▾ ';
}
.prompts-body {
padding: 0 16px 14px;
display: grid;
gap: 12px;
}
.prompt-item {
background: var(--panel2);
border-radius: 8px;
padding: 10px 12px;
}
.prompt-name {
font-size: 12px;
color: var(--muted);
text-transform: uppercase;
letter-spacing: 0.05em;
margin-bottom: 6px;
}
.prompt-text {
white-space: pre-wrap;
font-size: 13px;
line-height: 1.5;
}
.filters {
margin: 8px 0 16px;
display: flex;
gap: 8px;
flex-wrap: wrap;
align-items: center;
}
.filters label {
font-size: 12px;
color: var(--muted);
}
.filters select {
background: var(--panel);
color: var(--text);
border: 1px solid var(--border);
padding: 7px 10px;
border-radius: 8px;
font-size: 13px;
}
.filters button {
background: var(--panel);
color: var(--text);
border: 1px solid var(--border);
padding: 7px 14px;
border-radius: 8px;
cursor: pointer;
font-size: 13px;
}
.filters button.active {
background: #3b82f6;
border-color: #3b82f6;
color: #fff;
}
.count {
margin-left: auto;
color: var(--muted);
font-size: 12px;
}
.tbl-shell {
background: var(--panel);
border: 1px solid var(--border);
border-radius: 12px;
overflow: auto;
}
.r-model {
font-weight: 600;
white-space: nowrap;
}
.r-id {
font-weight: 700;
color: var(--muted);
}
.r-s {
white-space: nowrap;
}
.r-result {
font-weight: 800;
}
tr.ok .r-result {
color: var(--ok);
}
tr.no .r-result {
color: var(--no);
}
tr.pend .r-result {
color: var(--muted);
}
tr.pend {
opacity: 0.75;
}
.r-duck {
text-align: center;
white-space: nowrap;
}
.r-question {
max-width: 300px;
color: var(--text);
}
.r-expected {
max-width: 220px;
}
.r-answer {
max-width: 380px;
}
.r-answer,
.r-question {
white-space: pre-wrap;
}
</style>
+23
View File
@@ -0,0 +1,23 @@
import type { ReportRoot } from '@duck/types';
import { error } from '@sveltejs/kit';
import type { Load } from '@sveltejs/kit';
import { extractPerTask, getReviewTotals } from '$lib/reports';
export interface ReportsPageData {
report: ReportRoot;
rows: ReturnType<typeof extractPerTask>;
review: { reviewed: number; pending: number };
}
export const load: Load = async ({ fetch }): Promise<ReportsPageData> => {
const res = await fetch('/report.json');
if (!res.ok) {
throw error(res.status, `report.json not found (${res.status})`);
}
const report = (await res.json()) as ReportRoot;
return {
report,
rows: extractPerTask(report),
review: getReviewTotals(report)
};
};
File diff suppressed because it is too large Load Diff
+3
View File
@@ -0,0 +1,3 @@
# allow crawling everything by default
User-agent: *
Disallow:
+20
View File
@@ -0,0 +1,20 @@
{
"extends": "./.svelte-kit/tsconfig.json",
"compilerOptions": {
"rewriteRelativeImportExtensions": true,
"allowJs": true,
"checkJs": true,
"esModuleInterop": true,
"forceConsistentCasingInFileNames": true,
"resolveJsonModule": true,
"skipLibCheck": true,
"sourceMap": true,
"strict": true,
"moduleResolution": "bundler"
}
// Path aliases are handled by https://svelte.dev/docs/kit/configuration#alias
// except $lib which is handled by https://svelte.dev/docs/kit/configuration#files
//
// To make changes to top-level options such as include and exclude, we recommend extending
// the generated config; see https://svelte.dev/docs/kit/configuration#typescript
}
+17
View File
@@ -0,0 +1,17 @@
import adapter from '@sveltejs/adapter-vercel';
import { sveltekit } from '@sveltejs/kit/vite';
import { defineConfig } from 'vite';
export default defineConfig({
plugins: [
sveltekit({
compilerOptions: {
// Force runes mode for the project, except for libraries. Can be removed in svelte 6.
runes: ({ filename }) =>
filename.split(/[/\\]/).includes('node_modules') ? undefined : true
},
adapter: adapter()
})
]
});
+207
View File
@@ -0,0 +1,207 @@
# План миграции: единый SvelteKit-проект + типизированный eval
> Статус: **утверждён для проработки** (запуск реализации — по отдельной команде).
> Последнее обновление: 2026-09-04
---
## 1. Общая цель
Убрать избыточное разделение на два проекта и перевести всё в **один SvelteKit-проект**
на Vercel, а eval-harness — с JavaScript (`.mjs`) на **TypeScript**, с **общими типами**
между eval и сайтом.
Итог: один домен, один деплой, один контракт данных для отчётов (он не «ломается»,
потому что eval и рендер сидят на одних типах).
---
## 2. Целевая архитектура
### 2.1. Общий пакет типов — `packages/types`
Workspace-пакет `@duck/types`, используемый и eval, и `apps/web`.
```txt
packages/types/
package.json # name: "@duck/types" (через pnpm pkg)
tsconfig.json
src/index.ts # общий контракт
```
Типы (защищают контракт отчёта):
- `Task { id, question, answer }`
- `ScenarioName = "control" | "blind" | "mentor"`
- `ScenarioSummary { tasks, reviewed, correct, accuracy, pending, duckUsed, toolCalls, avgCalls }`
- `PerTaskRow { id, scenario, correct: boolean|null, duckUsed, toolCalls, promptTokens, genTokens, duckTokens, response, expected, question }`
- `ModelReport`, `ReportRoot { generatedAt, mcpUrl, prompts, models, aggregate, reviewedAt? }`
- `Reviews` (контракт `review.json`)
Роль: и `apps/eval` при записи `out/report.json`, и `apps/web` при рендере `/reports`
используют эти типы. Рассогласование схемы → ошибка на этапе компиляции, а не падение в проде.
### 2.2. Приложение — `apps/web` (единственный деплой на Vercel)
```txt
apps/web/ # SvelteKit + TypeScript + @sveltejs/adapter-vercel
src/routes/
+layout.svelte # общий nav-каркас
+page.svelte # / — домашняя (заглушка «скоро»)
/mcp/+page.svelte # /mcp — описание MCP + FAQ/README
/reports/+page.svelte # /reports — полные отчёты (рендер из report.json)
/api/mcp/+server.ts # /api/mcp — MCP endpoint (mcp-handler)
src/lib/
mcp.ts # перенос из apps/mcp/lib/mcp.ts
reports.ts # рендер-хелперы на базе @duck/types
static/report.json # коммитится — источник для /reports
```
**Домен:** `rubber-duck-mcp.vercel.app` — единственный. Всё на нём:
- `/` → домашняя
- `/mcp` → описание
- `/reports` → отчёты
- `/api/mcp` → MCP endpoint
### 2.3. Eval — типизация в TS
```txt
apps/eval/
src/*.ts # run, runner, report, review (из .mjs)
lib/mcpClient.ts # (из lib/mcpClient.mjs)
lib/ollama.ts # (из lib/ollama.mjs)
tsconfig.json
out/report.json # генерируется (НЕ коммитится)
data/tasks.json # (как есть)
```
---
## 3. Все решённые вопросы (закрытые решения)
| Вопрос | Решение |
| --------------------- | --------------------------------------------------------------------------------------- |
| Домен | единый `rubber-duck-mcp.vercel.app` |
| Куда класть SvelteKit | новая папка `apps/web` |
| Линковка Vercel | перелинковать существующий project `mcp` на `apps/web` (домен сохр.) |
| Frontend-домен | удалить `rubber-duck-frontend.vercel.app` и Vercel-проект `frontend` |
| `report.json` в проде | коммитить в `apps/web/static/report.json` |
| Страница `/reports` | полный формат (сводные таблицы по моделям + разбивка по задачам) |
| MCP-фреймворк | mcp-handler совместим с SvelteKit (web-standard `(Request) => Response`), Next не нужен |
| Язык eval | TypeScript + общий пакет типов |
---
## 4. Изменение запуска (было → стало)
Сейчас (JS):
```bash
node src/run.mjs
node src/review.mjs
```
Стало (TS): компиляция не требуется для dev — запуск через `tsx`:
```bash
# apps/eval
tsx src/run.ts
tsx src/review.ts
```
> Локальная генерация статического `report.html` (`report-html`) **убрана**: `/reports` на
> сайте рендерит те же данные из `report.json` через Svelte-компоненты на общих типах.
Скрипты в `apps/eval/package.json` и корневом `package.json` (`eval:*`) обновляются на `tsx`.
Добавляется:
- **typecheck**: `tsc --noEmit` для eval и web (контракт не ломается).
- dev-зависимости eval: `tsx`, `typescript`, `@types/node`.
---
## 5. Шаги реализации (поэтапно)
### Шаг 1 — `packages/types`
- Создать workspace-пакет `@duck/types` (манифест через `pnpm pkg`).
- Написать `src/index.ts` с общим контрактом + `tsconfig.json`.
- Убедиться, что pnpm-workspace (`packages: apps/*`) знает про `packages/*`.
### Шаг 2 — eval → TypeScript
- Переименовать `lib/*.mjs``.ts`, `src/*.mjs``.ts`.
- Подключить `@duck/types`, поправить импорты.
- Добавить `tsconfig.json`, зависимости (`tsx`, `typescript`, `@types/node`) через `pnpm pkg`/`pnpm add`.
- Обновить скрипты запуска (`eval:*``tsx ...`).
- Проверка: `tsc --noEmit` + короткий прогон `tsx src/run.ts --limit 1`.
### Шаг 3 — `apps/web` (SvelteKit)
- Инициализировать SvelteKit + TypeScript + `@sveltejs/adapter-vercel`.
- Подключить `@duck/types`.
### Шаг 4 — Перенос MCP
- `apps/mcp/lib/mcp.ts``apps/web/src/lib/mcp.ts`.
- `apps/mcp/app/api/mcp/route.ts``apps/web/src/routes/api/mcp/+server.ts`
(mcp-handler принимает `Request` → из SvelteKit передаём `event.request`).
- Убедиться, что `runtime = "nodejs"` задан.
### Шаг 5 — Страницы
- `+layout.svelte` — навигация;
- `/` — заглушка «скоро» (дизайн позже);
- `/mcp` — описание + FAQ + README (перенести текст из старой `page.tsx`);
- `/reports``load` из `static/report.json` через `lib/reports.ts` на общих типах;
рендер карточек/таблиц — Svelte-компоненты (визуальный стиль перенесён из прежнего `report-html.mjs`).
### Шаг 6 — Данные
- Скопировать `apps/eval/out/report.json``apps/web/static/report.json` и закоммитить.
- Добавить команду «sync отчёта на сайт» (копирует report.json после прогона eval).
### Шаг 7 — Деплой/конфиг
- `apps/web/package.json` + корневой: `dev`, `build`, `deploy``apps/web`;
`eval:*``tsx`; убрать `deploy-mcp`, `deploy-frontend`.
- Удалить `apps/frontend` + скрипты + Vercel-проект/домен frontend.
- Удалить `apps/mcp` (Next) после переноса.
- Перелинковать Vercel-проект `mcp` на `apps/web`; проверить, что домен назначен на проект.
### Шаг 8 — Проверка
- `tsc --noEmit` (eval + web).
- `pnpm dev` → проверить страницы и `/api/mcp`.
- `vercel --prod` → verify на домене: `/`, `/mcp`, `/reports`, `/api/mcp` (quack).
---
## 6. Что НЕ делаем сейчас
- Детальный дизайн страниц/домашней (делаем позднее, отдельно).
- Автоматическое обновление `report.json` на каждый прогон (только sync-команда).
- Расширенное наполнение FAQ/README (только перенос базового текста).
---
## 7. Риски и митигация
| Риск | Митигация |
| -------------------------------------------- | ------------------------------------------------------------------------- |
| `mcp-handler` под Next | фреймворк-агностичен (web-standard Request/Response), проверено по `d.ts` |
| Нужна компиляция для запуска eval | dev — `tsx`; при необходимости `tsc build` для prod-скриптов |
| Несовместимость схемы отчёта | общие типы `@duck/types` + `tsc --noEmit` в eval и web |
| Необратимое удаление Vercel-проектов | frontend/mcp удаляем аккуратно, после успешного переноса |
| `report.json` рассинхронизируется с прогоном | явная команда sync + коммит |
---
## 8. Порядок выполнения
Поэтапно, с проверкой на каждом шаге:
`packages/types` → eval-TS → `apps/web` (скелет) → перенос MCP → страницы → данные
→ деплой/конфиг → проверка.
(Реализацию начинать только по явной команде.)
+8 -10
View File
@@ -3,16 +3,14 @@
"version": "1.0.0", "version": "1.0.0",
"description": "Rubber duck debugging - frontend + MCP server monorepo", "description": "Rubber duck debugging - frontend + MCP server monorepo",
"scripts": { "scripts": {
"dev": "pnpm --dir apps/frontend dev", "dev": "pnpm --dir apps/web dev",
"build-frontend": "pnpm --dir apps/frontend build", "build": "pnpm --dir apps/web build",
"build-mcp": "pnpm --dir apps/mcp build", "deploy": "vercel --prod --yes",
"dev-mcp": "pnpm --dir apps/mcp dev", "typecheck": "pnpm --dir apps/eval run typecheck && pnpm --dir apps/web run typecheck",
"typecheck-mcp": "pnpm --dir apps/mcp exec tsc --noEmit", "eval:run": "pnpm --dir apps/eval run run",
"deploy-mcp": "vercel --cwd apps/mcp --prod --yes", "eval:review": "pnpm --dir apps/eval run review",
"deploy-frontend": "vercel --cwd apps/frontend --prod --yes", "eval:sync": "node scripts/sync-report.mjs",
"eval:run": "pnpm --dir apps/eval exec node src/run.mjs", "eval:typecheck": "pnpm --dir apps/eval run typecheck"
"eval:review": "pnpm --dir apps/eval exec node src/review.mjs",
"eval:html": "pnpm --dir apps/eval exec node src/report-html.mjs"
}, },
"license": "ISC", "license": "ISC",
"devEngines": { "devEngines": {
+18
View File
@@ -0,0 +1,18 @@
{
"name": "@duck/types",
"version": "0.1.0",
"private": true,
"type": "module",
"exports": {
".": {
"types": "./src/index.ts",
"default": "./src/index.ts"
}
},
"devDependencies": {
"typescript": "^5.9.3"
},
"scripts": {
"typecheck": "tsc --noEmit"
}
}
+127
View File
@@ -0,0 +1,127 @@
// ── Tasks ───────────────────────────────────────────────
export interface Task {
id: string;
question: string;
answer: string;
}
// ── Scenarios ───────────────────────────────────────────
export type ScenarioName = "control" | "thinking" | "blind" | "mentor";
export const SCENARIO_NAMES: readonly ScenarioName[] = [
"control",
"thinking",
"blind",
"mentor",
];
export const SCENARIO_LABELS: Record<ScenarioName, string> = {
control: "Без утки (контроль)",
thinking: "Думать вслух",
blind: "Слепая утка",
mentor: "Утка-помощник",
};
export const SCENARIO_DESCS: Record<ScenarioName, string> = {
control: "Модель решает задачу напрямую, без инструментов.",
thinking: "Модель выписывает рассуждения вслух, без дука.",
blind: "Модель объясняет подход коллеге, вызывает quack, не зная заранее ответ.",
mentor: "Модель работает в паре, зная, что ответ будет только «quack».",
};
// ── Per-task evaluation row ─────────────────────────────
export interface PerTaskRow {
id: string;
scenario: ScenarioName;
correct: boolean | null;
duckUsed: boolean;
toolCalls: number;
promptTokens: number;
genTokens: number;
duckTokens: number;
response: string;
expected: string;
question: string;
}
// ── Per-scenario summary (inside a model report) ────────
export interface ScenarioSummary {
tasks: number;
reviewed: number;
correct: number;
accuracy: number;
pending: number;
duckUsed: number;
toolCalls: number;
avgCalls: number;
avgPromptTokens: number;
avgGenTokens: number;
avgDuckTokens: number;
}
// ── Per-model report ────────────────────────────────────
export interface ModelReport {
model: string;
scenarios: Record<ScenarioName, ScenarioSummary>;
perTask: PerTaskRow[];
}
// ── Aggregate ───────────────────────────────────────────
export interface AggregateByModel {
tasks: number;
reviewed: number;
correct: number;
accuracy: number;
duckUsed: number;
}
export interface AggregateScenario {
tasks: number;
reviewed: number;
correct: number;
accuracy: number;
pending: number;
duckUsed: number;
toolCalls: number;
avgCalls: number;
byModel: Record<string, AggregateByModel>;
}
export interface AggregateModel {
tasks: number;
totalRows: number;
reviewed: number;
correct: number;
accuracy: number;
duckUsed: number;
toolCalls: number;
avgGenTokens: number;
}
export interface Aggregate {
perScenario: Record<ScenarioName, AggregateScenario>;
perModel: Record<string, AggregateModel>;
}
// ── Root report (report.json) ───────────────────────────
export interface ReportRoot {
generatedAt: string;
mcpUrl: string;
prompts: Record<ScenarioName, string>;
models: Record<string, ModelReport>;
aggregate: Aggregate;
reviewedAt?: string;
}
// ── Reviews (review.json) ───────────────────────────────
// Keys: "${model}|${taskId}|${scenario}"
export type ReviewKey = `${string}|${string}|${string}`;
export type Reviews = Record<ReviewKey, boolean>;
+12
View File
@@ -0,0 +1,12 @@
{
"compilerOptions": {
"target": "ES2022",
"module": "ES2022",
"moduleResolution": "bundler",
"strict": true,
"noEmit": true,
"isolatedModules": true,
"skipLibCheck": true
},
"include": ["src"]
}
+957 -109
View File
File diff suppressed because it is too large Load Diff
+3
View File
@@ -1,2 +1,5 @@
packages: packages:
- apps/* - apps/*
- packages/*
allowBuilds:
esbuild: true
+18
View File
@@ -0,0 +1,18 @@
import { copyFileSync, existsSync, mkdirSync } from 'node:fs';
import path from 'node:path';
import { fileURLToPath } from 'node:url';
const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
const src = path.join(root, 'apps', 'eval', 'out', 'report.json');
const destDir = path.join(root, 'apps', 'web', 'static');
const dest = path.join(destDir, 'report.json');
if (!existsSync(src)) {
console.error(`[sync-report] Not found: ${src}`);
console.error('Run the eval first (pnpm eval:run / eval:review) to produce an out/report.json.');
process.exit(1);
}
mkdirSync(destDir, { recursive: true });
copyFileSync(src, dest);
console.log(`[sync-report] ${path.relative(root, src)} -> ${path.relative(root, dest)}`);
+123 -26
View File
@@ -1,44 +1,141 @@
# 🎯 Суть эксперимента # 🎯 Эксперимент «Резиновая уточка»
Проверить, как промежуточный запрос к MPC-серверу (который на любое сообщение отвечает "quack") влияет на качество рассуждений и точность ответов LLM, особенно «слабых» моделей без встроенного механизма thinking. Проверить, как промежуточный запрос к MCP-серверу (отвечающему «quack») влияет на качество
рассуждений и точность ответов LLM — особенно «слабых» локальных моделей без встроенного
механизма thinking.
## 👥 Какие модели тестировать ## 🧪 Методология: 4 параллельных сценария
1. Локальные (для GTX 1660 Super, 6GB VRAM): Переменные, которыми мы управляем:
- Запускать через: Ollama (квантование Q4_K_M или Q5_K_M). - **«Думать вслух»** (writing out reasoning): модель пишет свои рассуждения явно или нет.
- Модели: Llama-3.2-3B-Instruct (идеально для теста) или Qwen-2.5-3B-Instruct (хорошая логика). - **«Дуб-инструмент»** (duck call): модель вызывает MCP-инструмент `quack` и получает кряк.
2. Коммерческие (бесплатные на OpenRouter): Чтобы отделить вклад каждой переменной, определяем **4 сценария**:
- Модели: Вбивать в поиск free и выбирать Mistral 7B Instruct, Llama 3 8B (Free) или Gemma 2 9B. | # | Сценарий | Думает вслух | Зовёт утку | Смысл |
| --- | ---------- | :----------: | :-------------------------------------------------------------------: | ----------------------------------------------------------------------------------------------------- |
| 1 | `control` | нет | нет | Базовый «прямой ответ». |
| 2 | `thinking` | **да** | нет | Контроль «думать вслух» (без утки). Отмеряет вклад проговаривания. |
| 3 | `blind` | да | **да** (не знает, что будет «кряк») | «Слепая» уточка — чистый тест влияния утки на фоне уже включённого мышления. |
| 4 | `mentor` | да | **да** (знает, что уточка отвечает только «quack», без полезной инфы) | «Уточка-помощник» — та же работа в паре, но модель заранее знает, что ответ будет бесполезным кряком. |
3. Эталон (для сравнения): Схема интерпретации разниц в accuracy:
- Любая модель со встроенным thinking (например, бесплатная DeepSeek-R1 на OpenRouter). Поможет понять, насколько уточка приближает слабую модель к «врожденному» мышлению. - `thinking control` → вклад «проговаривания мыслей вслух».
- `blind thinking` → вклад факта обращения к утке (на фоне «думать вслух»).
- `mentor thinking` → вклад знания о том, что от утки будет только бесполезный «quack»
(продолжает думать сам, не рассчитывая на подсказку).
> **Ограничение:** control и thinking структурно отличаются от сценариев с уткой. А вот
> `blind` и `mentor` намеренно сведены к одной структуре — различаются только тем, что
> mentor знает про «quack». Это держит сравнение чистым: эффект сводится только к
> информированности модели.
### Язык
Задачи и системные промпты — **на английском** (модели англоязычные). Отдельное
указание про язык ответа не даётся — модель естественно отвечает на английском,
раз и промпт, и задача сформулированы на нём.
### Формат ответа
**Свободный ответ** — модель формулирует финальный ответ естественно. Разметка — **ручная**
(reviewer оценивает каждый ответ). Никаких жёстких маркеров `ANSWER:`, никакого авто-парсинга.
Это осознанно: reviewer смотрит не только на финальное число, но и на ход рассуждений.
### Детерминизм
`temperature = 0` во всех сценариях (уже стоит по умолчанию в `ollama.chat`).
--- ---
## 🧪 Методика: 3 параллельных теста ## Промпты (системные, английский)
Для каждой задачи из вашего датасета (возьмите 20–30 сложных логических/математических задач) запустите три сценария: > Общее требование к финалу во всех сценариях: **чёткий конкретный ответ на вопрос**,
> а не догадка и не запрос подтверждения («ну вроде так, подтверди» — недопустимо).
- Тест 1: Контрольный (Прямой ответ) ### control — прямой ответ
- Промт: «Ты — опытный ассистент. Реши задачу пользователя максимально точно».
- Процесс: Модель решает задачу сразу в один шаг. > You are an experienced assistant. Solve the user's problem as accurately as possible.
- Тест 2: «Слепая» уточка (Чистый тест на симуляцию мышления) > Give a clear, concrete final answer to the question - a definite result, not a tentative
- Промт: «Ты решаешь сложную задачу. Чтобы решить её лучше, ты работаешь в паре с коллегой. Сначала подробно опиши коллеге суть своего подхода, первые шаги или сомнения. Дождись его ответа. Затем, учитывая реакцию коллеги, заверши решение и выдай финальный ответ пользователю». > guess or a request for confirmation.
- Процесс: Модель выгружает мысли → сервер отвечает "quack" → модель выдает итоговый ответ. Модель не знает заранее, что ей ответят «кряком», и думает сама.
- Тест 3: Уточка-помощник (Мета-инструкция) ### thinking — думать вслух, без утки (контроль)
- Промт: «Ты решаешь сложную задачу. Перед тем как выдать ответ, ты ОБЯЗАН детально расписать мысли и возможные ошибки для своей резиновой уточки (она ответит "quack"). Используй ответ уточки, чтобы проверить себя, найти баги и только после этого выдай идеальный ответ».
- Процесс: Модель целенаправленно использует утку для поиска своих же ошибок. > You are solving a difficult problem. Before giving your final answer, write out your
> reasoning step by step: your approach, each step, and any doubts or mistakes you notice
> along the way. Then give a clear, concrete final answer to the user's question - a
> definite result, not a request for confirmation.
### blind — «слепая» уточка
> You are solving a problem that the user asked you. To reason better, you use a separate
> tool named 'quack': you call it yourself to state your thinking out loud. Call the tool
> and spell out your approach, each step, and any doubts or mistakes you might be making,
> then wait for its short reply. Do not ask the user to confirm anything, and do not guess
> or invent the tool's reply yourself - the tool answers on your behalf. After the tool's
> reply, give the user a clear, concrete final answer to the question.
> **Важно для чистоты:** модель НЕ должна знать, что инструмент — уточка, отвечающая «кряк».
> Поэтому **описание инструмента (tool schema) нейтрально** — оно не раскрывает «только
> quack». Модель вызывает `quack` сама через tool-call, чтобы проговорить мысли, полагая,
> что получит короткий полезный ответ инструмента. Роли чётко разделены: юзеру — итоговый
> ответ, инструменту — озвучка мыслей. Модель НЕ строит диалог сама с собой и НЕ ждёт
> подтверждения от юзера.
### mentor — уточка-помощник
> You are solving a problem that the user asked you. To reason better, you use a separate
> tool named 'quack': you call it yourself to state your thinking out loud. Call the tool
> and spell out your approach, each step, and any doubts or mistakes you might be making,
> then wait for its short reply. Do not ask the user to confirm anything, and do not guess
> or invent the tool's reply yourself - the tool answers on your behalf. Note: the tool
> 'quack' is a rubber duck and will only ever reply with just 'quack' - it gives no useful
> information. Treat it as a way to voice your thoughts out loud, not as a source of
> answers. After the tool's reply, give the user a clear, concrete final answer to the
> question.
> **Ключевое:** blind и mentor структурно идентичны и отличаются **только** тем, что mentor
> заранее знает, что получит только «quack» без полезной информации. Никаких дополнительных
> директив (`MUST`, «double-check», «find bugs») — иначе они бы загрязняли сравнение.
--- ---
## 📊 Что фиксировать в результатах (Метрики) ## ⏱ Таймауты и ретраи
1. Точность (Accuracy): Вырос ли процент правильных ответов в Тесте 2 и Тесте 3 по сравнению с Тестом 1? - На один вызов модели — **таймаут 60 секунд** (1 минута).
2. Объем рассуждений (Token Count): Сколько токенов модель тратит на объяснение задачи утке? Становится ли её финальный текст длиннее и структурированнее? - При сбое/таймауте запрос **повторяется до 3 раз всего** (1-я попытка + 2 ретрая).
3. Поведение в Тесте 2: Как модель реагирует на "quack"? Игнорирует его, извиняется или сам факт написания первого сообщения помогает ей увидеть свои ошибки? - Если после всех попыток успеха нет — задача **пропускается** (строки в отчёте нет).
Рекомендация по настройке: для чистоты эксперимента во всех тестах выставляйте temperature = 0. ---
## 📊 Метрики
1. **Accuracy** — доля правильных ответов (по ручной разметке reviewer'а) в каждом сценарии.
Сравнение: `thinkingcontrol`, `blindthinking`, `mentorthinking`.
2. **Duck usage** — вызвал ли модель `quack` в blind/mentor (duckUsed, toolCalls). Те модели,
что не зовут утку, деградируют до `thinking` — это фиксируем как отдельное явление.
3. **Объём рассуждений** — prompt/gen/duck токены на сценарий (структурированность текста).
---
## 🤖 Модели (локально, GTX 1660 Super 4GB)
- `llama3.2:3b`
- `qwen3:1.7b`
- `qwen3:4b`
- `granite4.1:3b`
- `phi4-mini:3.8b`
> Известное явление: `phi4-mini` и `granite4.1` могут не вызывать `quack` — тогда их
> blind/mentor вырождаются в `thinking`. Это часть изучаемого феномена и фиксируется по
> `duckUsed`.
## 📋 Процесс
1. `pnpm eval:run` — прогнать модели по 4 сценариям (23+ задач на английском). `correct: null`.
2. `pnpm eval:review` — ручная разметка каждого ответа (y/e/Enter/слово).
3. `pnpm eval:html` — собрать отчёт: сводные таблицы и разбивка по задачам, accuracy по
размеченным, pending для неразмеченных.
4. (Сайт) скопировать `report.json` в `apps/web/static/report.json` для страницы `/reports`.
+6
View File
@@ -0,0 +1,6 @@
{
"framework": "sveltekit",
"installCommand": "pnpm install",
"buildCommand": "pnpm --dir apps/web build",
"outputDirectory": "apps/web/.vercel/output"
}