feat: update ai models tests

This commit is contained in:
2026-09-04 11:57:47 +05:00
parent 6bd1de2669
commit 111b718526
8 changed files with 441 additions and 103 deletions
+128 -12
View File
@@ -1,9 +1,15 @@
const DEFAULT_URL = "http://localhost:11434";
export class Ollama {
constructor({ url = DEFAULT_URL, timeoutMs = 600000 } = {}) {
constructor({ url = DEFAULT_URL, timeoutMs = 600000, idleTimeoutMs = null, maxTotalMs = null } = {}) {
this.url = url.replace(/\/$/, "");
this.timeoutMs = timeoutMs;
// Dynamic timeout: if set, the timer resets on every new token; the request
// only fails if this much time passes with NO progress (no new token).
this.idleTimeoutMs = idleTimeoutMs ?? timeoutMs;
// Absolute hard cap on one request's lifetime, even if tokens keep flowing
// (guards against a model looping/chatting forever). Defaults to 5 minutes.
this.maxTotalMs = maxTotalMs ?? 5 * 60_000;
}
async ping({ timeoutMs = 30000 } = {}) {
@@ -15,37 +21,147 @@ export class Ollama {
return { ok: true, models };
}
async chat({ model, messages, tools, temperature = 0, numCtx = 8192, timeoutMs }) {
/**
* Streaming chat with a dynamic (progress-based) timeout.
* The timeout only fires when `timeoutMs` passes with no new token/chunk.
* Optionally calls onChunk({token, content, type, toolCalls}) as data arrives.
* Returns the same shape as the old non-streaming chat().
*/
async chatStream({ model, messages, tools, temperature = 0, numCtx = 8192, timeoutMs, maxTotalMs, onChunk }) {
const body = {
model,
messages,
stream: false,
stream: true,
options: { temperature },
};
if (numCtx) body.options.num_ctx = numCtx;
if (tools && tools.length) body.tools = tools;
const limitMs = timeoutMs ?? this.idleTimeoutMs;
const totalCapMs = maxTotalMs ?? this.maxTotalMs;
const controller = new AbortController();
const startedAt = Date.now();
const res = await fetch(`${this.url}/api/chat`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify(body),
signal: AbortSignal.timeout(timeoutMs ?? this.timeoutMs),
signal: controller.signal,
});
if (!res.ok) {
const t = await res.text();
throw new Error(`Ollama chat HTTP ${res.status}: ${t.slice(0, 300)}`);
}
const data = await res.json();
const message = data.message ?? {};
if (!res.body || !res.body.getReader) {
throw new Error("Ollama streaming response has no body reader");
}
const reader = res.body.getReader();
const decoder = new TextDecoder();
let buffer = "";
let content = "";
let toolCalls = [];
let promptEvalCount = 0;
let evalCount = 0;
let lastActivity = Date.now();
const guardInterval = setInterval(() => {
if (Date.now() - startedAt > totalCapMs) {
const err = new Error(`Request exceeded hard cap of ${totalCapMs}ms; aborting`);
err.name = "TimeoutError";
controller.abort(err);
return;
}
if (Date.now() - lastActivity > limitMs) {
const err = new Error(`No progress from Ollama for ${limitMs}ms; aborting`);
err.name = "TimeoutError";
controller.abort(err);
}
}, Math.min(500, Math.max(100, Math.floor(Math.min(limitMs, totalCapMs) / 4))));
try {
for (;;) {
let chunk;
try {
chunk = await reader.read();
} catch (readErr) {
// Abort / network drop mid-stream. Surface whatever we got so the
// caller can tell "slowly progressing" from "hung".
const e = new Error(
`Aborted mid-stream after ${content.length} chars (toolCalls=${toolCalls.length}): ${readErr.message}`
);
e.name = "TimeoutError";
e.partialContent = content;
e.partialToolCalls = toolCalls;
e.partialPrompt = promptEvalCount;
e.partialGen = evalCount;
throw e;
}
const { done, value } = chunk;
if (done) break;
lastActivity = Date.now();
buffer += decoder.decode(value, { stream: true });
let idx;
while ((idx = buffer.indexOf("\n")) !== -1) {
const line = buffer.slice(0, idx).trim();
buffer = buffer.slice(idx + 1);
if (!line) continue;
let obj;
try {
obj = JSON.parse(line);
} catch {
continue;
}
if (obj.prompt_eval_count != null) promptEvalCount = obj.prompt_eval_count;
if (obj.eval_count != null) evalCount = obj.eval_count;
const msg = obj.message ?? {};
if (msg.content) {
content += msg.content;
onChunk?.({ token: msg.content, content, type: "content" });
}
if (msg.tool_calls && msg.tool_calls.length) {
toolCalls = msg.tool_calls;
onChunk?.({ token: null, content, type: "tool_calls", toolCalls: msg.tool_calls });
}
}
}
} finally {
clearInterval(guardInterval);
}
return {
role: message.role ?? "assistant",
content: message.content ?? "",
toolCalls: message.tool_calls ?? [],
promptEvalCount: data.prompt_eval_count ?? 0,
evalCount: data.eval_count ?? 0,
raw: data,
role: "assistant",
content,
toolCalls: Array.isArray(toolCalls) ? toolCalls : [],
promptEvalCount,
evalCount,
raw: { stream: true },
};
}
async chat({ model, messages, tools, temperature = 0, numCtx = 8192, timeoutMs }) {
return this.chatStream({
model,
messages,
tools,
temperature,
numCtx,
timeoutMs,
onChunk: null,
});
}
async warmup({ model, text = "say OK", timeoutMs = 60000 }) {
try {
return await this.chatStream({
model,
messages: [{ role: "user", content: text }],
timeoutMs,
});
} catch (err) {
return { error: err.message };
}
}
}
export async function listModels() {