Files
metona-ai-desktop/electron/harness/adapters/ollama.adapter.ts
T
thzxx 9329df68af
CI / 类型检查 + Lint + 单元测试 (push) Failing after 9m37s
CI / 全量测试 (Electron ABI) (push) Failing after 6m1s
CI / 产物编译验证 (push) Successful in 10m54s
feat: v0.8.3 工作空间切换链路根治 · 思考强度扩档 — 启动TDZ连锁/继承丢数/回显滞后三修复 · xhigh+true档位单源 · 安全防线ready后注册 · 2217 用例全量回归 + 生产模式E2E
2026-09-08 21:14:59 +08:00

823 lines
32 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/**
* Ollama Provider Adapter
*
* 本地推理引擎,支持 Tool Calling、Thinking 模式、NDJSON 流式。
* 无需 API Key,连接本地 http://localhost:11434。
*
* 完整实现 Ollama API 文档中所有端点和参数:
* - POST /api/chat (对话)
* - POST /api/generate (补全)
* - POST /api/embed (嵌入)
* - GET /api/tags (列出模型)
* - POST /api/show (模型详情)
* - POST /api/pull (下载模型)
* - GET /api/ps (运行中模型)
* - GET /api/version (版本)
* - think 参数(Thinking 模式)
* - options 参数(temperature/top_k/top_p/stop/num_ctx/num_predict
* - format 参数(structured output
* - images 参数(多模态)
* - tool_calls 流式处理
*
* @see apis/ollama-api-docs-20260518.html
*/
import { BaseAdapter } from './base-adapter';
import { truncatedArgumentsPayload, readStreamChunkWithIdleTimeout } from './shared/sse-stream';
import log from 'electron-log';
import { nanoid } from 'nanoid';
import type { MetonaRequest, MetonaResponse, MetonaStreamEvent } from '../types';
import { MetonaFinishReason, MetonaStreamEventType } from '../types';
import type { MetonaModelInfo } from '../types/metona-adapter';
export class OllamaAdapter extends BaseAdapter {
// H-2 修复: provider → providerId(规范要求)
override readonly providerId: string = 'ollama';
readonly supportedModels = ['qwen3:latest', 'gemma3:latest', 'deepseek-r1:latest'];
readonly supportsToolCalling = true;
readonly supportsThinking = true;
private baseURL: string;
constructor(config: ConstructorParameters<typeof BaseAdapter>[0]) {
super(config);
this.baseURL = config.baseURL || 'http://localhost:11434';
// v0.8.1: 上下文窗口不再从 /api/show 探测或写死默认值获取 —— 唯一合法来源是
// 设置面板「上下文长度」(llm.contextWindow → AdapterConfig.contextWindow
// 引擎侧经 contextLength=num_ctx 下发)。构造时仅探测能力(thinking/tools/vision
// 属协议正确性门控),不再缓存窗口数值。
this.probeCapabilitiesOnce();
}
/** /api/show 能力探测只发一次(构造函数发起),失败不重试(fail-open) */
private probeAttempted = false;
private probeInProgress = false;
/**
* /api/show capabilities 探测缓存 —— 模型是否支持思考(协议正确性门控,
* 非窗口/输出上限语义)。null = 未探测/探测失败(fail-open 放行,与
* listModels 能力回退策略一致);false = 服务端明确不支持 → 不发 think 参数。
*/
private cachedThinkingSupport: boolean | null = null;
/**
* v0.8.1: 构造时 fire-and-forget 探测一次默认模型能力(仅 thinking 门控消费)。
* 旧实现同时缓存 num_ctx 窗口数值 —— 已按"窗口唯一来源是设置面板"契约删除。
*/
private probeCapabilitiesOnce(): void {
if (this.probeAttempted) return;
this.probeAttempted = true;
this.probeInProgress = true;
void this.showModel(this.config.defaultModel)
.then((info) => {
if (Array.isArray(info?.capabilities)) {
this.cachedThinkingSupport = info!.capabilities.map(String).includes('thinking');
}
})
.catch(() => {
/* 模型探测失败不阻塞对话(fail-open) */
})
.finally(() => {
this.probeInProgress = false;
});
}
// ===== POST /api/chat =====
// H-2 修复: chat → send(规范要求)
async send(request: MetonaRequest): Promise<MetonaResponse> {
// #2 修复: toNativeRequest 改为 async(需下载 URL 图片转 base64
const nativeRequest = await this.toNativeRequest(request);
// #24 修复: 使用 fetchWithTimeout 替代 getFetchSignal + fetch,确保 timer 清理
const response = await this.fetchWithTimeout(
`${this.baseURL}/api/chat`,
{
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ ...nativeRequest, stream: false }),
},
this.config.timeoutMs ?? 300_000,
);
if (!response.ok) {
await this.throwHttpError(response, 'Ollama API error');
}
const data = (await response.json()) as Record<string, unknown>;
return this.toMetonaResponse(data, request.meta.requestId, request.meta.iteration);
}
// H-2 修复: chatStream → sendStream(规范要求)
async *sendStream(request: MetonaRequest): AsyncIterable<MetonaStreamEvent> {
// #2 修复: toNativeRequest 改为 async(需下载 URL 图片转 base64
const nativeRequest = await this.toNativeRequest(request);
// #24 修复: 使用 fetchWithTimeout 替代 getFetchSignal + fetch,确保 timer 清理
const response = await this.fetchWithTimeout(
`${this.baseURL}/api/chat`,
{
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ ...nativeRequest, stream: true }),
},
this.config.timeoutMs ?? 300_000,
);
if (!response.ok || !response.body) {
await this.throwHttpError(response, 'Ollama stream error');
}
// 非空断言:上方 if 已确保 response.body 不为 null
const reader = response.body!.getReader();
const decoder = new TextDecoder();
let seq = 0;
let buffer = '';
let streamEndedNormally = false;
while (true) {
// v0.7.4 P1-2: 空闲超时 — 本地模型加载/推理期间服务器可能长时间不推数据,
// 共享辅助在连续 60s 无数据时抛 SseUpstreamError(504) 进重试通道
const { done, value } = await readStreamChunkWithIdleTimeout(
reader,
60_000,
this.getExternalAbortSignal(),
);
if (done) break;
buffer += decoder.decode(value, { stream: true });
const lines = buffer.split('\n');
buffer = lines.pop() ?? '';
for (const line of lines) {
const trimmed = line.trim();
if (!trimmed) continue;
try {
const chunk = JSON.parse(trimmed);
// 思考内容
if (chunk.message?.thinking) {
yield {
type: MetonaStreamEventType.REASONING_DELTA,
requestId: request.meta.requestId,
sessionId: request.meta.sessionId,
iteration: request.meta.iteration,
seq: seq++,
timestamp: Date.now(),
delta: chunk.message.thinking,
};
}
// 文本内容
if (chunk.message?.content) {
yield {
type: MetonaStreamEventType.TEXT_DELTA,
requestId: request.meta.requestId,
sessionId: request.meta.sessionId,
iteration: request.meta.iteration,
seq: seq++,
timestamp: Date.now(),
delta: chunk.message.content,
};
}
// 工具调用(Ollama 在最后一个 chunk 中整块返回)
if (chunk.message?.tool_calls) {
for (const tc of chunk.message.tool_calls) {
const args = tc.function?.arguments;
// v0.6.4 缺口修复: NDJSON 路径的截断自愈 —— 原实现 JSON.parse 抛错会
// 落入外层 catch:该 tool call 整体静默丢弃,且同一行剩余处理
// (含 done/USAGE 检查)一并被跳过,与 v0.6.3 已根治的 OpenAI 共享层
// 旧行为完全相同。现独立捕获并转为 _truncatedArguments 自愈载荷,
// 同时保证本 chunk 的后续分支照常执行。
let parsedArgs: Record<string, unknown>;
if (typeof args === 'string') {
try {
parsedArgs = JSON.parse(args);
} catch (parseErr) {
const sample = args.slice(-120);
log.warn(
`[Ollama] Tool call args truncated (unparseable JSON, ${(parseErr as Error).message}). Tail: ...${sample}`,
);
parsedArgs = truncatedArgumentsPayload((parseErr as Error).message, sample);
}
} else {
parsedArgs = (args as Record<string, unknown>) ?? {};
}
yield {
type: MetonaStreamEventType.TOOL_CALL_COMPLETE,
requestId: request.meta.requestId,
sessionId: request.meta.sessionId,
iteration: request.meta.iteration,
seq: seq++,
timestamp: Date.now(),
toolCall: {
// L-9 修复: 统一使用 nanoid 生成工具调用 ID(与 sse-stream.ts 一致)
id: `tc_${nanoid(8)}`,
name: tc.function?.name ?? '',
args: parsedArgs,
iteration: request.meta.iteration,
timestamp: Date.now(),
},
};
}
}
// 流结束
if (chunk.done) {
streamEndedNormally = true;
// v0.8.0 P0-1: 采集 done_reason —— Ollama 的停止原因在最终 done chunk
// 携带;旧实现流式路径完全忽略(length 截断不可见),现归一化后随
// DONE 事件上交引擎
const doneReason = mapOllamaDoneReason(
chunk.done_reason as string | undefined,
Array.isArray(chunk.message?.tool_calls) && chunk.message.tool_calls.length > 0,
);
const finishReason = doneReason === MetonaFinishReason.LENGTH ? 'length' : undefined;
// 发送 usage 信息
yield {
type: MetonaStreamEventType.USAGE,
requestId: request.meta.requestId,
sessionId: request.meta.sessionId,
iteration: request.meta.iteration,
seq: seq++,
timestamp: Date.now(),
usage: {
inputTokens: chunk.prompt_eval_count ?? 0,
outputTokens: chunk.eval_count ?? 0,
totalTokens: (chunk.prompt_eval_count ?? 0) + (chunk.eval_count ?? 0),
},
};
yield {
type: MetonaStreamEventType.DONE,
requestId: request.meta.requestId,
sessionId: request.meta.sessionId,
iteration: request.meta.iteration,
seq: seq++,
timestamp: Date.now(),
// v0.8.0 P0-1: 仅 length 需要显式上交(stop/tool_calls 语义由
// 事件流本身表达;load/unload 属本地引擎状态非响应语义)
...(finishReason ? { finishReason } : {}),
};
return;
}
} catch (parseErr) {
// P2-8 修复: 与 sse-stream.ts 一致,记录解析失败行便于诊断
log.warn(
`[Ollama] Failed to parse NDJSON line: ${(parseErr as Error).message}`,
trimmed.slice(0, 200),
);
}
}
}
// 流未正常结束(连接断开等),补发 DONE 事件防止 Agent Loop 挂起
if (!streamEndedNormally) {
yield {
type: MetonaStreamEventType.DONE,
requestId: request.meta.requestId,
sessionId: request.meta.sessionId,
iteration: request.meta.iteration,
seq: seq++,
timestamp: Date.now(),
};
}
}
// ===== POST /api/generate =====
async generate(params: {
model: string;
prompt: string;
suffix?: string;
system?: string;
stream?: boolean;
think?: boolean | string;
format?: string | object;
images?: string[];
options?: Record<string, unknown>;
/** v0.8.0 P1-3.2: 外部取消信号(与固定 300s 超时合并,任一触发即中止) */
signal?: AbortSignal;
}): Promise<{
response: string;
thinking?: string;
done: boolean;
totalDuration: number;
evalCount: number;
}> {
const response = await fetch(`${this.baseURL}/api/generate`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ ...params, stream: false, signal: undefined }),
// v0.8.0 P1-3.2 根治: 旧实现固定 AbortSignal.timeout(300_000) —— 外部中断
// 无法取消该请求(最长 5 分钟资源悬挂)。现用 AbortSignal.any 合并外部
// signal 与超时信号,任一触发即中止(Node ≥20.3 / Electron 35 满足)。
signal: params.signal
? AbortSignal.any([params.signal, AbortSignal.timeout(300_000)])
: AbortSignal.timeout(300_000),
});
if (!response.ok) throw new Error(`Ollama generate error: ${response.status}`);
const data = (await response.json()) as {
response?: string;
thinking?: string;
done?: boolean;
total_duration?: number;
eval_count?: number;
};
return {
response: data.response ?? '',
thinking: data.thinking,
done: data.done ?? true,
totalDuration: data.total_duration ?? 0,
evalCount: data.eval_count ?? 0,
};
}
// ===== POST /api/embed =====
async embed(params: {
model: string;
input: string | string[];
dimensions?: number;
/** v0.8.0 P1-3.2: 外部取消信号(与固定 60s 超时合并) */
signal?: AbortSignal;
}): Promise<{ embeddings: number[][]; totalDuration: number }> {
const response = await fetch(`${this.baseURL}/api/embed`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify(params),
// v0.8.0 P1-3.2: 与 generate 同口径 —— 外部 signal 与超时合并
signal: params.signal
? AbortSignal.any([params.signal, AbortSignal.timeout(60_000)])
: AbortSignal.timeout(60_000),
});
if (!response.ok) throw new Error(`Ollama embed error: ${response.status}`);
const data = (await response.json()) as { embeddings?: number[][]; total_duration?: number };
return {
embeddings: data.embeddings ?? [],
totalDuration: data.total_duration ?? 0,
};
}
// ===== GET /api/tags =====
/**
* H-2 修复: 返回 MetonaModelInfo[](规范要求)
*
* Ollama /api/tags 返回模型列表含详细信息(name, size, details),
* 转换为 MetonaModelInfo 并补充默认元数据。
*/
async listModels(): Promise<MetonaModelInfo[]> {
try {
const response = await fetch(`${this.baseURL}/api/tags`, {
signal: AbortSignal.timeout(10_000),
});
if (response.ok) {
const data = (await response.json()) as {
models?: Array<{
name: string;
size?: number;
details?: { parameter_size?: string; quantization_level?: string; family?: string };
}>;
};
if (data.models?.length) {
// v0.6.4 P4-1: 能力标志改为逐模型 /api/show 实测探测;单个探测失败
// 该模型回退保守 true(不可用时行为与旧实现一致,fail-open 保可用性)
// v0.7.3 P1-4: supportsVision 随探测结果透出(undefined = 未知 → 前端保守放行),
// 供上传入口拒绝不支持图片的本地语言模型
// v0.8.1: 不再填充 contextWindow —— 窗口唯一来源是设置面板「上下文长度」
const enriched = await Promise.all(
data.models.map(async (m) => {
const caps = await this.probeCapabilities(m.name);
return {
id: m.name,
name: m.name,
supportsToolCalling: caps ? caps.supportsTools : true,
supportsThinking: caps ? caps.supportsThinking : true,
supportsVision: caps ? caps.supportsVision : undefined,
description: m.details
? `${m.details.family ?? 'unknown'} / ${m.details.parameter_size ?? '?'} / ${m.details.quantization_level ?? '?'}`
: undefined,
};
}),
);
return enriched;
}
}
} catch {
// API 不可用时降级
}
// 回退到 supportedModels
return this.supportedModels.map((id) => ({ id }));
}
/**
* H-2 修复: 获取上下文窗口大小(规范要求)
*
* v0.8.1 硬性契约: 唯一来源是设置面板「上下文长度」(llm.contextWindow),
* 未配置返回 0 —— 删除了旧的 4096 写死默认值与 /api/show num_ctx 探测缓存。
* Ollama 的 num_ctx 由引擎经 params.contextLength(同源配置)下发给服务端。
*/
override getContextWindow(): number {
if (typeof this.config.contextWindow === 'number' && this.config.contextWindow > 0) {
return this.config.contextWindow;
}
return 0;
}
/**
* v0.6.4 P4-1: 通过 /api/show 的 capabilities[] 动态探测模型真实能力。
* 此前 listModels 对所有本地模型硬编码 supportsToolCalling/supportsThinking:true
* (注释自知不准)—— 语言模型不支持 tools 时引擎仍下发工具定义,
* 造成"模型口头说调工具实际不调"的回归温床。探测失败返回 null 由调用方回退保守值。
*/
async probeCapabilities(model: string): Promise<{
supportsTools: boolean;
supportsVision: boolean;
supportsThinking: boolean;
} | null> {
const info = await this.showModel(model);
if (!info || !Array.isArray(info.capabilities)) return null;
const caps = new Set(info.capabilities.map((c) => String(c)));
return {
supportsTools: caps.has('tools'),
supportsVision: caps.has('vision'),
supportsThinking: caps.has('thinking'),
};
}
// ===== POST /api/show =====
async showModel(
model: string,
): Promise<{ parameters: string; template: string; capabilities?: string[] } | null> {
try {
const response = await fetch(`${this.baseURL}/api/show`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ model }),
signal: AbortSignal.timeout(10_000),
});
if (!response.ok) return null;
const data = (await response.json()) as {
parameters?: string;
template?: string;
capabilities?: string[];
};
return {
parameters: data.parameters ?? '',
template: data.template ?? '',
// v0.8.1: 保留 undefined 语义 —— 响应未携带 capabilities 字段 = 未知
//(fail-open),显式数组(含空数组 = 服务端权威"无任何能力")才参与门控。
// 旧实现 `?? []` 把"字段缺失"与"权威空"混同,探测竞态下会误判不支持思考。
capabilities: data.capabilities,
};
} catch {
return null;
}
}
// ===== POST /api/pull =====
/**
* v0.6.4 P4-1 重构:pull 支持外部取消信号 —— 原实现固定 600s 超时会掐死
* 大模型下载(进度不能续命、无取消通道),大仓/慢网络场景必然失败。
* 现契约:调用方通过 AbortSignal 控制生命周期(UI 取消按钮即可触发);
* 超时语义交给用户取消或服务端断流(读循环结束即完成),不再人为设上限。
*/
async pullModel(
model: string,
onProgress?: (progress: { status: string; completed?: number; total?: number }) => void,
signal?: AbortSignal,
): Promise<void> {
const response = await fetch(`${this.baseURL}/api/pull`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ model, stream: true }),
signal,
});
if (!response.ok || !response.body) throw new Error(`Ollama pull error: ${response.status}`);
const reader = response.body.getReader();
const decoder = new TextDecoder();
let buffer = '';
while (true) {
const { done, value } = await reader.read();
if (done) break;
buffer += decoder.decode(value, { stream: true });
const lines = buffer.split('\n');
buffer = lines.pop() ?? '';
for (const line of lines) {
if (!line.trim()) continue;
try {
const chunk = JSON.parse(line);
onProgress?.({ status: chunk.status, completed: chunk.completed, total: chunk.total });
} catch {
// L-3 修复: 添加日志便于诊断非标准行(如进度通知、空行等)
log.debug('[Ollama] skipped non-JSON line during pull:', line.slice(0, 100));
}
}
}
}
// ===== GET /api/ps =====
async listRunning(): Promise<
Array<{ name: string; size: number; sizeVram: number; contextLength: number }>
> {
try {
const response = await fetch(`${this.baseURL}/api/ps`, {
signal: AbortSignal.timeout(10_000),
});
if (!response.ok) return [];
const data = (await response.json()) as {
models?: Array<{
name: string;
size?: number;
size_vram?: number;
context_length?: number;
}>;
};
return (data.models ?? []).map((m) => ({
name: m.name ?? '',
size: m.size ?? 0,
sizeVram: m.size_vram ?? 0,
contextLength: m.context_length ?? 0,
}));
} catch {
return [];
}
}
// ===== GET /api/version =====
async getVersion(): Promise<string> {
try {
const response = await fetch(`${this.baseURL}/api/version`, {
signal: AbortSignal.timeout(5_000),
});
if (!response.ok) return 'unknown';
const data = (await response.json()) as { version?: string };
return data.version ?? 'unknown';
} catch {
return 'unknown';
}
}
// ========== 私有转换方法 ==========
/**
* #2 修复: 下载 http(s) URL 图片并转为纯 base64 字符串(不含 data: 前缀)
*
* Ollama API 的 images 字段要求纯 base64 字符串数组。
* 当 MetonaMessage.images 中存储的是 URL 时,需先下载转为 base64。
* 下载失败时返回空字符串(Ollama 会忽略空图片),不阻断整个请求。
*/
private async resolveImageToBase64(url: string): Promise<string> {
try {
// v0.8.2 P0-1: 图片 URL 下载收口到 SSRF 安全通道(此前直连 fetch 无校验、
// 无大小上限;现含 DNS pinning/重定向复检/10MB 上限/类型白名单),外部
// 中断信号由 BaseAdapter.fetchImageAsBase64 透传。
const { base64 } = await this.fetchImageAsBase64(url, 30_000);
return base64;
} catch (error) {
log.warn(
`[Ollama] Failed to download image ${url.slice(0, 100)}: ${(error as Error).message}`,
);
return '';
}
}
private async toNativeRequest(request: MetonaRequest): Promise<Record<string, unknown>> {
// 说明:能力/num_ctx 探测由构造函数 fire-and-forget 发起(probeAttempts 上限
// 1 次)—— 此处不再重复探测,避免每次请求都打 /api/show(也保持测试环境的
// fetch 捕获不受污染)。探测失败时 cachedThinkingSupport=null → 门控 fail-open。
const messages: Record<string, unknown>[] = [
{
role: 'system',
content: [
request.systemPrompt.roleDefinition,
request.systemPrompt.outputConstraints,
request.systemPrompt.safetyGuidelines,
request.systemPrompt.dynamicReminders,
]
.filter(Boolean)
.join('\n\n'),
},
];
// #2 修复: 改为 for 循环以支持 async 图片下载(map 回调无法 await)
for (const m of request.messages) {
if (m.role === 'system') continue;
// C-6 修复: Ollama API 不支持 null contentassistant 仅有 tool_calls 时转为空字符串
const msg: Record<string, unknown> = { role: m.role, content: m.content ?? '' };
// Ollama 图片使用 images 字段(纯 base64 数组,不含 data: 前缀)
if (m.images?.length) {
// #2 修复: 支持公网 URL 图片,下载后转为纯 base64
// 之前直接将 URL 字符串传给 Ollama,导致 base64 解码错误
const resolvedImages: string[] = [];
for (const img of m.images) {
const url = img.url;
if (url.startsWith('data:')) {
// data:image/png;base64,iVBOR... → iVBOR...
const base64Part = url.split(',')[1];
resolvedImages.push(base64Part ?? url);
} else if (url.startsWith('http://') || url.startsWith('https://')) {
// #2 修复: 公网 URL → 下载 → 纯 base64
const base64 = await this.resolveImageToBase64(url);
if (base64) resolvedImages.push(base64);
} else {
// 已是纯 base64 字符串(无 data: 前缀)
resolvedImages.push(url);
}
}
msg.images = resolvedImages;
}
// 工具结果
if (m.role === 'tool' && m.toolResult) {
msg.tool_call_id = m.toolResult.toolCallId;
// CE-2 修复: 工具失败时 result 为 null,优先用 error 字段作为 content
msg.content = m.toolResult.error
? m.toolResult.error
: typeof m.toolResult.result === 'string'
? m.toolResult.result
: JSON.stringify(m.toolResult.result);
}
// assistant 工具调用(Ollama REST API 要求 arguments 为 JSON 字符串)
if (m.role === 'assistant' && m.toolCalls?.length) {
msg.tool_calls = m.toolCalls.map((tc) => ({
function: { name: tc.name, arguments: JSON.stringify(tc.args) },
}));
}
// 推理内容回传(保持多轮推理链完整)
if (m.role === 'assistant' && m.reasoningContent) {
(msg as Record<string, unknown>).reasoning_content = m.reasoningContent;
}
messages.push(msg);
}
const body: Record<string, unknown> = {
model: this.config.defaultModel,
messages,
options: {
temperature: request.params.temperature,
num_predict: request.params.maxTokens,
...(request.params.topP != null && { top_p: request.params.topP }),
...(request.params.stopSequences?.length && { stop: request.params.stopSequences }),
...(request.params.contextLength != null && { num_ctx: request.params.contextLength }),
},
};
// Tool Calling
if (request.tools?.length) {
body.tools = request.tools.map((t) => ({
type: 'function',
function: {
name: t.name,
description: t.description,
parameters: t.parameters,
},
}));
}
// Thinking 模式
// v0.8.0 P0-3v0.8.0 修订后保留的唯一门控): /api/show capabilities 探测为
// 不支持思考(cachedThinkingSupport === false)时不发 think 参数 —— 与云端
// Provider 不同,这是 Ollama 服务端的**硬协议约束**(向无思考能力的模型发
// think 会导致每次请求 400 "does not support thinking",而非静默忽略),
// 故此处门控属协议正确性而非用户意图覆盖;探测为服务端实时真值(非静态
// 元信息)。未探测/探测失败(nullfail-open 放行,与 listModels 能力回退
// 策略一致。探测在适配器实例创建时 fire-and-forget 发起(probeCapabilitiesOnce)。
if (request.params.thinkingEnabled) {
const modelThinkingSupported = this.cachedThinkingSupport !== false;
if (modelThinkingSupported) {
// v0.8.3: 新增 xhigh / true 档 —— Qwen3 等新模型思考模板原生支持 xhigh
//(旧版仅下发 low/medium/high 时,模板会 500 "Unexpected reasoning effort");
// true = 不指定档位(think: true),由模型思考模板用自身默认档。
// max 语义为"应用内最高档"Ollama 侧同样以布尔 true 放行给服务端默认。
const effortMap: Record<string, string | boolean> = {
low: 'low',
medium: 'medium',
high: 'high',
xhigh: 'xhigh',
max: true,
true: true,
};
body.think = effortMap[request.params.thinkingEffort ?? 'high'] ?? true;
// v0.8.0 P0-3: 思考占用 num_predict 输出预算 —— 预算过小时显式告警
const numPredict = request.params.maxTokens;
if (typeof numPredict === 'number' && numPredict < 8192) {
log.warn(
`[Ollama] thinking enabled with small num_predict budget (${numPredict}) — reasoning may consume the entire budget and truncate the answer`,
);
}
} else {
log.warn(
`[Ollama] model "${this.config.defaultModel}" does not support thinking (per /api/show capabilities) — omitting think parameter (server would reject with 400 otherwise)`,
);
}
}
return body;
}
private toMetonaResponse(
data: Record<string, unknown>,
requestId: string,
iteration: number = 0,
): MetonaResponse {
const message = data.message as Record<string, unknown> | undefined;
const toolCalls = message?.tool_calls as Array<Record<string, unknown>> | undefined;
return {
meta: {
requestId,
provider: this.providerId,
model: (data.model as string) ?? this.config.defaultModel,
latencyMs: 0,
timestamp: Date.now(),
perfStats: {
loadDurationMs: data.load_duration ? (data.load_duration as number) / 1e6 : undefined,
promptEvalDurationMs: data.prompt_eval_duration
? (data.prompt_eval_duration as number) / 1e6
: undefined,
evalDurationMs: data.eval_duration ? (data.eval_duration as number) / 1e6 : undefined,
tokensPerSecond:
data.eval_count && data.eval_duration
? (data.eval_count as number) / ((data.eval_duration as number) / 1e9)
: undefined,
},
},
content: (message?.content as string) ?? '',
reasoningContent: message?.thinking as string | undefined,
toolCalls: toolCalls?.map((tc) => {
const fn = tc.function as Record<string, unknown>;
const rawArgs = fn?.arguments;
let args: Record<string, unknown> = {};
try {
args =
typeof rawArgs === 'string'
? JSON.parse(rawArgs)
: ((rawArgs as Record<string, unknown>) ?? {});
} catch (parseErr) {
// v0.6.4: 非流式路径截断自愈对齐 —— 原 catch 静默降级 {},与流式修复后的
// 行为不一致。统一转为 _truncatedArguments 错误参数。
const sample =
typeof rawArgs === 'string' ? rawArgs.slice(-120) : String(rawArgs).slice(-120);
log.warn(
`[Ollama] Non-stream tool call args truncated (unparseable JSON, ${(parseErr as Error).message}). Tail: ...${sample}`,
);
args = truncatedArgumentsPayload((parseErr as Error).message, sample);
}
return {
// L-9 修复(审计补充): 非流式路径统一使用 nanoid,与流式路径(sendStream)保持一致
id: `tc_${nanoid(8)}`,
name: (fn?.name as string) ?? '',
args,
iteration,
timestamp: Date.now(),
};
}),
usage: {
inputTokens: (data.prompt_eval_count as number) ?? 0,
outputTokens: (data.eval_count as number) ?? 0,
totalTokens: ((data.prompt_eval_count as number) ?? 0) + ((data.eval_count as number) ?? 0),
},
finishReason: mapOllamaDoneReason(
data.done_reason as string | undefined,
!!message?.tool_calls,
),
};
}
}
/**
* 映射 Ollama done_reason → MetonaFinishReason
*
* @see apis/ollama-api-docs-20260518.html — /api/chat 响应字段
*/
function mapOllamaDoneReason(
reason: string | undefined,
hasToolCalls: boolean,
): MetonaFinishReason {
if (hasToolCalls) return MetonaFinishReason.TOOL_CALLS;
switch (reason) {
case 'stop':
return MetonaFinishReason.STOP;
case 'length':
return MetonaFinishReason.LENGTH;
case 'load':
return MetonaFinishReason.STOP; // 冷启动加载完成,非错误
case 'unload':
return MetonaFinishReason.STOP;
default:
return MetonaFinishReason.STOP;
}
}