817 lines
31 KiB
TypeScript
817 lines
31 KiB
TypeScript
/**
|
||
* Ollama Provider Adapter
|
||
*
|
||
* 本地推理引擎,支持 Tool Calling、Thinking 模式、NDJSON 流式。
|
||
* 无需 API Key,连接本地 http://localhost:11434。
|
||
*
|
||
* 完整实现 Ollama API 文档中所有端点和参数:
|
||
* - POST /api/chat (对话)
|
||
* - POST /api/generate (补全)
|
||
* - POST /api/embed (嵌入)
|
||
* - GET /api/tags (列出模型)
|
||
* - POST /api/show (模型详情)
|
||
* - POST /api/pull (下载模型)
|
||
* - GET /api/ps (运行中模型)
|
||
* - GET /api/version (版本)
|
||
* - think 参数(Thinking 模式)
|
||
* - options 参数(temperature/top_k/top_p/stop/num_ctx/num_predict)
|
||
* - format 参数(structured output)
|
||
* - images 参数(多模态)
|
||
* - tool_calls 流式处理
|
||
*
|
||
* @see apis/ollama-api-docs-20260518.html
|
||
*/
|
||
|
||
import { BaseAdapter } from './base-adapter';
|
||
import { truncatedArgumentsPayload, readStreamChunkWithIdleTimeout } from './shared/sse-stream';
|
||
import log from 'electron-log';
|
||
import { nanoid } from 'nanoid';
|
||
import type { MetonaRequest, MetonaResponse, MetonaStreamEvent } from '../types';
|
||
import { MetonaFinishReason, MetonaStreamEventType } from '../types';
|
||
import type { MetonaModelInfo } from '../types/metona-adapter';
|
||
|
||
export class OllamaAdapter extends BaseAdapter {
|
||
// H-2 修复: provider → providerId(规范要求)
|
||
override readonly providerId: string = 'ollama';
|
||
readonly supportedModels = ['qwen3:latest', 'gemma3:latest', 'deepseek-r1:latest'];
|
||
readonly supportsToolCalling = true;
|
||
readonly supportsThinking = true;
|
||
|
||
private baseURL: string;
|
||
|
||
constructor(config: ConstructorParameters<typeof BaseAdapter>[0]) {
|
||
super(config);
|
||
this.baseURL = config.baseURL || 'http://localhost:11434';
|
||
// v0.8.1: 上下文窗口不再从 /api/show 探测或写死默认值获取 —— 唯一合法来源是
|
||
// 设置面板「上下文长度」(llm.contextWindow → AdapterConfig.contextWindow,
|
||
// 引擎侧经 contextLength=num_ctx 下发)。构造时仅探测能力(thinking/tools/vision,
|
||
// 属协议正确性门控),不再缓存窗口数值。
|
||
this.probeCapabilitiesOnce();
|
||
}
|
||
|
||
/** /api/show 能力探测只发一次(构造函数发起),失败不重试(fail-open) */
|
||
private probeAttempted = false;
|
||
private probeInProgress = false;
|
||
/**
|
||
* /api/show capabilities 探测缓存 —— 模型是否支持思考(协议正确性门控,
|
||
* 非窗口/输出上限语义)。null = 未探测/探测失败(fail-open 放行,与
|
||
* listModels 能力回退策略一致);false = 服务端明确不支持 → 不发 think 参数。
|
||
*/
|
||
private cachedThinkingSupport: boolean | null = null;
|
||
|
||
/**
|
||
* v0.8.1: 构造时 fire-and-forget 探测一次默认模型能力(仅 thinking 门控消费)。
|
||
* 旧实现同时缓存 num_ctx 窗口数值 —— 已按"窗口唯一来源是设置面板"契约删除。
|
||
*/
|
||
private probeCapabilitiesOnce(): void {
|
||
if (this.probeAttempted) return;
|
||
this.probeAttempted = true;
|
||
this.probeInProgress = true;
|
||
void this.showModel(this.config.defaultModel)
|
||
.then((info) => {
|
||
if (Array.isArray(info?.capabilities)) {
|
||
this.cachedThinkingSupport = info!.capabilities.map(String).includes('thinking');
|
||
}
|
||
})
|
||
.catch(() => {
|
||
/* 模型探测失败不阻塞对话(fail-open) */
|
||
})
|
||
.finally(() => {
|
||
this.probeInProgress = false;
|
||
});
|
||
}
|
||
|
||
// ===== POST /api/chat =====
|
||
|
||
// H-2 修复: chat → send(规范要求)
|
||
async send(request: MetonaRequest): Promise<MetonaResponse> {
|
||
// #2 修复: toNativeRequest 改为 async(需下载 URL 图片转 base64)
|
||
const nativeRequest = await this.toNativeRequest(request);
|
||
|
||
// #24 修复: 使用 fetchWithTimeout 替代 getFetchSignal + fetch,确保 timer 清理
|
||
const response = await this.fetchWithTimeout(
|
||
`${this.baseURL}/api/chat`,
|
||
{
|
||
method: 'POST',
|
||
headers: { 'Content-Type': 'application/json' },
|
||
body: JSON.stringify({ ...nativeRequest, stream: false }),
|
||
},
|
||
this.config.timeoutMs ?? 300_000,
|
||
);
|
||
|
||
if (!response.ok) {
|
||
await this.throwHttpError(response, 'Ollama API error');
|
||
}
|
||
|
||
const data = (await response.json()) as Record<string, unknown>;
|
||
return this.toMetonaResponse(data, request.meta.requestId, request.meta.iteration);
|
||
}
|
||
|
||
// H-2 修复: chatStream → sendStream(规范要求)
|
||
async *sendStream(request: MetonaRequest): AsyncIterable<MetonaStreamEvent> {
|
||
// #2 修复: toNativeRequest 改为 async(需下载 URL 图片转 base64)
|
||
const nativeRequest = await this.toNativeRequest(request);
|
||
|
||
// #24 修复: 使用 fetchWithTimeout 替代 getFetchSignal + fetch,确保 timer 清理
|
||
const response = await this.fetchWithTimeout(
|
||
`${this.baseURL}/api/chat`,
|
||
{
|
||
method: 'POST',
|
||
headers: { 'Content-Type': 'application/json' },
|
||
body: JSON.stringify({ ...nativeRequest, stream: true }),
|
||
},
|
||
this.config.timeoutMs ?? 300_000,
|
||
);
|
||
|
||
if (!response.ok || !response.body) {
|
||
await this.throwHttpError(response, 'Ollama stream error');
|
||
}
|
||
|
||
// 非空断言:上方 if 已确保 response.body 不为 null
|
||
const reader = response.body!.getReader();
|
||
const decoder = new TextDecoder();
|
||
let seq = 0;
|
||
let buffer = '';
|
||
let streamEndedNormally = false;
|
||
|
||
while (true) {
|
||
// v0.7.4 P1-2: 空闲超时 — 本地模型加载/推理期间服务器可能长时间不推数据,
|
||
// 共享辅助在连续 60s 无数据时抛 SseUpstreamError(504) 进重试通道
|
||
const { done, value } = await readStreamChunkWithIdleTimeout(
|
||
reader,
|
||
60_000,
|
||
this.getExternalAbortSignal(),
|
||
);
|
||
if (done) break;
|
||
|
||
buffer += decoder.decode(value, { stream: true });
|
||
const lines = buffer.split('\n');
|
||
buffer = lines.pop() ?? '';
|
||
|
||
for (const line of lines) {
|
||
const trimmed = line.trim();
|
||
if (!trimmed) continue;
|
||
|
||
try {
|
||
const chunk = JSON.parse(trimmed);
|
||
|
||
// 思考内容
|
||
if (chunk.message?.thinking) {
|
||
yield {
|
||
type: MetonaStreamEventType.REASONING_DELTA,
|
||
requestId: request.meta.requestId,
|
||
sessionId: request.meta.sessionId,
|
||
iteration: request.meta.iteration,
|
||
seq: seq++,
|
||
timestamp: Date.now(),
|
||
delta: chunk.message.thinking,
|
||
};
|
||
}
|
||
|
||
// 文本内容
|
||
if (chunk.message?.content) {
|
||
yield {
|
||
type: MetonaStreamEventType.TEXT_DELTA,
|
||
requestId: request.meta.requestId,
|
||
sessionId: request.meta.sessionId,
|
||
iteration: request.meta.iteration,
|
||
seq: seq++,
|
||
timestamp: Date.now(),
|
||
delta: chunk.message.content,
|
||
};
|
||
}
|
||
|
||
// 工具调用(Ollama 在最后一个 chunk 中整块返回)
|
||
if (chunk.message?.tool_calls) {
|
||
for (const tc of chunk.message.tool_calls) {
|
||
const args = tc.function?.arguments;
|
||
// v0.6.4 缺口修复: NDJSON 路径的截断自愈 —— 原实现 JSON.parse 抛错会
|
||
// 落入外层 catch:该 tool call 整体静默丢弃,且同一行剩余处理
|
||
// (含 done/USAGE 检查)一并被跳过,与 v0.6.3 已根治的 OpenAI 共享层
|
||
// 旧行为完全相同。现独立捕获并转为 _truncatedArguments 自愈载荷,
|
||
// 同时保证本 chunk 的后续分支照常执行。
|
||
let parsedArgs: Record<string, unknown>;
|
||
if (typeof args === 'string') {
|
||
try {
|
||
parsedArgs = JSON.parse(args);
|
||
} catch (parseErr) {
|
||
const sample = args.slice(-120);
|
||
log.warn(
|
||
`[Ollama] Tool call args truncated (unparseable JSON, ${(parseErr as Error).message}). Tail: ...${sample}`,
|
||
);
|
||
parsedArgs = truncatedArgumentsPayload((parseErr as Error).message, sample);
|
||
}
|
||
} else {
|
||
parsedArgs = (args as Record<string, unknown>) ?? {};
|
||
}
|
||
yield {
|
||
type: MetonaStreamEventType.TOOL_CALL_COMPLETE,
|
||
requestId: request.meta.requestId,
|
||
sessionId: request.meta.sessionId,
|
||
iteration: request.meta.iteration,
|
||
seq: seq++,
|
||
timestamp: Date.now(),
|
||
toolCall: {
|
||
// L-9 修复: 统一使用 nanoid 生成工具调用 ID(与 sse-stream.ts 一致)
|
||
id: `tc_${nanoid(8)}`,
|
||
name: tc.function?.name ?? '',
|
||
args: parsedArgs,
|
||
iteration: request.meta.iteration,
|
||
timestamp: Date.now(),
|
||
},
|
||
};
|
||
}
|
||
}
|
||
|
||
// 流结束
|
||
if (chunk.done) {
|
||
streamEndedNormally = true;
|
||
// v0.8.0 P0-1: 采集 done_reason —— Ollama 的停止原因在最终 done chunk
|
||
// 携带;旧实现流式路径完全忽略(length 截断不可见),现归一化后随
|
||
// DONE 事件上交引擎
|
||
const doneReason = mapOllamaDoneReason(
|
||
chunk.done_reason as string | undefined,
|
||
Array.isArray(chunk.message?.tool_calls) && chunk.message.tool_calls.length > 0,
|
||
);
|
||
const finishReason = doneReason === MetonaFinishReason.LENGTH ? 'length' : undefined;
|
||
// 发送 usage 信息
|
||
yield {
|
||
type: MetonaStreamEventType.USAGE,
|
||
requestId: request.meta.requestId,
|
||
sessionId: request.meta.sessionId,
|
||
iteration: request.meta.iteration,
|
||
seq: seq++,
|
||
timestamp: Date.now(),
|
||
usage: {
|
||
inputTokens: chunk.prompt_eval_count ?? 0,
|
||
outputTokens: chunk.eval_count ?? 0,
|
||
totalTokens: (chunk.prompt_eval_count ?? 0) + (chunk.eval_count ?? 0),
|
||
},
|
||
};
|
||
|
||
yield {
|
||
type: MetonaStreamEventType.DONE,
|
||
requestId: request.meta.requestId,
|
||
sessionId: request.meta.sessionId,
|
||
iteration: request.meta.iteration,
|
||
seq: seq++,
|
||
timestamp: Date.now(),
|
||
// v0.8.0 P0-1: 仅 length 需要显式上交(stop/tool_calls 语义由
|
||
// 事件流本身表达;load/unload 属本地引擎状态非响应语义)
|
||
...(finishReason ? { finishReason } : {}),
|
||
};
|
||
return;
|
||
}
|
||
} catch (parseErr) {
|
||
// P2-8 修复: 与 sse-stream.ts 一致,记录解析失败行便于诊断
|
||
log.warn(
|
||
`[Ollama] Failed to parse NDJSON line: ${(parseErr as Error).message}`,
|
||
trimmed.slice(0, 200),
|
||
);
|
||
}
|
||
}
|
||
}
|
||
|
||
// 流未正常结束(连接断开等),补发 DONE 事件防止 Agent Loop 挂起
|
||
if (!streamEndedNormally) {
|
||
yield {
|
||
type: MetonaStreamEventType.DONE,
|
||
requestId: request.meta.requestId,
|
||
sessionId: request.meta.sessionId,
|
||
iteration: request.meta.iteration,
|
||
seq: seq++,
|
||
timestamp: Date.now(),
|
||
};
|
||
}
|
||
}
|
||
|
||
// ===== POST /api/generate =====
|
||
|
||
async generate(params: {
|
||
model: string;
|
||
prompt: string;
|
||
suffix?: string;
|
||
system?: string;
|
||
stream?: boolean;
|
||
think?: boolean | string;
|
||
format?: string | object;
|
||
images?: string[];
|
||
options?: Record<string, unknown>;
|
||
/** v0.8.0 P1-3.2: 外部取消信号(与固定 300s 超时合并,任一触发即中止) */
|
||
signal?: AbortSignal;
|
||
}): Promise<{
|
||
response: string;
|
||
thinking?: string;
|
||
done: boolean;
|
||
totalDuration: number;
|
||
evalCount: number;
|
||
}> {
|
||
const response = await fetch(`${this.baseURL}/api/generate`, {
|
||
method: 'POST',
|
||
headers: { 'Content-Type': 'application/json' },
|
||
body: JSON.stringify({ ...params, stream: false, signal: undefined }),
|
||
// v0.8.0 P1-3.2 根治: 旧实现固定 AbortSignal.timeout(300_000) —— 外部中断
|
||
// 无法取消该请求(最长 5 分钟资源悬挂)。现用 AbortSignal.any 合并外部
|
||
// signal 与超时信号,任一触发即中止(Node ≥20.3 / Electron 35 满足)。
|
||
signal: params.signal
|
||
? AbortSignal.any([params.signal, AbortSignal.timeout(300_000)])
|
||
: AbortSignal.timeout(300_000),
|
||
});
|
||
|
||
if (!response.ok) throw new Error(`Ollama generate error: ${response.status}`);
|
||
const data = (await response.json()) as {
|
||
response?: string;
|
||
thinking?: string;
|
||
done?: boolean;
|
||
total_duration?: number;
|
||
eval_count?: number;
|
||
};
|
||
|
||
return {
|
||
response: data.response ?? '',
|
||
thinking: data.thinking,
|
||
done: data.done ?? true,
|
||
totalDuration: data.total_duration ?? 0,
|
||
evalCount: data.eval_count ?? 0,
|
||
};
|
||
}
|
||
|
||
// ===== POST /api/embed =====
|
||
|
||
async embed(params: {
|
||
model: string;
|
||
input: string | string[];
|
||
dimensions?: number;
|
||
/** v0.8.0 P1-3.2: 外部取消信号(与固定 60s 超时合并) */
|
||
signal?: AbortSignal;
|
||
}): Promise<{ embeddings: number[][]; totalDuration: number }> {
|
||
const response = await fetch(`${this.baseURL}/api/embed`, {
|
||
method: 'POST',
|
||
headers: { 'Content-Type': 'application/json' },
|
||
body: JSON.stringify(params),
|
||
// v0.8.0 P1-3.2: 与 generate 同口径 —— 外部 signal 与超时合并
|
||
signal: params.signal
|
||
? AbortSignal.any([params.signal, AbortSignal.timeout(60_000)])
|
||
: AbortSignal.timeout(60_000),
|
||
});
|
||
|
||
if (!response.ok) throw new Error(`Ollama embed error: ${response.status}`);
|
||
const data = (await response.json()) as { embeddings?: number[][]; total_duration?: number };
|
||
|
||
return {
|
||
embeddings: data.embeddings ?? [],
|
||
totalDuration: data.total_duration ?? 0,
|
||
};
|
||
}
|
||
|
||
// ===== GET /api/tags =====
|
||
|
||
/**
|
||
* H-2 修复: 返回 MetonaModelInfo[](规范要求)
|
||
*
|
||
* Ollama /api/tags 返回模型列表含详细信息(name, size, details),
|
||
* 转换为 MetonaModelInfo 并补充默认元数据。
|
||
*/
|
||
async listModels(): Promise<MetonaModelInfo[]> {
|
||
try {
|
||
const response = await fetch(`${this.baseURL}/api/tags`, {
|
||
signal: AbortSignal.timeout(10_000),
|
||
});
|
||
if (response.ok) {
|
||
const data = (await response.json()) as {
|
||
models?: Array<{
|
||
name: string;
|
||
size?: number;
|
||
details?: { parameter_size?: string; quantization_level?: string; family?: string };
|
||
}>;
|
||
};
|
||
if (data.models?.length) {
|
||
// v0.6.4 P4-1: 能力标志改为逐模型 /api/show 实测探测;单个探测失败
|
||
// 该模型回退保守 true(不可用时行为与旧实现一致,fail-open 保可用性)
|
||
// v0.7.3 P1-4: supportsVision 随探测结果透出(undefined = 未知 → 前端保守放行),
|
||
// 供上传入口拒绝不支持图片的本地语言模型
|
||
// v0.8.1: 不再填充 contextWindow —— 窗口唯一来源是设置面板「上下文长度」
|
||
const enriched = await Promise.all(
|
||
data.models.map(async (m) => {
|
||
const caps = await this.probeCapabilities(m.name);
|
||
return {
|
||
id: m.name,
|
||
name: m.name,
|
||
supportsToolCalling: caps ? caps.supportsTools : true,
|
||
supportsThinking: caps ? caps.supportsThinking : true,
|
||
supportsVision: caps ? caps.supportsVision : undefined,
|
||
description: m.details
|
||
? `${m.details.family ?? 'unknown'} / ${m.details.parameter_size ?? '?'} / ${m.details.quantization_level ?? '?'}`
|
||
: undefined,
|
||
};
|
||
}),
|
||
);
|
||
return enriched;
|
||
}
|
||
}
|
||
} catch {
|
||
// API 不可用时降级
|
||
}
|
||
// 回退到 supportedModels
|
||
return this.supportedModels.map((id) => ({ id }));
|
||
}
|
||
|
||
/**
|
||
* H-2 修复: 获取上下文窗口大小(规范要求)
|
||
*
|
||
* v0.8.1 硬性契约: 唯一来源是设置面板「上下文长度」(llm.contextWindow),
|
||
* 未配置返回 0 —— 删除了旧的 4096 写死默认值与 /api/show num_ctx 探测缓存。
|
||
* Ollama 的 num_ctx 由引擎经 params.contextLength(同源配置)下发给服务端。
|
||
*/
|
||
override getContextWindow(): number {
|
||
if (typeof this.config.contextWindow === 'number' && this.config.contextWindow > 0) {
|
||
return this.config.contextWindow;
|
||
}
|
||
return 0;
|
||
}
|
||
|
||
/**
|
||
* v0.6.4 P4-1: 通过 /api/show 的 capabilities[] 动态探测模型真实能力。
|
||
* 此前 listModels 对所有本地模型硬编码 supportsToolCalling/supportsThinking:true
|
||
* (注释自知不准)—— 语言模型不支持 tools 时引擎仍下发工具定义,
|
||
* 造成"模型口头说调工具实际不调"的回归温床。探测失败返回 null 由调用方回退保守值。
|
||
*/
|
||
async probeCapabilities(model: string): Promise<{
|
||
supportsTools: boolean;
|
||
supportsVision: boolean;
|
||
supportsThinking: boolean;
|
||
} | null> {
|
||
const info = await this.showModel(model);
|
||
if (!info || !Array.isArray(info.capabilities)) return null;
|
||
const caps = new Set(info.capabilities.map((c) => String(c)));
|
||
return {
|
||
supportsTools: caps.has('tools'),
|
||
supportsVision: caps.has('vision'),
|
||
supportsThinking: caps.has('thinking'),
|
||
};
|
||
}
|
||
|
||
// ===== POST /api/show =====
|
||
|
||
async showModel(
|
||
model: string,
|
||
): Promise<{ parameters: string; template: string; capabilities?: string[] } | null> {
|
||
try {
|
||
const response = await fetch(`${this.baseURL}/api/show`, {
|
||
method: 'POST',
|
||
headers: { 'Content-Type': 'application/json' },
|
||
body: JSON.stringify({ model }),
|
||
signal: AbortSignal.timeout(10_000),
|
||
});
|
||
if (!response.ok) return null;
|
||
const data = (await response.json()) as {
|
||
parameters?: string;
|
||
template?: string;
|
||
capabilities?: string[];
|
||
};
|
||
return {
|
||
parameters: data.parameters ?? '',
|
||
template: data.template ?? '',
|
||
// v0.8.1: 保留 undefined 语义 —— 响应未携带 capabilities 字段 = 未知
|
||
//(fail-open),显式数组(含空数组 = 服务端权威"无任何能力")才参与门控。
|
||
// 旧实现 `?? []` 把"字段缺失"与"权威空"混同,探测竞态下会误判不支持思考。
|
||
capabilities: data.capabilities,
|
||
};
|
||
} catch {
|
||
return null;
|
||
}
|
||
}
|
||
|
||
// ===== POST /api/pull =====
|
||
|
||
/**
|
||
* v0.6.4 P4-1 重构:pull 支持外部取消信号 —— 原实现固定 600s 超时会掐死
|
||
* 大模型下载(进度不能续命、无取消通道),大仓/慢网络场景必然失败。
|
||
* 现契约:调用方通过 AbortSignal 控制生命周期(UI 取消按钮即可触发);
|
||
* 超时语义交给用户取消或服务端断流(读循环结束即完成),不再人为设上限。
|
||
*/
|
||
async pullModel(
|
||
model: string,
|
||
onProgress?: (progress: { status: string; completed?: number; total?: number }) => void,
|
||
signal?: AbortSignal,
|
||
): Promise<void> {
|
||
const response = await fetch(`${this.baseURL}/api/pull`, {
|
||
method: 'POST',
|
||
headers: { 'Content-Type': 'application/json' },
|
||
body: JSON.stringify({ model, stream: true }),
|
||
signal,
|
||
});
|
||
|
||
if (!response.ok || !response.body) throw new Error(`Ollama pull error: ${response.status}`);
|
||
|
||
const reader = response.body.getReader();
|
||
const decoder = new TextDecoder();
|
||
let buffer = '';
|
||
|
||
while (true) {
|
||
const { done, value } = await reader.read();
|
||
if (done) break;
|
||
buffer += decoder.decode(value, { stream: true });
|
||
const lines = buffer.split('\n');
|
||
buffer = lines.pop() ?? '';
|
||
for (const line of lines) {
|
||
if (!line.trim()) continue;
|
||
try {
|
||
const chunk = JSON.parse(line);
|
||
onProgress?.({ status: chunk.status, completed: chunk.completed, total: chunk.total });
|
||
} catch {
|
||
// L-3 修复: 添加日志便于诊断非标准行(如进度通知、空行等)
|
||
log.debug('[Ollama] skipped non-JSON line during pull:', line.slice(0, 100));
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
// ===== GET /api/ps =====
|
||
|
||
async listRunning(): Promise<
|
||
Array<{ name: string; size: number; sizeVram: number; contextLength: number }>
|
||
> {
|
||
try {
|
||
const response = await fetch(`${this.baseURL}/api/ps`, {
|
||
signal: AbortSignal.timeout(10_000),
|
||
});
|
||
if (!response.ok) return [];
|
||
const data = (await response.json()) as {
|
||
models?: Array<{
|
||
name: string;
|
||
size?: number;
|
||
size_vram?: number;
|
||
context_length?: number;
|
||
}>;
|
||
};
|
||
return (data.models ?? []).map((m) => ({
|
||
name: m.name ?? '',
|
||
size: m.size ?? 0,
|
||
sizeVram: m.size_vram ?? 0,
|
||
contextLength: m.context_length ?? 0,
|
||
}));
|
||
} catch {
|
||
return [];
|
||
}
|
||
}
|
||
|
||
// ===== GET /api/version =====
|
||
|
||
async getVersion(): Promise<string> {
|
||
try {
|
||
const response = await fetch(`${this.baseURL}/api/version`, {
|
||
signal: AbortSignal.timeout(5_000),
|
||
});
|
||
if (!response.ok) return 'unknown';
|
||
const data = (await response.json()) as { version?: string };
|
||
return data.version ?? 'unknown';
|
||
} catch {
|
||
return 'unknown';
|
||
}
|
||
}
|
||
|
||
// ========== 私有转换方法 ==========
|
||
|
||
/**
|
||
* #2 修复: 下载 http(s) URL 图片并转为纯 base64 字符串(不含 data: 前缀)
|
||
*
|
||
* Ollama API 的 images 字段要求纯 base64 字符串数组。
|
||
* 当 MetonaMessage.images 中存储的是 URL 时,需先下载转为 base64。
|
||
* 下载失败时返回空字符串(Ollama 会忽略空图片),不阻断整个请求。
|
||
*/
|
||
private async resolveImageToBase64(url: string): Promise<string> {
|
||
try {
|
||
// v0.8.2 P0-1: 图片 URL 下载收口到 SSRF 安全通道(此前直连 fetch 无校验、
|
||
// 无大小上限;现含 DNS pinning/重定向复检/10MB 上限/类型白名单),外部
|
||
// 中断信号由 BaseAdapter.fetchImageAsBase64 透传。
|
||
const { base64 } = await this.fetchImageAsBase64(url, 30_000);
|
||
return base64;
|
||
} catch (error) {
|
||
log.warn(
|
||
`[Ollama] Failed to download image ${url.slice(0, 100)}: ${(error as Error).message}`,
|
||
);
|
||
return '';
|
||
}
|
||
}
|
||
|
||
private async toNativeRequest(request: MetonaRequest): Promise<Record<string, unknown>> {
|
||
// 说明:能力/num_ctx 探测由构造函数 fire-and-forget 发起(probeAttempts 上限
|
||
// 1 次)—— 此处不再重复探测,避免每次请求都打 /api/show(也保持测试环境的
|
||
// fetch 捕获不受污染)。探测失败时 cachedThinkingSupport=null → 门控 fail-open。
|
||
const messages: Record<string, unknown>[] = [
|
||
{
|
||
role: 'system',
|
||
content: [
|
||
request.systemPrompt.roleDefinition,
|
||
request.systemPrompt.outputConstraints,
|
||
request.systemPrompt.safetyGuidelines,
|
||
request.systemPrompt.dynamicReminders,
|
||
]
|
||
.filter(Boolean)
|
||
.join('\n\n'),
|
||
},
|
||
];
|
||
|
||
// #2 修复: 改为 for 循环以支持 async 图片下载(map 回调无法 await)
|
||
for (const m of request.messages) {
|
||
if (m.role === 'system') continue;
|
||
// C-6 修复: Ollama API 不支持 null content,assistant 仅有 tool_calls 时转为空字符串
|
||
const msg: Record<string, unknown> = { role: m.role, content: m.content ?? '' };
|
||
// Ollama 图片使用 images 字段(纯 base64 数组,不含 data: 前缀)
|
||
if (m.images?.length) {
|
||
// #2 修复: 支持公网 URL 图片,下载后转为纯 base64
|
||
// 之前直接将 URL 字符串传给 Ollama,导致 base64 解码错误
|
||
const resolvedImages: string[] = [];
|
||
for (const img of m.images) {
|
||
const url = img.url;
|
||
if (url.startsWith('data:')) {
|
||
// data:image/png;base64,iVBOR... → iVBOR...
|
||
const base64Part = url.split(',')[1];
|
||
resolvedImages.push(base64Part ?? url);
|
||
} else if (url.startsWith('http://') || url.startsWith('https://')) {
|
||
// #2 修复: 公网 URL → 下载 → 纯 base64
|
||
const base64 = await this.resolveImageToBase64(url);
|
||
if (base64) resolvedImages.push(base64);
|
||
} else {
|
||
// 已是纯 base64 字符串(无 data: 前缀)
|
||
resolvedImages.push(url);
|
||
}
|
||
}
|
||
msg.images = resolvedImages;
|
||
}
|
||
// 工具结果
|
||
if (m.role === 'tool' && m.toolResult) {
|
||
msg.tool_call_id = m.toolResult.toolCallId;
|
||
// CE-2 修复: 工具失败时 result 为 null,优先用 error 字段作为 content
|
||
msg.content = m.toolResult.error
|
||
? m.toolResult.error
|
||
: typeof m.toolResult.result === 'string'
|
||
? m.toolResult.result
|
||
: JSON.stringify(m.toolResult.result);
|
||
}
|
||
// assistant 工具调用(Ollama REST API 要求 arguments 为 JSON 字符串)
|
||
if (m.role === 'assistant' && m.toolCalls?.length) {
|
||
msg.tool_calls = m.toolCalls.map((tc) => ({
|
||
function: { name: tc.name, arguments: JSON.stringify(tc.args) },
|
||
}));
|
||
}
|
||
// 推理内容回传(保持多轮推理链完整)
|
||
if (m.role === 'assistant' && m.reasoningContent) {
|
||
(msg as Record<string, unknown>).reasoning_content = m.reasoningContent;
|
||
}
|
||
messages.push(msg);
|
||
}
|
||
|
||
const body: Record<string, unknown> = {
|
||
model: this.config.defaultModel,
|
||
messages,
|
||
options: {
|
||
temperature: request.params.temperature,
|
||
num_predict: request.params.maxTokens,
|
||
...(request.params.topP != null && { top_p: request.params.topP }),
|
||
...(request.params.stopSequences?.length && { stop: request.params.stopSequences }),
|
||
...(request.params.contextLength != null && { num_ctx: request.params.contextLength }),
|
||
},
|
||
};
|
||
|
||
// Tool Calling
|
||
if (request.tools?.length) {
|
||
body.tools = request.tools.map((t) => ({
|
||
type: 'function',
|
||
function: {
|
||
name: t.name,
|
||
description: t.description,
|
||
parameters: t.parameters,
|
||
},
|
||
}));
|
||
}
|
||
|
||
// Thinking 模式
|
||
// v0.8.0 P0-3(v0.8.0 修订后保留的唯一门控): /api/show capabilities 探测为
|
||
// 不支持思考(cachedThinkingSupport === false)时不发 think 参数 —— 与云端
|
||
// Provider 不同,这是 Ollama 服务端的**硬协议约束**(向无思考能力的模型发
|
||
// think 会导致每次请求 400 "does not support thinking",而非静默忽略),
|
||
// 故此处门控属协议正确性而非用户意图覆盖;探测为服务端实时真值(非静态
|
||
// 元信息)。未探测/探测失败(null)fail-open 放行,与 listModels 能力回退
|
||
// 策略一致。探测在适配器实例创建时 fire-and-forget 发起(probeCapabilitiesOnce)。
|
||
if (request.params.thinkingEnabled) {
|
||
const modelThinkingSupported = this.cachedThinkingSupport !== false;
|
||
if (modelThinkingSupported) {
|
||
const effortMap: Record<string, string | boolean> = {
|
||
low: 'low',
|
||
medium: 'medium',
|
||
high: 'high',
|
||
max: true,
|
||
};
|
||
body.think = effortMap[request.params.thinkingEffort ?? 'high'] ?? true;
|
||
// v0.8.0 P0-3: 思考占用 num_predict 输出预算 —— 预算过小时显式告警
|
||
const numPredict = request.params.maxTokens;
|
||
if (typeof numPredict === 'number' && numPredict < 8192) {
|
||
log.warn(
|
||
`[Ollama] thinking enabled with small num_predict budget (${numPredict}) — reasoning may consume the entire budget and truncate the answer`,
|
||
);
|
||
}
|
||
} else {
|
||
log.warn(
|
||
`[Ollama] model "${this.config.defaultModel}" does not support thinking (per /api/show capabilities) — omitting think parameter (server would reject with 400 otherwise)`,
|
||
);
|
||
}
|
||
}
|
||
|
||
return body;
|
||
}
|
||
|
||
private toMetonaResponse(
|
||
data: Record<string, unknown>,
|
||
requestId: string,
|
||
iteration: number = 0,
|
||
): MetonaResponse {
|
||
const message = data.message as Record<string, unknown> | undefined;
|
||
const toolCalls = message?.tool_calls as Array<Record<string, unknown>> | undefined;
|
||
return {
|
||
meta: {
|
||
requestId,
|
||
provider: this.providerId,
|
||
model: (data.model as string) ?? this.config.defaultModel,
|
||
latencyMs: 0,
|
||
timestamp: Date.now(),
|
||
perfStats: {
|
||
loadDurationMs: data.load_duration ? (data.load_duration as number) / 1e6 : undefined,
|
||
promptEvalDurationMs: data.prompt_eval_duration
|
||
? (data.prompt_eval_duration as number) / 1e6
|
||
: undefined,
|
||
evalDurationMs: data.eval_duration ? (data.eval_duration as number) / 1e6 : undefined,
|
||
tokensPerSecond:
|
||
data.eval_count && data.eval_duration
|
||
? (data.eval_count as number) / ((data.eval_duration as number) / 1e9)
|
||
: undefined,
|
||
},
|
||
},
|
||
content: (message?.content as string) ?? '',
|
||
reasoningContent: message?.thinking as string | undefined,
|
||
toolCalls: toolCalls?.map((tc) => {
|
||
const fn = tc.function as Record<string, unknown>;
|
||
const rawArgs = fn?.arguments;
|
||
let args: Record<string, unknown> = {};
|
||
try {
|
||
args =
|
||
typeof rawArgs === 'string'
|
||
? JSON.parse(rawArgs)
|
||
: ((rawArgs as Record<string, unknown>) ?? {});
|
||
} catch (parseErr) {
|
||
// v0.6.4: 非流式路径截断自愈对齐 —— 原 catch 静默降级 {},与流式修复后的
|
||
// 行为不一致。统一转为 _truncatedArguments 错误参数。
|
||
const sample =
|
||
typeof rawArgs === 'string' ? rawArgs.slice(-120) : String(rawArgs).slice(-120);
|
||
log.warn(
|
||
`[Ollama] Non-stream tool call args truncated (unparseable JSON, ${(parseErr as Error).message}). Tail: ...${sample}`,
|
||
);
|
||
args = truncatedArgumentsPayload((parseErr as Error).message, sample);
|
||
}
|
||
return {
|
||
// L-9 修复(审计补充): 非流式路径统一使用 nanoid,与流式路径(sendStream)保持一致
|
||
id: `tc_${nanoid(8)}`,
|
||
name: (fn?.name as string) ?? '',
|
||
args,
|
||
iteration,
|
||
timestamp: Date.now(),
|
||
};
|
||
}),
|
||
usage: {
|
||
inputTokens: (data.prompt_eval_count as number) ?? 0,
|
||
outputTokens: (data.eval_count as number) ?? 0,
|
||
totalTokens: ((data.prompt_eval_count as number) ?? 0) + ((data.eval_count as number) ?? 0),
|
||
},
|
||
finishReason: mapOllamaDoneReason(
|
||
data.done_reason as string | undefined,
|
||
!!message?.tool_calls,
|
||
),
|
||
};
|
||
}
|
||
}
|
||
|
||
/**
|
||
* 映射 Ollama done_reason → MetonaFinishReason
|
||
*
|
||
* @see apis/ollama-api-docs-20260518.html — /api/chat 响应字段
|
||
*/
|
||
function mapOllamaDoneReason(
|
||
reason: string | undefined,
|
||
hasToolCalls: boolean,
|
||
): MetonaFinishReason {
|
||
if (hasToolCalls) return MetonaFinishReason.TOOL_CALLS;
|
||
switch (reason) {
|
||
case 'stop':
|
||
return MetonaFinishReason.STOP;
|
||
case 'length':
|
||
return MetonaFinishReason.LENGTH;
|
||
case 'load':
|
||
return MetonaFinishReason.STOP; // 冷启动加载完成,非错误
|
||
case 'unload':
|
||
return MetonaFinishReason.STOP;
|
||
default:
|
||
return MetonaFinishReason.STOP;
|
||
}
|
||
}
|