|
@@ -15,19 +15,16 @@ export const PROVIDER_ID = 'zsjz';
|
|
|
|
|
|
|
|
export type AppliedProvider = AppliedProviderInfo;
|
|
export type AppliedProvider = AppliedProviderInfo;
|
|
|
|
|
|
|
|
-/** 端点协议判定:responses 直通;chat-only 走内置桥接;unsupported 双协议都没有;unreachable 网络不可达 */
|
|
|
|
|
-export type EndpointProtocol = 'responses' | 'chat-only' | 'unsupported' | 'unreachable';
|
|
|
|
|
|
|
+/**
|
|
|
|
|
+ * 端点判定。后端「模型管理」里的记录一律是 Chat 协议(没有 /responses),
|
|
|
|
|
+ * 所以这里只关心「能不能用 Chat」:chat → 走内置桥接;其余两种拒绝。
|
|
|
|
|
+ */
|
|
|
|
|
+export type EndpointProtocol = 'chat' | 'unsupported' | 'unreachable';
|
|
|
|
|
|
|
|
export interface EndpointProbeResult {
|
|
export interface EndpointProbeResult {
|
|
|
protocol: EndpointProtocol;
|
|
protocol: EndpointProtocol;
|
|
|
- /** /responses 的 HTTP 状态;未探或网络失败为 null */
|
|
|
|
|
- responsesStatus: number | null;
|
|
|
|
|
- /** /chat/completions 的 HTTP 状态;未探为 null */
|
|
|
|
|
|
|
+ /** /chat/completions 的 HTTP 状态;未探或网络失败为 null */
|
|
|
chatStatus: number | null;
|
|
chatStatus: number | null;
|
|
|
- /** /api/version 探到的 Ollama 版本;非 Ollama 或未探为 null */
|
|
|
|
|
- ollamaVersion: string | null;
|
|
|
|
|
- /** Ollama 低于 0.13.3(没有非状态化 /v1/responses):提示升级,但允许走桥接 */
|
|
|
|
|
- ollamaNeedsUpgrade: boolean;
|
|
|
|
|
detail: string;
|
|
detail: string;
|
|
|
}
|
|
}
|
|
|
|
|
|
|
@@ -42,7 +39,7 @@ function providerFile(): string {
|
|
|
return join(getCodexDataDir(), 'provider.json');
|
|
return join(getCodexDataDir(), 'provider.json');
|
|
|
}
|
|
}
|
|
|
|
|
|
|
|
-/** Codex 要求 base_url 是 API 根(形如 https://api.openai.com/v1),它自己再拼 /responses */
|
|
|
|
|
|
|
+/** Codex 要求 base_url 是 API 根(形如 https://api.openai.com/v1),它自己再拼路由 */
|
|
|
export function normalizeBaseUrl(raw: string | null | undefined): string | null {
|
|
export function normalizeBaseUrl(raw: string | null | undefined): string | null {
|
|
|
const value = (raw ?? '').trim().replace(/\/+$/u, '');
|
|
const value = (raw ?? '').trim().replace(/\/+$/u, '');
|
|
|
return value || null;
|
|
return value || null;
|
|
@@ -53,7 +50,7 @@ export function baseUrlHint(baseUrl: string | null): string | null {
|
|
|
if (!baseUrl) return '未配置 base_url';
|
|
if (!baseUrl) return '未配置 base_url';
|
|
|
if (!/^https?:\/\//u.test(baseUrl)) return 'base_url 必须以 http:// 或 https:// 开头';
|
|
if (!/^https?:\/\//u.test(baseUrl)) return 'base_url 必须以 http:// 或 https:// 开头';
|
|
|
if (!/\/v\d+$/u.test(baseUrl)) {
|
|
if (!/\/v\d+$/u.test(baseUrl)) {
|
|
|
- return 'base_url 通常应写到版本段(如 .../v1),否则 Codex 拼出的 /responses 可能 404';
|
|
|
|
|
|
|
+ return 'base_url 通常应写到版本段(如 .../v1),否则拼出的 /chat/completions 可能 404';
|
|
|
}
|
|
}
|
|
|
return null;
|
|
return null;
|
|
|
}
|
|
}
|
|
@@ -120,143 +117,101 @@ export function describeDroppedKeys(input: ApplyProviderInput): string[] {
|
|
|
return Object.keys(config ?? {}).filter((key) => !EXTRA_ALLOWLIST.includes(key));
|
|
return Object.keys(config ?? {}).filter((key) => !EXTRA_ALLOWLIST.includes(key));
|
|
|
}
|
|
}
|
|
|
|
|
|
|
|
-/** Ollama 自 0.13.3 起提供非状态化 /v1/responses(流式 / 工具调用 / reasoning summaries) */
|
|
|
|
|
-const OLLAMA_RESPONSES_MIN_VERSION: readonly number[] = [0, 13, 3];
|
|
|
|
|
|
|
+/**
|
|
|
|
|
+ * 探测只问「路由在不在」,绝不让端点真去生成:本地服务多是单槽推理,
|
|
|
|
|
+ * 一次生成能占住整个 HTTP 服务几十秒(实测 max_tokens:1 也要 14 秒,期间连 /models 都不应答)。
|
|
|
|
|
+ * 15 秒是留给排队的余量。
|
|
|
|
|
+ */
|
|
|
|
|
+const PROBE_TIMEOUT_MS = 15_000;
|
|
|
|
|
|
|
|
interface ProbeAttempt {
|
|
interface ProbeAttempt {
|
|
|
status: number | null;
|
|
status: number | null;
|
|
|
error: string | null;
|
|
error: string | null;
|
|
|
|
|
+ ms: number;
|
|
|
|
|
+ timedOut: boolean;
|
|
|
}
|
|
}
|
|
|
|
|
|
|
|
-async function postProbe(url: string, body: Record<string, unknown>, apiKey?: string | null): Promise<ProbeAttempt> {
|
|
|
|
|
|
|
+function describeFetchError(error: unknown): { message: string; timedOut: boolean } {
|
|
|
|
|
+ const err = error instanceof Error ? error : new Error(String(error));
|
|
|
|
|
+ return {
|
|
|
|
|
+ message: err.message,
|
|
|
|
|
+ timedOut: err.name === 'TimeoutError' || err.name === 'AbortError' || /timeout/iu.test(err.message),
|
|
|
|
|
+ };
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+async function probeFetch(url: string, init: RequestInit, timeoutMs: number): Promise<ProbeAttempt> {
|
|
|
|
|
+ const startedAt = Date.now();
|
|
|
try {
|
|
try {
|
|
|
- const response = await fetch(url, {
|
|
|
|
|
|
|
+ const response = await fetch(url, { ...init, signal: AbortSignal.timeout(timeoutMs) });
|
|
|
|
|
+ return { status: response.status, error: null, ms: Date.now() - startedAt, timedOut: false };
|
|
|
|
|
+ } catch (error) {
|
|
|
|
|
+ const { message, timedOut } = describeFetchError(error);
|
|
|
|
|
+ return { status: null, error: message, ms: Date.now() - startedAt, timedOut };
|
|
|
|
|
+ }
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+async function postProbe(url: string, body: Record<string, unknown>, apiKey?: string | null): Promise<ProbeAttempt> {
|
|
|
|
|
+ return probeFetch(
|
|
|
|
|
+ url,
|
|
|
|
|
+ {
|
|
|
method: 'POST',
|
|
method: 'POST',
|
|
|
headers: {
|
|
headers: {
|
|
|
'content-type': 'application/json',
|
|
'content-type': 'application/json',
|
|
|
...(apiKey ? { authorization: `Bearer ${apiKey}` } : {}),
|
|
...(apiKey ? { authorization: `Bearer ${apiKey}` } : {}),
|
|
|
},
|
|
},
|
|
|
body: JSON.stringify(body),
|
|
body: JSON.stringify(body),
|
|
|
- signal: AbortSignal.timeout(8_000),
|
|
|
|
|
- });
|
|
|
|
|
- return { status: response.status, error: null };
|
|
|
|
|
- } catch (error) {
|
|
|
|
|
- return { status: null, error: error instanceof Error ? error.message : String(error) };
|
|
|
|
|
- }
|
|
|
|
|
-}
|
|
|
|
|
-
|
|
|
|
|
-/** Ollama 的 /api/version 挂在 API 根之外(去掉 /v1 尾段);非 Ollama 端点返回 null */
|
|
|
|
|
-async function probeOllamaVersion(baseUrl: string): Promise<string | null> {
|
|
|
|
|
- const origin = baseUrl.replace(/\/v\d+$/u, '');
|
|
|
|
|
- try {
|
|
|
|
|
- const response = await fetch(`${origin}/api/version`, { signal: AbortSignal.timeout(5_000) });
|
|
|
|
|
- if (response.status !== 200) return null;
|
|
|
|
|
- const parsed = (await response.json()) as { version?: unknown };
|
|
|
|
|
- return typeof parsed.version === 'string' && parsed.version ? parsed.version : null;
|
|
|
|
|
- } catch {
|
|
|
|
|
- return null;
|
|
|
|
|
- }
|
|
|
|
|
-}
|
|
|
|
|
-
|
|
|
|
|
-function parseVersion(raw: string): number[] | null {
|
|
|
|
|
- const match = raw.trim().match(/^(\d+)\.(\d+)(?:\.(\d+))?/u);
|
|
|
|
|
- if (!match) return null;
|
|
|
|
|
- return [Number(match[1]), Number(match[2]), Number(match[3] ?? 0)];
|
|
|
|
|
|
|
+ },
|
|
|
|
|
+ PROBE_TIMEOUT_MS,
|
|
|
|
|
+ );
|
|
|
}
|
|
}
|
|
|
|
|
|
|
|
-function versionLt(a: number[], b: readonly number[]): boolean {
|
|
|
|
|
- for (let index = 0; index < 3; index += 1) {
|
|
|
|
|
- if (a[index] !== b[index]) return a[index] < b[index];
|
|
|
|
|
|
|
+/**
|
|
|
|
|
+ * 不可达的提示必须带上是哪个地址、等了多久——只说「端点不可达」时,
|
|
|
|
|
+ * 用户既不知道配错在哪,也无从判断是没起服务还是被防火墙慢慢吞掉。
|
|
|
|
|
+ */
|
|
|
|
|
+function unreachableDetail(url: string, attempt: ProbeAttempt): string {
|
|
|
|
|
+ if (attempt.timedOut) {
|
|
|
|
|
+ return `端点不可达:${url} 在 ${(attempt.ms / 1000).toFixed(1)} 秒内没有响应,请确认服务已启动、地址与端口正确`;
|
|
|
}
|
|
}
|
|
|
- return false;
|
|
|
|
|
|
|
+ return `端点不可达:${url} —— ${attempt.error ?? '网络错误'}`;
|
|
|
}
|
|
}
|
|
|
|
|
|
|
|
function probeResult(partial: Partial<EndpointProbeResult> & { protocol: EndpointProtocol; detail: string }): EndpointProbeResult {
|
|
function probeResult(partial: Partial<EndpointProbeResult> & { protocol: EndpointProtocol; detail: string }): EndpointProbeResult {
|
|
|
- return {
|
|
|
|
|
- responsesStatus: null,
|
|
|
|
|
- chatStatus: null,
|
|
|
|
|
- ollamaVersion: null,
|
|
|
|
|
- ollamaNeedsUpgrade: false,
|
|
|
|
|
- ...partial,
|
|
|
|
|
- };
|
|
|
|
|
|
|
+ return { chatStatus: null, ...partial };
|
|
|
}
|
|
}
|
|
|
|
|
|
|
|
/**
|
|
/**
|
|
|
- * 探测端点协议能力(四态判定)。
|
|
|
|
|
|
|
+ * 判定端点能不能用 Chat 协议——后端「模型管理」里的记录都是 Chat 端点,
|
|
|
|
|
+ * 有的就交给内置桥接。
|
|
|
*
|
|
*
|
|
|
- * Codex 0.155+ 只会发 Responses API:responses → 直通;chat-only → 内置桥接转换;
|
|
|
|
|
- * 其余两种拒绝。判定顺序:
|
|
|
|
|
- * 1. POST /responses(用**真实模型名**——Ollama 对未拉取的模型返回 404,不能直接判端点不支持);
|
|
|
|
|
- * 2. 404/405 且是 Ollama → 查 /api/version 消歧:≥0.13.3 说明 404 只是模型没拉(判 responses),
|
|
|
|
|
- * 低于 0.13.3 直接判 chat-only(/chat/completions 必有);
|
|
|
|
|
- * 3. 否则再 POST /chat/completions:路由存在判 chat-only,也不存在判 unsupported。
|
|
|
|
|
|
|
+ * 只看 /chat/completions 这条路由在不在:请求体故意不带 messages,
|
|
|
|
|
+ * 端点会在校验阶段就回 400(实测 15ms),不会真的开始生成。
|
|
|
|
|
+ * 代价是模型名写错要到真正提问时才暴露。
|
|
|
*/
|
|
*/
|
|
|
export async function probeEndpoint(params: {
|
|
export async function probeEndpoint(params: {
|
|
|
baseUrl: string;
|
|
baseUrl: string;
|
|
|
modelId?: string | null;
|
|
modelId?: string | null;
|
|
|
apiKey?: string | null;
|
|
apiKey?: string | null;
|
|
|
- providerType?: string | null;
|
|
|
|
|
}): Promise<EndpointProbeResult> {
|
|
}): Promise<EndpointProbeResult> {
|
|
|
const baseUrl = params.baseUrl.replace(/\/+$/u, '');
|
|
const baseUrl = params.baseUrl.replace(/\/+$/u, '');
|
|
|
|
|
+ const chatUrl = `${baseUrl}/chat/completions`;
|
|
|
const model = params.modelId?.trim() || 'probe';
|
|
const model = params.modelId?.trim() || 'probe';
|
|
|
- const isOllama = `${params.providerType ?? ''}`.trim().toUpperCase() === 'OLLAMA';
|
|
|
|
|
|
|
|
|
|
- const responses = await postProbe(
|
|
|
|
|
- `${baseUrl}/responses`,
|
|
|
|
|
- { model, input: 'ping', max_output_tokens: 16, stream: false },
|
|
|
|
|
- params.apiKey,
|
|
|
|
|
- );
|
|
|
|
|
- if (responses.status === null) {
|
|
|
|
|
- return probeResult({ protocol: 'unreachable', detail: `端点不可达:${responses.error ?? '网络错误'}` });
|
|
|
|
|
- }
|
|
|
|
|
- if (responses.status !== 404 && responses.status !== 405) {
|
|
|
|
|
- return probeResult({
|
|
|
|
|
- protocol: 'responses',
|
|
|
|
|
- responsesStatus: responses.status,
|
|
|
|
|
- detail: `端点已实现 Responses API(HTTP ${responses.status})`,
|
|
|
|
|
- });
|
|
|
|
|
- }
|
|
|
|
|
-
|
|
|
|
|
- if (isOllama) {
|
|
|
|
|
- const version = await probeOllamaVersion(baseUrl);
|
|
|
|
|
- if (version) {
|
|
|
|
|
- const parsed = parseVersion(version);
|
|
|
|
|
- if (parsed && !versionLt(parsed, OLLAMA_RESPONSES_MIN_VERSION)) {
|
|
|
|
|
- return probeResult({
|
|
|
|
|
- protocol: 'responses',
|
|
|
|
|
- responsesStatus: responses.status,
|
|
|
|
|
- ollamaVersion: version,
|
|
|
|
|
- detail: `Ollama ${version} 支持 Responses API;/responses 返回 404 通常只是模型「${model}」未拉取,请先 ollama pull ${model}`,
|
|
|
|
|
- });
|
|
|
|
|
- }
|
|
|
|
|
- return probeResult({
|
|
|
|
|
- protocol: 'chat-only',
|
|
|
|
|
- responsesStatus: responses.status,
|
|
|
|
|
- ollamaVersion: version,
|
|
|
|
|
- ollamaNeedsUpgrade: true,
|
|
|
|
|
- detail: `Ollama ${version} 低于 0.13.3,只提供 /chat/completions(建议升级 Ollama 以直连)`,
|
|
|
|
|
- });
|
|
|
|
|
- }
|
|
|
|
|
|
|
+ const chat = await postProbe(chatUrl, { model }, params.apiKey);
|
|
|
|
|
+ if (chat.status === null) {
|
|
|
|
|
+ return probeResult({ protocol: 'unreachable', detail: unreachableDetail(chatUrl, chat) });
|
|
|
}
|
|
}
|
|
|
-
|
|
|
|
|
- const chat = await postProbe(
|
|
|
|
|
- `${baseUrl}/chat/completions`,
|
|
|
|
|
- { model, messages: [{ role: 'user', content: 'ping' }], max_tokens: 1, stream: false },
|
|
|
|
|
- params.apiKey,
|
|
|
|
|
- );
|
|
|
|
|
- if (chat.status !== null && chat.status !== 404 && chat.status !== 405) {
|
|
|
|
|
|
|
+ if (chat.status !== 404 && chat.status !== 405) {
|
|
|
return probeResult({
|
|
return probeResult({
|
|
|
- protocol: 'chat-only',
|
|
|
|
|
- responsesStatus: responses.status,
|
|
|
|
|
|
|
+ protocol: 'chat',
|
|
|
chatStatus: chat.status,
|
|
chatStatus: chat.status,
|
|
|
- detail: `端点只提供 /chat/completions(/responses 返回 HTTP ${responses.status})`,
|
|
|
|
|
|
|
+ detail: `端点支持 Chat 协议(HTTP ${chat.status}),应用时经内置桥接转换为 Responses`,
|
|
|
});
|
|
});
|
|
|
}
|
|
}
|
|
|
return probeResult({
|
|
return probeResult({
|
|
|
protocol: 'unsupported',
|
|
protocol: 'unsupported',
|
|
|
- responsesStatus: responses.status,
|
|
|
|
|
chatStatus: chat.status,
|
|
chatStatus: chat.status,
|
|
|
- detail: `端点 Responses 与 Chat Completions 均不存在(HTTP ${responses.status}/${chat.status ?? '网络错误'}),无法对接`,
|
|
|
|
|
|
|
+ detail: `端点没有 /chat/completions 路由(HTTP ${chat.status}),本客户端只对接 Chat 协议端点`,
|
|
|
});
|
|
});
|
|
|
}
|
|
}
|
|
|
|
|
|
|
@@ -280,35 +235,26 @@ export async function applyProvider(
|
|
|
input: ApplyProviderInput & { modelRecordId?: string | number | null },
|
|
input: ApplyProviderInput & { modelRecordId?: string | number | null },
|
|
|
): Promise<AppliedProvider> {
|
|
): Promise<AppliedProvider> {
|
|
|
const spec = toProviderSpec(input);
|
|
const spec = toProviderSpec(input);
|
|
|
- /** 桥接时落盘存上游真实地址(展示用);桥接 URL 每次应用临时分配,不落盘 */
|
|
|
|
|
|
|
+ /** 落盘存上游真实地址(展示用);桥接 URL 每次应用临时分配,不落盘 */
|
|
|
const displayBaseUrl = spec.baseUrl;
|
|
const displayBaseUrl = spec.baseUrl;
|
|
|
- let bridged = false;
|
|
|
|
|
|
|
|
|
|
// 一定要先探端点:应用成功后才发现连不上,比这里直接报错更难排查
|
|
// 一定要先探端点:应用成功后才发现连不上,比这里直接报错更难排查
|
|
|
const probe = await probeEndpoint({
|
|
const probe = await probeEndpoint({
|
|
|
baseUrl: spec.baseUrl,
|
|
baseUrl: spec.baseUrl,
|
|
|
modelId: spec.model,
|
|
modelId: spec.model,
|
|
|
apiKey: spec.apiKey,
|
|
apiKey: spec.apiKey,
|
|
|
- providerType: input.providerType,
|
|
|
|
|
});
|
|
});
|
|
|
- if (probe.protocol !== 'responses' && probe.protocol !== 'chat-only') {
|
|
|
|
|
- throw new Error(probe.detail);
|
|
|
|
|
- }
|
|
|
|
|
- if (probe.protocol === 'chat-only') {
|
|
|
|
|
- // chat-only 端点:启动内置桥接,Codex 改连本地代理的 /v1/responses
|
|
|
|
|
- const bridge = await chatBridge.start({ upstreamBaseUrl: spec.baseUrl, headers: spec.httpHeaders ?? null });
|
|
|
|
|
- spec.baseUrl = bridge.url;
|
|
|
|
|
- bridged = true;
|
|
|
|
|
- }
|
|
|
|
|
|
|
+ if (probe.protocol !== 'chat') throw new Error(probe.detail);
|
|
|
|
|
|
|
|
- // 直通:确保上一轮的桥接已停
|
|
|
|
|
- if (!bridged) await chatBridge.stop();
|
|
|
|
|
|
|
+ // Codex 只会发 Responses,所以 Chat 端点一律经内置桥接:Codex 改连本地代理的 /v1/responses
|
|
|
|
|
+ const bridge = await chatBridge.start({ upstreamBaseUrl: spec.baseUrl, headers: spec.httpHeaders ?? null });
|
|
|
|
|
+ spec.baseUrl = bridge.url;
|
|
|
|
|
|
|
|
try {
|
|
try {
|
|
|
await runtime.applyProvider(spec);
|
|
await runtime.applyProvider(spec);
|
|
|
} catch (error) {
|
|
} catch (error) {
|
|
|
// 应用失败回滚桥接,避免留下一个指着旧上游的孤儿代理
|
|
// 应用失败回滚桥接,避免留下一个指着旧上游的孤儿代理
|
|
|
- if (bridged) await chatBridge.stop();
|
|
|
|
|
|
|
+ await chatBridge.stop();
|
|
|
throw error;
|
|
throw error;
|
|
|
}
|
|
}
|
|
|
|
|
|
|
@@ -321,7 +267,7 @@ export async function applyProvider(
|
|
|
providerId: spec.id,
|
|
providerId: spec.id,
|
|
|
name: spec.name ?? null,
|
|
name: spec.name ?? null,
|
|
|
appliedAt: new Date().toISOString(),
|
|
appliedAt: new Date().toISOString(),
|
|
|
- ...(bridged ? { bridged: true } : {}),
|
|
|
|
|
|
|
+ bridged: true,
|
|
|
};
|
|
};
|
|
|
await writeAppliedProvider(applied);
|
|
await writeAppliedProvider(applied);
|
|
|
return applied;
|
|
return applied;
|