providerService.ts 15 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365
  1. import { mkdir, readFile, writeFile } from 'node:fs/promises';
  2. import { join } from 'node:path';
  3. import { getCodexDataDir } from './codexHome';
  4. import { DEFAULT_MODEL_INSTRUCTIONS } from './defaultModelInstructions';
  5. import { DEFAULT_API_KEY_ENV, PROVIDER_EXTRA_ALLOWLIST, type CodexProviderSpec, type CodexRuntime, type JsonValue } from './codexRuntime';
  6. import type { AppliedProviderInfo, ApplyProviderInput } from './types';
  7. /**
  8. * 把后端「模型管理」里的一条记录翻译成 Codex 的 model provider。
  9. * 全程不做任何 Codex/OpenAI 账号登录:靠 wire_api="responses" + requires_openai_auth=false + env_key。
  10. * apiKey 只存在于内存与子进程环境变量,落盘的 AppliedProvider 不含任何密钥。
  11. */
  12. export const PROVIDER_ID = 'zsjz';
  13. export type AppliedProvider = AppliedProviderInfo;
  14. /**
  15. * 端点协议判定。Codex 0.155.1 起 `wire_api="chat"` 被删掉(实测 0.146/0.151/0.155 三个版本
  16. * 都在加载 config.toml 时就报 "wire_api = \"chat\" is no longer supported"),客户端也不做
  17. * Responses↔Chat 翻译 —— 只会 Chat 的端点一律拒绝应用,把原因说清楚。
  18. */
  19. export type EndpointProtocol = 'responses' | 'chat-only' | 'unsupported' | 'unreachable';
  20. export interface EndpointProbeResult {
  21. protocol: EndpointProtocol;
  22. /** /responses 的 HTTP 状态;未探或网络失败为 null */
  23. responsesStatus: number | null;
  24. /** /chat/completions 的 HTTP 状态;只在需要区分 chat-only 时才探,未探为 null */
  25. chatStatus: number | null;
  26. detail: string;
  27. }
  28. const EXTRA_ALLOWLIST: readonly string[] = PROVIDER_EXTRA_ALLOWLIST;
  29. /**
  30. * 自部署端点多是单槽推理:实测一轮 prompt 4.6 万 token、prefill 只有 ~29 token/s,
  31. * 首字之前可以静默十几分钟。Codex 在流建立后每次轮询都套一个
  32. * `timeout(stream_idle_timeout_ms, stream.next())`(rust-v0.155.1
  33. * codex-api/src/sse/responses.rs:580-606),默认 300_000(model-provider-info/src/lib.rs:29),
  34. * 到期给出可重试的 CodexErr::Stream(protocol/src/error.rs:372-413),按 stream_max_retries
  35. * (默认 5)重发整轮 —— 把几分钟的活儿重复五遍,还挤掉前缀缓存。
  36. *
  37. * 所以这里:空闲窗口放宽到远大于最差 prefill;两个重试都关到 0。
  38. * 注意不要再写 `supports_websockets=false`:那是 serde 默认值,用 -c 声明的 provider 从来不走
  39. * WebSocket,那个 15 秒是 websocket_connect_timeout_ms,与此无关(此前一条错误注释把六次修改带偏)。
  40. * 模型管理 config 里显式写的值优先。
  41. */
  42. const PROVIDER_DEFAULTS: Readonly<Record<string, JsonValue>> = Object.freeze({
  43. request_max_retries: 0,
  44. stream_max_retries: 0,
  45. stream_idle_timeout_ms: 1_800_000,
  46. });
  47. function providerFile(): string {
  48. return join(getCodexDataDir(), 'provider.json');
  49. }
  50. /** Codex 要求 base_url 是 API 根(形如 https://api.openai.com/v1),它自己再拼路由 */
  51. export function normalizeBaseUrl(raw: string | null | undefined): string | null {
  52. const value = (raw ?? '').trim().replace(/\/+$/u, '');
  53. return value || null;
  54. }
  55. /** 给页面的非阻断提示;返回 null 表示没问题 */
  56. export function baseUrlHint(baseUrl: string | null): string | null {
  57. if (!baseUrl) return '未配置 base_url';
  58. if (!/^https?:\/\//u.test(baseUrl)) return 'base_url 必须以 http:// 或 https:// 开头';
  59. if (!/\/v\d+$/u.test(baseUrl)) {
  60. return 'base_url 通常应写到版本段(如 .../v1),否则拼出的 /responses 可能 404';
  61. }
  62. return null;
  63. }
  64. function parseJsonRecord(raw: unknown): Record<string, unknown> | null {
  65. if (!raw) return null;
  66. if (typeof raw === 'object' && !Array.isArray(raw)) return raw as Record<string, unknown>;
  67. if (typeof raw !== 'string') return null;
  68. try {
  69. const parsed = JSON.parse(raw) as unknown;
  70. return parsed && typeof parsed === 'object' && !Array.isArray(parsed)
  71. ? (parsed as Record<string, unknown>)
  72. : null;
  73. } catch {
  74. return null;
  75. }
  76. }
  77. function readStringMap(raw: unknown): Record<string, string> | null {
  78. const record = parseJsonRecord(raw);
  79. if (!record) return null;
  80. const result: Record<string, string> = {};
  81. for (const [key, value] of Object.entries(record)) {
  82. if (typeof value === 'string' && value) result[key] = value;
  83. }
  84. return Object.keys(result).length ? result : null;
  85. }
  86. /**
  87. * 后端「模型管理」的一条记录一律翻译成自定义 provider:Codex 0.155.1 把 ollama / lmstudio
  88. * 等内置 id 列为保留字,覆盖 model_providers.<内置 id> 会让 app-server 启动即退出。
  89. * 地址缺失就是配置缺失,直接报错让页面提示,不做 localhost 兜底。
  90. *
  91. * model 目录(model_catalog_json)在 applyProvider 里另行生成落盘后挂到 spec 上:
  92. * 自定义模型 slug 不在 codex 内置目录里,线程会落到兜底元数据并打出
  93. * "Model metadata for `xxx` not found" 警告(上下文窗口/截断策略全靠猜)。
  94. */
  95. export function toProviderSpec(input: ApplyProviderInput): CodexProviderSpec {
  96. const model = `${input.modelId ?? ''}`.trim();
  97. if (!model) throw new Error('缺少 modelId');
  98. const baseUrl = normalizeBaseUrl(input.baseUrl);
  99. if (!baseUrl) {
  100. throw new Error(`模型「${input.name || model}」没有配置 base_url,请到「模型管理」补全(形如 https://host:port/v1)`);
  101. }
  102. const config = parseJsonRecord(input.config);
  103. const extra: Record<string, JsonValue> = { ...PROVIDER_DEFAULTS };
  104. for (const [key, value] of Object.entries(config ?? {})) {
  105. if (EXTRA_ALLOWLIST.includes(key)) extra[key] = value as JsonValue;
  106. }
  107. return {
  108. id: PROVIDER_ID,
  109. name: input.name ?? model,
  110. baseUrl,
  111. model,
  112. apiKey: input.apiKey ?? null,
  113. envKey: DEFAULT_API_KEY_ENV,
  114. httpHeaders: readStringMap(input.headersJson),
  115. extra: Object.keys(extra).length ? extra : null,
  116. };
  117. }
  118. /** config 里被丢弃的非白名单键,用于页面提示 */
  119. export function describeDroppedKeys(input: ApplyProviderInput): string[] {
  120. const config = parseJsonRecord(input.config);
  121. return Object.keys(config ?? {}).filter((key) => !EXTRA_ALLOWLIST.includes(key));
  122. }
  123. /**
  124. * 探测只问「路由在不在」,绝不让端点真去生成:本地服务多是单槽推理,
  125. * 一次生成能占住整个 HTTP 服务几十秒(实测 max_tokens:1 也要 14 秒,期间连 /models 都不应答)。
  126. * 15 秒是留给排队的余量。
  127. */
  128. const PROBE_TIMEOUT_MS = 15_000;
  129. interface ProbeAttempt {
  130. status: number | null;
  131. error: string | null;
  132. ms: number;
  133. timedOut: boolean;
  134. }
  135. function describeFetchError(error: unknown): { message: string; timedOut: boolean } {
  136. const err = error instanceof Error ? error : new Error(String(error));
  137. return {
  138. message: err.message,
  139. timedOut: err.name === 'TimeoutError' || err.name === 'AbortError' || /timeout/iu.test(err.message),
  140. };
  141. }
  142. async function probeFetch(url: string, init: RequestInit, timeoutMs: number): Promise<ProbeAttempt> {
  143. const startedAt = Date.now();
  144. try {
  145. const response = await fetch(url, { ...init, signal: AbortSignal.timeout(timeoutMs) });
  146. return { status: response.status, error: null, ms: Date.now() - startedAt, timedOut: false };
  147. } catch (error) {
  148. const { message, timedOut } = describeFetchError(error);
  149. return { status: null, error: message, ms: Date.now() - startedAt, timedOut };
  150. }
  151. }
  152. async function postProbe(url: string, body: Record<string, unknown>, apiKey?: string | null): Promise<ProbeAttempt> {
  153. return probeFetch(
  154. url,
  155. {
  156. method: 'POST',
  157. headers: {
  158. 'content-type': 'application/json',
  159. ...(apiKey ? { authorization: `Bearer ${apiKey}` } : {}),
  160. },
  161. body: JSON.stringify(body),
  162. },
  163. PROBE_TIMEOUT_MS,
  164. );
  165. }
  166. /**
  167. * 不可达的提示必须带上是哪个地址、等了多久——只说「端点不可达」时,
  168. * 用户既不知道配错在哪,也无从判断是没起服务还是被防火墙慢慢吞掉。
  169. */
  170. function unreachableDetail(url: string, attempt: ProbeAttempt): string {
  171. if (attempt.timedOut) {
  172. return `端点不可达:${url} 在 ${(attempt.ms / 1000).toFixed(1)} 秒内没有响应,请确认服务已启动、地址与端口正确`;
  173. }
  174. return `端点不可达:${url} —— ${attempt.error ?? '网络错误'}`;
  175. }
  176. function probeResult(
  177. partial: Partial<Omit<EndpointProbeResult, 'protocol' | 'detail'>> & {
  178. protocol: EndpointProtocol;
  179. detail: string;
  180. },
  181. ): EndpointProbeResult {
  182. return { responsesStatus: null, chatStatus: null, ...partial };
  183. }
  184. /**
  185. * 判定端点会不会说 Responses。只问「路由在不在」,绝不让端点真去生成:本地服务多是单槽推理,
  186. * 一次生成能占住整个 HTTP 服务几十秒(实测 max_tokens:1 也要 14 秒,期间连 /models 都不应答)。
  187. * 请求体故意不带 input,端点会在校验阶段就回 400(实测 15–27ms)。
  188. * 代价是模型名写错要到真正提问时才暴露。
  189. */
  190. export async function probeEndpoint(params: {
  191. baseUrl: string;
  192. modelId?: string | null;
  193. apiKey?: string | null;
  194. }): Promise<EndpointProbeResult> {
  195. const baseUrl = params.baseUrl.replace(/\/+$/u, '');
  196. const responsesUrl = `${baseUrl}/responses`;
  197. const chatUrl = `${baseUrl}/chat/completions`;
  198. const model = params.modelId?.trim() || 'probe';
  199. const responses = await postProbe(responsesUrl, { model }, params.apiKey);
  200. if (responses.status === null) {
  201. return probeResult({ protocol: 'unreachable', detail: unreachableDetail(responsesUrl, responses) });
  202. }
  203. if (responses.status !== 404 && responses.status !== 405) {
  204. return probeResult({
  205. protocol: 'responses',
  206. responsesStatus: responses.status,
  207. detail: `端点支持 Responses 协议(HTTP ${responses.status}),Codex 直连 ${responsesUrl}`,
  208. });
  209. }
  210. // /responses 不在:区分「只会 Chat」和「两条路由都没有」,前者的话要说得能让人去修端点
  211. const chat = await postProbe(chatUrl, { model }, params.apiKey);
  212. if (chat.status !== null && chat.status !== 404 && chat.status !== 405) {
  213. return probeResult({
  214. protocol: 'chat-only',
  215. responsesStatus: responses.status,
  216. chatStatus: chat.status,
  217. detail:
  218. `端点没有 /responses(HTTP ${responses.status}),只有 /chat/completions。` +
  219. 'Codex 0.155.1 已删除 wire_api="chat"(实测 0.146/0.151/0.155 一致),本客户端不做桥接。' +
  220. `请把端点换成会回答 POST ${responsesUrl} 的服务(自证:curl -X POST ${responsesUrl} -d '{"model":"${model}"}')`,
  221. });
  222. }
  223. return probeResult({
  224. protocol: 'unsupported',
  225. responsesStatus: responses.status,
  226. chatStatus: chat.status,
  227. detail: `端点没有 /responses 路由(HTTP ${responses.status}),Codex 只能按 Responses 协议对接 ${responsesUrl}`,
  228. });
  229. }
  230. export async function readAppliedProvider(): Promise<AppliedProvider | null> {
  231. try {
  232. const raw = JSON.parse(await readFile(providerFile(), 'utf8')) as Partial<AppliedProvider>;
  233. return raw.model ? (raw as AppliedProvider) : null;
  234. } catch {
  235. return null;
  236. }
  237. }
  238. async function writeAppliedProvider(value: AppliedProvider | null): Promise<void> {
  239. await mkdir(getCodexDataDir(), { recursive: true });
  240. await writeFile(providerFile(), JSON.stringify(value ?? {}, null, 2), 'utf8');
  241. }
  242. /**
  243. * 生成单条模型的目录条目,字段与 codex rust-v0.155.1 的 ModelInfo serde 形状对齐
  244. * (实测 0.155.1 会校验:catalog 模型必须携带 base_instructions/instructions_template,
  245. * 未知字段会报错,所以只写确认过的键)。元数据取 codex 对未知模型的兜底值,
  246. * 这样除 used_fallback 归零(警告消失、特性门控放行)外,请求行为与之前完全一致。
  247. */
  248. function buildModelCatalog(spec: CodexProviderSpec): Record<string, unknown> {
  249. return {
  250. models: [
  251. {
  252. slug: spec.model,
  253. display_name: spec.name ?? spec.model,
  254. description: null,
  255. default_reasoning_level: null,
  256. supported_reasoning_levels: [],
  257. shell_type: 'shell_command',
  258. visibility: 'list',
  259. supported_in_api: true,
  260. priority: 1000,
  261. additional_speed_tiers: [],
  262. service_tiers: [],
  263. default_service_tier: null,
  264. availability_nux: null,
  265. upgrade: null,
  266. default_reasoning_summary: 'none',
  267. support_verbosity: false,
  268. default_verbosity: null,
  269. supports_image_detail_original: false,
  270. supports_parallel_tool_calls: false,
  271. supports_reasoning_summaries: true,
  272. supports_search_tool: false,
  273. truncation_policy: { mode: 'bytes', limit: 10_000 },
  274. context_window: 272_000,
  275. max_context_window: 272_000,
  276. effective_context_window_percent: 95,
  277. experimental_supported_tools: [],
  278. input_modalities: ['text'],
  279. model_messages: { instructions_template: DEFAULT_MODEL_INSTRUCTIONS },
  280. },
  281. ],
  282. };
  283. }
  284. function catalogFile(): string {
  285. return join(getCodexDataDir(), 'model-catalog.json');
  286. }
  287. /**
  288. * 把目录文件落盘并返回路径。目录整体替换 codex 的内置目录(进程级),
  289. * 单条 provider 对应单条模型;换模型会走 applyProvider 重启子进程,天然重建。
  290. */
  291. async function writeModelCatalog(spec: CodexProviderSpec): Promise<string> {
  292. const path = catalogFile();
  293. await mkdir(getCodexDataDir(), { recursive: true });
  294. await writeFile(path, JSON.stringify(buildModelCatalog(spec), null, 2), 'utf8');
  295. return path;
  296. }
  297. /** 应用一条模型配置:先探端点,再重启子进程(api_key 走 env,换 key 必须重启) */
  298. export async function applyProvider(
  299. runtime: CodexRuntime,
  300. input: ApplyProviderInput & { modelRecordId?: string | number | null },
  301. ): Promise<AppliedProvider> {
  302. const spec = toProviderSpec(input);
  303. // 一定要先探端点:应用成功后才发现连不上,比这里直接报错更难排查
  304. const probe = await probeEndpoint({
  305. baseUrl: spec.baseUrl,
  306. modelId: spec.model,
  307. apiKey: spec.apiKey,
  308. });
  309. if (probe.protocol !== 'responses') throw new Error(probe.detail);
  310. spec.modelCatalogPath = await writeModelCatalog(spec);
  311. await runtime.applyProvider(spec);
  312. const applied: AppliedProvider = {
  313. model: spec.model,
  314. modelRecordId: input.modelRecordId === undefined || input.modelRecordId === null
  315. ? null
  316. : String(input.modelRecordId),
  317. baseUrl: spec.baseUrl,
  318. providerId: spec.id,
  319. name: spec.name ?? null,
  320. appliedAt: new Date().toISOString(),
  321. };
  322. await writeAppliedProvider(applied);
  323. return applied;
  324. }
  325. export async function clearProvider(runtime: CodexRuntime): Promise<void> {
  326. await runtime.applyProvider(null);
  327. await writeAppliedProvider(null);
  328. }