providerService.ts 16 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396
  1. import { mkdir, readFile, writeFile } from 'node:fs/promises';
  2. import { dirname, join, resolve } from 'node:path';
  3. import { getCodexDataDir } from './codexHome';
  4. import { DEFAULT_MODEL_INSTRUCTIONS } from './defaultModelInstructions';
  5. import { chatRouter } from './chatRouter/routerService';
  6. import { DEFAULT_API_KEY_ENV, PROVIDER_EXTRA_ALLOWLIST, type CodexProviderSpec, type CodexRuntime, type JsonValue } from './codexRuntime';
  7. import type { AppliedProviderInfo, ApplyProviderInput } from './types';
  8. /**
  9. * 把后端「模型管理」里的一条记录翻译成 Codex 的 model provider。
  10. * 全程不做任何 Codex/OpenAI 账号登录:靠 wire_api="responses" + requires_openai_auth=false + env_key。
  11. * apiKey 只存在于内存与子进程环境变量,落盘的 AppliedProvider 不含任何密钥。
  12. * 端点只会 Chat 协议时走内置路由层(chatRouter)桥接,对 codex 仍呈现为 Responses 端点。
  13. */
  14. export const PROVIDER_ID = 'zsjz';
  15. export type AppliedProvider = AppliedProviderInfo;
  16. /**
  17. * 端点协议判定。Codex 0.155.1 起 `wire_api="chat"` 被删掉(实测 0.146/0.151/0.155 三个版本
  18. * 都在加载 config.toml 时就报 "wire_api = \"chat\" is no longer supported"),因此对 codex
  19. * 永远说 Responses;只会 Chat 的端点由内置路由层翻译后对接,应用时不再拒绝。
  20. */
  21. export type EndpointProtocol = 'responses' | 'chat-only' | 'unsupported' | 'unreachable';
  22. export interface EndpointProbeResult {
  23. protocol: EndpointProtocol;
  24. /** /responses 的 HTTP 状态;未探或网络失败为 null */
  25. responsesStatus: number | null;
  26. /** /chat/completions 的 HTTP 状态;只在需要区分 chat-only 时才探,未探为 null */
  27. chatStatus: number | null;
  28. detail: string;
  29. }
  30. const EXTRA_ALLOWLIST: readonly string[] = PROVIDER_EXTRA_ALLOWLIST;
  31. /**
  32. * 自部署端点多是单槽推理:实测一轮 prompt 4.6 万 token、prefill 只有 ~29 token/s,
  33. * 首字之前可以静默十几分钟。Codex 在流建立后每次轮询都套一个
  34. * `timeout(stream_idle_timeout_ms, stream.next())`(rust-v0.155.1
  35. * codex-api/src/sse/responses.rs:580-606),默认 300_000(model-provider-info/src/lib.rs:29),
  36. * 到期给出可重试的 CodexErr::Stream(protocol/src/error.rs:372-413),按 stream_max_retries
  37. * (默认 5)重发整轮 —— 把几分钟的活儿重复五遍,还挤掉前缀缓存。
  38. *
  39. * 所以这里:空闲窗口放宽到远大于最差 prefill;两个重试都关到 0。
  40. * 注意不要再写 `supports_websockets=false`:那是 serde 默认值,用 -c 声明的 provider 从来不走
  41. * WebSocket,那个 15 秒是 websocket_connect_timeout_ms,与此无关(此前一条错误注释把六次修改带偏)。
  42. * 模型管理 config 里显式写的值优先。
  43. */
  44. const PROVIDER_DEFAULTS: Readonly<Record<string, JsonValue>> = Object.freeze({
  45. request_max_retries: 0,
  46. stream_max_retries: 0,
  47. stream_idle_timeout_ms: 1_800_000,
  48. });
  49. function providerFile(): string {
  50. return join(getCodexDataDir(), 'provider.json');
  51. }
  52. /** Codex 要求 base_url 是 API 根(形如 https://api.openai.com/v1),它自己再拼路由 */
  53. export function normalizeBaseUrl(raw: string | null | undefined): string | null {
  54. const value = (raw ?? '').trim().replace(/\/+$/u, '');
  55. return value || null;
  56. }
  57. /** 给页面的非阻断提示;返回 null 表示没问题 */
  58. export function baseUrlHint(baseUrl: string | null): string | null {
  59. if (!baseUrl) return '未配置 base_url';
  60. if (!/^https?:\/\//u.test(baseUrl)) return 'base_url 必须以 http:// 或 https:// 开头';
  61. if (!/\/v\d+$/u.test(baseUrl)) {
  62. return 'base_url 通常应写到版本段(如 .../v1),否则拼出的 /responses 可能 404';
  63. }
  64. return null;
  65. }
  66. function parseJsonRecord(raw: unknown): Record<string, unknown> | null {
  67. if (!raw) return null;
  68. if (typeof raw === 'object' && !Array.isArray(raw)) return raw as Record<string, unknown>;
  69. if (typeof raw !== 'string') return null;
  70. try {
  71. const parsed = JSON.parse(raw) as unknown;
  72. return parsed && typeof parsed === 'object' && !Array.isArray(parsed)
  73. ? (parsed as Record<string, unknown>)
  74. : null;
  75. } catch {
  76. return null;
  77. }
  78. }
  79. function readStringMap(raw: unknown): Record<string, string> | null {
  80. const record = parseJsonRecord(raw);
  81. if (!record) return null;
  82. const result: Record<string, string> = {};
  83. for (const [key, value] of Object.entries(record)) {
  84. if (typeof value === 'string' && value) result[key] = value;
  85. }
  86. return Object.keys(result).length ? result : null;
  87. }
  88. /**
  89. * 后端「模型管理」的一条记录一律翻译成自定义 provider:Codex 0.155.1 把 ollama / lmstudio
  90. * 等内置 id 列为保留字,覆盖 model_providers.<内置 id> 会让 app-server 启动即退出。
  91. * 地址缺失就是配置缺失,直接报错让页面提示,不做 localhost 兜底。
  92. *
  93. * model 目录(model_catalog_json)在 applyProvider 里另行生成落盘后挂到 spec 上:
  94. * 自定义模型 slug 不在 codex 内置目录里,线程会落到兜底元数据并打出
  95. * "Model metadata for `xxx` not found" 警告(上下文窗口/截断策略全靠猜)。
  96. */
  97. export function toProviderSpec(input: ApplyProviderInput): CodexProviderSpec {
  98. const model = `${input.modelId ?? ''}`.trim();
  99. if (!model) throw new Error('缺少 modelId');
  100. const baseUrl = normalizeBaseUrl(input.baseUrl);
  101. if (!baseUrl) {
  102. throw new Error(`模型「${input.name || model}」没有配置 base_url,请到「模型管理」补全(形如 https://host:port/v1)`);
  103. }
  104. const config = parseJsonRecord(input.config);
  105. const extra: Record<string, JsonValue> = { ...PROVIDER_DEFAULTS };
  106. for (const [key, value] of Object.entries(config ?? {})) {
  107. if (EXTRA_ALLOWLIST.includes(key)) extra[key] = value as JsonValue;
  108. }
  109. return {
  110. id: PROVIDER_ID,
  111. name: input.name ?? model,
  112. baseUrl,
  113. model,
  114. apiKey: input.apiKey ?? null,
  115. envKey: DEFAULT_API_KEY_ENV,
  116. httpHeaders: readStringMap(input.headersJson),
  117. extra: Object.keys(extra).length ? extra : null,
  118. };
  119. }
  120. /** config 里被丢弃的非白名单键,用于页面提示 */
  121. export function describeDroppedKeys(input: ApplyProviderInput): string[] {
  122. const config = parseJsonRecord(input.config);
  123. return Object.keys(config ?? {}).filter((key) => !EXTRA_ALLOWLIST.includes(key));
  124. }
  125. /**
  126. * 探测只问「路由在不在」,绝不让端点真去生成:本地服务多是单槽推理,
  127. * 一次生成能占住整个 HTTP 服务几十秒(实测 max_tokens:1 也要 14 秒,期间连 /models 都不应答)。
  128. * 15 秒是留给排队的余量。
  129. */
  130. const PROBE_TIMEOUT_MS = 15_000;
  131. interface ProbeAttempt {
  132. status: number | null;
  133. error: string | null;
  134. ms: number;
  135. timedOut: boolean;
  136. }
  137. function describeFetchError(error: unknown): { message: string; timedOut: boolean } {
  138. const err = error instanceof Error ? error : new Error(String(error));
  139. return {
  140. message: err.message,
  141. timedOut: err.name === 'TimeoutError' || err.name === 'AbortError' || /timeout/iu.test(err.message),
  142. };
  143. }
  144. async function probeFetch(url: string, init: RequestInit, timeoutMs: number): Promise<ProbeAttempt> {
  145. const startedAt = Date.now();
  146. try {
  147. const response = await fetch(url, { ...init, signal: AbortSignal.timeout(timeoutMs) });
  148. return { status: response.status, error: null, ms: Date.now() - startedAt, timedOut: false };
  149. } catch (error) {
  150. const { message, timedOut } = describeFetchError(error);
  151. return { status: null, error: message, ms: Date.now() - startedAt, timedOut };
  152. }
  153. }
  154. async function postProbe(url: string, body: Record<string, unknown>, apiKey?: string | null): Promise<ProbeAttempt> {
  155. return probeFetch(
  156. url,
  157. {
  158. method: 'POST',
  159. headers: {
  160. 'content-type': 'application/json',
  161. ...(apiKey ? { authorization: `Bearer ${apiKey}` } : {}),
  162. },
  163. body: JSON.stringify(body),
  164. },
  165. PROBE_TIMEOUT_MS,
  166. );
  167. }
  168. /**
  169. * 不可达的提示必须带上是哪个地址、等了多久——只说「端点不可达」时,
  170. * 用户既不知道配错在哪,也无从判断是没起服务还是被防火墙慢慢吞掉。
  171. */
  172. function unreachableDetail(url: string, attempt: ProbeAttempt): string {
  173. if (attempt.timedOut) {
  174. return `端点不可达:${url} 在 ${(attempt.ms / 1000).toFixed(1)} 秒内没有响应,请确认服务已启动、地址与端口正确`;
  175. }
  176. return `端点不可达:${url} —— ${attempt.error ?? '网络错误'}`;
  177. }
  178. function probeResult(
  179. partial: Partial<Omit<EndpointProbeResult, 'protocol' | 'detail'>> & {
  180. protocol: EndpointProtocol;
  181. detail: string;
  182. },
  183. ): EndpointProbeResult {
  184. return { responsesStatus: null, chatStatus: null, ...partial };
  185. }
  186. /**
  187. * 判定端点会不会说 Responses。只问「路由在不在」,绝不让端点真去生成:本地服务多是单槽推理,
  188. * 一次生成能占住整个 HTTP 服务几十秒(实测 max_tokens:1 也要 14 秒,期间连 /models 都不应答)。
  189. * 请求体故意不带 input,端点会在校验阶段就回 400(实测 15–27ms)。
  190. * 代价是模型名写错要到真正提问时才暴露。
  191. */
  192. export async function probeEndpoint(params: {
  193. baseUrl: string;
  194. modelId?: string | null;
  195. apiKey?: string | null;
  196. }): Promise<EndpointProbeResult> {
  197. const baseUrl = params.baseUrl.replace(/\/+$/u, '');
  198. const responsesUrl = `${baseUrl}/responses`;
  199. const chatUrl = `${baseUrl}/chat/completions`;
  200. const model = params.modelId?.trim() || 'probe';
  201. const responses = await postProbe(responsesUrl, { model }, params.apiKey);
  202. if (responses.status === null) {
  203. return probeResult({ protocol: 'unreachable', detail: unreachableDetail(responsesUrl, responses) });
  204. }
  205. if (responses.status !== 404 && responses.status !== 405) {
  206. return probeResult({
  207. protocol: 'responses',
  208. responsesStatus: responses.status,
  209. detail: `端点支持 Responses 协议(HTTP ${responses.status}),Codex 直连 ${responsesUrl}`,
  210. });
  211. }
  212. // /responses 不在:区分「只会 Chat」和「两条路由都没有」。前者由内置路由层桥接:
  213. // codex 仍说 Responses(0.155.1 已删 wire_api="chat"),路由层翻成 Chat 转给上游
  214. const chat = await postProbe(chatUrl, { model }, params.apiKey);
  215. if (chat.status !== null && chat.status !== 404 && chat.status !== 405) {
  216. return probeResult({
  217. protocol: 'chat-only',
  218. responsesStatus: responses.status,
  219. chatStatus: chat.status,
  220. detail:
  221. `端点没有 /responses(HTTP ${responses.status}),只有 /chat/completions(HTTP ${chat.status})。` +
  222. `应用后 Codex 将经内置路由层对接:${chatUrl}`,
  223. });
  224. }
  225. return probeResult({
  226. protocol: 'unsupported',
  227. responsesStatus: responses.status,
  228. chatStatus: chat.status,
  229. detail: `端点没有 /responses 路由(HTTP ${responses.status}),Codex 只能按 Responses 协议对接 ${responsesUrl}`,
  230. });
  231. }
  232. export async function readAppliedProvider(): Promise<AppliedProvider | null> {
  233. try {
  234. const raw = JSON.parse(await readFile(providerFile(), 'utf8')) as Partial<AppliedProvider>;
  235. return raw.model ? (raw as AppliedProvider) : null;
  236. } catch {
  237. return null;
  238. }
  239. }
  240. async function writeAppliedProvider(value: AppliedProvider | null): Promise<void> {
  241. await mkdir(getCodexDataDir(), { recursive: true });
  242. await writeFile(providerFile(), JSON.stringify(value ?? {}, null, 2), 'utf8');
  243. }
  244. /**
  245. * 生成单条模型的目录条目,字段与 codex rust-v0.155.1 的 ModelInfo serde 形状对齐
  246. * (实测 0.155.1 会校验:catalog 模型必须携带 base_instructions/instructions_template,
  247. * 未知字段会报错,所以只写确认过的键)。元数据取 codex 对未知模型的兜底值,
  248. * 这样除 used_fallback 归零(警告消失、特性门控放行)外,请求行为与之前完全一致。
  249. */
  250. function buildModelCatalog(spec: CodexProviderSpec): Record<string, unknown> {
  251. return {
  252. models: [
  253. {
  254. slug: spec.model,
  255. display_name: spec.name ?? spec.model,
  256. description: null,
  257. default_reasoning_level: null,
  258. supported_reasoning_levels: [],
  259. shell_type: 'shell_command',
  260. visibility: 'list',
  261. supported_in_api: true,
  262. priority: 1000,
  263. additional_speed_tiers: [],
  264. service_tiers: [],
  265. default_service_tier: null,
  266. availability_nux: null,
  267. upgrade: null,
  268. default_reasoning_summary: 'none',
  269. support_verbosity: false,
  270. default_verbosity: null,
  271. supports_image_detail_original: false,
  272. supports_parallel_tool_calls: false,
  273. supports_reasoning_summaries: true,
  274. supports_search_tool: false,
  275. truncation_policy: { mode: 'bytes', limit: 10_000 },
  276. context_window: 272_000,
  277. max_context_window: 272_000,
  278. effective_context_window_percent: 95,
  279. experimental_supported_tools: [],
  280. input_modalities: ['text'],
  281. model_messages: { instructions_template: DEFAULT_MODEL_INSTRUCTIONS },
  282. },
  283. ],
  284. };
  285. }
  286. function catalogFile(): string {
  287. // codex 的 model_catalog_json 要求是绝对路径(AbsolutePathBuf),ee-core 的数据目录
  288. // 在某些环境(vitest/未初始化 app)下会给出相对路径,这里统一 resolve 掉
  289. return resolve(getCodexDataDir(), 'model-catalog.json');
  290. }
  291. /**
  292. * 把目录文件落盘并返回路径。目录整体替换 codex 的内置目录(进程级),
  293. * 单条 provider 对应单条模型;换模型会走 applyProvider 重启子进程,天然重建。
  294. */
  295. async function writeModelCatalog(spec: CodexProviderSpec): Promise<string> {
  296. const path = catalogFile();
  297. await mkdir(dirname(path), { recursive: true });
  298. await writeFile(path, JSON.stringify(buildModelCatalog(spec), null, 2), 'utf8');
  299. return path;
  300. }
  301. /**
  302. * 应用一条模型配置:先探端点,再重启子进程(api_key 走 env,换 key 必须重启)。
  303. * 端点会 Responses → 直连;只会 Chat → 起内置路由层,codex 的 base_url 指向
  304. * 本地路由(真实上游地址只进路由层与落盘快照,绝不进 codex 参数)。
  305. */
  306. export async function applyProvider(
  307. runtime: CodexRuntime,
  308. input: ApplyProviderInput & { modelRecordId?: string | number | null },
  309. ): Promise<AppliedProvider> {
  310. const spec = toProviderSpec(input);
  311. // 一定要先探端点:应用成功后才发现连不上,比这里直接报错更难排查
  312. const probe = await probeEndpoint({
  313. baseUrl: spec.baseUrl,
  314. modelId: spec.model,
  315. apiKey: spec.apiKey,
  316. });
  317. if (probe.protocol !== 'responses' && probe.protocol !== 'chat-only') throw new Error(probe.detail);
  318. const upstreamBaseUrl = spec.baseUrl;
  319. if (probe.protocol === 'chat-only') {
  320. const router = await chatRouter.start({
  321. upstreamBaseUrl: spec.baseUrl,
  322. // 自定义头(headersJson)已随 spec.httpHeaders 进 codex 参数,codex 每次请求都会带上
  323. // 并到达路由层,这里只需给出白名单名单
  324. forwardHeaderNames: Object.keys(spec.httpHeaders ?? {}),
  325. });
  326. // codex 拿到的 base_url 是本地路由;真实上游地址保留在 applied 落盘里
  327. spec.baseUrl = router.url;
  328. } else {
  329. // 从桥接模型切回直连模型:旧路由不再需要,停掉以免状态面板报陈旧路由
  330. await chatRouter.stop();
  331. }
  332. spec.modelCatalogPath = await writeModelCatalog(spec);
  333. try {
  334. await runtime.applyProvider(spec);
  335. } catch (error) {
  336. if (probe.protocol === 'chat-only') await chatRouter.stop();
  337. throw error;
  338. }
  339. const applied: AppliedProvider = {
  340. model: spec.model,
  341. modelRecordId: input.modelRecordId === undefined || input.modelRecordId === null
  342. ? null
  343. : String(input.modelRecordId),
  344. // 落盘的是真实上游地址(与模型管理记录一致),路由地址不落盘
  345. baseUrl: upstreamBaseUrl,
  346. providerId: spec.id,
  347. name: spec.name ?? null,
  348. appliedAt: new Date().toISOString(),
  349. bridged: probe.protocol === 'chat-only',
  350. };
  351. await writeAppliedProvider(applied);
  352. return applied;
  353. }
  354. export async function clearProvider(runtime: CodexRuntime): Promise<void> {
  355. await runtime.applyProvider(null);
  356. await chatRouter.stop();
  357. await writeAppliedProvider(null);
  358. }