cc 2 viikkoa sitten
vanhempi
commit
26ecd3aaf6

+ 40 - 0
.workbuddy-ai/memory/2026-09-20.md

@@ -60,3 +60,43 @@ SSE 首帧下发的是 `data:/mcp/message?sessionId=…`(**相对路径,不
 `noSchemaLeaksInjectionSurface`、`numericSpecFieldsBecomeJsonNumbers`。
 另 4 个测试类(Call/GraphRender/Sql/Track)全绿。
 
+---
+
+## 新增:直连大模型 service(`com.zsjz.ai.module.agent.llm`)
+
+**需求**:封装一个不经过 Agent、直接调大模型拿结果的 service。
+
+**产出**:
+| 文件 | 职责 |
+|---|---|
+| `LlmService` | `chat` / `chatDetail` / `chatAs` / `stream`,核心是把 `Model#stream` 的分片拼成文本 |
+| `LlmRequest` | `@Builder(toBuilder=true)`:modelId / system / user / messages / temperature / maxTokens / timeout(默认 60s) |
+| `LlmResult` | record:text / modelId / modelName / inputTokens / outputTokens / elapsedMs / totalTokens() |
+
+设计要点:
+- 模型配置复用 **`agent_model` 表**(⚠️ 表名是 `agent_model` 不是 `model`),走 `AgentModelFactory.create()`,
+  与 Agent 链路同一套连接参数;不传 modelId 时取默认对话模型。
+- 失败语义**一律抛 `ServerException`(400/404/500/504),绝不返回 null** —— 与
+  `IntentService`/`FollowupService` 的 fail-open 刻意相反(这是调用方主动要结果,不是辅助链路)。
+- `chatAs` 不用 AgentScope 的 `getStructuredData`(那要 ReActAgent 的 `generate_response` 工具),
+  改为「提示词注入 schema + 宽松解析」(剥 ``` 围栏、截最外层 JSON)。
+
+**踩到并修掉的坑**:`Flux.blockLast(Duration)` 把流内**任何**错误都包成
+`IllegalStateException("Timeout on blocking read...")` ⇒ 模型 401 被误报成 504。
+改为 `.timeout(...)` + `.onErrorMap(TimeoutException.class, …)`。
+
+**验证**(19 例 mock 单测 + 6 例真实 HTTP 测试,全绿):
+本机 **Ollama 未启动**(11434 无监听)、且 shell 有 `http_proxy=127.0.0.1:58859`,
+默认模型 `Ollama:qwen` 连不上。于是新增
+`ai-server/src/test/resources/mock-openai-server.py`(HTTP/1.1 chunked 手写 SSE 的假 OpenAI 端点)
++ `LlmServiceHttpTest`(**不启 Spring 上下文**,秒级)真实跑通
+「DB 配置 → AgentModelFactory → HTTP SSE → 分片拼接 → usage 提取 → 错误码映射」。
+真厂商端点的 `LlmServiceLiveTest` 因本机无可用模型**未验证**。
+
+**顺带发现(未修)**:`com.zsjz.ai.module.plat.mapper.PhoneIspMapper` 是该包下唯一漏 `@Mapper`
+注解的接口,本项目不用 `@MapperScan` ⇒ `GlobalCache#initIspData()` 启动时抛
+`NoSuchBeanDefinitionException`,被 `AppLoadEndEventListener` try-catch 吞掉(日志「初始化系统文件失败」),
+**代价是手机号运营商映射从未加载**。测试里用 `@MockitoBean` 绕过。
+
+**环境提示**:探测本地端口必须 `curl --noproxy '*'`,否则 `http_proxy` 会把请求转给代理并返回 502。
+

+ 12 - 2
ai-server/pom.xml

@@ -27,6 +27,18 @@
         <agentscope.version>2.0.1</agentscope.version>
         <pgvector.version>0.1.6</pgvector.version>
     </properties>
+    <dependencyManagement>
+        <dependencies>
+            <dependency>
+                <groupId>org.springframework.ai</groupId>
+                <artifactId>spring-ai-bom</artifactId>
+                <version>1.1.8</version>
+                <type>pom</type>
+                <scope>import</scope>
+            </dependency>
+        </dependencies>
+
+    </dependencyManagement>
     <dependencies>
         <dependency>
             <groupId>org.springframework.boot</groupId>
@@ -106,13 +118,11 @@
         <dependency>
             <groupId>org.springframework.ai</groupId>
             <artifactId>spring-ai-starter-mcp-server-webmvc</artifactId>
-            <version>1.1.8</version>
         </dependency>
 
         <dependency>
             <groupId>org.springframework.ai</groupId>
             <artifactId>spring-ai-starter-mcp-server</artifactId>
-            <version>1.1.8</version>
         </dependency>
 
         <dependency>

+ 72 - 0
ai-server/src/main/java/com/zsjz/ai/module/agent/llm/LlmRequest.java

@@ -0,0 +1,72 @@
+package com.zsjz.ai.module.agent.llm;
+
+import io.agentscope.core.message.Msg;
+import lombok.Builder;
+import lombok.Getter;
+
+import java.time.Duration;
+import java.util.List;
+
+/**
+ * 直连大模型的请求参数({@link LlmService} 入参)。
+ *
+ * <p>最简用法只需要 {@code user}:
+ * <pre>{@code
+ * LlmRequest.builder().user("把这段通话记录总结成三句话").build()
+ * }</pre>
+ *
+ * <p>要完整多轮上下文就传 {@link #messages}(此时 {@link #system} / {@link #user} 被忽略)。
+ */
+@Getter
+@Builder(toBuilder = true)
+public class LlmRequest {
+
+    /**
+     * 默认单次调用超时
+     */
+    public static final Duration DEFAULT_TIMEOUT = Duration.ofSeconds(60);
+
+    /**
+     * 模型 ID({@code model.id})。
+     *
+     * <p>为空时使用「默认对话模型」({@code type=llm} 且 {@code default_model=true} 且启用)。
+     */
+    private final Long modelId;
+
+    /**
+     * 系统提示词(可选)
+     */
+    private final String system;
+
+    /**
+     * 用户提示词;与 {@link #messages} 二选一
+     */
+    private final String user;
+
+    /**
+     * 完整消息列表(多轮);非空时忽略 {@link #system} / {@link #user}
+     */
+    private final List<Msg> messages;
+
+    /**
+     * 采样温度;为空时用厂商默认
+     */
+    private final Double temperature;
+
+    /**
+     * 最大输出 token;为空时用厂商默认
+     */
+    private final Integer maxTokens;
+
+    /**
+     * 单次调用超时;为空时取 {@link #DEFAULT_TIMEOUT}
+     */
+    private final Duration timeout;
+
+    /**
+     * 实际生效的超时时间
+     */
+    public Duration timeoutOrDefault() {
+        return timeout == null ? DEFAULT_TIMEOUT : timeout;
+    }
+}

+ 26 - 0
ai-server/src/main/java/com/zsjz/ai/module/agent/llm/LlmResult.java

@@ -0,0 +1,26 @@
+package com.zsjz.ai.module.agent.llm;
+
+/**
+ * 直连大模型的结果。
+ *
+ * <p>只关心正文时用 {@link LlmService#chat} 直接拿 {@code String};
+ * 需要用量/耗时(记日志、算成本、展示"本次消耗 N tokens")时用
+ * {@link LlmService#chatDetail}。
+ *
+ * @param text         模型输出正文(多个流式分片已拼成完整文本)
+ * @param modelId      {@code model.id}
+ * @param modelName    模型名({@code provider:modelId} 形态,便于定位实际打到哪个厂商)
+ * @param inputTokens  输入 token 数;厂商未返回时为 0
+ * @param outputTokens 输出 token 数;厂商未返回时为 0
+ * @param elapsedMs    端到端耗时(毫秒)
+ */
+public record LlmResult(String text, Long modelId, String modelName,
+                        int inputTokens, int outputTokens, long elapsedMs) {
+
+    /**
+     * 输入 + 输出 token 合计
+     */
+    public int totalTokens() {
+        return inputTokens + outputTokens;
+    }
+}

+ 387 - 0
ai-server/src/main/java/com/zsjz/ai/module/agent/llm/LlmService.java

@@ -0,0 +1,387 @@
+package com.zsjz.ai.module.agent.llm;
+
+import com.zsjz.ai.common.exception.ServerException;
+import com.zsjz.ai.common.utils.Json;
+import com.zsjz.ai.module.agent.entity.AgentModel;
+import com.zsjz.ai.module.agent.mapper.AgentModelMapper;
+import com.zsjz.ai.module.agent.prompt.PromptHelper;
+import com.zsjz.ai.module.agent.service.AgentModelFactory;
+import com.zsjz.ai.module.agent.service.AgentModelService;
+import io.agentscope.core.message.ContentBlock;
+import io.agentscope.core.message.Msg;
+import io.agentscope.core.message.MsgRole;
+import io.agentscope.core.message.TextBlock;
+import io.agentscope.core.model.ChatResponse;
+import io.agentscope.core.model.ChatUsage;
+import io.agentscope.core.model.GenerateOptions;
+import io.agentscope.core.model.Model;
+import lombok.extern.slf4j.Slf4j;
+import org.springframework.stereotype.Service;
+import org.springframework.util.StringUtils;
+import reactor.core.publisher.Flux;
+
+import java.util.ArrayList;
+import java.util.List;
+import java.util.concurrent.TimeoutException;
+import java.util.concurrent.atomic.AtomicReference;
+
+/**
+ * 直连大模型服务:<b>不经过 Agent、不带工具、不做会话管理</b>,一问一答直接拿模型输出。
+ *
+ * <h3>和 Agent 链路的分工</h3>
+ * <ul>
+ *   <li>{@code AgentService} / {@code HarnessAgent}:需要工具调用、多轮记忆、工作空间的场景;</li>
+ *   <li><b>本服务</b>:意图识别、文本清洗、摘要、字段抽取、翻译、打标签这类「拿提示词换文本」的
+ *       纯 LLM 调用 —— 起一个 ReActAgent 是纯浪费(每轮都要构造工具 schema、跑 ReAct 循环)。</li>
+ * </ul>
+ *
+ * <h3>模型从哪来</h3>
+ * 复用 {@code model} 表的配置(provider / modelId / apiKey / baseUrl / config),
+ * 与 Agent 链路<b>同一套连接参数</b>,改配置两边同时生效:
+ * <ul>
+ *   <li>不传 {@code modelId} → 用「默认对话模型」({@code type=llm} + {@code default_model=true} + 启用);</li>
+ *   <li>传 {@code modelId} → 用指定模型(可以专门配一个便宜的小模型跑这类杂活)。</li>
+ * </ul>
+ *
+ * <h3>用法</h3>
+ * <pre>{@code
+ * // 1) 最常用:默认模型 + 一段提示词
+ * String answer = llmService.chat("把下面这段通话记录总结成三句话:\n" + raw);
+ *
+ * // 2) 指定模型 + 系统提示词
+ * String tag = llmService.chat(3L, "你是金融数据分类助手", "给这笔交易打一个类型标签");
+ *
+ * // 3) 结构化输出:模型返回 JSON,自动转成 Java 对象
+ * IntentResult r = llmService.chatAs(
+ *         LlmRequest.builder().system("...").user(question).build(), IntentResult.class);
+ *
+ * // 4) 流式(给前端 SSE 逐字上屏)
+ * llmService.stream(LlmRequest.builder().user(prompt).build())
+ *         .subscribe(chunk -> emitter.send(chunk));
+ *
+ * // 5) 需要 token 用量/耗时
+ * LlmResult detail = llmService.chatDetail(LlmRequest.builder().user(prompt).build());
+ * }</pre>
+ *
+ * <h3>失败语义</h3>
+ * 与 {@code IntentService} / {@code FollowupService} 的 fail-open 不同 —— 那两个是辅助链路,
+ * 失败要静默降级;本服务是<b>调用方主动要结果</b>,因此一律抛
+ * {@link ServerException}(400 参数问题 / 404 模型不存在 / 500 调用失败 / 504 超时),
+ * 绝不返回 {@code null} 让调用方猜。
+ *
+ * <p><b>注意</b>:本服务不做会话记忆、不做重试、不做限流 —— 需要这些请走 Agent 链路。
+ */
+@Slf4j
+@Service
+public class LlmService {
+
+    private final AgentModelService agentModelService;
+
+    private final AgentModelMapper agentModelMapper;
+
+    private final AgentModelFactory agentModelFactory;
+
+    public LlmService(AgentModelService agentModelService,
+                      AgentModelMapper agentModelMapper,
+                      AgentModelFactory agentModelFactory) {
+        this.agentModelService = agentModelService;
+        this.agentModelMapper = agentModelMapper;
+        this.agentModelFactory = agentModelFactory;
+    }
+
+    // ==================== 便捷入口 ====================
+
+    /**
+     * 用默认对话模型跑一段提示词,直接拿文本。
+     *
+     * @param prompt 用户提示词
+     * @return 模型输出正文
+     */
+    public String chat(String prompt) {
+        return chat(LlmRequest.builder().user(prompt).build());
+    }
+
+    /**
+     * 用指定模型跑一段提示词,直接拿文本。
+     *
+     * @param modelId 模型 ID({@code model.id})
+     * @param prompt  用户提示词
+     */
+    public String chat(Long modelId, String prompt) {
+        return chat(LlmRequest.builder().modelId(modelId).user(prompt).build());
+    }
+
+    /**
+     * 系统提示词 + 用户提示词。
+     *
+     * @param modelId      模型 ID,可为 null(用默认对话模型)
+     * @param systemPrompt 系统提示词,可为空
+     * @param userPrompt   用户提示词
+     */
+    public String chat(Long modelId, String systemPrompt, String userPrompt) {
+        return chat(LlmRequest.builder()
+                .modelId(modelId)
+                .system(systemPrompt)
+                .user(userPrompt)
+                .build());
+    }
+
+    /**
+     * 完整参数调用,直接拿文本。
+     */
+    public String chat(LlmRequest request) {
+        return chatDetail(request).text();
+    }
+
+    // ==================== 核心 ====================
+
+    /**
+     * 完整参数调用,返回文本 + 用量 + 耗时。
+     *
+     * <p>模型是流式的({@code Model#stream}),这里把分片按到达顺序拼成完整文本;
+     * 厂商返回 usage 时一并带出。
+     *
+     * @param request 请求参数
+     * @return 调用结果
+     * @throws ServerException 参数非法 / 无可用模型 / 模型配置不存在 / 调用失败 / 超时
+     */
+    public LlmResult chatDetail(LlmRequest request) {
+        if (request == null) {
+            throw new ServerException(400, "request 不能为空");
+        }
+        ResolvedModel resolved = resolveModel(request.getModelId());
+        List<Msg> messages = buildMessages(request);
+        GenerateOptions options = buildOptions(request);
+
+        // StringBuffer 而非 StringBuilder:流式分片可能在别的线程回调
+        StringBuffer text = new StringBuffer();
+        AtomicReference<ChatUsage> usage = new AtomicReference<>();
+        long start = System.currentTimeMillis();
+        try {
+            resolved.model().stream(messages, List.of(), options)
+                    .doOnNext(resp -> {
+                        appendText(resp, text);
+                        if (resp.getUsage() != null) {
+                            usage.set(resp.getUsage());
+                        }
+                    })
+                    // 用 timeout 操作符而不是 blockLast(Duration):后者把「流内错误」也统一抛成
+                    // IllegalStateException("Timeout on blocking read..."),没法与真正的超时区分 ——
+                    // 模型返回 401/429 会被误报成「超时」,排查方向直接跑偏。
+                    .timeout(request.timeoutOrDefault())
+                    .onErrorMap(TimeoutException.class, e -> new ServerException(504,
+                            "模型调用超时(" + request.timeoutOrDefault().toSeconds()
+                                    + "s): " + resolved.modelName()))
+                    .blockLast();
+        } catch (ServerException e) {
+            throw e;
+        } catch (Exception e) {
+            log.error("直连模型调用失败: model={}, error={}", resolved.modelName(), e.getMessage());
+            throw new ServerException(500, "模型调用失败[" + resolved.modelName() + "]: " + e.getMessage());
+        }
+
+        long elapsed = System.currentTimeMillis() - start;
+        ChatUsage u = usage.get();
+        int inputTokens = u == null ? 0 : u.getInputTokens();
+        int outputTokens = u == null ? 0 : u.getOutputTokens();
+        log.info("直连模型调用完成: model={}, 输入={}t, 输出={}t, 耗时={}ms",
+                resolved.modelName(), inputTokens, outputTokens, elapsed);
+        return new LlmResult(text.toString(), resolved.id(), resolved.modelName(),
+                inputTokens, outputTokens, elapsed);
+    }
+
+    /**
+     * 结构化输出:把 Java 类型的 JSON Schema 注入提示词,模型返回 JSON 后反序列化成对象。
+     *
+     * <p>不走 AgentScope 的 {@code getStructuredData}(那需要 ReActAgent 的
+     * {@code generate_response} 工具),而是「提示词约束 + 宽松解析」:
+     * 容忍 {@code ```json} 围栏、前后夹带的解释文字。
+     *
+     * @param request 请求参数;{@code system} 会被追加 schema 说明
+     * @param type    目标类型(POJO,字段用 public 或带 getter/setter)
+     * @param <T>     目标类型
+     * @return 反序列化后的对象
+     * @throws ServerException 模型未返回合法 JSON 或字段不匹配
+     */
+    public <T> T chatAs(LlmRequest request, Class<T> type) {
+        if (type == null) {
+            throw new ServerException(400, "type 不能为空");
+        }
+        String schema = PromptHelper.generateSchema(type);
+        String raw = chat(request.toBuilder().system(appendSchema(request.getSystem(), schema)).build());
+        return parseStructured(raw, type);
+    }
+
+    /**
+     * 流式调用:返回模型输出分片(增量文本),调用方自己决定怎么消费(SSE / 逐字上屏 / 拼接)。
+     *
+     * <p>注意:本方法<b>不做超时控制</b>(订阅方负责),也不抛 {@link ServerException} ——
+     * 错误以 {@code Flux.error} 形式下发,由订阅方决定降级策略。
+     *
+     * @param request 请求参数
+     * @return 增量文本流
+     */
+    public Flux<String> stream(LlmRequest request) {
+        if (request == null) {
+            throw new ServerException(400, "request 不能为空");
+        }
+        ResolvedModel resolved = resolveModel(request.getModelId());
+        List<Msg> messages = buildMessages(request);
+        GenerateOptions options = buildOptions(request);
+        return resolved.model().stream(messages, List.of(), options)
+                .flatMap(resp -> Flux.fromIterable(textChunks(resp)));
+    }
+
+    // ==================== 内部 ====================
+
+    /**
+     * 已解析的模型:ID + 配置 + 实例
+     */
+    private record ResolvedModel(Long id, AgentModel config, Model model) {
+
+        String modelName() {
+            return config.getProvider() + ":" + config.getModelId();
+        }
+    }
+
+    /**
+     * 解析要用的模型:显式 modelId 优先,否则取默认对话模型。
+     */
+    private ResolvedModel resolveModel(Long modelId) {
+        Long id = modelId;
+        if (id == null) {
+            id = agentModelService.getDefaultModelId();
+            if (id == null) {
+                throw new ServerException(400,
+                        "没有可用的对话模型:请在「模型配置」里把某个类型为 llm 的模型设为默认");
+            }
+        }
+        AgentModel config = agentModelMapper.selectById(id);
+        if (config == null) {
+            throw new ServerException(404, "模型配置不存在: " + id);
+        }
+        return new ResolvedModel(id, config, agentModelFactory.create(config));
+    }
+
+    /**
+     * 组装消息:给了 messages 就用它,否则 system + user。
+     */
+    private static List<Msg> buildMessages(LlmRequest request) {
+        if (request.getMessages() != null && !request.getMessages().isEmpty()) {
+            return request.getMessages();
+        }
+        if (!StringUtils.hasText(request.getUser())) {
+            throw new ServerException(400, "缺少提示词:请设置 user 或 messages");
+        }
+        List<Msg> messages = new ArrayList<>(2);
+        if (StringUtils.hasText(request.getSystem())) {
+            messages.add(Msg.builder().role(MsgRole.SYSTEM).textContent(request.getSystem()).build());
+        }
+        messages.add(Msg.builder().role(MsgRole.USER).textContent(request.getUser()).build());
+        return messages;
+    }
+
+    /**
+     * 组装生成参数;连接参数(apiKey / baseUrl / stream)已由
+     * {@code AgentModelFactory} 在构造 Model 时带上,这里只放采样参数。
+     */
+    private static GenerateOptions buildOptions(LlmRequest request) {
+        GenerateOptions.Builder builder = GenerateOptions.builder();
+        if (request.getTemperature() != null) {
+            builder.temperature(request.getTemperature());
+        }
+        if (request.getMaxTokens() != null) {
+            builder.maxTokens(request.getMaxTokens());
+        }
+        return builder.build();
+    }
+
+    /**
+     * 把一次响应的文本分片追加到缓冲区
+     */
+    private static void appendText(ChatResponse resp, StringBuffer target) {
+        for (String chunk : textChunks(resp)) {
+            target.append(chunk);
+        }
+    }
+
+    /**
+     * 取出一次响应里的全部文本分片(跳过 thinking / tool_use 等非文本块)
+     */
+    private static List<String> textChunks(ChatResponse resp) {
+        if (resp == null || resp.getContent() == null) {
+            return List.of();
+        }
+        List<String> chunks = new ArrayList<>();
+        for (ContentBlock block : resp.getContent()) {
+            if (block instanceof TextBlock t && t.getText() != null && !t.getText().isEmpty()) {
+                chunks.add(t.getText());
+            }
+        }
+        return chunks;
+    }
+
+    /**
+     * 在系统提示词后追加「只输出 JSON」约束与 schema
+     */
+    private static String appendSchema(String system, String schema) {
+        StringBuilder sb = new StringBuilder();
+        if (StringUtils.hasText(system)) {
+            sb.append(system).append("\n\n");
+        }
+        sb.append("输出要求:只输出一个 JSON 对象,不要输出任何解释文字,不要用 markdown 代码块包裹。\n")
+                .append("JSON 必须符合以下 schema:\n")
+                .append(schema);
+        return sb.toString();
+    }
+
+    /**
+     * 宽松解析模型输出:剥 {@code ```} 围栏 → 截取最外层 JSON → Jackson 反序列化
+     */
+    private static <T> T parseStructured(String raw, Class<T> type) {
+        String json = extractJson(raw);
+        if (json == null) {
+            throw new ServerException(500, "模型未返回合法 JSON: " + abbreviate(raw));
+        }
+        try {
+            return Json.objectMapper().readValue(json, type);
+        } catch (Exception e) {
+            throw new ServerException(500,
+                    "模型输出无法解析为 " + type.getSimpleName() + ": " + e.getMessage()
+                            + ",原始输出: " + abbreviate(raw));
+        }
+    }
+
+    /**
+     * 从可能夹带说明文字的模型输出里截出最外层 JSON
+     */
+    private static String extractJson(String raw) {
+        if (!StringUtils.hasText(raw)) {
+            return null;
+        }
+        String text = raw.trim();
+        if (text.startsWith("```")) {
+            int firstBreak = text.indexOf('\n');
+            int lastFence = text.lastIndexOf("```");
+            if (firstBreak > 0 && lastFence > firstBreak) {
+                text = text.substring(firstBreak + 1, lastFence).trim();
+            }
+        }
+        int startObj = text.indexOf('{');
+        int endObj = text.lastIndexOf('}');
+        int startArr = text.indexOf('[');
+        int endArr = text.lastIndexOf(']');
+        boolean arrayFirst = startArr >= 0 && (startObj < 0 || startArr < startObj);
+        if (arrayFirst) {
+            return endArr > startArr ? text.substring(startArr, endArr + 1) : null;
+        }
+        return endObj > startObj && startObj >= 0 ? text.substring(startObj, endObj + 1) : null;
+    }
+
+    private static String abbreviate(String s) {
+        if (s == null) {
+            return "";
+        }
+        return s.length() > 200 ? s.substring(0, 200) + "..." : s;
+    }
+}

+ 5 - 0
ai-server/src/main/java/com/zsjz/ai/module/plat/mapper/PhoneIspMapper.java

@@ -1,7 +1,10 @@
 package com.zsjz.ai.module.plat.mapper;
 
+import com.baomidou.dynamic.datasource.annotation.DS;
 import com.baomidou.mybatisplus.core.mapper.BaseMapper;
+import com.zsjz.ai.common.constants.StrConsts;
 import com.zsjz.ai.common.model.plat.entity.PhoneIsp;
+import org.apache.ibatis.annotations.Mapper;
 
 /**
  * 手机运营商 Mapper 接口
@@ -10,5 +13,7 @@ import com.zsjz.ai.common.model.plat.entity.PhoneIsp;
  * @version 1.0
  * @since 2025/7/28
  */
+@Mapper
+@DS(StrConsts.DS_KEY_SLAVE)
 public interface PhoneIspMapper extends BaseMapper<PhoneIsp> {
 }

+ 185 - 0
ai-server/src/test/java/com/zsjz/ai/module/agent/llm/LlmServiceHttpTest.java

@@ -0,0 +1,185 @@
+package com.zsjz.ai.module.agent.llm;
+
+import com.zsjz.ai.common.exception.ServerException;
+import com.zsjz.ai.module.agent.entity.AgentModel;
+import com.zsjz.ai.module.agent.mapper.AgentModelMapper;
+import com.zsjz.ai.module.agent.service.AgentModelFactory;
+import com.zsjz.ai.module.agent.service.AgentModelService;
+import org.junit.jupiter.api.BeforeAll;
+import org.junit.jupiter.api.DisplayName;
+import org.junit.jupiter.api.Test;
+import org.junit.jupiter.api.condition.EnabledIfSystemProperty;
+
+import java.net.URI;
+import java.net.http.HttpClient;
+import java.net.http.HttpRequest;
+import java.net.http.HttpResponse;
+import java.time.Duration;
+import java.util.ArrayList;
+import java.util.List;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertFalse;
+import static org.junit.jupiter.api.Assertions.assertThrows;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+import static org.junit.jupiter.api.Assertions.fail;
+import static org.mockito.Mockito.mock;
+import static org.mockito.Mockito.when;
+
+/**
+ * {@link LlmService} 的<b>真实 HTTP</b>链路测试。
+ *
+ * <p>三层验证是互补的,不要互相替代:
+ * <ul>
+ *   <li>{@link LlmServiceTest} —— 全 mock,验证分支逻辑(19 例);</li>
+ *   <li><b>本测试</b> —— 真 HTTP、假模型端点,验证 SSE 解析 / usage 提取 / 错误码映射;</li>
+ *   <li>{@link LlmServiceLiveTest} —— 真模型端点,验证 DB 里的 apiKey / baseUrl 确实可用。</li>
+ * </ul>
+ *
+ * <p>这里刻意<b>不启动 Spring 上下文</b>(那要 3 分钟,还得给 PhoneIspMapper 打补丁),
+ * 直接 {@code new LlmService(...)} + Mockito 假 mapper —— 秒级可跑,可反复跑。
+ *
+ * <p>跑法:
+ * <pre>
+ * # 终端 1
+ * python ai-server/src/test/resources/mock-openai-server.py 18999
+ * # 终端 2
+ * mvn -pl ai-server test -Dtest=LlmServiceHttpTest -Dmock.llm.port=18999
+ * </pre>
+ */
+@EnabledIfSystemProperty(named = "mock.llm.port", matches = "\\d+")
+class LlmServiceHttpTest {
+
+    /** mock 端点返回的分片,拼接后应等于这个常量 */
+    private static final String EXPECTED_TEXT = "DuckDB 是一个嵌入式的 OLAP 数据库。";
+
+    private static final long MODEL_ID = 99L;
+
+    @BeforeAll
+    static void mockEndpointMustBeUp() {
+        String url = "http://127.0.0.1:" + port() + "/";
+        try (HttpClient client = HttpClient.newBuilder()
+                .connectTimeout(Duration.ofSeconds(3)).build()) {
+            HttpResponse<String> resp = client.send(
+                    HttpRequest.newBuilder(URI.create(url)).GET().build(),
+                    HttpResponse.BodyHandlers.ofString());
+            assertEquals(200, resp.statusCode(), "mock 模型端点未正常响应");
+        } catch (Exception e) {
+            fail("mock 模型端点不可用(" + url + "):请先在另一个终端执行 "
+                    + "`python ai-server/src/test/resources/mock-openai-server.py " + port() + "`。"
+                    + " 原因: " + e);
+        }
+    }
+
+    // ==================== 用例 ====================
+
+    @Test
+    @DisplayName("真实 SSE:分片按序拼成完整文本")
+    void chatAssemblesStreamedChunks() {
+        LlmService llm = service(config("mock-model"));
+
+        String text = llm.chat("介绍一下 DuckDB");
+
+        assertEquals(EXPECTED_TEXT, text);
+    }
+
+    @Test
+    @DisplayName("真实 SSE:usage 与模型名被正确带回")
+    void chatDetailCarriesUsage() {
+        LlmService llm = service(config("mock-model"));
+
+        LlmResult result = llm.chatDetail(LlmRequest.builder().user("hi").build());
+
+        assertEquals(EXPECTED_TEXT, result.text());
+        assertEquals("openai:mock-model", result.modelName());
+        assertEquals(MODEL_ID, result.modelId());
+        assertEquals(11, result.inputTokens(), "应从末帧 usage 读到 prompt_tokens");
+        assertEquals(7, result.outputTokens(), "应从末帧 usage 读到 completion_tokens");
+        assertEquals(18, result.totalTokens());
+        assertTrue(result.elapsedMs() >= 0);
+    }
+
+    @Test
+    @DisplayName("真实 SSE:stream() 逐片下发而非一次性返回")
+    void streamYieldsMultipleChunks() {
+        LlmService llm = service(config("mock-model"));
+        List<String> chunks = new ArrayList<>();
+
+        llm.stream(LlmRequest.builder().user("hi").build())
+                .doOnNext(chunks::add)
+                .blockLast();
+
+        assertTrue(chunks.size() > 1, "应收到多个分片,实际 " + chunks.size() + " 个");
+        assertEquals(EXPECTED_TEXT, String.join("", chunks));
+    }
+
+    @Test
+    @DisplayName("模型返回 401 时报 500 且不能被误报成超时")
+    void modelErrorIsNotReportedAsTimeout() {
+        LlmService llm = service(config("mock-unauthorized"));
+
+        ServerException e = assertThrows(ServerException.class,
+                () -> llm.chatDetail(LlmRequest.builder().user("hi").build()));
+
+        assertEquals(500, e.getCode(), "模型侧错误应映射为 500,而不是 504");
+        assertFalse(e.getMessage().contains("超时"),
+                "401 被当成超时会让排查方向跑偏,实际消息: " + e.getMessage());
+    }
+
+    @Test
+    @DisplayName("超时确实映射为 504")
+    void slowModelBecomes504() {
+        LlmService llm = service(config("mock-slow"));
+
+        ServerException e = assertThrows(ServerException.class, () -> llm.chatDetail(
+                LlmRequest.builder().user("hi").timeout(Duration.ofSeconds(2)).build()));
+
+        assertEquals(504, e.getCode());
+        assertTrue(e.getMessage().contains("超时"), "消息里应点明超时: " + e.getMessage());
+    }
+
+    @Test
+    @DisplayName("模型配置查不到时抛 404,不返回 null")
+    void missingModelConfigBecomes404() {
+        AgentModelService svc = mock(AgentModelService.class);
+        AgentModelMapper mapper = mock(AgentModelMapper.class);
+        when(svc.getDefaultModelId()).thenReturn(MODEL_ID);
+        when(mapper.selectById(MODEL_ID)).thenReturn(null);
+        LlmService llm = new LlmService(svc, mapper, new AgentModelFactory());
+
+        ServerException e = assertThrows(ServerException.class, () -> llm.chat("hi"));
+
+        assertEquals(404, e.getCode());
+    }
+
+    // ==================== 装配 ====================
+
+    private static String port() {
+        return System.getProperty("mock.llm.port");
+    }
+
+    /**
+     * 造一条指向本地 mock 端点的模型配置
+     */
+    private static AgentModel config(String modelId) {
+        AgentModel config = new AgentModel();
+        config.setId(MODEL_ID);
+        config.setName("mock-" + modelId);
+        config.setProvider("openai");
+        config.setModelId(modelId);
+        config.setApiKey("mock-key");
+        config.setBaseUrl("http://127.0.0.1:" + port() + "/v1");
+        return config;
+    }
+
+    /**
+     * 用假 mapper 喂配置,但 model 实例、HTTP 传输、SSE 解析全是真的
+     */
+    private static LlmService service(AgentModel config) {
+        AgentModelService svc = mock(AgentModelService.class);
+        AgentModelMapper mapper = mock(AgentModelMapper.class);
+        when(svc.getDefaultModelId()).thenReturn(MODEL_ID);
+        when(mapper.selectById(MODEL_ID)).thenReturn(config);
+        return new LlmService(svc, mapper, new AgentModelFactory());
+    }
+}

+ 98 - 0
ai-server/src/test/java/com/zsjz/ai/module/agent/llm/LlmServiceLiveTest.java

@@ -0,0 +1,98 @@
+package com.zsjz.ai.module.agent.llm;
+
+import com.zsjz.ai.module.plat.mapper.PhoneIspMapper;
+import org.junit.jupiter.api.DisplayName;
+import org.junit.jupiter.api.Test;
+import org.junit.jupiter.api.condition.EnabledIfSystemProperty;
+import org.springframework.beans.factory.annotation.Autowired;
+import org.springframework.boot.test.context.SpringBootTest;
+import org.springframework.test.context.bean.override.mockito.MockitoBean;
+
+import static org.junit.jupiter.api.Assertions.assertFalse;
+import static org.junit.jupiter.api.Assertions.assertNotNull;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+
+/**
+ * {@link LlmService} 的真实链路冒烟测试。
+ *
+ * <p>与 {@link LlmServiceTest}(全 mock,验证逻辑)互补 —— 这里真的连一次模型,
+ * 证明「DB 模型配置 → ModelRegistry → 厂商接口 → 拼回文本」整条链路可用。
+ *
+ * <p><b>默认不执行</b>:需要真实的 {@code model} 表配置(含 apiKey)且会消耗 token。
+ * 手动跑:
+ * <pre>
+ * mvn -pl ai-server test -Dtest=LlmServiceLiveTest -Dllm.live=true
+ * </pre>
+ */
+@SpringBootTest(webEnvironment = SpringBootTest.WebEnvironment.NONE)
+@EnabledIfSystemProperty(named = "llm.live", matches = "true")
+class LlmServiceLiveTest {
+
+    /**
+     * 启动上下文必须的一个「补丁 bean」。
+     *
+     * <p>{@code PhoneIspMapper} 是 {@code com.zsjz.ai.module.plat.mapper} 下唯一<b>漏了
+     * {@code @Mapper} 注解</b>的接口(本项目不用 {@code @MapperScan},全靠该注解注册),
+     * 于是 {@code GlobalCache#initIspData()} 里的 {@code SpringUtil.getBean(PhoneIspMapper.class)}
+     * 在启动时抛 {@code NoSuchBeanDefinitionException}。线上之所以没炸,是因为
+     * {@code AppLoadEndEventListener} 把整个初始化 try-catch 掉了(日志里那句
+     * 「初始化系统文件失败」就是它)—— 代价是**手机号运营商映射从来没被加载过**。
+     *
+     * <p>这里用 mock 顶上,只为让本测试的上下文能起来,不改动生产代码。
+     */
+    @MockitoBean
+    private PhoneIspMapper phoneIspMapper;
+
+    @Autowired
+    private LlmService llmService;
+
+    @Test
+    @DisplayName("默认模型:一段提示词换一段文本")
+    void chatWithDefaultModel() {
+        LlmResult result = llmService.chatDetail(
+                LlmRequest.builder().user("用一句话(不超过 30 字)说明 DuckDB 是什么。").build());
+
+        System.out.println("[LIVE] model=" + result.modelName()
+                + ", tokens=" + result.inputTokens() + "+" + result.outputTokens()
+                + ", elapsed=" + result.elapsedMs() + "ms");
+        System.out.println("[LIVE] text=" + result.text());
+
+        assertNotNull(result.modelName(), "应能解析出实际模型名");
+        assertFalse(result.text().isBlank(), "模型不该返回空文本");
+    }
+
+    @Test
+    @DisplayName("结构化输出:真的能解析成 POJO")
+    void chatAsRealCall() {
+        LlmRequest request = LlmRequest.builder()
+                .system("你是金融交易分类助手。")
+                .user("给这笔交易打标签:2026-09-19 23:47,转账 98 万元,收款方为个人账户。")
+                .temperature(0.0)
+                .build();
+
+        SmokeResult result = llmService.chatAs(request, SmokeResult.class);
+
+        System.out.println("[LIVE] label=" + result.label + ", confidence=" + result.confidence);
+
+        assertNotNull(result.label, "结构化字段不该为空");
+        assertTrue(result.confidence >= 0, "confidence 应为数值");
+    }
+
+    @Test
+    @DisplayName("流式:分片能拼成非空文本")
+    void streamRealCall() {
+        StringBuilder sb = new StringBuilder();
+        llmService.stream(LlmRequest.builder().user("从 1 数到 5,只输出数字,用逗号分隔。").build())
+                .doOnNext(sb::append)
+                .blockLast();
+
+        System.out.println("[LIVE] stream=" + sb);
+        assertFalse(sb.toString().isBlank());
+    }
+
+    /** 结构化输出目标类型 */
+    public static class SmokeResult {
+        public String label;
+        public double confidence;
+    }
+}

+ 378 - 0
ai-server/src/test/java/com/zsjz/ai/module/agent/llm/LlmServiceTest.java

@@ -0,0 +1,378 @@
+package com.zsjz.ai.module.agent.llm;
+
+import com.zsjz.ai.common.exception.ServerException;
+import com.zsjz.ai.module.agent.entity.AgentModel;
+import com.zsjz.ai.module.agent.mapper.AgentModelMapper;
+import com.zsjz.ai.module.agent.service.AgentModelFactory;
+import com.zsjz.ai.module.agent.service.AgentModelService;
+import io.agentscope.core.message.ContentBlock;
+import io.agentscope.core.message.Msg;
+import io.agentscope.core.message.MsgRole;
+import io.agentscope.core.message.TextBlock;
+import io.agentscope.core.model.ChatResponse;
+import io.agentscope.core.model.ChatUsage;
+import io.agentscope.core.model.GenerateOptions;
+import io.agentscope.core.model.Model;
+import io.agentscope.core.model.ToolSchema;
+import org.junit.jupiter.api.BeforeEach;
+import org.junit.jupiter.api.DisplayName;
+import org.junit.jupiter.api.Test;
+import org.mockito.ArgumentCaptor;
+import reactor.core.publisher.Flux;
+
+import java.time.Duration;
+import java.util.List;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertFalse;
+import static org.junit.jupiter.api.Assertions.assertNull;
+import static org.junit.jupiter.api.Assertions.assertThrows;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+import static org.mockito.ArgumentMatchers.any;
+import static org.mockito.ArgumentMatchers.anyList;
+import static org.mockito.ArgumentMatchers.eq;
+import static org.mockito.Mockito.mock;
+import static org.mockito.Mockito.verify;
+import static org.mockito.Mockito.when;
+
+/**
+ * {@link LlmService} 的契约测试。
+ *
+ * <p>守住四件事:
+ * <ol>
+ *   <li><b>模型解析</b>:不传 modelId 走默认对话模型,传了就走指定模型;两者都拿不到时
+ *       必须抛可读异常(而不是 NPE / 返回 null);</li>
+ *   <li><b>流式分片拼接</b>:{@code Model#stream} 吐的是增量,服务要拼成完整文本,
+ *       且 usage 要带出来;</li>
+ *   <li><b>结构化输出</b>:容忍 {@code ```json} 围栏与夹带的解释文字;解析不了要抛异常;</li>
+ *   <li><b>失败语义</b>:一律 {@link ServerException},绝不静默返回 null
+ *       (与 IntentService/FollowupService 的 fail-open 不同,这是调用方主动要结果的场景)。</li>
+ * </ol>
+ */
+class LlmServiceTest {
+
+    private AgentModelService agentModelService;
+    private AgentModelMapper agentModelMapper;
+    private AgentModelFactory agentModelFactory;
+    private LlmService llmService;
+
+    @BeforeEach
+    void setUp() {
+        agentModelService = mock(AgentModelService.class);
+        agentModelMapper = mock(AgentModelMapper.class);
+        agentModelFactory = mock(AgentModelFactory.class);
+        llmService = new LlmService(agentModelService, agentModelMapper, agentModelFactory);
+    }
+
+    // ==================== 辅助 ====================
+
+    private AgentModel config(Long id, String provider, String modelId) {
+        AgentModel c = new AgentModel();
+        c.setId(id);
+        c.setProvider(provider);
+        c.setModelId(modelId);
+        c.setName("测试模型");
+        return c;
+    }
+
+    /**
+     * 让 resolveModel 走通:默认模型 ID = 1,配置存在,Model 实例已 mock
+     */
+    private Model stubDefaultModel() {
+        AgentModel cfg = config(1L, "openai", "gpt-4o-mini");
+        when(agentModelService.getDefaultModelId()).thenReturn(1L);
+        when(agentModelMapper.selectById(1L)).thenReturn(cfg);
+        Model model = mock(Model.class);
+        when(agentModelFactory.create(cfg)).thenReturn(model);
+        return model;
+    }
+
+    private static ChatResponse text(String content) {
+        return ChatResponse.builder()
+                .content(List.of(TextBlock.builder().text(content).build()))
+                .build();
+    }
+
+    // ==================== 1. 基本调用 ====================
+
+    @Test
+    @DisplayName("chat(prompt):用默认模型,把流式分片拼成完整文本")
+    void chatWithDefaultModel() {
+        Model model = stubDefaultModel();
+        when(model.stream(anyList(), anyList(), any())).thenReturn(Flux.just(
+                text("本案共"), text(" 38263 条"), text("通话记录")));
+
+        String result = llmService.chat("统计一下通话");
+
+        assertEquals("本案共 38263 条通话记录", result);
+
+        @SuppressWarnings("unchecked")
+        ArgumentCaptor<List<Msg>> captor = ArgumentCaptor.forClass(List.class);
+        verify(model).stream(captor.capture(), anyList(), any());
+        List<Msg> messages = captor.getValue();
+        assertEquals(1, messages.size(), "没给 system 时只应有一条 user 消息");
+        assertEquals(MsgRole.USER, messages.get(0).getRole());
+        assertEquals("统计一下通话", messages.get(0).getTextContent());
+    }
+
+    @Test
+    @DisplayName("chat(modelId, system, user):system 在前,且用的是指定模型")
+    void chatWithExplicitModelAndSystem() {
+        AgentModel cfg = config(7L, "dashscope", "qwen-plus");
+        when(agentModelMapper.selectById(7L)).thenReturn(cfg);
+        Model model = mock(Model.class);
+        when(agentModelFactory.create(cfg)).thenReturn(model);
+        when(model.stream(anyList(), anyList(), any())).thenReturn(Flux.just(text("已分类")));
+
+        String result = llmService.chat(7L, "你是金融分类助手", "给这笔交易打标签");
+
+        assertEquals("已分类", result);
+        // 显式传了 modelId 就不该再去查默认模型
+        verify(agentModelService, org.mockito.Mockito.never()).getDefaultModelId();
+
+        @SuppressWarnings("unchecked")
+        ArgumentCaptor<List<Msg>> captor = ArgumentCaptor.forClass(List.class);
+        verify(model).stream(captor.capture(), anyList(), any());
+        List<Msg> messages = captor.getValue();
+        assertEquals(2, messages.size());
+        assertEquals(MsgRole.SYSTEM, messages.get(0).getRole());
+        assertEquals("你是金融分类助手", messages.get(0).getTextContent());
+        assertEquals(MsgRole.USER, messages.get(1).getRole());
+    }
+
+    @Test
+    @DisplayName("messages 优先于 system/user")
+    void messagesWinOverSystemUser() {
+        Model model = stubDefaultModel();
+        when(model.stream(anyList(), anyList(), any())).thenReturn(Flux.just(text("ok")));
+
+        List<Msg> history = List.of(
+                Msg.builder().role(MsgRole.USER).textContent("第一轮").build(),
+                Msg.builder().role(MsgRole.ASSISTANT).textContent("回答").build(),
+                Msg.builder().role(MsgRole.USER).textContent("第二轮").build());
+        llmService.chat(LlmRequest.builder()
+                .system("不该出现")
+                .user("也不该出现")
+                .messages(history)
+                .build());
+
+        @SuppressWarnings("unchecked")
+        ArgumentCaptor<List<Msg>> captor = ArgumentCaptor.forClass(List.class);
+        verify(model).stream(captor.capture(), anyList(), any());
+        assertEquals(3, captor.getValue().size());
+        assertEquals("第一轮", captor.getValue().get(0).getTextContent());
+    }
+
+    @Test
+    @DisplayName("chatDetail 带出 usage 与耗时")
+    void chatDetailCarriesUsage() {
+        Model model = stubDefaultModel();
+        when(model.stream(anyList(), anyList(), any())).thenReturn(Flux.just(
+                ChatResponse.builder()
+                        .content(List.of(TextBlock.builder().text("结果").build()))
+                        .usage(ChatUsage.builder().inputTokens(120).outputTokens(8).build())
+                        .build()));
+
+        LlmResult result = llmService.chatDetail(LlmRequest.builder().user("hi").build());
+
+        assertEquals("结果", result.text());
+        assertEquals(1L, result.modelId());
+        assertEquals("openai:gpt-4o-mini", result.modelName());
+        assertEquals(120, result.inputTokens());
+        assertEquals(8, result.outputTokens());
+        assertEquals(128, result.totalTokens());
+        assertTrue(result.elapsedMs() >= 0);
+    }
+
+    @Test
+    @DisplayName("采样参数透传给 GenerateOptions")
+    void samplingOptionsAreForwarded() {
+        Model model = stubDefaultModel();
+        when(model.stream(anyList(), anyList(), any())).thenReturn(Flux.just(text("x")));
+
+        llmService.chat(LlmRequest.builder()
+                .user("hi").temperature(0.3).maxTokens(256).build());
+
+        ArgumentCaptor<GenerateOptions> captor = ArgumentCaptor.forClass(GenerateOptions.class);
+        verify(model).stream(anyList(), anyList(), captor.capture());
+        assertEquals(0.3, captor.getValue().getTemperature());
+        assertEquals(256, captor.getValue().getMaxTokens());
+    }
+
+    @Test
+    @DisplayName("stream() 原样下发分片")
+    void streamEmitsChunks() {
+        Model model = stubDefaultModel();
+        when(model.stream(anyList(), anyList(), any())).thenReturn(Flux.just(
+                text("逐"), text("字"), text("上屏")));
+
+        List<String> chunks = llmService.stream(LlmRequest.builder().user("hi").build())
+                .collectList().block();
+
+        assertEquals(List.of("逐", "字", "上屏"), chunks);
+    }
+
+    // ==================== 2. 结构化输出 ====================
+
+    /** 结构化输出的目标类型 */
+    public static class TagResult {
+        public String label;
+        public double confidence;
+    }
+
+    @Test
+    @DisplayName("chatAs:剥掉 ```json 围栏后反序列化")
+    void chatAsStripsFence() {
+        Model model = stubDefaultModel();
+        when(model.stream(anyList(), anyList(), any())).thenReturn(Flux.just(
+                text("```json\n{\"label\":\"大额交易\",\"confidence\":0.92}\n```")));
+
+        TagResult result = llmService.chatAs(LlmRequest.builder().user("打标签").build(), TagResult.class);
+
+        assertEquals("大额交易", result.label);
+        assertEquals(0.92, result.confidence);
+    }
+
+    @Test
+    @DisplayName("chatAs:容忍前后夹带的解释文字,且把 schema 注入 system")
+    void chatAsToleratesProseAndInjectsSchema() {
+        Model model = stubDefaultModel();
+        when(model.stream(anyList(), anyList(), any())).thenReturn(Flux.just(
+                text("好的,分析结果如下:{\"label\":\"夜间通话\",\"confidence\":0.8} 以上。")));
+
+        TagResult result = llmService.chatAs(LlmRequest.builder()
+                .system("你是分类助手").user("打标签").build(), TagResult.class);
+
+        assertEquals("夜间通话", result.label);
+
+        @SuppressWarnings("unchecked")
+        ArgumentCaptor<List<Msg>> captor = ArgumentCaptor.forClass(List.class);
+        verify(model).stream(captor.capture(), anyList(), any());
+        String system = captor.getValue().get(0).getTextContent();
+        assertTrue(system.startsWith("你是分类助手"), "原 system 必须保留: " + system);
+        assertTrue(system.contains("只输出一个 JSON 对象"), "缺少 JSON 约束: " + system);
+        assertTrue(system.contains("schema"), "缺少 schema 说明: " + system);
+    }
+
+    @Test
+    @DisplayName("chatAs:模型没吐 JSON 时抛 500 而不是返回 null")
+    void chatAsFailsOnNonJson() {
+        Model model = stubDefaultModel();
+        when(model.stream(anyList(), anyList(), any())).thenReturn(Flux.just(text("抱歉,我无法完成")));
+
+        ServerException e = assertThrows(ServerException.class,
+                () -> llmService.chatAs(LlmRequest.builder().user("x").build(), TagResult.class));
+        assertEquals(500, e.getCode());
+        assertTrue(e.getMessage().contains("未返回合法 JSON"), e.getMessage());
+    }
+
+    // ==================== 3. 失败语义 ====================
+
+    @Test
+    @DisplayName("没有默认模型时抛 400 并给出可操作的提示")
+    void noDefaultModel() {
+        when(agentModelService.getDefaultModelId()).thenReturn(null);
+
+        ServerException e = assertThrows(ServerException.class, () -> llmService.chat("hi"));
+        assertEquals(400, e.getCode());
+        assertTrue(e.getMessage().contains("默认"), e.getMessage());
+    }
+
+    @Test
+    @DisplayName("模型配置不存在时抛 404")
+    void modelConfigMissing() {
+        when(agentModelMapper.selectById(99L)).thenReturn(null);
+
+        ServerException e = assertThrows(ServerException.class, () -> llmService.chat(99L, "hi"));
+        assertEquals(404, e.getCode());
+    }
+
+    @Test
+    @DisplayName("缺提示词时抛 400")
+    void missingPrompt() {
+        stubDefaultModel();
+
+        ServerException e = assertThrows(ServerException.class,
+                () -> llmService.chat(LlmRequest.builder().build()));
+        assertEquals(400, e.getCode());
+        assertTrue(e.getMessage().contains("提示词"), e.getMessage());
+    }
+
+    @Test
+    @DisplayName("模型迟迟不返回时按 timeout 抛 504")
+    void timeoutBecomes504() {
+        Model model = stubDefaultModel();
+        when(model.stream(anyList(), anyList(), any())).thenReturn(Flux.never());
+
+        ServerException e = assertThrows(ServerException.class, () -> llmService.chat(
+                LlmRequest.builder().user("hi").timeout(Duration.ofMillis(150)).build()));
+        assertEquals(504, e.getCode());
+        assertTrue(e.getMessage().contains("超时"), e.getMessage());
+    }
+
+    @Test
+    @DisplayName("模型内部异常包成 500,不把 Reactor 的栈甩给调用方")
+    void modelErrorBecomes500() {
+        Model model = stubDefaultModel();
+        when(model.stream(anyList(), anyList(), any())).thenReturn(
+                Flux.error(new IllegalStateException("401 invalid api key")));
+
+        ServerException e = assertThrows(ServerException.class, () -> llmService.chat("hi"));
+        assertEquals(500, e.getCode());
+        assertTrue(e.getMessage().contains("401 invalid api key"), e.getMessage());
+    }
+
+    // ==================== 4. 边界 ====================
+
+    @Test
+    @DisplayName("非文本块(thinking / tool_use)不进正文")
+    void nonTextBlocksAreSkipped() {
+        Model model = stubDefaultModel();
+        ContentBlock thinking = mock(ContentBlock.class);
+        when(model.stream(anyList(), anyList(), any())).thenReturn(Flux.just(
+                ChatResponse.builder().content(List.of(thinking)).build(),
+                text("正文")));
+
+        assertEquals("正文", llmService.chat("hi"));
+    }
+
+    @Test
+    @DisplayName("timeout 默认 60s,显式设置时以显式值为准")
+    void timeoutDefaults() {
+        assertEquals(Duration.ofSeconds(60), LlmRequest.builder().build().timeoutOrDefault());
+        assertEquals(Duration.ofSeconds(5),
+                LlmRequest.builder().timeout(Duration.ofSeconds(5)).build().timeoutOrDefault());
+        assertNull(LlmRequest.builder().build().getTimeout());
+    }
+
+    @Test
+    @DisplayName("工具 schema 传空列表:直连不带工具")
+    void noToolsAreSent() {
+        Model model = stubDefaultModel();
+        when(model.stream(anyList(), anyList(), any())).thenReturn(Flux.just(text("x")));
+
+        llmService.chat("hi");
+
+        @SuppressWarnings("unchecked")
+        ArgumentCaptor<List<ToolSchema>> captor = ArgumentCaptor.forClass(List.class);
+        verify(model).stream(anyList(), captor.capture(), any());
+        assertTrue(captor.getValue().isEmpty(), "直连调用不该带任何工具");
+        assertFalse(captor.getValue() == null);
+    }
+
+    @Test
+    @DisplayName("chat(null) 抛 400 而不是 NPE")
+    void nullRequest() {
+        ServerException e = assertThrows(ServerException.class, () -> llmService.chat((LlmRequest) null));
+        assertEquals(400, e.getCode());
+    }
+
+    @Test
+    @DisplayName("stream() 的 Flux 可直接订阅,不需要 block")
+    void streamIsLazy() {
+        Model model = stubDefaultModel();
+        when(model.stream(anyList(), anyList(), any())).thenReturn(Flux.just(text("a")));
+
+        assertEquals(1, llmService.stream(LlmRequest.builder().user("hi").build()).count().block());
+    }
+}

+ 135 - 0
ai-server/src/test/resources/mock-openai-server.py

@@ -0,0 +1,135 @@
+#!/usr/bin/env python3
+# -*- coding: utf-8 -*-
+"""OpenAI 兼容的 mock 模型服务 —— 只为验证 LlmService 的「真实 HTTP 流式」链路。
+
+为什么需要它
+------------
+本机常常没有可用的大模型端点(Ollama 没启动、外网被 http_proxy 拦掉),
+但 LlmService 的核心风险恰恰在 HTTP 这一段(SSE 分片解析、usage 提取、
+错误码映射),全 mock 的单测覆盖不到。这个 mock 端点让 LlmService 真实走一遍:
+
+    DB 模型配置 -> AgentModelFactory -> HTTP SSE -> 分片拼接 -> LlmResult
+
+用法
+----
+    # 终端 1
+    python mock-openai-server.py 18999
+
+    # 终端 2
+    mvn -pl ai-server test -Dtest=LlmServiceHttpTest -Dmock.llm.port=18999
+
+模型名约定(测试用来触发特定分支)
+----------------------------------
+* 普通模型名          -> SSE 逐片吐 CHUNKS,末帧带 usage
+* mock-unauthorized   -> 返回 401,验证「模型报错不能被误报成超时」
+* mock-slow           -> 先睡 30s,验证超时分支(504)
+"""
+
+import json
+import sys
+import time
+from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
+
+# 分片拼接后应为:DuckDB 是一个嵌入式的 OLAP 数据库。
+CHUNKS = ["DuckDB", " 是", "一个", "嵌入式", "的 OLAP", " 数据库", "。"]
+PROMPT_TOKENS = 11
+
+
+class Handler(BaseHTTPRequestHandler):
+
+    protocol_version = "HTTP/1.1"
+
+    def log_message(self, fmt, *args):
+        sys.stderr.write("[mock-openai] " + (fmt % args) + "\n")
+        sys.stderr.flush()
+
+    # ---------- 工具 ----------
+
+    def _read_json(self):
+        length = int(self.headers.get("Content-Length") or 0)
+        raw = self.rfile.read(length) if length > 0 else b"{}"
+        try:
+            return json.loads(raw.decode("utf-8") or "{}")
+        except Exception:
+            return {}
+
+    def _write_json(self, status, payload):
+        body = json.dumps(payload, ensure_ascii=False).encode("utf-8")
+        self.send_response(status)
+        self.send_header("Content-Type", "application/json; charset=utf-8")
+        self.send_header("Content-Length", str(len(body)))
+        self.end_headers()
+        self.wfile.write(body)
+
+    def _chunk(self, data: bytes):
+        """写一个 HTTP/1.1 chunked 分片"""
+        self.wfile.write(("%X\r\n" % len(data)).encode("ascii"))
+        self.wfile.write(data)
+        self.wfile.write(b"\r\n")
+
+    # ---------- 路由 ----------
+
+    def do_GET(self):
+        """健康检查:Java 测试用它确认端口已就绪"""
+        self._write_json(200, {"status": "ok", "chunks": len(CHUNKS)})
+
+    def do_POST(self):
+        req = self._read_json()
+        model = req.get("model") or "mock-model"
+
+        if model == "mock-unauthorized":
+            self._write_json(401, {"error": {
+                "message": "Incorrect API key provided",
+                "type": "invalid_request_error",
+                "code": "invalid_api_key",
+            }})
+            return
+
+        self.send_response(200)
+        self.send_header("Content-Type", "text/event-stream; charset=utf-8")
+        self.send_header("Cache-Control", "no-cache")
+        self.send_header("Transfer-Encoding", "chunked")
+        self.end_headers()
+
+        if model == "mock-slow":
+            time.sleep(30)
+
+        base = {
+            "id": "chatcmpl-mock",
+            "object": "chat.completion.chunk",
+            "created": int(time.time()),
+            "model": model,
+        }
+        for piece in CHUNKS:
+            event = dict(base, choices=[{
+                "index": 0,
+                "delta": {"content": piece},
+                "finish_reason": None,
+            }])
+            self._chunk(("data: " + json.dumps(event, ensure_ascii=False) + "\n\n").encode("utf-8"))
+
+        final = dict(
+            base,
+            choices=[{"index": 0, "delta": {}, "finish_reason": "stop"}],
+            usage={
+                "prompt_tokens": PROMPT_TOKENS,
+                "completion_tokens": len(CHUNKS),
+                "total_tokens": PROMPT_TOKENS + len(CHUNKS),
+            },
+        )
+        self._chunk(("data: " + json.dumps(final, ensure_ascii=False) + "\n\n").encode("utf-8"))
+        self._chunk(b"data: [DONE]\n\n")
+        self.wfile.write(b"0\r\n\r\n")
+        self.wfile.flush()
+
+
+def main():
+    port = int(sys.argv[1]) if len(sys.argv) > 1 else 18999
+    server = ThreadingHTTPServer(("127.0.0.1", port), Handler)
+    sys.stderr.write("[mock-openai] listening on http://127.0.0.1:%d\n" % port)
+    sys.stderr.flush()
+    server.serve_forever()
+
+
+if __name__ == "__main__":
+    main()