|
|
@@ -6,6 +6,7 @@ import com.baomidou.mybatisplus.core.toolkit.Wrappers;
|
|
|
import com.zsjz.ai.common.base.BasicColumn;
|
|
|
import com.zsjz.ai.common.cache.GlobalCache;
|
|
|
import com.zsjz.ai.common.enums.FileCategoryEnum;
|
|
|
+import com.zsjz.ai.common.utils.Json;
|
|
|
import com.zsjz.ai.common.model.dm.dto.TableRuleDTO;
|
|
|
import com.zsjz.ai.common.model.dm.entity.FileInfo;
|
|
|
import com.zsjz.ai.common.model.plat.entity.TableInfo;
|
|
|
@@ -34,6 +35,7 @@ import com.zsjz.ai.module.aiclean.rule.AiRulePlan;
|
|
|
import com.zsjz.ai.module.aiclean.store.AiCleanStore;
|
|
|
import com.zsjz.ai.module.aiclean.verify.AiRuleVerifier;
|
|
|
import com.zsjz.ai.module.dm.mapper.FileInfoMapper;
|
|
|
+import com.zsjz.ai.module.agent.llm.LlmResult;
|
|
|
import lombok.RequiredArgsConstructor;
|
|
|
import lombok.extern.slf4j.Slf4j;
|
|
|
import org.springframework.stereotype.Service;
|
|
|
@@ -219,18 +221,29 @@ public class AiCleanService {
|
|
|
if (root == null || StrUtil.isBlank(root.getFilePath()) || !new File(root.getFilePath()).exists()) {
|
|
|
return new AiJudgeResult.Skip("源文件不存在或已被清理");
|
|
|
}
|
|
|
+ mark(job, AiCleanStatusEnum.PROBING);
|
|
|
SheetEvidence ev = SheetProbe.probe(job.getCaseId(), job.getFileId(), job.getSheetId(),
|
|
|
root.getFileName(), sheet.getSheetName(), new File(root.getFilePath()));
|
|
|
if (ev == null) {
|
|
|
return new AiJudgeResult.Skip("结构探测不可用");
|
|
|
}
|
|
|
+ mark(job, AiCleanStatusEnum.MATCHING);
|
|
|
List<String> headers = ev.primaryBlock().headers();
|
|
|
List<AiTemplateMatcher.AiTemplateWithFields> online = loadOnlineTemplates();
|
|
|
AiMatchOutcome outcome = matcher.match(ev, online, true);
|
|
|
job.setHeaderMd5(ev.headerMd5());
|
|
|
+ job.setHeaderRow(ev.primaryBlock().headerRow());
|
|
|
job.setMatchedBy(outcome.matchedBy());
|
|
|
job.setMatchTier(outcome.tier().name());
|
|
|
- job.setLlmCalls(outcome.llmCalled() ? 1 : 0);
|
|
|
+ // 用量遥测:模型名/token 随判定写回 job(finish 时落库)—— 成本看板的原料(评审 P1-5)
|
|
|
+ int llmCalls = outcome.llmCalled() ? 1 : 0;
|
|
|
+ LlmResult llm = outcome.llm();
|
|
|
+ if (llm != null) {
|
|
|
+ job.setModelName(llm.modelName());
|
|
|
+ job.setTokenIn(llm.inputTokens());
|
|
|
+ job.setTokenOut(llm.outputTokens());
|
|
|
+ }
|
|
|
+ job.setLlmCalls(llmCalls);
|
|
|
if (outcome.candidate() == null) {
|
|
|
return new AiJudgeResult.NeedReview("没有合适的模板(含 AI 模板库)");
|
|
|
}
|
|
|
@@ -240,23 +253,40 @@ public class AiCleanService {
|
|
|
}
|
|
|
// 选模板那次调用已经把列绑定一起返回了,复用它:单次模型调用实测就要几十秒,
|
|
|
// 为了拿 binds 再打一次等于把延迟和 token 直接翻倍
|
|
|
- AiRuleDraft draft = outcome.draft() != null ? outcome.draft()
|
|
|
+ AiTemplateMatcher.LlmAnswer answer = outcome.draft() != null
|
|
|
+ ? new AiTemplateMatcher.LlmAnswer(outcome.draft(), outcome.llm())
|
|
|
: matcher.askLlm(ev, headers, List.of(outcome.candidate()));
|
|
|
+ AiRuleDraft draft = answer == null ? null : answer.draft();
|
|
|
if (draft == null) {
|
|
|
- job.setLlmCalls(1);
|
|
|
+ // 调用发生但没拿到草稿(含重试耗尽):这次尝试也计入调用数
|
|
|
+ job.setLlmCalls(llmCalls + 1);
|
|
|
return new AiJudgeResult.Failed("模型未返回可解析的规则草稿");
|
|
|
}
|
|
|
- job.setLlmCalls(outcome.draft() != null ? 1 : 2);
|
|
|
+ if (outcome.draft() == null) {
|
|
|
+ // 第二次模型调用(腿 1/2 确定性命中但没有 binds 的场景):次数与用量都要补记
|
|
|
+ llmCalls++;
|
|
|
+ if (answer.usage() != null) {
|
|
|
+ job.setModelName(answer.usage().modelName());
|
|
|
+ job.setTokenIn(nzToken(job.getTokenIn()) + answer.usage().inputTokens());
|
|
|
+ job.setTokenOut(nzToken(job.getTokenOut()) + answer.usage().outputTokens());
|
|
|
+ }
|
|
|
+ }
|
|
|
+ job.setLlmCalls(llmCalls);
|
|
|
+ // 模型原始草稿落库:复盘「哪一版 prompt 判的」以及给后续提示词回归集当语料(评审 P1-5)
|
|
|
+ job.setDraft(StrUtil.maxLength(Json.toStr(draft), 4000));
|
|
|
List<RuleBind> binds = AiTemplateMatcher.filterBinds(draft.getBinds(), outcome.candidate(),
|
|
|
headers.size());
|
|
|
if (binds.isEmpty()) {
|
|
|
return new AiJudgeResult.NeedReview("模型未给出任何可用列绑定");
|
|
|
}
|
|
|
|
|
|
+ mark(job, AiCleanStatusEnum.VERIFYING);
|
|
|
List<Map<Integer, String>> sample = AiRowReader.read(new File(root.getFilePath()),
|
|
|
sheet.getSheetName(), ev.primaryBlock().headerRow(), SAMPLE);
|
|
|
VerifyReport report = verify(binds, outcome.candidate(), headers, sample, outcome);
|
|
|
job.setConfidence(java.math.BigDecimal.valueOf(report.confidence().overall()));
|
|
|
+ // 抽样报告落库(预览行截到 5 行防膨胀):排障时能看到「哪个列错误率多少、为什么被拒」
|
|
|
+ job.setDetail(detailJson(report));
|
|
|
if (report.rejectedCols() != null && !report.rejectedCols().isEmpty()) {
|
|
|
binds = binds.stream().filter(b -> !report.rejectedCols().contains(b.getFileColIndex())).toList();
|
|
|
}
|
|
|
@@ -309,37 +339,29 @@ public class AiCleanService {
|
|
|
/**
|
|
|
* 出口 B:建 AI 模板 + 动态表并写数。
|
|
|
*
|
|
|
+ * <p><b>流式管道</b>(评审 P0-1):Fesod 逐行回调里边清洗边写 DuckDB(攒批 flush),
|
|
|
+ * 不再把整个 sheet collect 成两份 Map 驻留堆内 —— 几十万行的流水表以前会在这里打爆内存。
|
|
|
+ *
|
|
|
+ * <p><b>写序与 {@link AiCleanStore} 的注释对齐</b>(评审发现的幽灵模板问题):
|
|
|
+ * 模板 id 先行生成(物理表名 {@code ai_t{id}} 依赖它)→ 建表 → 流式写数 →
|
|
|
+ * <b>成功才登记 ONLINE</b>。写数失败/零行就不登记,并删掉刚建的空表 ——
|
|
|
+ * 绝不留下「有元数据没物理表」的模板占着 {@code header_md5} 唯一键。
|
|
|
+ *
|
|
|
* <p>公开给融合链路用 —— 它自己决定何时刷批次计数,这里只负责「把这块数据落进 AI 表」
|
|
|
* 并把结果写回 job 的字段(不 finish,由调用方收尾)。
|
|
|
*
|
|
|
* @return 写入行数;失败返回 -1
|
|
|
*/
|
|
|
public long loadViaAiTemplate(AiCleanJob job, FileInfo root, FileInfo sheet, AiJudgeResult.AiPlan plan) {
|
|
|
+ mark(job, AiCleanStatusEnum.LOADING);
|
|
|
SheetEvidence ev = plan.ev();
|
|
|
List<RuleBind> binds = plan.binds();
|
|
|
List<String> headers = plan.headers();
|
|
|
AiRulePlan.Plan aiPlan = AiRulePlan.forAi(binds, headers, "ai_t");
|
|
|
- List<Map<String, Object>> rows = new ArrayList<>();
|
|
|
- for (Map<Integer, String> r : AiRowReader.read(new File(root.getFilePath()),
|
|
|
- sheet.getSheetName(), ev.primaryBlock().headerRow(), 0)) {
|
|
|
- Map<String, String> errors = new LinkedHashMap<>();
|
|
|
- Map<String, Object> cleaned = AiRecordCleaner.cleanRow(r, aiPlan.columns(), errors);
|
|
|
- if (cleaned == null) {
|
|
|
- continue;
|
|
|
- }
|
|
|
- cleaned.put("id", IdUtil.getSnowflakeNextId());
|
|
|
- cleaned.put("case_id", job.getCaseId());
|
|
|
- cleaned.put("file_id", job.getFileId());
|
|
|
- cleaned.put("sheet_id", job.getSheetId());
|
|
|
- cleaned.put("block_no", ev.primaryBlock().blockNo());
|
|
|
- // row_no 存原文件行号,是事后定位与撤销的抓手:读表器已经按表头行截断,这里用序号回填
|
|
|
- cleaned.put("row_no", rows.size() + ev.primaryBlock().headerRow() + 1);
|
|
|
- rows.add(cleaned);
|
|
|
- }
|
|
|
- if (rows.isEmpty()) {
|
|
|
- return -1L;
|
|
|
- }
|
|
|
+ // 模板 id 先行:物理表名依赖它;元数据登记挪到写数成功之后
|
|
|
AiCleanTemplate tpl = new AiCleanTemplate();
|
|
|
+ tpl.setId(IdUtil.getSnowflakeNextId());
|
|
|
+ tpl.setTableNameEn("ai_t" + tpl.getId());
|
|
|
tpl.setTableNameCn(StrUtil.blankToDefault(plan.draft().getReason(),
|
|
|
plan.candidate() == null ? "AI 识别表" : plan.candidate().nameCn()));
|
|
|
tpl.setHeaders(String.join(",", headers));
|
|
|
@@ -348,7 +370,6 @@ public class AiCleanService {
|
|
|
tpl.setBlockNo(ev.primaryBlock().blockNo());
|
|
|
tpl.setCategory(FileCategoryEnum.parse(plan.draft().getCategory()).name());
|
|
|
tpl.setMatchTier(job.getMatchTier());
|
|
|
- tpl.setStatus("ONLINE");
|
|
|
tpl.setVer(1);
|
|
|
tpl.setNeedsStruct(0);
|
|
|
tpl.setCaseIdSrc(job.getCaseId());
|
|
|
@@ -363,16 +384,56 @@ public class AiCleanService {
|
|
|
r.setVer(1);
|
|
|
ruleRows.add(r);
|
|
|
}
|
|
|
- Long tplId = store.register(tpl, aiPlan.fields(), ruleRows);
|
|
|
- long written = store.writeRows(tpl, aiPlan.fields(), rows, job.getId());
|
|
|
+ String err = AiDuckDb.createAiTable(tpl.getTableNameEn(), aiPlan.fields());
|
|
|
+ if (err != null) {
|
|
|
+ log.warn("AI 动态表建表失败,跳过写入: table={}, err={}", tpl.getTableNameEn(), err);
|
|
|
+ return -1L;
|
|
|
+ }
|
|
|
+ List<String> cols = aiPlan.fields().stream().map(AiCleanField::getColumnName).toList();
|
|
|
+ long written;
|
|
|
+ java.util.concurrent.atomic.AtomicLong total = new java.util.concurrent.atomic.AtomicLong();
|
|
|
+ try (AiDuckDb.BatchWriter writer = AiDuckDb.openWriter(tpl.getTableNameEn(), cols)) {
|
|
|
+ boolean ok = AiRowReader.forEachRow(new File(root.getFilePath()), sheet.getSheetName(),
|
|
|
+ ev.primaryBlock().headerRow(), (row, lineNo) -> {
|
|
|
+ Map<String, String> errors = new LinkedHashMap<>();
|
|
|
+ Map<String, Object> cleaned = AiRecordCleaner.cleanRow(row, aiPlan.columns(), errors);
|
|
|
+ if (cleaned == null) {
|
|
|
+ return;
|
|
|
+ }
|
|
|
+ cleaned.put("id", IdUtil.getSnowflakeNextId());
|
|
|
+ cleaned.put("case_id", job.getCaseId());
|
|
|
+ cleaned.put("file_id", job.getFileId());
|
|
|
+ cleaned.put("sheet_id", job.getSheetId());
|
|
|
+ cleaned.put("block_no", ev.primaryBlock().blockNo());
|
|
|
+ // row_no 存真实原文件行号(流式回调按物理行回填),是事后定位与撤销的抓手;
|
|
|
+ // 旧实现用「清洗后序号 + 表头行」近似,sheet 里有空行时对不上号
|
|
|
+ cleaned.put("row_no", lineNo);
|
|
|
+ cleaned.put("ai_job_id", job.getId());
|
|
|
+ cleaned.put("ai_rule_ver", tpl.getVer() == null ? 1 : tpl.getVer());
|
|
|
+ cleaned.put("category", tpl.getCategory());
|
|
|
+ // 写失败会在这里抛出,中断流式读取 —— 半截数据绝不能被当成成功
|
|
|
+ writer.add(cleaned);
|
|
|
+ total.incrementAndGet();
|
|
|
+ });
|
|
|
+ written = ok ? writer.finish() : -1L;
|
|
|
+ } catch (Exception e) {
|
|
|
+ log.warn("AI 流式入库失败: jobId={}, table={}, err={}", job.getId(), tpl.getTableNameEn(), e.getMessage());
|
|
|
+ return -1L;
|
|
|
+ }
|
|
|
if (written <= 0) {
|
|
|
- // 一行没落库不能算成功:全自动链路没有人看日志,写成 0 行会被上层当成清洗完成
|
|
|
+ // 一行没落库不能算成功(含「读到 0 行」):全自动链路没有人看日志,
|
|
|
+ // 写成 0 行会被上层当成清洗完成。空表删掉,模板不登记 —— 不留幽灵
|
|
|
+ AiDuckDb.dropTable(tpl.getTableNameEn());
|
|
|
return -1L;
|
|
|
}
|
|
|
- job.setAiTemplateId(tplId);
|
|
|
+ // 数据已落盘,现在才登记元数据为 ONLINE 并刷新治理节点(写序收口)
|
|
|
+ tpl.setStatus("ONLINE");
|
|
|
+ store.register(tpl, aiPlan.fields(), ruleRows);
|
|
|
+ store.refreshTree();
|
|
|
+ job.setAiTemplateId(tpl.getId());
|
|
|
job.setExitCode(AiCleanExitEnum.B.name());
|
|
|
- job.setRowTotal(rows.size());
|
|
|
- job.setRowLoaded((int) written);
|
|
|
+ job.setRowTotal((int) Math.min(total.get(), Integer.MAX_VALUE));
|
|
|
+ job.setRowLoaded((int) Math.min(written, Integer.MAX_VALUE));
|
|
|
return written;
|
|
|
}
|
|
|
|
|
|
@@ -402,9 +463,16 @@ public class AiCleanService {
|
|
|
cand.score(), total);
|
|
|
}
|
|
|
|
|
|
- /** 有合并区、或表头上方存在标题/说明行时认为需要结构归一 */
|
|
|
+ /**
|
|
|
+ * 主表块内有合并区、或表头上方存在标题/说明行时认为需要结构归一。
|
|
|
+ *
|
|
|
+ * <p>合并区只看<b>主块自己的</b>({@link SheetEvidence#primaryMergeCount()}):
|
|
|
+ * 旧实现用整个 sheet 的合并区数一票否决,远处一个无关合并格就能把整张表打入人工
|
|
|
+ * (评审 P2-7);而 Fesod 读数只会被本块范围内的合并区弄脏,块外的不影响主块。
|
|
|
+ * STRUCTURING 状态留给 P1 的结构归一出口(A2/A3),本版本仍不执行。
|
|
|
+ */
|
|
|
private boolean needsStructure(SheetEvidence ev) {
|
|
|
- return ev.mergeCount() > 0 || ev.primaryBlock().headerRow() > 1;
|
|
|
+ return ev.primaryMergeCount() > 0 || ev.primaryBlock().headerRow() > 1;
|
|
|
}
|
|
|
|
|
|
private List<AiTemplateMatcher.AiTemplateWithFields> loadOnlineTemplates() {
|
|
|
@@ -472,4 +540,39 @@ public class AiCleanService {
|
|
|
job.getId(), status, job.getExitCode(), job.getCostMs(), reason);
|
|
|
return job;
|
|
|
}
|
|
|
+
|
|
|
+ /**
|
|
|
+ * 过程状态落库:让批内能观测「哪张表卡在哪个阶段」——
|
|
|
+ * 以前 PROBING/MATCHING/VERIFYING/LOADING 四个中间状态从未被赋值,
|
|
|
+ * 单表判定(含几十秒的模型调用)期间 job 一直挂在 PENDING(评审 P1-6)。
|
|
|
+ * STRUCTURING 留给 P1 的结构归一出口,本版本不产生。
|
|
|
+ */
|
|
|
+ private void mark(AiCleanJob job, AiCleanStatusEnum status) {
|
|
|
+ job.setStatus(status);
|
|
|
+ job.setUpdateTime(LocalDateTime.now());
|
|
|
+ jobMapper.updateById(job);
|
|
|
+ }
|
|
|
+
|
|
|
+ /**
|
|
|
+ * 抽样报告 → 有界 JSON:预览行只留前 5 行(整份报告是给人复盘的,不是给程序回放的),
|
|
|
+ * 否则 300 行抽样全量进 {@code detail} 列,jobs 列表一次查询会被撑爆。
|
|
|
+ */
|
|
|
+ private static String detailJson(VerifyReport report) {
|
|
|
+ Map<String, Object> detail = new LinkedHashMap<>();
|
|
|
+ detail.put("sampled", report.sampled());
|
|
|
+ detail.put("valid", report.valid());
|
|
|
+ detail.put("invalid", report.invalid());
|
|
|
+ detail.put("estTotalRows", report.estTotalRows());
|
|
|
+ detail.put("perColErrorRate", report.perColErrorRate());
|
|
|
+ detail.put("rejectedCols", report.rejectedCols());
|
|
|
+ detail.put("columnErrors", report.columnErrors());
|
|
|
+ detail.put("confidence", report.confidence());
|
|
|
+ detail.put("previewRows", report.previewRows() == null ? List.of()
|
|
|
+ : report.previewRows().stream().limit(5).toList());
|
|
|
+ return StrUtil.maxLength(Json.toStr(detail), 8000);
|
|
|
+ }
|
|
|
+
|
|
|
+ private static int nzToken(Integer v) {
|
|
|
+ return v == null ? 0 : v;
|
|
|
+ }
|
|
|
}
|