feat(harness): RAG 忠实度评测 —— judge 拿检索原文评幻觉

补 Harness 已知洞:此前 LLM-judge 只看 input+output,看不到检索来源,幻觉其实没评。

- Evaluator.Score 增 sources 参数;有来源时走 llmJudgeGrounded:一次调用同时评 quality 质量 +
  faithfulness 忠实度,并列出 unsupported(未被来源支持的说法)→ 进 Flags。
  综合分(有来源)=0.3规则+0.35质量+0.35忠实;无来源时维持原 0.4规则+0.6质量。Result 增 Faithful 字段。
- 透传检索来源:runGraph 返回 (answer, refs, err),executeGraph 同步;Handle→evaluate(input,output,refs);
  refsOf(board)=检索资料+工具产出。compose 路径暂返回 nil refs(不评忠实度)。
- eval 日志增「忠实 X.XX,来源 N」。

测试:单测覆盖 grounded(quality/faithfulness/unsupported 解析 + 加权 + flags)与无来源跳过;
3 处测试 runGraph 三返回值更新。live 实测 RAG 任务忠实 1.00/来源 1,judge 正确判定无编造。
project_analysis Harness 清单勾掉该项。

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
Blizzard
2026-06-25 15:03:17 +08:00
parent 82b1d3e802
commit 446b784fc6
9 changed files with 148 additions and 44 deletions
@@ -33,7 +33,7 @@ func TestAgentCollaborationPassesOutput(t *testing.T) {
return "研究产出XYZ"
}}
o := &Orchestrator{pool: ll, breaker: harness.NewCircuitBreaker(), sink: &fakeSink{}}
ans, err := o.runGraph(context.Background(), &contract.Task{ID: "tc", Graph: []byte(graph)}, &execTracer{})
ans, _, err := o.runGraph(context.Background(), &contract.Task{ID: "tc", Graph: []byte(graph)}, &execTracer{})
if err != nil {
t.Fatal(err)
}
@@ -24,9 +24,11 @@ func registerFlowMerge() {
}
// executeGraph 按灰度开关选编排实现:compose.GraphPhase C)或自研 graph.go(默认/权威)。
func (o *Orchestrator) executeGraph(ctx context.Context, t *contract.Task, tr *execTracer) (string, error) {
// 返回 (成稿, 检索来源, error);来源供忠实度评测(compose 路径暂不提供来源 → 返回 nil)。
func (o *Orchestrator) executeGraph(ctx context.Context, t *contract.Task, tr *execTracer) (string, []string, error) {
if composeEnabled() {
return o.runComposeGraph(ctx, t, tr)
ans, err := o.runComposeGraph(ctx, t, tr)
return ans, nil, err
}
return o.runGraph(ctx, t, tr)
}
@@ -159,7 +161,8 @@ func (o *Orchestrator) runComposeGraph(ctx context.Context, t *contract.Task, tr
r, cerr := g.Compile(ctx, compose.WithNodeTriggerMode(compose.AllPredecessor))
if cerr != nil {
tr.info("task", "system", "compose 编译失败", "退回自研 graph.go"+cerr.Error())
return o.runGraph(ctx, t, tr)
ans, _, gerr := o.runGraph(ctx, t, tr) // 降级路径丢弃 refs(compose 路径暂不评忠实度)
return ans, gerr
}
if _, ierr := r.Invoke(ctx, flowSignal{}); ierr != nil {
tr.info("task", "system", "compose 执行告警", ierr.Error()) // 副作用已落 board;下方按需补一段答复
@@ -29,7 +29,7 @@ func runBoth(t *testing.T, graph string) (interp, comp string) {
task := &contract.Task{ID: "t_eq", Graph: []byte(graph)}
o1 := &Orchestrator{pool: echoLLM(), breaker: harness.NewCircuitBreaker(), sink: &fakeSink{}}
a1, err := o1.runGraph(context.Background(), task, &execTracer{})
a1, _, err := o1.runGraph(context.Background(), task, &execTracer{})
if err != nil {
t.Fatalf("runGraph: %v", err)
}
+12 -6
View File
@@ -42,7 +42,7 @@ type board struct {
// branch 按条件只激活选中的下游(剪枝)→ agent 节点流式回流 token。
//
// 逐节点点亮"运行·观测"。返回终端 agent 的完整产出(供写回历史)。
func (o *Orchestrator) runGraph(ctx context.Context, t *contract.Task, tr *execTracer) (string, error) {
func (o *Orchestrator) runGraph(ctx context.Context, t *contract.Task, tr *execTracer) (string, []string, error) {
flow, ferr := dsl.Parse(t.Graph)
plan := dsl.Compile(t.Graph)
b := &board{
@@ -57,7 +57,7 @@ func (o *Orchestrator) runGraph(ctx context.Context, t *contract.Task, tr *execT
b.profile = o.fetchMemory(ctx, b.uid, b.query)
b.history = o.fetchHistory(ctx, b.sid)
o.runConversation(ctx, t.ID, b, plan.System, tr, "agent")
return b.answer, b.fatalErr // 模型失败 → 上抛判 failed
return b.answer, refsOf(b), b.fatalErr // 模型失败 → 上抛判 failed
}
// 建邻接与入度(只认两端都存在的边)。保留整条边以便 branch 按 true/false 标签选路。
@@ -165,10 +165,10 @@ func (o *Orchestrator) runGraph(ctx context.Context, t *contract.Task, tr *execT
}
if b.rejected {
return b.answer, errRejected // 合法终态,Handle 据此判 rejected 并优雅收尾
return b.answer, nil, errRejected // 合法终态,Handle 据此判 rejected 并优雅收尾
}
if b.fatalErr != nil {
return b.answer, b.fatalErr // 上抛 → Handle 判 failed(带原因)
return b.answer, nil, b.fatalErr // 上抛 → Handle 判 failed(带原因)
}
// 图里无 agent 节点(纯工具/检索图)也要出一段模型答复,否则没有输出。
@@ -176,9 +176,15 @@ func (o *Orchestrator) runGraph(ctx context.Context, t *contract.Task, tr *execT
o.runConversation(ctx, t.ID, b, plan.System, tr, "agent")
}
if b.fatalErr != nil {
return b.answer, b.fatalErr
return b.answer, nil, b.fatalErr
}
return b.answer, nil
return b.answer, refsOf(b), nil // 成功:带回检索来源供忠实度评测
}
// refsOf 汇总本次执行的检索来源(检索资料 + 工具产出),供忠实度评测。
func refsOf(b *board) []string {
out := append([]string{}, b.refs...)
return append(out, b.toolOut...)
}
// retrieverNode 执行检索节点:kb 按 owner 作用域 → kb_search → 累计参考资料。
@@ -134,7 +134,7 @@ func TestRunGraph_BranchRouting(t *testing.T) {
ll := &fakeLLM{ready: true, stream: func(m []llm.ChatMessage) string { return m[0].Content }}
run := func(cond string) string {
o := newOrch(ll, &fakeTools{}, &fakeSink{}, &fakeExec{})
ans, err := o.runGraph(context.Background(), task(strings.Replace(g, "%s", cond, 1)), o.tracer("t1"))
ans, _, err := o.runGraph(context.Background(), task(strings.Replace(g, "%s", cond, 1)), o.tracer("t1"))
if err != nil {
t.Fatal(err)
}
@@ -160,7 +160,7 @@ func TestRunGraph_ToolFeedsAgent(t *testing.T) {
return &contract.ToolResult{OK: true, Content: "TOOLDATA"}
}}
o := newOrch(ll, ft, &fakeSink{}, &fakeExec{})
ans, err := o.runGraph(context.Background(), task(g), o.tracer("t1"))
ans, _, err := o.runGraph(context.Background(), task(g), o.tracer("t1"))
if err != nil {
t.Fatal(err)
}
@@ -186,7 +186,7 @@ func TestRunGraph_MapFanout(t *testing.T) {
return "正文XYZ", nil
}}
o := newOrch(ll, &fakeTools{}, &fakeSink{}, &fakeExec{})
ans, err := o.runGraph(context.Background(), task(g), o.tracer("t1"))
ans, _, err := o.runGraph(context.Background(), task(g), o.tracer("t1"))
if err != nil {
t.Fatal(err)
}
@@ -203,7 +203,7 @@ func TestRunGraph_OutputRedaction(t *testing.T) {
}}
fs := &fakeSink{}
o := newOrch(ll, &fakeTools{}, fs, &fakeExec{})
if _, err := o.runGraph(context.Background(), task(g), o.tracer("t1")); err != nil {
if _, _, err := o.runGraph(context.Background(), task(g), o.tracer("t1")); err != nil {
t.Fatal(err)
}
out := fs.text()
@@ -158,7 +158,7 @@ func (o *Orchestrator) Handle(ctx context.Context, t *contract.Task) error {
tr.info("task", "system", "任务受理", fmt.Sprintf("DSL %d 字节,按图执行", len(t.Graph)))
// 按 DSL 图执行:compose.GraphEINO_COMPOSE=1)或自研 graph.go(默认);agent 节点流式回流 token。
answer, err := o.executeGraph(tctx, t, tr)
answer, refs, err := o.executeGraph(tctx, t, tr)
if errors.Is(err, errRejected) {
// HITL 拒绝:合法终态,非故障。收尾流 + 置 rejected,不计熔断、不重投。
slog.InfoContext(ctx, "task rejected by approval", "task_id", t.ID)
@@ -189,21 +189,22 @@ func (o *Orchestrator) Handle(ctx context.Context, t *contract.Task) error {
// 写回阶段:离开热路径、异步落历史 + (TODO)抽取记忆。
go o.memorize(t, answer)
// 自动化评测:离开热路径,对本轮输出打分并记录(规则 + LLM-as-judge)。
go o.evaluate(t, dsl.Compile(t.Graph).Query, answer)
// 自动化评测:离开热路径,对本轮输出打分并记录(规则 + LLM-as-judge + RAG 忠实度)。
go o.evaluate(t, dsl.Compile(t.Graph).Query, answer, refs)
return nil
}
// evaluate 异步对一次输出做自动化评测并记录评分(off 热路径,不影响响应)。
func (o *Orchestrator) evaluate(t *contract.Task, input, output string) {
// sources 为本轮检索来源:非空时额外评忠实度(幻觉检测)。
func (o *Orchestrator) evaluate(t *contract.Task, input, output string, sources []string) {
if o.eval == nil {
return
}
ctx, cancel := context.WithTimeout(context.Background(), 15*time.Second)
defer cancel()
r := o.eval.Score(ctx, input, output)
log.Printf("[eval] task %s 综合 %.2f(规则 %.2f / LLM %.2fflags=%v %s",
t.ID, r.Overall, r.Rule, r.LLM, r.Flags, r.Reason)
r := o.eval.Score(ctx, input, output, sources)
log.Printf("[eval] task %s 综合 %.2f(规则 %.2f / LLM %.2f / 忠实 %.2f,来源 %dflags=%v %s",
t.ID, r.Overall, r.Rule, r.LLM, r.Faithful, len(sources), r.Flags, r.Reason)
}
// fetchMemory 经 MCP memory_get 工具召回用户常驻画像。