feat(harness): RAG 忠实度评测 —— judge 拿检索原文评幻觉
补 Harness 已知洞:此前 LLM-judge 只看 input+output,看不到检索来源,幻觉其实没评。 - Evaluator.Score 增 sources 参数;有来源时走 llmJudgeGrounded:一次调用同时评 quality 质量 + faithfulness 忠实度,并列出 unsupported(未被来源支持的说法)→ 进 Flags。 综合分(有来源)=0.3规则+0.35质量+0.35忠实;无来源时维持原 0.4规则+0.6质量。Result 增 Faithful 字段。 - 透传检索来源:runGraph 返回 (answer, refs, err),executeGraph 同步;Handle→evaluate(input,output,refs); refsOf(board)=检索资料+工具产出。compose 路径暂返回 nil refs(不评忠实度)。 - eval 日志增「忠实 X.XX,来源 N」。 测试:单测覆盖 grounded(quality/faithfulness/unsupported 解析 + 加权 + flags)与无来源跳过; 3 处测试 runGraph 三返回值更新。live 实测 RAG 任务忠实 1.00/来源 1,judge 正确判定无编造。 project_analysis Harness 清单勾掉该项。 Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -7,13 +7,14 @@ import (
|
||||
"strings"
|
||||
)
|
||||
|
||||
// Result 是一次输出评测的结果。Overall ∈ [0,1]。
|
||||
// Result 是一次输出评测的结果。各分 ∈ [0,1]。
|
||||
type Result struct {
|
||||
Overall float64 // 综合分(有 LLM 评审则 0.4*规则 + 0.6*LLM,否则=规则分)
|
||||
Rule float64 // 规则分
|
||||
LLM float64 // LLM-as-judge 分(0 表示未评/失败)
|
||||
Flags []string // 命中的规则问题
|
||||
Reason string // LLM 评语
|
||||
Overall float64 // 综合分(见 Score 的加权)
|
||||
Rule float64 // 规则分
|
||||
LLM float64 // LLM-as-judge 质量分(0=未评/失败)
|
||||
Faithful float64 // RAG 忠实度分(有检索来源时才评;0=未评/无来源)
|
||||
Flags []string // 命中的问题(规则 + 未被来源支持的说法)
|
||||
Reason string // LLM 评语
|
||||
}
|
||||
|
||||
// Evaluator 实现 LLM 自动化评测:规则检查(快、always-on)+ LLM-as-judge(模型就绪时)。
|
||||
@@ -28,16 +29,30 @@ func NewEvaluator(ready func() bool, chat func(ctx context.Context, sys, user st
|
||||
return &Evaluator{ready: ready, chat: chat}
|
||||
}
|
||||
|
||||
// Score 对一次推理输出综合打分。
|
||||
func (e *Evaluator) Score(ctx context.Context, input, output string) Result {
|
||||
// Score 对一次推理输出综合打分。sources 为检索到的资料(RAG 来源):非空时额外评忠实度
|
||||
// (回答是否基于来源、有无编造),并把"未被来源支持的说法"加进 Flags。
|
||||
func (e *Evaluator) Score(ctx context.Context, input, output string, sources []string) Result {
|
||||
rule, flags := ruleScore(output)
|
||||
res := Result{Rule: rule, Flags: flags, Overall: rule}
|
||||
if e.ready != nil && e.chat != nil && e.ready() {
|
||||
if s, reason, ok := e.llmJudge(ctx, input, output); ok {
|
||||
res.LLM = s
|
||||
res.Reason = reason
|
||||
res.Overall = 0.4*rule + 0.6*s
|
||||
if e.ready == nil || e.chat == nil || !e.ready() {
|
||||
return res
|
||||
}
|
||||
if len(sources) > 0 {
|
||||
// RAG 路径:一次 judge 同时评质量 + 忠实度 + 列未支持说法。
|
||||
if q, f, unsupported, reason, ok := e.llmJudgeGrounded(ctx, input, output, sources); ok {
|
||||
res.LLM, res.Faithful, res.Reason = q, f, reason
|
||||
res.Overall = 0.3*rule + 0.35*q + 0.35*f
|
||||
for _, u := range unsupported {
|
||||
res.Flags = append(res.Flags, "未被来源支持:"+u)
|
||||
}
|
||||
}
|
||||
return res
|
||||
}
|
||||
// 无来源:仅评质量(相关/准确/完整)。
|
||||
if s, reason, ok := e.llmJudge(ctx, input, output); ok {
|
||||
res.LLM = s
|
||||
res.Reason = reason
|
||||
res.Overall = 0.4*rule + 0.6*s
|
||||
}
|
||||
return res
|
||||
}
|
||||
@@ -114,6 +129,40 @@ func (e *Evaluator) llmJudge(ctx context.Context, input, output string) (float64
|
||||
return s, j.Reason, true
|
||||
}
|
||||
|
||||
// llmJudgeGrounded 给 judge 同时喂用户问题、检索资料、模型回答,一次返回:
|
||||
// quality 质量分、faithful 忠实度分(均归一到 [0,1])、unsupported 未被资料支持的说法、reason 评语。
|
||||
func (e *Evaluator) llmJudgeGrounded(ctx context.Context, input, output string, sources []string) (quality, faithful float64, unsupported []string, reason string, ok bool) {
|
||||
sys := "你是严格的 RAG 回答评审,重点核查回答是否严格基于检索资料、有无编造(幻觉)。"
|
||||
src := evalTruncate(strings.Join(sources, "\n---\n"), 3000)
|
||||
user := fmt.Sprintf(
|
||||
"【用户问题】%s\n\n【检索资料】\n%s\n\n【模型回答】%s\n\n"+
|
||||
"请评估两项:quality=回答质量(相关/准确/完整),faithfulness=忠实度(回答是否严格基于检索资料、有无编造)。"+
|
||||
"并列出 unsupported:回答中未被检索资料支持的具体说法(无则空数组)。"+
|
||||
"只输出 JSON:{\"quality\":1到5整数,\"faithfulness\":1到5整数,\"unsupported\":[\"...\"],\"reason\":\"一句话\"},不要多余文字。",
|
||||
evalTruncate(input, 400), src, evalTruncate(output, 1500))
|
||||
txt, err := e.chat(ctx, sys, user)
|
||||
if err != nil {
|
||||
return 0, 0, nil, "", false
|
||||
}
|
||||
var j struct {
|
||||
Quality float64 `json:"quality"`
|
||||
Faithfulness float64 `json:"faithfulness"`
|
||||
Unsupported []string `json:"unsupported"`
|
||||
Reason string `json:"reason"`
|
||||
}
|
||||
if json.Unmarshal([]byte(evalStripFence(txt)), &j) != nil || j.Quality <= 0 || j.Faithfulness <= 0 {
|
||||
return 0, 0, nil, "", false
|
||||
}
|
||||
norm := func(v float64) float64 {
|
||||
s := v / 5.0
|
||||
if s > 1 {
|
||||
s = 1
|
||||
}
|
||||
return s
|
||||
}
|
||||
return norm(j.Quality), norm(j.Faithfulness), j.Unsupported, j.Reason, true
|
||||
}
|
||||
|
||||
func evalTruncate(s string, n int) string {
|
||||
r := []rune(s)
|
||||
if len(r) <= n {
|
||||
|
||||
@@ -8,11 +8,11 @@ import (
|
||||
|
||||
func TestRuleScore(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
output string
|
||||
wantMax float64 // 期望分 ≤ 此值
|
||||
wantMin float64 // 期望分 ≥ 此值
|
||||
wantFlag string // 期望命中的标签(空=不校验)
|
||||
name string
|
||||
output string
|
||||
wantMax float64 // 期望分 ≤ 此值
|
||||
wantMin float64 // 期望分 ≥ 此值
|
||||
wantFlag string // 期望命中的标签(空=不校验)
|
||||
}{
|
||||
{"正常", "杭州是浙江省会,历史悠久,有西湖等名胜,是著名的旅游与电商之城。", 1.0, 1.0, ""},
|
||||
{"空", " ", 0, 0, "空输出"},
|
||||
@@ -40,7 +40,7 @@ func TestRuleScore_HeavyRepeat(t *testing.T) {
|
||||
|
||||
func TestScore_RuleOnly(t *testing.T) {
|
||||
e := NewEvaluator(nil, nil) // 无 LLM → 仅规则
|
||||
r := e.Score(context.Background(), "问题", "一段质量不错的较完整回答内容,长度足够,没有任何问题。")
|
||||
r := e.Score(context.Background(), "问题", "一段质量不错的较完整回答内容,长度足够,没有任何问题。", nil)
|
||||
if r.LLM != 0 || r.Overall != r.Rule {
|
||||
t.Errorf("无 LLM 时 Overall 应等于规则分, got overall=%.2f rule=%.2f llm=%.2f", r.Overall, r.Rule, r.LLM)
|
||||
}
|
||||
@@ -54,7 +54,7 @@ func TestScore_WithLLMJudge(t *testing.T) {
|
||||
return "```json\n{\"score\":4,\"reason\":\"相关且较完整\"}\n```", nil
|
||||
},
|
||||
)
|
||||
r := e.Score(context.Background(), "介绍杭州", "杭州是浙江省会,西湖闻名,历史与现代交融,电商发达。")
|
||||
r := e.Score(context.Background(), "介绍杭州", "杭州是浙江省会,西湖闻名,历史与现代交融,电商发达。", nil)
|
||||
if r.LLM <= 0 {
|
||||
t.Fatalf("应有 LLM 分, got %.2f", r.LLM)
|
||||
}
|
||||
@@ -75,12 +75,56 @@ func TestScore_LLMJudgeBadJSONFallsBack(t *testing.T) {
|
||||
func() bool { return true },
|
||||
func(ctx context.Context, sys, user string) (string, error) { return "我觉得还行吧", nil },
|
||||
)
|
||||
r := e.Score(context.Background(), "q", "一段足够长且正常的回答内容用于评测。")
|
||||
r := e.Score(context.Background(), "q", "一段足够长且正常的回答内容用于评测。", nil)
|
||||
if r.LLM != 0 || r.Overall != r.Rule {
|
||||
t.Errorf("LLM 返回非 JSON 应回退到规则分, got overall=%.2f llm=%.2f", r.Overall, r.LLM)
|
||||
}
|
||||
}
|
||||
|
||||
func TestScore_GroundedFaithfulness(t *testing.T) {
|
||||
// 有检索来源 → 走忠实度评测:judge 返回 quality/faithfulness/unsupported。
|
||||
var gotPrompt string
|
||||
e := NewEvaluator(
|
||||
func() bool { return true },
|
||||
func(ctx context.Context, sys, user string) (string, error) {
|
||||
gotPrompt = user
|
||||
return `{"quality":4,"faithfulness":2,"unsupported":["该产品支持离线模式"],"reason":"部分说法无资料支撑"}`, nil
|
||||
},
|
||||
)
|
||||
r := e.Score(context.Background(), "这产品支持什么", "它支持在线与离线模式。",
|
||||
[]string{"产品支持在线协作。", "产品提供云端存储。"})
|
||||
|
||||
if !strings.Contains(gotPrompt, "检索资料") || !strings.Contains(gotPrompt, "产品支持在线协作") {
|
||||
t.Fatalf("judge 提示词应包含检索资料, got: %q", gotPrompt)
|
||||
}
|
||||
if r.LLM < 0.79 || r.LLM > 0.81 { // quality 4/5
|
||||
t.Errorf("quality 应为 0.8, got %.2f", r.LLM)
|
||||
}
|
||||
if r.Faithful < 0.39 || r.Faithful > 0.41 { // faithfulness 2/5
|
||||
t.Errorf("忠实度应为 0.4, got %.2f", r.Faithful)
|
||||
}
|
||||
if !contains(r.Flags, "未被来源支持:该产品支持离线模式") {
|
||||
t.Errorf("未支持说法应进 Flags, got %v", r.Flags)
|
||||
}
|
||||
want := 0.3*r.Rule + 0.35*r.LLM + 0.35*r.Faithful
|
||||
if r.Overall < want-1e-9 || r.Overall > want+1e-9 {
|
||||
t.Errorf("综合分应为 0.3规则+0.35质量+0.35忠实=%.3f, got %.3f", want, r.Overall)
|
||||
}
|
||||
}
|
||||
|
||||
func TestScore_NoSourcesSkipsFaithfulness(t *testing.T) {
|
||||
e := NewEvaluator(
|
||||
func() bool { return true },
|
||||
func(ctx context.Context, sys, user string) (string, error) {
|
||||
return `{"score":5,"reason":"好"}`, nil
|
||||
},
|
||||
)
|
||||
r := e.Score(context.Background(), "q", "一段足够长且正常的回答内容用于评测。", nil)
|
||||
if r.Faithful != 0 {
|
||||
t.Errorf("无来源不应评忠实度, got %.2f", r.Faithful)
|
||||
}
|
||||
}
|
||||
|
||||
func contains(ss []string, want string) bool {
|
||||
for _, s := range ss {
|
||||
if s == want {
|
||||
|
||||
Reference in New Issue
Block a user