From 65f580c5bc0b6ef97ba6f49480c984d4c5350e25 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 9 Oct 2026 06:15:41 +0000 Subject: [PATCH] Add evaluation module: RelevancyEvaluator and FactCheckingEvaluator, golden dataset with a pass-rate gate, deterministic CI judge and simulated judge noise Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_01JXVi2GMQ7bR5EmbUFdDj7N --- README.md | 1 + evaluation/.gitignore | 1 + evaluation/README.md | 49 ++++++++++ evaluation/output/01-evaluator-prompts.txt | 42 +++++++++ evaluation/output/02-verdict-parsing.txt | 13 +++ evaluation/output/03-golden-run.txt | 37 ++++++++ evaluation/output/04-noisy-judge.txt | 17 ++++ evaluation/output/05-graded-and-composite.txt | 16 ++++ evaluation/output/06-judge-limits.txt | 8 ++ evaluation/output/07-ci-layers.txt | 7 ++ evaluation/output/08-live-judge-skipped.txt | 4 + evaluation/pom.xml | 70 ++++++++++++++ evaluation/scripts/run-all.sh | 16 ++++ .../com/ankurm/evaluation/CaseResult.java | 9 ++ .../ankurm/evaluation/CompositeEvaluator.java | 37 ++++++++ .../evaluation/ContainsFactsEvaluator.java | 30 ++++++ .../com/ankurm/evaluation/EvalReport.java | 22 +++++ .../com/ankurm/evaluation/EvalRunner.java | 38 ++++++++ .../com/ankurm/evaluation/GoldenCase.java | 13 +++ .../com/ankurm/evaluation/GoldenDataset.java | 28 ++++++ .../ankurm/evaluation/GradedEvaluator.java | 53 +++++++++++ .../com/ankurm/evaluation/RuleBasedJudge.java | 77 +++++++++++++++ .../ankurm/evaluation/SupportAssistant.java | 33 +++++++ .../evaluation/VerdictNormalizingModel.java | 33 +++++++ .../com/ankurm/evaluation/CiLayersTest.java | 51 ++++++++++ .../evaluation/EvaluatorAnatomyTest.java | 62 +++++++++++++ .../ankurm/evaluation/GoldenDatasetTest.java | 93 +++++++++++++++++++ .../evaluation/GradedCompositeTest.java | 65 +++++++++++++ .../ankurm/evaluation/JudgeLimitsTest.java | 45 +++++++++ .../com/ankurm/evaluation/LiveJudgeTest.java | 41 ++++++++ .../com/ankurm/evaluation/NoisyJudgeTest.java | 67 +++++++++++++ .../ankurm/evaluation/VerdictParsingTest.java | 64 +++++++++++++ .../ankurm/evaluation/support/Answers.java | 54 +++++++++++ .../ankurm/evaluation/support/FlakyJudge.java | 39 ++++++++ .../evaluation/support/RecordingModel.java | 45 +++++++++ .../ankurm/evaluation/support/Transcript.java | 47 ++++++++++ .../test/resources/golden/support-golden.json | 26 ++++++ 37 files changed, 1353 insertions(+) create mode 100644 evaluation/.gitignore create mode 100644 evaluation/README.md create mode 100644 evaluation/output/01-evaluator-prompts.txt create mode 100644 evaluation/output/02-verdict-parsing.txt create mode 100644 evaluation/output/03-golden-run.txt create mode 100644 evaluation/output/04-noisy-judge.txt create mode 100644 evaluation/output/05-graded-and-composite.txt create mode 100644 evaluation/output/06-judge-limits.txt create mode 100644 evaluation/output/07-ci-layers.txt create mode 100644 evaluation/output/08-live-judge-skipped.txt create mode 100644 evaluation/pom.xml create mode 100755 evaluation/scripts/run-all.sh create mode 100644 evaluation/src/main/java/com/ankurm/evaluation/CaseResult.java create mode 100644 evaluation/src/main/java/com/ankurm/evaluation/CompositeEvaluator.java create mode 100644 evaluation/src/main/java/com/ankurm/evaluation/ContainsFactsEvaluator.java create mode 100644 evaluation/src/main/java/com/ankurm/evaluation/EvalReport.java create mode 100644 evaluation/src/main/java/com/ankurm/evaluation/EvalRunner.java create mode 100644 evaluation/src/main/java/com/ankurm/evaluation/GoldenCase.java create mode 100644 evaluation/src/main/java/com/ankurm/evaluation/GoldenDataset.java create mode 100644 evaluation/src/main/java/com/ankurm/evaluation/GradedEvaluator.java create mode 100644 evaluation/src/main/java/com/ankurm/evaluation/RuleBasedJudge.java create mode 100644 evaluation/src/main/java/com/ankurm/evaluation/SupportAssistant.java create mode 100644 evaluation/src/main/java/com/ankurm/evaluation/VerdictNormalizingModel.java create mode 100644 evaluation/src/test/java/com/ankurm/evaluation/CiLayersTest.java create mode 100644 evaluation/src/test/java/com/ankurm/evaluation/EvaluatorAnatomyTest.java create mode 100644 evaluation/src/test/java/com/ankurm/evaluation/GoldenDatasetTest.java create mode 100644 evaluation/src/test/java/com/ankurm/evaluation/GradedCompositeTest.java create mode 100644 evaluation/src/test/java/com/ankurm/evaluation/JudgeLimitsTest.java create mode 100644 evaluation/src/test/java/com/ankurm/evaluation/LiveJudgeTest.java create mode 100644 evaluation/src/test/java/com/ankurm/evaluation/NoisyJudgeTest.java create mode 100644 evaluation/src/test/java/com/ankurm/evaluation/VerdictParsingTest.java create mode 100644 evaluation/src/test/java/com/ankurm/evaluation/support/Answers.java create mode 100644 evaluation/src/test/java/com/ankurm/evaluation/support/FlakyJudge.java create mode 100644 evaluation/src/test/java/com/ankurm/evaluation/support/RecordingModel.java create mode 100644 evaluation/src/test/java/com/ankurm/evaluation/support/Transcript.java create mode 100644 evaluation/src/test/resources/golden/support-golden.json diff --git a/README.md b/README.md index ded338e..96c952f 100644 --- a/README.md +++ b/README.md @@ -15,5 +15,6 @@ Runnable companion code for the Spring AI articles on [ankurm.com](https://ankur | [`chat-memory/`](chat-memory) | `MessageChatMemoryAdvisor`, `MessageWindowChatMemory`, the JDBC and Redis `ChatMemoryRepository`, per-user conversation IDs and a token-budget memory of our own, with the traps reproduced against a real PostgreSQL 16 and Redis Stack: a 36-character `conversation_id`, tool messages dropped on save, concurrent writers, a 1.x table under the 2.0 repository, and a Redis repository that silently steps aside for a custom `ChatMemory`. Spring Boot 4.1.1, Spring AI 2.0.1, Java 25. | [Chat Memory in Spring AI 2.0: JDBC, Redis and Windowed Conversations](https://ankurm.com/spring-ai-2-0-chat-memory-jdbc-redis-windowed-conversations/) | | [`advisors/`](advisors) | Three custom advisors -- a logger, a PII redactor (with a stream-safe restore) and a per-request / per-user token budget -- and tests for how the chain is ordered, what `BaseAdvisor` does on a stream, where an advisor sits relative to memory and the tool loop, and what a refusal looks like on a call, a stream and over HTTP (429). A recording stub model, no live model. Spring Boot 4.1.1, Spring AI 2.0.1, Java 25. | [Writing Custom Advisors in Spring AI 2.0: Logging, PII Redaction and Token Budgets](https://ankurm.com/spring-ai-2-0-custom-advisors-logging-pii-redaction-token-budgets/) | | [`vector-stores/`](vector-stores) | The same 30,000-document dataset behind `VectorStore` on pgvector, Redis, Qdrant and Elasticsearch: ingest time, recall@10, latency, metadata filtering and running cost, with the defaults that cost recall reproduced (Elasticsearch's quantised mapping, Redis `EF_RUNTIME`, pgvector post-filtering, Qdrant payload indexes). Spring Boot 4.1.1, Spring AI 2.0.1, Java 25. | [Choosing a Vector Store for Spring AI](https://ankurm.com/spring-ai-2-0-vector-store-comparison-pgvector-redis-qdrant-elasticsearch/) | +| [`evaluation/`](evaluation) | Testing an LLM app: `RelevancyEvaluator` and `FactCheckingEvaluator` (exactly what they send and which judge replies they accept), a 12-case golden dataset with a pass-rate gate, a deterministic judge for CI, simulated judge noise, a 1-5 graded evaluator and a composite. Stub models only; the one live-judge test is skipped without a key. Spring Boot 4.1.1, Spring AI 2.0.1, JUnit 6, Java 25. | [Testing LLM Apps in Java: Spring AI Evaluators and LLM-as-Judge in JUnit 6](https://ankurm.com/spring-ai-2-0-testing-llm-apps-evaluators-llm-as-judge-junit-6/) | Upgrading from Spring AI 1.x: [migration guide](https://ankurm.com/spring-ai-1-to-2-migration-guide/). diff --git a/evaluation/.gitignore b/evaluation/.gitignore new file mode 100644 index 0000000..2f7896d --- /dev/null +++ b/evaluation/.gitignore @@ -0,0 +1 @@ +target/ diff --git a/evaluation/README.md b/evaluation/README.md new file mode 100644 index 0000000..2b80495 --- /dev/null +++ b/evaluation/README.md @@ -0,0 +1,49 @@ +# evaluation + +Companion code for [Testing LLM Apps in Java: Spring AI Evaluators and LLM-as-Judge in JUnit 6](https://ankurm.com/spring-ai-2-0-testing-llm-apps-evaluators-llm-as-judge-junit-6/), part of the [Spring AI series](../README.md) on ankurm.com. + +A small retrieval-grounded support assistant, a 12-row golden dataset, and the tests that show what Spring AI's `RelevancyEvaluator` and `FactCheckingEvaluator` really send and accept, how to gate a build on a pass rate, and how to keep a model-graded test suite free and repeatable in CI. + +**No live model is used anywhere.** The "application model" is a scripted stub and the "judge" is either a recording stub or `RuleBasedJudge`, a deterministic word-overlap judge. Judge noise in `NoisyJudgeTest` is a seeded simulation. The one test that talks to a real model, `LiveJudgeTest`, is skipped unless `OPENAI_API_KEY` is set and was never run for the article. So nothing here says how well any real model judges. + +## Versions + +| Component | Version | +|---|---| +| Spring Boot | 4.1.1 | +| Spring AI | 2.0.1 (`spring-ai-client-chat`) | +| JUnit | 6.0.3 (managed by Boot 4.1.1; 6.1.3 is the latest on Maven Central) | +| Java | 25 (LTS) | + +## Quickstart + +```bash +scripts/run-all.sh # runs 21 tests (1 skipped) and regenerates output/01 .. 08 +``` + +Two consecutive runs produce byte-identical files. + +## What's here + +| File | What it shows | +|---|---| +| [`SupportAssistant.java`](src/main/java/com/ankurm/evaluation/SupportAssistant.java) | The application under test | +| [`EvalRunner.java`](src/main/java/com/ankurm/evaluation/EvalRunner.java), [`EvalReport.java`](src/main/java/com/ankurm/evaluation/EvalReport.java) | Run the golden set through three evaluators; `requirePassRate` is the build gate | +| [`RuleBasedJudge.java`](src/main/java/com/ankurm/evaluation/RuleBasedJudge.java) | A deterministic `ChatModel` that the real evaluators run against in CI | +| [`VerdictNormalizingModel.java`](src/main/java/com/ankurm/evaluation/VerdictNormalizingModel.java) | Turns "Yes." into "yes", which is all the built-in evaluators accept | +| [`GradedEvaluator.java`](src/main/java/com/ankurm/evaluation/GradedEvaluator.java) | A 1-5 judge, because the built-ins only return 0 or 1 | +| [`ContainsFactsEvaluator.java`](src/main/java/com/ankurm/evaluation/ContainsFactsEvaluator.java), [`CompositeEvaluator.java`](src/main/java/com/ankurm/evaluation/CompositeEvaluator.java) | A no-model check, and a combiner that restores the feedback the built-ins leave empty | +| [`src/test/resources/golden/support-golden.json`](src/test/resources/golden/support-golden.json) | The golden dataset | + +## Output files + +| File | Written by | +|---|---| +| [`01-evaluator-prompts.txt`](output/01-evaluator-prompts.txt) | `EvaluatorAnatomyTest`: the exact prompts, and the responses | +| [`02-verdict-parsing.txt`](output/02-verdict-parsing.txt) | `VerdictParsingTest`: which judge replies pass | +| [`03-golden-run.txt`](output/03-golden-run.txt) | `GoldenDatasetTest`: a healthy and a regressed build, and the gate | +| [`04-noisy-judge.txt`](output/04-noisy-judge.txt) | `NoisyJudgeTest`: simulated noise against five gates | +| [`05-graded-and-composite.txt`](output/05-graded-and-composite.txt) | `GradedCompositeTest` | +| [`06-judge-limits.txt`](output/06-judge-limits.txt) | `JudgeLimitsTest`: where the deterministic judge is wrong | +| [`07-ci-layers.txt`](output/07-ci-layers.txt) | `CiLayersTest`: model calls per testing layer | +| [`08-live-judge-skipped.txt`](output/08-live-judge-skipped.txt) | cut from the surefire report by `run-all.sh` | diff --git a/evaluation/output/01-evaluator-prompts.txt b/evaluation/output/01-evaluator-prompts.txt new file mode 100644 index 0000000..927f786 --- /dev/null +++ b/evaluation/output/01-evaluator-prompts.txt @@ -0,0 +1,42 @@ +# What RelevancyEvaluator and FactCheckingEvaluator send to the judge + +--- RelevancyEvaluator: prompt sent to the judge (1 message(s)) --- + Your task is to evaluate if the response for the query + is in line with the context information provided. + + You have two options to answer. Either YES or NO. + + Answer YES, if the response for the query + is in line with context information otherwise NO. + + Query: + How long do I have to request a refund on an annual plan? + + Response: + You can request a refund on an annual plan within 30 days of purchase. + + Context: + Annual plans can be refunded within 30 days of purchase. + Monthly plans are not refundable. + + Answer: + +--- RelevancyEvaluator: judge said "yes" --- +pass=true score=1.0 feedback='' metadata={} + +--- FactCheckingEvaluator: prompt sent to the judge --- + Evaluate whether or not the following claim is supported by the provided document. + Respond with "yes" if the claim is supported, or "no" if it is not. + + Document: + Annual plans can be refunded within 30 days of purchase. + Monthly plans are not refundable. + + Claim: + You can request a refund on an annual plan within 30 days of purchase. + +--- FactCheckingEvaluator: judge said "yes" --- +pass=true score=0.0 feedback='' metadata={} + +--- RelevancyEvaluator: judge said "no" --- +pass=false score=0.0 feedback='' diff --git a/evaluation/output/02-verdict-parsing.txt b/evaluation/output/02-verdict-parsing.txt new file mode 100644 index 0000000..7db7e94 --- /dev/null +++ b/evaluation/output/02-verdict-parsing.txt @@ -0,0 +1,13 @@ +# Which judge replies the built-in evaluators accept + +judge reply relevancy fact-chk wrapped +'yes' true true true +'YES' true true true +' yes\n' true true true +'Yes.' false false true +'yes!' false false true +'Yes, the response is in line with the context.' false false true +'**Yes**' false false true +'no' false false false +'No.' false false false +'' false false false diff --git a/evaluation/output/03-golden-run.txt b/evaluation/output/03-golden-run.txt new file mode 100644 index 0000000..1e7bee5 --- /dev/null +++ b/evaluation/output/03-golden-run.txt @@ -0,0 +1,37 @@ +# Golden dataset: 12 cases, deterministic judge, pass-rate gate at 0.90 + +== healthy build == +case relevant grounded has facts passed +refund-annual true true true true +refund-monthly true true true true +storage-pro true true true true +support-hours true true true true +sso-plan true true true true +data-region true true true true +api-rate-limit true true true true +backup-retention true true true true +macos-agent true true true true +reset-link true true true true +extra-seat true true true true +trial-card true true true true +pass rate: 1.00 failing: [] + +== regressed build == +case relevant grounded has facts passed +refund-annual false false false false +refund-monthly true true true true +storage-pro true true true true +support-hours true true true true +sso-plan false false false false +data-region true true true true +api-rate-limit true true true true +backup-retention true true true true +macos-agent true true true true +reset-link true true true true +extra-seat true true false false +trial-card true true true true +pass rate: 0.75 failing: [refund-annual, sso-plan, extra-seat] + +== the gate == +healthy build : requirePassRate(0.90) returned normally +regressed build : java.lang.AssertionError: pass rate 0.75 is below the 0.90 threshold; failing cases: [refund-annual, sso-plan, extra-seat] diff --git a/evaluation/output/04-noisy-judge.txt b/evaluation/output/04-noisy-judge.txt new file mode 100644 index 0000000..b030d3d --- /dev/null +++ b/evaluation/output/04-noisy-judge.txt @@ -0,0 +1,17 @@ +# Simulated judge noise: 200 runs of a 12-case suite per build + +noise-free pass rate: healthy build 1.00, regressed build 0.75 + +== 3% of judge verdicts flipped (mean pass rate: healthy 0.947, regressed 0.706) == +gate healthy build fails it regressed build fails it +every case must pass 96 of 200 runs 200 of 200 runs +pass rate >= 0.90 29 of 200 runs 200 of 200 runs +pass rate >= 0.85 29 of 200 runs 200 of 200 runs +pass rate >= 0.80 2 of 200 runs 200 of 200 runs + +== 8% of judge verdicts flipped (mean pass rate: healthy 0.859, regressed 0.640) == +gate healthy build fails it regressed build fails it +every case must pass 166 of 200 runs 200 of 200 runs +pass rate >= 0.90 102 of 200 runs 200 of 200 runs +pass rate >= 0.85 102 of 200 runs 200 of 200 runs +pass rate >= 0.80 47 of 200 runs 200 of 200 runs diff --git a/evaluation/output/05-graded-and-composite.txt b/evaluation/output/05-graded-and-composite.txt new file mode 100644 index 0000000..aacdfae --- /dev/null +++ b/evaluation/output/05-graded-and-composite.txt @@ -0,0 +1,16 @@ +# GradedEvaluator (minimum grade 4) and CompositeEvaluator + +judge reply pass score feedback +'5' true 1.0 grade 5 of 5 +'4/5' true 0.8 grade 4 of 5 +'Score: 4 - mostly right, misses the date' true 0.8 grade 4 of 5 +'3' false 0.6 grade 3 of 5 +'I would say excellent' false 0.0 judge did not return a grade: I would say excellent +'10' false 0.0 judge did not return a grade: 10 + +== CompositeEvaluator on the regressed 'extra-seat' answer ("12 USD" instead of "8 USD") == +answer : Extra seats cost 12 USD per month. +pass : false +score : 0.5 +feedback : contains-facts failed (missing: [8 USD]); +verdicts : {relevancy=true, contains-facts=false} diff --git a/evaluation/output/06-judge-limits.txt b/evaluation/output/06-judge-limits.txt new file mode 100644 index 0000000..fb12323 --- /dev/null +++ b/evaluation/output/06-judge-limits.txt @@ -0,0 +1,8 @@ +# RuleBasedJudge(0.6) on hand-labelled answers to: Can I get a refund on a monthly plan? + +answer type correct? judge says verdict +correct, same words true pass right +WRONG: negation dropped false pass WRONG +WRONG: roles swapped false pass WRONG +correct, paraphrased true fail WRONG +correct, but a refusal true fail WRONG diff --git a/evaluation/output/07-ci-layers.txt b/evaluation/output/07-ci-layers.txt new file mode 100644 index 0000000..fc47911 --- /dev/null +++ b/evaluation/output/07-ci-layers.txt @@ -0,0 +1,7 @@ +# Model calls per testing layer (stub app model, 12 golden cases) + +layer app calls judge calls network +1 unit: prompt and wiring, no evaluator 1 0 none +2 golden run, RuleBasedJudge (pull request) 12 24 none +3 golden run, real judge (nightly) n/a n/a needs OPENAI_API_KEY; skipped here +pass rate at layer 2: 1.00 diff --git a/evaluation/output/08-live-judge-skipped.txt b/evaluation/output/08-live-judge-skipped.txt new file mode 100644 index 0000000..4de7428 --- /dev/null +++ b/evaluation/output/08-live-judge-skipped.txt @@ -0,0 +1,4 @@ +# LiveJudgeTest as recorded by surefire when OPENAI_API_KEY is not set + +testcase: "realJudgeAgreesWithTheKnownGoodAndKnownBadBuilds" +skipped : "Environment variable [OPENAI_API_KEY] does not exist" diff --git a/evaluation/pom.xml b/evaluation/pom.xml new file mode 100644 index 0000000..144be29 --- /dev/null +++ b/evaluation/pom.xml @@ -0,0 +1,70 @@ + + + 4.0.0 + + + org.springframework.boot + spring-boot-starter-parent + 4.1.1 + + + + com.ankurm + evaluation + 1.0.0 + evaluation + Testing LLM apps in JUnit 6: Spring AI RelevancyEvaluator and FactCheckingEvaluator, LLM-as-judge, golden datasets and CI-safe deterministic judges. + + + 25 + 2.0.1 + + + + + + org.springframework.ai + spring-ai-bom + ${spring-ai.version} + pom + import + + + + + + + org.springframework.ai + spring-ai-client-chat + + + tools.jackson.core + jackson-databind + + + + org.springframework.ai + spring-ai-starter-model-openai + test + + + org.springframework.boot + spring-boot-starter-test + test + + + + + + + org.apache.maven.plugins + maven-surefire-plugin + + -Duser.timezone=UTC -Dstdout.encoding=UTF-8 -Dfile.encoding=UTF-8 + + + + + diff --git a/evaluation/scripts/run-all.sh b/evaluation/scripts/run-all.sh new file mode 100755 index 0000000..36df16b --- /dev/null +++ b/evaluation/scripts/run-all.sh @@ -0,0 +1,16 @@ +#!/usr/bin/env bash +# Regenerates every file under output/. Tests 01-07 are written by the suite itself through the +# Transcript helper; 08 is cut from the surefire report to show the live-judge test being skipped. +# No Docker, no database and no API key is needed. +set -euo pipefail +cd "$(dirname "$0")/.." +rm -rf target +mvn -q -B test 2>&1 | grep -E "Tests run:|BUILD|FAIL" || true +f=target/surefire-reports/TEST-com.ankurm.evaluation.LiveJudgeTest.xml +{ + echo "# LiveJudgeTest as recorded by surefire when OPENAI_API_KEY is not set" + echo + grep -o ' output/08-live-judge-skipped.txt +ls output diff --git a/evaluation/src/main/java/com/ankurm/evaluation/CaseResult.java b/evaluation/src/main/java/com/ankurm/evaluation/CaseResult.java new file mode 100644 index 0000000..99224b1 --- /dev/null +++ b/evaluation/src/main/java/com/ankurm/evaluation/CaseResult.java @@ -0,0 +1,9 @@ +package com.ankurm.evaluation; + +/** The outcome of running one golden case through the evaluators. */ +public record CaseResult(String id, String answer, boolean relevant, boolean grounded, boolean hasFacts) { + + public boolean passed() { + return relevant && grounded && hasFacts; + } +} diff --git a/evaluation/src/main/java/com/ankurm/evaluation/CompositeEvaluator.java b/evaluation/src/main/java/com/ankurm/evaluation/CompositeEvaluator.java new file mode 100644 index 0000000..198baf6 --- /dev/null +++ b/evaluation/src/main/java/com/ankurm/evaluation/CompositeEvaluator.java @@ -0,0 +1,37 @@ +package com.ankurm.evaluation; + +import java.util.LinkedHashMap; +import java.util.List; +import java.util.Map; + +import org.springframework.ai.evaluation.EvaluationRequest; +import org.springframework.ai.evaluation.EvaluationResponse; +import org.springframework.ai.evaluation.Evaluator; + +/** Runs several named evaluators on one request. It passes only if all of them pass; the score is their mean; per-evaluator verdicts go in metadata. */ +public class CompositeEvaluator implements Evaluator { + + private final Map parts; + + public CompositeEvaluator(Map parts) { + this.parts = parts; + } + + @Override + public EvaluationResponse evaluate(EvaluationRequest request) { + Map verdicts = new LinkedHashMap<>(); + boolean pass = true; + float total = 0; + StringBuilder feedback = new StringBuilder(); + for (var e : parts.entrySet()) { + EvaluationResponse r = e.getValue().evaluate(request); + verdicts.put(e.getKey(), r.isPass()); + pass &= r.isPass(); + total += r.getScore(); + if (!r.isPass()) { + feedback.append(e.getKey()).append(" failed").append(r.getFeedback().isEmpty() ? "; " : " (" + r.getFeedback() + "); "); + } + } + return new EvaluationResponse(pass, total / parts.size(), feedback.toString().strip(), verdicts); + } +} diff --git a/evaluation/src/main/java/com/ankurm/evaluation/ContainsFactsEvaluator.java b/evaluation/src/main/java/com/ankurm/evaluation/ContainsFactsEvaluator.java new file mode 100644 index 0000000..181d57e --- /dev/null +++ b/evaluation/src/main/java/com/ankurm/evaluation/ContainsFactsEvaluator.java @@ -0,0 +1,30 @@ +package com.ankurm.evaluation; + +import java.util.List; +import java.util.Locale; +import java.util.Map; + +import org.springframework.ai.evaluation.EvaluationRequest; +import org.springframework.ai.evaluation.EvaluationResponse; +import org.springframework.ai.evaluation.Evaluator; + +/** + * A plain-code {@link Evaluator}: the answer must contain every expected fact as a substring. + * No model call, no cost, no noise, and it fills in the feedback the built-in evaluators leave empty. + */ +public class ContainsFactsEvaluator implements Evaluator { + + private final List facts; + + public ContainsFactsEvaluator(List facts) { + this.facts = facts; + } + + @Override + public EvaluationResponse evaluate(EvaluationRequest request) { + String answer = request.getResponseContent().toLowerCase(Locale.ROOT); + List missing = facts.stream().filter(f -> !answer.contains(f.toLowerCase(Locale.ROOT))).toList(); + float score = facts.isEmpty() ? 1f : (facts.size() - missing.size()) / (float) facts.size(); + return new EvaluationResponse(missing.isEmpty(), score, missing.isEmpty() ? "all facts present" : "missing: " + missing, Map.of()); + } +} diff --git a/evaluation/src/main/java/com/ankurm/evaluation/EvalReport.java b/evaluation/src/main/java/com/ankurm/evaluation/EvalReport.java new file mode 100644 index 0000000..7a62d48 --- /dev/null +++ b/evaluation/src/main/java/com/ankurm/evaluation/EvalReport.java @@ -0,0 +1,22 @@ +package com.ankurm.evaluation; + +import java.util.List; + +/** All results of one run over the golden dataset, with the pass rate a build can gate on. */ +public record EvalReport(List results) { + + public double passRate() { + return results.isEmpty() ? 0 : results.stream().filter(CaseResult::passed).count() / (double) results.size(); + } + + public List failedIds() { + return results.stream().filter(r -> !r.passed()).map(CaseResult::id).toList(); + } + + /** Fails the build (with an {@link AssertionError} naming the failing cases) when the pass rate is below the threshold. */ + public void requirePassRate(double threshold) { + if (passRate() < threshold) { + throw new AssertionError("pass rate %.2f is below the %.2f threshold; failing cases: %s".formatted(passRate(), threshold, failedIds())); + } + } +} diff --git a/evaluation/src/main/java/com/ankurm/evaluation/EvalRunner.java b/evaluation/src/main/java/com/ankurm/evaluation/EvalRunner.java new file mode 100644 index 0000000..0981389 --- /dev/null +++ b/evaluation/src/main/java/com/ankurm/evaluation/EvalRunner.java @@ -0,0 +1,38 @@ +package com.ankurm.evaluation; + +import java.util.List; + +import org.springframework.ai.chat.client.ChatClient; +import org.springframework.ai.chat.evaluation.FactCheckingEvaluator; +import org.springframework.ai.chat.evaluation.RelevancyEvaluator; +import org.springframework.ai.chat.model.ChatModel; +import org.springframework.ai.evaluation.EvaluationRequest; + +/** Runs every golden case through the assistant, then through the relevancy, fact-checking and contains-facts evaluators. */ +public class EvalRunner { + + private final SupportAssistant assistant; + + private final RelevancyEvaluator relevancy; + + private final FactCheckingEvaluator factChecking; + + public EvalRunner(SupportAssistant assistant, ChatModel judge) { + this.assistant = assistant; + this.relevancy = new RelevancyEvaluator(ChatClient.builder(judge)); + this.factChecking = FactCheckingEvaluator.builder(ChatClient.builder(judge)).build(); + } + + public EvalReport run(List cases) { + return new EvalReport(cases.stream().map(this::runOne).toList()); + } + + private CaseResult runOne(GoldenCase c) { + String answer = assistant.answer(c.question(), c.documents()); + EvaluationRequest request = new EvaluationRequest(c.question(), c.documents(), answer); + boolean relevant = relevancy.evaluate(request).isPass(); + boolean grounded = factChecking.evaluate(request).isPass(); + boolean facts = new ContainsFactsEvaluator(c.expectedFacts()).evaluate(request).isPass(); + return new CaseResult(c.id(), answer, relevant, grounded, facts); + } +} diff --git a/evaluation/src/main/java/com/ankurm/evaluation/GoldenCase.java b/evaluation/src/main/java/com/ankurm/evaluation/GoldenCase.java new file mode 100644 index 0000000..b57989d --- /dev/null +++ b/evaluation/src/main/java/com/ankurm/evaluation/GoldenCase.java @@ -0,0 +1,13 @@ +package com.ankurm.evaluation; + +import java.util.List; + +import org.springframework.ai.document.Document; + +/** One row of the golden dataset: a question, the documents retrieval is expected to supply, and facts a good answer must contain. */ +public record GoldenCase(String id, String question, List context, List expectedFacts) { + + public List documents() { + return context.stream().map(t -> Document.builder().text(t).build()).toList(); + } +} diff --git a/evaluation/src/main/java/com/ankurm/evaluation/GoldenDataset.java b/evaluation/src/main/java/com/ankurm/evaluation/GoldenDataset.java new file mode 100644 index 0000000..2efc18e --- /dev/null +++ b/evaluation/src/main/java/com/ankurm/evaluation/GoldenDataset.java @@ -0,0 +1,28 @@ +package com.ankurm.evaluation; + +import java.io.IOException; +import java.io.InputStream; +import java.util.List; + +import tools.jackson.core.type.TypeReference; +import tools.jackson.databind.json.JsonMapper; + +/** Loads the golden dataset from a JSON file on the classpath. */ +public final class GoldenDataset { + + private GoldenDataset() { + } + + public static List load(String resource) { + try (InputStream in = GoldenDataset.class.getResourceAsStream(resource)) { + if (in == null) { + throw new IllegalArgumentException("missing resource " + resource); + } + return JsonMapper.builder().build().readValue(in, new TypeReference>() { + }); + } + catch (IOException e) { + throw new IllegalStateException(e); + } + } +} diff --git a/evaluation/src/main/java/com/ankurm/evaluation/GradedEvaluator.java b/evaluation/src/main/java/com/ankurm/evaluation/GradedEvaluator.java new file mode 100644 index 0000000..d91d31a --- /dev/null +++ b/evaluation/src/main/java/com/ankurm/evaluation/GradedEvaluator.java @@ -0,0 +1,53 @@ +package com.ankurm.evaluation; + +import java.util.Map; +import java.util.regex.Matcher; +import java.util.regex.Pattern; + +import org.springframework.ai.chat.client.ChatClient; +import org.springframework.ai.evaluation.EvaluationRequest; +import org.springframework.ai.evaluation.EvaluationResponse; +import org.springframework.ai.evaluation.Evaluator; + +/** + * An LLM-as-judge that grades 1-5 instead of yes/no, because the built-in evaluators can only + * return a score of 0 or 1. It pulls the first digit 1-5 out of the reply, so "4", "4/5" and + * "Score: 4 - mostly right" all work; a reply with no grade fails with the raw text as feedback. + */ +public class GradedEvaluator implements Evaluator { + + static final String PROMPT = """ + Grade how well the answer addresses the question, using only the context. + Reply with a single integer from 1 (useless) to 5 (complete and correct). + + Question: %s + + Answer: %s + + Context: %s + """; + + private static final Pattern GRADE = Pattern.compile("\\b([1-5])\\b"); + + private final ChatClient.Builder builder; + + private final int minimumGrade; + + public GradedEvaluator(ChatClient.Builder builder, int minimumGrade) { + this.builder = builder; + this.minimumGrade = minimumGrade; + } + + @Override + public EvaluationResponse evaluate(EvaluationRequest request) { + String reply = builder.build().prompt() + .user(PROMPT.formatted(request.getUserText(), request.getResponseContent(), doGetSupportingData(request))) + .call().content(); + Matcher m = GRADE.matcher(reply == null ? "" : reply); + if (!m.find()) { + return new EvaluationResponse(false, 0f, "judge did not return a grade: " + reply, Map.of()); + } + int grade = Integer.parseInt(m.group(1)); + return new EvaluationResponse(grade >= minimumGrade, grade / 5f, "grade " + grade + " of 5", Map.of("grade", grade)); + } +} diff --git a/evaluation/src/main/java/com/ankurm/evaluation/RuleBasedJudge.java b/evaluation/src/main/java/com/ankurm/evaluation/RuleBasedJudge.java new file mode 100644 index 0000000..46053d8 --- /dev/null +++ b/evaluation/src/main/java/com/ankurm/evaluation/RuleBasedJudge.java @@ -0,0 +1,77 @@ +package com.ankurm.evaluation; + +import java.util.Arrays; +import java.util.List; +import java.util.Locale; +import java.util.Set; +import java.util.regex.Matcher; +import java.util.regex.Pattern; +import java.util.stream.Collectors; + +import org.springframework.ai.chat.messages.AssistantMessage; +import org.springframework.ai.chat.model.ChatModel; +import org.springframework.ai.chat.model.ChatResponse; +import org.springframework.ai.chat.model.Generation; +import org.springframework.ai.chat.prompt.Prompt; + +/** + * A deterministic stand-in for a judge model, for continuous integration. It is a + * {@link ChatModel}, so the real {@code RelevancyEvaluator} and {@code FactCheckingEvaluator} + * run unchanged against it, prompt templates and all. It does not understand anything: it + * answers "yes" when enough of the claim's content words (and numbers) also occur in the context. + * That catches a swapped number or an off-topic answer and misses a negation -- see + * {@code JudgeLimitsTest}. + */ +public class RuleBasedJudge implements ChatModel { + + private static final Set STOP = Set.of("the", "and", "that", "this", "with", "from", "have", "your", "you", "are", "for", "can", "will", "not", "does", "what", "how", "long", "which"); + + private static final Pattern RELEVANCY = Pattern.compile("Response:\\s*(.*?)\\s*Context:\\s*(.*?)\\s*Answer:", Pattern.DOTALL); + + private static final Pattern FACT = Pattern.compile("Document:\\s*(.*?)\\s*Claim:\\s*(.*)", Pattern.DOTALL); + + private final double minimumOverlap; + + public RuleBasedJudge(double minimumOverlap) { + this.minimumOverlap = minimumOverlap; + } + + @Override + public ChatResponse call(Prompt prompt) { + String text = prompt.getInstructions().getLast().getText(); + String claim; + String evidence; + Matcher r = RELEVANCY.matcher(text); + Matcher f = FACT.matcher(text); + if (r.find()) { + claim = r.group(1); + evidence = r.group(2); + } + else if (f.find()) { + evidence = f.group(1); + claim = f.group(2); + } + else { + throw new IllegalArgumentException("RuleBasedJudge does not recognise this prompt: " + text); + } + return new ChatResponse(List.of(new Generation(new AssistantMessage(overlap(claim, evidence) >= minimumOverlap ? "yes" : "no")))); + } + + /** Share of the claim's content words that also appear in the evidence. */ + public static double overlap(String claim, String evidence) { + Set have = words(evidence); + Set need = words(claim); + if (need.isEmpty()) { + return 0; + } + long hit = need.stream().filter(have::contains).count(); + return hit / (double) need.size(); + } + + private static Set words(String s) { + return Arrays.stream(s.toLowerCase(Locale.ROOT).split("[^a-z0-9]+")) + .filter(w -> w.length() > 3 || w.matches("\\d+")) + .filter(w -> !STOP.contains(w)) + .collect(Collectors.toSet()); + } +} diff --git a/evaluation/src/main/java/com/ankurm/evaluation/SupportAssistant.java b/evaluation/src/main/java/com/ankurm/evaluation/SupportAssistant.java new file mode 100644 index 0000000..34f4f04 --- /dev/null +++ b/evaluation/src/main/java/com/ankurm/evaluation/SupportAssistant.java @@ -0,0 +1,33 @@ +package com.ankurm.evaluation; + +import java.util.List; +import java.util.stream.Collectors; + +import org.springframework.ai.chat.client.ChatClient; +import org.springframework.ai.document.Document; + +/** + * The application under test: a tiny retrieval-grounded support assistant. It receives the + * retrieved documents as an argument so a test controls exactly what the model is shown; in a real + * application a {@code QuestionAnswerAdvisor} would put them there. + */ +public class SupportAssistant { + + static final String SYSTEM = "Answer using only the context. If the context does not contain the answer, say you do not know."; + + private final ChatClient client; + + public SupportAssistant(ChatClient.Builder builder) { + this.client = builder.defaultSystem(SYSTEM).build(); + } + + public String answer(String question, List context) { + String joined = context.stream().map(Document::getText).collect(Collectors.joining("\n")); + return client.prompt() + .user(u -> u.text("Context:\n{context}\n\nQuestion: {question}") + .param("context", joined) + .param("question", question)) + .call() + .content(); + } +} diff --git a/evaluation/src/main/java/com/ankurm/evaluation/VerdictNormalizingModel.java b/evaluation/src/main/java/com/ankurm/evaluation/VerdictNormalizingModel.java new file mode 100644 index 0000000..d3a234c --- /dev/null +++ b/evaluation/src/main/java/com/ankurm/evaluation/VerdictNormalizingModel.java @@ -0,0 +1,33 @@ +package com.ankurm.evaluation; + +import java.util.List; +import java.util.Locale; + +import org.springframework.ai.chat.messages.AssistantMessage; +import org.springframework.ai.chat.model.ChatModel; +import org.springframework.ai.chat.model.ChatResponse; +import org.springframework.ai.chat.model.Generation; +import org.springframework.ai.chat.prompt.Prompt; + +/** + * Wraps a judge model and reduces its reply to the bare word the built-in evaluators compare + * against. They require the whole stripped reply to equal "yes" (ignoring case), so "Yes." or "Yes, + * the response matches" count as a failure. This keeps the first word, drops punctuation, and + * leaves everything else alone. + */ +public class VerdictNormalizingModel implements ChatModel { + + private final ChatModel delegate; + + public VerdictNormalizingModel(ChatModel delegate) { + this.delegate = delegate; + } + + @Override + public ChatResponse call(Prompt prompt) { + String reply = delegate.call(prompt).getResult().getOutput().getText(); + String first = reply == null ? "" : reply.strip().split("\\s+", 2)[0].replaceAll("[^A-Za-z]", "").toLowerCase(Locale.ROOT); + String verdict = first.equals("yes") || first.equals("no") ? first : reply; + return new ChatResponse(List.of(new Generation(new AssistantMessage(verdict)))); + } +} diff --git a/evaluation/src/test/java/com/ankurm/evaluation/CiLayersTest.java b/evaluation/src/test/java/com/ankurm/evaluation/CiLayersTest.java new file mode 100644 index 0000000..c042f13 --- /dev/null +++ b/evaluation/src/test/java/com/ankurm/evaluation/CiLayersTest.java @@ -0,0 +1,51 @@ +package com.ankurm.evaluation; + +import com.ankurm.evaluation.support.Answers; +import com.ankurm.evaluation.support.RecordingModel; +import com.ankurm.evaluation.support.Transcript; +import org.junit.jupiter.api.Test; +import org.springframework.ai.chat.client.ChatClient; +import org.springframework.ai.chat.messages.AssistantMessage; +import org.springframework.ai.chat.model.ChatResponse; +import org.springframework.ai.chat.model.Generation; + +import static org.assertj.core.api.Assertions.assertThat; + +/** Counts the model calls each testing layer makes, so "free in CI" is a measured number. Writes output/07. */ +class CiLayersTest { + + @Test + void modelCallsPerLayer() { + var cases = GoldenDataset.load("/golden/support-golden.json"); + + // Layer 1: prompt/plumbing unit test, no evaluator at all. + RecordingModel app = Answers.appModel(Answers.good()); + var assistant = new SupportAssistant(ChatClient.builder(app)); + assistant.answer(cases.getFirst().question(), cases.getFirst().documents()); + int unitAppCalls = app.callCount(); + + // Layer 2: golden run with the deterministic judge. + RecordingModel app2 = Answers.appModel(Answers.good()); + RuleBasedJudge rule = new RuleBasedJudge(0.6); + int[] judgeCalls = { 0 }; + var countingJudge = new org.springframework.ai.chat.model.ChatModel() { + @Override + public ChatResponse call(org.springframework.ai.chat.prompt.Prompt prompt) { + judgeCalls[0]++; + return rule.call(prompt); + } + }; + EvalReport report = new EvalRunner(new SupportAssistant(ChatClient.builder(app2)), countingJudge).run(cases); + + try (Transcript t = new Transcript("07-ci-layers.txt", "Model calls per testing layer (stub app model, 12 golden cases)")) { + t.line("%-44s %-12s %-12s %s", "layer", "app calls", "judge calls", "network"); + t.line("%-44s %-12d %-12d %s", "1 unit: prompt and wiring, no evaluator", unitAppCalls, 0, "none"); + t.line("%-44s %-12d %-12d %s", "2 golden run, RuleBasedJudge (pull request)", app2.callCount(), judgeCalls[0], "none"); + t.line("%-44s %-12s %-12s %s", "3 golden run, real judge (nightly)", "n/a", "n/a", "needs OPENAI_API_KEY; skipped here"); + t.line("pass rate at layer 2: %.2f", report.passRate()); + } + assertThat(unitAppCalls).isEqualTo(1); + assertThat(app2.callCount()).isEqualTo(12); + assertThat(judgeCalls[0]).isEqualTo(24); + } +} diff --git a/evaluation/src/test/java/com/ankurm/evaluation/EvaluatorAnatomyTest.java b/evaluation/src/test/java/com/ankurm/evaluation/EvaluatorAnatomyTest.java new file mode 100644 index 0000000..2fc5554 --- /dev/null +++ b/evaluation/src/test/java/com/ankurm/evaluation/EvaluatorAnatomyTest.java @@ -0,0 +1,62 @@ +package com.ankurm.evaluation; + +import java.util.List; + +import com.ankurm.evaluation.support.RecordingModel; +import com.ankurm.evaluation.support.Transcript; +import org.junit.jupiter.api.Test; +import org.springframework.ai.chat.client.ChatClient; +import org.springframework.ai.chat.evaluation.FactCheckingEvaluator; +import org.springframework.ai.chat.evaluation.RelevancyEvaluator; +import org.springframework.ai.document.Document; +import org.springframework.ai.evaluation.EvaluationRequest; +import org.springframework.ai.evaluation.EvaluationResponse; + +import static org.assertj.core.api.Assertions.assertThat; + +/** What the two built-in evaluators actually send to the judge and what they hand back. Writes output/01. */ +class EvaluatorAnatomyTest { + + private static final EvaluationRequest REQUEST = new EvaluationRequest( + "How long do I have to request a refund on an annual plan?", + List.of(Document.builder().text("Annual plans can be refunded within 30 days of purchase.").build(), + Document.builder().text("Monthly plans are not refundable.").build()), + "You can request a refund on an annual plan within 30 days of purchase."); + + @Test + void whatTheJudgeIsSentAndWhatComesBack() { + try (Transcript t = new Transcript("01-evaluator-prompts.txt", "What RelevancyEvaluator and FactCheckingEvaluator send to the judge")) { + RecordingModel judge = RecordingModel.replying("yes"); + + EvaluationResponse relevancy = new RelevancyEvaluator(ChatClient.builder(judge)).evaluate(REQUEST); + String relevancyPrompt = judge.lastUserText(); + EvaluationResponse fact = FactCheckingEvaluator.builder(ChatClient.builder(judge)).build().evaluate(REQUEST); + String factPrompt = judge.lastUserText(); + + t.line("--- RelevancyEvaluator: prompt sent to the judge (%d message(s)) ---", judge.prompts().getFirst().getInstructions().size()); + t.line(relevancyPrompt); + t.line("--- RelevancyEvaluator: judge said \"yes\" ---"); + t.line("pass=%s score=%s feedback='%s' metadata=%s", relevancy.isPass(), relevancy.getScore(), relevancy.getFeedback(), relevancy.getMetadata()); + t.blank(); + t.line("--- FactCheckingEvaluator: prompt sent to the judge ---"); + t.line(factPrompt); + t.line("--- FactCheckingEvaluator: judge said \"yes\" ---"); + t.line("pass=%s score=%s feedback='%s' metadata=%s", fact.isPass(), fact.getScore(), fact.getFeedback(), fact.getMetadata()); + t.blank(); + + EvaluationResponse failing = new RelevancyEvaluator(ChatClient.builder(RecordingModel.replying("no"))).evaluate(REQUEST); + t.line("--- RelevancyEvaluator: judge said \"no\" ---"); + t.line("pass=%s score=%s feedback='%s'", failing.isPass(), failing.getScore(), failing.getFeedback()); + + assertThat(relevancy.isPass()).isTrue(); + assertThat(relevancy.getScore()).isEqualTo(1.0f); + assertThat(failing.getScore()).isEqualTo(0.0f); + assertThat(failing.getFeedback()).isEmpty(); + assertThat(relevancyPrompt).contains("\tAnnual plans can be refunded within 30 days of purchase.\n\tMonthly plans are not refundable."); + assertThat(factPrompt).contains("Claim:").contains("Document:"); + // FactCheckingEvaluator builds its response without a score, so a pass still reports 0.0. + assertThat(fact.isPass()).isTrue(); + assertThat(fact.getScore()).isEqualTo(0.0f); + } + } +} diff --git a/evaluation/src/test/java/com/ankurm/evaluation/GoldenDatasetTest.java b/evaluation/src/test/java/com/ankurm/evaluation/GoldenDatasetTest.java new file mode 100644 index 0000000..6852be2 --- /dev/null +++ b/evaluation/src/test/java/com/ankurm/evaluation/GoldenDatasetTest.java @@ -0,0 +1,93 @@ +package com.ankurm.evaluation; + +import java.util.List; +import java.util.Map; + +import com.ankurm.evaluation.support.Answers; +import com.ankurm.evaluation.support.RecordingModel; +import com.ankurm.evaluation.support.Transcript; +import org.junit.jupiter.api.Test; +import org.junit.jupiter.params.ParameterizedTest; +import org.junit.jupiter.params.provider.MethodSource; +import org.springframework.ai.chat.client.ChatClient; +import org.springframework.ai.chat.evaluation.FactCheckingEvaluator; +import org.springframework.ai.chat.evaluation.RelevancyEvaluator; +import org.springframework.ai.evaluation.EvaluationRequest; + +import static org.assertj.core.api.Assertions.assertThat; +import static org.assertj.core.api.Assertions.assertThatThrownBy; + +/** A golden dataset run end to end, with a pass-rate gate. A healthy build passes it; a build with three bad answers fails it. Writes output/03. */ +class GoldenDatasetTest { + + static final List GOLDEN = GoldenDataset.load("/golden/support-golden.json"); + + private static EvalReport run(Map answers) { + SupportAssistant app = new SupportAssistant(ChatClient.builder(Answers.appModel(answers))); + return new EvalRunner(app, new RuleBasedJudge(0.6)).run(GOLDEN); + } + + @Test + void healthyAndRegressedBuilds() { + EvalReport healthy = run(Answers.good()); + EvalReport regressed = run(Answers.regressed()); + + try (Transcript t = new Transcript("03-golden-run.txt", "Golden dataset: %d cases, deterministic judge, pass-rate gate at 0.90".formatted(GOLDEN.size()))) { + for (var entry : List.of(Map.entry("healthy build", healthy), Map.entry("regressed build", regressed))) { + t.line("== %s ==", entry.getKey()); + t.line("%-18s %-9s %-9s %-9s %-7s", "case", "relevant", "grounded", "has facts", "passed"); + for (CaseResult r : entry.getValue().results()) { + t.line("%-18s %-9s %-9s %-9s %-7s", r.id(), r.relevant(), r.grounded(), r.hasFacts(), r.passed()); + } + t.line("pass rate: %.2f failing: %s", entry.getValue().passRate(), entry.getValue().failedIds()); + t.blank(); + } + t.line("== the gate =="); + healthy.requirePassRate(0.90); + t.line("healthy build : requirePassRate(0.90) returned normally"); + Throwable thrown = org.assertj.core.api.Assertions.catchThrowable(() -> regressed.requirePassRate(0.90)); + t.line("regressed build : %s", thrown); + } + + assertThat(healthy.passRate()).isEqualTo(1.0); + assertThat(regressed.failedIds()).containsExactly("refund-annual", "sso-plan", "extra-seat"); + assertThatThrownBy(() -> regressed.requirePassRate(0.90)).isInstanceOf(AssertionError.class).hasMessageContaining("sso-plan"); + } + + @Test + void whichEvaluatorCaughtWhat() { + EvalReport regressed = run(Answers.regressed()); + CaseResult hallucinated = regressed.results().stream().filter(r -> r.id().equals("refund-annual")).findFirst().orElseThrow(); + CaseResult offTopic = regressed.results().stream().filter(r -> r.id().equals("sso-plan")).findFirst().orElseThrow(); + CaseResult wrongNumber = regressed.results().stream().filter(r -> r.id().equals("extra-seat")).findFirst().orElseThrow(); + + assertThat(hallucinated.grounded()).isFalse(); + assertThat(offTopic.relevant()).isFalse(); + // A plausible wrong number slips past the judge (5 of 6 content words overlap) and is caught only by the exact-fact check. + assertThat(wrongNumber.relevant()).isTrue(); + assertThat(wrongNumber.grounded()).isTrue(); + assertThat(wrongNumber.hasFacts()).isFalse(); + } + + static List cases() { + return GOLDEN; + } + + /** The per-case style: one JUnit invocation per golden row, readable in any test report. Only suitable for a deterministic judge. */ + @ParameterizedTest(name = "[{index}] {0}") + @MethodSource("ids") + void healthyBuildPassesEveryCase(String id) { + GoldenCase c = GOLDEN.stream().filter(g -> g.id().equals(id)).findFirst().orElseThrow(); + RecordingModel app = Answers.appModel(Answers.good()); + String answer = new SupportAssistant(ChatClient.builder(app)).answer(c.question(), c.documents()); + EvaluationRequest request = new EvaluationRequest(c.question(), c.documents(), answer); + RuleBasedJudge judge = new RuleBasedJudge(0.6); + assertThat(new RelevancyEvaluator(ChatClient.builder(judge)).evaluate(request).isPass()).as("relevant").isTrue(); + assertThat(FactCheckingEvaluator.builder(ChatClient.builder(judge)).build().evaluate(request).isPass()).as("grounded").isTrue(); + assertThat(new ContainsFactsEvaluator(c.expectedFacts()).evaluate(request).isPass()).as("facts").isTrue(); + } + + static List ids() { + return GOLDEN.stream().map(GoldenCase::id).toList(); + } +} diff --git a/evaluation/src/test/java/com/ankurm/evaluation/GradedCompositeTest.java b/evaluation/src/test/java/com/ankurm/evaluation/GradedCompositeTest.java new file mode 100644 index 0000000..780dad0 --- /dev/null +++ b/evaluation/src/test/java/com/ankurm/evaluation/GradedCompositeTest.java @@ -0,0 +1,65 @@ +package com.ankurm.evaluation; + +import java.util.LinkedHashMap; +import java.util.List; +import java.util.Map; + +import com.ankurm.evaluation.support.Answers; +import com.ankurm.evaluation.support.RecordingModel; +import com.ankurm.evaluation.support.Transcript; +import org.junit.jupiter.api.Test; +import org.springframework.ai.chat.client.ChatClient; +import org.springframework.ai.chat.evaluation.RelevancyEvaluator; +import org.springframework.ai.document.Document; +import org.springframework.ai.evaluation.EvaluationRequest; +import org.springframework.ai.evaluation.EvaluationResponse; + +import static org.assertj.core.api.Assertions.assertThat; + +/** The two custom evaluators: a 1-5 grade (the built-ins only score 0 or 1) and a composite that restores feedback. Writes output/05. */ +class GradedCompositeTest { + + private static final EvaluationRequest REQUEST = new EvaluationRequest("Q?", List.of(Document.builder().text("ctx").build()), "answer"); + + @Test + void gradedEvaluatorReadsAGradeOutOfMessyReplies() { + Map expected = new LinkedHashMap<>(); + expected.put("5", true); + expected.put("4/5", true); + expected.put("Score: 4 - mostly right, misses the date", true); + expected.put("3", false); + expected.put("I would say excellent", false); + expected.put("10", false); + + try (Transcript t = new Transcript("05-graded-and-composite.txt", "GradedEvaluator (minimum grade 4) and CompositeEvaluator")) { + t.line("%-44s %-6s %-6s %s", "judge reply", "pass", "score", "feedback"); + for (var e : expected.entrySet()) { + EvaluationResponse r = new GradedEvaluator(ChatClient.builder(RecordingModel.replying(e.getKey())), 4).evaluate(REQUEST); + t.line("%-44s %-6s %-6s %s", "'" + e.getKey() + "'", r.isPass(), r.getScore(), r.getFeedback()); + assertThat(r.isPass()).as(e.getKey()).isEqualTo(e.getValue()); + } + + t.blank(); + t.line("== CompositeEvaluator on the regressed 'extra-seat' answer (\"12 USD\" instead of \"8 USD\") =="); + GoldenCase c = GoldenDatasetTest.GOLDEN.stream().filter(g -> g.id().equals("extra-seat")).findFirst().orElseThrow(); + String answer = Answers.regressed().get(c.question()); + EvaluationRequest request = new EvaluationRequest(c.question(), c.documents(), answer); + RuleBasedJudge judge = new RuleBasedJudge(0.6); + var composite = new CompositeEvaluator(new LinkedHashMap<>(Map.of()) { + { + put("relevancy", new RelevancyEvaluator(ChatClient.builder(judge))); + put("contains-facts", new ContainsFactsEvaluator(c.expectedFacts())); + } + }); + EvaluationResponse r = composite.evaluate(request); + t.line("answer : %s", answer); + t.line("pass : %s", r.isPass()); + t.line("score : %s", r.getScore()); + t.line("feedback : %s", r.getFeedback()); + t.line("verdicts : %s", r.getMetadata()); + assertThat(r.isPass()).isFalse(); + assertThat(r.getFeedback()).contains("contains-facts failed").contains("8 USD"); + assertThat(r.getMetadata()).containsEntry("relevancy", true).containsEntry("contains-facts", false); + } + } +} diff --git a/evaluation/src/test/java/com/ankurm/evaluation/JudgeLimitsTest.java b/evaluation/src/test/java/com/ankurm/evaluation/JudgeLimitsTest.java new file mode 100644 index 0000000..6541441 --- /dev/null +++ b/evaluation/src/test/java/com/ankurm/evaluation/JudgeLimitsTest.java @@ -0,0 +1,45 @@ +package com.ankurm.evaluation; + +import java.util.List; + +import com.ankurm.evaluation.support.Transcript; +import org.junit.jupiter.api.Test; +import org.springframework.ai.chat.client.ChatClient; +import org.springframework.ai.chat.evaluation.RelevancyEvaluator; +import org.springframework.ai.document.Document; +import org.springframework.ai.evaluation.EvaluationRequest; + +import static org.assertj.core.api.Assertions.assertThat; + +/** What the cheap deterministic judge gets wrong. This is the reason a real judge still runs somewhere (nightly). Writes output/06. */ +class JudgeLimitsTest { + + private record Probe(String label, String answer, boolean actuallyCorrect) { + } + + @Test + void theRuleBasedJudgeIsFooledInBothDirections() { + List context = List.of(Document.builder().text("Annual plans can be refunded within 30 days of purchase. Monthly plans are not refundable.").build()); + String question = "Can I get a refund on a monthly plan?"; + List probes = List.of( + new Probe("correct, same words", "Monthly plans are not refundable.", true), + new Probe("WRONG: negation dropped", "Monthly plans are refundable.", false), + new Probe("WRONG: roles swapped", "Annual plans are not refundable. Monthly plans can be refunded within 30 days of purchase.", false), + new Probe("correct, paraphrased", "No, you cannot get your money back on month-to-month subscriptions.", true), + new Probe("correct, but a refusal", "I do not have that information.", true)); + + try (Transcript t = new Transcript("06-judge-limits.txt", "RuleBasedJudge(0.6) on hand-labelled answers to: " + question)) { + t.line("%-26s %-9s %-14s %s", "answer type", "correct?", "judge says", "verdict"); + for (Probe p : probes) { + boolean judged = new RelevancyEvaluator(ChatClient.builder(new RuleBasedJudge(0.6))) + .evaluate(new EvaluationRequest(question, context, p.answer())).isPass(); + t.line("%-26s %-9s %-14s %s", p.label(), p.actuallyCorrect(), judged ? "pass" : "fail", judged == p.actuallyCorrect() ? "right" : "WRONG"); + } + } + + boolean negation = new RelevancyEvaluator(ChatClient.builder(new RuleBasedJudge(0.6))).evaluate(new EvaluationRequest(question, context, "Monthly plans are refundable.")).isPass(); + boolean paraphrase = new RelevancyEvaluator(ChatClient.builder(new RuleBasedJudge(0.6))).evaluate(new EvaluationRequest(question, context, "No, you cannot get your money back on month-to-month subscriptions.")).isPass(); + assertThat(negation).as("a dropped negation shares every content word, so the judge accepts it").isTrue(); + assertThat(paraphrase).as("a correct paraphrase shares almost no words, so the judge rejects it").isFalse(); + } +} diff --git a/evaluation/src/test/java/com/ankurm/evaluation/LiveJudgeTest.java b/evaluation/src/test/java/com/ankurm/evaluation/LiveJudgeTest.java new file mode 100644 index 0000000..473dee3 --- /dev/null +++ b/evaluation/src/test/java/com/ankurm/evaluation/LiveJudgeTest.java @@ -0,0 +1,41 @@ +package com.ankurm.evaluation; + +import com.ankurm.evaluation.support.Answers; +import org.junit.jupiter.api.Tag; +import org.junit.jupiter.api.Test; +import org.junit.jupiter.api.condition.EnabledIfEnvironmentVariable; +import org.springframework.ai.chat.client.ChatClient; +import org.springframework.ai.chat.model.ChatModel; +import org.springframework.beans.factory.annotation.Autowired; +import org.springframework.boot.autoconfigure.SpringBootApplication; +import org.springframework.boot.test.context.SpringBootTest; + +import static org.assertj.core.api.Assertions.assertThat; + +/** + * The nightly layer: the same golden dataset and the same evaluators, judged by a real model. + * Skipped unless {@code OPENAI_API_KEY} is set, so a pull-request build never spends money. It was + * compiled and its skip was recorded for this article, but it was NOT executed against a live + * model, so no live-judge result appears anywhere in the post. + */ +@Tag("live") +@EnabledIfEnvironmentVariable(named = "OPENAI_API_KEY", matches = ".+") +@SpringBootTest(classes = LiveJudgeTest.App.class) +class LiveJudgeTest { + + @SpringBootApplication + static class App { + } + + @Autowired + ChatModel model; + + @Test + void realJudgeAgreesWithTheKnownGoodAndKnownBadBuilds() { + var cases = GoldenDataset.load("/golden/support-golden.json"); + var healthy = new EvalRunner(new SupportAssistant(ChatClient.builder(Answers.appModel(Answers.good()))), model).run(cases); + var regressed = new EvalRunner(new SupportAssistant(ChatClient.builder(Answers.appModel(Answers.regressed()))), model).run(cases); + assertThat(healthy.passRate()).isGreaterThanOrEqualTo(0.85); + assertThat(regressed.passRate()).isLessThan(healthy.passRate()); + } +} diff --git a/evaluation/src/test/java/com/ankurm/evaluation/NoisyJudgeTest.java b/evaluation/src/test/java/com/ankurm/evaluation/NoisyJudgeTest.java new file mode 100644 index 0000000..8e32c24 --- /dev/null +++ b/evaluation/src/test/java/com/ankurm/evaluation/NoisyJudgeTest.java @@ -0,0 +1,67 @@ +package com.ankurm.evaluation; + +import java.util.Arrays; +import java.util.List; +import java.util.Map; + +import com.ankurm.evaluation.support.Answers; +import com.ankurm.evaluation.support.FlakyJudge; +import com.ankurm.evaluation.support.Transcript; +import org.junit.jupiter.api.Test; +import org.springframework.ai.chat.client.ChatClient; + +import static org.assertj.core.api.Assertions.assertThat; + +/** + * What a noisy judge does to a test suite. The noise is SIMULATED (a correct judge whose verdict + * is flipped 3 or 8 percent of the time, seeded), so the numbers show the effect of noise on the gating + * rule and say nothing about how often any real model is wrong. Writes output/04. + */ +class NoisyJudgeTest { + + static final List GOLDEN = GoldenDataset.load("/golden/support-golden.json"); + + static final int RUNS = 200; + + private static double passRate(Map answers, double flip, long seed) { + SupportAssistant app = new SupportAssistant(ChatClient.builder(Answers.appModel(answers))); + return new EvalRunner(app, new FlakyJudge(new RuleBasedJudge(0.6), flip, seed)).run(GOLDEN).passRate(); + } + + private static long below(double[] rates, double threshold) { + return Arrays.stream(rates).filter(r -> r < threshold).count(); + } + + @Test + void strictPerCaseAssertionsVersusAnAggregateThreshold() { + double[] flips = { 0.03, 0.08 }; + double[][] healthy = new double[flips.length][RUNS]; + double[][] regressed = new double[flips.length][RUNS]; + for (int f = 0; f < flips.length; f++) { + for (int i = 0; i < RUNS; i++) { + healthy[f][i] = passRate(Answers.good(), flips[f], i); + regressed[f][i] = passRate(Answers.regressed(), flips[f], 10_000 + i); + } + } + + try (Transcript t = new Transcript("04-noisy-judge.txt", "Simulated judge noise: %d runs of a %d-case suite per build".formatted(RUNS, GOLDEN.size()))) { + t.line("noise-free pass rate: healthy build 1.00, regressed build 0.75"); + for (int f = 0; f < flips.length; f++) { + t.blank(); + t.line("== %.0f%% of judge verdicts flipped (mean pass rate: healthy %.3f, regressed %.3f) ==", flips[f] * 100, + Arrays.stream(healthy[f]).average().orElse(0), Arrays.stream(regressed[f]).average().orElse(0)); + t.line("%-22s %-28s %s", "gate", "healthy build fails it", "regressed build fails it"); + for (double threshold : new double[] { 1.0, 0.90, 0.85, 0.80 }) { + String name = threshold == 1.0 ? "every case must pass" : "pass rate >= %.2f".formatted(threshold); + t.line("%-22s %-28s %s", name, "%d of %d runs".formatted(below(healthy[f], threshold), RUNS), "%d of %d runs".formatted(below(regressed[f], threshold), RUNS)); + } + } + } + + // At 3 percent noise a 0.80 gate has few false alarms and still catches the regression; "every case must pass" does not. + assertThat(below(healthy[0], 1.0)).isGreaterThan(below(healthy[0], 0.80)); + assertThat(below(regressed[0], 0.80)).isGreaterThan(below(healthy[0], 0.80)); + // At 8 percent noise even 0.80 raises false alarms: the gate has to be set from measured noise, not guessed. + assertThat(below(healthy[1], 0.80)).isGreaterThan(below(healthy[0], 0.80)); + } +} diff --git a/evaluation/src/test/java/com/ankurm/evaluation/VerdictParsingTest.java b/evaluation/src/test/java/com/ankurm/evaluation/VerdictParsingTest.java new file mode 100644 index 0000000..4af692a --- /dev/null +++ b/evaluation/src/test/java/com/ankurm/evaluation/VerdictParsingTest.java @@ -0,0 +1,64 @@ +package com.ankurm.evaluation; + +import java.util.LinkedHashMap; +import java.util.List; +import java.util.Map; + +import com.ankurm.evaluation.support.RecordingModel; +import com.ankurm.evaluation.support.Transcript; +import org.junit.jupiter.api.Test; +import org.springframework.ai.chat.client.ChatClient; +import org.springframework.ai.chat.evaluation.FactCheckingEvaluator; +import org.springframework.ai.chat.evaluation.RelevancyEvaluator; +import org.springframework.ai.chat.model.ChatModel; +import org.springframework.ai.document.Document; +import org.springframework.ai.evaluation.EvaluationRequest; + +import static org.assertj.core.api.Assertions.assertThat; + +/** The built-in evaluators pass only on a bare "yes". This shows which judge replies count, and the normalising wrapper that fixes it. Writes output/02. */ +class VerdictParsingTest { + + private static final EvaluationRequest REQUEST = new EvaluationRequest("Q?", List.of(Document.builder().text("ctx").build()), "answer"); + + private static boolean relevancy(ChatModel judge) { + return new RelevancyEvaluator(ChatClient.builder(judge)).evaluate(REQUEST).isPass(); + } + + private static boolean fact(ChatModel judge) { + return FactCheckingEvaluator.builder(ChatClient.builder(judge)).build().evaluate(REQUEST).isPass(); + } + + @Test + void onlyABareYesPasses() { + Map expectedRaw = new LinkedHashMap<>(); + expectedRaw.put("yes", true); + expectedRaw.put("YES", true); + expectedRaw.put(" yes\n", true); + expectedRaw.put("Yes.", false); + expectedRaw.put("yes!", false); + expectedRaw.put("Yes, the response is in line with the context.", false); + expectedRaw.put("**Yes**", false); + expectedRaw.put("no", false); + expectedRaw.put("No.", false); + expectedRaw.put("", false); + + try (Transcript t = new Transcript("02-verdict-parsing.txt", "Which judge replies the built-in evaluators accept")) { + t.line("%-52s %-9s %-9s %-9s", "judge reply", "relevancy", "fact-chk", "wrapped"); + for (var e : expectedRaw.entrySet()) { + ChatModel raw = RecordingModel.replying(e.getKey()); + boolean r = relevancy(raw); + boolean f = fact(raw); + boolean wrapped = relevancy(new VerdictNormalizingModel(raw)); + t.line("%-52s %-9s %-9s %-9s", "'" + e.getKey().replace("\n", "\\n") + "'", r, f, wrapped); + assertThat(r).as("relevancy for '%s'", e.getKey()).isEqualTo(e.getValue()); + assertThat(f).as("fact for '%s'", e.getKey()).isEqualTo(e.getValue()); + } + // The wrapper turns the punctuated, bold and sentence-style verdicts into passes, and leaves "no" alone. + assertThat(relevancy(new VerdictNormalizingModel(RecordingModel.replying("Yes.")))).isTrue(); + assertThat(relevancy(new VerdictNormalizingModel(RecordingModel.replying("Yes, the response is in line with the context.")))).isTrue(); + assertThat(relevancy(new VerdictNormalizingModel(RecordingModel.replying("No.")))).isFalse(); + assertThat(relevancy(new VerdictNormalizingModel(RecordingModel.replying("**Yes**")))).isTrue(); + } + } +} diff --git a/evaluation/src/test/java/com/ankurm/evaluation/support/Answers.java b/evaluation/src/test/java/com/ankurm/evaluation/support/Answers.java new file mode 100644 index 0000000..d1a2693 --- /dev/null +++ b/evaluation/src/test/java/com/ankurm/evaluation/support/Answers.java @@ -0,0 +1,54 @@ +package com.ankurm.evaluation.support; + +import java.util.HashMap; +import java.util.Map; + +import org.springframework.ai.chat.prompt.Prompt; + +/** + * What the pretend application model says for each golden question. {@link #good()} is a healthy + * build; {@link #regressed()} is the same build after a bad prompt change: one hallucinated + * answer, one off-topic answer, and one with a plausible but wrong number. + */ +public final class Answers { + + private Answers() { + } + + public static Map good() { + Map m = new HashMap<>(); + m.put("How long do I have to request a refund on an annual plan?", "You can request a refund on an annual plan within 30 days of purchase."); + m.put("Can I get a refund on a monthly plan?", "No. Monthly plans are not refundable."); + m.put("How much storage does the Pro plan include?", "The Pro plan includes 200 GB of storage."); + m.put("When is support open?", "Support is available Monday to Friday, 9:00 to 17:00 UTC."); + m.put("Which plan has SAML single sign-on?", "SAML single sign-on is available on the Enterprise plan only."); + m.put("Can I move my data to the US after signing up?", "No. The region can only be chosen at account creation and cannot be changed later."); + m.put("What happens if I exceed the API rate limit?", "Exceeding 600 requests per minute returns HTTP 429."); + m.put("How long are Pro backups kept?", "Pro backups are kept for 14 days."); + m.put("Is there a macOS agent?", "I do not know. Acme Cloud supports Linux and Windows agents."); + m.put("How long is a password reset link valid?", "A password reset link expires after 60 minutes."); + m.put("How much does an extra seat cost?", "Extra seats cost 8 USD per month."); + m.put("Do I need a credit card for the trial?", "No. The free trial does not require a credit card."); + return m; + } + + public static Map regressed() { + Map m = good(); + m.put("How long do I have to request a refund on an annual plan?", "You can request a refund within 60 days, including on monthly plans."); + m.put("Which plan has SAML single sign-on?", "Our team is happy to help with any sign-in question you have."); + m.put("How much does an extra seat cost?", "Extra seats cost 12 USD per month."); + return m; + } + + /** The scripted application model: finds the golden question in the user message and returns its answer. */ + public static RecordingModel appModel(Map answers) { + return new RecordingModel((Prompt p) -> { + String text = p.getInstructions().getLast().getText(); + return answers.entrySet().stream() + .filter(e -> text.contains("Question: " + e.getKey())) + .map(Map.Entry::getValue) + .findFirst() + .orElse("I do not know."); + }); + } +} diff --git a/evaluation/src/test/java/com/ankurm/evaluation/support/FlakyJudge.java b/evaluation/src/test/java/com/ankurm/evaluation/support/FlakyJudge.java new file mode 100644 index 0000000..dac12fd --- /dev/null +++ b/evaluation/src/test/java/com/ankurm/evaluation/support/FlakyJudge.java @@ -0,0 +1,39 @@ +package com.ankurm.evaluation.support; + +import java.util.List; +import java.util.Random; + +import org.springframework.ai.chat.messages.AssistantMessage; +import org.springframework.ai.chat.model.ChatModel; +import org.springframework.ai.chat.model.ChatResponse; +import org.springframework.ai.chat.model.Generation; +import org.springframework.ai.chat.prompt.Prompt; + +/** + * A SIMULATION of judge noise, not a measurement of any real model: it forwards to a correct + * judge, then with a fixed probability returns the opposite verdict. The random source is seeded, + * so runs are repeatable. The point is to show what noise does to a test suite, not to claim a rate. + */ +public class FlakyJudge implements ChatModel { + + private final ChatModel delegate; + + private final double flipRate; + + private final Random random; + + public FlakyJudge(ChatModel delegate, double flipRate, long seed) { + this.delegate = delegate; + this.flipRate = flipRate; + this.random = new Random(seed); + } + + @Override + public ChatResponse call(Prompt prompt) { + String verdict = delegate.call(prompt).getResult().getOutput().getText(); + if (random.nextDouble() < flipRate) { + verdict = verdict.equals("yes") ? "no" : "yes"; + } + return new ChatResponse(List.of(new Generation(new AssistantMessage(verdict)))); + } +} diff --git a/evaluation/src/test/java/com/ankurm/evaluation/support/RecordingModel.java b/evaluation/src/test/java/com/ankurm/evaluation/support/RecordingModel.java new file mode 100644 index 0000000..747732f --- /dev/null +++ b/evaluation/src/test/java/com/ankurm/evaluation/support/RecordingModel.java @@ -0,0 +1,45 @@ +package com.ankurm.evaluation.support; + +import java.util.List; +import java.util.concurrent.CopyOnWriteArrayList; +import java.util.function.Function; + +import org.springframework.ai.chat.messages.AssistantMessage; +import org.springframework.ai.chat.model.ChatModel; +import org.springframework.ai.chat.model.ChatResponse; +import org.springframework.ai.chat.model.Generation; +import org.springframework.ai.chat.prompt.Prompt; + +/** A scripted {@link ChatModel}: no network, no key. It records every prompt and answers with whatever the replier returns. */ +public class RecordingModel implements ChatModel { + + private final List prompts = new CopyOnWriteArrayList<>(); + + private final Function replier; + + public RecordingModel(Function replier) { + this.replier = replier; + } + + public static RecordingModel replying(String fixed) { + return new RecordingModel(p -> fixed); + } + + public List prompts() { + return prompts; + } + + public int callCount() { + return prompts.size(); + } + + public String lastUserText() { + return prompts.getLast().getInstructions().getLast().getText(); + } + + @Override + public ChatResponse call(Prompt prompt) { + prompts.add(prompt); + return new ChatResponse(List.of(new Generation(new AssistantMessage(replier.apply(prompt))))); + } +} diff --git a/evaluation/src/test/java/com/ankurm/evaluation/support/Transcript.java b/evaluation/src/test/java/com/ankurm/evaluation/support/Transcript.java new file mode 100644 index 0000000..3cfc614 --- /dev/null +++ b/evaluation/src/test/java/com/ankurm/evaluation/support/Transcript.java @@ -0,0 +1,47 @@ +package com.ankurm.evaluation.support; + +import java.io.IOException; +import java.io.PrintWriter; +import java.io.StringWriter; +import java.nio.file.Files; +import java.nio.file.Path; + +/** + * Writes a numbered transcript under {@code output/} (repository root, not {@code docs/}) and + * echoes it to the console. Every console block quoted in the article comes out of one of these + * files verbatim. + */ +public final class Transcript implements AutoCloseable { + + private final Path path; + private final StringWriter buffer = new StringWriter(); + private final PrintWriter out = new PrintWriter(buffer); + + public Transcript(String fileName, String title) { + this.path = Path.of("output", fileName); + out.println("# " + title); + out.println(); + } + + public Transcript line(String format, Object... args) { + out.println(args.length == 0 ? format : String.format(format, args)); + return this; + } + + public Transcript blank() { + out.println(); + return this; + } + + @Override + public void close() { + out.flush(); + try { + Files.createDirectories(path.getParent()); + Files.writeString(path, buffer.toString()); + } catch (IOException e) { + throw new IllegalStateException("could not write " + path, e); + } + System.out.print(buffer); + } +} diff --git a/evaluation/src/test/resources/golden/support-golden.json b/evaluation/src/test/resources/golden/support-golden.json new file mode 100644 index 0000000..531c95c --- /dev/null +++ b/evaluation/src/test/resources/golden/support-golden.json @@ -0,0 +1,26 @@ +[ + {"id": "refund-annual", "question": "How long do I have to request a refund on an annual plan?", + "context": ["Annual plans can be refunded within 30 days of purchase. Monthly plans are not refundable."], "expectedFacts": ["30 days"]}, + {"id": "refund-monthly", "question": "Can I get a refund on a monthly plan?", + "context": ["Annual plans can be refunded within 30 days of purchase. Monthly plans are not refundable."], "expectedFacts": ["not refundable"]}, + {"id": "storage-pro", "question": "How much storage does the Pro plan include?", + "context": ["The Free plan includes 5 GB of storage. The Pro plan includes 200 GB of storage."], "expectedFacts": ["200 GB"]}, + {"id": "support-hours", "question": "When is support open?", + "context": ["Support is available Monday to Friday, 9:00 to 17:00 UTC. Enterprise customers have 24/7 support."], "expectedFacts": ["Monday to Friday"]}, + {"id": "sso-plan", "question": "Which plan has SAML single sign-on?", + "context": ["Single sign-on with SAML is available on the Enterprise plan only."], "expectedFacts": ["Enterprise"]}, + {"id": "data-region", "question": "Can I move my data to the US after signing up?", + "context": ["Data is stored in the EU region by default. US storage can be selected at account creation and cannot be changed later."], "expectedFacts": ["cannot be changed"]}, + {"id": "api-rate-limit", "question": "What happens if I exceed the API rate limit?", + "context": ["The API allows 600 requests per minute per key. Exceeding the limit returns HTTP 429."], "expectedFacts": ["429"]}, + {"id": "backup-retention", "question": "How long are Pro backups kept?", + "context": ["Backups are kept for 14 days on the Pro plan and 35 days on the Enterprise plan."], "expectedFacts": ["14 days"]}, + {"id": "macos-agent", "question": "Is there a macOS agent?", + "context": ["Acme Cloud supports Linux and Windows agents."], "expectedFacts": ["do not know"]}, + {"id": "reset-link", "question": "How long is a password reset link valid?", + "context": ["Password reset links expire after 60 minutes."], "expectedFacts": ["60 minutes"]}, + {"id": "extra-seat", "question": "How much does an extra seat cost?", + "context": ["Each Pro plan includes 5 seats. Extra seats cost 8 USD per month."], "expectedFacts": ["8 USD"]}, + {"id": "trial-card", "question": "Do I need a credit card for the trial?", + "context": ["The free trial lasts 14 days and does not require a credit card."], "expectedFacts": ["does not require"]} +]