parts = new ArrayList<>();
+ JsonNode src = p == Provider.GEMINI ? b.path("generationConfig") : b;
+ for (String f : new String[] {"temperature", "top_p", "topP", "max_tokens", "max_completion_tokens", "maxOutputTokens"}) {
+ if (src.has(f)) {
+ parts.add(f + "=" + src.path(f).asString());
+ }
+ }
+ return parts.isEmpty() ? "nothing (the vendor's own defaults apply)" : String.join(", ", parts);
+ }
+}
diff --git a/providers/src/test/java/com/ankurm/providers/support/FakeProviderServer.java b/providers/src/test/java/com/ankurm/providers/support/FakeProviderServer.java
new file mode 100644
index 0000000..d8b3be9
--- /dev/null
+++ b/providers/src/test/java/com/ankurm/providers/support/FakeProviderServer.java
@@ -0,0 +1,287 @@
+package com.ankurm.providers.support;
+
+import java.io.IOException;
+import java.io.OutputStream;
+import java.net.InetSocketAddress;
+import java.nio.charset.StandardCharsets;
+import java.util.ArrayList;
+import java.util.HashSet;
+import java.util.List;
+import java.util.Map;
+import java.util.Set;
+import java.util.concurrent.ConcurrentHashMap;
+import java.util.concurrent.CopyOnWriteArrayList;
+import java.util.concurrent.atomic.AtomicInteger;
+
+import com.ankurm.providers.Provider;
+import com.sun.net.httpserver.HttpExchange;
+import com.sun.net.httpserver.HttpServer;
+import tools.jackson.databind.JsonNode;
+import tools.jackson.databind.json.JsonMapper;
+import tools.jackson.databind.node.ObjectNode;
+
+/**
+ * One local HTTP server that answers like OpenAI chat completions, Anthropic messages and Gemini
+ * generateContent. The REAL Spring AI model classes talk to it, so the requests it records are the
+ * requests those classes would send to the vendors.
+ *
+ * It is not a language model and not a vendor. Three things it does are SIMULATION, not
+ * measurement, and every number derived from them says so in the article:
+ *
+ * - token counts are {@code ceil(characters / 4)};
+ * - the answer is a fixed JSON string;
+ * - cache hits follow the rules the vendors DOCUMENT (see {@link #cacheRules}), applied to the
+ * request the client really sent. Whether a real vendor hits its cache is not tested here.
+ *
+ */
+public class FakeProviderServer implements AutoCloseable {
+
+ /** What one request carried, reduced to the things the article compares. */
+ public record Seen(Provider provider, String path, JsonNode body, String system, String user, String model) {
+ }
+
+ /** Vendor cache thresholds, in tokens, as documented on 2026-10-09 for the models named in {@link Provider}. */
+ public record CacheRules(int openAiMinTokens, int anthropicMinTokens, int geminiMinTokens) {
+ public static CacheRules documented() {
+ return new CacheRules(1024, 512, 4096);
+ }
+ }
+
+ private static final JsonMapper JSON = JsonMapper.builder().build();
+
+ private final HttpServer server;
+
+ private final List seen = new CopyOnWriteArrayList<>();
+
+ private final Map failuresLeft = new ConcurrentHashMap<>();
+
+ private final Map failureStatus = new ConcurrentHashMap<>();
+
+ private final Map> previousPrompts = new ConcurrentHashMap<>();
+
+ private final Map> anthropicWritten = new ConcurrentHashMap<>();
+
+ private volatile CacheRules cacheRules = CacheRules.documented();
+
+ private volatile String answer = "{\"title\":\"Login fails after password reset\",\"priority\":\"high\"}";
+
+ public FakeProviderServer() throws IOException {
+ server = HttpServer.create(new InetSocketAddress("127.0.0.1", 0), 0);
+ server.createContext("/", this::handle);
+ server.start();
+ }
+
+ public String url() {
+ return "http://127.0.0.1:" + server.getAddress().getPort();
+ }
+
+ public List seen() {
+ return seen;
+ }
+
+ public List seen(Provider p) {
+ return seen.stream().filter(s -> s.provider() == p).toList();
+ }
+
+ public FakeProviderServer answer(String json) {
+ this.answer = json;
+ return this;
+ }
+
+ public FakeProviderServer cacheRules(CacheRules r) {
+ this.cacheRules = r;
+ return this;
+ }
+
+ /** The next {@code count} requests to this provider fail with {@code status}. */
+ public FakeProviderServer failNext(Provider p, int status, int count) {
+ failureStatus.put(p, status);
+ failuresLeft.put(p, new AtomicInteger(count));
+ return this;
+ }
+
+ public void reset() {
+ seen.clear();
+ failuresLeft.clear();
+ previousPrompts.clear();
+ anthropicWritten.clear();
+ }
+
+ /** Estimated tokens: one per four characters, rounded up. Documented as a simulation. */
+ public static int tokens(String s) {
+ return (s.length() + 3) / 4;
+ }
+
+ @Override
+ public void close() {
+ server.stop(0);
+ }
+
+ private void handle(HttpExchange ex) throws IOException {
+ String path = ex.getRequestURI().getPath();
+ Provider p = path.contains("/chat/completions") ? Provider.OPENAI
+ : path.contains("/messages") ? Provider.ANTHROPIC
+ : path.contains(":generateContent") ? Provider.GEMINI : null;
+ String raw = new String(ex.getRequestBody().readAllBytes(), StandardCharsets.UTF_8);
+ if (p == null) {
+ reply(ex, 404, "{\"error\":\"unknown path " + path + "\"}");
+ return;
+ }
+ JsonNode body = JSON.readTree(raw);
+ Seen s = extract(p, path, body);
+ seen.add(s);
+ AtomicInteger left = failuresLeft.get(p);
+ if (left != null && left.getAndDecrement() > 0) {
+ reply(ex, failureStatus.get(p), errorBody(p, failureStatus.get(p)));
+ return;
+ }
+ reply(ex, 200, success(p, s));
+ }
+
+ private static Seen extract(Provider p, String path, JsonNode b) {
+ StringBuilder system = new StringBuilder();
+ StringBuilder user = new StringBuilder();
+ String model = b.path("model").asString("");
+ switch (p) {
+ case OPENAI -> {
+ for (JsonNode m : b.path("messages")) {
+ StringBuilder into = m.path("role").asString().equals("user") ? user : system;
+ text(m.path("content"), into);
+ }
+ }
+ case ANTHROPIC -> {
+ text(b.path("system"), system);
+ for (JsonNode m : b.path("messages")) {
+ text(m.path("content"), user);
+ }
+ }
+ case GEMINI -> {
+ for (JsonNode part : b.path("systemInstruction").path("parts")) {
+ system.append(part.path("text").asString(""));
+ }
+ for (JsonNode c : b.path("contents")) {
+ for (JsonNode part : c.path("parts")) {
+ user.append(part.path("text").asString(""));
+ }
+ }
+ String[] seg = path.split("/models/");
+ model = seg.length > 1 ? seg[1].replace(":generateContent", "") : "";
+ }
+ }
+ return new Seen(p, path, b, system.toString(), user.toString(), model);
+ }
+
+ private static void text(JsonNode content, StringBuilder into) {
+ if (content.isString()) {
+ into.append(content.asString());
+ } else {
+ for (JsonNode part : content) {
+ into.append(part.path("text").asString(""));
+ }
+ }
+ }
+
+ private String errorBody(Provider p, int status) {
+ return switch (p) {
+ case OPENAI -> "{\"error\":{\"message\":\"simulated " + status + "\",\"type\":\"server_error\",\"code\":null}}";
+ case ANTHROPIC -> "{\"type\":\"error\",\"error\":{\"type\":\"overloaded_error\",\"message\":\"simulated " + status + "\"}}";
+ case GEMINI -> "{\"error\":{\"code\":" + status + ",\"message\":\"simulated " + status + "\",\"status\":\"UNAVAILABLE\"}}";
+ };
+ }
+
+ private String success(Provider p, Seen s) {
+ String prompt = s.system() + "\n" + s.user();
+ int in = tokens(prompt);
+ int out = tokens(answer);
+ ObjectNode r = JSON.createObjectNode();
+ switch (p) {
+ case OPENAI -> {
+ int cached = prefixCached(p, prompt, cacheRules.openAiMinTokens());
+ r.put("id", "chatcmpl-fake").put("object", "chat.completion").put("created", 1700000000)
+ .put("model", s.model());
+ ObjectNode msg = r.putArray("choices").addObject().put("index", 0).put("finish_reason", "stop")
+ .putObject("message").put("role", "assistant").put("content", answer);
+ msg.putNull("refusal");
+ ObjectNode u = r.putObject("usage").put("prompt_tokens", in).put("completion_tokens", out)
+ .put("total_tokens", in + out);
+ u.putObject("prompt_tokens_details").put("cached_tokens", cached);
+ }
+ case ANTHROPIC -> {
+ int created = 0;
+ int read = 0;
+ String prefix = anthropicCachedPrefix(s.body());
+ if (prefix != null && tokens(prefix) >= cacheRules.anthropicMinTokens()) {
+ Set written = anthropicWritten.computeIfAbsent(p, k -> new HashSet<>());
+ if (written.contains(prefix)) {
+ read = tokens(prefix);
+ } else {
+ written.add(prefix);
+ created = tokens(prefix);
+ }
+ }
+ r.put("id", "msg_fake").put("type", "message").put("role", "assistant").put("model", s.model())
+ .put("stop_reason", "end_turn").putNull("stop_sequence");
+ r.putArray("content").addObject().put("type", "text").put("text", answer);
+ r.putObject("usage").put("input_tokens", in - created - read).put("output_tokens", out)
+ .put("cache_creation_input_tokens", created).put("cache_read_input_tokens", read);
+ }
+ case GEMINI -> {
+ int cached = prefixCached(p, prompt, cacheRules.geminiMinTokens());
+ ObjectNode cand = r.putArray("candidates").addObject().put("finishReason", "STOP").put("index", 0);
+ cand.putObject("content").put("role", "model").putArray("parts").addObject().put("text", answer);
+ r.putObject("usageMetadata").put("promptTokenCount", in).put("candidatesTokenCount", out)
+ .put("totalTokenCount", in + out).put("cachedContentTokenCount", cached);
+ r.put("modelVersion", s.model());
+ }
+ }
+ return JSON.writeValueAsString(r);
+ }
+
+ /** OpenAI and Gemini: documented automatic prefix caching. Cached = longest shared prefix with one of the last four prompts. */
+ private int prefixCached(Provider p, String prompt, int minTokens) {
+ List prev = previousPrompts.computeIfAbsent(p, k -> new ArrayList<>());
+ int best = 0;
+ synchronized (prev) {
+ for (String old : prev) {
+ int n = 0;
+ int max = Math.min(old.length(), prompt.length());
+ while (n < max && old.charAt(n) == prompt.charAt(n)) {
+ n++;
+ }
+ best = Math.max(best, n);
+ }
+ prev.add(prompt);
+ if (prev.size() > 4) {
+ prev.remove(0);
+ }
+ }
+ int t = best / 4;
+ return t >= minTokens ? t : 0;
+ }
+
+ /** Anthropic: text of tools-then-system blocks up to and including the last block carrying cache_control, else null. */
+ private static String anthropicCachedPrefix(JsonNode body) {
+ JsonNode sys = body.path("system");
+ if (!sys.isArray()) {
+ return null;
+ }
+ StringBuilder prefix = new StringBuilder();
+ String marked = null;
+ for (JsonNode block : sys) {
+ prefix.append(block.path("text").asString(""));
+ if (block.has("cache_control")) {
+ marked = prefix.toString();
+ }
+ }
+ return marked;
+ }
+
+ private void reply(HttpExchange ex, int status, String json) throws IOException {
+ byte[] out = json.getBytes(StandardCharsets.UTF_8);
+ ex.getResponseHeaders().add("Content-Type", "application/json");
+ ex.sendResponseHeaders(status, out.length);
+ try (OutputStream os = ex.getResponseBody()) {
+ os.write(out);
+ }
+ }
+}
diff --git a/providers/src/test/java/com/ankurm/providers/support/Fixtures.java b/providers/src/test/java/com/ankurm/providers/support/Fixtures.java
new file mode 100644
index 0000000..844bd10
--- /dev/null
+++ b/providers/src/test/java/com/ankurm/providers/support/Fixtures.java
@@ -0,0 +1,28 @@
+package com.ankurm.providers.support;
+
+/** Deterministic text of a chosen size, so the prompt length (and so the cost) is known exactly. */
+public final class Fixtures {
+
+ private Fixtures() {
+ }
+
+ private static final String[] RULES = {
+ "Tickets that mention data loss, a security problem or a payment failure are priority high.",
+ "Tickets about slow pages, wrong totals or broken exports are priority medium.",
+ "Questions about how to use a feature, cosmetic problems and feature requests are priority low.",
+ "Keep the title under eight words and write it in the present tense without the customer's name.",
+ "If the ticket contains several problems, summarise the most severe one and ignore the rest.",
+ "Never repeat an email address, phone number or order number from the ticket in the title.",
+ };
+
+ /** A support policy of at least {@code chars} characters, built from numbered rules. */
+ public static String policy(int chars) {
+ StringBuilder sb = new StringBuilder("You triage support tickets for an online shop. Follow every rule below.\n");
+ int n = 1;
+ while (sb.length() < chars) {
+ sb.append("Rule ").append(n).append(": ").append(RULES[(n - 1) % RULES.length]).append('\n');
+ n++;
+ }
+ return sb.toString();
+ }
+}
diff --git a/providers/src/test/java/com/ankurm/providers/support/Transcript.java b/providers/src/test/java/com/ankurm/providers/support/Transcript.java
new file mode 100644
index 0000000..a55656e
--- /dev/null
+++ b/providers/src/test/java/com/ankurm/providers/support/Transcript.java
@@ -0,0 +1,47 @@
+package com.ankurm.providers.support;
+
+import java.io.IOException;
+import java.io.PrintWriter;
+import java.io.StringWriter;
+import java.nio.file.Files;
+import java.nio.file.Path;
+
+/**
+ * Writes a numbered transcript under {@code output/} (repository root, not {@code docs/}) and
+ * echoes it to the console. Every console block quoted in the article comes out of one of these
+ * files verbatim.
+ */
+public final class Transcript implements AutoCloseable {
+
+ private final Path path;
+ private final StringWriter buffer = new StringWriter();
+ private final PrintWriter out = new PrintWriter(buffer);
+
+ public Transcript(String fileName, String title) {
+ this.path = Path.of("output", fileName);
+ out.println("# " + title);
+ out.println();
+ }
+
+ public Transcript line(String format, Object... args) {
+ out.println(args.length == 0 ? format : String.format(format, args));
+ return this;
+ }
+
+ public Transcript blank() {
+ out.println();
+ return this;
+ }
+
+ @Override
+ public void close() {
+ out.flush();
+ try {
+ Files.createDirectories(path.getParent());
+ Files.writeString(path, buffer.toString());
+ } catch (IOException e) {
+ throw new IllegalStateException("could not write " + path, e);
+ }
+ System.out.print(buffer);
+ }
+}