Add observability module: Spring AI built-in meters and spans, cost per endpoint from token usage, Prometheus and Grafana dashboard
Co-Authored-By: Claude Sonnet 5.5 <[email protected]> Claude-Session: https://claude.ai/code/session_01JXVi2GMQ7bR5EmbUFdDj7N
This commit is contained in:
Executable
+13
@@ -0,0 +1,13 @@
|
||||
#!/usr/bin/env bash
|
||||
# Sends a mixed workload so every dashboard panel has data. Usage: load.sh [rounds]
|
||||
H=http://127.0.0.1:8080
|
||||
for i in $(seq 1 "${1:-10}"); do
|
||||
curl -s -o /dev/null "$H/ask?q=what+is+observability+$i"
|
||||
curl -s -o /dev/null "$H/summarize" --get --data-urlencode "text=$(head -c 2000 /dev/zero | tr '\0' 'x') round $i"
|
||||
curl -s -o /dev/null "$H/weather?city=Pune"
|
||||
curl -s -o /dev/null "$H/stream?q=tell+me+something+$i"
|
||||
[ $((i % 4)) -eq 0 ] && curl -s -o /dev/null "$H/ask?q=slow+request+$i"
|
||||
[ $((i % 5)) -eq 0 ] && curl -s -o /dev/null "$H/ask?q=boom+$i"
|
||||
[ $((i % 6)) -eq 0 ] && curl -s -o /dev/null "$H/ask?q=nousage+$i"
|
||||
done
|
||||
echo "load done"
|
||||
Executable
+49
@@ -0,0 +1,49 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Generates dashboards/spring-ai-observability.json. Run it, commit the JSON; Grafana provisions the file as-is."""
|
||||
import json, pathlib
|
||||
|
||||
DS = {"type": "prometheus", "uid": "prom"}
|
||||
PANELS = [
|
||||
# (title, type, unit, queries[(expr, legend)], description)
|
||||
("Cost per hour by endpoint (USD, illustrative prices)", "timeseries", "currencyUSD",
|
||||
[("sum by (endpoint) (rate(app_ai_cost_usd_total[1m])) * 3600", "{{endpoint}}")],
|
||||
"Rate of app_ai_cost_usd_total scaled to an hourly figure. Prices come from ai.pricing in application.yml."),
|
||||
("Tokens per minute by endpoint and direction", "timeseries", "short",
|
||||
[("sum by (endpoint, type) (rate(app_ai_tokens_total[1m])) * 60", "{{endpoint}} {{type}}")],
|
||||
"Input and output tokens as the provider reported them."),
|
||||
("Cost per request by endpoint (USD)", "bargauge", "currencyUSD",
|
||||
[("sum by (endpoint) (increase(app_ai_cost_usd_total[5m])) / on (endpoint) label_replace(sum by (app_endpoint) (increase(spring_ai_chat_client_seconds_count{error=\"none\"}[5m])), \"endpoint\", \"$1\", \"app_endpoint\", \"(.*)\")", "{{endpoint}}")],
|
||||
"Cost divided by successful chat client calls. The two sides use different label names (endpoint vs app_endpoint), so label_replace renames one before the division."),
|
||||
("Model call latency p50 / p95 (s)", "timeseries", "s",
|
||||
[("histogram_quantile(0.50, sum by (le) (rate(gen_ai_client_operation_seconds_bucket{error=\"none\"}[1m])))", "p50"),
|
||||
("histogram_quantile(0.95, sum by (le) (rate(gen_ai_client_operation_seconds_bucket{error=\"none\"}[1m])))", "p95")],
|
||||
"gen_ai.client.operation, one sample per HTTP call to the provider. Needs the percentiles-histogram property."),
|
||||
("Failed model calls (share)", "stat", "percentunit",
|
||||
[("sum(rate(gen_ai_client_operation_seconds_count{error!=\"none\"}[5m])) / sum(rate(gen_ai_client_operation_seconds_count[5m]))", "failed")],
|
||||
"Calls whose error tag is not none, over all calls."),
|
||||
("Tool calls per minute", "timeseries", "short",
|
||||
[("sum by (spring_ai_tool_definition_name) (rate(spring_ai_tool_seconds_count[1m])) * 60", "{{spring_ai_tool_definition_name}}")],
|
||||
"spring.ai.tool, one series per tool name."),
|
||||
("Mean tool latency (s)", "timeseries", "s",
|
||||
[("sum by (spring_ai_tool_definition_name) (rate(spring_ai_tool_seconds_sum[1m])) / sum by (spring_ai_tool_definition_name) (rate(spring_ai_tool_seconds_count[1m]))", "{{spring_ai_tool_definition_name}}")],
|
||||
"Sum over count of the tool timer."),
|
||||
("Calls with no cost recorded", "timeseries", "short",
|
||||
[("sum by (endpoint, reason) (increase(app_ai_unpriced_total[5m]))", "{{endpoint}} {{reason}}")],
|
||||
"Calls that failed, reported no usage, or used a model with no configured price. Zero cost is never recorded silently."),
|
||||
]
|
||||
|
||||
panels = []
|
||||
for i, (title, kind, unit, queries, desc) in enumerate(PANELS):
|
||||
panels.append({
|
||||
"id": i + 1, "title": title, "type": kind, "description": desc, "datasource": DS,
|
||||
"gridPos": {"h": 8, "w": 12, "x": (i % 2) * 12, "y": (i // 2) * 8},
|
||||
"fieldConfig": {"defaults": {"unit": unit}, "overrides": []},
|
||||
"targets": [{"refId": chr(65 + j), "datasource": DS, "expr": e, "legendFormat": l, "editorMode": "code", "range": True}
|
||||
for j, (e, l) in enumerate(queries)],
|
||||
})
|
||||
|
||||
dash = {"uid": "spring-ai-observability", "title": "Spring AI: tokens, latency and cost", "schemaVersion": 39, "version": 1,
|
||||
"editable": True, "refresh": "5s", "time": {"from": "now-15m", "to": "now"}, "tags": ["spring-ai"], "panels": panels}
|
||||
out = pathlib.Path(__file__).resolve().parent.parent / "dashboards" / "spring-ai-observability.json"
|
||||
out.write_text(json.dumps(dash, indent=2) + "\n")
|
||||
print("wrote", out, len(panels), "panels")
|
||||
Executable
+13
@@ -0,0 +1,13 @@
|
||||
#!/usr/bin/env bash
|
||||
# Regenerates every file in output/: 01-07 from the test suite, 08-09 from the live Prometheus + Grafana stack.
|
||||
# Needs JDK 25 and Maven on PATH, plus the Prometheus and Grafana tarballs unpacked (see stack-up.sh).
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")/.."
|
||||
mvn -q -B test
|
||||
scripts/stack-up.sh
|
||||
trap 'scripts/stack-down.sh' EXIT
|
||||
scripts/load.sh 12
|
||||
sleep 8
|
||||
{ echo "# Prometheus metric families this app exposes at /actuator/prometheus (TYPE lines only)"; echo
|
||||
curl -s 127.0.0.1:8080/actuator/prometheus | grep -E '^# TYPE (gen_ai_|spring_ai_|app_ai_)' | sort; } > output/09-prometheus-families.txt
|
||||
scripts/verify-dashboard.py output/08-dashboard-queries.txt
|
||||
Executable
+4
@@ -0,0 +1,4 @@
|
||||
#!/usr/bin/env bash
|
||||
# Stops what stack-up.sh started, by PID (never by pattern).
|
||||
for f in /tmp/obs-run/*.pid; do [ -f "$f" ] && kill "$(cat "$f")" 2>/dev/null || true; done
|
||||
echo stopped
|
||||
Executable
+27
@@ -0,0 +1,27 @@
|
||||
#!/usr/bin/env bash
|
||||
# Starts the whole demo stack without Docker, all on 127.0.0.1:
|
||||
# fake OpenAI server :8099 -> the Spring app :8080 <-scraped every 2s- Prometheus :9090 <- Grafana :3000
|
||||
# PROM and GRAFANA default to where the article says to unpack the release tarballs.
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")/.."
|
||||
PROM="${PROM:-/tmp/tools/obs/prom}"; GRAFANA="${GRAFANA:-/tmp/tools/obs/grafana}"
|
||||
RUN=/tmp/obs-run; rm -rf "$RUN"; mkdir -p "$RUN/dashboards" "$RUN/prom" "$RUN/grafana/data" "$RUN/grafana/logs" "$RUN/grafana/plugins"
|
||||
cp dashboards/*.json "$RUN/dashboards/"
|
||||
mvn -q -B -DskipTests package
|
||||
JAR=$(ls target/*.jar | grep -v original | head -1)
|
||||
start() { local name=$1; shift; setsid nohup "$@" > "$RUN/$name.log" 2>&1 < /dev/null & echo $! > "$RUN/$name.pid"; }
|
||||
wait_for() { for i in $(seq 1 60); do curl -s -o /dev/null "$1" && return 0; sleep 1; done; echo "timeout waiting for $1" >&2; exit 1; }
|
||||
|
||||
# the fake server ships inside the same Boot jar; PropertiesLauncher lets us pick its main class
|
||||
start fake java -Dloader.main=com.ankurm.observability.fake.FakeOpenAiServer -cp "$JAR" org.springframework.boot.loader.launch.PropertiesLauncher 8099
|
||||
wait_for http://127.0.0.1:8099/
|
||||
start app java -jar "$JAR" --spring.profiles.active=stack
|
||||
wait_for http://127.0.0.1:8080/actuator/health
|
||||
start prometheus "$PROM/prometheus" --config.file=stack/prometheus.yml --storage.tsdb.path="$RUN/prom" --web.listen-address=127.0.0.1:9090
|
||||
wait_for http://127.0.0.1:9090/-/ready
|
||||
GF_PATHS_DATA="$RUN/grafana/data" GF_PATHS_LOGS="$RUN/grafana/logs" GF_PATHS_PLUGINS="$RUN/grafana/plugins" \
|
||||
GF_PATHS_PROVISIONING="$PWD/stack/grafana/provisioning" GF_SERVER_HTTP_ADDR=127.0.0.1 GF_SERVER_HTTP_PORT=3000 \
|
||||
GF_SECURITY_ADMIN_PASSWORD=admin GF_ANALYTICS_REPORTING_ENABLED=false GF_ANALYTICS_CHECK_FOR_UPDATES=false \
|
||||
start grafana "$GRAFANA/bin/grafana" server --homepath "$GRAFANA"
|
||||
wait_for http://127.0.0.1:3000/api/health
|
||||
echo "up: app :8080, prometheus :9090, grafana :3000 (admin/admin). pids in $RUN/*.pid"
|
||||
Executable
+38
@@ -0,0 +1,38 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Runs every dashboard query twice: straight against Prometheus, and through Grafana's /api/ds/query with the
|
||||
provisioned datasource (what a panel really does). Prints one line per panel query and exits non-zero if any
|
||||
query errors or returns no series. Usage: verify-dashboard.py [output-file]"""
|
||||
import base64, json, pathlib, sys, time, urllib.parse, urllib.request
|
||||
|
||||
root = pathlib.Path(__file__).resolve().parent.parent
|
||||
dash = json.loads((root / "dashboards" / "spring-ai-observability.json").read_text())
|
||||
auth = "Basic " + base64.b64encode(b"admin:admin").decode()
|
||||
|
||||
def get(url, data=None, headers=None):
|
||||
req = urllib.request.Request(url, data=data, headers=headers or {})
|
||||
with urllib.request.urlopen(req, timeout=30) as r:
|
||||
return json.loads(r.read())
|
||||
|
||||
lines, bad = [], 0
|
||||
now = int(time.time())
|
||||
lines.append(f"{'panel':<58} {'prom series':>11} {'grafana frames':>14}")
|
||||
for p in dash["panels"]:
|
||||
for t in p["targets"]:
|
||||
q = urllib.parse.urlencode({"query": t["expr"]})
|
||||
prom = get("http://127.0.0.1:9090/api/v1/query?" + q)
|
||||
n_prom = len(prom["data"]["result"]) if prom["status"] == "success" else -1
|
||||
body = json.dumps({"queries": [{"refId": t["refId"], "datasource": t["datasource"], "expr": t["expr"], "instant": False,
|
||||
"range": True, "interval": "", "intervalMs": 5000, "maxDataPoints": 200}],
|
||||
"from": str((now - 600) * 1000), "to": str(now * 1000)}).encode()
|
||||
gf = get("http://127.0.0.1:3000/api/ds/query", body, {"Content-Type": "application/json", "Authorization": auth})
|
||||
res = gf["results"][t["refId"]]
|
||||
n_gf = -1 if "error" in res else len(res.get("frames", []))
|
||||
name = f"{p['title'][:48]} [{t['refId']}]"
|
||||
lines.append(f"{name:<58} {n_prom:>11} {n_gf:>14}")
|
||||
if n_prom < 1 or n_gf < 1:
|
||||
bad += 1
|
||||
out = "\n".join(lines)
|
||||
print(out)
|
||||
if len(sys.argv) > 1:
|
||||
pathlib.Path(sys.argv[1]).write_text("# Every dashboard query, run against Prometheus and through Grafana's query API after scripts/load.sh\n\n" + out + "\n")
|
||||
sys.exit(1 if bad else 0)
|
||||
Reference in New Issue
Block a user