#!/usr/bin/env python3 """Generates dashboards/spring-ai-observability.json. Run it, commit the JSON; Grafana provisions the file as-is.""" import json, pathlib DS = {"type": "prometheus", "uid": "prom"} PANELS = [ # (title, type, unit, queries[(expr, legend)], description) ("Cost per hour by endpoint (USD, illustrative prices)", "timeseries", "currencyUSD", [("sum by (endpoint) (rate(app_ai_cost_usd_total[1m])) * 3600", "{{endpoint}}")], "Rate of app_ai_cost_usd_total scaled to an hourly figure. Prices come from ai.pricing in application.yml."), ("Tokens per minute by endpoint and direction", "timeseries", "short", [("sum by (endpoint, type) (rate(app_ai_tokens_total[1m])) * 60", "{{endpoint}} {{type}}")], "Input and output tokens as the provider reported them."), ("Cost per request by endpoint (USD)", "bargauge", "currencyUSD", [("sum by (endpoint) (increase(app_ai_cost_usd_total[5m])) / on (endpoint) label_replace(sum by (app_endpoint) (increase(spring_ai_chat_client_seconds_count{error=\"none\"}[5m])), \"endpoint\", \"$1\", \"app_endpoint\", \"(.*)\")", "{{endpoint}}")], "Cost divided by successful chat client calls. The two sides use different label names (endpoint vs app_endpoint), so label_replace renames one before the division."), ("Model call latency p50 / p95 (s)", "timeseries", "s", [("histogram_quantile(0.50, sum by (le) (rate(gen_ai_client_operation_seconds_bucket{error=\"none\"}[1m])))", "p50"), ("histogram_quantile(0.95, sum by (le) (rate(gen_ai_client_operation_seconds_bucket{error=\"none\"}[1m])))", "p95")], "gen_ai.client.operation, one sample per HTTP call to the provider. Needs the percentiles-histogram property."), ("Failed model calls (share)", "stat", "percentunit", [("sum(rate(gen_ai_client_operation_seconds_count{error!=\"none\"}[5m])) / sum(rate(gen_ai_client_operation_seconds_count[5m]))", "failed")], "Calls whose error tag is not none, over all calls."), ("Tool calls per minute", "timeseries", "short", [("sum by (spring_ai_tool_definition_name) (rate(spring_ai_tool_seconds_count[1m])) * 60", "{{spring_ai_tool_definition_name}}")], "spring.ai.tool, one series per tool name."), ("Mean tool latency (s)", "timeseries", "s", [("sum by (spring_ai_tool_definition_name) (rate(spring_ai_tool_seconds_sum[1m])) / sum by (spring_ai_tool_definition_name) (rate(spring_ai_tool_seconds_count[1m]))", "{{spring_ai_tool_definition_name}}")], "Sum over count of the tool timer."), ("Calls with no cost recorded", "timeseries", "short", [("sum by (endpoint, reason) (increase(app_ai_unpriced_total[5m]))", "{{endpoint}} {{reason}}")], "Calls that failed, reported no usage, or used a model with no configured price. Zero cost is never recorded silently."), ] panels = [] for i, (title, kind, unit, queries, desc) in enumerate(PANELS): panels.append({ "id": i + 1, "title": title, "type": kind, "description": desc, "datasource": DS, "gridPos": {"h": 8, "w": 12, "x": (i % 2) * 12, "y": (i // 2) * 8}, "fieldConfig": {"defaults": {"unit": unit}, "overrides": []}, "targets": [{"refId": chr(65 + j), "datasource": DS, "expr": e, "legendFormat": l, "editorMode": "code", "range": True} for j, (e, l) in enumerate(queries)], }) dash = {"uid": "spring-ai-observability", "title": "Spring AI: tokens, latency and cost", "schemaVersion": 39, "version": 1, "editable": True, "refresh": "5s", "time": {"from": "now-15m", "to": "now"}, "tags": ["spring-ai"], "panels": panels} out = pathlib.Path(__file__).resolve().parent.parent / "dashboards" / "spring-ai-observability.json" out.write_text(json.dumps(dash, indent=2) + "\n") print("wrote", out, len(panels), "panels")