Files
spring-ai/observability/scripts/make-dashboard.py
T

50 lines
3.7 KiB
Python
Executable File

#!/usr/bin/env python3
"""Generates dashboards/spring-ai-observability.json. Run it, commit the JSON; Grafana provisions the file as-is."""
import json, pathlib
DS = {"type": "prometheus", "uid": "prom"}
PANELS = [
# (title, type, unit, queries[(expr, legend)], description)
("Cost per hour by endpoint (USD, illustrative prices)", "timeseries", "currencyUSD",
[("sum by (endpoint) (rate(app_ai_cost_usd_total[1m])) * 3600", "{{endpoint}}")],
"Rate of app_ai_cost_usd_total scaled to an hourly figure. Prices come from ai.pricing in application.yml."),
("Tokens per minute by endpoint and direction", "timeseries", "short",
[("sum by (endpoint, type) (rate(app_ai_tokens_total[1m])) * 60", "{{endpoint}} {{type}}")],
"Input and output tokens as the provider reported them."),
("Cost per request by endpoint (USD)", "bargauge", "currencyUSD",
[("sum by (endpoint) (increase(app_ai_cost_usd_total[5m])) / on (endpoint) label_replace(sum by (app_endpoint) (increase(spring_ai_chat_client_seconds_count{error=\"none\"}[5m])), \"endpoint\", \"$1\", \"app_endpoint\", \"(.*)\")", "{{endpoint}}")],
"Cost divided by successful chat client calls. The two sides use different label names (endpoint vs app_endpoint), so label_replace renames one before the division."),
("Model call latency p50 / p95 (s)", "timeseries", "s",
[("histogram_quantile(0.50, sum by (le) (rate(gen_ai_client_operation_seconds_bucket{error=\"none\"}[1m])))", "p50"),
("histogram_quantile(0.95, sum by (le) (rate(gen_ai_client_operation_seconds_bucket{error=\"none\"}[1m])))", "p95")],
"gen_ai.client.operation, one sample per HTTP call to the provider. Needs the percentiles-histogram property."),
("Failed model calls (share)", "stat", "percentunit",
[("sum(rate(gen_ai_client_operation_seconds_count{error!=\"none\"}[5m])) / sum(rate(gen_ai_client_operation_seconds_count[5m]))", "failed")],
"Calls whose error tag is not none, over all calls."),
("Tool calls per minute", "timeseries", "short",
[("sum by (spring_ai_tool_definition_name) (rate(spring_ai_tool_seconds_count[1m])) * 60", "{{spring_ai_tool_definition_name}}")],
"spring.ai.tool, one series per tool name."),
("Mean tool latency (s)", "timeseries", "s",
[("sum by (spring_ai_tool_definition_name) (rate(spring_ai_tool_seconds_sum[1m])) / sum by (spring_ai_tool_definition_name) (rate(spring_ai_tool_seconds_count[1m]))", "{{spring_ai_tool_definition_name}}")],
"Sum over count of the tool timer."),
("Calls with no cost recorded", "timeseries", "short",
[("sum by (endpoint, reason) (increase(app_ai_unpriced_total[5m]))", "{{endpoint}} {{reason}}")],
"Calls that failed, reported no usage, or used a model with no configured price. Zero cost is never recorded silently."),
]
panels = []
for i, (title, kind, unit, queries, desc) in enumerate(PANELS):
panels.append({
"id": i + 1, "title": title, "type": kind, "description": desc, "datasource": DS,
"gridPos": {"h": 8, "w": 12, "x": (i % 2) * 12, "y": (i // 2) * 8},
"fieldConfig": {"defaults": {"unit": unit}, "overrides": []},
"targets": [{"refId": chr(65 + j), "datasource": DS, "expr": e, "legendFormat": l, "editorMode": "code", "range": True}
for j, (e, l) in enumerate(queries)],
})
dash = {"uid": "spring-ai-observability", "title": "Spring AI: tokens, latency and cost", "schemaVersion": 39, "version": 1,
"editable": True, "refresh": "5s", "time": {"from": "now-15m", "to": "now"}, "tags": ["spring-ai"], "panels": panels}
out = pathlib.Path(__file__).resolve().parent.parent / "dashboards" / "spring-ai-observability.json"
out.write_text(json.dumps(dash, indent=2) + "\n")
print("wrote", out, len(panels), "panels")