{ "uid": "spring-ai-observability", "title": "Spring AI: tokens, latency and cost", "schemaVersion": 39, "version": 1, "editable": true, "refresh": "5s", "time": { "from": "now-15m", "to": "now" }, "tags": [ "spring-ai" ], "panels": [ { "id": 1, "title": "Cost per hour by endpoint (USD, illustrative prices)", "type": "timeseries", "description": "Rate of app_ai_cost_usd_total scaled to an hourly figure. Prices come from ai.pricing in application.yml.", "datasource": { "type": "prometheus", "uid": "prom" }, "gridPos": { "h": 8, "w": 12, "x": 0, "y": 0 }, "fieldConfig": { "defaults": { "unit": "currencyUSD" }, "overrides": [] }, "targets": [ { "refId": "A", "datasource": { "type": "prometheus", "uid": "prom" }, "expr": "sum by (endpoint) (rate(app_ai_cost_usd_total[1m])) * 3600", "legendFormat": "{{endpoint}}", "editorMode": "code", "range": true } ] }, { "id": 2, "title": "Tokens per minute by endpoint and direction", "type": "timeseries", "description": "Input and output tokens as the provider reported them.", "datasource": { "type": "prometheus", "uid": "prom" }, "gridPos": { "h": 8, "w": 12, "x": 12, "y": 0 }, "fieldConfig": { "defaults": { "unit": "short" }, "overrides": [] }, "targets": [ { "refId": "A", "datasource": { "type": "prometheus", "uid": "prom" }, "expr": "sum by (endpoint, type) (rate(app_ai_tokens_total[1m])) * 60", "legendFormat": "{{endpoint}} {{type}}", "editorMode": "code", "range": true } ] }, { "id": 3, "title": "Cost per request by endpoint (USD)", "type": "bargauge", "description": "Cost divided by successful chat client calls. The two sides use different label names (endpoint vs app_endpoint), so label_replace renames one before the division.", "datasource": { "type": "prometheus", "uid": "prom" }, "gridPos": { "h": 8, "w": 12, "x": 0, "y": 8 }, "fieldConfig": { "defaults": { "unit": "currencyUSD" }, "overrides": [] }, "targets": [ { "refId": "A", "datasource": { "type": "prometheus", "uid": "prom" }, "expr": "sum by (endpoint) (increase(app_ai_cost_usd_total[5m])) / on (endpoint) label_replace(sum by (app_endpoint) (increase(spring_ai_chat_client_seconds_count{error=\"none\"}[5m])), \"endpoint\", \"$1\", \"app_endpoint\", \"(.*)\")", "legendFormat": "{{endpoint}}", "editorMode": "code", "range": true } ] }, { "id": 4, "title": "Model call latency p50 / p95 (s)", "type": "timeseries", "description": "gen_ai.client.operation, one sample per HTTP call to the provider. Needs the percentiles-histogram property.", "datasource": { "type": "prometheus", "uid": "prom" }, "gridPos": { "h": 8, "w": 12, "x": 12, "y": 8 }, "fieldConfig": { "defaults": { "unit": "s" }, "overrides": [] }, "targets": [ { "refId": "A", "datasource": { "type": "prometheus", "uid": "prom" }, "expr": "histogram_quantile(0.50, sum by (le) (rate(gen_ai_client_operation_seconds_bucket{error=\"none\"}[1m])))", "legendFormat": "p50", "editorMode": "code", "range": true }, { "refId": "B", "datasource": { "type": "prometheus", "uid": "prom" }, "expr": "histogram_quantile(0.95, sum by (le) (rate(gen_ai_client_operation_seconds_bucket{error=\"none\"}[1m])))", "legendFormat": "p95", "editorMode": "code", "range": true } ] }, { "id": 5, "title": "Failed model calls (share)", "type": "stat", "description": "Calls whose error tag is not none, over all calls.", "datasource": { "type": "prometheus", "uid": "prom" }, "gridPos": { "h": 8, "w": 12, "x": 0, "y": 16 }, "fieldConfig": { "defaults": { "unit": "percentunit" }, "overrides": [] }, "targets": [ { "refId": "A", "datasource": { "type": "prometheus", "uid": "prom" }, "expr": "sum(rate(gen_ai_client_operation_seconds_count{error!=\"none\"}[5m])) / sum(rate(gen_ai_client_operation_seconds_count[5m]))", "legendFormat": "failed", "editorMode": "code", "range": true } ] }, { "id": 6, "title": "Tool calls per minute", "type": "timeseries", "description": "spring.ai.tool, one series per tool name.", "datasource": { "type": "prometheus", "uid": "prom" }, "gridPos": { "h": 8, "w": 12, "x": 12, "y": 16 }, "fieldConfig": { "defaults": { "unit": "short" }, "overrides": [] }, "targets": [ { "refId": "A", "datasource": { "type": "prometheus", "uid": "prom" }, "expr": "sum by (spring_ai_tool_definition_name) (rate(spring_ai_tool_seconds_count[1m])) * 60", "legendFormat": "{{spring_ai_tool_definition_name}}", "editorMode": "code", "range": true } ] }, { "id": 7, "title": "Mean tool latency (s)", "type": "timeseries", "description": "Sum over count of the tool timer.", "datasource": { "type": "prometheus", "uid": "prom" }, "gridPos": { "h": 8, "w": 12, "x": 0, "y": 24 }, "fieldConfig": { "defaults": { "unit": "s" }, "overrides": [] }, "targets": [ { "refId": "A", "datasource": { "type": "prometheus", "uid": "prom" }, "expr": "sum by (spring_ai_tool_definition_name) (rate(spring_ai_tool_seconds_sum[1m])) / sum by (spring_ai_tool_definition_name) (rate(spring_ai_tool_seconds_count[1m]))", "legendFormat": "{{spring_ai_tool_definition_name}}", "editorMode": "code", "range": true } ] }, { "id": 8, "title": "Calls with no cost recorded", "type": "timeseries", "description": "Calls that failed, reported no usage, or used a model with no configured price. Zero cost is never recorded silently.", "datasource": { "type": "prometheus", "uid": "prom" }, "gridPos": { "h": 8, "w": 12, "x": 12, "y": 24 }, "fieldConfig": { "defaults": { "unit": "short" }, "overrides": [] }, "targets": [ { "refId": "A", "datasource": { "type": "prometheus", "uid": "prom" }, "expr": "sum by (endpoint, reason) (increase(app_ai_unpriced_total[5m]))", "legendFormat": "{{endpoint}} {{reason}}", "editorMode": "code", "range": true } ] } ] }