Add observability module: Spring AI built-in meters and spans, cost per endpoint from token usage, Prometheus and Grafana dashboard
Co-Authored-By: Claude Sonnet 5.5 <[email protected]> Claude-Session: https://claude.ai/code/session_01JXVi2GMQ7bR5EmbUFdDj7N
This commit is contained in:
@@ -0,0 +1,16 @@
|
||||
# Meters the framework creates for 3 requests (/ask, /weather with one tool call, /stream)
|
||||
|
||||
gen_ai.client.operation TIMER [error, gen_ai.operation.name, gen_ai.request.model, gen_ai.response.model, gen_ai.system]
|
||||
gen_ai.client.operation.active LONG_TASK_TIMER [gen_ai.operation.name, gen_ai.request.model, gen_ai.response.model, gen_ai.system]
|
||||
gen_ai.client.token.usage COUNTER [gen_ai.operation.name, gen_ai.request.model, gen_ai.response.model, gen_ai.system, gen_ai.token.type]
|
||||
spring.ai.advisor TIMER [error, gen_ai.operation.name, gen_ai.system, spring.ai.advisor.name, spring.ai.kind]
|
||||
spring.ai.advisor.active LONG_TASK_TIMER [gen_ai.operation.name, gen_ai.system, spring.ai.advisor.name, spring.ai.kind]
|
||||
spring.ai.chat.client TIMER [app.endpoint, error, gen_ai.operation.name, gen_ai.system, spring.ai.chat.client.stream, spring.ai.kind]
|
||||
spring.ai.chat.client.active LONG_TASK_TIMER [app.endpoint, gen_ai.operation.name, gen_ai.system, spring.ai.chat.client.stream, spring.ai.kind]
|
||||
spring.ai.tool TIMER [error, gen_ai.operation.name, gen_ai.system, spring.ai.kind, spring.ai.tool.definition.name, spring.ai.tool.type]
|
||||
spring.ai.tool.active LONG_TASK_TIMER [gen_ai.operation.name, gen_ai.system, spring.ai.kind, spring.ai.tool.definition.name, spring.ai.tool.type]
|
||||
|
||||
model calls recorded (gen_ai.client.operation, error=none): 4
|
||||
chat client calls recorded (spring.ai.chat.client): 3
|
||||
tool executions recorded (spring.ai.tool): 1
|
||||
tokens: input=24 output=29 total=53
|
||||
@@ -0,0 +1,24 @@
|
||||
# Span tree for GET /weather?city=Pune (one tool call, two model calls)
|
||||
|
||||
http get /weather [SERVER]
|
||||
spring_ai chat_client [INTERNAL]
|
||||
tool _calling [INTERNAL]
|
||||
call [INTERNAL]
|
||||
chat gpt-4o-mini [INTERNAL]
|
||||
POST [CLIENT]
|
||||
execute_tool getWeather [INTERNAL]
|
||||
call [INTERNAL]
|
||||
chat gpt-4o-mini [INTERNAL]
|
||||
POST [CLIENT]
|
||||
|
||||
attribute keys on the model-call spans:
|
||||
gen_ai.operation.name
|
||||
gen_ai.request.model
|
||||
gen_ai.response.finish_reasons
|
||||
gen_ai.response.id
|
||||
gen_ai.response.model
|
||||
gen_ai.system
|
||||
gen_ai.usage.input_tokens
|
||||
gen_ai.usage.output_tokens
|
||||
gen_ai.usage.total_tokens
|
||||
spring.ai.model.request.tool.names
|
||||
@@ -0,0 +1,10 @@
|
||||
# Tokens and cost per endpoint (illustrative price: 0.15 USD in / 0.60 USD out per million tokens)
|
||||
|
||||
endpoint input output cost USD
|
||||
ask 1 4 0.00000255
|
||||
summarize 511 13 0.00008445
|
||||
weather 19 18 0.00001365
|
||||
|
||||
framework model-level tokens: input=531 output=35
|
||||
our per-endpoint tokens : input=531 output=35
|
||||
total cost USD: 0.00010065
|
||||
@@ -0,0 +1,9 @@
|
||||
# Streaming asks for usage; a server that omits it is counted as unpriced
|
||||
|
||||
stream request body asks for usage (stream_options.include_usage): true
|
||||
request fragment: "stream":true
|
||||
|
||||
two calls to a server that omits the usage block (/ask and /stream):
|
||||
gen_ai.client.token.usage total, before 7 -> after 7
|
||||
app.ai.unpriced{reason=no-usage} for ask : 1
|
||||
app.ai.unpriced{reason=no-usage} for stream: 1
|
||||
@@ -0,0 +1,8 @@
|
||||
# One failed user request (the provider returns HTTP 500)
|
||||
|
||||
HTTP requests the provider received for that one call: 4
|
||||
model-call timers recorded:
|
||||
error=InternalServerException response.model=none count=1
|
||||
chat-client timers with an error: 1
|
||||
app.ai.unpriced{endpoint=ask,reason=no-usage}: 1
|
||||
app.ai.cost.usd recorded for the failed call: 0
|
||||
@@ -0,0 +1,7 @@
|
||||
# Where does the prompt text "ticket-4711-card-ending-0042" appear in telemetry?
|
||||
|
||||
setting log lines with it / span attributes with it / span events with it
|
||||
defaults 0 / 0 / 0
|
||||
spring.ai.chat.observations.log-prompt=true (+ client) 1 / 0 / 0
|
||||
|
||||
span attributes carrying the text when enabled: []
|
||||
@@ -0,0 +1,7 @@
|
||||
# Bucket series for gen_ai_client_operation_seconds after 2 calls
|
||||
|
||||
setting _bucket series / _sum series
|
||||
defaults 0 / 1
|
||||
percentiles-histogram.gen_ai.client.operation=true 69 / 1
|
||||
|
||||
first bucket series when enabled: gen_ai_client_operation_seconds_bucket{le="0.001"}
|
||||
@@ -0,0 +1,12 @@
|
||||
# Every dashboard query, run against Prometheus and through Grafana's query API after scripts/load.sh
|
||||
|
||||
panel prom series grafana frames
|
||||
Cost per hour by endpoint (USD, illustrative pri [A] 4 4
|
||||
Tokens per minute by endpoint and direction [A] 8 8
|
||||
Cost per request by endpoint (USD) [A] 4 4
|
||||
Model call latency p50 / p95 (s) [A] 1 1
|
||||
Model call latency p50 / p95 (s) [B] 1 1
|
||||
Failed model calls (share) [A] 1 1
|
||||
Tool calls per minute [A] 1 1
|
||||
Mean tool latency (s) [A] 1 1
|
||||
Calls with no cost recorded [A] 1 1
|
||||
@@ -0,0 +1,26 @@
|
||||
# Prometheus metric families this app exposes at /actuator/prometheus (TYPE lines only)
|
||||
|
||||
# TYPE app_ai_cost_usd_total counter
|
||||
# TYPE app_ai_tokens_total counter
|
||||
# TYPE app_ai_unpriced_total counter
|
||||
# TYPE gen_ai_client_operation_active_seconds histogram
|
||||
# TYPE gen_ai_client_operation_active_seconds_gcount gauge
|
||||
# TYPE gen_ai_client_operation_active_seconds_gsum gauge
|
||||
# TYPE gen_ai_client_operation_active_seconds_max gauge
|
||||
# TYPE gen_ai_client_operation_seconds histogram
|
||||
# TYPE gen_ai_client_operation_seconds_max gauge
|
||||
# TYPE gen_ai_client_token_usage_total counter
|
||||
# TYPE spring_ai_advisor_active_seconds summary
|
||||
# TYPE spring_ai_advisor_active_seconds_max gauge
|
||||
# TYPE spring_ai_advisor_seconds summary
|
||||
# TYPE spring_ai_advisor_seconds_max gauge
|
||||
# TYPE spring_ai_chat_client_active_seconds histogram
|
||||
# TYPE spring_ai_chat_client_active_seconds_gcount gauge
|
||||
# TYPE spring_ai_chat_client_active_seconds_gsum gauge
|
||||
# TYPE spring_ai_chat_client_active_seconds_max gauge
|
||||
# TYPE spring_ai_chat_client_seconds histogram
|
||||
# TYPE spring_ai_chat_client_seconds_max gauge
|
||||
# TYPE spring_ai_tool_active_seconds summary
|
||||
# TYPE spring_ai_tool_active_seconds_max gauge
|
||||
# TYPE spring_ai_tool_seconds summary
|
||||
# TYPE spring_ai_tool_seconds_max gauge
|
||||
Reference in New Issue
Block a user