# Scale orders on in-flight requests per pod, not CPU. Target: 5 in flight on average. apiVersion: autoscaling/v2 kind: HorizontalPodAutoscaler metadata: {name: orders, namespace: demo} spec: scaleTargetRef: {apiVersion: apps/v1, kind: Deployment, name: orders} minReplicas: 1 maxReplicas: 4 metrics: - type: Pods pods: metric: {name: app_inflight_requests} target: {type: AverageValue, averageValue: "5"} behavior: scaleUp: stabilizationWindowSeconds: 0 scaleDown: # The default is 300 s. Shortened so the demo shows a scale-down inside a few minutes; # keep the default (or longer) in production. stabilizationWindowSeconds: 30