# Token Rate Limiting
#
# Reserves an estimated token cost at admission time and reconciles
# that reservation against actual provider-reported usage once the
# response completes. Rejects with 429 when the bucket can't cover
# the estimate.
#
# Usage:
#   cargo run -p praxis-ai-proxy -- -c examples/configs/token-rate-limit.yaml
#   curl -i http://localhost:8080/v1/chat/completions -d '{"model":"gpt-4","messages":[{"role":"user","content":"hi"}]}'
#
# This is the agreed M1/M2/M6 core of the token rate limiting proposal
# (00121_token-rate-limiting.md in praxis-proxy/enhancements, tracked by
# epic ai#121), plus M3 estimation, M4 token-type weights, and the M7
# observability contract from ai#883:
#   - M1: a single sliding-window budget (one catch-all rule)
#   - M2: reservation-based admission, reconciled against actual usage
#   - M4: per-type weights at reconciliation (cached input cheaper, etc.)
#   - M6: 429 responses with token-denominated rate limit headers
#   - M7: group-level Prometheus metrics, request spans, and accounting logs
#
# `rules:` (per `ai#789`/`praxis#551`) lets each rule pick its own
# admission algorithm and match condition; a rule with no `match:` is a
# catch-all applying to every request that reaches it. This example
# uses one such catch-all rule. `window`/`capacity` match the proposal's
# own field names: "Windows are sliding: a `window: 1h` budget tracks
# usage in the most recent 60 minutes from the current instant." State
# defaults to in-process; set a filter-level `backend: {kind: valkey,
# url: ..., namespace: ...}` (a sibling of `rules:`, not per-rule) to
# share every rule's budget across every gateway instance/replica over
# one shared connection. The Valkey backend uses plain commands; two
# admissions racing on one key may briefly overshoot the budget by
# their estimates.
#
# Set `key:` to partition each matching rule's budget. Scalars
# (`global`, `authenticated_subject`, `ip`, `model`), a header mapping
# (`header: x-tenant-id`), or a list of those (composite, e.g. subject
# + model) are accepted. Missing dimensions reject by default; set
# `missing: fallback` to drop an absent dimension (header-absent traffic
# then shares the global bucket). Values are hashed before storage.
# The default is `key: global`, one shared budget per rule.
#
# `default_weights` (M4) scale provider-reported token types when the
# reservation is reconciled. Cache read/write are a breakdown of
# token.input (not an addition); reasoning is nested in token.output
# on OpenAI/Anthropic and additive on Google. Omitted types default to
# 1.0. Admission still reserves `reserved_tokens` unweighted (M3).
#
# Observability requires no filter configuration. Metrics use bounded
# rule/algorithm/result/backend labels and never identify a user. The
# budget_remaining is the remaining budget for the key of the most
# recent admission decision on this replica; active_keys shows how
# many keys the backend currently retains. Both are snapshots. Build
# with `opentelemetry` to add the request-scoped decision span. Structured
# best-effort audit records use the dedicated
# `praxis_ai::token_rate_limit::accounting` log target at INFO (one line
# per admission or settlement); keep only failures with
# `runtime.log_overrides: {"praxis_ai::token_rate_limit::accounting": "warn"}`.
#
# Deliberately not yet built (see filters/src/token_rate_limit/mod.rs
# for the full rationale): CEL-expression matchers/keys, and
# billing-grade metering (S3).
#
# token_rate_limit is declared *before* token_count: response hooks
# run in reverse declared order, so token_count's on_response_body
# (which writes token.total to filter_metadata) runs before
# token_rate_limit's on_response_body reads it back to reconcile the
# reservation. This mirrors the token_usage_headers/token_count
# ordering in examples/configs/token-counting.yaml.
#
# Assumes request identity is already resolved upstream (this filter
# doesn't authenticate callers) -- a catch-all rule like the one below
# reserves quota for every request that reaches it, including probes
# and health checks. Scope with an explicit `match:` per rule, or place
# an identity/auth filter earlier in the pipeline. Tracked in grid#101.
#
# The X-RateLimit-*-Tokens response headers always reflect the
# reservation-time snapshot, never this response's own reconciliation
# — headers are committed before the body (and thus actual usage) is
# known. Reconciliation still affects every *subsequent* request's
# admission decision.
#
# Supported token_count provider values:
#   openai | anthropic | google | bedrock | azure

listeners:
  - name: default
    address: "127.0.0.1:8080"
    filter_chains:
      - main

filter_chains:
  - name: main
    filters:
      - filter: router
        routes:
          - path: "/v1/chat/completions"
            cluster: backend

      - filter: token_rate_limit
        key: global
        default_weights:
          input: 1.0
          output: 1.0
          cached_input: 0.1    # prompt-cache hits (token.cache_read)
          cache_write: 1.25    # prompt-cache writes (token.cache_write)
          reasoning: 0.9       # thinking tokens (token.reasoning)
        rules:
          - name: default
            algorithm: sliding_window
            window: 1h             # sliding window duration
            capacity: 100000       # max tokens admitted within `window`
            reserved_tokens: 500   # fixed cost reserved per request at admission

      - filter: token_count
        provider: openai   # openai | anthropic | google | bedrock | azure

      - filter: access_log

      - filter: load_balancer
        clusters:
          - name: backend
            endpoints:
              - "127.0.0.1:3000"

insecure_options:
  allow_private_endpoints: true # example proxies to a local backend
