# Token Counting
#
# Extracts token usage from AI inference responses (streaming and
# non-streaming) and makes counts available to downstream filters
# via filter metadata as token.input, token.output, and token.total.
#
# For providers that support prompt caching, the cached portion of the
# input is also reported as token.cache_read and token.cache_write.
# Both are a breakdown of token.input, not an addition to it, so
# summing them with token.input would double-count.
#
# Each cache key is set only when the provider reported that count. An
# absent key means the response carried no cache information; a zero
# means the provider reported a cache miss. Providers whose API has no
# cache write concept, such as Google, never set token.cache_write;
# OpenAI reports cache writes as cache_write_tokens on both Chat
# Completions and Responses API usage.
#
# Reasoning / thinking tokens are reported as token.reasoning when the
# provider sends them:
#   - OpenAI / Azure Chat Completions:
#     usage.completion_tokens_details.reasoning_tokens
#     (a breakdown of token.output, not an addition)
#   - OpenAI / Azure Responses API:
#     usage.output_tokens_details.reasoning_tokens
#     (a breakdown of token.output, not an addition)
#   - Anthropic: usage.output_tokens_details.thinking_tokens
#     (a breakdown of token.output; on streams, the final message_delta)
#   - Google: usageMetadata.thoughtsTokenCount
#     (separate from candidatesTokenCount / token.output; already in
#     the provider total when totalTokenCount is present, otherwise
#     included in the computed fallback total)
#   - Bedrock Converse: not reported (no documented field)
# An absent key means the provider did not report reasoning; a zero
# means it reported none.
#
# Usage:
#   cargo run -p praxis-ai-proxy -- -c examples/configs/token-counting.yaml
#   curl http://localhost:8080/v1/chat/completions -d '{"model":"gpt-4","messages":[{"role":"user","content":"hi"}]}'
#
# token_count reads the response body, extracts provider-specific
# token usage, and writes the counts to filter_metadata.
#
# token_usage_headers injects Praxis-Token-Input, Praxis-Token-Output,
# and Praxis-Token-Total into the downstream response — but only when
# counts are available during the response-header phase. token_count
# extracts counts from the response body inside on_response_body,
# which runs after response headers have already been sent
# downstream, so for every provider currently supported here the
# counts remain in filter_metadata for other response-phase filters
# to read, but the Praxis-Token-* headers are not injected.
#
# access_log demonstrates that token_count composes cleanly alongside
# other response-phase filters in the same pipeline.
#
# Supported provider values:
#   openai | anthropic | google | bedrock | azure

listeners:
  - name: default
    address: "127.0.0.1:8080"
    filter_chains:
      - main

filter_chains:
  - name: main
    filters:
      - filter: router
        routes:
          - path_prefix: "/"
            cluster: backend

      # token_usage_headers is declared before token_count: response
      # hooks run in reverse declared order, so token_count's
      # on_response_body (which sets filter_metadata) runs before
      # token_usage_headers reads it.
      - filter: token_usage_headers

      - filter: token_count
        provider: openai   # openai | anthropic | google | bedrock | azure

      - filter: access_log

      - filter: load_balancer
        clusters:
          - name: backend
            endpoints:
              - "127.0.0.1:3000"

insecure_options:
  allow_private_endpoints: true # example proxies to local backends
