# Stream Usage Injection
#
# Ensures every streaming OpenAI Chat Completions request carries
# stream_options.include_usage = true so the upstream response includes
# token usage in the final SSE event.
#
# Without this opt-in, OpenAI streams carry no usage object and the
# gateway meters a legitimate 0 — the client is responsible for
# requesting it, but a metering gateway cannot rely on client behavior.
#
# Usage:
#   cargo run -p praxis-ai-proxy -- -c examples/configs/stream-usage-inject.yaml
#   curl http://localhost:8080/v1/chat/completions \
#     -d '{"model":"gpt-4","stream":true,"messages":[{"role":"user","content":"hi"}]}'
#
# The filter is a no-op for non-streaming requests, requests that
# already carry the opt-in, provider-invalid stream_options values,
# and non-JSON bodies.
#
# The filter buffers the request body up to max_body_bytes (default
# 10 MiB, ceiling 64 MiB), and that cap applies to every request
# traversing the chain — not only streaming chat completions. Raise it
# on a deployment with larger legitimate requests:
#   - filter: stream_usage_inject
#     max_body_bytes: 10485760

listeners:
  - name: default
    address: "127.0.0.1:8080"
    filter_chains:
      - main

filter_chains:
  - name: main
    filters:
      - filter: router
        routes:
          - path_prefix: "/"
            cluster: backend

      - filter: stream_usage_inject

      - filter: token_count
        provider: openai

      - filter: token_usage_headers

      - filter: load_balancer
        clusters:
          - name: backend
            endpoints:
              - "127.0.0.1:3000"

insecure_options:
  allow_private_endpoints: true # example proxies to local backends
