# Responses and Chat Completions Model Rewrite
#
# Rewrites or injects the top-level `model` field in Responses API
# and Chat Completions request bodies before forwarding to the
# inference backend.
#
# This pipeline uses `openai_responses_format` to classify the
# request, then `openai_responses_model_rewrite` to apply alias
# mapping. The router uses the effective model header to select
# the backend cluster, so routing reflects the rewritten model,
# not the original client-facing name.
#
# Use case: Codex or other Responses API clients send
# `model: "codex-mini-2026-06-24"` and the proxy transparently
# rewrites it to a locally-hosted model before forwarding. The same
# alias table covers `POST /v1/chat/completions`, so a gateway that
# advertises one client-facing name can forward the provider-specific
# target model on either API.
#
# Build:
#   cargo build -p praxis-ai-proxy

listeners:
  - name: ai-gateway
    address: "127.0.0.1:8080"
    filter_chains: [model-rewrite-pipeline]

filter_chains:
  - name: model-rewrite-pipeline
    filters:
      - filter: openai_responses_format
        on_invalid: continue
        headers:
          format: x-praxis-ai-format
          model: x-praxis-ai-model

      - filter: openai_responses_model_rewrite
        default_model: "llama-3.3-70b"
        model_aliases:
          "codex-*": "llama-3.3-70b"
          "gpt-4.1-mini": "qwen-2.5-72b"
        headers:
          effective_model: x-praxis-ai-effective-model
          original_model: x-praxis-ai-original-model

      - filter: router
        routes:
          - path: "/v1/responses"
            headers:
              x-praxis-ai-effective-model: "llama-3.3-70b"
            cluster: "llama-backend"
          - path: "/v1/responses"
            headers:
              x-praxis-ai-effective-model: "qwen-2.5-72b"
            cluster: "qwen-backend"
          - path: "/v1/chat/completions"
            headers:
              x-praxis-ai-effective-model: "llama-3.3-70b"
            cluster: "llama-backend"
          - path: "/v1/chat/completions"
            headers:
              x-praxis-ai-effective-model: "qwen-2.5-72b"
            cluster: "qwen-backend"
          - path_prefix: "/"
            cluster: "default-backend"

      - filter: load_balancer
        clusters:
          - name: "llama-backend"
            endpoints:
              - "127.0.0.1:3001"
          - name: "qwen-backend"
            endpoints:
              - "127.0.0.1:3002"
          - name: "default-backend"
            endpoints:
              - "127.0.0.1:3003"

insecure_options:
  allow_private_endpoints: true # example proxies to local backends
