# Anthropic Messages Full-Flow Agentic (Web Search) — protocol-adaptive gateway
#
# A single Anthropic Messages gateway that runs the server-owned web-search loop
# through Praxis core's iterative_request_router (IRR), serves BOTH streaming and
# buffered clients from one pipeline, and adapts to BOTH backend wire formats
# from one config:
#
#   * a backend that speaks Anthropic Messages natively (for example vLLM's
#     `/v1/messages` endpoint), and
#   * a Chat-Completions-only backend (for example a vLLM deployment serving only
#     `/v1/chat/completions`), reached through the Anthropic <-> Chat Completions
#     translation filters.
#
# The top-level `router` classifies each request and freezes ONE logical cluster
# for it from the model, so the inference step resolves the same backend on every
# re-entry. Inside the step a single `load_balancer` with
# `cluster_source: bound_upstream` resolves that frozen binding to a concrete
# backend and publishes its `application_protocol`; the protocol-adapter filters
# after it are gated on `selected_upstream.application_protocol`, so the
# translation path runs only for the Chat Completions backend and stays fully
# inert on the native Anthropic path.
#
# `anthropic_web_search` selects the transport per request from the client's
# `stream` flag, with no operator opt-in:
#
#   * `"stream": true`  -> the terminal answer is streamed to the client as one
#     coherent Anthropic Messages SSE lifecycle. Intermediate model/search
#     transitions stay internal and the managed `WebSearch` tool-use block is
#     suppressed.
#   * `"stream": false` -> the same loop runs buffered and returns one final
#     Anthropic Messages JSON object.
#
# Filter ordering inside the inference step is deliberate. Response filters run in
# the reverse of request order, so `anthropic_web_search` is placed FIRST in the
# step: on the request phase it parses/rebuilds the Anthropic body before any
# translation, and on the response phase it runs LAST, after the response has been
# translated back into Anthropic Messages (on the chat path) or observed natively
# (on the native path). Each nested web-search callout continues the request trace
# with the same request/trace IDs and a fresh outbound span ID.
#
# Requires WEB_SEARCH_API_KEY (the search provider key; this example uses Tavily,
# whose key travels in the request body) and, for the Chat Completions backend,
# VLLM_API_KEY (the backend Bearer token injected toward that cluster). The native
# Messages backend listens on 127.0.0.1:8000 and the Chat Completions backend on
# 127.0.0.1:8001.
#
# Run the deterministic native model mock:
#   cargo run -p praxis-test-utils --example anthropic_messages_web_search_mock
#
# In a second terminal, run Praxis:
#   WEB_SEARCH_API_KEY="$TAVILY_API_KEY" VLLM_API_KEY="$VLLM_API_KEY" \
#     cargo run -p praxis-ai-proxy -- \
#     -c examples/configs/anthropic/full-flow-agentic.yaml
#
# Native Anthropic backend (streaming), routed by default:
#   curl -N http://127.0.0.1:8080/v1/messages \
#     -H 'content-type: application/json' \
#     -d '{"model":"openai/gpt-oss-20b","max_tokens":1024,"stream":true,"messages":[{"role":"user","content":"Use web search to look up potato, then summarize in one sentence."}],"tools":[{"name":"WebSearch","description":"Search the web","input_schema":{"type":"object","properties":{"query":{"type":"string"}},"required":["query"]}}]}'
#
# Chat-Completions-only backend (buffered), routed by model:
#   curl http://127.0.0.1:8080/v1/messages \
#     -H 'content-type: application/json' \
#     -d '{"model":"Qwen/Qwen3-8B","max_tokens":512,"messages":[{"role":"user","content":"Use web search to look up potato, then summarize in one sentence."}],"tools":[{"name":"WebSearch","description":"Search the web","input_schema":{"type":"object","properties":{"query":{"type":"string"}},"required":["query"]}}],"tool_choice":{"type":"tool","name":"WebSearch"}}'
#
# Send the same body with "stream": true (or false) to switch transport.

listeners:
  - name: anthropic-full-flow-agentic
    address: "127.0.0.1:8080"
    filter_chains: [full-flow-agentic]

filter_chains:
  - name: full-flow-agentic
    filters:
      - filter: trace_context
        # Establish typed correlation before IRR snapshots request extensions;
        # Praxis core creates a fresh span ID for each search callout hop.

      - filter: anthropic_messages_format
        on_invalid: reject
      - filter: anthropic_validate

      # Top-level binding router: classify the request and freeze ONE logical
      # cluster for it so the inference step resolves the same backend on every
      # re-entry. The classifier promotes the model to `x-praxis-ai-model`.
      - filter: router
        routes:
          # Chat-Completions-only backends: route the model to the translation
          # path. Replace this model policy with your own deployment's models.
          - path_prefix: "/v1/messages"
            headers:
              x-praxis-ai-model: "Qwen/Qwen3-8B"
            cluster: chat-completions-backend
          # Default: a backend that speaks the Anthropic Messages wire format
          # natively.
          - path_prefix: "/v1/messages"
            cluster: messages-backend

      - filter: iterative_request_router
        initial_step: inference
        max_iterations: 6
        timeout_ms: 90000
        steps:
          - name: inference
            filters:
              # FIRST in the step so it runs LAST on the response phase, observing
              # the response already translated back to Anthropic Messages on the
              # chat path (or the native Anthropic response on the native path).
              - filter: anthropic_web_search
                provider: tavily
                api_key: ${WEB_SEARCH_API_KEY}
                default_context_size: medium
                timeout_ms: 10000

              # Resolve the frozen binding to a concrete backend and publish its
              # `application_protocol`, which gates the protocol adapters below.
              # This unconditional load balancer also satisfies core's requirement
              # that a selected_upstream-gated filter follow an unconditional one.
              - filter: load_balancer
                cluster_source: bound_upstream
                clusters:
                  - name: messages-backend
                    http:
                      application_protocol: anthropic_messages
                    endpoints: ["127.0.0.1:8000"]
                  - name: chat-completions-backend
                    http:
                      application_protocol: openai_chat_completions
                    endpoints: ["127.0.0.1:8001"]

              # Native path only: set the default Anthropic version the backend
              # expects and restore the caller's version/beta across re-entry.
              - filter: anthropic_messages_protocol
                default_version: "2023-06-01"
                conditions:
                  - when:
                      selected_upstream:
                        application_protocol: anthropic_messages

              # Chat path only: translate the Anthropic request into Chat
              # Completions and the non-streaming response back into Anthropic
              # Messages. Also strips the client's native Anthropic auth header.
              - filter: anthropic_messages_to_chat_completions
                max_body_bytes: 1048576
                conditions:
                  - when:
                      selected_upstream:
                        application_protocol: openai_chat_completions

              # Chat path only: convert streaming Chat Completions SSE back into
              # Anthropic Messages SSE.
              - filter: anthropic_messages_to_chat_completions_stream
                max_partial_event_bytes: 10485760
                max_tool_blocks: 10000
                conditions:
                  - when:
                      selected_upstream:
                        application_protocol: openai_chat_completions

              # Chat path only: map the Anthropic message endpoint onto the
              # backend's Chat Completions endpoint.
              - filter: path_rewrite
                replace:
                  pattern: "^/v1/messages$"
                  replacement: "/v1/chat/completions"
                conditions:
                  - when:
                      selected_upstream:
                        application_protocol: openai_chat_completions
                      path_prefix: "/v1/messages"
                      methods: [POST]

              # Inject the backend's own Bearer token for the Chat Completions
              # cluster after the load balancer selects it; strip any
              # client-supplied Authorization first. A security filter cannot carry
              # request conditions, so it is scoped by cluster instead: on the
              # native path `ctx.cluster` is `messages-backend`, so this is a no-op.
              - filter: credential_injection
                clusters:
                  - name: chat-completions-backend
                    header: Authorization
                    env_var: VLLM_API_KEY
                    header_prefix: "Bearer "
                    strip_client_credential: true
            on_result:
              - filter: anthropic_web_search
                key: action
                value: loop
                next: inference
              - default: true
                done: true

insecure_options:
  allow_private_endpoints: true # example proxies to local backends
