# Responses API Full Flow — Unified Agentic Gateway
# Requires `--features openai-file-resolve-filter,openai-conversations,openai-mcp-tools,store-sqlite` because these filters are opt-in.
#
# Runs the complete Responses API pipeline through an agentic
# iterative_request_router that executes hosted file_search, web_search,
# and MCP tool calls in a model-tool-model loop, persisting both buffered
# and streaming (`stream: true`) responses. This is the single config an
# operator would deploy for a Responses API gateway that must serve the
# full agentic loop, the Conversations API, and the dedicated
# prompts/embeddings/files/vector_stores services behind one listener. The
# request trace is continued through every nested web-search, OGX, and MCP
# callout with a fresh span ID per hop.
#
# Pre-IRR filters run once on the client request to set up shared state
# (ResponsesState) which persists across all IRR iterations via extension
# swapping:
#
#   0. state_owner maps deployment-specific trusted ingress headers into one
#      normalized tenant/issuer/subject context for owner-aware filters.
#      callout_credentials captures per-user Brave and OGX provider keys from
#      trusted x-user-brave-key and x-user-ogx-key ingress headers into the
#      `brave_search` and `ogx_files` slots, then strips both source headers
#      before inference or IRR processing. The trusted boundary MUST
#      unconditionally delete-then-set those source headers on every request;
#      append-on-success is insufficient because this filter cannot authenticate
#      header provenance. The IRR carries the resulting typed request extension
#      across iterations, while each callout explicitly selects its slot for its
#      exact-authority provider destination.
#      project_state_owner_headers then projects that context as x-tenant-id and
#      x-user-id for OGX and explicitly allowlisted callouts. The raw ingress
#      names are stripped, and the projection is repeated inside the IRR step
#      because it owns a separate destination request lifecycle.
#      These adapters do not themselves change persistence predicates: this
#      context-only configuration must be combined with owner-scoped store
#      enforcement before it is exposed as a multi-owner state service.
#
#   0. openai_operation publishes the typed registry match consumed by
#      openai_conversations. It must remain earlier in this filter chain.
#
#   1. openai_conversations handles CRUD for the Conversations API
#      (/v1/conversations) and sets has_conversation metadata when a
#      Responses request carries a conversation field. On the response
#      path it appends input+output items back to the conversation after
#      a successful non-streaming response.
#
#   2. openai_responses_format classifies the request and promotes
#      format, model, stream, and mode to internal routing headers.
#      on_invalid is left at the default (passthrough) so that
#      Conversations API traffic is not rejected before the conversations
#      filter can handle it in the body phase.
#
#   3. openai_responses_validate checks JSON syntax, generates response
#      and conversation IDs, and writes responses.* metadata for
#      downstream filters. Provider-owned parameter combinations pass
#      through.
#
#   4. openai_tool_parse parses the tools array and tool_choice from
#      Responses API requests and promotes summary facts (has_tools,
#      has_web_search, tool_choice, function_count, etc.) to metadata and
#      filter results. Does not mutate the request body.
#
#   5. openai_response_store persists responses to SQLite and registers
#      the store backend so downstream filters (rehydrate, compact) can
#      read from it. It runs PRE-IRR (before the iterative_request_router
#      below), so on the response path — which runs in reverse config
#      order — it persists AFTER the IRR step has composed the terminal
#      event. This is what makes streaming persistence work: the store
#      reads the accumulated response object that openai_stream_events
#      builds inside the IRR.
#
#   6. openai_responses_rehydrate loads conversation context from
#      previous_response_id or conversation field by fetching stored
#      responses/conversations and prepending their message history to the
#      current request. previous_response_id takes precedence when both
#      are present.
#
#   7. openai_file_resolve resolves file_id references in the current
#      input and rehydrated history through a configured Files API. Runs
#      after rehydrate so stored history is available, and before
#      responses_proxy so the rebuilt upstream body contains file_data or
#      image_url rather than stale file_id references.
#
#   8. openai_doc_extract converts input_file content parts to input_text
#      for backends that do not natively support input_file (e.g. vLLM,
#      llm-d). Text-safe content (text/*, application/json,
#      application/xml) is decoded from base64, validated as UTF-8, and
#      forwarded as plain text. Unsupported formats are left unchanged
#      (on_unsupported: continue). This filter is optional — omit it when
#      the inference backend natively supports input_file.
#
#   9. openai_mcp_tool_resolve resolves MCP tool entries from the tools
#      array by calling tools/list on each upstream MCP server. Runs after
#      rehydrate so previous_tools is available for cross-request caching.
#      Writes mcp_tool_map to ResponsesState for downstream tool dispatch.
#      Projected identity headers are forwarded only to operator-configured
#      connector_id targets, never arbitrary client-selected server_url values.
#
#   10. The bypass gateway (a no-op `headers` carrier hosting a
#      branch_chains bypass) routes everything EXCEPT a classified
#      `POST /v1/responses` create request directly to a backend. The IRR
#      below fully owns the sub-request lifecycle, which is incompatible
#      with WebSocket upgrades and with the dedicated
#      prompts/embeddings/files/vector_stores services — so those, plus
#      `GET /v1/responses` WebSocket handshakes, take the bypass branch.
#      The carrier's `unless` condition skips it only when the request is
#      `POST /v1/responses` AND classified as x-praxis-ai-format:
#      openai_responses, so genuine create requests fall through to the
#      IRR while a Chat Completions or Anthropic Messages body posted to
#      /v1/responses (or a GET WebSocket handshake) still runs the
#      carrier. The `unless` gate is a positive allow-list of one format:
#      any body that is NOT classified openai_responses (Chat Completions,
#      Anthropic Messages, or any future/unclassified format) runs the
#      carrier and gets a route-miss (404) — no per-format reject rule is
#      needed. Discrimination lives here, PRE-IRR, because the IRR strips
#      x-praxis-* headers from its sub-request (see step 11). Inside the
#      bypass, the `/v1/responses` route matches on the WebSocket
#      `Upgrade` header (routes cannot match on HTTP method), so a WS
#      handshake reaches the backend while a non-Responses body gets a
#      route-miss (404). The bypass router/load_balancer live INSIDE the
#      branch: a top-level router or load_balancer cannot coexist with an
#      IRR in the same chain, and the branch rejoins at `terminal` so its
#      selected cluster is forwarded upstream directly.
#
#   11. The iterative_request_router owns the agentic loop for classified
#      openai_responses create requests. openai_responses_proxy rebuilds
#      the outbound body from ResponsesState and automatically selects the
#      typed streaming transport for an effective `stream: true` request,
#      so each backend SSE chunk reaches openai_stream_events
#      incrementally. openai_agentic_loop is the sole loop owner, sole
#      response parser, and sole transition authority: it extracts
#      completed function_call, web_search_call, file_search_call, and MCP
#      items, records the assignment each request-phase dispatcher must
#      execute next, and publishes the single loop/done decision. Because
#      the IRR is the terminal filter, all backend routing for create
#      requests lives inside the IRR step. That step router is path-only:
#      the IRR strips every x-praxis-* reserved-prefix header from its
#      sub-request before the step runs, so it cannot route on
#      x-praxis-ai-format. Responses-vs-Chat discrimination therefore
#      happens PRE-IRR at the bypass carrier (step 10); only a classified
#      openai_responses create request reaches this step.
#
# IRR inference filter chain (canonical #1046 order):
#   Request phase (forward):
#     openai_stream_events → openai_web_search → openai_mcp_dispatch
#       → openai_file_search_callout → openai_agentic_loop
#       → headers → router → load_balancer → openai_responses_proxy
#   Response phase (reverse):
#     openai_responses_proxy → load_balancer → router → headers
#       → openai_agentic_loop → (dispatchers have no response phase)
#       → openai_stream_events
#   The proxy's request-body hook still participates in the pipeline body
#   pre-read. Its request-header hook runs after load_balancer selection, so it
#   can fail closed unless the selected cluster explicitly declares both
#   application_protocol: openai_responses and application_provider: openai.
#   The three dispatchers (openai_web_search, openai_mcp_dispatch,
#   openai_file_search_callout) have no response phase: each executes the
#   loop owner's assigned calls at request-body EOS on the next iteration,
#   reconciling results in place inside ResponsesState.accumulated_output.
#   Each is inert unless the model emits the matching hosted tool call.
#
# Transition rules (evaluated after each step's response-body):
#   openai_agentic_loop.action = "loop" → next: inference
#   default                             → done (exit to client)
#   The loop owner is the sole transition authority (issue #1046); the
#   dispatch filters never drive the IRR transition.
#
# Config knobs:
#   max_infer_iters: application-level iteration cap on openai_agentic_loop
#   max_iterations:  infrastructure-level safety cap on the IRR. It must be
#                    at least max_infer_iters + 1 (initial inference plus
#                    the allowed tool-backed inference continuations). Here
#                    7 + 1 = 8.
#
# Streaming persistence:
#   Removing openai_stream_events (or moving openai_response_store after
#   the IRR) silently breaks streaming persistence: the store persists
#   ResponsesState.response_object, and only openai_stream_events
#   populates it for `stream: true` responses. openai_stream_events is
#   also REQUIRED for loop correctness — openai_responses_proxy selects
#   typed streaming automatically for `stream: true`, which commits
#   `response.completed` to the client as it arrives, so a loop-terminal
#   error detected by openai_agentic_loop can only reach the client
#   through this filter's logical-stream finalizer. When it is absent from
#   a streaming step, openai_agentic_loop fails closed with a 500 before
#   any backend request. Keep the store PRE-IRR and openai_stream_events
#   INSIDE the IRR step so both buffered and streaming responses are
#   persisted and retrievable via GET /v1/responses/{id} (served pre-IRR
#   by openai_response_store).
#
# Security: StreamBuffer body callouts run before this listener's
# header-phase filters. Deploy this example behind an outer authentication
# and authorization boundary before enabling the required
# allow_pre_security_callout acknowledgement below.
#
# A request is stateful when any of these hold:
#   - previous_response_id is set
#   - tools array is non-empty
#   - store is true (the OpenAI spec default when omitted)
#   - background is true
#   - conversation is set
#   - prompt.id is set
#
# Example requests:
#
#   # Stateful (store defaults to true) — validated, looped, persisted
#   curl -X POST http://localhost:8080/v1/responses \
#     -H "Content-Type: application/json" \
#     -d '{"model":"gpt-4.1","input":"Hello, world!"}'
#
#   # Streaming with persistence (store defaults to true)
#   curl -N -X POST http://localhost:8080/v1/responses \
#     -H "Content-Type: application/json" \
#     -d '{"model":"gpt-4.1","input":"Hello","stream":true}'
#
#   # Stateless (store=false, no stateful markers) — validated and routed
#   curl -X POST http://localhost:8080/v1/responses \
#     -H "Content-Type: application/json" \
#     -d '{"model":"gpt-4.1","input":"Hello","store":false}'
#
#   # Multi-turn with previous_response_id
#   curl -X POST http://localhost:8080/v1/responses \
#     -H "Content-Type: application/json" \
#     -d '{"model":"gpt-4.1","input":"What next?","previous_response_id":"resp_abc"}'
#
#   # File search — model issues file_search_call, executed via vector store
#   curl -X POST http://localhost:8080/v1/responses \
#     -H "Content-Type: application/json" \
#     -H "Authorization: Bearer token" \
#     -d '{
#       "model": "gpt-4.1",
#       "input": "What does the README say about deployment?",
#       "tools": [{"type": "file_search", "vector_store_ids": ["vs_abc"]}]
#     }'
#
#   # MCP tool loop — model issues an MCP call, executed via the server URL
#   curl -X POST http://localhost:8080/v1/responses \
#     -H "Content-Type: application/json" \
#     -d '{
#       "model": "gpt-4.1",
#       "input": "What is the weather in SF?",
#       "tools": [{"type": "mcp", "server_label": "weather",
#                  "server_url": "http://mcp-weather:8001/mcp",
#                  "allowed_tools": ["get_weather"],
#                  "require_approval": "never"}]
#     }'
#
#   # Retrieve a stored response (buffered or streamed)
#   curl http://localhost:8080/v1/responses/<response-id>
#
#   # Responses WebSocket (requires a WebSocket-capable backend) —
#   # takes the bypass branch, never the IRR
#   websocat -H="Authorization: Bearer $OPENAI_API_KEY" \
#     ws://localhost:8080/v1/responses
#
# Web search requires the trusted boundary to provide x-user-brave-key. The
# WEB_SEARCH_API_KEY field remains required by the provider configuration, so
# set it before starting the proxy; when `user_credential: brave_search` is
# configured, a missing per-user slot fails closed instead of falling back to
# that shared key. openai_web_search stays inert until the model emits a
# web_search_call.
#
# Listener and cluster transport timeouts are intentionally omitted.
# Configured downstream read and upstream idle/read/write timeouts remain
# active after a 101 upgrade and can terminate long-lived WebSockets.
#
# Direct OpenAI upstream variant:
#   - Replace the local inference-backend endpoint with api.openai.com:443
#     and configure tls.sni: api.openai.com (in BOTH the bypass branch's
#     load_balancer and the IRR step's load_balancer).
#   - Uncomment the `http.application_protocol: openai_responses` and
#     `http.application_provider: openai` declarations on BOTH
#     inference-backend clusters below. The canonical IRR ordering already
#     permits `prompt.id` once both capabilities are declared; endpoint
#     identity does not participate in the trust decision.
#   - Add this headers filter inside the IRR step after the router to
#     replace the downstream Host:
#
#       - filter: headers
#         request_set:
#           - name: Host
#             value: api.openai.com
#
#   - Either pass through the client's Authorization header or add a
#     credential_injection filter inside the IRR step to inject a
#     server-owned credential:
#
#       - filter: credential_injection
#         clusters:
#           - name: inference-backend
#             header: Authorization
#             env_var: OPENAI_API_KEY
#             header_prefix: "Bearer "
#
#   - Configure the inference cluster as:
#
#       - name: "inference-backend"
#         endpoints:
#           - "api.openai.com:443"
#         tls:
#           sni: api.openai.com
#
# Build:
#   cargo build -p praxis-ai-proxy --features openai-file-resolve-filter,openai-conversations,openai-mcp-tools,store-sqlite

listeners:
  - name: ai-gateway
    address: "127.0.0.1:8080"
    filter_chains: [full-flow-agentic-pipeline]

filter_chains:
  - name: full-flow-agentic-pipeline
    filters:
      - filter: trace_context
        # Establish typed correlation before IRR snapshots request extensions.
        # Child callouts retain this request/trace identity while Praxis core
        # creates a fresh span ID for each outbound hop.

      - filter: state_owner
        # These source names are deployment-specific. The authentication
        # boundary must overwrite them and prevent clients from bypassing it.
        mode: trusted_headers
        tenant:
          header: x-auth-tenant
        issuer:
          static: urn:example:gateway
        subject:
          header: x-auth-user

      - filter: callout_credentials
        # The trusted authentication boundary must delete any client-supplied
        # instance and then set exactly one authenticated value on every
        # request. This filter establishes typed request context and strips the
        # ingress header; it does not authenticate the header's provenance.
        credentials:
          - slot: brave_search
            source_header: x-user-brave-key
          - slot: ogx_files
            source_header: x-user-ogx-key
          - slot: mcp_gateway
            source_header: x-user-mcp-key
        assertions:
          # The same establishing filter keeps assertions in a separate typed
          # slot map. The trusted boundary MUST delete any client-supplied
          # x-mcp-authorized value and then set the verified, short-lived
          # assertion. Praxis forwards it unchanged only to configured
          # connector_id destinations; the MCP Gateway decides authorization.
          - slot: mcp_gateway
            source_header: x-mcp-authorized

      # Emit the normalized owner in the contract expected by OGX and by the
      # explicitly configured downstream callouts below. Raw x-auth-* inputs
      # are stripped and never used directly as destination assertions.
      - filter: project_state_owner_headers
        tenant_header: x-tenant-id
        subject_header: x-user-id

      # Publishes the typed operation consumed by openai_conversations.
      - filter: openai_operation

      - filter: openai_conversations
        backend: sqlite
        database_url: "sqlite://responses.db?mode=rwc"
        conversations_table: openai_conversations
        items_table: openai_conversation_items

      - filter: openai_responses_format
        on_invalid: continue
        headers:
          format: x-praxis-ai-format
          model: x-praxis-ai-model
          stream: x-praxis-ai-stream
          mode: x-praxis-responses-mode

      - filter: openai_responses_validate

      - filter: openai_tool_parse

      - filter: openai_response_store
        backend: sqlite
        # In-memory:
        #   database_url: "sqlite::memory:"
        # File-backed:
        database_url: "sqlite://responses.db?mode=rwc"
        responses_table: openai_responses
        conversations_table: openai_conversations

      - filter: openai_responses_rehydrate

      - filter: openai_file_resolve
        files_api_url: "http://127.0.0.1:9999"
        # Require the caller-scoped OGX credential captured above. It is
        # injected only on configured file_id callouts; file_url downloads
        # remain on the separate credential-free resolver.
        user_credential: ogx_files
        allow_pre_security_callout: true
        # file_id callouts run through this outbound filter chain via the
        # filtered subrequest executor, which enforces destination authority,
        # DNS/SSRF (from the pipeline's allow_private_upstreams), TLS/SNI, and
        # Host centrally — this holds even with an empty or omitted chain. The
        # chain only re-projects the trusted tenant/subject identity from the
        # StateOwner extension; client file_url downloads never traverse it.
        outbound_chain:
          name: files-api-outbound
          filters:
            - filter: project_state_owner_headers
              tenant_header: x-tenant-id
              subject_header: x-user-id
        on_missing: reject
        timeout_ms: 10000

      - filter: openai_doc_extract
        allow_pre_security_callout: true
        on_unsupported: continue

      - filter: openai_mcp_tool_resolve
        user_credential: mcp_gateway
        authorization_assertion: mcp_gateway
        forward_headers:
          - x-tenant-id
          - x-user-id
        # Ambient trusted headers are forwarded only for connector-backed MCP
        # entries; arbitrary client-supplied server_url targets never receive
        # them. Clients reference this destination with connector_id.
        connectors:
          - id: trusted-mcp
            server_url: http://127.0.0.1:8001/mcp

      # Bypass gateway: a no-op carrier that runs for everything EXCEPT a
      # classified `POST /v1/responses` create request. When it runs, its
      # unconditional branch routes the request directly (WebSocket
      # upgrades, /v1/prompts, /v1/embeddings, /v1/files, /v1/vector_stores)
      # and rejoins at `terminal`, forwarding the selected cluster upstream
      # without entering the IRR. The router and load_balancer live inside
      # the branch because a top-level router/load_balancer cannot coexist
      # with the IRR below. The `unless` condition skips this carrier only
      # for `POST /v1/responses` bearing x-praxis-ai-format:
      # openai_responses, so create requests fall through to the IRR while a
      # non-Responses body on /v1/responses (Chat Completions, Anthropic
      # Messages, or unclassified) still runs the carrier and gets a
      # route-miss (404). Discrimination lives here because the IRR strips
      # x-praxis-* headers from its sub-request; inside the branch the
      # /v1/responses route matches the WebSocket Upgrade header (routes
      # cannot match on HTTP method), so a WS handshake reaches the backend
      # while a non-Responses body does not.
      - filter: headers
        conditions:
          - unless:
              path: "/v1/responses"
              methods: [POST]
              headers:
                x-praxis-ai-format: "openai_responses"
        branch_chains:
          - name: bypass-irr
            rejoin: terminal
            chains:
              - name: bypass-chain
                filters:
                  - filter: router
                    routes:
                      - path_prefix: "/v1/prompts"
                        cluster: "prompts-api"
                      - path_prefix: "/v1/embeddings"
                        cluster: "embeddings-api"
                      - path_prefix: "/v1/files"
                        cluster: "files-api"
                      # WebSocket handshakes are GET /v1/responses with an
                      # Upgrade header. Routes cannot match on HTTP method, so
                      # the Upgrade header discriminates: a WS handshake routes
                      # to the backend, while a non-Responses body posted to
                      # /v1/responses carries no Upgrade header and gets a
                      # route-miss (404) here instead of reaching the backend.
                      - path: "/v1/responses"
                        headers:
                          upgrade: "websocket"
                        cluster: "inference-backend"
                      - path_prefix: "/v1/vector_stores"
                        cluster: "vector-stores-backend"
                  - filter: load_balancer
                    clusters:
                      - name: "prompts-api"
                        endpoints:
                          - "127.0.0.1:9998"
                      - name: "embeddings-api"
                        endpoints:
                          - "127.0.0.1:9997"
                      - name: "files-api"
                        endpoints:
                          - "127.0.0.1:9999"
                      - name: "inference-backend"
                        # Optional provider capability. Uncomment only when
                        # this backend implements OpenAI-managed features such
                        # as prompt IDs; generic/vLLM backends should leave it
                        # absent so prompt templates fail closed.
                        # http:
                        #   application_protocol: openai_responses
                        #   application_provider: openai
                        endpoints:
                          - "127.0.0.1:3001"
                      - name: "vector-stores-backend"
                        endpoints:
                          - "127.0.0.1:3002"

      # Classified openai_responses POST /v1/responses create requests flow
      # here. The IRR runs the agentic model-tool-model loop: each round
      # streams through openai_stream_events, the three request-phase
      # dispatchers execute the loop owner's assigned hosted-tool calls, and
      # openai_agentic_loop parses each response and publishes the single
      # loop/done transition. The pre-IRR openai_response_store persists the
      # composed object on the response path (reverse order). The step
      # router is path-only because the IRR strips x-praxis-* headers from
      # its sub-request; the bypass carrier above has already excluded
      # non-Responses traffic.
      - filter: iterative_request_router
        initial_step: inference
        # Safety cap: at least max_infer_iters + 1 (7 + 1 = 8).
        max_iterations: 8
        # The IRR's 30s default is too short for production model streams.
        # Allow six minutes total for time-to-first-byte and logical
        # streaming across the loop; each step inherits a five-minute
        # deadline (the vLLM integration tests already run at 300s).
        timeout_ms: 360000
        step_timeout_ms: 300000
        max_response_bytes: 67108864
        max_stream_response_bytes: 67108864
        max_state_bytes: 136314880
        steps:
          - name: inference
            filters:
              # Re-project inside the destination-owned subrequest. Core moves
              # the normalized StateOwner extension into each IRR step; this
              # strips any still-uncommitted raw ingress identity headers and
              # recreates only the OGX/callout contract.
              - filter: project_state_owner_headers
                tenant_header: x-tenant-id
                subject_header: x-user-id

              # Composes every inference SSE stream into one logical
              # Responses lifecycle: preserves one response identity across
              # rounds and withholds per-round terminal events until the IRR
              # transition is known. A single inference round is just a
              # one-round logical stream, so this filter always composes and
              # must run inside this IRR step. REQUIRED for both streaming
              # persistence and loop-terminal error delivery (see the
              # "Streaming persistence" note above); openai_agentic_loop
              # fails closed with a 500 on an effective stream:true request
              # when it is absent.
              - filter: openai_stream_events

              # Request-phase dispatcher: runs first on re-entry to execute
              # web_search_call assignments prepared by the loop owner. Inert
              # until the model emits a web_search_call.
              - filter: openai_web_search
                provider: brave
                api_key: ${WEB_SEARCH_API_KEY}
                # Require the per-user key captured by the outer
                # callout_credentials filter. The secret is injected only at
                # the exact resolved provider authority.
                user_credential: brave_search
                max_calls_per_round: 32
              # Request-phase dispatcher: runs second on re-entry to execute
              # MCP calls prepared and classified by the loop owner. Inert
              # until the model emits an MCP tool call.
              - filter: openai_mcp_dispatch
                user_credential: mcp_gateway
                authorization_assertion: mcp_gateway
                forward_headers:
                  - x-tenant-id
                  - x-user-id
                max_calls_per_round: 32
                max_parallel_calls: 8
                max_result_bytes: 1048576
                max_total_result_bytes: 8388608
              # Request-phase dispatcher: at request-body EOS on each IRR
              # re-entry it executes the file_search_call items the loop owner
              # assigned in the prior response, reconciling each in place
              # inside ResponsesState.accumulated_output. It never parses the
              # response body and never decides whether another round runs,
              # so it is inert unless the model emits a hosted file_search_call.
              - filter: openai_file_search_callout
                vector_store_url: http://127.0.0.1:3002
                user_credential: ogx_files
                # Every vector-store sub-request runs through the filtered
                # subrequest executor, which enforces destination authority,
                # DNS/SSRF, TLS/SNI, and Host centrally (destination from
                # vector_store_url) — this holds even with an empty or omitted
                # chain. The outbound chain only re-projects the trusted
                # tenant/subject identity from the StateOwner extension.
                outbound_chain:
                  name: vector-store-outbound
                  filters:
                    - filter: project_state_owner_headers
                      tenant_header: x-tenant-id
                      subject_header: x-user-id
                timeout_ms: 5000
                max_response_bytes: 10485760
                max_total_response_bytes: 67108864
                max_state_bytes: 136314880
                on_failure: closed
              # Sole loop owner: parses each model response, records the
              # hosted-tool assignments for the dispatchers, and publishes the
              # single continuation signal (action=loop|done). max_infer_iters
              # must stay below the IRR max_iterations safety cap (7 + 1 = 8).
              - filter: openai_agentic_loop
                max_infer_iters: 7
              - filter: headers
                request_set:
                  - name: Content-Type
                    value: application/json
              # Path-only routing. The x-praxis-ai-format header cannot be
              # matched here: the IRR strips every x-praxis-* reserved-prefix
              # header from the sub-request before this step's router runs, so
              # a header match would always miss (404). Responses-vs-Chat
              # discrimination therefore happens PRE-IRR at the bypass carrier
              # (only a classified openai_responses POST reaches this step).
              - filter: router
                routes:
                  - path: "/v1/responses"
                    cluster: "inference-backend"
              - filter: load_balancer
                clusters:
                  - name: "inference-backend"
                    # Optional provider capability. Uncomment only when this
                    # backend can resolve OpenAI-managed prompt IDs.
                    # Endpoint hostname, port, TLS, and SNI are not the trust
                    # boundary.
                    # http:
                    #   application_protocol: openai_responses
                    #   application_provider: openai
                    endpoints:
                      - "127.0.0.1:3001"
              # Selects typed streaming automatically when the effective
              # outbound body carries `"stream": true`; otherwise buffers.
              # Its request hook runs after load-balancer selection. Prompt
              # templates fail closed unless the selected cluster enables both
              # optional application declarations above.
              - filter: openai_responses_proxy
            on_result:
              # openai_agentic_loop is the sole loop authority (issue #1046):
              # it parses each model response, records assignments for the
              # request-phase dispatchers (web_search, mcp_dispatch,
              # file_search_callout), and publishes the single continuation
              # signal. The IRR transitions only on the owner's action, never
              # on a dispatcher's.
              - filter: openai_agentic_loop
                key: action
                value: loop
                next: inference
              - default: true
                done: true

insecure_options:
  allow_private_endpoints: true # example proxies to local backends
  # Central SSRF control: permits resolved callout addresses (vector_store_url,
  # file_id Files API) to be private/loopback at connect time (needed for
  # local/private vector stores and Files API backends).
  allow_private_upstreams: true
