# Responses File Search with a Chat Completions Backend
# Requires `--features openai-responses` because these filters are opt-in.
#
# Accepts finite OpenAI Responses requests with hosted file search while
# targeting a backend that only implements /v1/chat/completions. The
# compatibility filter privately exposes file_search to that backend as a
# function tool; openai_agentic_loop then normalizes the returned function
# call into a file_search_call and records it as an assignment, and
# openai_file_search_callout executes it before one more finite inference
# round (#1046).
#
# Request order (forward):
#   1. openai_responses_format classifies the client request.
#   2. openai_responses_validate creates canonical ResponsesState.
#   3. iterative_request_router owns the finite model-search-model loop.
#   4. openai_file_search_callout dispatches the owner's pending hosted
#      search assignments at request-body EOS on IRR re-entry.
#   5. openai_agentic_loop prepares the next inference round.
#   6. responses_to_chat_completions synthesizes the private function tool.
#   7. path_rewrite selects /v1/chat/completions explicitly.
#
# Response filters run in reverse order. Each Chat Completions response is
# therefore converted into a Responses resource by
# responses_to_chat_completions before openai_agentic_loop inspects it. A
# returned function_call(name="file_search") is normalized to
# file_search_call, recorded as an assignment, executed through the vector
# store on the next re-entry, and retained in the client-visible final
# Responses output. openai_agentic_loop is the sole loop owner and publishes
# the single continuation signal (action=loop|done); openai_file_search_callout
# is a request-phase dispatcher with no response phase.
#
# This pipeline is finite only. Streaming file-search orchestration is not
# supported.

listeners:
  - name: responses-file-search-chat-gateway
    address: "127.0.0.1:8080"
    filter_chains: [responses-file-search-chat]

filter_chains:
  - name: responses-file-search-chat
    filters:
      - filter: openai_responses_format
        on_invalid: reject
        headers:
          format: x-praxis-ai-format
          model: x-praxis-ai-model
          stream: x-praxis-ai-stream

      - filter: openai_responses_validate

      - filter: iterative_request_router
        initial_step: inference
        max_iterations: 8
        timeout_ms: 120000
        step_timeout_ms: 60000
        max_response_bytes: 67108864
        max_state_bytes: 136314880
        steps:
          - name: inference
            filters:
              # Request-phase dispatcher: at request-body EOS on each IRR
              # re-entry it executes the file_search_call items the loop owner
              # assigned in the prior (translated) response, reconciling each in
              # place inside ResponsesState.accumulated_output. It never parses
              # the response body and never decides whether another round runs.
              - filter: openai_file_search_callout
                vector_store_url: http://127.0.0.1:8001
                # Every vector-store sub-request runs through this chain. It
                # must be inline ({ name, filters }): this filter runs inside an
                # iterative_request_router step, whose pipeline is built with an
                # empty named-chain map, so a top-level filter_chains reference
                # cannot resolve here. The destination comes from
                # vector_store_url, so the chain never selects an upstream.
                outbound_chain:
                  name: vector-store-outbound
                  filters:
                    - filter: headers
                      request_set:
                        - name: X-Vector-Store-Client
                          value: praxis-ai-gateway
                timeout_ms: 5000
                max_response_bytes: 10485760
                max_total_response_bytes: 67108864
                max_state_bytes: 136314880
                on_failure: closed
                forward_headers:
                  - authorization

              # Sole loop owner: on the response phase it runs after
              # responses_to_chat_completions has converted the Chat Completions
              # reply into a Responses resource, so it parses the normalized
              # file_search_call items, records assignments for the dispatcher,
              # and publishes the single continuation signal (action=loop|done).
              # max_infer_iters must stay below the IRR max_iterations cap.
              - filter: openai_agentic_loop
                max_infer_iters: 7

              - filter: responses_to_chat_completions
                max_rewritten_body_bytes: 67108864

              - filter: path_rewrite
                replace:
                  pattern: "^/v1/responses/?$"
                  replacement: "/v1/chat/completions"
                conditions:
                  - when:
                      path_prefix: "/v1/responses"
                      methods: [POST]

              - filter: headers
                request_set:
                  - name: Content-Type
                    value: application/json

              - filter: router
                routes:
                  - path: "/v1/chat/completions"
                    cluster: "chat-completions-backend"

              - filter: load_balancer
                clusters:
                  - name: "chat-completions-backend"
                    endpoints:
                      - "127.0.0.1:3001"
            on_result:
              - filter: openai_agentic_loop
                key: action
                value: loop
                next: inference
              - default: true
                done: true

insecure_options:
  allow_private_endpoints: true # example proxies to local backends
  # Central SSRF control for outbound callouts: permits the vector_store_url's
  # resolved address to be private/loopback at connect time. Required only when
  # the vector store runs on a private network (e.g. local development).
  allow_private_upstreams: true
