File Search Chat Completions
Accepts finite OpenAI Responses requests with hosted file search while targeting a backend that only implements /v1/chat/completions
Category: Setup-dependent integration
Task: Accepts finite OpenAI Responses requests with hosted file search while targeting a backend that only implements /v1/chat/completions
Prerequisites: The external service, credentials, or certificates referenced by this configuration.
This configuration comes from the selected release. The example has not been run here; external services are not bundled.
Download the source file.
# Responses File Search with a Chat Completions Backend
# Requires `--features openai-responses` because these filters are opt-in.
#
# Accepts finite OpenAI Responses requests with hosted file search while
# targeting a backend that only implements /v1/chat/completions. The
# compatibility filter privately exposes file_search to that backend as a
# function tool; openai_agentic_loop then normalizes the returned function
# call into a file_search_call and records it as an assignment, and
# openai_file_search_callout executes it before one more finite inference
# round (#1046).
#
# Request order (forward):
# 1. openai_responses_format classifies the client request.
# 2. openai_responses_validate creates canonical ResponsesState.
# 3. iterative_request_router owns the finite model-search-model loop.
# 4. openai_file_search_callout dispatches the owner's pending hosted
# search assignments at request-body EOS on IRR re-entry.
# 5. openai_agentic_loop prepares the next inference round.
# 6. responses_to_chat_completions synthesizes the private function tool.
# 7. path_rewrite selects /v1/chat/completions explicitly.
#
# Response filters run in reverse order. Each Chat Completions response is
# therefore converted into a Responses resource by
# responses_to_chat_completions before openai_agentic_loop inspects it. A
# returned function_call(name="file_search") is normalized to
# file_search_call, recorded as an assignment, executed through the vector
# store on the next re-entry, and retained in the client-visible final
# Responses output. openai_agentic_loop is the sole loop owner and publishes
# the single continuation signal (action=loop|done); openai_file_search_callout
# is a request-phase dispatcher with no response phase.
#
# This pipeline is finite only. Streaming file-search orchestration is not
# supported.
listeners:
- name: responses-file-search-chat-gateway
address: "127.0.0.1:8080"
filter_chains: [responses-file-search-chat]
filter_chains:
- name: responses-file-search-chat
filters:
- filter: openai_responses_format
on_invalid: reject
headers:
format: x-praxis-ai-format
model: x-praxis-ai-model
stream: x-praxis-ai-stream
- filter: openai_responses_validate
- filter: iterative_request_router
initial_step: inference
max_iterations: 8
timeout_ms: 120000
step_timeout_ms: 60000
max_response_bytes: 67108864
max_state_bytes: 136314880
steps:
- name: inference
filters:
# Request-phase dispatcher: at request-body EOS on each IRR
# re-entry it executes the file_search_call items the loop owner
# assigned in the prior (translated) response, reconciling each in
# place inside ResponsesState.accumulated_output. It never parses
# the response body and never decides whether another round runs.
- filter: openai_file_search_callout
vector_store_url: http://127.0.0.1:8001
# Every vector-store sub-request runs through this chain. It
# must be inline ({ name, filters }): this filter runs inside an
# iterative_request_router step, whose pipeline is built with an
# empty named-chain map, so a top-level filter_chains reference
# cannot resolve here. The destination comes from
# vector_store_url, so the chain never selects an upstream.
outbound_chain:
name: vector-store-outbound
filters:
- filter: headers
request_set:
- name: X-Vector-Store-Client
value: praxis-ai-gateway
timeout_ms: 5000
max_response_bytes: 10485760
max_total_response_bytes: 67108864
max_state_bytes: 136314880
on_failure: closed
forward_headers:
- authorization
# Sole loop owner: on the response phase it runs after
# responses_to_chat_completions has converted the Chat Completions
# reply into a Responses resource, so it parses the normalized
# file_search_call items, records assignments for the dispatcher,
# and publishes the single continuation signal (action=loop|done).
# max_infer_iters must stay below the IRR max_iterations cap.
- filter: openai_agentic_loop
max_infer_iters: 7
- filter: responses_to_chat_completions
max_rewritten_body_bytes: 67108864
- filter: path_rewrite
replace:
pattern: "^/v1/responses/?$"
replacement: "/v1/chat/completions"
conditions:
- when:
path_prefix: "/v1/responses"
methods: [POST]
- filter: headers
request_set:
- name: Content-Type
value: application/json
- filter: router
routes:
- path: "/v1/chat/completions"
cluster: "chat-completions-backend"
- filter: load_balancer
clusters:
- name: "chat-completions-backend"
endpoints:
- "127.0.0.1:3001"
on_result:
- filter: openai_agentic_loop
key: action
value: loop
next: inference
- default: true
done: true
insecure_options:
allow_private_endpoints: true # example proxies to local backends
# Central SSRF control for outbound callouts: permits the vector_store_url's
# resolved address to be private/loopback at connect time. Required only when
# the vector store runs on a private network (e.g. local development).
allow_private_upstreams: true