# Streaming Event Accumulation
# Requires `--features store-sqlite` because these filters are opt-in.
#
# Demonstrates the `openai_stream_events` filter, which composes the
# current iterative-request-router (IRR) execution into one logical
# Responses SSE stream: it parses each backend SSE chunk, accumulates
# state (response object, output items, tool calls, usage) into
# ResponsesState, normalizes the per-round lifecycle, and preserves
# parser state through stream completion. A single inference round is
# just a one-round logical stream.
#
# `openai_stream_events` always composes a logical stream, so it must
# run inside an `iterative_request_router` step; placing it anywhere
# else fails closed at request time. A `stream: true` request arms it.
# It must sit after `load_balancer` in that step so IRR body hooks see the
# selected peer when needed; `openai_responses_proxy` buffers the request
# body, so IRR otherwise runs that phase before load balancing.
#
# The `openai_response_store` filter runs pre-IRR (registering the
# store backend) and persists the accumulated ResponsesState once the
# logical stream terminates. Response body filters run in reverse
# config order, so the store persists after the step composes the
# terminal event.
#
# Because the store sits after `openai_stream_events` on the response
# path, it also captures each normalized SSE event (with its stamped
# sequence number) into a per-response replay log. A completed response
# created with `stream: true` can then be replayed verbatim with
# `GET /v1/responses/{id}?stream=true`: the stored events are streamed
# back in original order, terminating in the same terminal event, with
# no delta reconstruction. `starting_after=N` resumes after a cursor,
# returning only events whose `sequence_number > N`. The replay log is
# bounded by `max_event_count` and `max_event_bytes` (defaults below); a
# stream that exceeds either bound stops capture, so its terminal event
# is never recorded and it becomes non-replayable (a subsequent
# `?stream=true` retrieval returns 400 invalid_request_error). Responses
# created without `stream: true` have no replay log and return the same
# 400. The plain JSON `GET /v1/responses/{id}` retrieval is unchanged.
#
# `openai_responses_proxy` automatically selects Praxis's
# typed streaming transport for the effective `stream: true` request, so
# each SSE chunk reaches `openai_stream_events` incrementally.
#
# For a complete multi-turn setup with rehydrate and the agentic loop,
# see agentic-loop.yaml. For terminal streaming without persistence, see
# irr-terminal-streaming.yaml.
#
# Example requests:
#
#   # For local testing, encode the readable [tenant, issuer, subject] tuple.
#   # In production, a trusted authentication boundary constructs this value.
#   OWNER_COMPONENTS='["tenant-a","https://issuer.example","user-123"]'
#   OWNER_PAYLOAD=$(
#     printf '%s' "$OWNER_COMPONENTS" |
#       openssl base64 -A |
#       tr '+/' '-_' |
#       tr -d '='
#   )
#   OWNER_ASSERTION="v1.${OWNER_PAYLOAD}"
#
#   # Streaming with persistence (store defaults to true)
#   curl -N http://localhost:8080/v1/responses \
#     -H "Content-Type: application/json" \
#     -H "x-authenticated-state-owner: ${OWNER_ASSERTION}" \
#     -d '{"model":"Qwen/Qwen3-0.6B","input":"Say hello","stream":true}'
#
#   # Non-streaming (still persisted via the buffered path)
#   curl http://localhost:8080/v1/responses \
#     -H "Content-Type: application/json" \
#     -H "x-authenticated-state-owner: ${OWNER_ASSERTION}" \
#     -d '{"model":"Qwen/Qwen3-0.6B","input":"Say hello"}'
#
#   # Retrieve stored response (plain JSON)
#   curl http://localhost:8080/v1/responses/<response-id> \
#     -H "x-authenticated-state-owner: ${OWNER_ASSERTION}"
#
#   # Replay the stored SSE event log for a stream:true response
#   curl -N "http://localhost:8080/v1/responses/<response-id>?stream=true" \
#     -H "x-authenticated-state-owner: ${OWNER_ASSERTION}"
#
#   # Resume replay after sequence number 5
#   curl -N "http://localhost:8080/v1/responses/<response-id>?stream=true&starting_after=5" \
#     -H "x-authenticated-state-owner: ${OWNER_ASSERTION}"
#
# Requires the ai-inference feature:
#   cargo build -p praxis-ai-proxy --features store-sqlite

listeners:
  - name: ai-gateway
    address: "127.0.0.1:8080"
    filter_chains: [responses-pipeline]

filter_chains:
  - name: responses-pipeline
    filters:
      - filter: state_owner
        mode: trusted_owner
        header: x-authenticated-state-owner

      - filter: openai_responses_request
        on_invalid: reject
        headers:
          format: x-praxis-ai-format
          model: x-praxis-ai-model
          stream: x-praxis-ai-stream
          mode: x-praxis-responses-mode

      - filter: openai_response_store
        backend: sqlite
        database_url: "sqlite://responses.db?mode=rwc"
        responses_table: openai_responses
        conversations_table: openai_conversations
        # Replay-log bounds for `stream: true` responses. A stream that exceeds
        # either bound stops event capture, so its terminal event is never
        # recorded and it becomes non-replayable. Defaults shown.
        # max_event_count: 10000
        # max_event_bytes: 16777216

      - filter: iterative_request_router
        initial_step: inference
        max_iterations: 1
        # The IRR's default 30s end-to-end deadline is too short for model
        # inference. This single step inherits a 6-minute total deadline for
        # time-to-first-byte plus response streaming.
        timeout_ms: 360000
        steps:
          - name: inference
            filters:
              - filter: openai_responses_proxy

              - filter: headers
                request_set:
                  - name: Content-Type
                    value: application/json

              - filter: router
                routes:
                  - path: "/v1/responses"
                    cluster: "inference-backend"

              - filter: load_balancer
                clusters:
                  - name: "inference-backend"
                    # Cluster ceiling. openai_stream_events caps the live body
                    # at leftover timeout_secs after the first SSE chunk.
                    read_timeout_ms: 300000
                    endpoints:
                      - "127.0.0.1:8000"

              # After load_balancer so IRR body hooks see the selected peer.
              # openai_responses_proxy buffers the request body, so IRR runs
              # that phase before load balancing; an earlier placement never
              # sees ctx.upstream. timeout_secs is an absolute deadline from
              # the first SSE chunk (default 300). Each chunk recaps that
              # absolute cutoff onto the live body through cap_stream_deadline.
              - filter: openai_stream_events
                # max_tool_call_argument_bytes: 1048576
                # Aggregate accumulation budget bounding total process memory even
                # when every individual SSE event is within max_buffer_bytes. The
                # stream fails closed once either dimension is exceeded.
                # max_accumulated_bytes: 67108864
                # max_output_items: 100000
            on_result:
              - default: true
                done: true

insecure_options:
  allow_private_endpoints: true # example proxies to local backends
