# Response Store
# Requires `--features store-sqlite` because these filters are opt-in.
#
# Persists streaming and non-streaming Responses API responses to a database,
# serves stored data via GET endpoints, and handles DELETE /v1/responses/{id}
# locally. Supports SQLite (file-backed or in-memory) and PostgreSQL backends:
#
#   POST /v1/responses          — proxied to backend, response persisted
#   GET  /v1/responses/{id}     — served from local store (200 or 404)
#   GET  /v1/responses/{id}/input_items — paginated input items
#   DELETE /v1/responses/{id}   — deleted from local store (200 or 404)
#
# Request order:
#   1. openai_responses_request classifies the create body, publishes the
#      classification metadata `openai_response_store` reads, and creates the
#      `ResponsesState` the streaming accumulator needs.
#   2. state_owner assigns the persisted-state owner.
#   3. openai_response_store initializes the store before inference and, on the
#      response path, persists the finished response.
#   4. iterative_request_router runs one inference step (max_iterations: 1) that
#      selects the response transport and, for a `stream: true` request, composes
#      the backend SSE into one logical stream the accumulator captures.
#
# Persisted state always requires an explicit owner source. This standalone
# example uses `single_tenant`, which deliberately shares one owner and
# is not user isolation. See `state-ownership.yaml` for a trusted multi-user
# boundary and the versioned assertion format.
#
# POST persistence:
#   `openai_responses_proxy` always advertises the streaming capability and
#   selects Praxis's typed streaming transport for an effective `stream: true`
#   request; otherwise it selects the buffered transport. For a streaming
#   request, `openai_stream_events` (inside the IRR step, after `load_balancer`)
#   parses each backend SSE chunk and accumulates the terminal `ResponsesState`;
#   response body filters run in reverse config order, so the pre-IRR
#   `openai_response_store` persists the accumulated response once the logical
#   stream terminates. A finite request is persisted from its buffered JSON
#   body instead. Non-2xx responses and content types other than JSON or
#   event-stream are skipped.
#
# DELETE handling:
#   DELETE /v1/responses/{id} is intercepted during `on_request`
#   and short-circuited with a 200 (deleted) or 404 (not found)
#   JSON response. The request never reaches the backend.
#
#   curl -X DELETE http://127.0.0.1:8080/v1/responses/resp_abc
#
# The store is lazily initialized on the first qualifying
# request. If initialization fails (bad URL, permissions),
# the failure is permanent and the filter becomes a no-op.

listeners:
  - name: ai-gateway
    address: "127.0.0.1:8080"
    filter_chains: [responses-pipeline]

filter_chains:
  - name: responses-pipeline
    filters:
      - filter: openai_responses_request
        on_invalid: reject

      - filter: state_owner
        mode: single_tenant
        tenant_id: default

      - filter: openai_response_store
        backend: sqlite
        # In-memory:
        #   database_url: "sqlite::memory:"
        # File-backed:
        database_url: "sqlite://responses.db?mode=rwc"
        responses_table: openai_responses
        conversations_table: openai_conversations
        #
        # Connection pool tuning (optional, sqlx defaults apply when omitted):
        #   pool:
        #     max_connections: 10      # maximum pool connections (default: 10)
        #     min_connections: 0       # minimum idle connections (default: 0)
        #     idle_timeout_secs: 600   # seconds before idle connections close (default: 600, 0 to disable)
        #     acquire_timeout_secs: 30 # seconds to wait for a connection (default: 30)
        #
        # At-rest payload compression (optional, defaults to none):
        #   Compresses the responses table's stored JSON payloads
        #   (response_object, input, messages) with zstd. These columns are
        #   binary (BLOB on sqlite, BYTEA on postgres): uncompressed rows hold
        #   raw JSON bytes and compressed rows hold a raw zstd frame. The read
        #   path detects the zstd frame magic and decompresses transparently, so
        #   uncompressed rows stay readable after enabling compression. Works
        #   with both the sqlite and postgres backends.
        #   compression:
        #     algorithm: zstd          # none (default) or zstd
        #     level: 3                 # optional zstd level (default: 3)
        #
        #   Upgrading an existing database: schema version 3 and later store
        #   these responses columns as binary. Databases created under an older
        #   schema version must be migrated before the proxy will start — see
        #   docs/store/schema-migration.md.
        #
        # PostgreSQL backend:
        #   backend: postgres
        #   database_url: "postgres://user:pass@db.example.com:5432/praxis"
        #   responses_table: openai_responses
        #   conversations_table: openai_conversations
        #   allow_private_database_url: true  # required for DNS, local, or private DB targets
        #   ssl_mode: verify-full      # disable, prefer, require, verify-ca, verify-full (default: verify-full)
        #   ssl_root_cert: /path/to/ca.pem  # optional, for verify-ca / verify-full

      - filter: iterative_request_router
        initial_step: inference
        max_iterations: 1
        # The IRR's default 30s end-to-end deadline is too short for model
        # inference. This single step inherits a 6-minute total deadline for
        # time-to-first-byte plus response streaming.
        timeout_ms: 360000
        steps:
          - name: inference
            filters:
              # Selects the response transport per request: Praxis's typed
              # streaming transport for an effective `stream: true`, otherwise the
              # buffered transport. Named `inference` so branch chains can rejoin.
              - filter: openai_responses_proxy

              - filter: router
                routes:
                  - path: "/v1/responses"
                    cluster: "inference-backend"

              - filter: load_balancer
                clusters:
                  - name: "inference-backend"
                    # Cluster ceiling. openai_stream_events caps the live body
                    # at leftover timeout_secs after the first SSE chunk.
                    read_timeout_ms: 300000
                    endpoints:
                      - "127.0.0.1:8000"

              # After load_balancer so IRR body hooks see the selected peer.
              # openai_responses_proxy buffers the request body, so IRR runs that
              # phase before load balancing; an earlier placement never sees
              # ctx.upstream. It arms only for a streaming request and passes a
              # finite response through untouched. timeout_secs is an absolute
              # deadline from the first SSE chunk (default 300).
              - filter: openai_stream_events
            on_result:
              - default: true
                done: true

insecure_options:
  allow_private_endpoints: true # example proxies to local backends
