components:
  parameters: {}
  responses:
    BadRequest:
      content:
        application/json:
          example:
            error:
              code: context_length_exceeded
              message: prompt_tokens + max_tokens (9000) exceeds the model's context
                length (8192).
              type: invalid_request_error
          schema:
            $ref: '#/components/schemas/ErrorEnvelope'
      description: 'Invalid request: a guardrail was violated (max_tokens > cap, context

        exceeded) or the body is malformed.

        '
    ExternalProviderNotEnabled:
      content:
        application/json:
          example:
            error:
              code: external_provider_not_enabled
              message: The model 'frontier-x-large' is served by an external inference
                provider, outside Shadow's own EU-sovereign infrastructure. Your organization
                has not enabled external providers. Contact your administrator to
                enable them.
              param: model
              type: invalid_request_error
          schema:
            $ref: '#/components/schemas/ErrorEnvelope'
      description: 'The model is served by an external inference provider and the
        caller''s

        organization has not enabled external providers (PAAS-206 / PAAS-32 X1).


        Deliberately a 403 that NAMES the model, where an internal-only model answers
        404:

        the model''s existence is not a secret here — only its enablement is — and
        the

        error is the enablement funnel. Raised before the rate limiter and before
        any

        upstream call, so a refused request costs nothing and, in particular, spends
        no

        supplier money. Retrying will not help; an administrator enabling external

        providers for the organization will.

        '
    ModelNotFound:
      content:
        application/json:
          example:
            error:
              code: model_not_found
              message: The model 'gpt-5' does not exist.
              type: invalid_request_error
          schema:
            $ref: '#/components/schemas/ErrorEnvelope'
      description: Model unknown to the catalog.
    ModelUnavailable:
      content:
        application/json:
          example:
            error:
              code: model_unavailable
              message: The model is currently unavailable. Please retry.
              type: api_error
          schema:
            $ref: '#/components/schemas/ErrorEnvelope'
      description: 'Model unavailable (failed cold start, no replica). The client
        may

        retry.

        '
    NotFound:
      content:
        application/json:
          schema:
            $ref: '#/components/schemas/ErrorEnvelope'
      description: Resource not found.
    PayloadTooLarge:
      content:
        application/json:
          example:
            error:
              code: request_too_large
              message: Request body exceeds the maximum allowed size.
              type: invalid_request_error
          schema:
            $ref: '#/components/schemas/ErrorEnvelope'
      description: Request body beyond the limit (size guardrail).
    RateLimited:
      content:
        application/json:
          example:
            error:
              code: rate_limit_exceeded
              message: Rate limit reached for requests per minute. Please retry after
                12 second(s).
              type: rate_limit_error
          schema:
            $ref: '#/components/schemas/ErrorEnvelope'
      description: "Rate limit reached. Codes:\n- `rate_limit_exceeded`: the key's\
        \ or the user's requests per minute (RPM)\n  or tokens per minute (TPM) are\
        \ used up; retry after `Retry-After`.\n- `insufficient_quota`: the user's\
        \ daily token quota is used up; it\n  resets at midnight UTC.\nLimits resolve\
        \ per key first, then per user, then to the platform\ndefaults. The OpenAI\
        \ SDKs honour `Retry-After` natively (automatic\nbackoff). The video route\
        \ also returns its own 429,\n`video_jobs_quota_exceeded`, when too many jobs\
        \ are still running; that\ncap is separate from rate limiting.\n"
      headers:
        Retry-After:
          description: Seconds to wait before retrying.
          schema:
            type: integer
    Unauthorized:
      content:
        application/json:
          example:
            error:
              code: invalid_api_key
              message: Invalid API key.
              type: authentication_error
          schema:
            $ref: '#/components/schemas/ErrorEnvelope'
      description: Missing, malformed, unknown, or revoked key.
    UpstreamError:
      content:
        application/json:
          example:
            error:
              code: upstream_error
              message: The upstream model server returned an error.
              type: api_error
          schema:
            $ref: '#/components/schemas/ErrorEnvelope'
      description: Unrecoverable upstream error (vLLM/Forge).
    UpstreamTimeout:
      content:
        application/json:
          example:
            error:
              code: upstream_timeout
              message: The request to the upstream model server timed out.
              type: api_error
          schema:
            $ref: '#/components/schemas/ErrorEnvelope'
      description: Upstream timeout exceeded (cold start too long, slow generation).
  schemas:
    ChatChoice:
      properties:
        finish_reason:
          description: e.g. stop, length, content_filter.
          nullable: true
          type: string
        index:
          type: integer
        message:
          $ref: '#/components/schemas/ChatMessage'
      required:
      - index
      - message
      type: object
    ChatCompletion:
      properties:
        choices:
          items:
            $ref: '#/components/schemas/ChatChoice'
          type: array
        created:
          type: integer
        id:
          type: string
        model:
          type: string
        object:
          enum:
          - chat.completion
          type: string
        routing:
          allOf:
          - $ref: '#/components/schemas/ResidencyRouting'
          description: 'Present ONLY when the request was served by an external inference
            provider

            (PAAS-206). Mutually exclusive with `pricing` by construction: a spot
            price

            describes our own fleet''s capacity market, and an external call is not
            in it.

            On a stream the same object arrives as a `choices: []` chunk just before

            `data: [DONE]`; on both surfaces the `X-Shadow-Residency: external` header

            carries it too, which is the half that survives a stream breaking mid-answer.

            '
        usage:
          $ref: '#/components/schemas/Usage'
      required:
      - id
      - object
      - created
      - model
      - choices
      type: object
    ChatCompletionRequest:
      additionalProperties: true
      properties:
        frequency_penalty:
          type: number
        max_completion_tokens:
          description: 'OpenAI''s current name for `max_tokens`, and the same knob
            here: same cap, same

            context check, and an error about it names this field. Sent together,
            this one

            wins (as upstream).

            '
          minimum: 1
          type: integer
        max_tokens:
          description: 'Capped by the gateway: ≤ the model''s `max_output_tokens`
            (published on

            `GET /v1/models`; the global guard-rail unless the model raises it) AND

            prompt_tokens + max_tokens ≤ the model''s max_model_len.


            OMITTED is the normal case and costs nothing: the engine then fills the
            window it

            has left, computed from the real prompt length. The gateway states a ceiling

            upstream only when its own cap is the binding one — it never derives a
            ceiling

            from its own prompt ESTIMATE, which is how a request with no `max_tokens`
            used to

            be refused outright on a model whose cap equals its window.

            '
          minimum: 1
          type: integer
        messages:
          items:
            $ref: '#/components/schemas/ChatMessage'
          minItems: 1
          type: array
        model:
          description: Public model_id from the catalog.
          example: qwen2.5-3b
          type: string
        n:
          default: 1
          description: 'Number of completions. n>1 ⇒ num_requests=n for metering.


            Requires a non-zero `temperature`: below 1e-5 the engine samples greedily,
            every

            completion would be identical, and it refuses the request rather than
            return

            copies. Asking for both is a 400 naming `n`, raised before any GPU is
            woken.

            '
          minimum: 1
          type: integer
        presence_penalty:
          type: number
        seed:
          type: integer
        stop:
          oneOf:
          - type: string
          - items:
              type: string
            type: array
        stream:
          default: false
          description: 'If true, response as an SSE stream (native relay — see STREAMING.md).

            '
          type: boolean
        stream_options:
          $ref: '#/components/schemas/StreamOptions'
        temperature:
          maximum: 2
          minimum: 0
          type: number
        tools:
          description: 'Function declarations, relayed to vLLM as-is, AND/OR the gateway''s
            built-in

            server tools (`{"type": "web_search"}`, `{"type": "web_fetch"}`,

            `{"type": "code_interpreter"}`) — the same vocabulary /v1/responses accepts.

            A built-in entry is executed by the gateway inside its own tool loop and

            stripped from the upstream body; an unknown

            `type` is a 400 with `param: "tools"`, which narrows the older "out-of-scope

            fields are relayed as-is" contract on purpose (a `{"type": …}` vLLM does
            not

            understand is an upstream 400 at best and silently ignored at worst).

            `{"type": "web_search"}` and `web_search_options` are the same switch:

            agreeing duplicates are merged, disagreeing ones are a 400.

            '
          items:
            oneOf:
            - $ref: '#/components/schemas/ServerToolDeclaration'
            - type: object
          type: array
        top_p:
          type: number
        web_search_options:
          additionalProperties: true
          description: 'Opt-in web search (PAAS-5). When present, the gateway injects
            a `web_search` tool, executes model-emitted searches server-side via the
            Staan WebSearch4AI vendor (queries LEAVE Shadow infrastructure — see docs/SERVER-TOOLS.md),
            and returns a cited answer. Each executed search is billed at defaults.price_web_search.
            Requires the account flag `web_search_enabled` and a model with native
            tool calling. Mirrors OpenAI''s field of the same name.

            '
          properties:
            search_context_size:
              description: Results per search (3/5/8). Default medium.
              enum:
              - low
              - medium
              - high
              type: string
          type: object
      required:
      - model
      - messages
      type: object
    ChatMessage:
      additionalProperties: true
      properties:
        annotations:
          description: 'Web-search citations (OpenAI url_citation shape). `index`
            matches the inline `[n]` markers in the content.

            '
          items:
            properties:
              type:
                enum:
                - url_citation
                type: string
              url_citation:
                properties:
                  index:
                    type: integer
                  title:
                    type: string
                  url:
                    type: string
                type: object
            type: object
          type: array
        content:
          description: 'A string, or an array of parts for a multimodal prompt. An

            `{"type": "image_url", "image_url": {"url": …}}` part accepts a base64
            data URI

            (`data:image/png;base64,…`) or an `http(s)` URL, on any model whose

            `architecture.input_modalities` includes `image`.


            Both are resolved by the GATEWAY before the model sees them: a data URI
            is decoded

            and identified from its own bytes (the declared media type is ignored
            — clients

            label JPEGs `image/png` routinely), and a remote URL is fetched here,
            through the

            egress guard, with a real User-Agent. So a bad image is a 400 `invalid_image`
            that

            names the offending part, never an opaque error from inside the serving
            container,

            and a URL that only some origins will serve is not a coin toss. Formats:
            PNG, JPEG,

            GIF, WEBP, BMP, TIFF. Limits: 8 MB per image, 8 fetched URLs per request
            (data URIs

            are bounded by the request-body limit instead).

            '
          nullable: true
          oneOf:
          - type: string
          - items:
              type: object
            type: array
        name:
          type: string
        role:
          example: user
          type: string
      required:
      - role
      type: object
    Completion:
      properties:
        choices:
          items:
            properties:
              finish_reason:
                nullable: true
                type: string
              index:
                type: integer
              text:
                type: string
            type: object
          type: array
        created:
          type: integer
        id:
          type: string
        model:
          type: string
        object:
          enum:
          - text_completion
          type: string
        usage:
          $ref: '#/components/schemas/Usage'
      required:
      - id
      - object
      - created
      - model
      - choices
      type: object
    CompletionRequest:
      additionalProperties: true
      properties:
        max_tokens:
          type: integer
        model:
          type: string
        n:
          default: 1
          type: integer
        prompt:
          oneOf:
          - type: string
          - items:
              type: string
            type: array
        seed:
          type: integer
        stop:
          oneOf:
          - type: string
          - items:
              type: string
            type: array
        stream:
          default: false
          type: boolean
        stream_options:
          $ref: '#/components/schemas/StreamOptions'
        temperature:
          type: number
        top_p:
          type: number
      required:
      - model
      - prompt
      type: object
    DedicatedCreateRequest:
      properties:
        alias_slug:
          description: Alias suffix — the full alias is "ded/<slug>", globally unique.
          pattern: ^[a-z0-9-]{3,24}$
          type: string
        engine:
          $ref: '#/components/schemas/DedicatedEngineConfig'
          description: 'Optional Class-B engine document, validated against what the
            model

            offers (`ModelInfo.dedicated_engine`) and baked by the FIRST deploy —

            the one moment an engine setting costs no extra weight reload

            (code `engine_setting_unavailable` on a setting the model does not

            offer). Omitted → the recipe''s defaults.

            '
        max_replicas:
          description: Autoscaling ceiling, capped by `max_replicas_cap` (default
            16).
          maximum: 32
          minimum: 1
          type: integer
        min_replicas:
          description: 'Replicas kept running at all times. 0 = scale-to-zero: nothing
            is

            billed at rest, but the first call has to boot a replica. Capped by

            the organisation''s `max_replicas_cap` (default 16).

            '
          maximum: 32
          minimum: 0
          type: integer
        mode:
          enum:
          - spot
          - reserved
          - hybrid
          type: string
        model_id:
          description: 'Active catalog model, excluding `gpt-oss-20b`, `orpheus-tts-3b`,

            `ltx-video-2b` and `gpt-oss-120b` (code `model_not_eligible`).

            '
          type: string
        reserved_floor:
          description: Hybrid mode only — reserved floor (≤ max_replicas).
          maximum: 32
          minimum: 0
          type: integer
        scaling:
          $ref: '#/components/schemas/DedicatedScalingPolicy'
          description: 'Optional autoscaling policy. Stored on the endpoint and asserted

            once provisioning succeeds — the deploy bakes its own autoscaling,

            which would otherwise win silently. Omitted → the model''s default

            preset (`scaling_bounds.default_profile`).

            '
      required:
      - model_id
      - alias_slug
      - mode
      - min_replicas
      - max_replicas
      type: object
    DedicatedEndpoint:
      description: 'PRIVATE instance of a catalog model, addressed by `alias` (the
        value

        of the `model` field in OpenAI requests) and billed per GPU-card-minute

        instead of per token. The name of the underlying private deployment is

        OPAQUE and is never exposed on customer routes.

        '
      properties:
        alias:
          description: '"ded/<slug>" — to be put in the `model` field of requests.'
          type: string
        capacity_converging:
          description: 'HYBRID ONLY, and false while replicas created under an EARLIER

            reserved floor are still running on the wrong capacity class. A

            replica''s class is fixed when it starts and is never rewritten, so

            raising the floor live converges only as the endpoint adds

            replicas — this flag is how the customer is told the floor they

            just set is not yet the floor they have.

            '
          type: boolean
        cards_per_replica:
          description: GPU cards per replica (inherited from the model; multiplies
            the price).
          type: integer
        created_at:
          format: date-time
          type: string
        endpoint_id:
          description: Identifier "ded_<hex>".
          type: string
        engine:
          allOf:
          - $ref: '#/components/schemas/DedicatedEngineConfig'
          description: 'The customer''s Class-B engine document (SPEC-DEDICATED §12.3),
            or

            `null` when the endpoint runs the recipe''s own defaults. A COMPLETE

            REPLACEMENT — what it does not name is the default. Changed only

            through `PUT …/engine` (a billed redeploy), never by the PATCH.

            '
          nullable: true
        engine_bounds:
          $ref: '#/components/schemas/DedicatedEngineBounds'
          description: 'Per setting, whether THIS model offers it, why not when it
            does not,

            the default the recipe runs with and the admissible values. Render

            the engine controls from these and only these: availability is a

            fact of the catalogue AND of the deploy script, never of a name.

            '
        engine_changed_at:
          description: 'When the engine settings were last changed, `null` if never.
            Shares

            the redeploy cooldown with `mode_changed_at` (`mode_change_too_soon`).

            '
          format: date-time
          nullable: true
          type: string
        live:
          description: 'Short-TTL snapshot of the running replicas; null while nothing
            is

            deployed yet (or if the snapshot could not be read).

            '
          nullable: true
          properties:
            replicas_ready:
              description: Replicas able to serve a request right now.
              type: integer
            replicas_total:
              description: 'Replicas placed on a GPU. A replica still booting counts
                — it

                holds its GPU, and it bills.

                '
              type: integer
          type: object
        max_replicas:
          description: 'Absolute ceiling is 32; the EFFECTIVE cap is per-organisation

            (default 16, 32 for an internal organisation) and readable as

            `max_replicas_cap` on `GET /api/dedicated/rates`.

            '
          maximum: 32
          minimum: 1
          type: integer
        min_replicas:
          minimum: 0
          type: integer
        mode:
          enum:
          - spot
          - reserved
          - hybrid
          type: string
        mode_changed_at:
          description: 'When the capacity mode was last changed, `null` if it never
            was.

            The cooldown clock: each change restarts the endpoint and bills a

            weight reload, so another one is refused for a while afterwards

            (`mode_change_too_soon` — SPEC-DEDICATED §12.9 anti-abuse).

            '
          format: date-time
          nullable: true
          type: string
        model_id:
          description: Cloned catalog model (serving recipe).
          type: string
        month_cost:
          description: 'Current month''s cost, in the `currency` of

            `GET /api/dedicated/rates` (card-minutes × dedicated rate card).

            Counts every minute a replica held a GPU, boot windows included.


            Currency-neutral replacement for `month_cost_usd`, which is

            DEPRECATED and still answers the SAME number (D10/D11). Additive on

            purpose: a client reading the name the service stopped writing gets

            a silent zero, and a cost that silently reads zero is PAAS-601.

            '
          type: number
        month_cost_rate_matches_card:
          description: 'Whether every minute in `month_cost` found a rate for THIS
            endpoint''s GPU

            card and capacity class (`hourly_by_gpu_type`, PAAS-278). `false` means
            some

            minutes had none and count as ZERO — which is also what is invoiced for

            them — so the figure is the bill, but the endpoint sits on a card or class

            nobody priced (one created before its card had a rate).


            ABSENT when the model has left the catalogue: its card is then unknown,
            and

            "we cannot tell" is not the same claim as "the rate was right". Also absent

            on gateways predating the field.

            '
          type: boolean
        month_cost_usd:
          deprecated: true
          description: 'DEPRECATED — read `month_cost`. Same number, and no longer
            USD: the

            rate card is EUR since D10/D11 while this field name is not. Kept

            so no client is cut over on a flag day; removed in a later MR.

            '
          type: number
        month_minutes:
          description: 'Replica-minutes sampled over the current month, per capacity
            class.

            A minute is sampled for every replica holding a GPU, whether or not

            it was ready to serve.

            '
          properties:
            reserved:
              type: number
            spot:
              type: number
          required:
          - reserved
          - spot
          type: object
        previous_mode:
          description: 'The mode in force before the last change, `null` if there
            was none.

            The ROLLBACK TARGET: a redeploy that fails to boot puts the endpoint

            back on the configuration that was serving rather than leaving it in

            `error` holding cards (§12.9).

            '
          enum:
          - spot
          - reserved
          - hybrid
          nullable: true
          type: string
        reserved_floor:
          description: Always-on floor in reserved (hybrid mode only, 0 otherwise).
          type: integer
        scaling:
          allOf:
          - $ref: '#/components/schemas/DedicatedScalingState'
          description: 'The customer''s STORED autoscaling policy, or `null` when
            they never

            set one — the endpoint then runs on

            `scaling_bounds.profiles[scaling_bounds.default_profile]`, which the

            gateway asserts for it. Keys the customer never set are ABSENT here,

            not null.

            '
          nullable: true
        scaling_bounds:
          $ref: '#/components/schemas/DedicatedScalingBounds'
          description: 'Per-model ceilings, queue-signal capability and resolved presets.

            Size every scaling control from these — the admission bound is per

            model, so a constant would offer values the gateway refuses.

            '
        status:
          description: '`active` means the endpoint can serve a request RIGHT NOW

            (`live.replicas_ready >= 1`). `warming` means the deployment exists

            and its replicas hold their GPUs — so it IS billing — while the

            model weights load, but none can serve yet: right after a create or

            a resume, and while a lost replica is rebuilt. A call made during

            `warming` gets a retryable `503 model_warming` with a `Retry-After`

            header, not a dropped connection. `paused` and `deleted` hold no

            GPU and bill nothing.


            `updating` is DERIVED the same way `warming` is (PAAS-183): the

            STORED lifecycle is still `active` — the GPUs are held, the endpoint

            keeps billing and the reconciler keeps treating it as deployed —

            while a redeploy is in flight for it (a capacity-mode change). It

            OUTRANKS `warming`: a customer in the middle of a billed weight

            reload needs to be told that, not that a replica is starting. A

            `paused` endpoint never presents `updating`.

            '
          enum:
          - provisioning
          - warming
          - updating
          - active
          - paused
          - error
          - deleted
          type: string
      required:
      - endpoint_id
      - model_id
      - alias
      - mode
      - min_replicas
      - max_replicas
      - reserved_floor
      - status
      - created_at
      - cards_per_replica
      - live
      - month_minutes
      - month_cost
      - month_cost_usd
      - scaling
      - scaling_bounds
      - mode_changed_at
      - previous_mode
      - engine
      - engine_bounds
      - engine_changed_at
      type: object
    DedicatedEngineBounds:
      description: 'The engine controls resolved for ONE model — the single definition
        of

        what is offered, served on the endpoint and on `ModelInfo` so both

        wizards render from the same verdicts.

        '
      properties:
        max_model_len:
          $ref: '#/components/schemas/DedicatedEngineSetting'
        max_num_seqs:
          $ref: '#/components/schemas/DedicatedEngineSetting'
        prefix_caching:
          $ref: '#/components/schemas/DedicatedEngineSetting'
        speculative_tokens:
          $ref: '#/components/schemas/DedicatedEngineSetting'
      required:
      - max_model_len
      - max_num_seqs
      - speculative_tokens
      - prefix_caching
      type: object
    DedicatedEngineConfig:
      additionalProperties: false
      description: 'The engine document: a COMPLETE REPLACEMENT of the recipe''s defaults
        for

        the settings it names. Only settings the model offers may appear

        (`DedicatedEngineBounds.<setting>.available`).

        '
      properties:
        max_model_len:
          description: 'Context window in tokens, `[1024, the catalogue''s max_model_len]`.

            Lowering it frees KV cache — which is what pays a bigger batch; the

            catalogue value is the ceiling because nothing above it was ever

            validated in quality.

            '
          minimum: 1024
          type: integer
        max_num_seqs:
          description: 'Generation batch (`--max-num-seqs`), `[1, the model''s

            max_concurrent_inputs]`. Offered only where the catalogue declares

            the recipe''s default batch.

            '
          minimum: 1
          type: integer
        prefix_caching:
          description: '`--enable-prefix-caching`. Offered only where the cache can
            hit:

            `inert` on hybrid Gated-DeltaNet/Mamba models (the recurrent state

            does not exist at the shared-prefix boundary), `locked_off` where

            the flag off is a bug fix.

            '
          type: boolean
        speculative_tokens:
          description: 'Speculative depth k through the model''s OWN multi-token-prediction

            head; 0 = off. Offered only on checkpoints that carry the head, and

            only in the depths the catalogue declares (`choices`) — on

            qwen3.6-35b-a3b that is `2`: 2.6 tokens per step, decode −26 %,

            TTFT +12 %, greedy traffic only (speculation collapses under

            sampling; `min_p` and `logit_bias` are ignored while it is on).

            '
          minimum: 0
          type: integer
      type: object
    DedicatedEngineSetting:
      description: 'One engine setting resolved for ONE model. `available` false comes
        with

        a stable `reason`: `script_ignores_setting` (the deploy script does not

        read the variable, or does not carry it into the runtime through

        `SI_RUNTIME_SECRET` — a stored value would be obeyed by nobody),

        `default_unknown` (the catalogue does not declare the recipe''s batch),

        `no_mtp_head` (nothing to draft with), `prefix_cache_inert`,

        `prefix_cache_locked_off`, `model_unknown`. Render an unavailable

        setting DISABLED with its reason, not hidden; never send it.

        '
      properties:
        available:
          type: boolean
        choices:
          description: 'For `speculative_tokens`: the admissible depths, 0 first.'
          items:
            type: integer
          type: array
        default:
          description: The value the recipe runs with when the document does not name
            the setting.
          nullable: true
        max:
          nullable: true
          type: integer
        min:
          nullable: true
          type: integer
        reason:
          nullable: true
          type: string
      required:
      - available
      - reason
      - default
      type: object
    DedicatedModeChange:
      allOf:
      - $ref: '#/components/schemas/DedicatedModeTarget'
      - properties:
          confirm_redeploy:
            description: 'Must be exactly `true` (400, code `confirmation_required`).
              Send

              it only after showing the customer the preview''s cost and

              unavailable window; a console must never set it implicitly.

              '
            enum:
            - true
            type: boolean
        required:
        - confirm_redeploy
        type: object
      description: '`PUT /api/dedicated/{endpoint_id}/mode` body: a target plus an
        EXPLICIT

        confirmation. Distinct from the PATCH rescale on purpose

        (SPEC-DEDICATED §12.10) — a billed reboot must never be reachable from a

        payload that reads like a slider move.

        '
    DedicatedModePreview:
      description: 'Cost/downtime quote for a capacity-mode change. READ-ONLY: nothing
        is

        deployed, stored or billed by asking for it.

        '
      properties:
        allowed:
          description: 'Whether the `PUT` would be accepted right now. False does
            NOT mean

            the quote is wrong — the figures are still what the change would

            cost once the blocker clears.

            '
          type: boolean
        blocked_reason:
          description: 'The stable code the `PUT` would answer with (`endpoint_paused`,

            `endpoint_busy`, `endpoint_not_ready`, `mode_change_too_soon`,

            `mode_unchanged`, `mode_not_priced`, `model_not_eligible`, …), or

            `null`. It lets a

            control be disabled WITH a reason instead of failing on submit.

            '
          nullable: true
          type: string
        cooldown_seconds_remaining:
          description: 'Seconds before another mode change is accepted; 0 when none
            is

            pending. Each change restarts the endpoint and bills a weight

            reload, so they are rate-limited (§12.9 anti-abuse).

            '
          type: integer
        endpoint_id:
          type: string
        from:
          description: The configuration serving right now.
          properties:
            mode:
              enum:
              - spot
              - reserved
              - hybrid
              type: string
            reserved_floor:
              type: integer
          required:
          - mode
          - reserved_floor
          type: object
        kind:
          description: 'Always `redeploy`, and the enum is closed deliberately: a
            replica''s

            capacity class is chosen when the replica is created and never

            rewritten, so NO cross-mode transition can be applied to a running

            replica. That includes spot ↔ hybrid, which reads like a live

            change and is not — raising the reserved floor creates no reserved

            replica until the endpoint next scales UP, and the shed order keeps

            the replicas that should have been replaced alive longest.

            '
          enum:
          - redeploy
          type: string
        rate_per_card_minute_after:
          description: 'The rate that applies afterwards — the DURABLE part of the
            decision,

            as opposed to the one-off reload cost.

            '
          type: number
        rate_per_card_minute_before:
          description: The per-card-minute rate in force today.
          type: number
        reload:
          description: 'What the redeploy itself costs. The replicas are REPLACED:
            they hold

            their GPUs for the whole weight reload and those card-minutes bill

            (§4), priced the way the minute sampler bills the TARGET

            configuration — the reserved floor''s share at the reserved rate,

            the rest at the spot rate.

            '
          properties:
            cards_per_replica:
              description: GPU cards per replica; it multiplies the reload's cost.
              type: integer
            currency:
              description: 'Same pin as `GET /api/dedicated/rates` — `eur` since D10/D11,

                over unchanged rates.

                '
              enum:
              - eur
              type: string
            estimate_source:
              description: '`observed` = timed on THIS endpoint''s own last start-up.

                `default` = a platform average, because this endpoint has never

                been timed. NEVER present a `default` as if it had been

                measured: on a large model it can be off by minutes, and that is

                the kind of number that ends in a support ticket.

                '
              enum:
              - observed
              - default
              type: string
            estimated_boot_seconds:
              description: How long one replica takes to load its weights and become
                ready.
              type: integer
            estimated_cost:
              description: 'cards_per_replica × (estimated_boot_seconds / 60) ×

                (replicas_reserved × rate_per_card_minute_reserved +

                replicas_spot × rate_per_card_minute_spot). An ESTIMATE — it

                moves with the observed boot time — and the number to print on

                the confirmation.


                Currency-neutral replacement for `estimated_cost_usd`, which is

                DEPRECATED and carries the SAME number.

                '
              type: number
            estimated_cost_usd:
              deprecated: true
              description: DEPRECATED — read `estimated_cost`. Same number, not USD.
              type: number
            rate_per_card_minute:
              description: 'The BURST rate of the target class — the one rate the
                class

                decision turns on, and a display. It is not the rate the whole

                reload bills at: on a hybrid target the floor''s replicas bill

                at `rate_per_card_minute_reserved`.

                '
              type: number
            rate_per_card_minute_reserved:
              type: number
            rate_per_card_minute_spot:
              type: number
            replicas_reloaded:
              description: 'Replicas that will be replaced — `live.replicas_total`
                while the

                endpoint is running, `min_replicas` when it is at rest.

                '
              type: integer
            replicas_reserved:
              description: 'Of those, how many bill at the reserved rate: all of them
                on a

                reserved target, `min(replicas_reloaded, reserved_floor)` on a

                hybrid one, none on spot.

                '
              type: integer
            replicas_spot:
              description: The rest, billed at the spot rate.
              type: integer
          required:
          - replicas_reloaded
          - replicas_reserved
          - replicas_spot
          - cards_per_replica
          - estimated_boot_seconds
          - estimate_source
          - rate_per_card_minute
          - rate_per_card_minute_reserved
          - rate_per_card_minute_spot
          - currency
          - estimated_cost
          - estimated_cost_usd
          type: object
        to:
          description: 'The configuration that would be deployed. `reserved_floor`
            is forced

            to 0 outside hybrid — a floor is meaningless there, and it is

            persisted as 0 rather than kept as a value nothing reads.

            '
          properties:
            mode:
              enum:
              - spot
              - reserved
              - hybrid
              type: string
            reserved_floor:
              type: integer
          required:
          - mode
          - reserved_floor
          type: object
        unavailable_window_seconds:
          description: 'How long the endpoint cannot serve. Calls made inside it get
            a

            retryable error, not a dropped connection.

            '
          type: integer
      required:
      - endpoint_id
      - from
      - to
      - kind
      - allowed
      - blocked_reason
      - reload
      - unavailable_window_seconds
      - rate_per_card_minute_before
      - rate_per_card_minute_after
      - cooldown_seconds_remaining
      type: object
    DedicatedModeTarget:
      description: 'The capacity configuration being ASKED FOR. The same shape on
        the

        preview and on the change itself, so a quote and the write it justifies

        cannot describe two different transitions.

        '
      properties:
        mode:
          description: 'Target capacity class. On `PUT` it must DIFFER from the endpoint''s

            current mode (400, code `mode_unchanged`): a no-op would still

            redeploy and bill a weight reload for nothing.

            '
          enum:
          - spot
          - reserved
          - hybrid
          type: string
        reserved_floor:
          description: 'HYBRID ONLY — refused outside it — and then REQUIRED and ≥
            1 (400,

            code `reserved_floor_required`). A hybrid endpoint with a floor of 0

            deploys every replica on preemptible capacity, which is what spot

            mode already is: it would be billed and labelled as something it is

            not. Capped by `max_replicas` AND by the organisation''s effective

            replica cap (`max_replicas_cap` on `GET /api/dedicated/rates`,

            code `replica_cap_exceeded`).

            '
          maximum: 32
          minimum: 1
          type: integer
      required:
      - mode
      type: object
    DedicatedPatchRequest:
      description: 'Omitted fields = unchanged. `reserved_floor` rejected outside
        hybrid mode.


        A `mode` key is REFUSED here (400, code `mode_change_requires_redeploy`),

        before anything else is applied: changing the capacity mode is not a

        live change — it restarts the endpoint and the weight reload is billed —

        so it has its own verb, `PUT /api/dedicated/{endpoint_id}/mode`, which

        quotes the cost first. It used to be accepted and silently ignored,

        which is the failure mode §12.7 exists to forbid.


        ONE ASYMMETRY, and it is deliberate: the replica bounds keep

        omitted-means-unchanged, while `scaling` is a COMPLETE REPLACEMENT of

        the stored policy — the scheduling backend swaps its policy document

        wholesale, so merging on our side would leave the endpoint running a

        policy nobody ever sent. Send `"scaling": null` to drop the override and

        go back to the model''s default preset; OMIT the key to leave the policy

        alone. An empty object is refused (400): clearing is done with null,

        never with `{}`.

        '
      properties:
        max_replicas:
          maximum: 32
          minimum: 1
          type: integer
        min_replicas:
          maximum: 32
          minimum: 0
          type: integer
        reserved_floor:
          maximum: 32
          minimum: 0
          type: integer
        scaling:
          allOf:
          - $ref: '#/components/schemas/DedicatedScalingPolicy'
          description: 'COMPLETE REPLACEMENT of the stored policy. `null` restores
            the

            model''s default preset — an OBSERVABLE change, not a no-op: the

            endpoint goes back to the gateway''s own default profile, never to

            the scheduling backend''s null policy (the documented cause of

            replica flapping, PAAS-197).

            '
          nullable: true
      type: object
    DedicatedPercentiles:
      description: 'A distribution over the requests that REPORTED the figure, never
        over

        every request: `measured_requests` is the denominator. The percentiles

        are `null` when it is 0 — an upstream that never reported a value has no

        p50, not a 0 ms one.

        '
      properties:
        max:
          nullable: true
          type: integer
        measured_requests:
          type: integer
        p50:
          description: Median (the typical request).
          nullable: true
          type: integer
        p95:
          description: The slowest 5 %.
          nullable: true
          type: integer
      required:
      - p50
      - p95
      - measured_requests
      type: object
    DedicatedPrefixCache:
      description: 'Prefix reuse over the window: the share of the prompt the engine
        had

        already processed and did not process again. On a dedicated endpoint

        reuse buys SPEED, not money — tokens are not billed here at all.

        '
      properties:
        block_source:
          description: '`observed` or `null` — there is no third source. The absence
            of a

            value is reported as such rather than filled in.

            '
          enum:
          - observed
          nullable: true
          type: string
        block_tokens:
          description: 'The model''s reuse granularity, OBSERVED on a live replica:
            prefixes

            are reused whole blocks at a time, and a shared prefix shorter than

            one block is never reused at all — which is the one actionable

            sentence this whole block exists to say. `null` when it has never

            been sampled, and deliberately NOT inferred from the catalogue''s

            declared architecture, which is wrong on the very models that show a

            zero hit rate (§12.13). A guessed block size turns that sentence

            into a lie.

            '
          nullable: true
          type: integer
        cached_prompt_tokens:
          type: integer
        coverage:
          description: 'measured_requests / requests.ok. A 0 % hit rate is only readable

            next to this: without it, "no reuse at all" and "the figure was

            never reported" look the same.

            '
          type: number
        hit_rate:
          description: 'cached_prompt_tokens / prompt_tokens, or `null` when nothing
            in the

            window reported a cached count. `null` and `0` are OPPOSITE

            findings — one says nothing was measured, the other says reuse was

            measured and there was none — and a client must render them

            differently.

            '
          nullable: true
          type: number
        measured_requests:
          description: Completed requests that reported a cached count.
          type: integer
        prompt_tokens:
          description: 'Prompt tokens over the rows that reported a cached count —
            the

            denominator of `hit_rate`, and NOT the window''s total prompt tokens

            (`tokens.prompt`), which includes rows that reported nothing.

            '
          type: integer
      required:
      - prompt_tokens
      - cached_prompt_tokens
      - hit_rate
      - measured_requests
      - coverage
      - block_tokens
      - block_source
      type: object
    DedicatedReplicaHealth:
      description: 'Live per-replica speed. It answers the one failure nothing else
        sees: a

        replica that is up, ready and answering at a fraction of its peers''

        speed never makes anything queue, so the autoscaler is blind to it and

        the endpoint will not replace it on its own.

        '
      properties:
        expected_tok_s:
          description: The speed this model is expected to reach on this hardware.
          nullable: true
          type: number
        fastest_tok_s:
          nullable: true
          type: number
        queue:
          description: Requests in flight and waiting; null when unreadable.
          nullable: true
          properties:
            running:
              type: integer
            waiting:
              type: integer
          required:
          - running
          - waiting
          type: object
        reason:
          description: 'Why there is no sample: `no_ready_replica` (nothing is running
            — the

            endpoint was not woken to find out) or `no_metrics` (this model''s

            server publishes no such figures, or the scrape failed).

            '
          enum:
          - no_ready_replica
          - no_metrics
          nullable: true
          type: string
        replicas:
          items:
            $ref: '#/components/schemas/DedicatedReplicaSample'
          type: array
        replicas_sampled:
          type: integer
        sampled:
          description: 'False = deliberately NOT scraped. A page poll must never wake
            a

            scale-to-zero endpoint, so an endpoint at rest reports

            `no_ready_replica` instead of a sample — a state, not an error.

            '
          type: boolean
        sampled_at:
          format: date-time
          type: string
        slowest_tok_s:
          nullable: true
          type: number
        verdict:
          description: 'Same computation, same threshold and same function as the
            operator''s

            DEGRADED REPLICA alert — a customer page and an internal log line

            must not disagree about the same replica. `unknown` = not enough

            generated tokens to judge yet.

            '
          enum:
          - ok
          - degraded
          - unknown
          type: string
      required:
      - sampled_at
      - sampled
      - reason
      - queue
      - replicas
      - expected_tok_s
      - slowest_tok_s
      - fastest_tok_s
      - replicas_sampled
      - verdict
      type: object
    DedicatedReplicaSample:
      description: One replica's live generation sample.
      properties:
        age_s:
          description: Seconds since this sample was taken.
          type: number
        tok_s:
          description: Generated tokens per second.
          type: number
        tokens:
          description: Tokens this replica has generated since it started.
          type: integer
      required:
      - tok_s
      - tokens
      - age_s
      type: object
    DedicatedScalingBounds:
      description: 'Everything needed to render correct scaling controls WITHOUT hardcoding

        anything: the per-model ceilings, whether the model publishes a

        waiting-request signal, and the three presets already resolved FOR THIS

        MODEL. Served on the endpoint AND on `ModelInfo.dedicated_scaling`, so

        the create wizard and the rescale wizard size their controls from the

        same numbers.

        '
      properties:
        buffer_replicas_max:
          type: integer
        default_profile:
          description: 'The preset in force while the endpoint stores no policy of
            its own

            (`scaling: null`). An endpoint is never left on the scheduling

            backend''s null policy — that is the documented cause of the replica

            flapping this area exists to fix (PAAS-197).

            '
          enum:
          - economical
          - balanced
          - latency_first
          type: string
        max_queue_depth_max:
          type: integer
        profiles:
          description: 'The three presets resolved for THIS model, keyed by name —
            the

            single definition of their numbers, computed from the model''s own

            admission bound. A client RENDERS them; it never recomputes them.

            `vllm_target_waiting` is absent from every preset when

            `queue_signal` is false.

            '
          properties:
            balanced:
              $ref: '#/components/schemas/DedicatedScalingPolicy'
            economical:
              $ref: '#/components/schemas/DedicatedScalingPolicy'
            latency_first:
              $ref: '#/components/schemas/DedicatedScalingPolicy'
          required:
          - economical
          - balanced
          - latency_first
          type: object
        queue_signal:
          description: 'Whether this model''s server publishes a waiting-request count.

            FALSE → `vllm_target_waiting` is refused

            (`queue_signal_unavailable`) and the control must not be offered:

            it would otherwise be accepted and do nothing, the silent failure

            SPEC §12.7 exists to forbid.

            '
          type: boolean
        scale_down_idle_seconds_max:
          type: integer
        scale_down_idle_seconds_min:
          type: integer
        scaledown_window_seconds_max:
          type: integer
        scaledown_window_seconds_min:
          type: integer
        target_inputs_max:
          description: 'min(100, the admission bound) — PER ENDPOINT on `DedicatedEndpoint`

            (the bound follows the batch the endpoint runs: `min(2 ×

            engine.max_num_seqs, the model''s)`, §12.6), per model on

            `ModelInfo`. A hardcoded ceiling would let a control offer a value

            the gateway refuses with `target_inputs_exceeds_admission`. The

            presets in `profiles` shrink with it.

            '
          type: integer
        vllm_target_waiting_max:
          type: integer
      required:
      - target_inputs_max
      - buffer_replicas_max
      - scale_down_idle_seconds_min
      - scale_down_idle_seconds_max
      - scaledown_window_seconds_min
      - scaledown_window_seconds_max
      - max_queue_depth_max
      - vllm_target_waiting_max
      - queue_signal
      - default_profile
      - profiles
      type: object
    DedicatedScalingEvent:
      description: 'One replica''s life: when it was started, when it became ready,
        when it

        was released. The customer-facing shape — the scheduling backend''s

        internal join keys are stripped (SPEC-DEDICATED §5.4).

        '
      properties:
        boot_seconds:
          description: 'ready_at − scaled_up_at. Null while the replica never became
            ready.

            These are BILLED minutes: the replica held its cards throughout.

            '
          nullable: true
          type: number
        calls_total:
          description: Requests this replica handled.
          type: integer
        capacity_class:
          description: 'The class this replica ACTUALLY ran on, read from its own
            run — the

            only place upstream exposes it. Fetched for running replicas only.

            Null on a released replica and on any run that could not be read,

            and NEVER substituted from the endpoint''s current `mode`, which is

            exactly the question this field exists to answer independently.

            '
          enum:
          - spot
          - reserved
          nullable: true
          type: string
        drain_requested_at:
          format: date-time
          nullable: true
          type: string
        error_code:
          nullable: true
          type: string
        ready_at:
          description: Null while the replica never became ready.
          format: date-time
          nullable: true
          type: string
        replica_id:
          description: Opaque identifier — a stable list key that addresses nothing.
          type: string
        scaled_down_at:
          description: Null while the replica is still running.
          format: date-time
          nullable: true
          type: string
        scaled_up_at:
          format: date-time
          type: string
        served_seconds:
          description: '(scaled_down_at or now) − ready_at. Null before the replica
            was ready.

            '
          nullable: true
          type: number
        state:
          description: 'Lifecycle state as reported upstream. Tolerant of unknown
            values —

            new states must not break a table.

            '
          type: string
        version:
          description: 'The endpoint''s OWN deployment count at the time this replica
            was

            created. It is what makes "this replica came from the configuration

            before your mode change" legible without exposing anything internal.

            '
          type: integer
      required:
      - replica_id
      - version
      - scaled_up_at
      - ready_at
      - drain_requested_at
      - scaled_down_at
      - state
      - error_code
      - calls_total
      - boot_seconds
      - served_seconds
      - capacity_class
      type: object
    DedicatedScalingHistory:
      description: 'Replica lifecycle rows, newest first. `items: []` means this endpoint
        has

        never started a replica — a history that could not be READ is a 503, not

        an empty list.

        '
      properties:
        items:
          items:
            $ref: '#/components/schemas/DedicatedScalingEvent'
          type: array
        truncated:
          description: Whether `limit` cut the list short.
          type: boolean
      required:
      - items
      - truncated
      type: object
    DedicatedScalingPolicy:
      description: 'Autoscaling settings of ONE dedicated endpoint (SPEC-DEDICATED
        §12.2).

        Applied LIVE: no redeploy, no interruption, no request dropped, and

        nothing billed by the change itself — only the replicas it ends up

        running cost anything.


        Every field is optional, and an ABSENT field means "the platform default

        governs it", which is not the same as 0. Unknown fields are REFUSED

        (400) rather than ignored, because the scheduling backend rejects them

        too and the whole replacement document would be lost with them.

        '
      properties:
        buffer_replicas:
          description: 'Hot spare replicas held ahead of demand, and only while the
            endpoint

            is actually serving traffic. They absorb a burst without anyone

            waiting for a boot, and they bill like any other replica.

            '
          maximum: 10
          minimum: 0
          type: integer
        max_queue_depth:
          description: 'Queue length past which a new request is refused with an immediate

            503 instead of queued. A safety valve, not a performance knob:

            `0` = unlimited, the default, and no preset sets it.

            '
          maximum: 10000
          minimum: 0
          type: integer
        scale_down_idle_seconds:
          description: 'How long a replica stays warm after traffic stops. The most
            direct

            cost lever (SPEC §12.2), and the ONE scale-down control this console

            renders: releasing a replica stops its meter, but the next request

            then pays a boot — and on a dedicated endpoint those boot minutes

            bill too.

            '
          maximum: 3600
          minimum: 30
          type: integer
        scaledown_window_seconds:
          description: 'Policy-level scale-down window. Accepted for parity with SPEC
            §12.2

            but NEVER set by a preset and never sent by this console: it needs a

            scheduling backend from 2026-08-11 or newer, and an older one

            refuses the whole policy document (`scaling_policy_unsupported`) —

            the document being a complete replacement, everything in it goes

            with it. Prefer `scale_down_idle_seconds`, which has no such

            requirement.

            '
          maximum: 3600
          minimum: 30
          type: integer
        target_inputs:
          description: 'Requests one replica handles at once before another replica
            is

            started — the main lever between waiting time and GPU-minutes. Low:

            a second replica starts while the first is still mostly free. High:

            a replica is filled to the brim before another is paid for.


            The MAXIMUM IS PER MODEL — `scaling_bounds.target_inputs_max`,

            i.e. min(100, the model''s admission bound) — and a higher value is

            refused with `target_inputs_exceeds_admission`: a replica cannot

            accept more requests at once than its admission bound, so the target

            could never be reached. Never hardcode the ceiling.

            '
          maximum: 100
          minimum: 1
          type: integer
        vllm_target_waiting:
          description: 'Waiting requests tolerated before capacity is added — a faster
            and

            more direct trigger than average load, working alongside

            `target_inputs`.


            Only accepted on a model whose server publishes a waiting-request

            count; anywhere else the setting would be stored and do NOTHING, so

            it is refused with `queue_signal_unavailable` (SPEC §12.7). Gate the

            control on `scaling_bounds.queue_signal` — neither the tier nor the

            model id answers the question.

            '
          maximum: 1000
          minimum: 1
          type: integer
      type: object
    DedicatedScalingState:
      allOf:
      - $ref: '#/components/schemas/DedicatedScalingPolicy'
      - description: 'What an endpoint currently STORES, plus the preset those values

          match. Keys the customer never set are ABSENT, not null.

          '
        properties:
          profile:
            description: 'The preset these values match EXACTLY, or `custom` when
              they

              match none. Resolved server-side against

              `scaling_bounds.profiles`, so a client never recomputes a

              preset''s numbers.

              '
            enum:
            - economical
            - balanced
            - latency_first
            - custom
            type: string
        required:
        - profile
        type: object
    DedicatedTelemetry:
      description: 'What the endpoint actually did over `[from, to)`, aggregated from
        its own

        metered requests. Measured, never projected.

        '
      properties:
        capacity_verdict:
          description: 'Resolved SERVER-SIDE so a console never re-derives it.

            `constrained` — requests waited for a replica to start, and that

            share is what the scaling settings can remove. `no_wake_observed`

            — enough measured requests and none waited for a replica to wake:

            the OBSERVATION, and only that. `cold_wait_ms` measures waiting

            for capacity to wake, not queueing inside a replica that is

            already warm, so this does NOT say the endpoint is oversized or

            that more replicas would change nothing — a saturated but

            continuously warm endpoint produces the same figures. `unknown` —

            too few measured requests to say anything.

            '
          enum:
          - no_wake_observed
          - constrained
          - unknown
          type: string
        cold_wait_ms:
          allOf:
          - $ref: '#/components/schemas/DedicatedPercentiles'
          - properties:
              requests_with_wake:
                description: Requests that waited for a replica to start.
                type: integer
            required:
            - requests_with_wake
            type: object
          description: 'The share of TTFT that was a wait for a replica to be there
            — the

            only half any scaling setting can move. The rest is the model

            reading the prompt, which no capacity setting touches; splitting the

            figure is what keeps the answer honest when the honest answer is

            "the capacity half was zero".

            '
        from:
          format: date-time
          type: string
        latency_ms:
          allOf:
          - $ref: '#/components/schemas/DedicatedPercentiles'
          description: End-to-end response time.
        prefix_cache:
          $ref: '#/components/schemas/DedicatedPrefixCache'
        requests:
          description: 'Requests metered in the window. `incomplete` is a generation
            cut

            short (a preemption, a disconnect); `error` is a request that failed.

            '
          properties:
            error:
              type: integer
            incomplete:
              type: integer
            ok:
              type: integer
          required:
          - ok
          - incomplete
          - error
          type: object
        to:
          format: date-time
          type: string
        tokens:
          description: 'Tokens served in the window. INFORMATIONAL on a dedicated
            endpoint —

            it is billed per GPU-card-minute and tokens are not billed at all.

            '
          properties:
            completion:
              type: integer
            prompt:
              type: integer
          required:
          - prompt
          - completion
          type: object
        ttft_ms:
          allOf:
          - $ref: '#/components/schemas/DedicatedPercentiles'
          description: 'Delay before the FIRST token, as opposed to the total answer
            time.

            '
        window:
          enum:
          - 1h
          - 24h
          - 7d
          type: string
      required:
      - window
      - from
      - to
      - requests
      - tokens
      - prefix_cache
      - ttft_ms
      - cold_wait_ms
      - latency_ms
      - capacity_verdict
      type: object
    ErrorEnvelope:
      properties:
        error:
          properties:
            code:
              description: 'Stable machine code. E.g. `invalid_api_key`, `model_not_found`,

                `model_not_supported_for_endpoint`, `context_length_exceeded`,

                `request_too_large`, `insufficient_permissions`, `invalid_image`,

                `upstream_error`, `model_unavailable`, `model_warming`,

                `upstream_timeout`, `invalid_scopes`, `session_required`,

                `insufficient_scope`.


                `model_warming` (503, with a `Retry-After` header) — the model

                is still cold-starting and the gateway could not wait for a

                ready replica within the edge''s read-timeout budget (PAAS-183).

                Retry after `Retry-After` seconds; the OpenAI SDKs do so

                natively. Distinct from `model_unavailable`, which is not a

                warm-up in progress.


                `session_required` (403) — a credential/org/billing/admin

                operation (editing an API key from the dashboard, say) reached

                with an `sk-blade-…` API key instead of a console session; these

                operations are session-only and not grantable via any key

                scope (PaaS-81). `insufficient_scope` (403) — an API key

                reached a route its granted `scopes` do not cover.


                `invalid_image` is a 400 about one image part of a multimodal prompt
                — an

                undecodable data URI, a URL that could not be fetched, an origin that
                answered

                something other than an image. `param` locates the part

                (`messages[0].content[1].image_url`).


                `model_not_found` means the id is unknown;

                `model_not_supported_for_endpoint` means it exists but is served

                elsewhere, and the message names the route that serves it.


                This envelope is the ONLY error shape on the API — including the

                statuses the framework raises by itself (unknown route, wrong

                method), which used to answer `{"detail": …}`.

                '
              nullable: true
              type: string
            message:
              description: Human-readable message.
              type: string
            param:
              description: Offending parameter (OpenAI compat), if applicable.
              nullable: true
              type: string
            type:
              description: 'OpenAI error family. E.g. `invalid_request_error`,

                `api_error`, `authentication_error`.

                '
              type: string
          required:
          - message
          - type
          type: object
      required:
      - error
      type: object
    Model:
      properties:
        architecture:
          description: What the model takes and returns.
          properties:
            input_modalities:
              items:
                enum:
                - text
                - image
                - audio
                type: string
              type: array
            output_modalities:
              items:
                enum:
                - text
                - image
                - audio
                - video
                - embedding
                - score
                - json
                type: string
              type: array
          type: object
        context_length:
          description: 'The model''s context window (`max_model_len`). Absent on the
            models where

            it means nothing (image, video, and the utility heads, which carry 0).

            Published because a client that cannot read it invents one: a 64k model

            assumed to be 128k reports a context fill ratio that is simply wrong.

            '
          example: 65536
          type: integer
        context_window:
          description: 'The same number as `context_length`, under the name the LiteLLM/LangChain
            family

            of clients reads. A mirror, not a second quantity.

            '
          example: 65536
          type: integer
        created:
          description: Epoch (OpenAI compat; time added to the catalog).
          type: integer
        id:
          example: qwen2.5-3b
          type: string
        max_model_len:
          description: 'The same number again, under vLLM''s own spelling — anything
            written against a bare

            vLLM `/v1/models` finds it here too.

            '
          example: 65536
          type: integer
        max_output_tokens:
          description: 'Ceiling on what the model may emit. On a chat model that is
            the `max_tokens`

            guard-rail — the global one, or the model''s own when it declares a higher
            one —

            and the effective limit on a request is `min(max_output_tokens, context
            -

            prompt_tokens)`. On a document head (`kind: parse`) it is the ceiling
            the serving

            recipe applies to `max_new_tokens`; those heads have no context window
            to report

            (their input is an image), and four of them have no output ceiling either,
            so the

            field is simply absent there. Absent on a dedicated-endpoint alias. It
            used to be

            discoverable only by triggering its 400.

            '
          example: 16384
          type: integer
        object:
          enum:
          - model
          type: string
        owned_by:
          example: shadow-inference
          type: string
        pricing_mode:
          description: 'What a POOL request on this model is billed on (PAAS-364).
            `spot` scales

            `price_in`/`price_out` by the model''s card-class coefficient — the value
            every

            model on our own fleet takes today. `fixed` bills those prices flat: the

            catalogue declares it per model, and the usage event is audited as

            (''fixed'', 1.0). `external` is derived, never declared, and means the
            model is

            served by a third party (PAAS-206) whose price is a supplier rate card.


            These are the same strings `usage_events.pricing_mode` persists, deliberately:

            the mode a request was billed under is greppable in the file that declared
            it.

            The customer-facing name of `spot` is "dynamic pricing" (PAAS-280) and
            is a

            DISPLAY label applied in the console — `spot` here also distinguishes
            it from

            the dedicated preemptible capacity class of the same name.

            '
          enum:
          - spot
          - fixed
          - external
          type: string
        publisher:
          description: 'Who published the served weights ("Mistral AI", "Alibaba Qwen",
            "NVIDIA"…), from

            the catalogue (PAAS-82). Display-only. ABSENT when the catalogue does
            not

            establish one — gate on the key like `attribution`.

            '
          example: Mistral AI
          type: string
        reasoning:
          description: Chat models only.
          properties:
            controllable:
              description: 'True when `chat_template_kwargs: {"enable_thinking": bool}`
                turns the

                deliberation on and off — the reasoning control that acts on this
                fleet.

                '
              type: boolean
            separate_channel:
              description: 'True when the deliberation comes back in `reasoning_content`
                (and, for

                compatibility, `reasoning`) instead of inline in `content`. When false,

                the deliberation IS part of the message and no client can split it.

                '
              type: boolean
          type: object
        release_date:
          description: 'Upstream release date of the served checkpoint (ISO 8601,
            `YYYY-MM-DD`), from

            the catalogue (PAAS-82) — only ever a SOURCED date (the publisher''s own

            announcement, or the PAAS-143 recency audit), never an estimate. ABSENT
            when

            no sourced date exists: render nothing rather than an approximation.

            '
          example: '2025-07-15'
          format: date
          type: string
        routing:
          allOf:
          - $ref: '#/components/schemas/ResidencyRouting'
          description: 'Present ONLY on a model served outside our own infrastructure.
            A model

            carrying this key is listed only to organizations that have enabled external

            providers; calling one without that enablement returns 403

            `external_provider_not_enabled`.

            '
        specialties:
          description: 'Editorial specialty tags the UI cannot derive from `tier`/`kind`
            alone (`code`…).

            Open value set, lowercase slugs. ABSENT when the catalogue states none.

            '
          example:
          - code
          items:
            type: string
          type: array
        spot_enabled:
          description: 'True for a model whose pool price MOVES — served on our own
            fleet and

            `pricing_mode: spot` — so `spot_multiplier` and `gpu_class` accompany
            it.


            FALSE in two cases, which publish the same ABSENT keys rather than null
            or

            1.0, because a client reading `spot_multiplier` must be able to treat
            its

            presence as "this price moves": a model served by an external inference

            provider (PAAS-206), which runs on cards we do not own so there is no
            class

            signal at all and which publishes `routing` instead; and a model declaring

            `pricing_mode: fixed` (PAAS-364), which runs on our cards but on a flat
            rate

            card. In both, `price_in`/`price_out` are the price, unscaled.


            Prefer `pricing_mode`: it says which of the three a row is, where this
            boolean

            only says the price does not move.

            '
          type: boolean
        supported_parameters:
          description: 'Request parameters this model HONOURS, on chat models only.
            Deliberately

            narrower than what the API accepts: `reasoning_effort` reaches the engine,

            passes its enum validation and then changes nothing on a model whose chat

            template does not read it, so it is not listed. `structured_outputs` appears

            only where grammar-backed `json_schema` decoding was measured to hold,
            and

            `tools`/`tool_choice` only where the serving stack can extract tool calls.

            '
          example:
          - max_tokens
          - temperature
          - tools
          - tool_choice
          - structured_outputs
          items:
            type: string
          type: array
      required:
      - id
      - object
      - owned_by
      type: object
    ModelList:
      properties:
        data:
          items:
            $ref: '#/components/schemas/Model'
          type: array
        object:
          enum:
          - list
          type: string
      required:
      - object
      - data
      type: object
    ResidencyRouting:
      description: 'Residency disclosure for a request served outside Shadow''s own
        EU-sovereign

        infrastructure (PAAS-206). Published in three places, always identically:
        the

        `X-Shadow-Residency: external` response header, a `routing` object on a

        non-streaming body, and one `choices: []` extension chunk emitted just before

        `data: [DONE]` on a stream. It also appears on the catalogue entry itself

        (`GET /v1/models`), so the disclosure is readable BEFORE a request is sent,
        not

        only after.


        Absent on every request served by our own fleet: sovereign is the product''s

        standing claim, not a per-response one.

        '
      properties:
        notice:
          description: Disclosure sentence. Render verbatim; do not summarise it.
          example: This request was served by an external inference provider, outside
            Shadow's own EU-sovereign infrastructure.
          type: string
        provider_class:
          enum:
          - external
          type: string
        region:
          description: 'Where the provider operates, as the provider states it. Informational,
            and

            explicitly NOT a claim of sovereignty.

            '
          example: EU — France (OVHcloud AI Endpoints)
          type: string
        residency:
          enum:
          - external
          type: string
      required:
      - provider_class
      - residency
      - notice
      type: object
    Response:
      additionalProperties: true
      properties:
        created_at:
          type: integer
        error:
          description: 'Set when `status: failed` — no answer was produced.'
          nullable: true
          properties:
            code:
              type: string
            message:
              type: string
          type: object
        id:
          example: resp_9f2c1e7a4b8d4f0e9c3a5b7d1e2f4a60
          type: string
        incomplete_details:
          description: 'Set when `status: incomplete`. `reason` is `max_output_tokens`
            (the answer was

            truncated) or `upstream_error` (the model stream ended mid-answer; the
            partial

            text is billed).

            '
          nullable: true
          properties:
            reason:
              type: string
          type: object
        model:
          type: string
        object:
          enum:
          - response
          type: string
        output:
          description: 'The items produced, in order: executed server-tool calls,
            then any client

            `function_call`, then the assistant `message`.

            '
          items:
            $ref: '#/components/schemas/ResponseOutputItem'
          type: array
        output_text:
          description: Convenience aggregate of every `output_text` part.
          type: string
        previous_response_id:
          nullable: true
          type: string
        status:
          enum:
          - completed
          - incomplete
          - failed
          - in_progress
          type: string
        store:
          description: Always false (see ResponsesRequest.store).
          type: boolean
        usage:
          nullable: true
          properties:
            input_tokens:
              type: integer
            output_tokens:
              type: integer
            total_tokens:
              type: integer
          type: object
      required:
      - id
      - object
      - created_at
      - status
      - model
      - output
      type: object
    ResponseOutputItem:
      additionalProperties: true
      description: 'One `output` item. `type: message` carries the answer; `type:
        <tool>_call`

        (`web_search_call`, `web_fetch_call`, `code_interpreter_call`) reports one

        EXECUTED server-tool call; `type: function_call` hands a client function call

        back to the caller.

        '
      properties:
        action:
          description: 'Server-tool items only: what the tool did — `{"type": "search",
            "query": …}`

            for web_search and `{"type": "fetch", "url": …}` for web_fetch (an extension:

            OpenAI publishes no fetch tool). A tool with no verb of its own falls
            back to

            `{"type": "call", "arguments": {…}}`, which is what `code_interpreter_call`

            carries today: `{"type": "call", "arguments": {"code": "…"}}`, the program

            the model ran. Read `type` before keying on a field — the fallback is
            a

            documented shape, not a placeholder to be surprised by.

            '
          type: object
        arguments:
          description: '`function_call` items only: the raw JSON string the model
            emitted.'
          type: string
        call_id:
          description: 'The id the MODEL used for the call. On a `function_call` item
            it is what the

            caller must echo in its `function_call_output`.

            '
          type: string
        content:
          description: '`message` items only: `output_text` parts with their annotations.'
          items:
            properties:
              annotations:
                items:
                  $ref: '#/components/schemas/UrlCitation'
                type: array
              text:
                type: string
              type:
                enum:
                - output_text
                type: string
            type: object
          type: array
        id:
          type: string
        name:
          description: '`function_call` items only.'
          type: string
        role:
          description: '`message` items only.'
          type: string
        status:
          description: '`failed` on a tool call means the tool errored (vendor down,
            blocked URL,

            unreadable page) — the request itself still succeeds and the model answers

            around it. There is no third state for "ran, then hit its ceiling": a

            `code_interpreter_call` that timed out, or whose program was killed, is

            `completed` (it executed and its seconds are billed) and says so in its

            result; the `incomplete` metering status lives on the usage event, not
            on

            the item.

            '
          enum:
          - in_progress
          - completed
          - failed
          type: string
        type:
          example: web_search_call
          type: string
      required:
      - id
      - type
      type: object
    ResponsesRequest:
      additionalProperties: false
      properties:
        input:
          description: 'A plain string (one user turn), or the item array. Item types
            translated:

            `message` (roles user/assistant/system/developer — `developer` maps to
            a

            system turn), `function_call` and `function_call_output`, which is how
            a

            client answers its own function tool without a stored session. Content
            parts:

            `input_text`, `output_text`, `input_image`.

            '
          oneOf:
          - type: string
          - items:
              type: object
            minItems: 1
            type: array
        instructions:
          description: 'Prepended as a system turn. When server tools are declared,
            the gateway''s tool

            guidance is merged INTO it rather than added as a second system message
            (several

            chat templates reject a system turn that is not first).

            '
          type: string
        max_output_tokens:
          description: Same cap as chat `max_tokens` (≤ the model's `max_output_tokens`,
            prompt + output ≤ max_model_len).
          minimum: 1
          type: integer
        metadata:
          description: Echoed back unchanged; not stored.
          type: object
        model:
          description: 'Public model_id from the catalog, `kind: chat`. `ded/<slug>`
            aliases are

            refused here (400) — use /v1/chat/completions. The legacy `:spot` suffix
            is

            a deprecated no-op (every pool price is spot).

            '
          example: llama-3.1-8b
          type: string
        parallel_tool_calls:
          type: boolean
        previous_response_id:
          description: 'Not supported — a non-null value returns 400 `unsupported_parameter`.
            Replay

            the conversation in `input` instead.

            '
          nullable: true
          type: string
        reasoning:
          additionalProperties: false
          description: '`{"effort": "none"|"minimal"|"low"|"medium"|"high"}`. Translated
            onto the

            model''s OWN control — `chat_template_kwargs.enable_thinking`, the one
            that

            acts on this fleet — so `none` turns the deliberation off and anything
            else

            turns it on. Only the extremes are actionable, because the control is
            an

            on/off switch and grading it would advertise a granularity nothing implements.


            A model with no such control (`reasoning.controllable: false` on

            `GET /v1/models`) returns 400 `unsupported_parameter` rather than accepting

            the parameter and ignoring it. `summary` is not supported: this surface

            streams the model''s raw deliberation and nothing summarises it.

            '
          properties:
            effort:
              enum:
              - none
              - minimal
              - low
              - medium
              - high
              type: string
          type: object
        seed:
          type: integer
        store:
          default: false
          description: 'Must be false (or absent). `true` returns 400 `unsupported_parameter`:
            nothing

            is persisted, so a stored response could never be retrieved or continued.

            '
          type: boolean
        stream:
          default: false
          type: boolean
        temperature:
          type: number
        text:
          description: 'Only `{"format": {"type": "text"}}` is accepted; a `json_schema`
            format

            returns 400 (structured decoding is not translated on this surface).

            '
          type: object
        tool_choice:
          description: 'Relayed to the model. Cannot be combined with a built-in tool
            (400

            `unsupported_parameter`): the gateway owns the loop and sets `tool_choice`

            per round itself, so a caller''s value would be overwritten.

            '
          oneOf:
          - type: string
          - type: object
        tools:
          description: 'Function declarations in the Responses (flattened) shape —
            `{"type":

            "function", "name", "parameters"}` — and/or built-in server tools.

            '
          items:
            oneOf:
            - $ref: '#/components/schemas/ServerToolDeclaration'
            - type: object
          type: array
        top_p:
          type: number
        user:
          type: string
      required:
      - model
      - input
      type: object
    ServerToolDeclaration:
      additionalProperties: true
      description: 'A gateway-executed ("built-in") tool, declared in `tools` alongside
        ordinary

        `function` tools. vLLM only understands `{"type": "function"}`, so the gateway

        strips these entries from the upstream body and injects the tool''s own function

        schema per round. An unknown `type` is a 400 with `param: "tools"` — never

        relayed and never ignored. Each declaration requires the matching account

        entitlement and a model with native tool calling (`tool_call_parser`).


        `{"type": "code_interpreter"}` runs model-written Python in a Forge sandbox
        and

        takes `max_uses` and `timeout_s` — in particular no `container` handle: every

        call gets a sandbox created for it and terminated after it, so there is no

        session to name or reuse, nothing survives between two runs, and the model''s

        `code` argument is the only input that crosses the boundary (never the API
        key,

        never the transcript). CPU-only, gVisor (uid 10001), 20 s per run and 45 s
        of

        sandbox time per request, billed per SECOND at `defaults.price_code_second`.

        Its public-internet egress is OPEN and cannot currently be switched off, so
        a

        program can print text it downloaded: results are treated as third-party content

        for the rest of the request. Read docs/SERVER-TOOLS.md before assuming this

        sandbox is isolated from the network.

        '
      properties:
        max_uses:
          description: 'Extension (no OpenAI equivalent): tightens this tool''s per-request
            call

            ceiling. It can only LOWER the documented budget, never raise it. Defaults:

            6 for `web_search`, 3 for `web_fetch`, 4 for `code_interpreter`.

            '
          minimum: 1
          type: integer
        search_context_size:
          description: '`web_search` only: results per search (3/5/8).'
          enum:
          - low
          - medium
          - high
          type: string
        timeout_s:
          description: '`code_interpreter` only, and an extension: the wall clock
            granted to ONE run,

            clamped to [5, 20] seconds. Like `max_uses` it can only tighten — a caller

            cannot buy more sandbox — and it is bounded again by the 45 seconds of

            sandbox time a whole request may consume, whichever runs out first.

            '
          maximum: 20
          minimum: 5
          type: number
        type:
          enum:
          - web_search
          - web_fetch
          - code_interpreter
          type: string
      required:
      - type
      type: object
    StreamOptions:
      properties:
        include_usage:
          description: 'If true, a final SSE chunk carries `usage` (before `[DONE]`).
            The

            gateway ALWAYS knows the exact usage (non-streamed upstream call);

            this option only controls whether it is re-emitted in the stream.

            '
          type: boolean
      type: object
    UrlCitation:
      description: 'A web citation on an `output_text` part. `index` is the `[n]`
        marker number the

        model writes inline and is authoritative (identical to the chat surface''s

        `url_citation.index`); `start_index`/`end_index` are the best-effort character

        span of that marker in `text`. A source the model never referenced carries
        an

        EMPTY span at the END of the text — deliberately not `0..0`, which would read
        as

        "cited in the first character". See docs/SERVER-TOOLS.md.

        '
      properties:
        end_index:
          type: integer
        index:
          type: integer
        start_index:
          type: integer
        title:
          type: string
        type:
          enum:
          - url_citation
          type: string
        url:
          type: string
      type: object
    Usage:
      description: 'vLLM token count. Source of truth for metering (SPEC §6.1).

        Always present in non-stream mode; re-emitted in stream mode if

        include_usage. Invariant: total_tokens = prompt_tokens +

        completion_tokens. For n>1, vLLM aggregates the n choices into a

        SINGLE `usage` (completion_tokens = sum). Since P0, the inference

        endpoints may respond 429 (RPM/TPM/daily quota — see

        components/responses/RateLimited); by default no limit is set (M1

        behavior unchanged until admin/ops configures one). This `usage`

        remains the source of the TPM debit (real throughput debited

        post-response, prompt+max_tokens estimate in the pre-check).

        '
      properties:
        completion_tokens:
          type: integer
        completion_tokens_details:
          description: '`reasoning_tokens` is how many of `completion_tokens` went
            into the deliberation

            channel (`reasoning_content`) rather than into the answer. Present on
            the models

            that have such a channel (`reasoning.separate_channel` on GET /v1/models),
            ABSENT

            elsewhere — never a fabricated 0, which reads as "this model did not reason"
            on a

            response that was mostly deliberation.


            Counted by the gateway from the engine''s token stream, and published
            only when

            that stream is one token per delta (it is, on this fleet — checked per
            response).

            It counts the tokens that reached the channel: the `</think>` the reasoning
            parser

            consumes to close it is generated and billed, and is not in this figure
            — one

            token, measured. `completion_tokens` itself is always the engine''s own
            number.

            '
          properties:
            reasoning_tokens:
              type: integer
          type: object
        prompt_tokens:
          type: integer
        prompt_tokens_details:
          description: 'Present when the serving engine reports it. `cached_tokens`
            is the part of the

            prompt the GPU''s prefix cache served instead of recomputing — the only
            visibility

            a client has into a cache hit, and it is billed like any other prompt
            token.

            '
          properties:
            cached_tokens:
              type: integer
          type: object
        total_tokens:
          type: integer
      required:
      - prompt_tokens
      - completion_tokens
      - total_tokens
      type: object
    VideoGenerationJob:
      properties:
        created:
          type: integer
        error:
          properties:
            message:
              type: string
          type: object
        id:
          description: vgen_…
          type: string
        model:
          type: string
        object:
          enum:
          - video.generation
          type: string
        status:
          enum:
          - queued
          - running
          - completed
          - failed
          - cancelled
          type: string
        video:
          description: Present when status=completed.
          properties:
            gen_seconds:
              type: number
            seconds:
              type: number
            url:
              description: Presigned S3 URL (expires — re-GET the job to refresh)
              type: string
          type: object
      type: object
  securitySchemes:
    ApiKeyAuth:
      bearerFormat: sk-blade-…
      description: 'Shadow Inference API key. `Authorization: Bearer sk-blade-<base62>`.

        Only the SHA-256 is stored server-side. The same key authenticates

        the /v1 API and the /api dashboard.

        '
      scheme: bearer
      type: http
info:
  description: 'OpenAI-compatible LLM-as-a-service API, sitting in front of the Forge

    serverless GPU platform. Auth via `sk-blade-…` key. Token metering per

    user × model.

    '
  title: Shadow Inference Gateway API
  version: 1.0.0-m1
openapi: 3.0.3
paths:
  /api/dedicated:
    get:
      operationId: listDedicated
      responses:
        '200':
          content:
            application/json:
              schema:
                items:
                  $ref: '#/components/schemas/DedicatedEndpoint'
                type: array
          description: Requester's non-deleted endpoints, all statuses.
        '401':
          $ref: '#/components/responses/Unauthorized'
      summary: Requester's dedicated endpoints (live state + current month's cost).
      tags:
      - bff
      - dedicated
    post:
      description: 'ASYNCHRONOUS provisioning. The response comes back with

        `status: provisioning` while the private deployment is created, then

        `warming` while the replicas pull the model weights, then `active`

        once at least one replica can serve (or `error`). Eligible models:

        active catalog EXCEPT `gpt-oss-20b`, `orpheus-tts-3b`,

        `ltx-video-2b` and `gpt-oss-120b`. Quota: 2 active endpoints per ORGANISATION
        (403,

        code `dedicated_quota_exceeded`); replica bounds are additionally

        capped per organisation (400, code `replica_cap_exceeded` — see

        `max_replicas_cap` on `GET /api/dedicated/rates`). The full alias

        `ded/<slug>` is unique GLOBALLY (collision → 409). The model''s GPU card

        must be priced for every class the requested `mode` bills at (400, code

        `mode_not_priced` otherwise — e.g. `spot` or `hybrid` on an H100, which

        is priced reserved only; see `hourly_by_gpu_type` on `GET /api/dedicated/rates`).


        BILLING STARTS HERE. GPU card-minutes accrue from the moment a replica

        holds a GPU — the weight-loading/boot window included, while the

        endpoint still presents as `warming` and cannot serve — and stop only

        when the replicas are released (`pause`, a rescale to 0 replicas, or

        `DELETE`). Tokens are NOT billed on a dedicated endpoint.

        '
      operationId: createDedicated
      requestBody:
        content:
          application/json:
            schema:
              $ref: '#/components/schemas/DedicatedCreateRequest'
        required: true
      responses:
        '201':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/DedicatedEndpoint'
          description: Endpoint created, provisioning in progress.
        '400':
          $ref: '#/components/responses/BadRequest'
        '401':
          $ref: '#/components/responses/Unauthorized'
        '403':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorEnvelope'
          description: Dedicated endpoint quota reached (code `dedicated_quota_exceeded`).
        '409':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorEnvelope'
          description: Alias already taken (code `alias_taken`).
      summary: Deploy a dedicated (private) instance of a catalog model.
      tags:
      - bff
      - dedicated
  /api/dedicated/rates:
    get:
      description: 'Source for the wizard''s cost estimator. Cost for one minute =

        Σ (replicas_holding_a_gpu × cards_per_replica × rate(class)) — a

        replica still booting holds its GPU and counts. A scale-to-zero

        endpoint at 0 replicas costs 0.

        '
      operationId: dedicatedRates
      responses:
        '200':
          content:
            application/json:
              schema:
                properties:
                  applies_to_gpu_type:
                    description: 'The GPU card the flat `spot` and `reserved` belong
                      to. They are

                      that card''s price only; a model on another card is priced by
                      its

                      own row of `hourly_by_gpu_type` (PAAS-278).

                      '
                    example: RTX-A4500
                    type: string
                  currency:
                    description: 'Currency the rate card, `month_cost` and `estimated_cost`

                      are denominated in.


                      CHANGED FROM `usd` TO `eur` (D10/D11, 2026-09-25). The

                      RATES THEMSELVES DID NOT MOVE — only the unit they are

                      declared in. infer''s meter, its spend cap and billing''s

                      rate card all price these same `models.yaml` numbers in

                      EUR; this route was the last surface publishing them as

                      dollars, and one number under two labels is what stalled

                      the usage export for 2.5 days (PAAS-607).


                      On this card it is a correction rather than a repricing:

                      the A4500''s two scalars are its negotiated list prices,

                      EUR0.40/h and EUR0.55/h per card, divided by sixty. They

                      were computed in euros and published as dollars.


                      This enum is the ONE place the currency is pinned —

                      `scripts/gen_docs.py::dedicated_currency` reads it to

                      generate the public docs bundle and refuses a second

                      member, so nothing may restate a currency the contract

                      does not declare.

                      '
                    enum:
                    - eur
                    type: string
                  hourly_by_gpu_type:
                    additionalProperties:
                      properties:
                        reserved:
                          type: number
                        spot:
                          type: number
                      type: object
                    description: 'THE rate card (PAAS-278): GPU card (`gpu_type`,
                      the catalogue

                      literal) → capacity class → price per card per HOUR, exactly
                      as

                      set and billed. Usage is metered per card-minute; a minute costs

                      the hourly price / 60, and 60 card-minutes cost it exactly.
                      A

                      class ABSENT under a card is not offered on it: create and mode

                      change refuse that mode with `mode_not_priced` (a `hybrid`

                      endpoint needs both classes — its burst runs on spot).

                      '
                    example:
                      H100:
                        reserved: 3.0
                      RTX-2000-ADA-GENERATION:
                        reserved: 0.35
                        spot: 0.28
                      RTX-A4500:
                        reserved: 0.55
                        spot: 0.4
                    type: object
                  max_replicas_cap:
                    description: 'The requester organisation''s EFFECTIVE replica
                      cap

                      (default 16 for a billable organisation, 32 — the absolute

                      ceiling — for an internal one; raisable on request up to

                      32). Bounds above it are refused with `replica_cap_exceeded`.

                      '
                    type: integer
                  reserved:
                    description: 'Per card, per MINUTE, in reserved (never preempted)
                      — for the

                      `applies_to_gpu_type` card ONLY (hourly / 60). See `spot`.

                      '
                    type: number
                  spot:
                    description: 'Per card, per MINUTE, in spot (preemptible) — for
                      the

                      `applies_to_gpu_type` card ONLY: its hourly rate / 60, NOT

                      rounded (0.4 / 60 = 0.00666…). Kept for existing clients; read

                      `hourly_by_gpu_type` to price a model on any card.

                      '
                    type: number
                  unpriced_gpu_types:
                    description: 'Fleet cards present in the catalogue with NO rate
                      in any class —

                      they cannot be provisioned. A card priced for one class only
                      is

                      not listed (see `hourly_by_gpu_type`). CPU-only entries are
                      absent by

                      construction: they hold no card and accrue no card-minutes.

                      '
                    example: []
                    items:
                      type: string
                    type: array
                required:
                - spot
                - reserved
                - currency
                - max_replicas_cap
                type: object
          description: Price per GPU card and class (models.yaml defaults.dedicated_hourly_rates).
        '401':
          $ref: '#/components/responses/Unauthorized'
      summary: Dedicated rate card (€/GPU card/minute, per capacity class).
      tags:
      - bff
      - dedicated
  /api/dedicated/{endpoint_id}:
    delete:
      description: 'Releases every replica: the card-minute meter stops here. The
        row is

        kept (`status: deleted`) for billing history — minutes already sampled

        remain billable. The alias stops resolving (`404 model_not_found`) and

        is NOT reusable.

        '
      operationId: deleteDedicated
      responses:
        '200':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/DedicatedEndpoint'
          description: Endpoint deleted.
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          $ref: '#/components/responses/NotFound'
      summary: Full teardown (scale 0, deployment removed, status deleted).
      tags:
      - bff
      - dedicated
    get:
      operationId: getDedicated
      responses:
        '200':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/DedicatedEndpoint'
          description: The endpoint, if it belongs to the requester.
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          $ref: '#/components/responses/NotFound'
      summary: Detail of a dedicated endpoint (live replicas, minutes/cost for the
        month).
      tags:
      - bff
      - dedicated
    parameters:
    - in: path
      name: endpoint_id
      required: true
      schema:
        type: string
    patch:
      description: 'Applies the new bounds hot, without a redeploy (a `paused` endpoint
        is

        NOT rescaled — `resume` re-applies the bounds). Omitted fields are left

        unchanged. Scaling UP adds replicas that bill from the moment they hold

        a GPU, before they are ready to serve; scaling DOWN stops the meter for

        the replicas that are released. Changing `mode` requires a redeploy and

        is not supported here.


        The autoscaling policy (`scaling`) rides this same call and is applied

        the same way — live, within seconds, no redeploy, no interruption, and

        nothing billed by the change itself. A `paused` or still-provisioning

        endpoint STORES it and gets it asserted on its next resume. Unlike the

        replica bounds, `scaling` REPLACES the stored policy wholesale; `null`

        restores the model''s default preset.


        Stable error codes specific to the policy:

        `target_inputs_exceeds_admission` (above the model''s admission bound —

        `scaling_bounds.target_inputs_max` names it),

        `queue_signal_unavailable` (the model''s server publishes no

        waiting-request count), `scaling_policy_unsupported` (a field the

        endpoint''s scheduling backend does not know yet) and

        `scaling_policy_rejected`. The last two mean NOTHING was changed: the

        policy is replaced wholesale, so a refusal loses the whole document and

        the stored one is left exactly as it was.

        '
      operationId: patchDedicated
      requestBody:
        content:
          application/json:
            schema:
              $ref: '#/components/schemas/DedicatedPatchRequest'
        required: true
      responses:
        '200':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/DedicatedEndpoint'
          description: Endpoint updated.
        '400':
          $ref: '#/components/responses/BadRequest'
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          $ref: '#/components/responses/NotFound'
      summary: LIVE rescale (min/max/reserved_floor) without a redeploy.
      tags:
      - bff
      - dedicated
  /api/dedicated/{endpoint_id}/mode:
    put:
      description: 'Applies the transition quoted by `POST …/mode/preview`. A distinct
        verb

        from the PATCH rescale on purpose (SPEC-DEDICATED §12.10): PATCH is the

        hot, free, uninterrupted change, this is the billed reboot, and keeping

        them apart is what makes it impossible — for a client as much as for our

        own code — to trigger a billed restart while believing a slider was

        moved. `confirm_redeploy: true` must be sent explicitly for the same

        reason.


        The endpoint is redeployed IN PLACE: same private deployment, same

        alias, same `endpoint_id`, same billing history. The replicas are

        replaced, hold their GPUs while the weights reload, and those minutes

        bill.


        The answer comes back immediately with `status: updating` — a DERIVED

        status, exactly like `warming`: the stored lifecycle is still `active`

        because the GPUs are held and the endpoint keeps billing throughout.

        Calls made during the reload get a retryable error, not a dropped

        connection.


        If the new configuration fails to boot, the endpoint is put back to the

        one that was serving (`previous_mode` / `previous_reserved_floor`) and

        restarted ONCE; that second start-up is billed too. Only if the rollback

        also fails does the endpoint end in `error`.


        Stable error codes: `mode_unchanged` (400 — a no-op would redeploy and

        bill a weight reload for nothing), `reserved_floor_required` (400),

        `replica_cap_exceeded` (400), `model_not_eligible` (400),

        `mode_not_priced` (400 — the model''s card has no rate for a class the

        target mode bills at, e.g. spot or hybrid on an H100),

        `confirmation_required` (400), `endpoint_paused` / `endpoint_busy` /

        `endpoint_not_ready` (409) and `mode_change_too_soon` (429).

        '
      operationId: changeDedicatedMode
      parameters:
      - in: path
        name: endpoint_id
        required: true
        schema:
          type: string
      requestBody:
        content:
          application/json:
            schema:
              $ref: '#/components/schemas/DedicatedModeChange'
        required: true
      responses:
        '200':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/DedicatedEndpoint'
          description: 'Redeploy started. The body is the endpoint with the new mode
            already

            stored and a PRESENTED status of `updating`.

            '
        '400':
          $ref: '#/components/responses/BadRequest'
        '401':
          $ref: '#/components/responses/Unauthorized'
        '402':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorEnvelope'
          description: 'Monthly spending cap reached. A redeploy restarts the GPU-minute

            meter, exactly like `resume`, so it is refused for the same reason —

            and the over-budget sweep would re-pause the endpoint within the

            minute anyway.

            '
        '403':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorEnvelope'
          description: 'Org role `admin` required (code `insufficient_permissions`).

            '
        '404':
          $ref: '#/components/responses/NotFound'
        '409':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorEnvelope'
          description: 'The endpoint cannot be redeployed right now, and NOTHING was

            changed: `endpoint_busy` (still provisioning, or a change is already

            in flight — a second concurrent deploy on the same app would race

            and the loser would write `error` over a change about to succeed),

            `endpoint_paused` (it holds no GPU; a redeploy would bring the

            replicas back and silently restart the meter on an endpoint that was

            deliberately stopped — resume it first) or `endpoint_not_ready` (its

            last deployment failed, so the restart would fail identically and

            the start-up minutes would still be billed).

            '
        '429':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorEnvelope'
          description: 'Cooldown (code `mode_change_too_soon`). Each change restarts
            the

            endpoint and bills a weight reload; without a ceiling a customer can

            reboot their own GPUs in a loop and self-bill a month of compute in

            a night (§12.9 anti-abuse). `cooldown_seconds_remaining` on the

            preview says when the next one is accepted.

            '
      summary: Change the capacity mode — a BILLED redeploy, not a live rescale.
      tags:
      - bff
      - dedicated
  /api/dedicated/{endpoint_id}/mode/preview:
    post:
      description: 'A QUOTE: nothing is deployed, nothing is stored, nothing is billed
        by

        this call. It exists because the capacity mode is the one dedicated

        setting that is NOT a live change. A replica''s capacity class is chosen

        when the replica is CREATED and never rewritten afterwards, so moving an

        endpoint between classes means replacing every replica — including

        spot ↔ hybrid, which reads live and is not (SPEC-DEDICATED §12.1).


        THE REDEPLOY IS BILLED: the replicas hold their GPUs for the whole

        weight reload, and held cards bill (§4). `reload.estimated_cost` is

        that amount, at the NEW class''s rate, and the console is expected to

        print it on the confirmation before sending the `PUT`.


        `allowed: false` + `blocked_reason` reports — WITHOUT a failed write —

        the stable code the `PUT` would answer with (`endpoint_paused`,

        `mode_change_too_soon`, `endpoint_busy`, …), so a control can be

        disabled with a reason instead of failing on submit.


        `reload.estimate_source` MUST be surfaced: `observed` was timed on this

        endpoint''s own last start-up, `default` is a platform average and can be

        off by minutes on a large model.


        THE ALIAS SURVIVES. The private deployment is redeployed in place, so

        `alias`, `endpoint_id` and the billing history are unchanged and nothing

        the caller hardcoded has to be touched — which is the whole reason this

        route exists (the alternative is delete + recreate, which loses the

        alias).

        '
      operationId: previewDedicatedMode
      parameters:
      - in: path
        name: endpoint_id
        required: true
        schema:
          type: string
      requestBody:
        content:
          application/json:
            schema:
              $ref: '#/components/schemas/DedicatedModeTarget'
        required: true
      responses:
        '200':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/DedicatedModePreview'
          description: 'The quote. Nothing was applied — including when `allowed`
            is false.

            '
        '400':
          $ref: '#/components/responses/BadRequest'
        '401':
          $ref: '#/components/responses/Unauthorized'
        '403':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorEnvelope'
          description: 'Org role `admin` required — a mode change restarts a billing

            endpoint (code `insufficient_permissions`).

            '
        '404':
          $ref: '#/components/responses/NotFound'
      summary: Quote the cost and the downtime of a capacity-mode change. Applies
        NOTHING.
      tags:
      - bff
      - dedicated
  /api/dedicated/{endpoint_id}/pause:
    post:
      operationId: pauseDedicated
      parameters:
      - in: path
        name: endpoint_id
        required: true
        schema:
          type: string
      responses:
        '200':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/DedicatedEndpoint'
          description: Endpoint paused.
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          $ref: '#/components/responses/NotFound'
      summary: Scale to zero — releases the GPUs and stops the meter, keeps the config.
        Idempotent.
      tags:
      - bff
      - dedicated
  /api/dedicated/{endpoint_id}/replicas:
    get:
      description: 'A replica can be ready, answering, and generating at a fraction
        of the

        speed of its peers. Nothing else notices, because a slow replica never

        makes anything queue: the autoscaler sees no backlog and the endpoint

        never replaces it. `verdict` comes from the SAME function and the same

        threshold as the operator''s DEGRADED REPLICA alert, so this page and

        that log line cannot disagree about the same replica.


        THIS ROUTE NEVER WAKES THE ENDPOINT. With no ready replica it answers

        200 with `sampled: false` and `reason: no_ready_replica` — a normal

        state at rest, not an error. Sampling goes through the serve ingress,

        which would wake a scale-to-zero endpoint and start its meter on the

        customer whose page happens to be open.

        '
      operationId: dedicatedReplicas
      parameters:
      - in: path
        name: endpoint_id
        required: true
        schema:
          type: string
      responses:
        '200':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/DedicatedReplicaHealth'
          description: 'A live sample, or a stated reason why there is none. Always
            200 —

            "no replica to measure" is a state, not a failure.

            '
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          $ref: '#/components/responses/NotFound'
      summary: Per-replica generation speed, sampled live — finds the replica that
        is up but slow.
      tags:
      - bff
      - dedicated
  /api/dedicated/{endpoint_id}/resume:
    post:
      description: 'The endpoint restarts from zero replicas: it presents `warming`
        and

        bills again from the moment the first replica holds a GPU, before it

        can serve.

        '
      operationId: resumeDedicated
      parameters:
      - in: path
        name: endpoint_id
        required: true
        schema:
          type: string
      responses:
        '200':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/DedicatedEndpoint'
          description: Endpoint reactivated.
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          $ref: '#/components/responses/NotFound'
      summary: Restore the persisted min/max bounds and set the endpoint back to active.
      tags:
      - bff
      - dedicated
  /api/dedicated/{endpoint_id}/scaling-history:
    get:
      description: 'One row per replica the endpoint has run, newest first. `boot_seconds`

        (started → ready) and `served_seconds` (ready → released) are derived

        here so the two are computed once: those start-up minutes were BILLED —

        the replica held its cards for the whole boot.


        AN EMPTY LIST AND AN UNREADABLE HISTORY ARE DIFFERENT SENTENCES.

        `items: []` states that this endpoint has never started a replica, which

        is alarming and almost always false; a history that could not be read is

        a 503 (`scaling_history_unavailable`), never an empty list.


        The scheduling backend''s internal join keys (deployment id, endpoint id,

        run id) are STRIPPED here (SPEC-DEDICATED §5.4) — `replica_id` is an

        opaque list key that addresses nothing, and `version` is the endpoint''s

        OWN deployment count, which is what makes "this replica came from the

        configuration before your mode change" legible. The admin mirror keeps

        the raw identifiers.


        `capacity_class` is read from the replica''s own run, and only for

        replicas still running (bounded by `max_replicas`, so a 200-row history

        is not 200 upstream reads). A released replica reports `null`, and so

        does a run we could not read — it is NEVER substituted from the

        endpoint''s current `mode`, which is precisely the question this field

        answers independently.

        '
      operationId: dedicatedScalingHistory
      parameters:
      - in: path
        name: endpoint_id
        required: true
        schema:
          type: string
      - description: 'Rows returned, newest first. Validated HERE so a refusal names
          our

          field and the scheduling backend''s own wording never reaches a

          customer (PAAS-68).

          '
        in: query
        name: limit
        required: false
        schema:
          default: 50
          maximum: 200
          minimum: 1
          type: integer
      responses:
        '200':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/DedicatedScalingHistory'
          description: Replica lifecycle rows, newest first.
        '400':
          $ref: '#/components/responses/BadRequest'
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          $ref: '#/components/responses/NotFound'
        '503':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorEnvelope'
          description: 'The scaling history could not be read (code

            `scaling_history_unavailable`). Nothing was changed; retry shortly.

            Deliberately NOT an empty list: "we could not ask" is not "it is not

            there".

            '
      summary: When the endpoint added or released a replica, and how long each took
        to be ready.
      tags:
      - bff
      - dedicated
  /api/dedicated/{endpoint_id}/telemetry:
    get:
      description: 'READ-ONLY, aggregated from this endpoint''s own metered requests
        over the

        window. MEASURED, never projected: an endpoint that served nothing

        reports zeroes and nulls rather than an estimate.


        THREE routes rather than one composite (this, `scaling-history` and

        `replicas`): they have different freshness, different cost and different

        failure modes — a control-plane blip must not blank the database-backed

        numbers a customer came for.


        NULL IS A VALUE HERE, and never a 0. `prefix_cache.hit_rate`, every

        percentile and `block_tokens` are null when nothing reported the figure,

        which is a different fact from "measured, and it was zero" — the columns

        behind them are nullable with no default for exactly that reason. A 0 %

        prefix-cache hit rate is a real, actionable finding; rendering "not

        measured" as 0 % would make the two indistinguishable.

        `prefix_cache.coverage` is what makes a 0 % readable.

        '
      operationId: dedicatedTelemetry
      parameters:
      - in: path
        name: endpoint_id
        required: true
        schema:
          type: string
      - description: 'Trailing window. A CLOSED set, validated rather than passed
          to a

          filter: an unrecognised value that quietly matched nothing would

          read as "this endpoint had no traffic", the silent failure §12.7

          exists to forbid.

          '
        in: query
        name: window
        required: false
        schema:
          default: 24h
          enum:
          - 1h
          - 24h
          - 7d
          type: string
      responses:
        '200':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/DedicatedTelemetry'
          description: Aggregates over `[from, to)`.
        '400':
          $ref: '#/components/responses/BadRequest'
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          $ref: '#/components/responses/NotFound'
      summary: What the endpoint actually did — prefix reuse, TTFT, latency, request
        counts.
      tags:
      - bff
      - dedicated
  /v1/audio/speech:
    post:
      description: 'JSON `{model, input, voice?}` → audio bytes (`audio/wav`). Served
        by

        the Orpheus wrapper (voices: tara, leah, jess, leo, dan, mia, zac, zoe;

        inline emotion tags `<laugh>`, `<sigh>`…). `input` ≤ 4096 characters.

        '
      operationId: createSpeech
      requestBody:
        content:
          application/json:
            schema:
              properties:
                input:
                  maxLength: 4096
                  type: string
                model:
                  type: string
                voice:
                  default: tara
                  type: string
              required:
              - model
              - input
              type: object
        required: true
      responses:
        '200':
          content:
            audio/wav:
              schema:
                format: binary
                type: string
          description: WAV audio (mono 24 kHz).
        '400':
          $ref: '#/components/responses/BadRequest'
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          $ref: '#/components/responses/ModelNotFound'
        '429':
          $ref: '#/components/responses/RateLimited'
        '502':
          $ref: '#/components/responses/UpstreamError'
        '503':
          $ref: '#/components/responses/ModelUnavailable'
        '504':
          $ref: '#/components/responses/UpstreamTimeout'
      summary: Text-to-speech (TTS, OpenAI-compatible).
      tags:
      - openai
  /v1/audio/transcriptions:
    post:
      description: 'Multipart `file=<audio>` (wav, mp3, m4a, flac, ogg — 25 MB max)
        +

        `model=<model_id tier=audio>`. The gateway proxies the multipart as-is

        to the vLLM replica (Whisper/Voxtral), which returns `{text, usage:{type:

        "duration", seconds}}`. v1 metering: estimated from the transcribed text

        (metering="estimated") — the "audio seconds" unit is coming in M2.

        '
      operationId: createTranscription
      requestBody:
        content:
          multipart/form-data:
            schema:
              properties:
                file:
                  format: binary
                  type: string
                model:
                  type: string
              required:
              - file
              - model
              type: object
        required: true
      responses:
        '200':
          content:
            application/json:
              schema:
                properties:
                  text:
                    type: string
                  usage:
                    properties:
                      seconds:
                        type: integer
                      type:
                        type: string
                    type: object
                type: object
          description: Transcription result.
        '400':
          $ref: '#/components/responses/BadRequest'
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          $ref: '#/components/responses/ModelNotFound'
        '413':
          $ref: '#/components/responses/PayloadTooLarge'
        '429':
          $ref: '#/components/responses/RateLimited'
        '502':
          $ref: '#/components/responses/UpstreamError'
        '503':
          $ref: '#/components/responses/ModelUnavailable'
        '504':
          $ref: '#/components/responses/UpstreamTimeout'
      summary: Audio transcription (ASR, OpenAI-compatible).
      tags:
      - openai
  /v1/chat/completions:
    post:
      description: "OpenAI-compatible. Supported fields: `model`, `messages`, `stream`,\n\
        `stream_options`, `max_tokens`, `temperature`, `top_p`, `stop`, `seed`,\n\
        `frequency_penalty`, `presence_penalty`, `n`, `web_search_options`,\n`tools`,\
        \ `tool_choice`. Out-of-scope fields (`response_format`…) are\nrelayed as-is\
        \ to vLLM if it supports them, otherwise a clear error is\nreturned (SPEC\
        \ §5.1). On a Harmony-format model whose `auto` tool calls need it,\nwhen\
        \ `tool_choice` is `auto` or absent and no function tool sets `strict`, the\n\
        gateway sends every function tool with `strict: true` (here and on\n`/v1/responses`),\
        \ which constrains the call to the Harmony format; set `strict`\non any tool\
        \ yourself to opt out.\n\n**The pass-through is deliberate, and two things\
        \ keep it honest.** An allowlist\nwould refuse the engine extras callers legitimately\
        \ use (`chat_template_kwargs`,\n`top_k`, `repetition_penalty`, the guided-decoding\
        \ family), so an unknown field is\nrelayed rather than rejected. What a caller\
        \ gets instead is the answer BEFORE and\nAFTER: `supported_parameters` on\
        \ `GET /v1/models` lists, per model, what this\ngateway stands behind, and\
        \ any parameter it had to drop is named on the response in\n`x-shadow-ignored-params`.\
        \ A field that is neither listed nor announced was relayed\nto the engine\
        \ — the one case left where \"accepted\" does not prove \"applied\".\n\nGuardrails\
        \ (400 otherwise): `model` known, body size ≤ limit,\n`max_tokens` ≤ the model's\
        \ `max_output_tokens`,\n`prompt_tokens + max_tokens ≤ max_model_len`, every\
        \ `messages[].role` in the enum.\n\n**Two parameters are answered per MODEL**,\
        \ and `supported_parameters` on\n`GET /v1/models` says which beforehand:\n\
        \n- `reasoning_effort` — HONOURED where the model exposes a reasoning control:\n\
        \  the gateway translates it onto that control (`chat_template_kwargs.\n \
        \ enable_thinking`) instead of relaying a field the engine validates and then\n\
        \  ignores. `off` is a synonym of `none`. On a model whose chat template reads\n\
        \  the grade itself (the Harmony models, e.g. `gpt-oss-20b`) it is relayed:\n\
        \  `minimal` becomes `low`, and `none` is dropped because no grade turns that\n\
        \  deliberation off. On a model with no such control, or `none` on a Harmony\n\
        \  model, it is dropped and named in the `x-shadow-ignored-params` response\n\
        \  header, never a silent no-op. An unknown value is a 400.\n- `stop` — REFUSED\
        \ (400 `unsupported_parameter`) on a model whose deliberation\n  comes back\
        \ in its own channel (`reasoning.separate_channel: true`). Stop\n  sequences\
        \ are matched against that channel too, so a sequence the model\n  happens\
        \ to think about ends the generation before the answer exists, with the\n\
        \  `finish_reason: \"stop\"` of a normal completion — a failure the client\
        \ cannot\n  detect. Reproduced outside this platform (OpenRouter, qwen3-30b-a3b)\
        \ and open\n  upstream as vllm-project/vllm#38499.\n\n`tools` carries TWO\
        \ vocabularies. A `{\"type\": \"function\"}` entry is a\nCLIENT function:\
        \ relayed as-is, and the model's call comes back for the\ncaller to answer.\
        \ An entry naming a SERVER tool\n(`{\"type\": \"web_search\"}`, `{\"type\"\
        : \"web_fetch\", \"max_uses\": 3}`,\n`{\"type\": \"code_interpreter\"}` —\
        \ see docs/SERVER-TOOLS.md) is executed by\nthe gateway inside its own loop\
        \ and never reaches vLLM, which understands\nfunction declarations only. Any\
        \ other `type` is a 400 rather than a\npass-through: a declaration nothing\
        \ implements would otherwise be dropped\nin silence while the caller believed\
        \ a tool had run. Each server tool\nneeds its own account flag (`web_search_enabled`,\
        \ `web_fetch_enabled`,\n`code_interpreter_enabled`) and a model with native\
        \ tool calling: a model\nwhose tool calls the serving stack cannot extract\
        \ is refused up front\n(400) rather than answered with raw JSON — see the\
        \ per-model table in\ndocs/SERVER-TOOLS.md.\n\n`web_search_options` (PAAS-5,\
        \ opt-in, see docs/SERVER-TOOLS.md): the\nlegacy entry point for the same\
        \ web_search tool. When present, the\ngateway runs a tool loop against the\
        \ Staan WebSearch4AI vendor and\nreturns a cited answer (`[n]` inline markers\
        \ +\n`message.annotations`). Requires the account flag `web_search_enabled`\n\
        and a model with native tool calling.\n\nSTREAMING with any server tool declared:\
        \ the gateway is no longer a\nrelay, and the loop publishes its activity as\
        \ extension chunks with\n`\"choices\": []` and a `\"web_search\"` object (full\
        \ ordering in\nSTREAMING.md §2.2 bis):\n\n- `{\"status\": \"searching\", \"\
        query\": \"...\"}` — a web_search call about to run;\n- `{\"status\": \"fetching\"\
        , \"url\": \"...\"}` — a web_fetch call about to run;\n- `{\"status\": \"\
        running_code\", \"language\": \"python\", \"code_chars\": <n>}` —\na code_interpreter\
        \ run about to start. The PROGRAM is not echoed here: the\nclient already\
        \ has it in the tool call, and an activity chunk is not the\nplace to re-broadcast\
        \ the user's data.\nThe key stays `web_search` for EVERY tool: it is the name\
        \ the first tool\nshipped with, and a rename would break deployed clients.\
        \ A replacement\nnaming the tool and the call's outcome will be emitted ALONGSIDE\
        \ it for\none release before this one is retired.\n- `{\"status\": \"done\"\
        , \"searches\": <n>, \"price_per_search\": <float>}` —\nonce per request,\
        \ even when no call ran. `searches` counts web_search\nONLY (it is a billed\
        \ unit, not a total of tool calls), so a fetch-only or\ncompute-only turn\
        \ legitimately ends on `searches: 0`.\n\nPer-call OUTCOMES are not published\
        \ on this surface: a failed call is\nindistinguishable from a successful one\
        \ here, and the request still\nreturns 200 (a tool failure is never a failed\
        \ request). A client that\nneeds to know WHICH tool ran and whether it succeeded\
        \ should use\nPOST /v1/responses, whose output items name the tool and carry\n\
        `status: \"failed\"`. A final chunk carries `delta.annotations` with the\n\
        `url_citation` list. In the non-stream response, the same information is\n\
        exposed as a top-level `web_search` object (`{\"searches\": <n>,\n\"price_per_search\"\
        : <float>}`) alongside\n`choices[].message.annotations`.\n\nIf `stream=false`\
        \ → JSON `ChatCompletion` response.\nIf `stream=true`  → `text/event-stream`\
        \ SSE stream of `ChatCompletionChunk`\nterminated by `data: [DONE]` (native\
        \ relay of the vLLM stream — see\nSTREAMING.md). With `stream_options.include_usage=true`,\
        \ a final chunk\ncarries `usage`.\n"
      operationId: createChatCompletion
      requestBody:
        content:
          application/json:
            schema:
              $ref: '#/components/schemas/ChatCompletionRequest'
        required: true
      responses:
        '200':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ChatCompletion'
            text/event-stream:
              schema:
                description: 'Sequence of `data: <ChatCompletionChunk JSON>\n\n` lines,

                  an optional final `usage` chunk, then `data: [DONE]\n\n`.

                  '
                type: string
          description: 'Completion. JSON if `stream=false`; SSE stream if `stream=true`.

            '
        '400':
          $ref: '#/components/responses/BadRequest'
        '401':
          $ref: '#/components/responses/Unauthorized'
        '403':
          $ref: '#/components/responses/ExternalProviderNotEnabled'
        '404':
          $ref: '#/components/responses/ModelNotFound'
        '413':
          $ref: '#/components/responses/PayloadTooLarge'
        '429':
          $ref: '#/components/responses/RateLimited'
        '502':
          $ref: '#/components/responses/UpstreamError'
        '503':
          $ref: '#/components/responses/ModelUnavailable'
        '504':
          $ref: '#/components/responses/UpstreamTimeout'
      summary: Chat completions (core route, stream + non-stream).
      tags:
      - openai
  /v1/completions:
    post:
      description: 'Legacy OpenAI-compatible. `prompt` string or array of strings.
        Same

        guardrails, same stream/non-stream modes (native SSE relay) as chat.

        '
      operationId: createCompletion
      requestBody:
        content:
          application/json:
            schema:
              $ref: '#/components/schemas/CompletionRequest'
        required: true
      responses:
        '200':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/Completion'
            text/event-stream:
              schema:
                type: string
          description: Completion (JSON or SSE depending on `stream`).
        '400':
          $ref: '#/components/responses/BadRequest'
        '401':
          $ref: '#/components/responses/Unauthorized'
        '403':
          $ref: '#/components/responses/ExternalProviderNotEnabled'
        '404':
          $ref: '#/components/responses/ModelNotFound'
        '413':
          $ref: '#/components/responses/PayloadTooLarge'
        '429':
          $ref: '#/components/responses/RateLimited'
        '502':
          $ref: '#/components/responses/UpstreamError'
        '503':
          $ref: '#/components/responses/ModelUnavailable'
        '504':
          $ref: '#/components/responses/UpstreamTimeout'
      summary: Legacy completions (compat for older clients).
      tags:
      - openai
  /v1/detect:
    post:
      description: 'Image in → structured regions out. JSON `{model, image, ...knobs}`
        where

        `image` is base64 or a `data:image/…;base64,…` URI. Serves detection heads

        (e.g. `tatr` — table structure). Synchronous, one JSON document, NO streaming

        (`stream:true` → 400). Model addressing is exact (no default). Billed per

        processed image.

        '
      operationId: detect
      requestBody:
        content:
          application/json:
            schema:
              properties:
                image:
                  description: base64 or data-URI image
                  type: string
                model:
                  type: string
              required:
              - model
              - image
              type: object
        required: true
      responses:
        '200':
          content:
            application/json:
              schema:
                properties:
                  model:
                    type: string
                  object:
                    example: detection
                    type: string
                  regions:
                    items:
                      properties:
                        box:
                          description: '[l, t, r, b] in pixels'
                          items:
                            type: number
                          type: array
                        label:
                          type: string
                        score:
                          type: number
                      type: object
                    type: array
                type: object
          description: Detection result.
        '400':
          $ref: '#/components/responses/BadRequest'
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          $ref: '#/components/responses/ModelNotFound'
        '502':
          $ref: '#/components/responses/UpstreamError'
        '503':
          $ref: '#/components/responses/ModelUnavailable'
        '504':
          $ref: '#/components/responses/UpstreamTimeout'
      summary: Object/region detection (document-CV utility heads).
      tags:
      - openai
  /v1/models:
    get:
      description: 'OpenAI format `{object:"list", data:[Model…]}`. The list comes
        from

        `models.yaml` (the gateway catalog), not vLLM.

        '
      operationId: listModels
      responses:
        '200':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ModelList'
          description: Catalog.
        '401':
          $ref: '#/components/responses/Unauthorized'
      summary: List the models in the catalog.
      tags:
      - openai
  /v1/parse:
    post:
      description: 'Image in → structured text/table out. JSON `{model, image, ...knobs}`.

        Serves parsing heads (`trocr` — handwriting; `got-ocr-2` — general OCR;

        `deplot` — chart to data table; `pix2text` — math formula to LaTeX;

        `tableformer` — table structure recognition).

        The response carries a polymorphic `text`/`table`/`tables`/`output` field
        per

        model. Synchronous, one JSON document, NO streaming (`stream:true` → 400).

        Billed per processed image.


        **`tableformer` additionally requires `table_bboxes`** — it recognises the

        structure inside a table it is GIVEN, it does not locate the table. Get the

        boxes from `tatr` on `/v1/detect`, or pass the full-page box. A body without
        it

        is rejected with 400 immediately (no replica is woken).

        '
      operationId: parse
      requestBody:
        content:
          application/json:
            schema:
              properties:
                image:
                  description: base64 or data-URI image
                  type: string
                max_new_tokens:
                  description: Generation ceiling for the text-generating heads (`trocr`,
                    `deplot`). Clamped to the model's own bound.
                  type: integer
                model:
                  type: string
                table_bboxes:
                  description: 'REQUIRED for `tableformer`: pixel boxes of the tables
                    to recognise. Ignored by the other parse heads.'
                  items:
                    description: '[l, t, r, b] in pixels'
                    items:
                      type: number
                    type: array
                  type: array
              required:
              - model
              - image
              type: object
        required: true
      responses:
        '200':
          content:
            application/json:
              schema:
                properties:
                  model:
                    type: string
                  object:
                    example: parse
                    type: string
                type: object
          description: Parse result (shape varies by model).
        '400':
          $ref: '#/components/responses/BadRequest'
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          $ref: '#/components/responses/ModelNotFound'
        '502':
          $ref: '#/components/responses/UpstreamError'
        '503':
          $ref: '#/components/responses/ModelUnavailable'
        '504':
          $ref: '#/components/responses/UpstreamTimeout'
      summary: Structured parsing (document-CV utility heads).
      tags:
      - openai
  /v1/responses:
    post:
      description: 'OpenAI-compatible Responses surface, implemented as a TRANSLATION
        LAYER over the

        same gateway-owned server-tool loop `/v1/chat/completions` uses: the tools,
        the

        budgets, the citations and the billing are identical, only the request and
        event

        vocabularies differ (see docs/SERVER-TOOLS.md).


        Accepted: `model`, `input` (a string, or an array of `message` /

        `function_call` / `function_call_output` items), `instructions`, `tools`,

        `tool_choice`, `parallel_tool_calls`, `max_output_tokens`, `temperature`,

        `top_p`, `seed`, `stream`, `store` (false), `metadata`, `text`

        (`{"format": {"type": "text"}}` only), `user`, `reasoning`

        (`{"effort": …}`, see the schema).


        **Stateless by design.** `store: true` and `previous_response_id` return 400

        `unsupported_parameter`: no conversation is persisted, so accepting either
        would

        mean answering a different question than the client believes it asked. `store`

        is echoed as `false` in every response, including when it was omitted (OpenAI

        defaults it to true). Any parameter not in the list above is a 400

        `unknown_parameter` rather than being silently ignored.


        **Not available here:** `ded/<slug>` aliases (a pricing surface of

        `/v1/chat/completions`, refused with an explicit 400 rather than served at
        a

        different price), and `n > 1` (the surface returns one response). Pool pricing
        is

        spot like everywhere else: the multiplier in force at submission is stamped
        on

        the usage event; the legacy `:spot` suffix is accepted as a deprecated no-op.


        `tools` accepts `function` declarations (flattened, the Responses shape) plus
        the

        built-in server tools — `{"type": "web_search"}`, `{"type": "web_fetch"}`,

        `{"type": "code_interpreter"}` — which require the matching account entitlement

        and a model with native tool calling. `tool_choice` cannot be combined with
        a

        built-in tool (400): the gateway drives the loop and sets `tool_choice` itself
        on

        every round.


        **Output** is an item list: one `web_search_call` / `web_fetch_call` /

        `code_interpreter_call` per executed server-tool call (`action` says what
        it did,

        `status` whether it worked — a tool failure is a failed ITEM, never a failed

        request), then a `reasoning` item when the model deliberated in its own channel,

        one `function_call` per client function the model asked for, and the assistant

        `message` whose `output_text` content part carries `url_citation` annotations.


        The `reasoning` item carries the model''s RAW deliberation in

        `content[].reasoning_text` (`summary` stays empty — nothing here summarises
        it),

        and streams as `response.reasoning_text.delta` events that close on

        `response.reasoning_text.done` before the answer''s first token, so a client
        can

        show the thinking live and collapse it when the answer starts. It appears
        only

        for models whose serving stack splits that channel

        (`reasoning.separate_channel` on `GET /v1/models`); on every other model the

        deliberation is part of the message and nothing can separate it.


        `usage.output_tokens_details.reasoning_tokens` is present only when the engine

        reported the split. It used to be hard-coded to 0, which reads as "this model
        did

        not reason" on a response whose tokens were nearly all deliberation. The same
        rule

        holds on `/v1/chat/completions`, where `usage.completion_tokens_details` is
        relayed

        verbatim when the engine sends it and omitted when it does not: splitting
        the count

        here would need a tokenizer, and an estimate presented as accounting is worse
        than

        an absent field.


        **Citations** carry BOTH shapes: `index` is the `[n]` marker number the model

        writes inline (authoritative, the same value the chat surface returns) and

        `start_index`/`end_index` are the best-effort character span of that marker
        in the

        answer. A source the model never referenced has an EMPTY span at the end of
        the

        text — models do cite numbers they never fetched, and there is no honest offset

        for those. See docs/SERVER-TOOLS.md.


        If `stream=false` → one JSON `Response`. If `stream=true` → a NAMED-event
        SSE

        stream (`event:` + `data:` lines) with `sequence_number` on every event:

        `response.created`, `response.output_item.added`/`.done`,

        `response.<tool>_call.in_progress`/`.completed`, `response.content_part.added`/

        `.done`, `response.output_text.delta`/`.done`,

        `response.output_text.annotation.added`, and exactly one terminal event —

        `response.completed`, `response.incomplete` (truncated answer, or the model

        stream breaking after partial text) or `response.failed` (no answer produced).

        There is no `[DONE]` sentinel: the terminal event is the terminator.


        Metering: `usage_events.endpoint = ''responses''`, one token event per request

        (streamed or not), plus one row per executed server-tool call under that tool''s

        own endpoint — the same rows the chat surface writes.

        '
      operationId: createResponse
      requestBody:
        content:
          application/json:
            schema:
              $ref: '#/components/schemas/ResponsesRequest'
        required: true
      responses:
        '200':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/Response'
            text/event-stream:
              schema:
                description: 'Sequence of `event: <type>\ndata: <JSON>\n\n` blocks,
                  ending with

                  `response.completed`, `response.incomplete` or `response.failed`.

                  '
                type: string
          description: 'The response. JSON if `stream=false`; a named-event SSE stream
            if `stream=true`.

            '
        '400':
          $ref: '#/components/responses/BadRequest'
        '401':
          $ref: '#/components/responses/Unauthorized'
        '403':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorEnvelope'
          description: 'The account is not entitled to a declared server tool, the
            model is disabled

            by an administrator, or the model is served by an external inference provider

            the organization has not enabled (`external_provider_not_enabled`, PAAS-206
            —

            see the response of the same name).

            '
        '404':
          $ref: '#/components/responses/ModelNotFound'
        '413':
          $ref: '#/components/responses/PayloadTooLarge'
        '429':
          $ref: '#/components/responses/RateLimited'
        '502':
          $ref: '#/components/responses/UpstreamError'
        '503':
          $ref: '#/components/responses/ModelUnavailable'
        '504':
          $ref: '#/components/responses/UpstreamTimeout'
      summary: Responses API (server tools, item-based output) — stateless only.
      tags:
      - openai
  /v1/video/generations:
    get:
      operationId: listVideoGenerations
      responses:
        '200':
          content:
            application/json:
              schema:
                properties:
                  data:
                    items:
                      $ref: '#/components/schemas/VideoGenerationJob'
                    type: array
                  object:
                    type: string
                type: object
          description: List.
      summary: List the requester's 20 most recent video jobs.
      tags:
      - video
    post:
      description: 'Submit→poll: generation (1-3 min) runs as an async Forge job —
        the

        202 returns immediately with a job_id. Quota: 2 non-terminal jobs

        per user (429 code video_jobs_quota_exceeded). Limits:

        704×480, 161 frames, 50 steps, prompt ≤ 2000 chars.

        '
      operationId: createVideoGeneration
      requestBody:
        content:
          application/json:
            schema:
              properties:
                fps:
                  maximum: 30
                  minimum: 8
                  type: integer
                height:
                  maximum: 480
                  type: integer
                model:
                  type: string
                negative_prompt:
                  maxLength: 2000
                  type: string
                num_frames:
                  description: rounded to 8k+1
                  maximum: 161
                  type: integer
                num_inference_steps:
                  maximum: 50
                  type: integer
                prompt:
                  maxLength: 2000
                  type: string
                seed:
                  type: integer
                width:
                  maximum: 704
                  type: integer
              required:
              - model
              - prompt
              type: object
        required: true
      responses:
        '202':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/VideoGenerationJob'
          description: Job created.
        '400':
          $ref: '#/components/responses/BadRequest'
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          $ref: '#/components/responses/ModelNotFound'
        '429':
          $ref: '#/components/responses/RateLimited'
      summary: Submit a text-to-video job (asynchronous, 202).
      tags:
      - video
  /v1/video/generations/{job_id}:
    get:
      description: 'Reconciles state with the control plane on EVERY read (the presigned

        S3 URL is re-resolved — the one returned previously expires).

        Uniform 404 if the job does not exist or belongs to someone else.

        '
      operationId: getVideoGeneration
      parameters:
      - in: path
        name: job_id
        required: true
        schema:
          type: string
      responses:
        '200':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/VideoGenerationJob'
          description: Job.
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          $ref: '#/components/responses/ModelNotFound'
      summary: State of a video job (+ presigned mp4 URL when completed).
      tags:
      - video
  /v1/video/generations/{job_id}/cancel:
    post:
      operationId: cancelVideoGeneration
      parameters:
      - in: path
        name: job_id
        required: true
        schema:
          type: string
      responses:
        '200':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/VideoGenerationJob'
          description: Job (status cancelled if the cancellation took effect).
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          $ref: '#/components/responses/ModelNotFound'
      summary: Cancel a video job (best-effort).
      tags:
      - video
security:
- ApiKeyAuth: []
servers:
- description: Gateway root (port 8000).
  url: /
tags:
- description: OpenAI-compatible surface (/v1), consumed by the OpenAI SDK.
  name: openai
- description: Backend-for-frontend for the dashboard (/api).
  name: bff
- description: Routes reserved for the admin role.
  name: admin
