# GENERATED from the gateway's contract by apps/gateway/src/contract/openapi.ts.
# Do not edit by hand: run `npm run openapi -w @qirnas/gateway`. CI fails when it is stale.
openapi: "3.1.0"
info:
  title: MindLab API
  version: "1.0.0"
  description: |-
    OpenAI-compatible chat completions for the Qirnas models: point any OpenAI SDK at this server's
    `/v1` and use your API key.

    **Authentication.** `Authorization: Bearer <API key>` on every request.

    **Models.** `GET /v1/models` lists the models your key can call, with each one's context window,
    output limits, capabilities and status. `qirnas` is a permanent alias for the default model:
    send it, or no `model` at all, and the default model answers. A model id the gateway does not know
    is answered by the default model too, but that is deprecated: such a response carries a
    `Deprecation` header (RFC 9745) and a `Link` to the changelog, which will announce, at least 60
    days ahead, when unknown ids start getting 404 `model_not_found`. Send an id from
    `GET /v1/models`, or `qirnas`.

    **Request ids.** Every response carries `x-request-id`. Send your own (1 to 128 of
    `A-Z a-z 0-9 . _ : -`) to line your logs up with ours; anything else is replaced by a generated
    id. Error bodies repeat it as `request_id`: quote it when you report a problem.

    **Errors.** Every error has the same body: `{"error": {"message", "type", "code", "param"},
    "request_id"}` (OpenAI's error object). Branch on `error.code`, never on the message; codes are
    only ever added. The body also carries `statusCode`, `message` and `code` for older clients.

    **Streaming.** With `stream: true` the answer arrives as server-sent events, one
    `data: <ChatCompletionChunk>` per event, ending with `data: [DONE]`. While a model starts (a
    cold GPU can take minutes) the gateway sends an SSE comment line, `: keep-alive`, every
    15 s: skip lines that start with `:`. A failure after the stream began arrives
    in-band as one `data: <InBandError>` event, then `data: [DONE]`; its `error.code` is one of
    `context_length_exceeded`, `upstream_error`, `upstream_timeout`. A failure before the stream began is a normal HTTP
    error. With `stream_options.include_usage` the last chunk before `[DONE]` carries the usage.

    **Tool calling.** Offer `tools` (and optionally `tool_choice`); when the model calls one the
    answer has `finish_reason: "tool_calls"` and `message.tool_calls`. Run the tools, append the
    assistant message and one `role: "tool"` message per call (with its `tool_call_id`), and call
    again.

    **Scope-gated parameters.** `reasoning` and `think` are part of the schema but need a key
    holding the scope of the same name; without it, `true` is refused with 403 `scope_required`.

    **Embeddings and search.** `POST /v1/embeddings` (OpenAI-compatible) and `POST /v1/search`
    need a key holding the `embeddings` or `search` scope; without it they answer 403
    `scope_required`. Their capacity is shared by all callers: besides your key's own limit
    (`rate_limit_exceeded`) they can answer 429 `capacity_exceeded` and 503
    `service_unavailable`, with `Retry-After`.
servers:
  - url: "https://api.mindlabsa.com"
security:
  - bearerAuth: []
paths:
  "/v1/chat/completions":
    post:
      operationId: createChatCompletion
      summary: Create a chat completion
      description: "Single-shot (JSON) or streamed (server-sent events). The gateway answers a successful request with 201 (any 2xx is success). The request body is at most 256 KB (else 413 `request_too_large`)."
      parameters:
        - "$ref": "#/components/parameters/RequestId"
      requestBody:
        required: true
        content:
          application/json:
            schema:
              "$ref": "#/components/schemas/ChatCompletionRequest"
      responses:
        "201":
          description: "The answer: a `ChatCompletion`, or with `stream: true` a stream of `ChatCompletionChunk` events (see the API description for the event format)."
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
            Deprecation:
              "$ref": "#/components/headers/Deprecation"
            Link:
              "$ref": "#/components/headers/Link"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ChatCompletion"
            text/event-stream:
              schema:
                type: string
                description: "`data: <ChatCompletionChunk JSON>` events, `: keep-alive` comments, at most one `data: <InBandError JSON>` event, then `data: [DONE]`."
        "400":
          description: |-
            Error. `error.code` is one of:
            - `invalid_request`: The body is not valid JSON, a field failed validation, or two fields cannot be combined. `param` names the first offending field (e.g. `temperature`, `messages.0.role`). Fix the request; do not retry it.
            - `context_length_exceeded`: The messages plus the output budget (`max_tokens`, or the model's `default_max_tokens`) do not fit the model's `context_window` (see GET /v1/models). Shorten the messages or lower `max_tokens`. `param` is `messages`.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
            Deprecation:
              "$ref": "#/components/headers/Deprecation"
            Link:
              "$ref": "#/components/headers/Link"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "401":
          description: |-
            Error. `error.code` is one of:
            - `missing_api_key`: No `Authorization: Bearer <key>` header was sent.
            - `invalid_api_key`: The API key is unknown, revoked or expired.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "403":
          description: |-
            Error. `error.code` is one of:
            - `scope_required`: The request needs a scope the API key does not hold. `param` names what needs it: `reasoning` (the `reasoning` scope) or `think` (the `think` scope), or `model` for a model that runs think mode (the `think` scope). Send the request without it, or use a key that holds the scope. Or the route itself needs one: `POST /v1/embeddings` needs `embeddings` and `POST /v1/search` needs `search`, in every enforcement mode (`param` is absent then).
            - `model_not_permitted`: The API key may not use this model: it is outside the key's model list, or it is an outside-provider model the key holds no grant for. `param` is `model`. GET /v1/models lists the models you can call.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
            Deprecation:
              "$ref": "#/components/headers/Deprecation"
            Link:
              "$ref": "#/components/headers/Link"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "404":
          description: |-
            Error. `error.code` is one of:
            - `model_not_found`: The model is not available to this API key. GET /v1/models lists the models you can call. (A model id the gateway does not know is served by the default model, with a `Deprecation` header.)
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "413":
          description: |-
            Error. `error.code` is one of:
            - `request_too_large`: The request body is larger than the gateway accepts: 256 KB for `POST /v1/chat/completions` and `POST /v1/embeddings`, 100 KB elsewhere.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "429":
          description: |-
            Error. `error.code` is one of:
            - `rate_limit_exceeded`: This API key reached its rate limit for the current window (one owner's keys may share a limit). Wait for the `Retry-After` seconds, then retry.
            - `upstream_rate_limited`: The model's server asked the gateway to slow down. Retry shortly, with backoff. Mid-stream it arrives in-band as `upstream_error`.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
            Retry-After:
              "$ref": "#/components/headers/RetryAfter"
            Deprecation:
              "$ref": "#/components/headers/Deprecation"
            Link:
              "$ref": "#/components/headers/Link"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "500":
          description: |-
            Error. `error.code` is one of:
            - `internal_error`: An unexpected gateway error. Retry; report the `request_id` if it persists.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
            Deprecation:
              "$ref": "#/components/headers/Deprecation"
            Link:
              "$ref": "#/components/headers/Link"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "502":
          description: |-
            Error. `error.code` is one of:
            - `upstream_error`: The model's server, or the search or embeddings service, failed or returned something unusable (including a stream that stalled). Retry.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
            Deprecation:
              "$ref": "#/components/headers/Deprecation"
            Link:
              "$ref": "#/components/headers/Link"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "503":
          description: |-
            Error. `error.code` is one of:
            - `model_unavailable`: The model cannot be served: it is blocked by the gateway's upstream checks or not configured. Retrying does not help until an operator acts.
            - `model_offline`: The model exists but its GPU is switched off (`status: offline` in GET /v1/models). Use another model, or retry later.
            - `no_model_available`: No model was named (or the alias `qirnas`, or an unknown id, was used) and the default model is not currently available.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
            Deprecation:
              "$ref": "#/components/headers/Deprecation"
            Link:
              "$ref": "#/components/headers/Link"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "504":
          description: |-
            Error. `error.code` is one of:
            - `upstream_timeout`: The model, or the search or embeddings service, did not answer in time. For a model this is typically a GPU cold start that outlasted the wait: retrying in a minute usually succeeds.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
            Deprecation:
              "$ref": "#/components/headers/Deprecation"
            Link:
              "$ref": "#/components/headers/Link"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
  "/v1/models":
    get:
      operationId: listModels
      summary: List the models this key can call
      description: "Exactly the models a chat completion can name with this key, led by the alias `qirnas` (the default model's entry under its permanent name)."
      parameters:
        - "$ref": "#/components/parameters/RequestId"
      responses:
        "200":
          description: The models.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ModelList"
        "401":
          description: |-
            Error. `error.code` is one of:
            - `missing_api_key`: No `Authorization: Bearer <key>` header was sent.
            - `invalid_api_key`: The API key is unknown, revoked or expired.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "429":
          description: |-
            Error. `error.code` is one of:
            - `rate_limit_exceeded`: This API key reached its rate limit for the current window (one owner's keys may share a limit). Wait for the `Retry-After` seconds, then retry.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
            Retry-After:
              "$ref": "#/components/headers/RetryAfter"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "500":
          description: |-
            Error. `error.code` is one of:
            - `internal_error`: An unexpected gateway error. Retry; report the `request_id` if it persists.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
  "/v1/embeddings":
    post:
      operationId: createEmbeddings
      summary: Create embeddings
      description: "OpenAI-compatible embeddings from `bge-m3` (1024 dimensions, normalized to unit length), the one model this route serves. Needs a key holding the `embeddings` scope, else 403 `scope_required` (whatever the key's other scopes; no enforcement mode lets a key without it through). Up to 64 inputs and 32,000 characters per request, in a body of at most 256 KB (else 413 `request_too_large`). An input longer than 4096 tokens is truncated, not refused. The capacity is shared by all callers: a busy moment is 429 `capacity_exceeded` with `Retry-After`. A 502 or 504 carries `x-should-retry: false` (the work was already spent): send the request again later, or smaller, not at once."
      parameters:
        - "$ref": "#/components/parameters/RequestId"
      requestBody:
        required: true
        content:
          application/json:
            schema:
              "$ref": "#/components/schemas/EmbeddingsRequest"
      responses:
        "200":
          description: "One embedding per input, in order."
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/EmbeddingsResponse"
        "400":
          description: |-
            Error. `error.code` is one of:
            - `invalid_request`: The body is not valid JSON, a field failed validation, or two fields cannot be combined. `param` names the first offending field (e.g. `temperature`, `messages.0.role`). Fix the request; do not retry it.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "401":
          description: |-
            Error. `error.code` is one of:
            - `missing_api_key`: No `Authorization: Bearer <key>` header was sent.
            - `invalid_api_key`: The API key is unknown, revoked or expired.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "403":
          description: |-
            Error. `error.code` is one of:
            - `scope_required`: The request needs a scope the API key does not hold. `param` names what needs it: `reasoning` (the `reasoning` scope) or `think` (the `think` scope), or `model` for a model that runs think mode (the `think` scope). Send the request without it, or use a key that holds the scope. Or the route itself needs one: `POST /v1/embeddings` needs `embeddings` and `POST /v1/search` needs `search`, in every enforcement mode (`param` is absent then).
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "413":
          description: |-
            Error. `error.code` is one of:
            - `request_too_large`: The request body is larger than the gateway accepts: 256 KB for `POST /v1/chat/completions` and `POST /v1/embeddings`, 100 KB elsewhere.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "429":
          description: |-
            Error. `error.code` is one of:
            - `rate_limit_exceeded`: This API key reached its rate limit for the current window (one owner's keys may share a limit). Wait for the `Retry-After` seconds, then retry.
            - `capacity_exceeded`: The platform's shared capacity for this service (all callers together) is in use. It is not your key's own limit (that is `rate_limit_exceeded`). Wait for the `Retry-After` seconds, then retry.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
            Retry-After:
              "$ref": "#/components/headers/RetryAfter"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "500":
          description: |-
            Error. `error.code` is one of:
            - `internal_error`: An unexpected gateway error. Retry; report the `request_id` if it persists.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "502":
          description: |-
            Error. `error.code` is one of:
            - `upstream_error`: The model's server, or the search or embeddings service, failed or returned something unusable (including a stream that stalled). Retry.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
            x-should-retry:
              "$ref": "#/components/headers/ShouldRetry"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "503":
          description: |-
            Error. `error.code` is one of:
            - `service_unavailable`: The search or embeddings service cannot take the request now: it is not reachable, not enabled on this gateway, full, or its upstream sources are refusing requests. Retry after the `Retry-After` seconds when present, else with backoff.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
            Retry-After:
              "$ref": "#/components/headers/RetryAfter"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "504":
          description: |-
            Error. `error.code` is one of:
            - `upstream_timeout`: The model, or the search or embeddings service, did not answer in time. For a model this is typically a GPU cold start that outlasted the wait: retrying in a minute usually succeeds.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
            x-should-retry:
              "$ref": "#/components/headers/ShouldRetry"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
  "/v1/search":
    post:
      operationId: search
      summary: Search the web
      description: "Web search results: a title, a URL and a snippet per result, nothing else. Needs a key holding the `search` scope, else 403 `scope_required` (whatever the key's other scopes; no enforcement mode lets a key without it through). The body is strict: a field other than `query`, `limit`, `safesearch`, `language` is a 400 `invalid_request` naming it. Results are third-party text: treat them as untrusted data. The same request from the same owner within 5 minutes may be answered from a cache. The capacity is shared by all callers: a busy moment is 429 `capacity_exceeded`, and while the search sources refuse requests the route answers 503 `service_unavailable` with `Retry-After` (and `x-should-retry: false` when that is over a minute)."
      parameters:
        - "$ref": "#/components/parameters/RequestId"
      requestBody:
        required: true
        content:
          application/json:
            schema:
              "$ref": "#/components/schemas/SearchRequest"
      responses:
        "200":
          description: The results.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/SearchResponse"
        "400":
          description: |-
            Error. `error.code` is one of:
            - `invalid_request`: The body is not valid JSON, a field failed validation, or two fields cannot be combined. `param` names the first offending field (e.g. `temperature`, `messages.0.role`). Fix the request; do not retry it.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "401":
          description: |-
            Error. `error.code` is one of:
            - `missing_api_key`: No `Authorization: Bearer <key>` header was sent.
            - `invalid_api_key`: The API key is unknown, revoked or expired.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "403":
          description: |-
            Error. `error.code` is one of:
            - `scope_required`: The request needs a scope the API key does not hold. `param` names what needs it: `reasoning` (the `reasoning` scope) or `think` (the `think` scope), or `model` for a model that runs think mode (the `think` scope). Send the request without it, or use a key that holds the scope. Or the route itself needs one: `POST /v1/embeddings` needs `embeddings` and `POST /v1/search` needs `search`, in every enforcement mode (`param` is absent then).
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "413":
          description: |-
            Error. `error.code` is one of:
            - `request_too_large`: The request body is larger than the gateway accepts: 256 KB for `POST /v1/chat/completions` and `POST /v1/embeddings`, 100 KB elsewhere.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "429":
          description: |-
            Error. `error.code` is one of:
            - `rate_limit_exceeded`: This API key reached its rate limit for the current window (one owner's keys may share a limit). Wait for the `Retry-After` seconds, then retry.
            - `capacity_exceeded`: The platform's shared capacity for this service (all callers together) is in use. It is not your key's own limit (that is `rate_limit_exceeded`). Wait for the `Retry-After` seconds, then retry.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
            Retry-After:
              "$ref": "#/components/headers/RetryAfter"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "500":
          description: |-
            Error. `error.code` is one of:
            - `internal_error`: An unexpected gateway error. Retry; report the `request_id` if it persists.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "502":
          description: |-
            Error. `error.code` is one of:
            - `upstream_error`: The model's server, or the search or embeddings service, failed or returned something unusable (including a stream that stalled). Retry.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "503":
          description: |-
            Error. `error.code` is one of:
            - `service_unavailable`: The search or embeddings service cannot take the request now: it is not reachable, not enabled on this gateway, full, or its upstream sources are refusing requests. Retry after the `Retry-After` seconds when present, else with backoff.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
            Retry-After:
              "$ref": "#/components/headers/RetryAfter"
            x-should-retry:
              "$ref": "#/components/headers/ShouldRetry"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
        "504":
          description: |-
            Error. `error.code` is one of:
            - `upstream_timeout`: The model, or the search or embeddings service, did not answer in time. For a model this is typically a GPU cold start that outlasted the wait: retrying in a minute usually succeeds.
          headers:
            x-request-id:
              "$ref": "#/components/headers/RequestId"
          content:
            application/json:
              schema:
                "$ref": "#/components/schemas/ErrorEnvelope"
components:
  securitySchemes:
    bearerAuth:
      type: http
      scheme: bearer
      description: Your API key.
  parameters:
    RequestId:
      name: x-request-id
      in: header
      required: false
      description: "Your id for this request, echoed back in the response header and error bodies. Anything that does not match the pattern is replaced by a generated id."
      schema:
        type: string
        pattern: "^[A-Za-z0-9._:-]{1,128}$"
  headers:
    RequestId:
      description: "This request's id: yours when you sent a valid one, else generated."
      schema:
        type: string
    Deprecation:
      description: "Only when the request's `model` is an id the gateway does not know (the alias `qirnas` is known): the default model answers such a request, but naming an unknown id is deprecated (RFC 9745). The value is when that was deprecated, as a structured-field date: `@1790467200` (2026-09-27T00:00:00Z). The errors such a request gets once it is authenticated and validated carry it too (each error status that can lists it); the ones sent before that do not: 401, 413, 429 `rate_limit_exceeded`, and a 400 for a field that fails validation. No `Sunset` header yet: the changelog announces the date unknown ids stop working at least 60 days ahead."
      schema:
        type: string
        examples:
          - "@1790467200"
    Link:
      description: "Sent with `Deprecation`: the changelog (https://developer.mindlabsa.com/changelog), as `<https://developer.mindlabsa.com/changelog>; rel=\"deprecation\"; type=\"text/html\"`."
      schema:
        type: string
        examples:
          - "<https://developer.mindlabsa.com/changelog>; rel=\"deprecation\"; type=\"text/html\""
    RetryAfter:
      description: "With `rate_limit_exceeded`: the whole seconds, at least 1, until the limit that refused the request allows another: the general request limit of this key (or of its owner), or on `POST /v1/embeddings` and `POST /v1/search` its owner's quota for that route. With `capacity_exceeded`, and with `service_unavailable` on those two routes: the whole seconds to wait before retrying (such a 503 may also come without it: then retry with backoff). Not sent with `upstream_rate_limited`: retry that one shortly, with backoff."
      schema:
        type: integer
        minimum: 1
    ShouldRetry:
      description: "`false`: do not retry this request at once (the work behind it was already spent, or the wait is long); the OpenAI SDKs honour it. Send it again later, or smaller."
      schema:
        type: string
        enum:
          - "false"
  schemas:
    ChatCompletionRequest:
      type: object
      required:
        - messages
      description: "Fields not listed here are ignored (OpenAI clients send extras), except that a listed field with an invalid value is a 400 `invalid_request` naming it in `param`. `null` for `top_p`, `presence_penalty`, `frequency_penalty`, `seed`, `stop`, `stream_options`, `user`, `reasoning`, `think` is the same as not sending the field, and so is an empty `stop` or `user`."
      properties:
        model:
          type: string
          description: "A model id from GET /v1/models, or `qirnas` for the default model. Omitted, or an id the gateway does not know: the default model (an unknown id is deprecated: see the `Deprecation` response header)."
        messages:
          type: array
          minItems: 1
          items:
            "$ref": "#/components/schemas/ChatMessage"
          description: "The conversation so far. Without a system message of yours, the model's own system prompt is used."
        temperature:
          type: number
          minimum: 0
          maximum: 2
          description: Sampling temperature.
        top_p:
          type: number
          minimum: 0
          maximum: 1
          description: "Nucleus sampling: the probability mass to sample from. `0`: only the most likely token."
        max_tokens:
          type: integer
          minimum: 1
          description: "The most tokens to generate. Omitted: the model's `default_max_tokens`. Above its `max_output_tokens`: lowered to it (see GET /v1/models)."
        presence_penalty:
          type: number
          minimum: -2
          maximum: 2
          description: "Penalizes tokens that already appeared, encouraging new topics."
        frequency_penalty:
          type: number
          minimum: -2
          maximum: 2
          description: "Penalizes tokens by how often they appeared. Omitted: the model's own default."
        seed:
          type: integer
          minimum: -9007199254740991
          maximum: 9007199254740991
          description: "Makes sampling repeatable for the same request and model, as far as the model server can."
        stop:
          description: "Up to 4 sequences where generation stops (not included in the answer). An empty string or list is the same as not sending `stop`."
          oneOf:
            - type: string
              maxLength: 256
            - type: array
              maxItems: 4
              items:
                type: string
                minLength: 1
                maxLength: 256
        stream:
          type: boolean
          description: Stream the answer as server-sent events.
        stream_options:
          "$ref": "#/components/schemas/StreamOptions"
          description: "Only with `stream: true`."
        tools:
          type: array
          items:
            "$ref": "#/components/schemas/Tool"
          description: Functions the model may call.
        tool_choice:
          "$ref": "#/components/schemas/ToolChoice"
        user:
          type: string
          maxLength: 256
          description: "An opaque id for your end user, so abuse can be traced to one of your users without knowing who they are. Send a stable hash, never an email address or a name. It is not sent to the model; the gateway logs only a keyed hash of it."
        reasoning:
          type: boolean
          default: false
          x-required-scope: reasoning
          description: "`true`: the model reasons before it answers (better answers to hard questions, more latency and tokens; the reasoning itself is not returned). Needs a key with the `reasoning` scope, else 403 `scope_required`."
        think:
          type: boolean
          default: false
          x-required-scope: think
          description: "`true`: think-harder mode. The model answers several times and the answer they agree on is returned; usage counts every attempt. Not with `tools`. Needs a key with the `think` scope, else 403 `scope_required`."
    ChatMessage:
      type: object
      required:
        - role
      properties:
        role:
          type: string
          enum:
            - system
            - user
            - assistant
            - tool
        content:
          type:
            - string
            - "null"
          description: Null for an assistant message that only calls tools.
        tool_call_id:
          type: string
          description: "For a `tool` message: the id of the call this is the result of."
        tool_calls:
          type: array
          items:
            "$ref": "#/components/schemas/ToolCall"
          description: "For an `assistant` message sent back: the tool calls it made."
    Tool:
      type: object
      required:
        - type
        - function
      properties:
        type:
          type: string
          enum:
            - function
        function:
          type: object
          required:
            - name
          properties:
            name:
              type: string
            description:
              type: string
            parameters:
              type: object
              description: "The function's arguments, as a JSON Schema object."
    ToolChoice:
      description: "`auto` (the model decides), `none`, `required`, or one named function to call."
      oneOf:
        - type: string
          enum:
            - auto
            - none
            - required
        - type: object
          required:
            - type
            - function
          properties:
            type:
              type: string
              enum:
                - function
            function:
              type: object
              required:
                - name
              properties:
                name:
                  type: string
    ToolCall:
      type: object
      required:
        - id
        - type
        - function
      properties:
        id:
          type: string
        type:
          type: string
          enum:
            - function
        function:
          type: object
          required:
            - name
            - arguments
          properties:
            name:
              type: string
            arguments:
              type: string
              description: "The arguments, JSON-encoded."
    ToolCallDelta:
      type: object
      required:
        - index
      description: "A fragment of a streamed tool call: join the fragments that share an index."
      properties:
        index:
          type: integer
        id:
          type: string
        type:
          type: string
          enum:
            - function
        function:
          type: object
          properties:
            name:
              type: string
            arguments:
              type: string
    StreamOptions:
      type: object
      properties:
        include_usage:
          type: boolean
          description: "Send one more chunk before `[DONE]`: empty `choices` and the request's `usage`."
    Usage:
      type: object
      required:
        - prompt_tokens
        - completion_tokens
        - total_tokens
      properties:
        prompt_tokens:
          type: integer
        completion_tokens:
          type: integer
        total_tokens:
          type: integer
    ChatCompletion:
      type: object
      required:
        - id
        - object
        - model
        - choices
        - usage
      properties:
        id:
          type: string
        object:
          type: string
          enum:
            - chat.completion
        model:
          type: string
          description: "Always `qirnas`."
        choices:
          type: array
          items:
            type: object
            required:
              - index
              - message
              - finish_reason
            properties:
              index:
                type: integer
              message:
                type: object
                required:
                  - role
                  - content
                properties:
                  role:
                    type: string
                    enum:
                      - assistant
                  content:
                    type:
                      - string
                      - "null"
                  tool_calls:
                    type: array
                    items:
                      "$ref": "#/components/schemas/ToolCall"
              finish_reason:
                type: string
                description: "`stop`, `length` (max_tokens reached) or `tool_calls`."
        usage:
          "$ref": "#/components/schemas/Usage"
    ChatCompletionChunk:
      type: object
      required:
        - id
        - object
        - model
        - choices
      properties:
        id:
          type: string
          description: The same for every chunk of one answer.
        object:
          type: string
          enum:
            - chat.completion.chunk
        model:
          type: string
          description: "Always `qirnas`."
        choices:
          type: array
          description: "One choice; empty only in the `include_usage` chunk."
          items:
            type: object
            required:
              - index
              - delta
              - finish_reason
            properties:
              index:
                type: integer
              delta:
                type: object
                properties:
                  content:
                    type: string
                  tool_calls:
                    type: array
                    items:
                      "$ref": "#/components/schemas/ToolCallDelta"
              finish_reason:
                type:
                  - string
                  - "null"
        usage:
          "$ref": "#/components/schemas/Usage"
          description: "Only in the last chunk, when `stream_options.include_usage` was sent."
    Model:
      type: object
      required:
        - id
        - object
        - created
        - owned_by
        - provider
        - self_hosted
        - upstream_revision
        - context_window
        - max_output_tokens
        - default_max_tokens
        - capabilities
        - status
      properties:
        id:
          type: string
          description: "The id to send as `model`."
        object:
          type: string
          enum:
            - model
        created:
          type: integer
          description: Unix seconds the model was added (0 when unknown).
        owned_by:
          type: string
          description: "The same as `provider` (OpenAI's field)."
        provider:
          type: string
          description: "Who serves the model. `unknown` until it is reviewed."
        self_hosted:
          type: boolean
          description: "Weights we control, served by an endpoint we operate or rent. A model is ours only when this is true and `provider` is `mindlab`. Not a statement about where data is processed."
        upstream_revision:
          type:
            - string
            - "null"
          pattern: "^[0-9a-f]{64}$"
          description: "For a model that is not self-hosted: a fingerprint of where its requests go, which changes when the model is pointed at another upstream. null for a self-hosted model."
        context_window:
          type:
            - integer
            - "null"
          description: "Tokens the messages and the answer must fit in together; null when not recorded. Beyond it: 400 `context_length_exceeded`."
        max_output_tokens:
          type: integer
          description: "The largest `max_tokens` the model is given; a larger request is lowered."
        default_max_tokens:
          type: integer
          description: "`max_tokens` for a request that sends none."
        capabilities:
          type: array
          items:
            type: string
          description: "What the model supports, e.g. `tools`, `streaming`."
        status:
          type: string
          enum:
            - "on"
            - offline
          description: "`offline`: switched off for now. A call by this id gets 503 `model_offline`; by the alias `qirnas` (or with no model), 503 `no_model_available`."
    ModelList:
      type: object
      required:
        - object
        - data
      properties:
        object:
          type: string
          enum:
            - list
        data:
          type: array
          items:
            "$ref": "#/components/schemas/Model"
    ErrorCode:
      type: string
      enum:
        - invalid_request
        - context_length_exceeded
        - missing_api_key
        - invalid_api_key
        - permission_denied
        - scope_required
        - model_not_permitted
        - not_found
        - model_not_found
        - request_too_large
        - rate_limit_exceeded
        - upstream_rate_limited
        - client_closed_request
        - internal_error
        - upstream_error
        - model_unavailable
        - model_offline
        - no_model_available
        - upstream_timeout
        - capacity_exceeded
        - service_unavailable
      description: |-
        - `invalid_request` (400): The body is not valid JSON, a field failed validation, or two fields cannot be combined. `param` names the first offending field (e.g. `temperature`, `messages.0.role`). Fix the request; do not retry it.
        - `context_length_exceeded` (400): The messages plus the output budget (`max_tokens`, or the model's `default_max_tokens`) do not fit the model's `context_window` (see GET /v1/models). Shorten the messages or lower `max_tokens`. `param` is `messages`.
        - `missing_api_key` (401): No `Authorization: Bearer <key>` header was sent.
        - `invalid_api_key` (401): The API key is unknown, revoked or expired.
        - `permission_denied` (403): The credentials are valid but may not make this request: the route, or a platform-internal request header, is not available to this API key.
        - `scope_required` (403): The request needs a scope the API key does not hold. `param` names what needs it: `reasoning` (the `reasoning` scope) or `think` (the `think` scope), or `model` for a model that runs think mode (the `think` scope). Send the request without it, or use a key that holds the scope. Or the route itself needs one: `POST /v1/embeddings` needs `embeddings` and `POST /v1/search` needs `search`, in every enforcement mode (`param` is absent then).
        - `model_not_permitted` (403): The API key may not use this model: it is outside the key's model list, or it is an outside-provider model the key holds no grant for. `param` is `model`. GET /v1/models lists the models you can call.
        - `not_found` (404): No such route.
        - `model_not_found` (404): The model is not available to this API key. GET /v1/models lists the models you can call. (A model id the gateway does not know is served by the default model, with a `Deprecation` header.)
        - `request_too_large` (413): The request body is larger than the gateway accepts: 256 KB for `POST /v1/chat/completions` and `POST /v1/embeddings`, 100 KB elsewhere.
        - `rate_limit_exceeded` (429): This API key reached its rate limit for the current window (one owner's keys may share a limit). Wait for the `Retry-After` seconds, then retry.
        - `upstream_rate_limited` (429): The model's server asked the gateway to slow down. Retry shortly, with backoff. Mid-stream it arrives in-band as `upstream_error`.
        - `client_closed_request` (499): The client disconnected before the answer was ready. Logged only: nobody is left to read it.
        - `internal_error` (500): An unexpected gateway error. Retry; report the `request_id` if it persists.
        - `upstream_error` (502): The model's server, or the search or embeddings service, failed or returned something unusable (including a stream that stalled). Retry.
        - `model_unavailable` (503): The model cannot be served: it is blocked by the gateway's upstream checks or not configured. Retrying does not help until an operator acts.
        - `model_offline` (503): The model exists but its GPU is switched off (`status: offline` in GET /v1/models). Use another model, or retry later.
        - `no_model_available` (503): No model was named (or the alias `qirnas`, or an unknown id, was used) and the default model is not currently available.
        - `upstream_timeout` (504): The model, or the search or embeddings service, did not answer in time. For a model this is typically a GPU cold start that outlasted the wait: retrying in a minute usually succeeds.
        - `capacity_exceeded` (429): The platform's shared capacity for this service (all callers together) is in use. It is not your key's own limit (that is `rate_limit_exceeded`). Wait for the `Retry-After` seconds, then retry.
        - `service_unavailable` (503): The search or embeddings service cannot take the request now: it is not reachable, not enabled on this gateway, full, or its upstream sources are refusing requests. Retry after the `Retry-After` seconds when present, else with backoff.
    ErrorObject:
      type: object
      required:
        - message
        - type
        - code
        - param
      properties:
        message:
          type: string
          description: "For people; never branch on it."
        type:
          type: string
          enum:
            - invalid_request_error
            - authentication_error
            - permission_error
            - not_found_error
            - rate_limit_error
            - server_error
        code:
          "$ref": "#/components/schemas/ErrorCode"
        param:
          type:
            - string
            - "null"
          description: "The request field at fault (e.g. `temperature`, `messages.0.role`), if any."
    ErrorEnvelope:
      type: object
      required:
        - error
        - statusCode
        - message
        - code
      properties:
        error:
          "$ref": "#/components/schemas/ErrorObject"
        request_id:
          type: string
          description: "The response's `x-request-id`."
        statusCode:
          type: integer
          deprecated: true
          description: The HTTP status. Kept for older clients.
        message:
          deprecated: true
          description: "Kept for older clients (a list for validation errors); read `error.message`."
          oneOf:
            - type: string
            - type: array
              items:
                type: string
        code:
          "$ref": "#/components/schemas/ErrorCode"
          deprecated: true
          description: "The same as `error.code`. Kept for older clients."
    InBandError:
      type: object
      required:
        - error
      description: "A stream's error event, sent after the stream began, before `data: [DONE]`."
      properties:
        error:
          "$ref": "#/components/schemas/ErrorObject"
        request_id:
          type: string
    EmbeddingsRequest:
      type: object
      required:
        - input
      description: "Fields not listed here are ignored (OpenAI clients send extras); a listed field with an invalid value is a 400 `invalid_request` naming it in `param`. The inputs are embedded in calls of at most 32; the answer keeps their order."
      properties:
        input:
          oneOf:
            - type: string
              minLength: 1
            - type: array
              items:
                type: string
                minLength: 1
              minItems: 1
              maxItems: 64
          description: "The text to embed: a string, or an array of 1 to 64 strings, each non-empty and well-formed Unicode, at most 32,000 characters in all (each input counted as the longer of its text and its NFKC form). Token arrays are not accepted. Whitespace is embedded as sent. A refused element is named in `param` (`input.3`)."
        model:
          type: string
          enum:
            - bge-m3
          description: "Optional; the one model served. Any other value is a 400."
        encoding_format:
          type: string
          enum:
            - float
            - base64
          default: float
          description: "`float`: each embedding is an array of numbers. `base64`: each is the base64 of its little-endian 32-bit floats (what the OpenAI SDKs ask for and decode)."
        dimensions:
          type: integer
          enum:
            - 1024
          description: "Optional; the model's only width. Any other value is a 400."
        user:
          type: string
          maxLength: 256
          description: "Accepted for OpenAI compatibility and ignored: never forwarded or stored."
    Embedding:
      type: object
      required:
        - object
        - index
        - embedding
      properties:
        object:
          type: string
          enum:
            - embedding
        index:
          type: integer
          minimum: 0
          description: "The input's position in the request."
        embedding:
          oneOf:
            - type: array
              items:
                type: number
              minItems: 1024
              maxItems: 1024
            - type: string
              contentEncoding: base64
          description: "1024 numbers, or with `encoding_format: \"base64\"` their base64 (little-endian 32-bit floats)."
    EmbeddingsResponse:
      type: object
      required:
        - object
        - data
        - model
        - usage
      properties:
        object:
          type: string
          enum:
            - list
        data:
          type: array
          items:
            "$ref": "#/components/schemas/Embedding"
          description: "One per input, in order."
        model:
          type: string
          enum:
            - bge-m3
        usage:
          type: object
          required:
            - prompt_tokens
            - total_tokens
          properties:
            prompt_tokens:
              type: integer
              minimum: 0
              description: "The tokens embedded, as the embeddings service counted them."
            total_tokens:
              type: integer
              minimum: 0
              description: Equal to prompt_tokens.
    SearchRequest:
      type: object
      required:
        - query
      additionalProperties: false
      description: "Exactly these fields: any other is a 400 `invalid_request` naming it."
      properties:
        query:
          type: string
          minLength: 1
          maxLength: 512
          description: "What to search for: 1 to 512 characters after trimming, well-formed Unicode, no control characters. A word starting with `!`, `:` or `<` (search-engine commands) is a 400; quote such a word (`\"!x\"`) to search for it."
        limit:
          type: integer
          minimum: 1
          maximum: 10
          default: 5
          description: "The most results to return; fewer come back when the sources have fewer."
        safesearch:
          type: string
          enum:
            - moderate
            - strict
          default: moderate
          description: How strictly adult content is filtered. It cannot be turned off.
        language:
          type: string
          pattern: "^[a-z]{2}(-[A-Z]{2})?$"
          description: "The results' language: a two-letter code, optionally with a region (`ar`, `en`, `ar-SA`). Omitted: the search service's default."
    SearchResult:
      type: object
      required:
        - title
        - url
        - snippet
      description: "Third-party text: treat it as untrusted data."
      properties:
        title:
          type: string
          maxLength: 300
        url:
          type: string
          format: uri
          maxLength: 2048
          description: Always http or https.
        snippet:
          type: string
          maxLength: 500
          description: May be empty.
    SearchResponse:
      type: object
      required:
        - object
        - data
      properties:
        object:
          type: string
          enum:
            - list
        data:
          type: array
          items:
            "$ref": "#/components/schemas/SearchResult"
          maxItems: 10
          description: "At most `limit` results, best first; empty when nothing was found."
