{
  "openapi": "3.1.0",
  "info": {
    "title": "Quorum API",
    "version": "1.0.0",
    "summary": "Route a question through a panel of frontier models and get one deliberated answer.",
    "description": "Quorum's HTTP API. `POST /v1/chat/completions` is OpenAI-shaped, so most SDKs work by changing the base URL and the model name — the `model` field selects a Quorum **mode** (a configured panel) rather than a single model. An additive `quorum` block on every response carries what the OpenAI shape has no vocabulary for: which engines actually answered, how many rounds ran, whether they converged, and what it cost.\n\nThis document is written from the handler source, not from memory. Where the API does not do something — token streaming, for one — it is marked here rather than left implied.\n\nThree things that surprise people:\n\n1. **Requests carrying a browser `Origin` header are rejected with 403.** This API is server-to-server. A real key presented from a browser is treated as a leaked key, not a legitimate call.\n2. **`stream: true` returns 400, not a stream.** Provider token streaming is not implemented; failing loudly beats returning a plain body to a client that is waiting for SSE.\n3. **A deliberation is slow by design.** Measured on production 2026-08-20: p50 28.5s, p90 88.2s, p99 229.3s across 3,845 deliberations. Set your client timeout accordingly, and use `POST /v1/estimate` — free — to decide whether a question is worth the wait before spending it.",
    "contact": { "name": "Quorum", "url": "https://www.quorum.dog/help" },
    "license": { "name": "Proprietary", "url": "https://www.quorum.dog/terms.html" }
  },
  "servers": [
    { "url": "https://www.quorum.dog", "description": "Production" }
  ],
  "externalDocs": {
    "description": "The library",
    "url": "https://www.quorum.dog/docs"
  },
  "tags": [
    { "name": "Deliberation", "description": "Convene a panel, or price one first." },
    { "name": "Discovery", "description": "What this key may call." },
    { "name": "Receipts", "description": "What actually happened on a past call." },
    { "name": "Testing", "description": "Measure a mode against a benchmark set. Long-running." }
  ],
  "security": [{ "ApiKeyAuth": [] }],
  "paths": {
    "/v1/chat/completions": {
      "post": {
        "operationId": "createChatCompletion",
        "tags": ["Deliberation"],
        "summary": "Convene the panel",
        "description": "Runs a real deliberation and bills for it. Seconds, not milliseconds — see the timing note on this document.\n\nSend an `Idempotency-Key` header on anything a retry could duplicate. A repeat of a succeeded key replays the original answer with `quorum.replayed: true`; it is not re-run and not re-billed. A repeat of a key still in flight returns 409.",
        "parameters": [
          {
            "name": "Idempotency-Key",
            "in": "header",
            "required": false,
            "schema": { "type": "string" },
            "description": "Deduplicates retries. If omitted, one is derived from org, key, model and messages — so an identical repeat is deduplicated whether you asked for it or not."
          }
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": { "$ref": "#/components/schemas/ChatCompletionRequest" },
              "examples": {
                "basic": {
                  "summary": "A question worth a panel",
                  "value": {
                    "model": "quorum-standard",
                    "messages": [
                      { "role": "user", "content": "We are choosing between Postgres row-level security and application-layer authorisation for a multi-tenant SaaS. Which, and what breaks either way?" }
                    ]
                  }
                }
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "The panel answered.",
            "content": {
              "application/json": { "schema": { "$ref": "#/components/schemas/ChatCompletion" } }
            }
          },
          "400": { "$ref": "#/components/responses/BadRequest" },
          "401": { "$ref": "#/components/responses/Unauthorized" },
          "403": { "$ref": "#/components/responses/Forbidden" },
          "404": { "$ref": "#/components/responses/ModelNotFound" },
          "409": {
            "description": "A request with this `Idempotency-Key` is already running. Wait for it rather than retrying.",
            "content": { "application/json": { "schema": { "$ref": "#/components/schemas/Error" } } }
          },
          "413": {
            "description": "The prompt exceeds this key's `max_input_tokens` (8,000 unless raised).",
            "content": { "application/json": { "schema": { "$ref": "#/components/schemas/Error" } } }
          },
          "429": { "$ref": "#/components/responses/RateLimited" },
          "500": { "$ref": "#/components/responses/ServerError" },
          "503": { "$ref": "#/components/responses/Unavailable" }
        }
      }
    },
    "/v1/estimate": {
      "post": {
        "operationId": "estimate",
        "tags": ["Deliberation"],
        "summary": "Price a question without running it",
        "description": "Classifies the prompt — depth, difficulty, task type — and returns what a real call would cost. Always free: `billed_usd` is `0`.\n\nIt runs the same classifier a real deliberation runs, so the classification is the one that would actually apply, not a second implementation of it. Measured at roughly 0.9 s and about $0.00006 of real cost per call, which is why it is free — but it is still authenticated and still rate-limited, because tiny multiplied by unlimited automated volume is not tiny.\n\nThis is the front half of the escalation pattern: classify everything, deliberate only what earns it. Where that threshold sits is your decision, not ours.",
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": { "$ref": "#/components/schemas/EstimateRequest" }
            }
          }
        },
        "responses": {
          "200": {
            "description": "Classification and projected price.",
            "content": { "application/json": { "schema": { "$ref": "#/components/schemas/Estimate" } } }
          },
          "400": { "$ref": "#/components/responses/BadRequest" },
          "401": { "$ref": "#/components/responses/Unauthorized" },
          "403": { "$ref": "#/components/responses/Forbidden" },
          "404": { "$ref": "#/components/responses/ModelNotFound" },
          "429": { "$ref": "#/components/responses/RateLimited" },
          "500": { "$ref": "#/components/responses/ServerError" },
          "503": { "$ref": "#/components/responses/Unavailable" }
        }
      }
    },
    "/v1/models": {
      "get": {
        "operationId": "listModels",
        "tags": ["Discovery"],
        "summary": "List the modes this key may call",
        "description": "Scoped to what your organisation is actually entitled to, not every mode that exists.\n\nEach entry discloses the provider families the mode is designed to fan out to, before you send it any data. That is the pre-call half of subprocessor transparency; `quorum.engines` on a completion response is the after-the-fact half, and the two can differ when a seat falls back.\n\nRead `pricing.q_surcharge_by_depth` when it is present — it is what most calls are billed at. `q_surcharge_usd` is the flat fallback and ceiling.",
        "responses": {
          "200": {
            "description": "The modes available to this key.",
            "content": { "application/json": { "schema": { "$ref": "#/components/schemas/ModelList" } } }
          },
          "401": { "$ref": "#/components/responses/Unauthorized" },
          "403": { "$ref": "#/components/responses/Forbidden" },
          "429": { "$ref": "#/components/responses/RateLimited" },
          "500": { "$ref": "#/components/responses/ServerError" },
          "503": { "$ref": "#/components/responses/Unavailable" }
        }
      }
    },
    "/v1/receipts/{request_id}": {
      "get": {
        "operationId": "getReceipt",
        "tags": ["Receipts"],
        "summary": "What happened on a past call",
        "description": "Per-seat models, judge scores, whether a seat fell back to a different engine, latency and cost for a completed request.\n\nEvery other endpoint makes Quorum do more; this one makes the *caller* able to do more. An agent that can see judge scores and convergence can escalate, re-ask, or flag for human review — decisions it cannot make from a bare answer string.\n\nScoped to your organisation in the lookup itself: another org's `request_id` returns 404, never a partial leak.",
        "parameters": [
          {
            "name": "request_id",
            "in": "path",
            "required": true,
            "schema": { "type": "string" },
            "description": "From `quorum.request_id` on a completion response."
          },
          {
            "name": "include_transcript",
            "in": "query",
            "required": false,
            "schema": { "type": "boolean", "default": false },
            "description": "Include the full text each seat produced. Off by default — a full transcript is a labelled multi-model comparison dataset, not something handed over merely because a caller knows a request id."
          }
        ],
        "responses": {
          "200": {
            "description": "The receipt, or an explanation of why there is none.",
            "content": { "application/json": { "schema": { "$ref": "#/components/schemas/ReceiptResponse" } } }
          },
          "400": { "$ref": "#/components/responses/BadRequest" },
          "401": { "$ref": "#/components/responses/Unauthorized" },
          "403": { "$ref": "#/components/responses/Forbidden" },
          "404": {
            "description": "No receipt for this `request_id` under this organisation.",
            "content": { "application/json": { "schema": { "$ref": "#/components/schemas/Error" } } }
          },
          "429": { "$ref": "#/components/responses/RateLimited" },
          "500": { "$ref": "#/components/responses/ServerError" },
          "503": { "$ref": "#/components/responses/Unavailable" }
        }
      }
    },
    "/v1/tests": {
      "post": {
        "operationId": "createTestRun",
        "tags": ["Testing"],
        "summary": "Measure a mode against a benchmark set",
        "description": "Runs a mode over questions drawn from a real benchmark dataset and has a blind multi-judge panel score the results. Billed per question at 25% off standard pricing, for the run as configured.\n\n**This is not a request-response call.** The shared execution engine processes roughly one question per minute per batch, so a 60-question run takes about an hour. You get a `batch_id` and a `status_url` immediately; poll it. That constraint is real and disclosed rather than hidden behind a spinner.\n\nSend `dry_run: true` to get the full price and duration back without starting anything or billing anything.",
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": { "$ref": "#/components/schemas/TestRunRequest" },
              "examples": {
                "dryRun": {
                  "summary": "Price it first",
                  "value": { "model": "quorum-standard", "buckets": { "light": 10, "medium": 10, "hard": 10 }, "dry_run": true }
                }
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "Launched, or priced when `dry_run` was set.",
            "content": { "application/json": { "schema": { "$ref": "#/components/schemas/TestRunAccepted" } } }
          },
          "400": { "$ref": "#/components/responses/BadRequest" },
          "401": { "$ref": "#/components/responses/Unauthorized" },
          "403": { "$ref": "#/components/responses/Forbidden" },
          "404": { "$ref": "#/components/responses/ModelNotFound" },
          "429": { "$ref": "#/components/responses/RateLimited" },
          "500": { "$ref": "#/components/responses/ServerError" },
          "503": { "$ref": "#/components/responses/Unavailable" }
        }
      },
      "get": {
        "operationId": "getTestRun",
        "tags": ["Testing"],
        "summary": "Progress of a test run",
        "parameters": [
          { "name": "batch_id", "in": "query", "required": true, "schema": { "type": "string", "format": "uuid" } }
        ],
        "responses": {
          "200": {
            "description": "Batch progress and, once complete, results.",
            "content": { "application/json": { "schema": { "type": "object", "properties": { "ok": { "type": "boolean" }, "batch": { "type": "object", "additionalProperties": true } } } } }
          },
          "401": { "$ref": "#/components/responses/Unauthorized" },
          "403": { "$ref": "#/components/responses/Forbidden" },
          "404": { "description": "No such batch under this organisation.", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/Error" } } } },
          "429": { "$ref": "#/components/responses/RateLimited" }
        }
      }
    },
    "/v1/certify": {
      "post": {
        "operationId": "createCertification",
        "tags": ["Testing"],
        "summary": "Certify a mode — fixed shape, $50",
        "description": "A fixed 150-question run — roughly 50 each of light, medium and hard — scored by three judges at two repeats each, ending in a certification verdict. $50 flat.\n\nSame execution engine as `/v1/tests` and the same throughput: about 150 minutes, not instant. Poll `status_url`.\n\nThis is **not** the Marketplace badge. That one requires a published listing and moderation. This certifies a mode your organisation runs privately, with no listing involved.",
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "type": "object",
                "required": ["model"],
                "properties": {
                  "model": { "type": "string", "description": "An `api_model_id` from /v1/models.", "examples": ["quorum-standard"] }
                }
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "Certification run launched.",
            "content": { "application/json": { "schema": { "$ref": "#/components/schemas/CertificationAccepted" } } }
          },
          "400": { "$ref": "#/components/responses/BadRequest" },
          "401": { "$ref": "#/components/responses/Unauthorized" },
          "403": { "$ref": "#/components/responses/Forbidden" },
          "404": { "$ref": "#/components/responses/ModelNotFound" },
          "429": { "$ref": "#/components/responses/RateLimited" },
          "500": { "$ref": "#/components/responses/ServerError" }
        }
      },
      "get": {
        "operationId": "getCertification",
        "tags": ["Testing"],
        "summary": "Progress of a certification run",
        "parameters": [
          { "name": "batch_id", "in": "query", "required": true, "schema": { "type": "string", "format": "uuid" } }
        ],
        "responses": {
          "200": {
            "description": "Batch progress, plus a `certification` verdict once complete.",
            "content": { "application/json": { "schema": { "type": "object", "properties": { "ok": { "type": "boolean" }, "batch": { "type": "object", "additionalProperties": true }, "certification": { "type": "object", "additionalProperties": true } } } } }
          },
          "401": { "$ref": "#/components/responses/Unauthorized" },
          "403": { "$ref": "#/components/responses/Forbidden" },
          "404": { "description": "No such batch under this organisation.", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/Error" } } } },
          "429": { "$ref": "#/components/responses/RateLimited" }
        }
      }
    }
  },
  "components": {
    "securitySchemes": {
      "ApiKeyAuth": {
        "type": "http",
        "scheme": "bearer",
        "description": "`Authorization: Bearer qk_live_…` (or `qk_test_…`). Server-to-server only — a request carrying a browser `Origin` header is rejected with 403 `browser_origin_not_allowed`, because a real key sent from a browser is a leaked key."
      }
    },
    "schemas": {
      "Message": {
        "type": "object",
        "required": ["role", "content"],
        "properties": {
          "role": { "type": "string", "enum": ["system", "user", "assistant"] },
          "content": { "type": "string" }
        }
      },
      "ChatCompletionRequest": {
        "type": "object",
        "required": ["model", "messages"],
        "properties": {
          "model": {
            "type": "string",
            "description": "An `api_model_id` from /v1/models — a mode, not a single model.",
            "examples": ["quorum-standard"]
          },
          "messages": {
            "type": "array",
            "minItems": 1,
            "items": { "$ref": "#/components/schemas/Message" }
          },
          "stream": {
            "type": "boolean",
            "default": false,
            "description": "Only `false` is accepted. `true` returns 400 `stream_not_supported` — provider token streaming is not implemented, and a silent non-stream would break an SDK waiting for SSE."
          },
          "quorum": {
            "type": "object",
            "description": "Optional per-call overrides, within what the mode and your key permit.",
            "additionalProperties": true
          }
        }
      },
      "ChatCompletion": {
        "type": "object",
        "description": "OpenAI's shape, plus an additive `quorum` block. Nothing in the OpenAI-shaped part was renamed.",
        "properties": {
          "id": { "type": "string" },
          "object": { "type": "string", "const": "chat.completion" },
          "created": { "type": "integer", "description": "Unix seconds." },
          "model": { "type": "string" },
          "choices": {
            "type": "array",
            "items": {
              "type": "object",
              "properties": {
                "index": { "type": "integer" },
                "message": { "$ref": "#/components/schemas/Message" },
                "finish_reason": {
                  "type": "string",
                  "enum": ["stop", "cost_cap", "length"],
                  "description": "`stop` means the answer is complete. `cost_cap` means the run stopped at the mode's cost ceiling — the answer is real, but shorter deliberation than the mode would otherwise have run; treat it as a signal, not an error. `length` ADDED 2026-09-17, ADDITIVE: the answer was CUT OFF by the model's output limit and is incomplete — the same value api.openai.com uses, so an SDK already handles it. It is MEASURED, never inferred: it is set from the field the provider itself used to report the cut (Google `MAX_TOKENS`, Bedrock and Anthropic `max_tokens`, OpenAI and Groq `length`), never from the answer's length or its shape. It describes THE CALL THAT PRODUCED THE TEXT YOU RECEIVED — the synthesis on a panel run, the winning seat when no synthesis ran — so a losing seat that was cut does not set it. `length` OUTRANKS `cost_cap` when both apply: a cost cap is a deliberate stop with a real answer, a truncation is an accident with an unusable one, and you need the second. A run that was also capped still records that on its receipt. ON AN IDEMPOTENT REPLAY (`replayed: true`) this value is READ FROM THE STORED REQUEST, so a retry reports what the first call reported — that was not true before 2026-09-17, when a replay always said `stop`. A replayed request made before that date recorded no truncation measurement and reports `stop` as this endpoint's default rather than as a verified fact; it says so in `quorum.note` and omits `quorum.answer_truncated` entirely, which is how you tell the two apart."
                }
              }
            }
          },
          "usage": {
            "type": "object",
            "properties": {
              "prompt_tokens": { "type": "integer" },
              "completion_tokens": { "type": "integer" },
              "total_tokens": { "type": "integer" }
            }
          },
          "quorum": {
            "type": "object",
            "description": "What the OpenAI shape has no field for.",
            "properties": {
              "request_id": { "type": "string", "description": "Pass to /v1/receipts/{request_id}." },
              "mode_key": { "type": "string" },
              "mode_config_hash": { "type": "string", "description": "Short hash of the mode's engine and deliberation config. A change here means the panel itself changed, as opposed to ordinary prompt-to-prompt variance." },
              "rounds": { "type": "integer", "description": "How many deliberation rounds actually ran. Most calls resolve in one; a second round means the panel disagreed enough to warrant it." },
              "seats": { "type": "integer", "description": "How many seats CONVENED for this request. Unchanged: it counts every seat the panel sat, including one that then produced nothing." },
              "seats_answered": { "type": ["integer", "null"], "description": "How many of those seats actually produced an answer. ADDED 2026-09-17, additive only. `seats_answered < seats` means the deliberation you paid panel price for was smaller than the panel that convened — a three-engine run can quietly become two, or none, and until now nothing in the response said so. A seat whose first attempt was cut and then RETRIED still counts here: the retry answered, so the seat contributed, and the discarded attempt is wasted cost rather than a missing panellist. `null` only when the run reported no seat array at all, which is not the same claim as zero. ON AN IDEMPOTENT REPLAY (`replayed: true`) this field is read from the stored request and is ABSENT — not null — when that request predates 2026-09-17 and never recorded it; absence there means \"not recorded\", while `0` still means no seat answered." },
              "answer_truncated": { "type": "boolean", "description": "ADDED 2026-09-17, ADDITIVE. Present ONLY on an idempotent replay (`replayed: true`), where it is the truncation fact read from the stored request: `true` means the answer you are being handed was cut off by the model's output limit, `false` means it was not. It is the same measurement that set `finish_reason` on the original call, so a retry reports what the first call reported instead of a hardcoded `stop`. On a first (non-replayed) call the equivalent fact is `finish_reason` itself, which is why the field does not appear there. IT IS ABSENT WHEN THE STORED REQUEST NEVER RECORDED IT — requests made before 2026-09-17 could not measure it — and absence is deliberate: those runs are unknown, not known-clean, and publishing `false` would state a measurement nobody took. A caller that must be certain the answer is whole should branch on `'answer_truncated' in response.quorum` before trusting a replayed `finish_reason` of `stop`." },
              "converged": { "type": "boolean", "description": "Whether the seats agreed. `false` is information, not failure — it is the signal that the question is genuinely contested." },
              "capped": { "type": "boolean" },
              "surcharge_usd": { "type": "number" },
              "billed_usd": { "type": "number" },
              "latency_ms": { "type": "integer" },
              "engines": { "type": "array", "items": { "type": "string" }, "description": "The provider families that actually answered. May differ from the mode's declared design when a seat fell back." },
              "persisted": { "type": "boolean" },
              "replayed": { "type": "boolean", "description": "Present and `true` only on an idempotent replay: the original answer, not re-run and not re-billed." }
            }
          }
        }
      },
      "EstimateRequest": {
        "type": "object",
        "required": ["model", "messages"],
        "properties": {
          "model": { "type": "string", "examples": ["quorum-standard"] },
          "messages": { "type": "array", "minItems": 1, "items": { "$ref": "#/components/schemas/Message" } }
        }
      },
      "Estimate": {
        "type": "object",
        "properties": {
          "request_id": { "type": "string" },
          "model": { "type": "string" },
          "mode_key": { "type": "string" },
          "depth": { "type": "string", "enum": ["light", "medium", "deep"], "description": "Drives which per-depth surcharge applies." },
          "difficulty_score": { "type": "number", "description": "0–1. Production distribution as of 2026-08-20 (n=3,947): 45% light, 33% medium, 22% deep; mean 0.456." },
          "task_type": { "type": "string" },
          "estimated_price_usd": { "type": "number" },
          "billed_usd": { "type": "number", "const": 0, "description": "Always zero. This endpoint is free." },
          "timing_ms": {
            "type": "object",
            "description": "Returned so you can measure this round trip yourself rather than taking our word for it.",
            "properties": {
              "total": { "type": "integer" },
              "classify": { "type": "integer" }
            }
          }
        }
      },
      "ModelList": {
        "type": "object",
        "properties": {
          "object": { "type": "string", "const": "list" },
          "data": { "type": "array", "items": { "$ref": "#/components/schemas/Model" } }
        }
      },
      "Model": {
        "type": "object",
        "properties": {
          "id": { "type": "string", "description": "Use this as `model` on other endpoints." },
          "object": { "type": "string", "const": "model" },
          "owned_by": { "type": "string", "const": "quorum" },
          "quorum": {
            "type": "object",
            "properties": {
              "mode_key": { "type": "string" },
              "name": { "type": "string" },
              "description": { "type": ["string", "null"] },
              "price_tier": { "type": ["string", "null"] },
              "pricing": {
                "type": "object",
                "properties": {
                  "q_surcharge_usd": { "type": ["number", "null"], "description": "Flat fallback and ceiling." },
                  "q_surcharge_by_depth": { "type": ["object", "null"], "additionalProperties": { "type": "number" }, "description": "What most calls are actually billed at. Read this in preference to the flat value when present." },
                  "q_surcharge_express_usd": { "type": ["number", "null"] }
                }
              },
              "providers": { "type": "array", "items": { "type": "string" }, "description": "Provider families this mode is designed to use, disclosed before you send data. Real per-call assignment can differ under fallback." }
            }
          }
        }
      },
      "ReceiptResponse": {
        "type": "object",
        "properties": {
          "ok": { "type": "boolean" },
          "receipt_available": { "type": "boolean", "description": "Present and `false` when the request never produced a deliberation — it failed, or was capped before firing. `reason` says which." },
          "reason": { "type": "string" },
          "receipt": { "$ref": "#/components/schemas/Receipt" }
        }
      },
      "Receipt": {
        "type": "object",
        "properties": {
          "request_id": { "type": "string" },
          "mode_key": { "type": "string" },
          "mode_config_hash": { "type": "string" },
          "status": { "type": "string" },
          "billed_usd": { "type": "number", "description": "What this call cost you. BREAKING CHANGE 2026-08-26: this schema and this endpoint also carried a second money field reporting Quorum's own cost of running the call. It was published in error, it is cost-of-goods data, and it has been removed. `billed_usd` is unchanged and is the only per-call money figure the receipt returns; nothing else in the receipt changed." },
          "token_charge_usd": { "type": ["number", "null"], "description": "The metered pass-through half of `billed_usd`: inference you consumed, priced at the published per-million-token rate. ADDED 2026-09-08, additive only -- `billed_usd` is unchanged. `token_charge_usd + q_surcharge_usd` equals `billed_usd` on a normal call; when `status` is `capped` the charge was held at the disclosed ceiling and the two components will sum to more than you were billed. Both are reported as recorded, never back-derived from the total. `null` on calls made before this split was recorded, which is not the same as $0.00. This is what YOU were charged for tokens; Quorum's own cost of serving the call is not part of this API." },
          "q_surcharge_usd": { "type": ["number", "null"], "description": "The Quorum fee half of `billed_usd` -- what the deliberation itself costs, over and above the tokens. Set per Mode, and may vary with depth or express routing on the same Mode. ADDED 2026-09-08, additive only. `null` on calls made before this split was recorded." },
          "prompt_tokens": { "type": ["integer", "null"], "description": "Input tokens metered on this request, as recorded when it closed. ADDED 2026-09-13, additive only. This is the REQUEST-LEVEL counter and it is deliberately not the sum of `seats[].tokens_in`: the judge, the synthesis and the classifier consumed tokens on the same request and sit on no seat, so adding the array up returns less than was metered. `null` means no count was recorded, which is not the same claim as zero tokens. This is your own consumption, not a cost figure -- what serving the call cost Quorum is not part of this API." },
          "completion_tokens": { "type": ["integer", "null"], "description": "Output tokens metered on this request, as recorded when it closed. ADDED 2026-09-13, additive only. Same request-level scope and the same null-is-not-zero rule as `prompt_tokens`." },
          "rounds": { "type": ["integer", "null"] },
          "seat_count": { "type": ["integer", "null"], "description": "How many seats this request convened. ADDED 2026-09-13, additive only. RECORDED WHEN THE REQUEST WAS MADE, not counted from the `seats` array below -- and you want this one rather than `seats.length`, because that array carries one row per seat PER ROUND. A three-seat panel that ran two rounds returns four seat rows, so its length is four and this field is three. Counting the array is what once made a receipt claim a fourth panellist who never sat. `null` on a request whose seat count was not recorded." },
          "lane": { "type": ["string", "null"], "enum": ["express", "panel", null], "description": "Which lane answered this call. ADDED 2026-09-10, additive only. `panel` means several engines answered independently and a judge read them. `express` means one engine answered on its own, with no judge and no synthesis pass -- that is a deliberate route for easy questions, not a degraded panel and not an error. USUALLY INFERRED, SOMETIMES EVIDENCED, and the distinction is published rather than glossed. One stored field can name the lane positively: a deliberation that took the express route records `best_seat_judged_by` as `express_lane` on the winning seat's row. WHEN THAT MARKER IS PRESENT IT IS TRUSTED, because only the express route writes it. WHEN IT IS ABSENT THE LANE IS DERIVED from the rows the deliberation wrote -- one seat, no judge, no synthesis, and no evidence that a second engine was dispatched -- because the marker is incomplete: it is only recorded from 2026-09-08, and only on calls that stored a best seat, so its absence is not a statement that the express route did not run. Two consequences of the derived half, which you should know before you build on it. (1) A Mode configured with a single seat reports `express` even though the express route never ran, because the two are indistinguishable in the record. (2) A panel whose other seats failed without leaving a trace also reports `express`. A marked call has neither ambiguity. `null` means the lane could not be determined -- typically a receipt whose transcript could not be read, or a call older than the rows this is derived from. Treat a derived lane as a strong hint, not as billing-grade provenance. ONE FURTHER SCOPE NOTE, true of every field on this receipt: where a request_id belongs to a conversation that continued, the receipt describes ONE turn of it -- the latest -- and every field agrees on which. It never mixes two turns." },
          "difficulty_score": { "type": ["number", "null"] },
          "task_type": { "type": ["string", "null"] },
          "hallucination_risk": { "type": ["string", "null"] },
          "minority_insight_likely": { "type": ["boolean", "null"] },
          "best_seat": {
            "type": ["object", "null"],
            "properties": {
              "model_used": { "type": "string" },
              "judge_score": { "type": ["number", "null"] },
              "content": { "type": "string", "description": "Only with `include_transcript=true`." }
            }
          },
          "seats": {
            "type": "array",
            "items": {
              "type": "object",
              "properties": {
                "role": { "type": ["string", "null"], "description": "Which seat this row is: `seat1`, `seat2` or `seat3`. ADDED 2026-09-13, additive only. THIS IS THE ONLY FIELD THAT IDENTIFIES A SEAT, and you need it because this array holds one row per seat PER ROUND -- the same seat answering in round 2 is a second row. Numbering the array by position therefore invents panellists, and `model_used` cannot substitute: two seats may hold the same engine, and a seat that fell back changes engine mid-run. `null` on rows written before this field was published." },
                "model_used": { "type": "string" },
                "round_number": { "type": ["integer", "null"] },
                "judge_score": { "type": ["number", "null"] },
                "is_best_seat": { "type": ["boolean", "null"] },
                "fallback_used": { "type": ["boolean", "null"], "description": "True when the intended engine was unavailable and another took the seat." },
                "original_model": { "type": ["string", "null"], "description": "What was meant to sit there, when `fallback_used`." },
                "tokens_in": { "type": ["integer", "null"], "description": "Input tokens this seat consumed on its own provider call. ADDED 2026-09-08. Per seat, never a running total -- sum the array yourself if you want the deliberation's input. `null` means no usage was recorded for that seat, which is not the same as zero. Seats only: the judge and the synthesis are not in this array." },
                "tokens_out": { "type": ["integer", "null"], "description": "Output tokens this seat produced on its own provider call. ADDED 2026-09-08. Where a model reasons before answering, the provider bills that thinking as output and it is already inside this number. Per seat, never a running total; `null` means not recorded." },
                "started_at": { "type": ["string", "null"], "format": "date-time", "description": "Wall-clock instant this seat's provider call began. ADDED 2026-09-08, additive only. Use it with `ended_at` to place the seats on one timeline: seats sharing a `started_at` were dispatched together, which `latency_ms` alone cannot tell you. `null` on calls made before this was recorded (2026-09-03)." },
                "ended_at": { "type": ["string", "null"], "format": "date-time", "description": "Wall-clock instant this seat's provider call finished, taken after the response body was read. ADDED 2026-09-08, additive only. `ended_at - started_at` can EXCEED `latency_ms`: latency times only the final attempt, while this window spans every attempt including a generation that was discarded and re-run. Reported as recorded and never derived -- a seat with a `started_at` and a null `ended_at` means the end was not recorded, not that it ended at `started_at + latency_ms`." },
                "latency_ms": { "type": ["integer", "null"] },
                "content": { "type": "string", "description": "Only with `include_transcript=true`." }
              }
            }
          },
          "staff": {
            "type": "object",
            "description": "The engines that worked on your answer WITHOUT sitting in a seat: the judge that read the seats, the synthesis that wrote the text you received, and the grounding search. ADDED 2026-09-10, additive only -- nothing already in this schema changed. Seats are in `seats`; these are not seats and are deliberately not mixed into that array, so a judge is never counted as one more opinion. Which engine ran and how long it took, and nothing else: no verdict text, no synthesis draft, and no cost figure.",
            "properties": {
              "judge": {
                "type": ["object", "null"],
                "description": "The engine that scored the seats and picked the best answer. `null` when no judge ran at all -- which is the normal case on the `express` lane, where there is a single answer and nothing to compare it against. Not an error.",
                "properties": {
                  "model_used": { "type": ["string", "null"], "description": "The engine that judged. `null` when the judge call was made but the engine was not recorded -- notably a judge call that ran and failed, where naming a model by inference would be a guess. `null` here does not mean no judge ran; the parent object being `null` means that." },
                  "latency_ms": { "type": ["integer", "null"], "description": "TOTAL time the judge spent on this call, summed across every round it read. A round is one pass in which the seats answer, and the judge reading that pass is part of the round, so a multi-round deliberation has the judge running more than once and this is the sum, not one call. `null` when no duration was recorded, which is not the same as zero." },
                  "fallback_used": { "type": "boolean", "description": "True when the Mode's configured judge failed and its backup judged instead." },
                  "original_model": { "type": ["string", "null"], "description": "The configured judge engine, when `fallback_used`." }
                }
              },
              "synthesis": {
                "type": ["object", "null"],
                "description": "The engine that combined the seats into the single answer you received. `null` when nothing was synthesised -- on the `express` lane one engine answers and its text is shipped as written, and where the seats already agreed there may be nothing to combine. The synthesis is not a deliberation round and is never counted in `rounds`.",
                "properties": {
                  "model_used": { "type": ["string", "null"], "description": "The engine that wrote the synthesis. `null` when a synthesis ran but its engine was not recorded; older calls stored a placeholder instead of a model name and that is reported as `null` rather than passed through as if it were an engine." },
                  "latency_ms": { "type": ["integer", "null"], "description": "How long the synthesis took, including any fallback attempt. `null` when not recorded." },
                  "fallback_used": { "type": "boolean", "description": "True when the configured synthesis engine was unavailable and another wrote the synthesis." },
                  "original_model": { "type": ["string", "null"], "description": "The configured synthesis engine, when `fallback_used`." }
                }
              },
              "grounding": {
                "type": "object",
                "description": "Whether a web search was run before the seats answered, and by which engine. ALWAYS PRESENT, so that `fired: false` is a statement rather than a gap.",
                "properties": {
                  "fired": { "type": "boolean", "description": "Whether grounding ran. `false` IS THE ORDINARY ANSWER: grounding is gated by the classifier and most questions do not need a web search, so `false` means it was not called for on this question. It is not a failure, an outage or a missing capability, and it should not be surfaced as one." },
                  "model_used": { "type": ["string", "null"], "description": "The search engine that grounded the answer, when `fired` is true. `null` whenever `fired` is false, and also when a search ran without its source being recorded. No duration is reported: grounding is dispatched without being waited on, so no meaningful per-call duration exists to publish." }
                }
              }
            }
          },
          "latency_ms": { "type": ["integer", "null"] },
          "created_at": { "type": "string", "format": "date-time" },
          "completed_at": { "type": ["string", "null"], "format": "date-time" }
        }
      },
      "TestRunRequest": {
        "type": "object",
        "required": ["model", "buckets"],
        "properties": {
          "model": { "type": "string", "examples": ["quorum-standard"] },
          "buckets": {
            "type": "object",
            "description": "Questions per difficulty bucket. Each 1–100; 300 total across all buckets.",
            "properties": {
              "light": { "type": "integer", "minimum": 1, "maximum": 100 },
              "medium": { "type": "integer", "minimum": 1, "maximum": 100 },
              "hard": { "type": "integer", "minimum": 1, "maximum": 100 }
            },
            "additionalProperties": false
          },
          "judges": {
            "type": "array",
            "items": { "type": "string", "enum": ["nova_pro", "mistral_large", "llama4_maverick"] },
            "description": "Defaults to all three. Judging is blind, anonymised and order-randomised."
          },
          "judge_repeats": { "type": "integer", "minimum": 1, "maximum": 3, "default": 1, "description": "Repeat scoring to measure judge consistency." },
          "judging_protocol": { "type": "string", "enum": ["pairwise", "plurality"], "default": "pairwise" },
          "dry_run": { "type": "boolean", "default": false, "description": "Price and duration only. Nothing runs, nothing is billed." }
        }
      },
      "TestRunAccepted": {
        "type": "object",
        "properties": {
          "ok": { "type": "boolean" },
          "dry_run": { "type": "boolean" },
          "batch_id": { "type": "string", "format": "uuid" },
          "mode_key": { "type": "string" },
          "total_questions": { "type": "integer" },
          "per_bucket": { "type": "object", "additionalProperties": true },
          "total_cost_usd": { "type": "number", "description": "On a dry run." },
          "billed_usd": { "type": "number", "description": "On a real run. Charged up front for the run as configured." },
          "estimated_minutes": { "type": "integer" },
          "status_url": { "type": "string" },
          "note": { "type": "string" }
        }
      },
      "CertificationAccepted": {
        "type": "object",
        "properties": {
          "ok": { "type": "boolean" },
          "batch_id": { "type": "string", "format": "uuid" },
          "model": { "type": "string" },
          "mode_key": { "type": "string" },
          "total_questions": { "type": "integer", "const": 150 },
          "billed_usd": { "type": "number", "const": 50 },
          "status_url": { "type": "string" },
          "note": { "type": "string" }
        }
      },
      "Error": {
        "type": "object",
        "description": "OpenAI's error envelope, plus `quorum.request_id` where one exists — so a failure is as traceable as a success.",
        "properties": {
          "error": {
            "type": "object",
            "properties": {
              "message": { "type": "string" },
              "type": {
                "type": "string",
                "enum": ["invalid_request_error", "authentication_error", "permission_error", "rate_limit_error", "api_error"]
              },
              "code": {
                "type": "string",
                "description": "The stable identifier. Branch on this, not on `message`.",
                "enum": [
                  "invalid_api_key",
                  "revoked_api_key",
                  "org_suspended",
                  "browser_origin_not_allowed",
                  "insufficient_scope",
                  "invalid_model",
                  "invalid_messages",
                  "model_not_found",
                  "stream_not_supported",
                  "context_length_exceeded",
                  "request_in_progress",
                  "missing_request_id",
                  "receipt_not_found",
                  "invalid_buckets",
                  "invalid_bucket",
                  "invalid_bucket_count",
                  "too_many_questions",
                  "invalid_judges",
                  "method_not_allowed",
                  "concurrency_limit_exceeded",
                  "rate_limit_exceeded",
                  "service_unavailable",
                  "internal_error"
                ]
              },
              "param": { "type": ["string", "null"] }
            }
          },
          "quorum": {
            "type": "object",
            "properties": { "request_id": { "type": ["string", "null"] } }
          }
        }
      }
    },
    "responses": {
      "BadRequest": {
        "description": "The request was malformed. `error.code` says how.",
        "content": { "application/json": { "schema": { "$ref": "#/components/schemas/Error" } } }
      },
      "Unauthorized": {
        "description": "Missing, invalid or revoked key. Not retryable — fix the key.",
        "content": { "application/json": { "schema": { "$ref": "#/components/schemas/Error" } } }
      },
      "Forbidden": {
        "description": "The key is real but not permitted: organisation suspended, key lacks the scope, or the request carried a browser `Origin` header.",
        "content": { "application/json": { "schema": { "$ref": "#/components/schemas/Error" } } }
      },
      "ModelNotFound": {
        "description": "No mode with that `api_model_id` is exposed to this organisation. Call /v1/models to see what is.",
        "content": { "application/json": { "schema": { "$ref": "#/components/schemas/Error" } } }
      },
      "RateLimited": {
        "description": "Rate or concurrency ceiling hit. Defaults are 60 requests/minute per key and 5 concurrent requests per organisation. A `Retry-After` header is sent; back off and retry.",
        "headers": {
          "Retry-After": { "schema": { "type": "integer" }, "description": "Seconds." }
        },
        "content": { "application/json": { "schema": { "$ref": "#/components/schemas/Error" } } }
      },
      "ServerError": {
        "description": "Our fault. Safe to retry with the same `Idempotency-Key`.",
        "content": { "application/json": { "schema": { "$ref": "#/components/schemas/Error" } } }
      },
      "Unavailable": {
        "description": "A dependency was unreachable and the gate failed closed rather than letting the call through unchecked. Retryable.",
        "content": { "application/json": { "schema": { "$ref": "#/components/schemas/Error" } } }
      }
    }
  }
}
