{
  "info": {
    "description": "Offline speech recognition. This description is generated from the dialects this server has mounted and lists every endpoint it serves; request and response schemas are deliberately absent rather than hand-written and wrong. The prose at /docs is the reference.",
    "title": "NanoASR",
    "version": "v1.1.1"
  },
  "openapi": "3.1.0",
  "paths": {
    "/": {
      "get": {
        "description": "Served without a credential.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Redirects to /docs.",
        "tags": [
          "server"
        ]
      }
    },
    "/api/v1/catalog": {
      "get": {
        "description": "",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "List the models available for download.",
        "tags": [
          "native"
        ]
      }
    },
    "/api/v1/config": {
      "get": {
        "description": "Requires an administrative API key.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "The effective configuration, with secrets redacted.",
        "tags": [
          "native"
        ]
      }
    },
    "/api/v1/jobs": {
      "get": {
        "description": "Scoped to the calling key unless the key is administrative. Paged by cursor.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "List this key's jobs.",
        "tags": [
          "native"
        ]
      },
      "post": {
        "description": "Takes the same fields as transcribe, plus webhook_url for a signed delivery on completion. The upload is held on disk until the job reaches a terminal state and is then deleted.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Queue the same work and return a job.",
        "tags": [
          "native"
        ]
      }
    },
    "/api/v1/jobs/{id}": {
      "delete": {
        "description": "A diarization pass already under way cannot be interrupted and runs to its end.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Cancel a queued or running job.",
        "tags": [
          "native"
        ]
      },
      "get": {
        "description": "",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Fetch one job and its result.",
        "tags": [
          "native"
        ]
      }
    },
    "/api/v1/jobs/{id}/events": {
      "get": {
        "description": "text/event-stream of numbered job states. Last-Event-ID resumes rather than replays. A job that has already finished yields one catch-up event and closes.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Follow a job as server-sent events.",
        "tags": [
          "native"
        ]
      }
    },
    "/api/v1/models": {
      "get": {
        "description": "",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "List installed models and what is resident.",
        "tags": [
          "native"
        ]
      }
    },
    "/api/v1/models/{id}/download": {
      "post": {
        "description": "Requires an administrative API key.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Download a catalog model, streaming progress as SSE.",
        "tags": [
          "native"
        ]
      }
    },
    "/api/v1/models/{id}/load": {
      "post": {
        "description": "Requires an administrative API key.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Load a model into memory now.",
        "tags": [
          "native"
        ]
      }
    },
    "/api/v1/models/{id}/pin": {
      "post": {
        "description": "Requires an administrative API key.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Keep a model resident, or stop keeping it.",
        "tags": [
          "native"
        ]
      }
    },
    "/api/v1/models/{id}/reload": {
      "post": {
        "description": "The new instance is loaded and warmed before the pointer moves. Requires an administrative API key.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Swap in another revision without dropping requests.",
        "tags": [
          "native"
        ]
      }
    },
    "/api/v1/models/{id}/unload": {
      "post": {
        "description": "Requires an administrative API key.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Release a model.",
        "tags": [
          "native"
        ]
      }
    },
    "/api/v1/transcribe": {
      "post": {
        "description": "multipart/form-data with file, plus model, language, channel_mode, diarize, num_speakers, punctuate, itn, hotwords, decoding_method and word_timestamps. Cancelling the request stops the decode between batches rather than at the end of the file.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Transcribe an uploaded file and wait for the result.",
        "tags": [
          "native"
        ]
      }
    },
    "/asr": {
      "post": {
        "description": "multipart with audio_file, plus output (txt, json, srt, vtt, tsv), task, language, word_timestamps and encode. task=translate is refused rather than answered with a transcription.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Transcribe and return the transcript as a file.",
        "tags": [
          "era"
        ]
      }
    },
    "/asr_task": {
      "post": {
        "description": "",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Queue the same work and return a task id.",
        "tags": [
          "era"
        ]
      }
    },
    "/asr_task/{task_id}": {
      "get": {
        "description": "",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Poll a queued task.",
        "tags": [
          "era"
        ]
      }
    },
    "/detect-language": {
      "post": {
        "description": "Answered from the model's declared languages, not from an acoustic language identifier: these are monolingual models, so the honest answer is what the model was trained on.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Report the language of the audio.",
        "tags": [
          "era"
        ]
      }
    },
    "/docs": {
      "get": {
        "description": "Served without a credential.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "This page.",
        "tags": [
          "server"
        ]
      }
    },
    "/docs/openapi.json": {
      "get": {
        "description": "Served without a credential.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "The same endpoint list, as OpenAPI.",
        "tags": [
          "server"
        ]
      }
    },
    "/healthz": {
      "get": {
        "description": "Served without a credential.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Liveness, with the native library versions.",
        "tags": [
          "server"
        ]
      }
    },
    "/readyz": {
      "get": {
        "description": "503 when the queue is full, which is when a load balancer should stop sending work here. Served without a credential.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Readiness, and the queue depth.",
        "tags": [
          "server"
        ]
      }
    },
    "/ui": {
      "get": {
        "description": "Served without a credential because a browser does not send a bearer token when it loads a script tag. The SPA discovers whether the API needs a key from the first 401 it gets. Served without a credential.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "The test UI.",
        "tags": [
          "server"
        ]
      }
    },
    "/v1/audio/transcriptions": {
      "post": {
        "description": "multipart/form-data with file, and optionally model, language, response_format (json, verbose_json, text, srt, vtt) and repeated timestamp_granularities[] (word, segment). prompt is applied as a comma-separated hotword list rather than as an LM prompt, and temperature is accepted and ignored because the decoders do not sample; both are reported back as warnings. Asking for word timings from a model that cannot produce them yields segment timings and a warning, or 422 with X-NanoASR-Strict: 1.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Transcribe an uploaded file.",
        "tags": [
          "openai"
        ]
      }
    },
    "/v1/audio/translations": {
      "post": {
        "description": "Answers 501. Translation needs a model that writes a language it did not hear.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Not implemented: this build ships transcription models only.",
        "tags": [
          "openai"
        ]
      }
    },
    "/v1/models": {
      "get": {
        "description": "Transcription models only: the supporting models (VAD, punctuation, diarization) and the streaming models are left out, because passing one here would fail. Each entry carries the NanoASR state, languages and whether it produces word timestamps.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "List the models that can be passed as `model`.",
        "tags": [
          "openai"
        ]
      }
    },
    "/v1/models/{id}": {
      "get": {
        "description": "",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Describe one model.",
        "tags": [
          "openai"
        ]
      }
    },
    "/v1/realtime": {
      "get": {
        "description": "Client events: session.update (or transcription_session.update), input_audio_buffer.append with base64 audio, .commit and .clear. A binary frame is accepted as raw audio in the declared format, which saves base64's third of overhead. Server events: transcription_session.created and .updated, input_audio_buffer.speech_started, .speech_stopped, .committed, .cleared, conversation.item.created, conversation.item.input_audio_transcription.delta and .completed, error, and nanoasr.warning for a parameter that was accepted and ignored. Every delta also carries the full hypothesis in nanoasr.text, because a streaming decoder can retract a word it already sent and no sequence of deltas expresses that. input_audio_format takes pcm16 (24 kHz by default), g711_ulaw, g711_alaw, or an object {\"type\":\"audio/pcm\",\"rate\":16000}. The model, the decoding method and the endpoint timings belong to the server and cannot be chosen per session; prompt is not applied. Authenticate with Authorization: Bearer, or from a browser with the subprotocol openai-insecure-api-key.\u003ckey\u003e. Called with the query string intent=transcription.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Open a streaming recognition session (websocket upgrade).",
        "tags": [
          "realtime"
        ]
      }
    },
    "/v1/realtime/transcription_sessions": {
      "post": {
        "description": "Answers 501 and says how to connect with an ordinary API key instead.",
        "responses": {
          "default": {
            "description": "See the description."
          }
        },
        "summary": "Not implemented: this server mints no ephemeral tokens.",
        "tags": [
          "realtime"
        ]
      }
    }
  }
}
