{
  "openapi": "3.1.0",
  "info": {
    "title": "Create transcription API",
    "version": "1.0.0"
  },
  "servers": [
    {
      "url": "https://api.cometapi.com"
    }
  ],
  "security": [
    {
      "bearerAuth": []
    }
  ],
  "paths": {
    "/v1/audio/transcriptions": {
      "post": {
        "summary": "Create transcription",
        "operationId": "create_transcription",
        "requestBody": {
          "required": true,
          "content": {
            "multipart/form-data": {
              "schema": {
                "type": "object",
                "properties": {
                  "file": {
                    "format": "binary",
                    "type": "string",
                    "description": "The audio file to transcribe. Supported formats: flac, mp3, mp4, mpeg, mpga, m4a, ogg, wav, webm."
                  },
                  "model": {
                    "type": "string",
                    "description": "The speech-to-text model to use. Choose a current speech model from the [Models page](/overview/models).",
                    "default": "whisper-1"
                  },
                  "language": {
                    "type": "string",
                    "description": "The language of the input audio in ISO-639-1 format (e.g., `en`, `zh`, `ja`). Supplying the language improves accuracy and latency."
                  },
                  "prompt": {
                    "type": "string",
                    "description": "Optional text to guide the model's style or continue a previous audio segment. The prompt should match the audio language."
                  },
                  "response_format": {
                    "type": "string",
                    "description": "The output format for the transcription.",
                    "enum": [
                      "json",
                      "text",
                      "srt",
                      "verbose_json",
                      "vtt"
                    ],
                    "default": "json"
                  },
                  "temperature": {
                    "type": "number",
                    "description": "Sampling temperature between 0 and 1. Higher values produce more random output; lower values are more focused. When set to 0, the model auto-adjusts temperature using log probability.",
                    "minimum": 0,
                    "maximum": 1,
                    "default": 0
                  }
                },
                "required": [
                  "file",
                  "model"
                ]
              }
            }
          }
        },
        "responses": {
          "200": {
            "description": "The transcription result.",
            "content": {
              "application/json": {
                "schema": {
                  "type": "object",
                  "required": [
                    "text"
                  ],
                  "properties": {
                    "text": {
                      "type": "string",
                      "description": "The transcribed text."
                    }
                  }
                },
                "examples": {
                  "Default": {
                    "summary": "Transcription result",
                    "value": {
                      "text": "Hello, welcome to CometAPI."
                    }
                  }
                }
              }
            }
          }
        },
        "x-codeSamples": [
          {
            "lang": "python",
            "label": "Create transcription",
            "source": "import os\nfrom openai import OpenAI\n\nclient = OpenAI(\n    api_key=os.environ[\"COMETAPI_KEY\"],\n    base_url=\"https://api.cometapi.com/v1\"\n)\n\naudio_file = open(\"audio.mp3\", \"rb\")\ntranscription = client.audio.transcriptions.create(\n    model=\"whisper-1\",\n    file=audio_file\n)\nprint(transcription.text)"
          },
          {
            "lang": "javascript",
            "label": "Create transcription",
            "source": "import OpenAI from \"openai\";\nimport fs from \"fs\";\n\nconst client = new OpenAI({\n  apiKey: process.env.COMETAPI_KEY,\n  baseURL: \"https://api.cometapi.com/v1\"\n});\n\nconst transcription = await client.audio.transcriptions.create({\n  model: \"whisper-1\",\n  file: fs.createReadStream(\"audio.mp3\")\n});\nconsole.log(transcription.text);"
          },
          {
            "lang": "shell",
            "label": "Create transcription",
            "source": "curl -X POST https://api.cometapi.com/v1/audio/transcriptions \\\n  -H \"Authorization: Bearer $COMETAPI_KEY\" \\\n  -F model=\"whisper-1\" \\\n  -F file=\"@audio.mp3\""
          }
        ]
      }
    }
  },
  "components": {
    "securitySchemes": {
      "bearerAuth": {
        "type": "http",
        "scheme": "bearer",
        "description": "Bearer token authentication. Use your CometAPI key."
      }
    }
  }
}
