Audio
Speech to Text
Transcribe audio into text, on three models behind one synchronous endpoint
POST
/
v1
/
audio
/
transcriptions
curl https://api.qingbo.ai/v1/audio/transcriptions \
-H "Authorization: Bearer YOUR_API_KEY" \
-F file="@audio.mp3" \
-F model="gpt-4o-transcribe"
from openai import OpenAI
client = OpenAI(
base_url="https://api.qingbo.ai/v1",
api_key="YOUR_API_KEY"
)
with open("audio.mp3", "rb") as f:
transcript = client.audio.transcriptions.create(
model="gpt-4o-transcribe",
file=f
)
print(transcript.text)
import OpenAI from 'openai';
import fs from 'fs';
const client = new OpenAI({
baseURL: 'https://api.qingbo.ai/v1',
apiKey: 'YOUR_API_KEY'
});
const transcript = await client.audio.transcriptions.create({
model: 'gpt-4o-transcribe',
file: fs.createReadStream('audio.mp3')
});
console.log(transcript.text);
package main
import (
"bytes"
"fmt"
"io"
"mime/multipart"
"net/http"
"os"
)
func main() {
file, _ := os.Open("audio.mp3")
defer file.Close()
body := &bytes.Buffer{}
writer := multipart.NewWriter(body)
part, _ := writer.CreateFormFile("file", "audio.mp3")
io.Copy(part, file)
writer.WriteField("model", "gpt-4o-transcribe")
writer.Close()
req, _ := http.NewRequest("POST", "https://api.qingbo.ai/v1/audio/transcriptions", body)
req.Header.Set("Authorization", "Bearer YOUR_API_KEY")
req.Header.Set("Content-Type", writer.FormDataContentType())
resp, _ := http.DefaultClient.Do(req)
defer resp.Body.Close()
result, _ := io.ReadAll(resp.Body)
fmt.Println(string(result))
}
import java.net.http.*;
import java.net.URI;
import java.nio.file.*;
public class Main {
public static void main(String[] args) throws Exception {
String boundary = "----FormBoundary" + System.currentTimeMillis();
byte[] fileBytes = Files.readAllBytes(Path.of("audio.mp3"));
String body = "--" + boundary + "\r\n"
+ "Content-Disposition: form-data; name=\"file\"; filename=\"audio.mp3\"\r\n"
+ "Content-Type: audio/mpeg\r\n\r\n";
String fieldPart = "\r\n--" + boundary + "\r\n"
+ "Content-Disposition: form-data; name=\"model\"\r\n\r\ngpt-4o-transcribe\r\n"
+ "--" + boundary + "--\r\n";
byte[] requestBody = new byte[body.getBytes().length + fileBytes.length + fieldPart.getBytes().length];
System.arraycopy(body.getBytes(), 0, requestBody, 0, body.getBytes().length);
System.arraycopy(fileBytes, 0, requestBody, body.getBytes().length, fileBytes.length);
System.arraycopy(fieldPart.getBytes(), 0, requestBody, body.getBytes().length + fileBytes.length, fieldPart.getBytes().length);
HttpClient client = HttpClient.newHttpClient();
HttpRequest request = HttpRequest.newBuilder()
.uri(URI.create("https://api.qingbo.ai/v1/audio/transcriptions"))
.header("Authorization", "Bearer YOUR_API_KEY")
.header("Content-Type", "multipart/form-data; boundary=" + boundary)
.POST(HttpRequest.BodyPublishers.ofByteArray(requestBody))
.build();
HttpResponse<String> response = client.send(request,
HttpResponse.BodyHandlers.ofString());
System.out.println(response.body());
}
}
{
"text": "Welcome to the QWave speech-to-text API."
}
Upstream shutdown notice: OpenAI has announced that
whisper-1, gpt-4o-transcribe and gpt-4o-mini-transcribe shut down on 2027-02-26; they are delisted here the same day. Replacement models will be announced in Service Notices ahead of time. Keep model configurable in new integrations.task_id and no polling. All three models share this endpoint — switching between them means changing model only. For the opposite direction, reading text aloud, use Text-to-Speech.
curl https://api.qingbo.ai/v1/audio/transcriptions \
-H "Authorization: Bearer YOUR_API_KEY" \
-F file="@audio.mp3" \
-F model="gpt-4o-transcribe"
from openai import OpenAI
client = OpenAI(
base_url="https://api.qingbo.ai/v1",
api_key="YOUR_API_KEY"
)
with open("audio.mp3", "rb") as f:
transcript = client.audio.transcriptions.create(
model="gpt-4o-transcribe",
file=f
)
print(transcript.text)
import OpenAI from 'openai';
import fs from 'fs';
const client = new OpenAI({
baseURL: 'https://api.qingbo.ai/v1',
apiKey: 'YOUR_API_KEY'
});
const transcript = await client.audio.transcriptions.create({
model: 'gpt-4o-transcribe',
file: fs.createReadStream('audio.mp3')
});
console.log(transcript.text);
package main
import (
"bytes"
"fmt"
"io"
"mime/multipart"
"net/http"
"os"
)
func main() {
file, _ := os.Open("audio.mp3")
defer file.Close()
body := &bytes.Buffer{}
writer := multipart.NewWriter(body)
part, _ := writer.CreateFormFile("file", "audio.mp3")
io.Copy(part, file)
writer.WriteField("model", "gpt-4o-transcribe")
writer.Close()
req, _ := http.NewRequest("POST", "https://api.qingbo.ai/v1/audio/transcriptions", body)
req.Header.Set("Authorization", "Bearer YOUR_API_KEY")
req.Header.Set("Content-Type", writer.FormDataContentType())
resp, _ := http.DefaultClient.Do(req)
defer resp.Body.Close()
result, _ := io.ReadAll(resp.Body)
fmt.Println(string(result))
}
import java.net.http.*;
import java.net.URI;
import java.nio.file.*;
public class Main {
public static void main(String[] args) throws Exception {
String boundary = "----FormBoundary" + System.currentTimeMillis();
byte[] fileBytes = Files.readAllBytes(Path.of("audio.mp3"));
String body = "--" + boundary + "\r\n"
+ "Content-Disposition: form-data; name=\"file\"; filename=\"audio.mp3\"\r\n"
+ "Content-Type: audio/mpeg\r\n\r\n";
String fieldPart = "\r\n--" + boundary + "\r\n"
+ "Content-Disposition: form-data; name=\"model\"\r\n\r\ngpt-4o-transcribe\r\n"
+ "--" + boundary + "--\r\n";
byte[] requestBody = new byte[body.getBytes().length + fileBytes.length + fieldPart.getBytes().length];
System.arraycopy(body.getBytes(), 0, requestBody, 0, body.getBytes().length);
System.arraycopy(fileBytes, 0, requestBody, body.getBytes().length, fileBytes.length);
System.arraycopy(fieldPart.getBytes(), 0, requestBody, body.getBytes().length + fileBytes.length, fieldPart.getBytes().length);
HttpClient client = HttpClient.newHttpClient();
HttpRequest request = HttpRequest.newBuilder()
.uri(URI.create("https://api.qingbo.ai/v1/audio/transcriptions"))
.header("Authorization", "Bearer YOUR_API_KEY")
.header("Content-Type", "multipart/form-data; boundary=" + boundary)
.POST(HttpRequest.BodyPublishers.ofByteArray(requestBody))
.build();
HttpResponse<String> response = client.send(request,
HttpResponse.BodyHandlers.ofString());
System.out.println(response.body());
}
}
{
"text": "Welcome to the QWave speech-to-text API."
}
Request parameters
The request body usesmultipart/form-data.
file
required
The audio file to transcribe. Supported formats: mp3, mp4, mpeg, mpga, m4a, wav, webm.
string
required
Model ID:
gpt-4o-transcribe, gpt-4o-mini-transcribe or whisper-1.400 and is not billed.
Pricing
Billed by audio length in minutes. The three models have different rates; transcript length, spoken language and file format make no difference to the price. Rates are inprice_config on GET /v1/models and in the console’s Model Market. The billing unit is quota, at 500,000 quota = 1 USD, not dollars.
A failed request is not billed.
Response
string
The transcribed text.
Available models
| Model ID | Description |
|---|---|
gpt-4o-transcribe | Current-generation speech-to-text, with a markedly lower word-error rate than Whisper and more robust handling of accents and noisy audio |
gpt-4o-mini-transcribe | Budget tier at half the rate of gpt-4o-transcribe, still ahead of Whisper on accuracy |
whisper-1 | OpenAI Whisper, 90+ languages, tolerant of accents and background noise. Priced above both tiers above |
Related
⌘I
curl https://api.qingbo.ai/v1/audio/transcriptions \
-H "Authorization: Bearer YOUR_API_KEY" \
-F file="@audio.mp3" \
-F model="gpt-4o-transcribe"
from openai import OpenAI
client = OpenAI(
base_url="https://api.qingbo.ai/v1",
api_key="YOUR_API_KEY"
)
with open("audio.mp3", "rb") as f:
transcript = client.audio.transcriptions.create(
model="gpt-4o-transcribe",
file=f
)
print(transcript.text)
import OpenAI from 'openai';
import fs from 'fs';
const client = new OpenAI({
baseURL: 'https://api.qingbo.ai/v1',
apiKey: 'YOUR_API_KEY'
});
const transcript = await client.audio.transcriptions.create({
model: 'gpt-4o-transcribe',
file: fs.createReadStream('audio.mp3')
});
console.log(transcript.text);
package main
import (
"bytes"
"fmt"
"io"
"mime/multipart"
"net/http"
"os"
)
func main() {
file, _ := os.Open("audio.mp3")
defer file.Close()
body := &bytes.Buffer{}
writer := multipart.NewWriter(body)
part, _ := writer.CreateFormFile("file", "audio.mp3")
io.Copy(part, file)
writer.WriteField("model", "gpt-4o-transcribe")
writer.Close()
req, _ := http.NewRequest("POST", "https://api.qingbo.ai/v1/audio/transcriptions", body)
req.Header.Set("Authorization", "Bearer YOUR_API_KEY")
req.Header.Set("Content-Type", writer.FormDataContentType())
resp, _ := http.DefaultClient.Do(req)
defer resp.Body.Close()
result, _ := io.ReadAll(resp.Body)
fmt.Println(string(result))
}
import java.net.http.*;
import java.net.URI;
import java.nio.file.*;
public class Main {
public static void main(String[] args) throws Exception {
String boundary = "----FormBoundary" + System.currentTimeMillis();
byte[] fileBytes = Files.readAllBytes(Path.of("audio.mp3"));
String body = "--" + boundary + "\r\n"
+ "Content-Disposition: form-data; name=\"file\"; filename=\"audio.mp3\"\r\n"
+ "Content-Type: audio/mpeg\r\n\r\n";
String fieldPart = "\r\n--" + boundary + "\r\n"
+ "Content-Disposition: form-data; name=\"model\"\r\n\r\ngpt-4o-transcribe\r\n"
+ "--" + boundary + "--\r\n";
byte[] requestBody = new byte[body.getBytes().length + fileBytes.length + fieldPart.getBytes().length];
System.arraycopy(body.getBytes(), 0, requestBody, 0, body.getBytes().length);
System.arraycopy(fileBytes, 0, requestBody, body.getBytes().length, fileBytes.length);
System.arraycopy(fieldPart.getBytes(), 0, requestBody, body.getBytes().length + fileBytes.length, fieldPart.getBytes().length);
HttpClient client = HttpClient.newHttpClient();
HttpRequest request = HttpRequest.newBuilder()
.uri(URI.create("https://api.qingbo.ai/v1/audio/transcriptions"))
.header("Authorization", "Bearer YOUR_API_KEY")
.header("Content-Type", "multipart/form-data; boundary=" + boundary)
.POST(HttpRequest.BodyPublishers.ofByteArray(requestBody))
.build();
HttpResponse<String> response = client.send(request,
HttpResponse.BodyHandlers.ofString());
System.out.println(response.body());
}
}
{
"text": "Welcome to the QWave speech-to-text API."
}