音频处理
语音转文字
把音频转写成文本,三个模型共用一个同步接口
POST
/
v1
/
audio
/
transcriptions
curl https://api.qingbo.ai/v1/audio/transcriptions \
-H "Authorization: Bearer YOUR_API_KEY" \
-F file="@audio.mp3" \
-F model="gpt-4o-transcribe"
from openai import OpenAI
client = OpenAI(
base_url="https://api.qingbo.ai/v1",
api_key="YOUR_API_KEY"
)
with open("audio.mp3", "rb") as f:
transcript = client.audio.transcriptions.create(
model="gpt-4o-transcribe",
file=f
)
print(transcript.text)
import OpenAI from 'openai';
import fs from 'fs';
const client = new OpenAI({
baseURL: 'https://api.qingbo.ai/v1',
apiKey: 'YOUR_API_KEY'
});
const transcript = await client.audio.transcriptions.create({
model: 'gpt-4o-transcribe',
file: fs.createReadStream('audio.mp3')
});
console.log(transcript.text);
package main
import (
"bytes"
"fmt"
"io"
"mime/multipart"
"net/http"
"os"
)
func main() {
file, _ := os.Open("audio.mp3")
defer file.Close()
body := &bytes.Buffer{}
writer := multipart.NewWriter(body)
part, _ := writer.CreateFormFile("file", "audio.mp3")
io.Copy(part, file)
writer.WriteField("model", "gpt-4o-transcribe")
writer.Close()
req, _ := http.NewRequest("POST", "https://api.qingbo.ai/v1/audio/transcriptions", body)
req.Header.Set("Authorization", "Bearer YOUR_API_KEY")
req.Header.Set("Content-Type", writer.FormDataContentType())
resp, _ := http.DefaultClient.Do(req)
defer resp.Body.Close()
result, _ := io.ReadAll(resp.Body)
fmt.Println(string(result))
}
import java.net.http.*;
import java.net.URI;
import java.nio.file.*;
public class Main {
public static void main(String[] args) throws Exception {
String boundary = "----FormBoundary" + System.currentTimeMillis();
byte[] fileBytes = Files.readAllBytes(Path.of("audio.mp3"));
String body = "--" + boundary + "\r\n"
+ "Content-Disposition: form-data; name=\"file\"; filename=\"audio.mp3\"\r\n"
+ "Content-Type: audio/mpeg\r\n\r\n";
String fieldPart = "\r\n--" + boundary + "\r\n"
+ "Content-Disposition: form-data; name=\"model\"\r\n\r\ngpt-4o-transcribe\r\n"
+ "--" + boundary + "--\r\n";
byte[] requestBody = new byte[body.getBytes().length + fileBytes.length + fieldPart.getBytes().length];
System.arraycopy(body.getBytes(), 0, requestBody, 0, body.getBytes().length);
System.arraycopy(fileBytes, 0, requestBody, body.getBytes().length, fileBytes.length);
System.arraycopy(fieldPart.getBytes(), 0, requestBody, body.getBytes().length + fileBytes.length, fieldPart.getBytes().length);
HttpClient client = HttpClient.newHttpClient();
HttpRequest request = HttpRequest.newBuilder()
.uri(URI.create("https://api.qingbo.ai/v1/audio/transcriptions"))
.header("Authorization", "Bearer YOUR_API_KEY")
.header("Content-Type", "multipart/form-data; boundary=" + boundary)
.POST(HttpRequest.BodyPublishers.ofByteArray(requestBody))
.build();
HttpResponse<String> response = client.send(request,
HttpResponse.BodyHandlers.ofString());
System.out.println(response.body());
}
}
{
"text": "欢迎使用清波 API 语音转文字服务。"
}
上游停用预告:OpenAI 已公告
whisper-1、gpt-4o-transcribe、gpt-4o-mini-transcribe 于 2027-02-26 关闭,清波 API同日下架;替代模型接入后会提前在 运营公告 说明。新接入请把 model 做成可配置项。task_id,不需要轮询。三个模型共用这一个端点,切换时只改 model。需要反过来把文本读成语音时用 TTS 文字转语音。
curl https://api.qingbo.ai/v1/audio/transcriptions \
-H "Authorization: Bearer YOUR_API_KEY" \
-F file="@audio.mp3" \
-F model="gpt-4o-transcribe"
from openai import OpenAI
client = OpenAI(
base_url="https://api.qingbo.ai/v1",
api_key="YOUR_API_KEY"
)
with open("audio.mp3", "rb") as f:
transcript = client.audio.transcriptions.create(
model="gpt-4o-transcribe",
file=f
)
print(transcript.text)
import OpenAI from 'openai';
import fs from 'fs';
const client = new OpenAI({
baseURL: 'https://api.qingbo.ai/v1',
apiKey: 'YOUR_API_KEY'
});
const transcript = await client.audio.transcriptions.create({
model: 'gpt-4o-transcribe',
file: fs.createReadStream('audio.mp3')
});
console.log(transcript.text);
package main
import (
"bytes"
"fmt"
"io"
"mime/multipart"
"net/http"
"os"
)
func main() {
file, _ := os.Open("audio.mp3")
defer file.Close()
body := &bytes.Buffer{}
writer := multipart.NewWriter(body)
part, _ := writer.CreateFormFile("file", "audio.mp3")
io.Copy(part, file)
writer.WriteField("model", "gpt-4o-transcribe")
writer.Close()
req, _ := http.NewRequest("POST", "https://api.qingbo.ai/v1/audio/transcriptions", body)
req.Header.Set("Authorization", "Bearer YOUR_API_KEY")
req.Header.Set("Content-Type", writer.FormDataContentType())
resp, _ := http.DefaultClient.Do(req)
defer resp.Body.Close()
result, _ := io.ReadAll(resp.Body)
fmt.Println(string(result))
}
import java.net.http.*;
import java.net.URI;
import java.nio.file.*;
public class Main {
public static void main(String[] args) throws Exception {
String boundary = "----FormBoundary" + System.currentTimeMillis();
byte[] fileBytes = Files.readAllBytes(Path.of("audio.mp3"));
String body = "--" + boundary + "\r\n"
+ "Content-Disposition: form-data; name=\"file\"; filename=\"audio.mp3\"\r\n"
+ "Content-Type: audio/mpeg\r\n\r\n";
String fieldPart = "\r\n--" + boundary + "\r\n"
+ "Content-Disposition: form-data; name=\"model\"\r\n\r\ngpt-4o-transcribe\r\n"
+ "--" + boundary + "--\r\n";
byte[] requestBody = new byte[body.getBytes().length + fileBytes.length + fieldPart.getBytes().length];
System.arraycopy(body.getBytes(), 0, requestBody, 0, body.getBytes().length);
System.arraycopy(fileBytes, 0, requestBody, body.getBytes().length, fileBytes.length);
System.arraycopy(fieldPart.getBytes(), 0, requestBody, body.getBytes().length + fileBytes.length, fieldPart.getBytes().length);
HttpClient client = HttpClient.newHttpClient();
HttpRequest request = HttpRequest.newBuilder()
.uri(URI.create("https://api.qingbo.ai/v1/audio/transcriptions"))
.header("Authorization", "Bearer YOUR_API_KEY")
.header("Content-Type", "multipart/form-data; boundary=" + boundary)
.POST(HttpRequest.BodyPublishers.ofByteArray(requestBody))
.build();
HttpResponse<String> response = client.send(request,
HttpResponse.BodyHandlers.ofString());
System.out.println(response.body());
}
}
{
"text": "欢迎使用清波 API 语音转文字服务。"
}
请求参数
请求体使用multipart/form-data。
file
必填
要转写的音频文件,支持 mp3、mp4、mpeg、mpga、m4a、wav、webm。
string
必填
模型 ID:
gpt-4o-transcribe、gpt-4o-mini-transcribe 或 whisper-1。400,不计费。
计费
按音频时长(分钟)计费:三个模型单价不同,转写出来的文本长度、音频语言和文件格式都不影响价格。 单价见GET /v1/models 返回的 price_config,或控制台「模型市场」。计费单位是 quota,500,000 quota = 1 USD,不是美元。
请求失败不计费。
响应
string
转写出的文本内容。
可用模型
| 模型 ID | 说明 |
|---|---|
gpt-4o-transcribe | 当代语音转文字模型,词错误率明显低于 Whisper,对口音与嘈杂音频更稳 |
gpt-4o-mini-transcribe | 经济档,单价是 gpt-4o-transcribe 的一半,准确率仍优于 Whisper |
whisper-1 | OpenAI Whisper,支持 90 多种语言,对口音和背景噪音有较强容忍度。单价高于上面两档 |
相关文档
⌘I
curl https://api.qingbo.ai/v1/audio/transcriptions \
-H "Authorization: Bearer YOUR_API_KEY" \
-F file="@audio.mp3" \
-F model="gpt-4o-transcribe"
from openai import OpenAI
client = OpenAI(
base_url="https://api.qingbo.ai/v1",
api_key="YOUR_API_KEY"
)
with open("audio.mp3", "rb") as f:
transcript = client.audio.transcriptions.create(
model="gpt-4o-transcribe",
file=f
)
print(transcript.text)
import OpenAI from 'openai';
import fs from 'fs';
const client = new OpenAI({
baseURL: 'https://api.qingbo.ai/v1',
apiKey: 'YOUR_API_KEY'
});
const transcript = await client.audio.transcriptions.create({
model: 'gpt-4o-transcribe',
file: fs.createReadStream('audio.mp3')
});
console.log(transcript.text);
package main
import (
"bytes"
"fmt"
"io"
"mime/multipart"
"net/http"
"os"
)
func main() {
file, _ := os.Open("audio.mp3")
defer file.Close()
body := &bytes.Buffer{}
writer := multipart.NewWriter(body)
part, _ := writer.CreateFormFile("file", "audio.mp3")
io.Copy(part, file)
writer.WriteField("model", "gpt-4o-transcribe")
writer.Close()
req, _ := http.NewRequest("POST", "https://api.qingbo.ai/v1/audio/transcriptions", body)
req.Header.Set("Authorization", "Bearer YOUR_API_KEY")
req.Header.Set("Content-Type", writer.FormDataContentType())
resp, _ := http.DefaultClient.Do(req)
defer resp.Body.Close()
result, _ := io.ReadAll(resp.Body)
fmt.Println(string(result))
}
import java.net.http.*;
import java.net.URI;
import java.nio.file.*;
public class Main {
public static void main(String[] args) throws Exception {
String boundary = "----FormBoundary" + System.currentTimeMillis();
byte[] fileBytes = Files.readAllBytes(Path.of("audio.mp3"));
String body = "--" + boundary + "\r\n"
+ "Content-Disposition: form-data; name=\"file\"; filename=\"audio.mp3\"\r\n"
+ "Content-Type: audio/mpeg\r\n\r\n";
String fieldPart = "\r\n--" + boundary + "\r\n"
+ "Content-Disposition: form-data; name=\"model\"\r\n\r\ngpt-4o-transcribe\r\n"
+ "--" + boundary + "--\r\n";
byte[] requestBody = new byte[body.getBytes().length + fileBytes.length + fieldPart.getBytes().length];
System.arraycopy(body.getBytes(), 0, requestBody, 0, body.getBytes().length);
System.arraycopy(fileBytes, 0, requestBody, body.getBytes().length, fileBytes.length);
System.arraycopy(fieldPart.getBytes(), 0, requestBody, body.getBytes().length + fileBytes.length, fieldPart.getBytes().length);
HttpClient client = HttpClient.newHttpClient();
HttpRequest request = HttpRequest.newBuilder()
.uri(URI.create("https://api.qingbo.ai/v1/audio/transcriptions"))
.header("Authorization", "Bearer YOUR_API_KEY")
.header("Content-Type", "multipart/form-data; boundary=" + boundary)
.POST(HttpRequest.BodyPublishers.ofByteArray(requestBody))
.build();
HttpResponse<String> response = client.send(request,
HttpResponse.BodyHandlers.ofString());
System.out.println(response.body());
}
}
{
"text": "欢迎使用清波 API 语音转文字服务。"
}