feat(media): transcribe audio via Whisper API before reaching agent

Move audio transcription from agent-driven (exec + local whisper) to
Manager-layer processing via OpenAI Whisper API. Voice messages are
now transcribed automatically before the agent sees them, so the
agent only receives text. Local whisper skill remains as fallback
when API key is not configured. Also changed default model from
turbo to base for faster first-time experience.

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
Naiyuan Qing
2026-02-09 10:06:11 +08:00
parent 922a3b2bb7
commit 4e5780692e
3 changed files with 89 additions and 29 deletions

View File

@@ -1,10 +1,9 @@
---
name: Audio Transcription
description: Transcribe audio files using OpenAI Whisper CLI
description: Transcribe audio files using local Whisper CLI (fallback when API is unavailable)
version: 1.0.0
metadata:
emoji: "🎙️"
always: true
requires:
anyBins:
- whisper
@@ -16,11 +15,6 @@ metadata:
bins: [whisper]
label: "Install OpenAI Whisper via Homebrew"
os: [darwin]
- id: pip-whisper
kind: uv
package: openai-whisper
bins: [whisper]
label: "Install OpenAI Whisper via pip/uv"
tags:
- audio
- transcription
@@ -29,26 +23,14 @@ userInvocable: false
disableModelInvocation: false
---
## Audio Transcription
## Audio Transcription (Local Fallback)
When you receive a message indicating an audio or voice message file (e.g., `[audio message received]` with a `File:` path), you should transcribe it.
Voice messages from channels are normally transcribed automatically via the OpenAI Whisper API before reaching you. This skill is only needed when the API is unavailable.
### How to Transcribe
Run the following command using the `exec` tool:
If you receive `[audio message received]` with a `File:` path (instead of `[Voice Message]` with a transcript), it means the API transcription was not available. Use local whisper to transcribe:
```
whisper "<file_path>" --model turbo --output_format txt --output_dir /tmp
whisper "<file_path>" --model base --output_format txt --output_dir /tmp
```
Then read the resulting `.txt` file (same name as input, in `/tmp/`) to get the transcript.
### Response Format
After transcription, respond naturally based on the transcribed content. If the user said something in the voice message, respond to it as if they had typed it.
If transcription fails, let the user know and suggest they check their Whisper installation.
### Supported Formats
Whisper supports: mp3, mp4, mpeg, mpga, m4a, wav, webm, ogg, oga, flac
Then read the `.txt` file from `/tmp/` and respond based on the transcribed content.

View File

@@ -20,6 +20,7 @@ import { listChannels } from "./registry.js";
import { loadChannelsConfig } from "./config.js";
import { MessageAggregator, DEFAULT_CHUNKER_CONFIG } from "../hub/message-aggregator.js";
import type { AsyncAgent } from "../agent/async-agent.js";
import { transcribeAudio } from "../media/transcribe.js";
interface AccountHandle {
channelId: string;
@@ -275,7 +276,7 @@ export class ChannelManager {
}
}
/** Download media file and forward to agent */
/** Download media file, process it, and forward result to agent */
private async routeMedia(
plugin: ChannelPlugin,
accountId: string,
@@ -294,8 +295,23 @@ export class ChannelManager {
const mimeType = media.mimeType ?? "image/jpeg";
const caption = media.caption || "User sent an image.";
agent.writeWithImages(caption, [{ type: "image", data: base64, mimeType }]);
} else if (media.type === "audio") {
// Audio: transcribe via Whisper API before reaching agent
const transcript = await transcribeAudio(filePath);
if (transcript) {
const parts = ["[Voice Message]", `Transcript: ${transcript}`];
if (media.caption) parts.push(`Caption: ${media.caption}`);
agent.write(parts.join("\n"));
} else {
// No API key configured — fall back to file path
const parts = ["[audio message received]", `File: ${filePath}`];
if (media.mimeType) parts.push(`Type: ${media.mimeType}`);
if (media.duration) parts.push(`Duration: ${media.duration}s`);
if (media.caption) parts.push(`Caption: ${media.caption}`);
agent.write(parts.join("\n"));
}
} else {
// Audio/video/document: tell agent the file path, let it handle via skills
// Video/document: tell agent the file path
const parts: string[] = [];
parts.push(`[${media.type} message received]`);
parts.push(`File: ${filePath}`);
@@ -306,9 +322,8 @@ export class ChannelManager {
}
} catch (err) {
const msg = err instanceof Error ? err.message : String(err);
console.error(`[Channels] Failed to download media: ${msg}`);
// Fallback: send text-only if download fails
agent.write(message.text || `[Failed to download ${media.type}]`);
console.error(`[Channels] Failed to process media: ${msg}`);
agent.write(message.text || `[Failed to process ${media.type}]`);
}
}

63
src/media/transcribe.ts Normal file
View File

@@ -0,0 +1,63 @@
/**
* Audio transcription via OpenAI Whisper API.
*
* Called by ChannelManager before the message reaches the Agent,
* so the Agent only ever sees text.
*/
import { readFile } from "node:fs/promises";
import { basename } from "node:path";
import { credentialManager } from "../agent/credentials.js";
/**
* Transcribe an audio file using OpenAI Whisper API.
*
* @param filePath - Local path to the audio file
* @returns Transcribed text, or null if no API key configured
*/
export async function transcribeAudio(filePath: string): Promise<string | null> {
const config = credentialManager.getLlmProviderConfig("openai");
const apiKey = config?.apiKey;
if (!apiKey) return null;
const fileBuffer = await readFile(filePath);
const fileName = basename(filePath);
// Build multipart form data manually (no external dependency)
const boundary = `----FormBoundary${Date.now()}`;
const parts: Buffer[] = [];
// file field
parts.push(Buffer.from(
`--${boundary}\r\nContent-Disposition: form-data; name="file"; filename="${fileName}"\r\nContent-Type: application/octet-stream\r\n\r\n`,
));
parts.push(fileBuffer);
parts.push(Buffer.from("\r\n"));
// model field
parts.push(Buffer.from(
`--${boundary}\r\nContent-Disposition: form-data; name="model"\r\n\r\nwhisper-1\r\n`,
));
// closing boundary
parts.push(Buffer.from(`--${boundary}--\r\n`));
const body = Buffer.concat(parts);
const res = await fetch("https://api.openai.com/v1/audio/transcriptions", {
method: "POST",
headers: {
Authorization: `Bearer ${apiKey}`,
"Content-Type": `multipart/form-data; boundary=${boundary}`,
},
body,
});
if (!res.ok) {
const errText = await res.text().catch(() => "");
throw new Error(`Whisper API error: HTTP ${res.status} ${errText}`);
}
const result = (await res.json()) as { text: string };
return result.text;
}