feat: activate gemini voice tts path

This commit is contained in:
Yahya Alhinai
2026-06-28 02:11:02 +00:00
parent 4726e8ce80
commit 89893110f1
11 changed files with 151 additions and 61 deletions
+3 -1
View File
@@ -7,8 +7,10 @@ LIVEKIT_API_SECRET=
# --- Gemini (vision + event detection + voice) --- # --- Gemini (vision + event detection + voice) ---
GEMINI_API_KEY= GEMINI_API_KEY=
# GOOGLE_API_KEY= also works as a local alias, but GEMINI_API_KEY is the
# canonical deployment secret name used by DigitalOcean and docs.
GEMINI_VISION_MODEL=gemini-2.0-flash GEMINI_VISION_MODEL=gemini-2.0-flash
GEMINI_LIVE_MODEL=gemini-live-2.5-flash GEMINI_LIVE_MODEL=gemini-3.1-flash-tts-preview
# --- GitHub (repo state + sync PR artifacts) --- # --- GitHub (repo state + sync PR artifacts) ---
GITHUB_TOKEN= GITHUB_TOKEN=
+1 -1
View File
@@ -83,6 +83,6 @@ export class PodMan {
reliable: true, reliable: true,
topic: DATA_TOPIC, topic: DATA_TOPIC,
}); });
await speak(this.room, message); // gemini-3.1-flash-live voice into the room await speak(this.room, message); // Gemini voice audio into the room
} }
} }
+9 -2
View File
@@ -5,6 +5,13 @@ function req(name: string): string {
if (!v) throw new Error(`Missing required env var: ${name}`); if (!v) throw new Error(`Missing required env var: ${name}`);
return v; return v;
} }
function reqAny(primary: string, aliases: string[] = []): string {
for (const name of [primary, ...aliases]) {
const v = process.env[name];
if (v) return v;
}
throw new Error(`Missing required env var: ${primary}`);
}
function opt(name: string, fallback = ''): string { function opt(name: string, fallback = ''): string {
return process.env[name] ?? fallback; return process.env[name] ?? fallback;
} }
@@ -15,9 +22,9 @@ export const env = {
LIVEKIT_API_KEY: req('LIVEKIT_API_KEY'), LIVEKIT_API_KEY: req('LIVEKIT_API_KEY'),
LIVEKIT_API_SECRET: req('LIVEKIT_API_SECRET'), LIVEKIT_API_SECRET: req('LIVEKIT_API_SECRET'),
// Gemini // Gemini
GEMINI_API_KEY: req('GEMINI_API_KEY'), GEMINI_API_KEY: reqAny('GEMINI_API_KEY', ['GOOGLE_API_KEY', 'GOOGLE_GENERATIVE_AI_API_KEY']),
GEMINI_VISION_MODEL: opt('GEMINI_VISION_MODEL', 'gemini-2.0-flash'), GEMINI_VISION_MODEL: opt('GEMINI_VISION_MODEL', 'gemini-2.0-flash'),
GEMINI_LIVE_MODEL: opt('GEMINI_LIVE_MODEL', 'gemini-live-2.5-flash'), GEMINI_LIVE_MODEL: opt('GEMINI_LIVE_MODEL', 'gemini-3.1-flash-tts-preview'),
// GitHub // GitHub
GITHUB_TOKEN: req('GITHUB_TOKEN'), GITHUB_TOKEN: req('GITHUB_TOKEN'),
GITHUB_REPO: req('GITHUB_REPO'), // owner/name GITHUB_REPO: req('GITHUB_REPO'), // owner/name
+70 -32
View File
@@ -1,3 +1,4 @@
import { Buffer } from 'node:buffer';
import { import {
AudioFrame, AudioFrame,
AudioSource, AudioSource,
@@ -12,6 +13,7 @@ import { env } from '../env.js';
const SAMPLE_RATE = 24_000; const SAMPLE_RATE = 24_000;
const CHANNELS = 1; const CHANNELS = 1;
const FRAME_SAMPLES = SAMPLE_RATE / 10;
const encoder = new TextEncoder(); const encoder = new TextEncoder();
const ai = new GoogleGenAI({ apiKey: env.GEMINI_API_KEY }); const ai = new GoogleGenAI({ apiKey: env.GEMINI_API_KEY });
@@ -44,6 +46,72 @@ function audioFrames(message: LiveServerMessage): AudioFrame[] {
return out; return out;
} }
function framesFromPcmBase64(data: string, mimeType?: string): AudioFrame[] {
const frame = audioFrameFromBase64(data, mimeType);
if (!frame) return [];
const samples = frame.data;
const frames: AudioFrame[] = [];
for (let offset = 0; offset < samples.length; offset += FRAME_SAMPLES) {
const chunk = samples.subarray(offset, Math.min(offset + FRAME_SAMPLES, samples.length));
frames.push(new AudioFrame(chunk, SAMPLE_RATE, CHANNELS, chunk.length / CHANNELS));
}
return frames;
}
async function generateTtsFrames(message: string): Promise<AudioFrame[]> {
const res = await ai.models.generateContent({
model: env.GEMINI_LIVE_MODEL,
contents: [{ parts: [{ text: message }] }],
config: {
responseModalities: [Modality.AUDIO],
speechConfig: { voiceConfig: { prebuiltVoiceConfig: { voiceName: 'Kore' } } },
},
});
const parts = res.candidates?.[0]?.content?.parts ?? [];
return parts.flatMap((part) =>
framesFromPcmBase64(part.inlineData?.data ?? '', part.inlineData?.mimeType),
);
}
async function speakWithTts(source: AudioSource, message: string): Promise<void> {
for (const frame of await generateTtsFrames(message)) {
await source.captureFrame(frame);
}
}
async function speakWithLive(source: AudioSource, message: string): Promise<void> {
let done: () => void = () => {};
const donePromise = new Promise<void>((resolve) => {
done = resolve;
});
const session: Session = await ai.live.connect({
model: env.GEMINI_LIVE_MODEL,
config: { responseModalities: [Modality.AUDIO] },
callbacks: {
onmessage: (event) => {
void (async () => {
for (const frame of audioFrames(event)) await source.captureFrame(frame);
if (event.serverContent?.turnComplete || event.serverContent?.generationComplete) done();
})();
},
onerror: (event) => {
console.warn(`[voice] Gemini Live error: ${event.message}`);
done();
},
onclose: done,
},
});
session.sendClientContent({
turns: [{ role: 'user', parts: [{ text: message }] }],
turnComplete: true,
});
await Promise.race([donePromise, new Promise((resolve) => setTimeout(resolve, 15_000))]);
session.close();
}
/** /**
* Speak a message into the LiveKit room using Gemini Live audio. A data-channel * Speak a message into the LiveKit room using Gemini Live audio. A data-channel
* VOICE_CUE is sent first so clients still get the cue if audio generation or * VOICE_CUE is sent first so clients still get the cue if audio generation or
@@ -60,38 +128,8 @@ export async function speak(room: Room, message: string): Promise<void> {
try { try {
const publication = await room.localParticipant.publishTrack(track, options); const publication = await room.localParticipant.publishTrack(track, options);
let done: () => void = () => {}; if (env.GEMINI_LIVE_MODEL.includes('tts')) await speakWithTts(source, message);
const donePromise = new Promise<void>((resolve) => { else await speakWithLive(source, message);
done = resolve;
});
let session: Session | null = null;
session = await ai.live.connect({
model: env.GEMINI_LIVE_MODEL,
config: { responseModalities: [Modality.AUDIO] },
callbacks: {
onmessage: (event) => {
void (async () => {
for (const frame of audioFrames(event)) await source.captureFrame(frame);
if (event.serverContent?.turnComplete || event.serverContent?.generationComplete)
done();
})();
},
onerror: (event) => {
console.warn(`[voice] Gemini Live error: ${event.message}`);
done();
},
onclose: done,
},
});
session.sendClientContent({
turns: [{ role: 'user', parts: [{ text: message }] }],
turnComplete: true,
});
await Promise.race([donePromise, new Promise((resolve) => setTimeout(resolve, 15_000))]);
session.close();
if (publication.sid) await room.localParticipant.unpublishTrack(publication.sid, true); if (publication.sid) await room.localParticipant.unpublishTrack(publication.sid, true);
await source.close(); await source.close();
} catch (err) { } catch (err) {
+3 -3
View File
@@ -205,9 +205,9 @@ From the remote plan snapshot and health check on `2026-06-27`:
not an implemented research agent yet. not an implemented research agent yet.
- Deployment reliability is partial; API health is reachable, but API/static - Deployment reliability is partial; API health is reachable, but API/static
site/worker together must still be reverified before demo. site/worker together must still be reverified before demo.
- Env docs are inconsistent: backend defaults are `gemini-3.5-flash` and - Env docs now align on `gemini-3.5-flash` for vision and
`gemini-3.1-flash-live-preview`, while `.env.example` still lists older `gemini-3.1-flash-tts-preview` for voice. The backend still preserves a Gemini
Gemini model names. Live path for future available Live models.
### Not yet proven ### Not yet proven
+1 -1
View File
@@ -69,7 +69,7 @@ LIVEKIT_API_SECRET=...
GEMINI_API_KEY=... GEMINI_API_KEY=...
GEMINI_VISION_MODEL=gemini-2.0-flash GEMINI_VISION_MODEL=gemini-2.0-flash
GEMINI_LIVE_MODEL=gemini-live-2.5-flash GEMINI_LIVE_MODEL=gemini-3.1-flash-tts-preview
GITHUB_TOKEN=... GITHUB_TOKEN=...
GITHUB_REPO=karti-ai/podman GITHUB_REPO=karti-ai/podman
+3 -3
View File
@@ -109,11 +109,11 @@ Respond with the message text only.
--- ---
## 4. Voice Output — Gemini Live 2.5 via LiveKit ## 4. Voice Output — Gemini TTS via LiveKit
**Model:** `gemini-live-2.5-flash` (confirm exact ID from LiveKit Agents docs) **Model:** `gemini-3.1-flash-tts-preview`
**Integration:** LiveKit Agents framework — Hermes runs as a LiveKit Agent with Gemini Live 2.5 as the voice provider **Integration:** Hermes generates Gemini TTS audio and publishes it as a LiveKit audio track. The code still preserves a Gemini Live path for future available Live models.
**Flow:** **Flow:**
+4 -4
View File
@@ -89,11 +89,11 @@ Hermes uses the same endpoint. Grants:
--- ---
## Gemini Live 2.5 model ## Gemini voice model
- Model ID: `gemini-live-2.5-flash` — confirm exact ID from LiveKit Agents + Gemini docs at build time - Model ID: `gemini-3.1-flash-tts-preview`
- LiveKit Agents has native Gemini Live integration — no manual audio encoding needed - Hermes generates Gemini TTS audio and publishes it as a LiveKit audio track.
- Hermes passes text string → Agents handles streaming audio publication - The backend keeps a Gemini Live path for future model availability, but the verified deployment path uses TTS.
--- ---
@@ -72,7 +72,7 @@ PodMan is a real-time AI team coordination agent for software teams. Engineers j
- **Vision:** `gemini-2.0-flash` — screen → `{ currentFile, inferredTask, terminalVisible, recentTerminalOutput, confidence }` - **Vision:** `gemini-2.0-flash` — screen → `{ currentFile, inferredTask, terminalVisible, recentTerminalOutput, confidence }`
- **Event detection:** `gemini-2.0-flash` — all engineer states → `{ event, involvedEngineers, file, reason }` - **Event detection:** `gemini-2.0-flash` — all engineer states → `{ event, involvedEngineers, file, reason }`
- **Nudge generation:** `gemini-2.0-flash` — event → spoken message text - **Nudge generation:** `gemini-2.0-flash` — event → spoken message text
- **Voice:** `gemini-live-2.5-flash` via LiveKit Agents — text → streaming audio - **Voice:** `gemini-3.1-flash-tts-preview` via LiveKit audio publication — text → audio
### MongoDB Atlas (4 collections) ### MongoDB Atlas (4 collections)
+2 -2
View File
@@ -48,7 +48,7 @@ services:
- { key: LIVEKIT_API_SECRET, scope: RUN_TIME, type: SECRET } - { key: LIVEKIT_API_SECRET, scope: RUN_TIME, type: SECRET }
- { key: GEMINI_API_KEY, scope: RUN_TIME, type: SECRET } - { key: GEMINI_API_KEY, scope: RUN_TIME, type: SECRET }
- { key: GEMINI_VISION_MODEL, scope: RUN_TIME, value: gemini-2.0-flash } - { key: GEMINI_VISION_MODEL, scope: RUN_TIME, value: gemini-2.0-flash }
- { key: GEMINI_LIVE_MODEL, scope: RUN_TIME, value: gemini-live-2.5-flash } - { key: GEMINI_LIVE_MODEL, scope: RUN_TIME, value: gemini-3.1-flash-tts-preview }
- { key: GITHUB_TOKEN, scope: RUN_TIME, type: SECRET } - { key: GITHUB_TOKEN, scope: RUN_TIME, type: SECRET }
- { key: GITHUB_REPO, scope: RUN_TIME, value: karti-ai/podman } - { key: GITHUB_REPO, scope: RUN_TIME, value: karti-ai/podman }
- { key: MONGODB_URI, scope: RUN_TIME, type: SECRET } - { key: MONGODB_URI, scope: RUN_TIME, type: SECRET }
@@ -73,7 +73,7 @@ workers:
- { key: LIVEKIT_API_SECRET, scope: RUN_TIME, type: SECRET } - { key: LIVEKIT_API_SECRET, scope: RUN_TIME, type: SECRET }
- { key: GEMINI_API_KEY, scope: RUN_TIME, type: SECRET } - { key: GEMINI_API_KEY, scope: RUN_TIME, type: SECRET }
- { key: GEMINI_VISION_MODEL, scope: RUN_TIME, value: gemini-2.0-flash } - { key: GEMINI_VISION_MODEL, scope: RUN_TIME, value: gemini-2.0-flash }
- { key: GEMINI_LIVE_MODEL, scope: RUN_TIME, value: gemini-live-2.5-flash } - { key: GEMINI_LIVE_MODEL, scope: RUN_TIME, value: gemini-3.1-flash-tts-preview }
- { key: GITHUB_TOKEN, scope: RUN_TIME, type: SECRET } - { key: GITHUB_TOKEN, scope: RUN_TIME, type: SECRET }
- { key: GITHUB_REPO, scope: RUN_TIME, value: karti-ai/podman } - { key: GITHUB_REPO, scope: RUN_TIME, value: karti-ai/podman }
- { key: MONGODB_URI, scope: RUN_TIME, type: SECRET } - { key: MONGODB_URI, scope: RUN_TIME, type: SECRET }
+54 -11
View File
@@ -1,4 +1,5 @@
#!/usr/bin/env node #!/usr/bin/env node
import { Buffer } from 'node:buffer';
import { existsSync } from 'node:fs'; import { existsSync } from 'node:fs';
import { readFile } from 'node:fs/promises'; import { readFile } from 'node:fs/promises';
import { MongoClient } from 'mongodb'; import { MongoClient } from 'mongodb';
@@ -15,7 +16,6 @@ const requiredEnv = [
'LIVEKIT_URL', 'LIVEKIT_URL',
'LIVEKIT_API_KEY', 'LIVEKIT_API_KEY',
'LIVEKIT_API_SECRET', 'LIVEKIT_API_SECRET',
'GEMINI_API_KEY',
'GITHUB_TOKEN', 'GITHUB_TOKEN',
'GITHUB_REPO', 'GITHUB_REPO',
'MONGODB_URI', 'MONGODB_URI',
@@ -33,6 +33,19 @@ function isSet(name) {
return !!process.env[name]?.trim(); return !!process.env[name]?.trim();
} }
function configuredGeminiKey() {
const candidates = ['GEMINI_API_KEY', 'GOOGLE_API_KEY', 'GOOGLE_GENERATIVE_AI_API_KEY'];
for (const name of candidates) {
const value = process.env[name]?.trim();
if (!value) continue;
if (/replace|todo|example|your|xxx/i.test(value) || value.length < 20) {
throw new Error(`${name} looks like a placeholder or truncated key`);
}
return { name, value };
}
throw new Error('GEMINI_API_KEY is not set');
}
async function check(name, fn) { async function check(name, fn) {
try { try {
const detail = await fn(); const detail = await fn();
@@ -205,12 +218,12 @@ async function checkGitHub() {
} }
async function checkGeminiVision() { async function checkGeminiVision() {
if (!isSet('GEMINI_API_KEY')) throw new Error('GEMINI_API_KEY is not set'); const key = configuredGeminiKey();
const model = process.env.GEMINI_VISION_MODEL ?? 'gemini-2.0-flash'; const model = process.env.GEMINI_VISION_MODEL ?? 'gemini-2.0-flash';
const res = await doFetch( const res = await doFetch(
`https://generativelanguage.googleapis.com/v1beta/models/${encodeURIComponent( `https://generativelanguage.googleapis.com/v1beta/models/${encodeURIComponent(
model, model,
)}:generateContent?key=${encodeURIComponent(process.env.GEMINI_API_KEY)}`, )}:generateContent?key=${encodeURIComponent(key.value)}`,
{ {
method: 'POST', method: 'POST',
headers: { 'content-type': 'application/json' }, headers: { 'content-type': 'application/json' },
@@ -224,19 +237,39 @@ async function checkGeminiVision() {
return model; return model;
} }
async function checkGeminiLiveListed() { async function checkGeminiVoiceModel() {
if (!isSet('GEMINI_API_KEY')) throw new Error('GEMINI_API_KEY is not set'); const key = configuredGeminiKey();
const model = process.env.GEMINI_LIVE_MODEL ?? 'gemini-live-2.5-flash'; const model = process.env.GEMINI_LIVE_MODEL ?? 'gemini-3.1-flash-tts-preview';
const res = await doFetch( const res = await doFetch(
`https://generativelanguage.googleapis.com/v1beta/models?key=${encodeURIComponent( `https://generativelanguage.googleapis.com/v1beta/models?key=${encodeURIComponent(key.value)}`,
process.env.GEMINI_API_KEY,
)}`,
); );
if (!res.ok) throw new Error(await responseError('Gemini model list', res)); if (!res.ok) throw new Error(await responseError('Gemini model list', res));
const body = await res.json(); const body = await res.json();
const names = (body.models ?? []).map((m) => m.name?.replace(/^models\//, '')); const names = (body.models ?? []).map((m) => m.name?.replace(/^models\//, ''));
if (!names.includes(model)) throw new Error(`${model} not present in Gemini model list`); if (!names.includes(model)) throw new Error(`${model} not present in Gemini model list`);
return model; if (!model.includes('tts')) return model;
const tts = await doFetch(
`https://generativelanguage.googleapis.com/v1beta/models/${encodeURIComponent(
model,
)}:generateContent?key=${encodeURIComponent(key.value)}`,
{
method: 'POST',
headers: { 'content-type': 'application/json' },
body: JSON.stringify({
contents: [{ parts: [{ text: 'Say clearly: PodMan voice check.' }] }],
generationConfig: {
responseModalities: ['AUDIO'],
speechConfig: { voiceConfig: { prebuiltVoiceConfig: { voiceName: 'Kore' } } },
},
}),
},
);
if (!tts.ok) throw new Error(await responseError('Gemini voice check', tts));
const ttsBody = await tts.json();
const audio = ttsBody.candidates?.[0]?.content?.parts?.[0]?.inlineData?.data;
if (!audio) throw new Error('Gemini voice response had no audio');
return `${model}, generated ${Buffer.from(audio, 'base64').byteLength} audio bytes`;
} }
async function checkVoyage() { async function checkVoyage() {
@@ -262,6 +295,16 @@ await check('workspace', checkWorkspace);
for (const name of requiredEnv) { for (const name of requiredEnv) {
add(`env:${name}`, isSet(name) ? 'ok' : 'fail', isSet(name) ? 'set' : 'missing'); add(`env:${name}`, isSet(name) ? 'ok' : 'fail', isSet(name) ? 'set' : 'missing');
} }
try {
const key = configuredGeminiKey();
add(
'env:GEMINI_API_KEY',
'ok',
key.name === 'GEMINI_API_KEY' ? 'set' : `using ${key.name} alias`,
);
} catch (err) {
add('env:GEMINI_API_KEY', 'fail', summarizeError(err));
}
for (const name of optionalEnv) { for (const name of optionalEnv) {
add(`env:${name}`, isSet(name) ? 'ok' : 'warn', isSet(name) ? 'set' : 'optional'); add(`env:${name}`, isSet(name) ? 'ok' : 'warn', isSet(name) ? 'set' : 'optional');
} }
@@ -281,7 +324,7 @@ await check('livekit room service', checkLiveKitApi);
await check('mongo ping', checkMongo); await check('mongo ping', checkMongo);
await check('github repo access', checkGitHub); await check('github repo access', checkGitHub);
await check('gemini vision model', checkGeminiVision); await check('gemini vision model', checkGeminiVision);
await check('gemini live model listed', checkGeminiLiveListed); await check('gemini voice model', checkGeminiVoiceModel);
if (isSet('VOYAGE_API_KEY')) { if (isSet('VOYAGE_API_KEY')) {
await check('voyage embeddings', checkVoyage); await check('voyage embeddings', checkVoyage);