feat: activate gemini voice tts path

This commit is contained in:
Yahya Alhinai
2026-06-28 02:11:02 +00:00
parent 4726e8ce80
commit 89893110f1
11 changed files with 151 additions and 61 deletions
+1 -1
View File
@@ -83,6 +83,6 @@ export class PodMan {
reliable: true,
topic: DATA_TOPIC,
});
await speak(this.room, message); // gemini-3.1-flash-live voice into the room
await speak(this.room, message); // Gemini voice audio into the room
}
}
+9 -2
View File
@@ -5,6 +5,13 @@ function req(name: string): string {
if (!v) throw new Error(`Missing required env var: ${name}`);
return v;
}
function reqAny(primary: string, aliases: string[] = []): string {
for (const name of [primary, ...aliases]) {
const v = process.env[name];
if (v) return v;
}
throw new Error(`Missing required env var: ${primary}`);
}
function opt(name: string, fallback = ''): string {
return process.env[name] ?? fallback;
}
@@ -15,9 +22,9 @@ export const env = {
LIVEKIT_API_KEY: req('LIVEKIT_API_KEY'),
LIVEKIT_API_SECRET: req('LIVEKIT_API_SECRET'),
// Gemini
GEMINI_API_KEY: req('GEMINI_API_KEY'),
GEMINI_API_KEY: reqAny('GEMINI_API_KEY', ['GOOGLE_API_KEY', 'GOOGLE_GENERATIVE_AI_API_KEY']),
GEMINI_VISION_MODEL: opt('GEMINI_VISION_MODEL', 'gemini-2.0-flash'),
GEMINI_LIVE_MODEL: opt('GEMINI_LIVE_MODEL', 'gemini-live-2.5-flash'),
GEMINI_LIVE_MODEL: opt('GEMINI_LIVE_MODEL', 'gemini-3.1-flash-tts-preview'),
// GitHub
GITHUB_TOKEN: req('GITHUB_TOKEN'),
GITHUB_REPO: req('GITHUB_REPO'), // owner/name
+70 -32
View File
@@ -1,3 +1,4 @@
import { Buffer } from 'node:buffer';
import {
AudioFrame,
AudioSource,
@@ -12,6 +13,7 @@ import { env } from '../env.js';
const SAMPLE_RATE = 24_000;
const CHANNELS = 1;
const FRAME_SAMPLES = SAMPLE_RATE / 10;
const encoder = new TextEncoder();
const ai = new GoogleGenAI({ apiKey: env.GEMINI_API_KEY });
@@ -44,6 +46,72 @@ function audioFrames(message: LiveServerMessage): AudioFrame[] {
return out;
}
function framesFromPcmBase64(data: string, mimeType?: string): AudioFrame[] {
const frame = audioFrameFromBase64(data, mimeType);
if (!frame) return [];
const samples = frame.data;
const frames: AudioFrame[] = [];
for (let offset = 0; offset < samples.length; offset += FRAME_SAMPLES) {
const chunk = samples.subarray(offset, Math.min(offset + FRAME_SAMPLES, samples.length));
frames.push(new AudioFrame(chunk, SAMPLE_RATE, CHANNELS, chunk.length / CHANNELS));
}
return frames;
}
async function generateTtsFrames(message: string): Promise<AudioFrame[]> {
const res = await ai.models.generateContent({
model: env.GEMINI_LIVE_MODEL,
contents: [{ parts: [{ text: message }] }],
config: {
responseModalities: [Modality.AUDIO],
speechConfig: { voiceConfig: { prebuiltVoiceConfig: { voiceName: 'Kore' } } },
},
});
const parts = res.candidates?.[0]?.content?.parts ?? [];
return parts.flatMap((part) =>
framesFromPcmBase64(part.inlineData?.data ?? '', part.inlineData?.mimeType),
);
}
async function speakWithTts(source: AudioSource, message: string): Promise<void> {
for (const frame of await generateTtsFrames(message)) {
await source.captureFrame(frame);
}
}
async function speakWithLive(source: AudioSource, message: string): Promise<void> {
let done: () => void = () => {};
const donePromise = new Promise<void>((resolve) => {
done = resolve;
});
const session: Session = await ai.live.connect({
model: env.GEMINI_LIVE_MODEL,
config: { responseModalities: [Modality.AUDIO] },
callbacks: {
onmessage: (event) => {
void (async () => {
for (const frame of audioFrames(event)) await source.captureFrame(frame);
if (event.serverContent?.turnComplete || event.serverContent?.generationComplete) done();
})();
},
onerror: (event) => {
console.warn(`[voice] Gemini Live error: ${event.message}`);
done();
},
onclose: done,
},
});
session.sendClientContent({
turns: [{ role: 'user', parts: [{ text: message }] }],
turnComplete: true,
});
await Promise.race([donePromise, new Promise((resolve) => setTimeout(resolve, 15_000))]);
session.close();
}
/**
* Speak a message into the LiveKit room using Gemini Live audio. A data-channel
* VOICE_CUE is sent first so clients still get the cue if audio generation or
@@ -60,38 +128,8 @@ export async function speak(room: Room, message: string): Promise<void> {
try {
const publication = await room.localParticipant.publishTrack(track, options);
let done: () => void = () => {};
const donePromise = new Promise<void>((resolve) => {
done = resolve;
});
let session: Session | null = null;
session = await ai.live.connect({
model: env.GEMINI_LIVE_MODEL,
config: { responseModalities: [Modality.AUDIO] },
callbacks: {
onmessage: (event) => {
void (async () => {
for (const frame of audioFrames(event)) await source.captureFrame(frame);
if (event.serverContent?.turnComplete || event.serverContent?.generationComplete)
done();
})();
},
onerror: (event) => {
console.warn(`[voice] Gemini Live error: ${event.message}`);
done();
},
onclose: done,
},
});
session.sendClientContent({
turns: [{ role: 'user', parts: [{ text: message }] }],
turnComplete: true,
});
await Promise.race([donePromise, new Promise((resolve) => setTimeout(resolve, 15_000))]);
session.close();
if (env.GEMINI_LIVE_MODEL.includes('tts')) await speakWithTts(source, message);
else await speakWithLive(source, message);
if (publication.sid) await room.localParticipant.unpublishTrack(publication.sid, true);
await source.close();
} catch (err) {