feat: pod-wide Lyria background music (replaces test-audio drums)

The 'Test audio' button becomes 'Background Music': each pod gets a calm looping track from Gemini Lyria 3 that sings the pod name once up front then stays instrumental. Backend GET /api/pods/:id/music generates via the Gemini interactions endpoint and caches the MP3 per pod in Mongo (pod_music); the frontend fetches and loops it via Web Audio, published pod-wide on the existing podman-beat track.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
Kartikeya
2026-06-28 02:55:19 -07:00
parent 604bf9d5ac
commit ab8ea07c12
7 changed files with 218 additions and 43 deletions
+16
View File
@@ -29,6 +29,7 @@ import { loadPodGraph, reachFrom } from './graph/store.js';
import { listPodActivity } from './activity/store.js';
import { getMemberWorkHistory } from './activity/member-history.js';
import { speakInRoom } from './voice/live.js';
import { getPodMusic } from './voice/music.js';
import { notifyHermesInterventionInRoom } from './action/hermes.js';
import {
activeLiveConversation,
@@ -406,6 +407,21 @@ app.get('/api/internal/hermes/jobs/:jobId/events/stream', async (req, res) => {
});
});
// Per-pod background music (Lyria), generated once and cached. Streams MP3 the
// frontend loops as a pod-wide LiveKit track (replaces the synthesized beat).
app.get('/api/pods/:id/music', async (req, res) => {
try {
const pod = await getPod(req.params.id);
if (!pod) return res.status(404).json({ error: 'pod not found' });
const mp3 = await getPodMusic(pod.id, pod.name);
res.set('Content-Type', 'audio/mpeg');
res.set('Cache-Control', 'public, max-age=86400');
res.send(mp3);
} catch (e) {
res.status(500).json({ error: (e as Error).message });
}
});
app.post('/api/pods/:id/hermes/notify', async (req, res) => {
const podId = req.params.id;
const pod = await getPod(podId);
+94
View File
@@ -0,0 +1,94 @@
import { Buffer } from 'node:buffer';
import { getDb } from '../memory/db.js';
import { env } from '../env.js';
// Lyria 3 is reached via the Gemini "interactions" endpoint (not :predict, which
// is the Vertex path). The clip model returns a ~30s base64 MP3.
const MUSIC_MODEL = process.env.GEMINI_MUSIC_MODEL ?? 'lyria-3-clip-preview';
const INTERACTIONS_URL = 'https://generativelanguage.googleapis.com/v1beta/interactions';
interface PodMusicDoc {
podId: string;
name: string; // pod name the vocal was generated for
model: string;
mp3Base64: string;
createdAt: string;
}
interface InteractionContent {
type?: string;
data?: string;
text?: string;
}
interface InteractionResponse {
steps?: Array<{ content?: InteractionContent[] }>;
output_audio?: { data?: string };
}
/**
* Background "hold music" prompt: opens with the pod name sung once, then a calm
* instrumental bed that loops. Keep it unobtrusive — this is fill, not a song.
*/
function musicPrompt(podName: string): string {
return [
'Calm soothing instrumental background hold music for a tech app, like gentle on-hold lobby music.',
`It opens in the first three seconds with a soft gentle voice clearly saying the words "${podName}" one time,`,
'and after that opening it is purely instrumental with warm electric piano, gentle synth pads and a soft relaxed beat.',
'Unobtrusive, pleasant and steady with no climax, designed to loop seamlessly as quiet background fill.',
'No other lyrics or vocals after the opening.',
].join(' ');
}
function extractAudioBase64(data: InteractionResponse): string | null {
for (const step of data.steps ?? []) {
for (const c of step.content ?? []) {
if (c.type === 'audio' && c.data) return c.data;
}
}
return data.output_audio?.data ?? null;
}
async function generate(podName: string): Promise<Buffer> {
const res = await fetch(INTERACTIONS_URL, {
method: 'POST',
headers: { 'Content-Type': 'application/json', 'x-goog-api-key': env.GEMINI_API_KEY },
body: JSON.stringify({ model: MUSIC_MODEL, input: musicPrompt(podName) }),
});
if (!res.ok) {
throw new Error(`Lyria ${res.status}: ${(await res.text()).slice(0, 300)}`);
}
const data = (await res.json()) as InteractionResponse;
const b64 = extractAudioBase64(data);
if (!b64) throw new Error('Lyria returned no audio');
return Buffer.from(b64, 'base64');
}
/**
* The pod's background-music MP3, generated by Lyria on first request and cached
* in the `pod_music` collection. Regenerated if the pod name changes so the sung
* name stays correct. Lyria generation is slow (~20s); the cache makes every
* call after the first instant.
*/
export async function getPodMusic(podId: string, podName: string): Promise<Buffer> {
const db = await getDb();
const col = db.collection<PodMusicDoc>('pod_music');
const cached = await col.findOne({ podId });
if (cached && cached.name === podName && cached.model === MUSIC_MODEL && cached.mp3Base64) {
return Buffer.from(cached.mp3Base64, 'base64');
}
const mp3 = await generate(podName);
await col.updateOne(
{ podId },
{
$set: {
podId,
name: podName,
model: MUSIC_MODEL,
mp3Base64: mp3.toString('base64'),
createdAt: new Date().toISOString(),
},
},
{ upsert: true },
);
return mp3;
}