diff --git a/README.md b/README.md index 89d8437..7645109 100644 --- a/README.md +++ b/README.md @@ -87,7 +87,9 @@ Quit Hype Ghost. OBS plugin (local whisper.cpp — audio never leaves your PC), add its Transcription filter to your mic source, output to a text file or a (hidden) text source, and point the wizard's voice-awareness step (or the `transcript` config section) at it. Then answering the ghost - out loud *is* replying. + out loud *is* replying. Everything it hears also shows up as faint 🎙 lines in the deck + feed (toggle in Settings → Voice), so when the cast reacts oddly you can see exactly what + the transcription thought you said. - **Party / co-op audio (second channel):** streaming alongside friends? Route their audio (Discord/TeamSpeak/party chat) to a separate OBS source, add a *second* LocalVocal Transcription filter to it, output to a **different** file/text source than your mic, and @@ -124,6 +126,7 @@ Quit Hype Ghost. | `twitch.decapi` | Fetch viewer count + stream info keylessly via [DecAPI](https://decapi.me), a community-run Twitch API proxy (default true). Only your public channel name is sent to it. Set false to opt out. | | `twitch.clientId` / `clientSecret` | *(Optional, power users)* App credentials from dev.twitch.tv — when set, the official Helix API is used directly instead of DecAPI. | | `transcript.mode` | `off`, `file` (tail LocalVocal's .txt/.srt output), or `textSource` (poll a text source over OBS WebSocket). | +| `transcript.showInFeed` | Echo what voice awareness hears into the deck feed as faint 🎙 lines (party audio as 🎧) so you can spot mishears the cast might be reacting to (default true). Deck only — never in the cast's chat history, the recap, or the on-stream overlay. | | `transcript2.*` | *(Optional)* A **second** transcription channel for party/co-op audio (a separate audio device with its own LocalVocal filter). Same `mode`/`file`/`textSource`/`pollSeconds` as `transcript`, plus `label` — how the cast refers to those people (e.g. "my co-op squad"). Treated as *other people*, never as the streamer. | | `cadence.soloSeconds` / `quietSeconds` | Average gap between messages when alone / when real viewers are present. | | `cadence.jitter`, `burstChance`, `lullChance` | Natural rhythm: ±jitter on normal gaps, occasional quick bursts (0.3–0.6x) and long lulls (1.6–3x). | diff --git a/config.example.json b/config.example.json index e187635..c50146c 100644 --- a/config.example.json +++ b/config.example.json @@ -94,7 +94,8 @@ "mode": "off", "file": "", "textSource": "LocalVocal Captions", - "pollSeconds": 2 + "pollSeconds": 2, + "showInFeed": true }, "transcript2": { "mode": "off", diff --git a/public/dashboard.html b/public/dashboard.html index 3a9d100..33640b9 100644 --- a/public/dashboard.html +++ b/public/dashboard.html @@ -120,6 +120,14 @@ .msg.streamer .meta { flex-direction: row-reverse; } .msg.streamer .nm { color: var(--accent-2); } .msg.streamer .bubble { background: var(--accent-soft); border-color: var(--accent-line); border-radius: var(--r-lg); border-top-right-radius: 4px; } + /* "heard" rows — what voice awareness transcribed. Faint and italic on + purpose: this is sensor readout (possibly misheard), not chat. Mic lines + sit right like your typed messages; party lines sit left (other people). */ + .heard { align-self: flex-end; display: flex; align-items: baseline; gap: 8px; max-width: 640px; font-size: var(--fs-sm); font-style: italic; color: var(--faint); animation: rise .28s ease-out; } + .heard.party { align-self: flex-start; } + .heard .ic { font-style: normal; flex: none; } + .heard .txt { overflow-wrap: anywhere; min-width: 0; } + .heard .tm { font-style: normal; font-size: var(--fs-xs); flex: none; } .moment-card { align-self: center; display: flex; align-items: center; gap: 10px; padding: 7px 15px; border-radius: var(--r-full); background: var(--glass); border: 1px solid var(--accent-line); font-size: var(--fs-sm); animation: pop .4s cubic-bezier(.2,1.3,.5,1); } @keyframes pop { from { opacity: 0; transform: scale(.85); } to { opacity: 1; transform: none; } } .moment-card .spark { background: var(--aurora); -webkit-background-clip: text; background-clip: text; color: transparent; font-weight: 800; } @@ -260,7 +268,8 @@ next:'up next', obsOk:'OBS connected', obsNo:'OBS offline', brainOk:'Brain ready', brainNo:'No brain', viewersManual:'manual', micOff:'off', micQuiet:'quiet', micWait:'waiting', party:'Party', partyQuiet:'quiet', partyWait:'waiting', chatQuiet:'quiet', chatActive:'active', lost:'Connection lost — reconnecting…', msgs:'msgs', watching:'watching', idle:'idle', noMoments:'No moments yet — the cast flags clip-worthy plays here.', - previewBody:'Preview mode — your cast has no brain yet, so the room stays quiet.', previewCta:'Connect one in Settings' }, + previewBody:'Preview mode — your cast has no brain yet, so the room stays quiet.', previewCta:'Connect one in Settings', + heardTip:'What voice awareness transcribed — the cast may be reacting to this. If it misheard you, that explains their reaction. (Deck only — never shown on stream.)' }, es: { brain:'Cerebro', viewers:'Espectadores', mic:'Micro', chat:'Chat', cast:'El elenco', energy:'Energía', chill:'Tranqui', balanced:'Normal', hype:'Máx', room:'Sala', modeAuto:'Auto', modeSolo:'Solo', modeViewers:'Público', say:'Di algo', pause:'Pausar', resume:'Reanudar', moments:'Momentos', recap:'Exportar', send:'Enviar', you:'Tú', @@ -269,7 +278,8 @@ next:'siguiente', obsOk:'OBS conectado', obsNo:'OBS desconectado', brainOk:'Cerebro listo', brainNo:'Sin cerebro', viewersManual:'manual', micOff:'apagado', micQuiet:'silencio', micWait:'esperando', party:'Grupo', partyQuiet:'silencio', partyWait:'esperando', chatQuiet:'tranquilo', chatActive:'activo', lost:'Conexión perdida — reconectando…', msgs:'msjs', watching:'mirando', idle:'inactivo', noMoments:'Aún no hay momentos — el elenco marcará jugadas destacadas aquí.', - previewBody:'Modo vista previa — tu elenco aún no tiene cerebro, así que la sala sigue en silencio.', previewCta:'Conéctalo en Ajustes' }, + previewBody:'Modo vista previa — tu elenco aún no tiene cerebro, así que la sala sigue en silencio.', previewCta:'Conéctalo en Ajustes', + heardTip:'Lo que la escucha de voz transcribió — el elenco puede estar reaccionando a esto. Si te entendió mal, eso explica su reacción. (Solo en la consola — nunca en el stream.)' }, }; let L = STR.en; const T = (k) => (L[k] ?? STR.en[k] ?? k); @@ -393,6 +403,24 @@ renderMoments(); } + // What voice awareness heard (mic or party channel) — deck-only, ephemeral + // (not in history, not on the overlay, gone on reload). Capped so hours of + // talking can't grow the DOM without bound. + const MAX_HEARD_ROWS = 30; + function addHeard(h) { + clearEmpty(); + const rows = feed.querySelectorAll('.heard'); + if (rows.length >= MAX_HEARD_ROWS) rows[0].remove(); + const div = document.createElement('div'); + div.className = 'heard' + (h.channel === 'party' ? ' party' : ''); + div.title = T('heardTip'); + const time = new Date(h.ts).toLocaleTimeString([], { hour: '2-digit', minute: '2-digit' }); + div.innerHTML = `${h.channel === 'party' ? '🎧' : '🎙️'}${time}`; + div.querySelector('.txt').textContent = (h.channel === 'party' && h.label ? h.label + ' · ' : '') + '“' + h.text + '”'; + feed.appendChild(div); + feed.scrollTop = feed.scrollHeight; + } + function addSystem(text) { clearEmpty(); const div = document.createElement('div'); @@ -537,6 +565,7 @@ else if (msg.type === 'moment') { momentsData.push(msg.moment); addMoment(msg.moment, true); } else if (msg.type === 'state') { const s = state._startedAt; state = msg.state; state._startedAt = state.startedAt || s || Date.now(); renderState(); } else if (msg.type === 'system') { addSystem(msg.text); } + else if (msg.type === 'heard') { addHeard(msg.heard); } }; ws.onclose = () => { if (!announcedDrop) { announcedDrop = true; toast(T('lost')); } setTimeout(connect, 3000); }; } diff --git a/public/settings.html b/public/settings.html index 425951d..8167212 100644 --- a/public/settings.html +++ b/public/settings.html @@ -212,6 +212,10 @@

🎙️ Voice awareness

+
+ + +

🎧 Second channel — party / co-op audio

diff --git a/src/server.js b/src/server.js index 5d8c9e7..2d39911 100644 --- a/src/server.js +++ b/src/server.js @@ -234,7 +234,7 @@ export function startServer(opts = {}) { const HOT_PATHS = [ 'energy', 'talkingPoints', 'theme.', 'overlay.', 'moments.', 'memory.', 'stream.', 'app.costMeter', 'app.uiLanguage', 'app.fontScale', 'app.autoPause', 'app.autoPauseMinutes', 'app.autoUpdate', - 'app.ttsVoice', 'app.ttsRate', 'app.ttsOutputDevice', + 'app.ttsVoice', 'app.ttsRate', 'app.ttsOutputDevice', 'transcript.showInFeed', 'cadence.soloSeconds', 'cadence.quietSeconds', 'cadence.jitter', 'cadence.burstChance', 'cadence.lullChance', 'cadence.replyDelaySeconds', 'cadence.minVoiceReplyGapSeconds', 'cadence.minScreenshotGapSeconds', 'cadence.minPartyNudgeGapSeconds', @@ -556,6 +556,13 @@ export function startServer(opts = {}) { obs, onSpeech: (line) => { state.lastHeard = { text: line, ts: Date.now() }; + // Deck-only echo of what was transcribed, so mishears are visible the + // moment they happen. Never enters `history` (the brain already gets + // the transcript window separately) and never reaches the overlay + // (its client ignores the type). Read live — the toggle hot-applies. + if (config.transcript.showInFeed !== false) { + broadcast({ type: 'heard', heard: { channel: 'mic', text: line, ts: Date.now() } }); + } broadcastState(); loop.onSpeech(); }, @@ -571,6 +578,13 @@ export function startServer(opts = {}) { obs, onSpeech: (line) => { state.lastHeardParty = { text: line, ts: Date.now() }; + // Same deck-only echo as the mic channel, labeled as "other people". + if (config.transcript.showInFeed !== false) { + broadcast({ + type: 'heard', + heard: { channel: 'party', label: config.transcript2?.label || 'Party', text: line, ts: Date.now() }, + }); + } broadcastState(); loop.onPartySpeech(); }, diff --git a/test/transcript.test.js b/test/transcript.test.js index d82b307..e6ba39a 100644 --- a/test/transcript.test.js +++ b/test/transcript.test.js @@ -88,3 +88,33 @@ test('getWindow prunes entries older than the window and honors sinceTs', (t) => assert.equal(feed.getWindow(), 'new speech'); assert.equal(feed.getWindow(Date.now() + 1000), ''); // nothing newer than the future }); + +// textSource mode must drive the exact same onSpeech path as file mode — the +// deck's "heard" feed echo and voice replies hang off that callback, so both +// transcription modes get identical behavior. +test('textSource mode fires onSpeech on changed captions, same as file mode', async (t) => { + let sourceText = 'stale pre-launch caption'; + const heard = []; + const feed = new TranscriptFeed({ + mode: 'textSource', + textSource: 'LocalVocal Captions', + pollSeconds: 999, + windowSeconds: 120, + obs: { getTextSourceText: async () => sourceText }, + onSpeech: (line) => heard.push(line), + }); + // Prime exactly like start() does: the pre-launch caption is not fresh speech. + feed.lastSourceText = await feed.obs.getTextSourceText(feed.textSource); + await feed.pollTextSource(); + assert.deepEqual(heard, [], 'primed caption must not fire'); + sourceText = 'did you see that dragon'; + await feed.pollTextSource(); + await feed.pollTextSource(); // unchanged caption fires only once + assert.deepEqual(heard, ['did you see that dragon']); + sourceText = null; // OBS unreachable mid-stream + await feed.pollTextSource(); + sourceText = '00:01:02,000 --> 00:01:04,000'; // SRT timing junk is filtered here too + await feed.pollTextSource(); + assert.deepEqual(heard, ['did you see that dragon']); + assert.equal(feed.getWindow(), 'did you see that dragon'); +});