fix: full review pass (main process, voice pipeline, Hermes client, HUD) and e2e-validated local voice (v2.1.0.1)

Electron main
- utility-process channel: sherpa output copied into V8 buffers (Electron rejects external
  buffers), audio exchanged as base64; worker restart on timeout, identity-safe exit handling,
  deterministic native unload by restart
- webhook: loopback by default without secret, UTF-8-safe body assembly, clean restart, port validation
- files IPC: realpath-based root check, openPath allow-list (reveal-only outside document folders),
  Windows reserved names; store: debounced async atomic writes with backup of corrupt files;
  logger: streaming writes with rotation; log message size cap
- HTTP stream proxy: socket released on idle timeout / renderer destroyed, id validation
- model manager: inactivity timeout, retrying rm/rename (Windows locks), engine stopped before
  replacing a model; WAV decoder handles float/24-bit; IPC payload validation
- window: opaque rounded window on Windows, navigation lock-down, visibility events;
  Ctrl+Alt+Escape instead of the Task Manager shortcut; single-instance guard; EVEFLOW_USER_DATA

Voice pipeline
- abort semantics (SendHandle.aborted), abort before the stream opens, session id prefixes per transport
- hands-free re-arm after replies without speech, start/stop race, no chime on auto re-arm,
  no silence shipped to STT (400 ms pre-roll), no transcription of empty manual stops
- TTS: bounded prefetch, cancellable segments, non-interrupting notices, volume applied at play time
- SSE CRLF split, usage in chat completions, finish_reason length, phonetic regex hoisted

HUD
- core renderer: no canvas shadows, cached colours, reusable spectrum buffer, theme read on change,
  30 fps idle, stops when the window is hidden; ping flashes on send / tool / speech
- deltas coalesced per animation frame; stable auto-scroll; narrow selectors everywhere
- bundled fonts (offline), reduce-motion fix, error toasts, ops drawer below 1180 px,
  interim transcript and first-token latency in the core caption, Ctrl+K, dialog semantics,
  switch/aria roles, compact widget cleanup, Whisper small recommended for French
- docs/ROADMAP.md: audit results, e2e results and the plan towards a real JARVIS

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_017Wn5VX9HNbJ7N54hR24u9Y
This commit is contained in:
Claude committed 2026-09-03 18:08:09 +00:00
1 parent b1b8d0d5d4
commit 5f0fa7a1e2
55 files changed
+1181 -433

No files matched your search

+30 -7
View File
@@ -16,7 +16,7 @@ interface ModelRef {
type Request =
| { id: number; type: 'status' }
| { id: number; type: 'transcribe'; model: ModelRef; wav: Uint8Array; language: string }
| { id: number; type: 'transcribe'; model: ModelRef; wav: Uint8Array | string; language: string }
| { id: number; type: 'synthesize'; model: ModelRef; text: string; speaker: number; speed: number }
| { id: number; type: 'unload'; modelId?: string };
@@ -32,7 +32,7 @@ type Sherpa = {
OfflineTts: new (config: unknown) => {
numSpeakers: number;
sampleRate: number;
generate: (req: { text: string; sid: number; speed: number }) => { samples: Float32Array; sampleRate: number };
generate: (req: { text: string; sid: number; speed: number; enableExternalBuffer?: boolean }) => { samples: Float32Array; sampleRate: number };
};
version: string;
};
@@ -136,16 +136,20 @@ function getSynthesizer(model: ModelRef) {
// ── audio helpers ──────────────────────────────────────────────────────────
function decodeWav(bytes: Uint8Array): { samples: Float32Array; sampleRate: number } {
const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength);
if (String.fromCharCode(bytes[0], bytes[1], bytes[2], bytes[3]) !== 'RIFF') throw new Error('WAV invalide');
if (String.fromCharCode(bytes[0], bytes[1], bytes[2], bytes[3]) !== 'RIFF' || String.fromCharCode(bytes[8], bytes[9], bytes[10], bytes[11]) !== 'WAVE') {
throw new Error('WAV invalide');
}
let offset = 12;
let sampleRate = 16000;
let channels = 1;
let bits = 16;
let format = 1; // 1 = PCM, 3 = IEEE float
let data: { start: number; length: number } | null = null;
while (offset + 8 <= bytes.byteLength) {
const id = String.fromCharCode(bytes[offset], bytes[offset + 1], bytes[offset + 2], bytes[offset + 3]);
const size = view.getUint32(offset + 4, true);
if (id === 'fmt ') {
format = view.getUint16(offset + 8, true);
channels = view.getUint16(offset + 10, true);
sampleRate = view.getUint32(offset + 12, true);
bits = view.getUint16(offset + 22, true);
@@ -163,7 +167,16 @@ function decodeWav(bytes: Uint8Array): { samples: Float32Array; sampleRate: numb
let sum = 0;
for (let c = 0; c < channels; c++) {
const pos = data.start + (i * channels + c) * bytesPerSample;
sum += bits === 16 ? view.getInt16(pos, true) / 32768 : bits === 32 ? view.getInt32(pos, true) / 2147483648 : (bytes[pos] - 128) / 128;
sum +=
format === 3 && bits === 32
? view.getFloat32(pos, true)
: bits === 16
? view.getInt16(pos, true) / 32768
: bits === 24
? (((bytes[pos] | (bytes[pos + 1] << 8) | (bytes[pos + 2] << 16)) << 8) >> 8) / 8388608
: bits === 32
? view.getInt32(pos, true) / 2147483648
: (bytes[pos] - 128) / 128;
}
samples[i] = sum / channels;
}
@@ -225,8 +238,11 @@ function handle(req: Request): unknown {
}
case 'transcribe': {
const started = Date.now();
const { samples, sampleRate } = decodeWav(req.wav);
// Audio crosses the process boundary as base64: V8 refuses to serialize external buffers.
const bytes = typeof req.wav === 'string' ? new Uint8Array(Buffer.from(req.wav, 'base64')) : req.wav;
const { samples, sampleRate } = decodeWav(bytes);
const pcm = resampleTo16k(samples, sampleRate);
if (pcm.length < 1600) throw new Error('Audio trop court');
const recognizer = getRecognizer(req.model, req.language);
const stream = recognizer.createStream();
stream.acceptWaveform({ sampleRate: 16000, samples: pcm });
@@ -238,8 +254,15 @@ function handle(req: Request): unknown {
const started = Date.now();
const tts = getSynthesizer(req.model);
const sid = Math.max(0, Math.min(tts.numSpeakers - 1, Math.floor(req.speaker)));
const audio = tts.generate({ text: req.text, sid, speed: Math.max(0.5, Math.min(2, req.speed || 1)) });
return { wav: encodeWav(audio.samples, audio.sampleRate), sampleRate: audio.sampleRate, durationMs: Date.now() - started, audioSec: audio.samples.length / audio.sampleRate };
// Electron forbids N-API external buffers: ask sherpa-onnx to copy the samples into a V8 buffer.
const audio = tts.generate({ text: req.text, sid, speed: Math.max(0.5, Math.min(2, req.speed || 1)), enableExternalBuffer: false });
const wav = encodeWav(audio.samples, audio.sampleRate);
return {
wav: Buffer.from(wav.buffer, wav.byteOffset, wav.byteLength).toString('base64'),
sampleRate: audio.sampleRate,
durationMs: Date.now() - started,
audioSec: audio.samples.length / audio.sampleRate
};
}
case 'unload': {
if (req.modelId) {