mirror of
https://github.com/tinyhumansai/openhuman.git
synced 2026-07-27 21:08:00 +00:00
feat(mascot): mic-only composer with cloud STT + cartoon voice (#1253)
This commit is contained in:
+8
-14
@@ -14,10 +14,6 @@ import Rewards from './pages/Rewards';
|
||||
import Settings from './pages/Settings';
|
||||
import Skills from './pages/Skills';
|
||||
import Welcome from './pages/Welcome';
|
||||
import { APP_ENVIRONMENT } from './utils/config';
|
||||
|
||||
/** /human is mascot work-in-progress — only mount the route pre-prod. */
|
||||
const HUMAN_ROUTE_ENABLED = APP_ENVIRONMENT !== 'production';
|
||||
|
||||
const AppRoutes = () => {
|
||||
return (
|
||||
@@ -52,16 +48,14 @@ const AppRoutes = () => {
|
||||
}
|
||||
/>
|
||||
|
||||
{HUMAN_ROUTE_ENABLED && (
|
||||
<Route
|
||||
path="/human"
|
||||
element={
|
||||
<ProtectedRoute requireAuth={true}>
|
||||
<HumanPage />
|
||||
</ProtectedRoute>
|
||||
}
|
||||
/>
|
||||
)}
|
||||
<Route
|
||||
path="/human"
|
||||
element={
|
||||
<ProtectedRoute requireAuth={true}>
|
||||
<HumanPage />
|
||||
</ProtectedRoute>
|
||||
}
|
||||
/>
|
||||
|
||||
<Route path="/intelligence" element={<Navigate to="/settings/intelligence" replace />} />
|
||||
|
||||
|
||||
@@ -7,10 +7,6 @@ import { useCoreState } from '../providers/CoreStateProvider';
|
||||
import { useAppSelector } from '../store/hooks';
|
||||
import { selectUnreadCount } from '../store/notificationSlice';
|
||||
import { isAccountsFullscreen } from '../utils/accountsFullscreen';
|
||||
import { APP_ENVIRONMENT } from '../utils/config';
|
||||
|
||||
/** /human is mascot work-in-progress — only surface it pre-prod. */
|
||||
const HUMAN_TAB_ENABLED = APP_ENVIRONMENT !== 'production';
|
||||
|
||||
const tabs = [
|
||||
{
|
||||
@@ -196,53 +192,51 @@ const BottomTabBar = () => {
|
||||
if (!e.currentTarget.contains(e.relatedTarget as Node)) setRevealed(false);
|
||||
}}>
|
||||
<nav className="pointer-events-auto inline-flex items-center gap-1 rounded-sm border border-stone-300 bg-stone-200 shadow-soft px-1 py-1">
|
||||
{tabs
|
||||
.filter(tab => tab.id !== 'human' || HUMAN_TAB_ENABLED)
|
||||
.map(tab => {
|
||||
const active = isActive(tab.path);
|
||||
const showBadge = tab.id === 'notifications' && unreadCount > 0;
|
||||
// data-walkthrough attributes for the Joyride walkthrough steps.
|
||||
// Maps tab ids to their walkthrough target names.
|
||||
const walkthroughAttr: Record<string, string> = {
|
||||
chat: 'tab-chat',
|
||||
skills: 'tab-skills',
|
||||
notifications: 'tab-notifications',
|
||||
settings: 'tab-settings',
|
||||
};
|
||||
return (
|
||||
<button
|
||||
key={tab.id}
|
||||
data-walkthrough={walkthroughAttr[tab.id]}
|
||||
onClick={() => navigate(tab.path)}
|
||||
className={`group relative flex items-center px-2 py-2 rounded-sm text-sm transition-colors duration-500 ease-[cubic-bezier(0.22,1,0.36,1)] cursor-pointer ${
|
||||
{tabs.map(tab => {
|
||||
const active = isActive(tab.path);
|
||||
const showBadge = tab.id === 'notifications' && unreadCount > 0;
|
||||
// data-walkthrough attributes for the Joyride walkthrough steps.
|
||||
// Maps tab ids to their walkthrough target names.
|
||||
const walkthroughAttr: Record<string, string> = {
|
||||
chat: 'tab-chat',
|
||||
skills: 'tab-skills',
|
||||
notifications: 'tab-notifications',
|
||||
settings: 'tab-settings',
|
||||
};
|
||||
return (
|
||||
<button
|
||||
key={tab.id}
|
||||
data-walkthrough={walkthroughAttr[tab.id]}
|
||||
onClick={() => navigate(tab.path)}
|
||||
className={`group relative flex items-center px-2 py-2 rounded-sm text-sm transition-colors duration-500 ease-[cubic-bezier(0.22,1,0.36,1)] cursor-pointer ${
|
||||
active
|
||||
? 'bg-white text-stone-900 font-semibold shadow-sm'
|
||||
: 'bg-transparent text-stone-500 hover:bg-stone-300/50 hover:text-stone-700'
|
||||
}`}
|
||||
aria-label={
|
||||
tab.id === 'notifications' && unreadCount > 0
|
||||
? `${tab.label} (${unreadCount} unread)`
|
||||
: tab.label
|
||||
}>
|
||||
<span className="relative inline-flex flex-shrink-0">
|
||||
{tab.icon}
|
||||
{showBadge && (
|
||||
<span className="absolute -top-1 -right-1 min-w-[14px] h-[14px] px-1 rounded-full bg-coral-500 text-[9px] font-bold text-white flex items-center justify-center leading-none">
|
||||
{unreadCount > 9 ? '9+' : unreadCount}
|
||||
</span>
|
||||
)}
|
||||
</span>
|
||||
<span
|
||||
className={`overflow-hidden whitespace-nowrap transition-[max-width,margin-left,opacity] duration-500 ease-[cubic-bezier(0.22,1,0.36,1)] ${
|
||||
active
|
||||
? 'bg-white text-stone-900 font-semibold shadow-sm'
|
||||
: 'bg-transparent text-stone-500 hover:bg-stone-300/50 hover:text-stone-700'
|
||||
}`}
|
||||
aria-label={
|
||||
tab.id === 'notifications' && unreadCount > 0
|
||||
? `${tab.label} (${unreadCount} unread)`
|
||||
: tab.label
|
||||
}>
|
||||
<span className="relative inline-flex flex-shrink-0">
|
||||
{tab.icon}
|
||||
{showBadge && (
|
||||
<span className="absolute -top-1 -right-1 min-w-[14px] h-[14px] px-1 rounded-full bg-coral-500 text-[9px] font-bold text-white flex items-center justify-center leading-none">
|
||||
{unreadCount > 9 ? '9+' : unreadCount}
|
||||
</span>
|
||||
)}
|
||||
</span>
|
||||
<span
|
||||
className={`overflow-hidden whitespace-nowrap transition-[max-width,margin-left,opacity] duration-500 ease-[cubic-bezier(0.22,1,0.36,1)] ${
|
||||
active
|
||||
? 'max-w-[160px] ml-2 opacity-100'
|
||||
: 'max-w-0 ml-0 opacity-0 group-hover:max-w-[160px] group-hover:ml-2 group-hover:opacity-100 group-focus-visible:max-w-[160px] group-focus-visible:ml-2 group-focus-visible:opacity-100'
|
||||
}`}>
|
||||
{tab.label}
|
||||
</span>
|
||||
</button>
|
||||
);
|
||||
})}
|
||||
? 'max-w-[160px] ml-2 opacity-100'
|
||||
: 'max-w-0 ml-0 opacity-0 group-hover:max-w-[160px] group-hover:ml-2 group-hover:opacity-100 group-focus-visible:max-w-[160px] group-focus-visible:ml-2 group-focus-visible:opacity-100'
|
||||
}`}>
|
||||
{tab.label}
|
||||
</span>
|
||||
</button>
|
||||
);
|
||||
})}
|
||||
</nav>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
@@ -50,7 +50,7 @@ const HumanPage = () => {
|
||||
{/* Chat sidebar — vertically centered above the BottomTabBar (~80px). */}
|
||||
<div className="absolute right-4 top-0 bottom-20 z-10 flex items-center">
|
||||
<aside className="w-[420px] h-[min(720px,calc(100vh-160px))] rounded-2xl border border-stone-300 bg-white shadow-soft flex flex-col overflow-hidden">
|
||||
<Conversations variant="sidebar" />
|
||||
<Conversations variant="sidebar" composer="mic-cloud" />
|
||||
</aside>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
@@ -0,0 +1,204 @@
|
||||
import { fireEvent, render, screen, waitFor } from '@testing-library/react';
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest';
|
||||
|
||||
import { MicCloudComposer } from './MicCloudComposer';
|
||||
|
||||
// transcribeCloud + encodeBlobToWav are the network/heavy boundaries — mock
|
||||
// them here so we can drive the state machine without touching real APIs.
|
||||
const transcribeCloudMock = vi.fn();
|
||||
const encodeBlobToWavMock = vi.fn();
|
||||
vi.mock('./voice/sttClient', () => ({
|
||||
transcribeCloud: (...args: unknown[]) => transcribeCloudMock(...args),
|
||||
}));
|
||||
vi.mock('./voice/wavEncoder', () => ({
|
||||
encodeBlobToWav: (...args: unknown[]) => encodeBlobToWavMock(...args),
|
||||
}));
|
||||
|
||||
interface FakeRecorder {
|
||||
state: 'inactive' | 'recording' | 'paused';
|
||||
mimeType: string;
|
||||
ondataavailable: ((e: { data: Blob }) => void) | null;
|
||||
onstop: (() => void) | null;
|
||||
start: () => void;
|
||||
stop: () => void;
|
||||
}
|
||||
|
||||
function makeFakeRecorder(mime: string): FakeRecorder {
|
||||
const rec: FakeRecorder = {
|
||||
state: 'inactive',
|
||||
mimeType: mime,
|
||||
ondataavailable: null,
|
||||
onstop: null,
|
||||
start() {
|
||||
rec.state = 'recording';
|
||||
},
|
||||
stop() {
|
||||
rec.state = 'inactive';
|
||||
// Simulate the browser delivering one chunk + the onstop callback.
|
||||
rec.ondataavailable?.({ data: new Blob([new Uint8Array([1, 2, 3])], { type: mime }) });
|
||||
rec.onstop?.();
|
||||
},
|
||||
};
|
||||
return rec;
|
||||
}
|
||||
|
||||
const fakeStream = { getTracks: () => [{ stop: vi.fn() }] } as unknown as MediaStream;
|
||||
|
||||
describe('MicCloudComposer', () => {
|
||||
let recorder: FakeRecorder;
|
||||
let getUserMediaMock: ReturnType<typeof vi.fn>;
|
||||
// Snapshot the descriptor so afterEach can restore it — without this, the
|
||||
// first test that overrides `navigator.mediaDevices` leaks the override
|
||||
// into siblings and makes the suite order-dependent.
|
||||
let originalMediaDevicesDescriptor: PropertyDescriptor | undefined;
|
||||
|
||||
beforeEach(() => {
|
||||
originalMediaDevicesDescriptor = Object.getOwnPropertyDescriptor(
|
||||
globalThis.navigator,
|
||||
'mediaDevices'
|
||||
);
|
||||
transcribeCloudMock.mockReset();
|
||||
encodeBlobToWavMock.mockReset();
|
||||
recorder = makeFakeRecorder('audio/webm;codecs=opus');
|
||||
|
||||
getUserMediaMock = vi.fn().mockResolvedValue(fakeStream);
|
||||
// jsdom's `navigator` is a real object — stub the property in place so
|
||||
// the real prototype chain (React's userAgent reads, etc.) keeps working.
|
||||
Object.defineProperty(globalThis.navigator, 'mediaDevices', {
|
||||
value: { getUserMedia: getUserMediaMock },
|
||||
configurable: true,
|
||||
writable: true,
|
||||
});
|
||||
|
||||
// `new MediaRecorder(...)` requires a real constructor; `vi.fn(() => x)`
|
||||
// returns an object but isn't constructible. Use a class wrapper.
|
||||
class FakeRecorderCtor {
|
||||
constructor() {
|
||||
return recorder as unknown as MediaRecorder;
|
||||
}
|
||||
static isTypeSupported(m: string) {
|
||||
return m.startsWith('audio/webm');
|
||||
}
|
||||
}
|
||||
vi.stubGlobal('MediaRecorder', FakeRecorderCtor);
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
if (originalMediaDevicesDescriptor) {
|
||||
Object.defineProperty(globalThis.navigator, 'mediaDevices', originalMediaDevicesDescriptor);
|
||||
} else {
|
||||
delete (globalThis.navigator as { mediaDevices?: MediaDevices }).mediaDevices;
|
||||
}
|
||||
vi.unstubAllGlobals();
|
||||
});
|
||||
|
||||
it('renders the idle "Tap and speak" state', () => {
|
||||
render(<MicCloudComposer disabled={false} onSubmit={vi.fn()} />);
|
||||
expect(screen.getByText('Tap and speak')).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('shows a "Waiting" label when disabled', () => {
|
||||
render(<MicCloudComposer disabled={true} onSubmit={vi.fn()} />);
|
||||
expect(screen.getByText(/waiting/i)).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('does not start recording when disabled', () => {
|
||||
render(<MicCloudComposer disabled={true} onSubmit={vi.fn()} />);
|
||||
fireEvent.click(screen.getByRole('button', { name: /start recording/i }));
|
||||
expect(getUserMediaMock).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('starts recording on tap, then transcribes + submits on second tap', async () => {
|
||||
transcribeCloudMock.mockResolvedValueOnce('hello world');
|
||||
const onSubmit = vi.fn();
|
||||
const onError = vi.fn();
|
||||
render(<MicCloudComposer disabled={false} onSubmit={onSubmit} onError={onError} />);
|
||||
|
||||
fireEvent.click(screen.getByRole('button', { name: /start recording/i }));
|
||||
await waitFor(() => expect(getUserMediaMock).toHaveBeenCalled());
|
||||
expect(onError).not.toHaveBeenCalled();
|
||||
await waitFor(() =>
|
||||
expect(screen.getByRole('button', { name: /stop recording and send/i })).toBeInTheDocument()
|
||||
);
|
||||
expect(getUserMediaMock).toHaveBeenCalledWith({
|
||||
audio: expect.objectContaining({
|
||||
channelCount: 1,
|
||||
echoCancellation: true,
|
||||
noiseSuppression: true,
|
||||
autoGainControl: true,
|
||||
}),
|
||||
});
|
||||
|
||||
fireEvent.click(screen.getByRole('button', { name: /stop recording and send/i }));
|
||||
await waitFor(() => expect(onSubmit).toHaveBeenCalledWith('hello world'));
|
||||
expect(transcribeCloudMock).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('forwards the language prop to transcribeCloud', async () => {
|
||||
transcribeCloudMock.mockResolvedValueOnce('hi');
|
||||
render(<MicCloudComposer disabled={false} onSubmit={vi.fn()} language="es" />);
|
||||
fireEvent.click(screen.getByRole('button', { name: /start recording/i }));
|
||||
await waitFor(() =>
|
||||
expect(screen.getByRole('button', { name: /stop recording and send/i })).toBeInTheDocument()
|
||||
);
|
||||
fireEvent.click(screen.getByRole('button', { name: /stop recording and send/i }));
|
||||
await waitFor(() => expect(transcribeCloudMock).toHaveBeenCalled());
|
||||
const opts = transcribeCloudMock.mock.calls[0][1];
|
||||
expect(opts).toEqual({ language: 'es' });
|
||||
});
|
||||
|
||||
it('surfaces a permission-denied error via onError', async () => {
|
||||
getUserMediaMock.mockRejectedValueOnce(new Error('NotAllowed'));
|
||||
const onError = vi.fn();
|
||||
render(<MicCloudComposer disabled={false} onSubmit={vi.fn()} onError={onError} />);
|
||||
fireEvent.click(screen.getByRole('button', { name: /start recording/i }));
|
||||
await waitFor(() => expect(onError).toHaveBeenCalledWith(expect.stringMatching(/permission/i)));
|
||||
});
|
||||
|
||||
it('falls back to wav re-encode when the native attempt fails', async () => {
|
||||
transcribeCloudMock
|
||||
.mockRejectedValueOnce(new Error('codec not accepted'))
|
||||
.mockResolvedValueOnce('after fallback');
|
||||
encodeBlobToWavMock.mockResolvedValueOnce(
|
||||
new Blob([new Uint8Array([0])], { type: 'audio/wav' })
|
||||
);
|
||||
const onSubmit = vi.fn();
|
||||
render(<MicCloudComposer disabled={false} onSubmit={onSubmit} />);
|
||||
fireEvent.click(screen.getByRole('button', { name: /start recording/i }));
|
||||
await waitFor(() =>
|
||||
expect(screen.getByRole('button', { name: /stop recording and send/i })).toBeInTheDocument()
|
||||
);
|
||||
fireEvent.click(screen.getByRole('button', { name: /stop recording and send/i }));
|
||||
await waitFor(() => expect(onSubmit).toHaveBeenCalledWith('after fallback'));
|
||||
expect(encodeBlobToWavMock).toHaveBeenCalledTimes(1);
|
||||
expect(transcribeCloudMock).toHaveBeenCalledTimes(2);
|
||||
});
|
||||
|
||||
it('reports an error when transcription returns empty text', async () => {
|
||||
transcribeCloudMock.mockResolvedValueOnce('');
|
||||
const onError = vi.fn();
|
||||
const onSubmit = vi.fn();
|
||||
render(<MicCloudComposer disabled={false} onSubmit={onSubmit} onError={onError} />);
|
||||
fireEvent.click(screen.getByRole('button', { name: /start recording/i }));
|
||||
await waitFor(() =>
|
||||
expect(screen.getByRole('button', { name: /stop recording and send/i })).toBeInTheDocument()
|
||||
);
|
||||
fireEvent.click(screen.getByRole('button', { name: /stop recording and send/i }));
|
||||
await waitFor(() =>
|
||||
expect(onError).toHaveBeenCalledWith(expect.stringMatching(/no speech detected/i))
|
||||
);
|
||||
expect(onSubmit).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('reports a microphone-unavailable error when getUserMedia is missing', () => {
|
||||
Object.defineProperty(globalThis.navigator, 'mediaDevices', {
|
||||
value: undefined,
|
||||
configurable: true,
|
||||
writable: true,
|
||||
});
|
||||
const onError = vi.fn();
|
||||
render(<MicCloudComposer disabled={false} onSubmit={vi.fn()} onError={onError} />);
|
||||
fireEvent.click(screen.getByRole('button', { name: /start recording/i }));
|
||||
expect(onError).toHaveBeenCalledWith(expect.stringMatching(/not available/i));
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,336 @@
|
||||
import debug from 'debug';
|
||||
import { useEffect, useRef, useState } from 'react';
|
||||
|
||||
import { transcribeCloud } from './voice/sttClient';
|
||||
import { encodeBlobToWav } from './voice/wavEncoder';
|
||||
|
||||
const composerLog = debug('human:mic-composer');
|
||||
|
||||
/** MIME types MediaRecorder will be asked to use, in priority order.
|
||||
*
|
||||
* AAC-in-MP4 is preferred because the hosted STT upstream (GMI Whisper)
|
||||
* rejected Opus-in-WebM with "Invalid JSON payload" — AAC is far more
|
||||
* broadly accepted by OpenAI-compatible audio endpoints. We fall through
|
||||
* to WebM/Opus on Chromium builds that haven't shipped MP4 recording, then
|
||||
* to whatever the browser picks by default. */
|
||||
const PREFERRED_MIMES = ['audio/mp4;codecs=mp4a.40.2', 'audio/mp4', 'audio/webm;codecs=opus'];
|
||||
|
||||
function pickRecorderMime(): string {
|
||||
if (typeof MediaRecorder === 'undefined') return '';
|
||||
for (const mime of PREFERRED_MIMES) {
|
||||
if (MediaRecorder.isTypeSupported(mime)) return mime;
|
||||
}
|
||||
return '';
|
||||
}
|
||||
|
||||
export interface MicCloudComposerProps {
|
||||
/** Disabled while a turn is in flight or the welcome message is pending. */
|
||||
disabled: boolean;
|
||||
/** Receives the transcribed text — same callback the textarea send uses. */
|
||||
onSubmit: (text: string) => Promise<void> | void;
|
||||
/** Surfaced when the mic flow fails so the parent can show a banner. */
|
||||
onError?: (message: string) => void;
|
||||
/** ISO 639-1 language hint forwarded to Scribe. Defaults to `'en'` —
|
||||
* passing a hint is meaningfully more accurate than auto-detect on
|
||||
* short utterances. Set to empty string to let Scribe auto-detect. */
|
||||
language?: string;
|
||||
}
|
||||
|
||||
type RecordingState = 'idle' | 'recording' | 'transcribing';
|
||||
|
||||
/**
|
||||
* Tap-to-toggle mic composer for the mascot page. Captures audio via the
|
||||
* browser's `MediaRecorder`, hands the resulting Blob to the cloud STT proxy
|
||||
* (`openhuman.voice_cloud_transcribe`), then forwards the transcript through
|
||||
* `onSubmit` so it joins the agent's normal send pipeline.
|
||||
*
|
||||
* Single button, single decision: tap once to start recording, tap again to
|
||||
* stop and send. No textarea — that's the whole point of the mascot tab.
|
||||
*/
|
||||
export function MicCloudComposer({
|
||||
disabled,
|
||||
onSubmit,
|
||||
onError,
|
||||
language = 'en',
|
||||
}: MicCloudComposerProps) {
|
||||
const [state, setState] = useState<RecordingState>('idle');
|
||||
const recorderRef = useRef<MediaRecorder | null>(null);
|
||||
const streamRef = useRef<MediaStream | null>(null);
|
||||
const chunksRef = useRef<Blob[]>([]);
|
||||
// Tracks unmount so async callbacks (recorder.onstop, finalizeRecording)
|
||||
// don't fire setState/onSubmit on a dead component — without this, the
|
||||
// user navigating away mid-recording can dispatch an unintended message.
|
||||
const disposedRef = useRef(false);
|
||||
// Guards against rapid re-taps during the `getUserMedia` permission prompt.
|
||||
// Without this, two awaited `getUserMedia` calls can resolve back-to-back
|
||||
// and leave one of the granted streams orphaned (mic indicator stuck on).
|
||||
const startInFlightRef = useRef(false);
|
||||
|
||||
// If the component unmounts mid-record, release the mic so the OS indicator
|
||||
// doesn't get stuck on.
|
||||
useEffect(() => {
|
||||
disposedRef.current = false;
|
||||
return () => {
|
||||
disposedRef.current = true;
|
||||
// Detach onstop first — `recorder.stop()` below is what would fire it,
|
||||
// and we don't want finalizeRecording running post-unmount.
|
||||
if (recorderRef.current) recorderRef.current.onstop = null;
|
||||
stopStream();
|
||||
try {
|
||||
recorderRef.current?.stop();
|
||||
} catch {
|
||||
// recorder may already be inactive
|
||||
}
|
||||
recorderRef.current = null;
|
||||
};
|
||||
}, []);
|
||||
|
||||
function stopStream() {
|
||||
if (streamRef.current) {
|
||||
for (const track of streamRef.current.getTracks()) {
|
||||
try {
|
||||
track.stop();
|
||||
} catch {
|
||||
// already stopped
|
||||
}
|
||||
}
|
||||
streamRef.current = null;
|
||||
}
|
||||
}
|
||||
|
||||
async function startRecording() {
|
||||
if (state !== 'idle' || disabled || startInFlightRef.current) return;
|
||||
if (typeof navigator === 'undefined' || !navigator.mediaDevices?.getUserMedia) {
|
||||
onError?.('Microphone access is not available in this runtime.');
|
||||
return;
|
||||
}
|
||||
startInFlightRef.current = true;
|
||||
|
||||
let stream: MediaStream;
|
||||
try {
|
||||
// Audio constraints tuned for STT accuracy:
|
||||
// - mono: Scribe processes a single channel, stereo just doubles upload
|
||||
// - 48kHz: matches Opus's native rate, no resample artifacts
|
||||
// - {echo,noise,gain}: huge accuracy win on real-world mic input
|
||||
// (untreated room noise + low-volume speech is the #1 reason
|
||||
// transcription drops words in our flow)
|
||||
stream = await navigator.mediaDevices.getUserMedia({
|
||||
audio: {
|
||||
channelCount: 1,
|
||||
sampleRate: 48000,
|
||||
echoCancellation: true,
|
||||
noiseSuppression: true,
|
||||
autoGainControl: true,
|
||||
},
|
||||
});
|
||||
} catch (err) {
|
||||
startInFlightRef.current = false;
|
||||
const msg = err instanceof Error ? err.message : String(err);
|
||||
composerLog('getUserMedia rejected: %s', msg);
|
||||
onError?.(`Microphone permission denied: ${msg}`);
|
||||
return;
|
||||
}
|
||||
|
||||
// Component unmounted while waiting for permission — release the granted
|
||||
// stream instead of leaking it (mic indicator would otherwise stay on).
|
||||
if (disposedRef.current) {
|
||||
startInFlightRef.current = false;
|
||||
stream.getTracks().forEach(t => t.stop());
|
||||
return;
|
||||
}
|
||||
|
||||
const mime = pickRecorderMime();
|
||||
// 128kbps Opus is well above the threshold where Scribe's accuracy
|
||||
// plateaus; MediaRecorder's default for voice can be as low as 32kbps,
|
||||
// which audibly muddies consonants.
|
||||
const recorderOptions: MediaRecorderOptions = { audioBitsPerSecond: 128_000 };
|
||||
if (mime) recorderOptions.mimeType = mime;
|
||||
let recorder: MediaRecorder;
|
||||
try {
|
||||
recorder = new MediaRecorder(stream, recorderOptions);
|
||||
} catch (err) {
|
||||
stream.getTracks().forEach(t => t.stop());
|
||||
startInFlightRef.current = false;
|
||||
const msg = err instanceof Error ? err.message : String(err);
|
||||
onError?.(`Failed to start recorder: ${msg}`);
|
||||
return;
|
||||
}
|
||||
|
||||
chunksRef.current = [];
|
||||
recorder.ondataavailable = (e: BlobEvent) => {
|
||||
if (e.data && e.data.size > 0) chunksRef.current.push(e.data);
|
||||
};
|
||||
recorder.onstop = () => {
|
||||
void finalizeRecording();
|
||||
};
|
||||
|
||||
streamRef.current = stream;
|
||||
recorderRef.current = recorder;
|
||||
recorder.start();
|
||||
setState('recording');
|
||||
startInFlightRef.current = false;
|
||||
composerLog('recording started mime=%s', recorder.mimeType || '(default)');
|
||||
}
|
||||
|
||||
function stopRecording() {
|
||||
const recorder = recorderRef.current;
|
||||
if (!recorder || recorder.state === 'inactive') return;
|
||||
setState('transcribing');
|
||||
try {
|
||||
recorder.stop();
|
||||
} catch (err) {
|
||||
// If `stop()` throws, `onstop` never fires → finalizeRecording never
|
||||
// resets `state`, leaving the UI stuck on "Transcribing…". Recover here.
|
||||
composerLog('recorder.stop threw: %s', err);
|
||||
const msg = err instanceof Error ? err.message : String(err);
|
||||
onError?.(`Failed to stop recording: ${msg}`);
|
||||
stopStream();
|
||||
recorderRef.current = null;
|
||||
setState('idle');
|
||||
}
|
||||
}
|
||||
|
||||
async function finalizeRecording() {
|
||||
// Component was torn down mid-recording — clean up resources without
|
||||
// touching React state (which would log a warning) or `onSubmit`
|
||||
// (which would dispatch a message to a thread the user has left).
|
||||
if (disposedRef.current) {
|
||||
stopStream();
|
||||
recorderRef.current = null;
|
||||
chunksRef.current = [];
|
||||
return;
|
||||
}
|
||||
const recorder = recorderRef.current;
|
||||
recorderRef.current = null;
|
||||
stopStream();
|
||||
const chunks = chunksRef.current;
|
||||
chunksRef.current = [];
|
||||
|
||||
const mime = recorder?.mimeType || 'audio/webm';
|
||||
const blob = new Blob(chunks, { type: mime });
|
||||
composerLog('recording stopped chunks=%d bytes=%d', chunks.length, blob.size);
|
||||
|
||||
if (blob.size === 0) {
|
||||
setState('idle');
|
||||
onError?.('No audio captured. Try holding the mic a little longer.');
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
const transcript = await transcribeWithFallback(blob);
|
||||
if (!transcript) {
|
||||
onError?.('No speech detected. Try again.');
|
||||
setState('idle');
|
||||
return;
|
||||
}
|
||||
await onSubmit(transcript);
|
||||
} catch (err) {
|
||||
const msg = err instanceof Error ? err.message : String(err);
|
||||
composerLog('transcribe failed: %s', msg);
|
||||
onError?.(`Voice transcription failed: ${msg}`);
|
||||
} finally {
|
||||
setState('idle');
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Send the recorder's native blob first (Opus-in-WebM ~3KB/sec) — Scribe
|
||||
* accepts it natively and it uploads ~30x faster than the 16kHz mono WAV
|
||||
* we used to transcode (~32KB/sec). If that ever fails (older STT
|
||||
* provider behind a feature flag, codec mismatch, …), retry once with a
|
||||
* re-encoded WAV so we don't regress correctness for the speed win.
|
||||
*/
|
||||
async function transcribeWithFallback(blob: Blob): Promise<string> {
|
||||
const startedAt = Date.now();
|
||||
const opts = language ? { language } : undefined;
|
||||
try {
|
||||
composerLog(
|
||||
'transcribe attempt=native bytes=%d mime=%s lang=%s',
|
||||
blob.size,
|
||||
blob.type,
|
||||
language || 'auto'
|
||||
);
|
||||
const text = await transcribeCloud(blob, opts);
|
||||
composerLog('transcribe ok attempt=native ms=%d', Math.round(Date.now() - startedAt));
|
||||
return text;
|
||||
} catch (err) {
|
||||
const msg = err instanceof Error ? err.message : String(err);
|
||||
composerLog('transcribe failed attempt=native — falling back to wav: %s', msg);
|
||||
const reEncodeStart = Date.now();
|
||||
const wav = await encodeBlobToWav(blob);
|
||||
composerLog(
|
||||
'wav fallback bytes=%d encode_ms=%d',
|
||||
wav.size,
|
||||
Math.round(Date.now() - reEncodeStart)
|
||||
);
|
||||
const text = await transcribeCloud(wav, opts);
|
||||
composerLog(
|
||||
'transcribe ok attempt=wav-fallback total_ms=%d',
|
||||
Math.round(Date.now() - startedAt)
|
||||
);
|
||||
return text;
|
||||
}
|
||||
}
|
||||
|
||||
const isRecording = state === 'recording';
|
||||
const isBusy = state === 'transcribing';
|
||||
const buttonDisabled = disabled || isBusy;
|
||||
|
||||
const label = isBusy
|
||||
? 'Transcribing…'
|
||||
: isRecording
|
||||
? 'Tap to send'
|
||||
: disabled
|
||||
? 'Waiting for the agent…'
|
||||
: 'Tap and speak';
|
||||
|
||||
return (
|
||||
<div className="flex items-center justify-center gap-3">
|
||||
<button
|
||||
type="button"
|
||||
aria-label={isRecording ? 'Stop recording and send' : 'Start recording'}
|
||||
onClick={() => (isRecording ? stopRecording() : void startRecording())}
|
||||
disabled={buttonDisabled}
|
||||
className={`relative w-14 h-14 flex items-center justify-center rounded-full text-white shadow-soft transition-colors disabled:opacity-40 disabled:cursor-not-allowed ${
|
||||
isRecording ? 'bg-coral-500 hover:bg-coral-400' : 'bg-primary-500 hover:bg-primary-600'
|
||||
}`}>
|
||||
{isRecording && (
|
||||
<span className="absolute inset-0 rounded-full bg-coral-500/40 animate-ping" />
|
||||
)}
|
||||
{isBusy ? (
|
||||
<svg className="w-5 h-5 animate-spin" fill="none" viewBox="0 0 24 24">
|
||||
<circle
|
||||
className="opacity-25"
|
||||
cx="12"
|
||||
cy="12"
|
||||
r="10"
|
||||
stroke="currentColor"
|
||||
strokeWidth="4"
|
||||
/>
|
||||
<path
|
||||
className="opacity-75"
|
||||
fill="currentColor"
|
||||
d="M4 12a8 8 0 018-8V0C5.373 0 0 5.373 0 12h4z"
|
||||
/>
|
||||
</svg>
|
||||
) : (
|
||||
<svg
|
||||
className="relative w-6 h-6"
|
||||
fill="none"
|
||||
stroke="currentColor"
|
||||
strokeWidth={1.8}
|
||||
viewBox="0 0 24 24">
|
||||
<path
|
||||
strokeLinecap="round"
|
||||
strokeLinejoin="round"
|
||||
d="M12 18.75a6 6 0 006-6v-1.5m-6 7.5a6 6 0 01-6-6v-1.5m6 7.5v3.75m-3.75 0h7.5M12 15.75a3 3 0 01-3-3V4.5a3 3 0 116 0v8.25a3 3 0 01-3 3z"
|
||||
/>
|
||||
</svg>
|
||||
)}
|
||||
</button>
|
||||
<span className="text-xs text-stone-500 select-none">{label}</span>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
export default MicCloudComposer;
|
||||
@@ -0,0 +1,95 @@
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest';
|
||||
|
||||
import { callCoreRpc } from '../../../services/coreRpcClient';
|
||||
import { transcribeCloud } from './sttClient';
|
||||
|
||||
vi.mock('../../../services/coreRpcClient', () => ({ callCoreRpc: vi.fn() }));
|
||||
|
||||
describe('transcribeCloud', () => {
|
||||
beforeEach(() => {
|
||||
(callCoreRpc as ReturnType<typeof vi.fn>).mockReset();
|
||||
});
|
||||
it('routes through openhuman.voice_cloud_transcribe with base64 + mime', async () => {
|
||||
const mock = callCoreRpc as ReturnType<typeof vi.fn>;
|
||||
mock.mockResolvedValueOnce({ text: 'hello there' });
|
||||
const blob = new Blob([new Uint8Array([1, 2, 3, 4, 5])], { type: 'audio/webm;codecs=opus' });
|
||||
|
||||
const text = await transcribeCloud(blob);
|
||||
|
||||
expect(text).toBe('hello there');
|
||||
expect(mock).toHaveBeenCalledTimes(1);
|
||||
const call = mock.mock.calls[0][0] as {
|
||||
method: string;
|
||||
params: { audio_base64: string; mime_type: string; file_name: string };
|
||||
};
|
||||
expect(call.method).toBe('openhuman.voice_cloud_transcribe');
|
||||
// `audio/webm;codecs=opus` should collapse to the bare type the backend
|
||||
// allow-list accepts.
|
||||
expect(call.params.mime_type).toBe('audio/webm');
|
||||
expect(call.params.file_name).toBe('audio.webm');
|
||||
expect(call.params.audio_base64).toBe(btoa('\x01\x02\x03\x04\x05'));
|
||||
});
|
||||
|
||||
it('rejects empty blobs without hitting the core', async () => {
|
||||
const mock = callCoreRpc as ReturnType<typeof vi.fn>;
|
||||
const blob = new Blob([], { type: 'audio/webm' });
|
||||
await expect(transcribeCloud(blob)).rejects.toThrow(/empty/);
|
||||
expect(mock).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('forwards the optional model + language hints', async () => {
|
||||
const mock = callCoreRpc as ReturnType<typeof vi.fn>;
|
||||
mock.mockResolvedValueOnce({ text: 'hi' });
|
||||
const blob = new Blob([new Uint8Array([9])], { type: 'audio/webm' });
|
||||
|
||||
await transcribeCloud(blob, { model: 'scribe_v1', language: 'en' });
|
||||
const params = mock.mock.calls[0][0].params as Record<string, unknown>;
|
||||
expect(params.model).toBe('scribe_v1');
|
||||
expect(params.language).toBe('en');
|
||||
});
|
||||
|
||||
it('trims whitespace off the returned transcript', async () => {
|
||||
const mock = callCoreRpc as ReturnType<typeof vi.fn>;
|
||||
mock.mockResolvedValueOnce({ text: ' spacey ' });
|
||||
const blob = new Blob([new Uint8Array([1])], { type: 'audio/webm' });
|
||||
expect(await transcribeCloud(blob)).toBe('spacey');
|
||||
});
|
||||
|
||||
// Per-mime extension heuristic — the upstream STT provider sniffs the file
|
||||
// extension when the container isn't unambiguous, so each branch matters.
|
||||
it.each([
|
||||
['audio/webm', 'audio.webm'],
|
||||
['video/webm', 'audio.webm'],
|
||||
['audio/ogg', 'audio.ogg'],
|
||||
['audio/mpeg', 'audio.mp3'],
|
||||
['audio/wav', 'audio.wav'],
|
||||
['audio/x-wav', 'audio.wav'],
|
||||
['audio/mp4', 'audio.m4a'],
|
||||
['audio/x-m4a', 'audio.m4a'],
|
||||
['audio/flac', 'audio.flac'],
|
||||
['application/octet-stream', 'audio.webm'],
|
||||
])('derives file_name for mime %s -> %s', async (mime, expected) => {
|
||||
const mock = callCoreRpc as ReturnType<typeof vi.fn>;
|
||||
mock.mockResolvedValueOnce({ text: 'ok' });
|
||||
const blob = new Blob([new Uint8Array([1])], { type: mime });
|
||||
await transcribeCloud(blob);
|
||||
const params = mock.mock.calls[0][0].params as Record<string, unknown>;
|
||||
expect(params.file_name).toBe(expected);
|
||||
});
|
||||
|
||||
it('honors an explicit fileName override', async () => {
|
||||
const mock = callCoreRpc as ReturnType<typeof vi.fn>;
|
||||
mock.mockResolvedValueOnce({ text: 'ok' });
|
||||
const blob = new Blob([new Uint8Array([1])], { type: 'audio/webm' });
|
||||
await transcribeCloud(blob, { fileName: 'custom-clip.webm', mimeType: 'audio/webm' });
|
||||
const params = mock.mock.calls[0][0].params as Record<string, unknown>;
|
||||
expect(params.file_name).toBe('custom-clip.webm');
|
||||
});
|
||||
|
||||
it('returns empty string when backend response has no text field', async () => {
|
||||
const mock = callCoreRpc as ReturnType<typeof vi.fn>;
|
||||
mock.mockResolvedValueOnce({});
|
||||
const blob = new Blob([new Uint8Array([1])], { type: 'audio/webm' });
|
||||
expect(await transcribeCloud(blob)).toBe('');
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,102 @@
|
||||
import debug from 'debug';
|
||||
|
||||
import { callCoreRpc } from '../../../services/coreRpcClient';
|
||||
|
||||
const sttLog = debug('human:stt');
|
||||
|
||||
export interface CloudTranscribeOptions {
|
||||
/** Override the backend STT model id. Default is whatever the backend
|
||||
* resolves `whisper-v1` to today. */
|
||||
model?: string;
|
||||
/** BCP-47 language hint, e.g. `'en'`. */
|
||||
language?: string;
|
||||
/** Defaults derived from the recorded blob. */
|
||||
mimeType?: string;
|
||||
fileName?: string;
|
||||
}
|
||||
|
||||
export interface CloudTranscribeResult {
|
||||
text: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Transcribe a recorded audio blob via the Rust core's cloud STT proxy.
|
||||
*
|
||||
* The blob is read into a base64 string and shipped over JSON-RPC; the core
|
||||
* decodes it and POSTs `multipart/form-data` to the hosted backend's
|
||||
* `/openai/v1/audio/transcriptions` endpoint. Going through the core keeps
|
||||
* the provider API key off the desktop app and reuses the same auth flow as
|
||||
* `synthesizeSpeech`.
|
||||
*/
|
||||
export async function transcribeCloud(
|
||||
blob: Blob,
|
||||
opts: CloudTranscribeOptions = {}
|
||||
): Promise<string> {
|
||||
if (!blob || blob.size === 0) {
|
||||
throw new Error('audio blob is empty');
|
||||
}
|
||||
const encodeStart = Date.now();
|
||||
const audio_base64 = await blobToBase64(blob);
|
||||
const encodeMs = Math.round(Date.now() - encodeStart);
|
||||
|
||||
const params: Record<string, unknown> = { audio_base64 };
|
||||
// MediaRecorder mime types include codec parameters (e.g. `audio/webm;codecs=opus`)
|
||||
// — the backend's allow-list expects the bare type, so strip the suffix.
|
||||
const mime = (opts.mimeType ?? blob.type ?? 'audio/webm').split(';')[0].trim() || 'audio/webm';
|
||||
params.mime_type = mime;
|
||||
params.file_name = opts.fileName ?? `audio.${guessExtension(mime)}`;
|
||||
if (opts.model) params.model = opts.model;
|
||||
if (opts.language) params.language = opts.language;
|
||||
|
||||
sttLog(
|
||||
'transcribe bytes=%d mime=%s base64_ms=%d (b64_size=%d)',
|
||||
blob.size,
|
||||
mime,
|
||||
encodeMs,
|
||||
audio_base64.length
|
||||
);
|
||||
|
||||
const rpcStart = Date.now();
|
||||
const result = await callCoreRpc<CloudTranscribeResult>({
|
||||
method: 'openhuman.voice_cloud_transcribe',
|
||||
params,
|
||||
});
|
||||
const text = result?.text?.trim() ?? '';
|
||||
sttLog('transcribed chars=%d rpc_ms=%d', text.length, Math.round(Date.now() - rpcStart));
|
||||
return text;
|
||||
}
|
||||
|
||||
async function blobToBase64(blob: Blob): Promise<string> {
|
||||
const buf = await blob.arrayBuffer();
|
||||
const bytes = new Uint8Array(buf);
|
||||
// Chunked to avoid `Maximum call stack` on large clips when spread into
|
||||
// String.fromCharCode in one go.
|
||||
const CHUNK = 0x8000;
|
||||
let binary = '';
|
||||
for (let i = 0; i < bytes.length; i += CHUNK) {
|
||||
binary += String.fromCharCode(...bytes.subarray(i, i + CHUNK));
|
||||
}
|
||||
return btoa(binary);
|
||||
}
|
||||
|
||||
function guessExtension(mime: string): string {
|
||||
switch (mime) {
|
||||
case 'audio/webm':
|
||||
case 'video/webm':
|
||||
return 'webm';
|
||||
case 'audio/ogg':
|
||||
return 'ogg';
|
||||
case 'audio/mpeg':
|
||||
return 'mp3';
|
||||
case 'audio/wav':
|
||||
case 'audio/x-wav':
|
||||
return 'wav';
|
||||
case 'audio/mp4':
|
||||
case 'audio/x-m4a':
|
||||
return 'm4a';
|
||||
case 'audio/flac':
|
||||
return 'flac';
|
||||
default:
|
||||
return 'webm';
|
||||
}
|
||||
}
|
||||
@@ -1,7 +1,12 @@
|
||||
import { describe, expect, it, vi } from 'vitest';
|
||||
|
||||
import { callCoreRpc } from '../../../services/coreRpcClient';
|
||||
import { proceduralVisemes, synthesizeSpeech, visemesFromAlignment } from './ttsClient';
|
||||
import {
|
||||
prepareForSpeech,
|
||||
proceduralVisemes,
|
||||
synthesizeSpeech,
|
||||
visemesFromAlignment,
|
||||
} from './ttsClient';
|
||||
|
||||
vi.mock('../../../services/coreRpcClient', () => ({ callCoreRpc: vi.fn() }));
|
||||
|
||||
@@ -16,19 +21,19 @@ describe('synthesizeSpeech (core RPC)', () => {
|
||||
const r = await synthesizeSpeech('hello', { voiceId: 'v1', modelId: 'm1' });
|
||||
expect(mock).toHaveBeenCalledWith({
|
||||
method: 'openhuman.voice_reply_synthesize',
|
||||
params: { text: 'hello', voice_id: 'v1', model_id: 'm1' },
|
||||
params: { text: 'hello.', voice_id: 'v1', model_id: 'm1' },
|
||||
});
|
||||
expect(r.audio_base64).toBe('AAA=');
|
||||
expect(r.visemes).toHaveLength(1);
|
||||
});
|
||||
|
||||
it('omits options that were not provided', async () => {
|
||||
it('falls back to the configured mascot voice when no voiceId is given', async () => {
|
||||
const mock = callCoreRpc as ReturnType<typeof vi.fn>;
|
||||
mock.mockResolvedValueOnce({ audio_base64: 'BBB=', audio_mime: 'audio/mpeg', visemes: [] });
|
||||
await synthesizeSpeech('hi');
|
||||
expect(mock).toHaveBeenCalledWith({
|
||||
method: 'openhuman.voice_reply_synthesize',
|
||||
params: { text: 'hi' },
|
||||
params: { text: 'hi.', voice_id: 'ljX1ZrXuDIIRVcmiVSyR' },
|
||||
});
|
||||
});
|
||||
|
||||
@@ -39,6 +44,48 @@ describe('synthesizeSpeech (core RPC)', () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe('prepareForSpeech', () => {
|
||||
it('inserts an ellipsis pause between paragraphs so the voice beats between thoughts', () => {
|
||||
expect(prepareForSpeech('First thought.\n\nSecond thought.')).toBe(
|
||||
'First thought. ... Second thought.'
|
||||
);
|
||||
});
|
||||
|
||||
it('strips markdown emphasis, headings, lists, and quotes', () => {
|
||||
const md = '# Title\n\n- one\n- two\n\n**bold** and _italic_ and `code`.';
|
||||
const out = prepareForSpeech(md);
|
||||
expect(out).not.toMatch(/[#*_`-]/);
|
||||
expect(out).toContain('bold and italic and code.');
|
||||
expect(out).toContain('one');
|
||||
expect(out).toContain('two');
|
||||
});
|
||||
|
||||
it('drops fenced code blocks and replaces bare URLs with a stand-in', () => {
|
||||
const md = 'See ```js\nconst x = 1;\n``` and visit https://example.com for more.';
|
||||
const out = prepareForSpeech(md);
|
||||
expect(out).not.toContain('const x');
|
||||
expect(out).not.toContain('https://');
|
||||
expect(out).toContain('a link');
|
||||
});
|
||||
|
||||
it('keeps the label of a markdown link and discards the URL', () => {
|
||||
expect(prepareForSpeech('See [the docs](https://example.com).')).toBe('See the docs.');
|
||||
});
|
||||
|
||||
it('appends a terminator when the message ends mid-thought', () => {
|
||||
expect(prepareForSpeech('hello there')).toBe('hello there.');
|
||||
// Already terminated → leave it.
|
||||
expect(prepareForSpeech('hello there!')).toBe('hello there!');
|
||||
expect(prepareForSpeech('hello there?')).toBe('hello there?');
|
||||
});
|
||||
|
||||
it('treats a single newline as a soft wrap, not a pause', () => {
|
||||
expect(prepareForSpeech('one line\nstill the same sentence.')).toBe(
|
||||
'one line still the same sentence.'
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
describe('visemesFromAlignment', () => {
|
||||
it('returns empty for empty input', () => {
|
||||
expect(visemesFromAlignment([])).toEqual([]);
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
import debug from 'debug';
|
||||
|
||||
import { callCoreRpc } from '../../../services/coreRpcClient';
|
||||
import { MASCOT_VOICE_ID } from '../../../utils/config';
|
||||
|
||||
const ttsLog = debug('human:tts');
|
||||
|
||||
@@ -45,11 +46,18 @@ export interface TtsOptions {
|
||||
* issues and keeps auth in one place.
|
||||
*/
|
||||
export async function synthesizeSpeech(text: string, opts: TtsOptions = {}): Promise<TtsResponse> {
|
||||
const params: Record<string, unknown> = { text };
|
||||
if (opts.voiceId) params.voice_id = opts.voiceId;
|
||||
const voiceId = opts.voiceId ?? MASCOT_VOICE_ID;
|
||||
// `prepareForSpeech` collapses to '' on replies that are pure code/markdown
|
||||
// formatting. The core RPC rejects empty text, which would propagate as a
|
||||
// visible error for what was effectively a no-op reply. Fall back to the
|
||||
// raw trimmed text, then to a single ellipsis (so the mascot just exhales)
|
||||
// before letting an empty payload reach the upstream.
|
||||
const spoken = prepareForSpeech(text) || text.trim() || '...';
|
||||
const params: Record<string, unknown> = { text: spoken };
|
||||
if (voiceId) params.voice_id = voiceId;
|
||||
if (opts.modelId) params.model_id = opts.modelId;
|
||||
if (opts.outputFormat) params.output_format = opts.outputFormat;
|
||||
ttsLog('synthesize chars=%d voice=%s', text.length, opts.voiceId ?? 'default');
|
||||
ttsLog('synthesize chars=%d (raw=%d) voice=%s', spoken.length, text.length, voiceId ?? 'default');
|
||||
|
||||
const result = await callCoreRpc<TtsResponse>({
|
||||
method: 'openhuman.voice_reply_synthesize',
|
||||
@@ -103,6 +111,50 @@ export function visemesFromAlignment(alignment: AlignmentFrame[]): VisemeFrame[]
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Reshape an assistant message into something the TTS engine can read with
|
||||
* natural cadence. The agent's reply is markdown — raw `**bold**`, headings,
|
||||
* code fences, link syntax, and `\n\n` paragraph breaks all confuse
|
||||
* ElevenLabs' prosody model and collapse the pauses between sentences. We
|
||||
* strip the formatting and translate paragraph boundaries into an explicit
|
||||
* `...` pause, which ElevenLabs honors as a beat between thoughts.
|
||||
*
|
||||
* Exported for tests so the mapping can be pinned without going through the
|
||||
* full RPC stack.
|
||||
*/
|
||||
export function prepareForSpeech(raw: string): string {
|
||||
let s = raw ?? '';
|
||||
// Drop fenced code blocks entirely — reading symbols out loud is painful and
|
||||
// they almost never carry the intent of the reply.
|
||||
s = s.replace(/```[\s\S]*?```/g, ' ');
|
||||
// Inline code → keep the contents, drop the backticks.
|
||||
s = s.replace(/`([^`]+)`/g, '$1');
|
||||
// Markdown links `[label](url)` → just the label.
|
||||
s = s.replace(/\[([^\]]+)\]\(([^)]+)\)/g, '$1');
|
||||
// Bare URLs read terribly — replace with a short stand-in.
|
||||
s = s.replace(/https?:\/\/\S+/g, 'a link');
|
||||
// Headings, blockquotes, list bullets at line start.
|
||||
s = s.replace(/^\s{0,3}#{1,6}\s+/gm, '');
|
||||
s = s.replace(/^\s{0,3}>\s?/gm, '');
|
||||
s = s.replace(/^\s*[-*+]\s+/gm, '');
|
||||
s = s.replace(/^\s*\d+\.\s+/gm, '');
|
||||
// Emphasis markers — keep the words, drop the wrappers.
|
||||
s = s.replace(/(\*\*|__)(.*?)\1/g, '$2');
|
||||
s = s.replace(/(\*|_)(.*?)\1/g, '$2');
|
||||
// Convert paragraph breaks into an explicit ellipsis pause before we collapse
|
||||
// whitespace, otherwise the double newline becomes a single space.
|
||||
s = s.replace(/\n{2,}/g, ' ... ');
|
||||
// Single newlines inside a paragraph are just soft wraps in markdown.
|
||||
s = s.replace(/\n+/g, ' ');
|
||||
// Ensure a sentence terminator at the very end so the voice doesn't trail
|
||||
// upward like an unfinished thought.
|
||||
s = s.trim();
|
||||
if (s.length > 0 && !/[.!?…]$/.test(s)) s += '.';
|
||||
// Collapse any runs of whitespace introduced by the substitutions above.
|
||||
s = s.replace(/[ \t]{2,}/g, ' ');
|
||||
return s;
|
||||
}
|
||||
|
||||
function alignmentLetterToCode(chunk: string): string {
|
||||
const ch = chunk.replace(/[^a-zA-Z]/g, '').slice(-1);
|
||||
return letterToOculusViseme(ch);
|
||||
|
||||
@@ -0,0 +1,110 @@
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest';
|
||||
|
||||
import { encodeBlobToWav } from './wavEncoder';
|
||||
|
||||
// jsdom doesn't ship Web Audio. We only need a thin stub that lets the
|
||||
// encoder reach the WAV header + sample copy paths — the actual decode +
|
||||
// resample is tested as a black box (input bytes → WAV bytes round-trip).
|
||||
|
||||
interface FakeAudioBuffer {
|
||||
sampleRate: number;
|
||||
length: number;
|
||||
numberOfChannels: number;
|
||||
getChannelData(c: number): Float32Array;
|
||||
}
|
||||
|
||||
function createFakeBuffer(sampleRate: number, channels: Float32Array[]): FakeAudioBuffer {
|
||||
return {
|
||||
sampleRate,
|
||||
length: channels[0].length,
|
||||
numberOfChannels: channels.length,
|
||||
getChannelData: (c: number) => channels[c],
|
||||
};
|
||||
}
|
||||
|
||||
describe('encodeBlobToWav', () => {
|
||||
let decodedBuffer: FakeAudioBuffer;
|
||||
|
||||
beforeEach(() => {
|
||||
// Default: stereo at 48kHz so we exercise both the resample-via-render
|
||||
// path and the mono mixdown.
|
||||
decodedBuffer = createFakeBuffer(48_000, [
|
||||
new Float32Array([0, 0.5, -0.5, 1, -1]),
|
||||
new Float32Array([0, 0.5, -0.5, 1, -1]),
|
||||
]);
|
||||
|
||||
class FakeOfflineAudioContext {
|
||||
constructor(
|
||||
public numberOfChannels: number,
|
||||
public length: number,
|
||||
public sampleRate: number
|
||||
) {}
|
||||
decodeAudioData = vi.fn(async () => decodedBuffer as unknown as AudioBuffer);
|
||||
createBufferSource() {
|
||||
return { buffer: null as AudioBuffer | null, connect: vi.fn(), start: vi.fn() };
|
||||
}
|
||||
destination = {} as AudioNode;
|
||||
startRendering = vi.fn(async () => {
|
||||
// Resampled buffer at the constructor's target sample rate (16kHz),
|
||||
// mono. Use a tiny known signal so we can assert the WAV bytes.
|
||||
return createFakeBuffer(this.sampleRate, [
|
||||
new Float32Array([0, 0.5, -0.5]),
|
||||
]) as unknown as AudioBuffer;
|
||||
});
|
||||
}
|
||||
|
||||
vi.stubGlobal('OfflineAudioContext', FakeOfflineAudioContext);
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.unstubAllGlobals();
|
||||
});
|
||||
|
||||
it('rejects empty blobs', async () => {
|
||||
const blob = new Blob([], { type: 'audio/webm' });
|
||||
await expect(encodeBlobToWav(blob)).rejects.toThrow(/empty/);
|
||||
});
|
||||
|
||||
it('produces a 16kHz mono WAV blob with a valid RIFF/WAVE header', async () => {
|
||||
const input = new Blob([new Uint8Array([1, 2, 3, 4])], { type: 'audio/webm' });
|
||||
const out = await encodeBlobToWav(input);
|
||||
|
||||
expect(out.type).toBe('audio/wav');
|
||||
const buf = await out.arrayBuffer();
|
||||
const view = new DataView(buf);
|
||||
const decoder = new TextDecoder('ascii');
|
||||
expect(decoder.decode(buf.slice(0, 4))).toBe('RIFF');
|
||||
expect(decoder.decode(buf.slice(8, 12))).toBe('WAVE');
|
||||
expect(decoder.decode(buf.slice(12, 16))).toBe('fmt ');
|
||||
// PCM format = 1, mono = 1 channel, 16kHz, 16-bit
|
||||
expect(view.getUint16(20, true)).toBe(1);
|
||||
expect(view.getUint16(22, true)).toBe(1);
|
||||
expect(view.getUint32(24, true)).toBe(16_000);
|
||||
expect(view.getUint16(34, true)).toBe(16);
|
||||
expect(decoder.decode(buf.slice(36, 40))).toBe('data');
|
||||
});
|
||||
|
||||
it('skips the resample render pass when the source is already at 16kHz', async () => {
|
||||
decodedBuffer = createFakeBuffer(16_000, [new Float32Array([0, 0.25, -0.25])]);
|
||||
const input = new Blob([new Uint8Array([9])], { type: 'audio/wav' });
|
||||
const out = await encodeBlobToWav(input);
|
||||
// 3 samples × 2 bytes/sample + 44-byte header
|
||||
expect(out.size).toBe(44 + 6);
|
||||
const view = new DataView(await out.arrayBuffer());
|
||||
// Sample at offset 44 (first sample) should be 0
|
||||
expect(view.getInt16(44, true)).toBe(0);
|
||||
// setInt16 truncates toward zero rather than rounding, so 0.25 * 0x7fff
|
||||
// (= 8191.75) lands at 8191 in the file. Pin the truncation behavior
|
||||
// explicitly so a future "let's round" change has to flag this.
|
||||
expect(view.getInt16(46, true)).toBe(Math.trunc(0.25 * 0x7fff));
|
||||
expect(view.getInt16(48, true)).toBe(Math.trunc(-0.25 * 0x8000));
|
||||
});
|
||||
|
||||
it('clamps samples that drift outside [-1, 1] from accumulator rounding', async () => {
|
||||
decodedBuffer = createFakeBuffer(16_000, [new Float32Array([1.5, -1.5])]);
|
||||
const input = new Blob([new Uint8Array([1])], { type: 'audio/wav' });
|
||||
const view = new DataView(await (await encodeBlobToWav(input)).arrayBuffer());
|
||||
expect(view.getInt16(44, true)).toBe(0x7fff); // clamped to +1
|
||||
expect(view.getInt16(46, true)).toBe(-0x8000); // clamped to -1
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,122 @@
|
||||
/**
|
||||
* Re-encode a recorded audio blob (any container the browser's
|
||||
* `decodeAudioData` understands — WebM/Opus, MP4/AAC, OGG, …) to a
|
||||
* **16kHz mono 16-bit PCM WAV** blob.
|
||||
*
|
||||
* Why this exists: the hosted STT upstream (GMI Whisper) rejects
|
||||
* Opus-in-WebM payloads with "Invalid JSON payload", and Chromium-based
|
||||
* runtimes (including the CEF webview Tauri ships) don't reliably support
|
||||
* `MediaRecorder` with MP4. WAV at Whisper's native 16kHz is the most
|
||||
* portable thing we can hand the backend without standing up an ffmpeg
|
||||
* dependency in the desktop app.
|
||||
*
|
||||
* Implementation: `OfflineAudioContext` decodes + resamples in one pass,
|
||||
* then we mix to mono and write a standard RIFF/WAVE header in front of
|
||||
* the 16-bit little-endian samples. Synchronous after the decode promise
|
||||
* resolves so we can pipe it straight into the STT client.
|
||||
*/
|
||||
|
||||
const TARGET_SAMPLE_RATE = 16_000;
|
||||
|
||||
export async function encodeBlobToWav(blob: Blob): Promise<Blob> {
|
||||
if (!blob || blob.size === 0) {
|
||||
throw new Error('audio blob is empty');
|
||||
}
|
||||
const arrayBuffer = await blob.arrayBuffer();
|
||||
// `decodeAudioData` consumes the buffer, so use a copy if the caller
|
||||
// happens to reuse `blob` afterwards.
|
||||
const decoded = await decodeToBuffer(arrayBuffer.slice(0));
|
||||
const mono = mixDownToMono(decoded);
|
||||
const wav = buildWav(mono, TARGET_SAMPLE_RATE);
|
||||
return new Blob([wav], { type: 'audio/wav' });
|
||||
}
|
||||
|
||||
/**
|
||||
* Decode arbitrary compressed audio into an `AudioBuffer` at
|
||||
* `TARGET_SAMPLE_RATE`. Uses `OfflineAudioContext` so the resample
|
||||
* happens during decode rather than via a separate render step.
|
||||
*/
|
||||
async function decodeToBuffer(arrayBuffer: ArrayBuffer): Promise<AudioBuffer> {
|
||||
// OfflineAudioContext requires concrete length/channels up front, but
|
||||
// `decodeAudioData` returns a buffer at the source rate. Trick: decode
|
||||
// with a throwaway `AudioContext`, then render through an OfflineAC at
|
||||
// 16kHz to perform the resample.
|
||||
const tmp = new OfflineAudioContext(1, 1, TARGET_SAMPLE_RATE);
|
||||
const decoded = await tmp.decodeAudioData(arrayBuffer);
|
||||
|
||||
if (decoded.sampleRate === TARGET_SAMPLE_RATE) {
|
||||
return decoded;
|
||||
}
|
||||
|
||||
const offline = new OfflineAudioContext(
|
||||
decoded.numberOfChannels,
|
||||
Math.ceil((decoded.length * TARGET_SAMPLE_RATE) / decoded.sampleRate),
|
||||
TARGET_SAMPLE_RATE
|
||||
);
|
||||
const source = offline.createBufferSource();
|
||||
source.buffer = decoded;
|
||||
source.connect(offline.destination);
|
||||
source.start(0);
|
||||
return offline.startRendering();
|
||||
}
|
||||
|
||||
function mixDownToMono(buffer: AudioBuffer): Float32Array {
|
||||
if (buffer.numberOfChannels === 1) {
|
||||
return buffer.getChannelData(0);
|
||||
}
|
||||
const length = buffer.length;
|
||||
const mono = new Float32Array(length);
|
||||
const channels: Float32Array[] = [];
|
||||
for (let c = 0; c < buffer.numberOfChannels; c++) {
|
||||
channels.push(buffer.getChannelData(c));
|
||||
}
|
||||
for (let i = 0; i < length; i++) {
|
||||
let sum = 0;
|
||||
for (let c = 0; c < channels.length; c++) sum += channels[c][i];
|
||||
mono[i] = sum / channels.length;
|
||||
}
|
||||
return mono;
|
||||
}
|
||||
|
||||
function buildWav(samples: Float32Array, sampleRate: number): ArrayBuffer {
|
||||
const bytesPerSample = 2; // 16-bit PCM
|
||||
const numChannels = 1;
|
||||
const dataBytes = samples.length * bytesPerSample;
|
||||
const buffer = new ArrayBuffer(44 + dataBytes);
|
||||
const view = new DataView(buffer);
|
||||
|
||||
// RIFF chunk descriptor
|
||||
writeString(view, 0, 'RIFF');
|
||||
view.setUint32(4, 36 + dataBytes, true);
|
||||
writeString(view, 8, 'WAVE');
|
||||
|
||||
// fmt sub-chunk (PCM)
|
||||
writeString(view, 12, 'fmt ');
|
||||
view.setUint32(16, 16, true); // sub-chunk size
|
||||
view.setUint16(20, 1, true); // PCM format
|
||||
view.setUint16(22, numChannels, true);
|
||||
view.setUint32(24, sampleRate, true);
|
||||
view.setUint32(28, sampleRate * numChannels * bytesPerSample, true); // byte rate
|
||||
view.setUint16(32, numChannels * bytesPerSample, true); // block align
|
||||
view.setUint16(34, bytesPerSample * 8, true); // bits per sample
|
||||
|
||||
// data sub-chunk
|
||||
writeString(view, 36, 'data');
|
||||
view.setUint32(40, dataBytes, true);
|
||||
|
||||
let offset = 44;
|
||||
for (let i = 0; i < samples.length; i++, offset += 2) {
|
||||
// Clamp + scale to signed 16-bit. Reverse-clipping protects against
|
||||
// floats slightly outside [-1, 1] from accumulator rounding.
|
||||
const s = Math.max(-1, Math.min(1, samples[i]));
|
||||
view.setInt16(offset, s < 0 ? s * 0x8000 : s * 0x7fff, true);
|
||||
}
|
||||
|
||||
return buffer;
|
||||
}
|
||||
|
||||
function writeString(view: DataView, offset: number, value: string) {
|
||||
for (let i = 0; i < value.length; i++) {
|
||||
view.setUint8(offset + i, value.charCodeAt(i));
|
||||
}
|
||||
}
|
||||
@@ -10,6 +10,7 @@ import PillTabBar from '../components/PillTabBar';
|
||||
import UpsellBanner from '../components/upsell/UpsellBanner';
|
||||
import { dismissBanner, shouldShowBanner } from '../components/upsell/upsellDismissState';
|
||||
import UsageLimitModal from '../components/upsell/UsageLimitModal';
|
||||
import MicCloudComposer from '../features/human/MicCloudComposer';
|
||||
// [#1123] Commented out — welcome-agent onboarding replaced by Joyride walkthrough
|
||||
// import { ONBOARDING_WELCOME_THREAD_LABEL } from '../constants/onboardingChat';
|
||||
import { useStickToBottom } from '../hooks/useStickToBottom';
|
||||
@@ -86,6 +87,14 @@ interface ConversationsProps {
|
||||
* another page (e.g. /accounts).
|
||||
*/
|
||||
variant?: 'page' | 'sidebar';
|
||||
/**
|
||||
* Composer mode. `text` (default) uses the textarea + send button.
|
||||
* `mic-cloud` swaps the entire composer for a single mic button that
|
||||
* captures audio via `MediaRecorder`, transcribes it through the cloud
|
||||
* STT proxy, then routes the transcript through the same send path.
|
||||
* Used by the mascot tab so the only interaction is voice.
|
||||
*/
|
||||
composer?: 'text' | 'mic-cloud';
|
||||
}
|
||||
|
||||
export function isComposerInteractionBlocked(args: {
|
||||
@@ -122,7 +131,7 @@ export function isComposerInteractionBlocked(args: {
|
||||
// );
|
||||
// }
|
||||
|
||||
const Conversations = ({ variant = 'page' }: ConversationsProps = {}) => {
|
||||
const Conversations = ({ variant = 'page', composer = 'text' }: ConversationsProps = {}) => {
|
||||
const dispatch = useAppDispatch();
|
||||
const navigate = useNavigate();
|
||||
const {
|
||||
@@ -1584,7 +1593,16 @@ const Conversations = ({ variant = 'page' }: ConversationsProps = {}) => {
|
||||
</div>
|
||||
)}
|
||||
|
||||
{inputMode === 'text' ? (
|
||||
{composer === 'mic-cloud' ? (
|
||||
<MicCloudComposer
|
||||
// Without `!selectedThreadId`, a mic submit before a thread is
|
||||
// ready hits `handleSendMessage`'s early return and the
|
||||
// transcript is silently dropped — the user spoke into the void.
|
||||
disabled={composerInteractionBlocked || !selectedThreadId}
|
||||
onSubmit={text => handleSendMessage(text)}
|
||||
onError={message => setSendError(chatSendError('voice_transcription', message))}
|
||||
/>
|
||||
) : inputMode === 'text' ? (
|
||||
<div className="flex items-end gap-3">
|
||||
<div className="relative flex flex-1 items-center justify-center rounded-xl border border-stone-200 bg-white transition-all focus-within:border-primary-500/50 focus-within:ring-1 focus-within:ring-primary-500/50">
|
||||
<div
|
||||
|
||||
@@ -129,6 +129,7 @@ vi.mock('../utils/config', () => ({
|
||||
LATEST_APP_DOWNLOAD_URL: 'https://github.com/tinyhumansai/openhuman/releases/latest',
|
||||
APP_VERSION: '0.0.0-test',
|
||||
DEV_JWT_TOKEN: undefined,
|
||||
MASCOT_VOICE_ID: 'ljX1ZrXuDIIRVcmiVSyR',
|
||||
}));
|
||||
|
||||
vi.mock('../services/backendUrl', () => ({
|
||||
|
||||
@@ -139,3 +139,11 @@ export const LATEST_APP_DOWNLOAD_URL =
|
||||
* Sentry pipeline end-to-end. Has no effect in normal builds.
|
||||
*/
|
||||
export const SENTRY_SMOKE_TEST = import.meta.env.VITE_SENTRY_SMOKE_TEST === 'true';
|
||||
|
||||
/**
|
||||
* ElevenLabs voice ID used for the mascot's reply speech. Picked to sound
|
||||
* like a friendly cartoon character rather than a human narrator. Override
|
||||
* with `VITE_MASCOT_VOICE_ID` to A/B alternative voices without a code change.
|
||||
*/
|
||||
export const MASCOT_VOICE_ID =
|
||||
(import.meta.env.VITE_MASCOT_VOICE_ID as string | undefined)?.trim() || 'ljX1ZrXuDIIRVcmiVSyR';
|
||||
|
||||
@@ -183,6 +183,22 @@ impl BackendOAuthClient {
|
||||
Ok(Self { client, base })
|
||||
}
|
||||
|
||||
/// Borrow the underlying `reqwest::Client` for callers that need to
|
||||
/// drive a non-JSON request shape (e.g. `multipart/form-data` uploads
|
||||
/// for cloud STT) without re-implementing TLS/proxy plumbing.
|
||||
pub fn raw_client(&self) -> &Client {
|
||||
&self.client
|
||||
}
|
||||
|
||||
/// Resolve a backend-relative path against the configured base URL.
|
||||
/// Mirrors what `authed_json` does internally so callers using
|
||||
/// `raw_client()` don't have to assemble URLs by hand.
|
||||
pub fn url_for(&self, path: &str) -> Result<Url> {
|
||||
self.base
|
||||
.join(path.trim_start_matches('/'))
|
||||
.with_context(|| format!("build URL for {path}"))
|
||||
}
|
||||
|
||||
/// Returns the URL for initiating a login flow for a specific provider.
|
||||
pub fn login_url(&self, provider: &str) -> Result<Url> {
|
||||
let p = provider.trim().trim_matches('/');
|
||||
|
||||
@@ -0,0 +1,176 @@
|
||||
//! Cloud speech-to-text — proxies the hosted backend's
|
||||
//! `/openai/v1/audio/transcriptions` endpoint so the desktop UI can transcribe
|
||||
//! mic input without shipping a provider API key. Mirrors the shape of
|
||||
//! `reply_speech.rs`, but uploads multipart form data instead of JSON.
|
||||
//!
|
||||
//! Used by the mascot's mic-only composer (`HumanPage`) — recording is
|
||||
//! captured via `MediaRecorder` in the renderer, base64-encoded, then sent
|
||||
//! through this RPC. The transcribed text is fed straight into the agent's
|
||||
//! existing send pipeline.
|
||||
|
||||
use base64::{engine::general_purpose::STANDARD as BASE64, Engine};
|
||||
use log::debug;
|
||||
use reqwest::header::AUTHORIZATION;
|
||||
use reqwest::multipart::{Form, Part};
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use crate::api::config::effective_api_url;
|
||||
use crate::api::jwt::get_session_token;
|
||||
use crate::api::BackendOAuthClient;
|
||||
use crate::openhuman::config::Config;
|
||||
use crate::rpc::RpcOutcome;
|
||||
|
||||
const LOG_PREFIX: &str = "[voice_cloud_stt]";
|
||||
|
||||
/// Default model id sent to the backend. The backend's controller currently
|
||||
/// resolves this to whichever provider it has configured for audio
|
||||
/// transcription (today: GMI Whisper). Callers can override.
|
||||
const DEFAULT_MODEL: &str = "whisper-v1";
|
||||
|
||||
/// Caller-tunable knobs.
|
||||
#[derive(Debug, Default, Clone)]
|
||||
pub struct CloudTranscribeOptions {
|
||||
pub model: Option<String>,
|
||||
pub language: Option<String>,
|
||||
pub mime_type: Option<String>,
|
||||
/// Original file name hint (e.g. `audio.webm`). Some upstream providers
|
||||
/// sniff the extension; without one we fall back to `audio.webm`.
|
||||
pub file_name: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct CloudTranscribeResult {
|
||||
pub text: String,
|
||||
}
|
||||
|
||||
/// Decode + upload audio bytes to the backend STT endpoint.
|
||||
///
|
||||
/// `audio_base64` is what comes off the wire from the renderer — keeping the
|
||||
/// UI side base64 means we don't have to reach for a binary RPC channel.
|
||||
pub async fn transcribe_cloud(
|
||||
config: &Config,
|
||||
audio_base64: &str,
|
||||
opts: &CloudTranscribeOptions,
|
||||
) -> Result<RpcOutcome<CloudTranscribeResult>, String> {
|
||||
let trimmed = audio_base64.trim();
|
||||
if trimmed.is_empty() {
|
||||
return Err("audio_base64 is required".to_string());
|
||||
}
|
||||
let audio_bytes = BASE64
|
||||
.decode(trimmed)
|
||||
.map_err(|e| format!("invalid base64 audio: {e}"))?;
|
||||
if audio_bytes.is_empty() {
|
||||
return Err("decoded audio is empty".to_string());
|
||||
}
|
||||
|
||||
let token = get_session_token(config)
|
||||
.map_err(|e| e.to_string())?
|
||||
.and_then(|t| {
|
||||
let s = t.trim().to_string();
|
||||
if s.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(s)
|
||||
}
|
||||
})
|
||||
.ok_or_else(|| "no backend session token; sign in first".to_string())?;
|
||||
|
||||
let api_url = effective_api_url(&config.api_url);
|
||||
let client = BackendOAuthClient::new(&api_url).map_err(|e| e.to_string())?;
|
||||
let url = client
|
||||
.url_for("/openai/v1/audio/transcriptions")
|
||||
.map_err(|e| e.to_string())?;
|
||||
|
||||
let mime = opts
|
||||
.mime_type
|
||||
.as_deref()
|
||||
.map(str::trim)
|
||||
.filter(|s| !s.is_empty())
|
||||
.unwrap_or("audio/webm")
|
||||
.to_string();
|
||||
let file_name = opts
|
||||
.file_name
|
||||
.as_deref()
|
||||
.map(str::trim)
|
||||
.filter(|s| !s.is_empty())
|
||||
.unwrap_or("audio.webm")
|
||||
.to_string();
|
||||
let model = opts
|
||||
.model
|
||||
.as_deref()
|
||||
.map(str::trim)
|
||||
.filter(|s| !s.is_empty())
|
||||
.unwrap_or(DEFAULT_MODEL)
|
||||
.to_string();
|
||||
|
||||
let bytes_len = audio_bytes.len();
|
||||
let part = Part::bytes(audio_bytes)
|
||||
.file_name(file_name.clone())
|
||||
.mime_str(&mime)
|
||||
.map_err(|e| format!("invalid mime '{mime}': {e}"))?;
|
||||
|
||||
let mut form = Form::new().part("file", part).text("model", model.clone());
|
||||
if let Some(lang) = opts
|
||||
.language
|
||||
.as_deref()
|
||||
.map(str::trim)
|
||||
.filter(|s| !s.is_empty())
|
||||
{
|
||||
form = form.text("language", lang.to_string());
|
||||
}
|
||||
|
||||
debug!(
|
||||
"{LOG_PREFIX} POST {} mime={} bytes={} model={}",
|
||||
url.path(),
|
||||
mime,
|
||||
bytes_len,
|
||||
model
|
||||
);
|
||||
|
||||
let upload_started = std::time::Instant::now();
|
||||
let response = client
|
||||
.raw_client()
|
||||
.post(url.clone())
|
||||
.header(AUTHORIZATION, format!("Bearer {token}"))
|
||||
.multipart(form)
|
||||
.send()
|
||||
.await
|
||||
.map_err(|e| format!("backend transcription request failed: {e}"))?;
|
||||
|
||||
let status = response.status();
|
||||
let body = response
|
||||
.text()
|
||||
.await
|
||||
.map_err(|e| format!("read transcription response failed: {e}"))?;
|
||||
let upload_ms = upload_started.elapsed().as_millis();
|
||||
debug!(
|
||||
"{LOG_PREFIX} backend responded status={} upload_round_trip_ms={} body_bytes={}",
|
||||
status,
|
||||
upload_ms,
|
||||
body.len()
|
||||
);
|
||||
if !status.is_success() {
|
||||
return Err(format!(
|
||||
"POST /openai/v1/audio/transcriptions failed ({status}): {body}"
|
||||
));
|
||||
}
|
||||
|
||||
let parsed: serde_json::Value = serde_json::from_str(&body)
|
||||
.map_err(|e| format!("parse transcription response failed: {e}; body={body}"))?;
|
||||
// A 200 with no string `text` field is a backend contract break — surface
|
||||
// it as an error rather than swallowing it as a successful empty
|
||||
// transcription, which would look to the caller like "no speech detected".
|
||||
let text = parsed
|
||||
.get("text")
|
||||
.and_then(|v| v.as_str())
|
||||
.ok_or_else(|| format!("transcription response missing string `text`: {body}"))?
|
||||
.trim()
|
||||
.to_string();
|
||||
|
||||
debug!("{LOG_PREFIX} transcribed chars={}", text.len());
|
||||
|
||||
Ok(RpcOutcome::single_log(
|
||||
CloudTranscribeResult { text },
|
||||
"cloud STT via POST /openai/v1/audio/transcriptions",
|
||||
))
|
||||
}
|
||||
@@ -6,6 +6,7 @@
|
||||
|
||||
pub mod audio_capture;
|
||||
pub(crate) mod cli;
|
||||
pub mod cloud_transcribe;
|
||||
pub mod dictation_listener;
|
||||
pub mod hallucination;
|
||||
pub mod hotkey;
|
||||
|
||||
@@ -44,6 +44,19 @@ struct TtsParams {
|
||||
output_path: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
struct CloudTranscribeParams {
|
||||
audio_base64: String,
|
||||
#[serde(default)]
|
||||
mime_type: Option<String>,
|
||||
#[serde(default)]
|
||||
file_name: Option<String>,
|
||||
#[serde(default)]
|
||||
model: Option<String>,
|
||||
#[serde(default)]
|
||||
language: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
struct ReplySynthesizeParams {
|
||||
text: String,
|
||||
@@ -84,6 +97,7 @@ pub fn all_voice_controller_schemas() -> Vec<ControllerSchema> {
|
||||
voice_schemas("voice_transcribe_bytes"),
|
||||
voice_schemas("voice_tts"),
|
||||
voice_schemas("voice_reply_synthesize"),
|
||||
voice_schemas("voice_cloud_transcribe"),
|
||||
voice_schemas("voice_server_start"),
|
||||
voice_schemas("voice_server_stop"),
|
||||
voice_schemas("voice_server_status"),
|
||||
@@ -113,6 +127,10 @@ pub fn all_voice_registered_controllers() -> Vec<RegisteredController> {
|
||||
schema: voice_schemas("voice_reply_synthesize"),
|
||||
handler: handle_voice_reply_synthesize,
|
||||
},
|
||||
RegisteredController {
|
||||
schema: voice_schemas("voice_cloud_transcribe"),
|
||||
handler: handle_voice_cloud_transcribe,
|
||||
},
|
||||
RegisteredController {
|
||||
schema: voice_schemas("voice_server_start"),
|
||||
handler: handle_voice_server_start,
|
||||
@@ -213,6 +231,24 @@ pub fn voice_schemas(function: &str) -> ControllerSchema {
|
||||
"ReplySpeechResult: { audio_base64, audio_mime, visemes, alignment? }.",
|
||||
)],
|
||||
},
|
||||
"voice_cloud_transcribe" => ControllerSchema {
|
||||
namespace: "voice",
|
||||
function: "cloud_transcribe",
|
||||
description:
|
||||
"Transcribe audio bytes via the hosted backend's STT endpoint. Used by the \
|
||||
mascot's mic-only composer so we don't ship a provider API key in the desktop app.",
|
||||
inputs: vec![
|
||||
required_string(
|
||||
"audio_base64",
|
||||
"Base64-encoded audio bytes (e.g. webm/opus from MediaRecorder).",
|
||||
),
|
||||
optional_string("mime_type", "Audio MIME type (default: audio/webm)."),
|
||||
optional_string("file_name", "Original filename hint (default: audio.webm)."),
|
||||
optional_string("model", "Backend STT model id (default: whisper-v1)."),
|
||||
optional_string("language", "BCP-47 language hint, e.g. 'en'."),
|
||||
],
|
||||
outputs: vec![json_output("result", "CloudTranscribeResult: { text }.")],
|
||||
},
|
||||
"voice_server_start" => ControllerSchema {
|
||||
namespace: "voice",
|
||||
function: "server_start",
|
||||
@@ -344,6 +380,27 @@ fn handle_voice_reply_synthesize(params: Map<String, Value>) -> ControllerFuture
|
||||
})
|
||||
}
|
||||
|
||||
fn handle_voice_cloud_transcribe(params: Map<String, Value>) -> ControllerFuture {
|
||||
Box::pin(async move {
|
||||
let config = config_rpc::load_config_with_timeout().await?;
|
||||
let p = deserialize_params::<CloudTranscribeParams>(params)?;
|
||||
let opts = crate::openhuman::voice::cloud_transcribe::CloudTranscribeOptions {
|
||||
model: p.model,
|
||||
language: p.language,
|
||||
mime_type: p.mime_type,
|
||||
file_name: p.file_name,
|
||||
};
|
||||
to_json(
|
||||
crate::openhuman::voice::cloud_transcribe::transcribe_cloud(
|
||||
&config,
|
||||
&p.audio_base64,
|
||||
&opts,
|
||||
)
|
||||
.await?,
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
fn handle_voice_server_start(params: Map<String, Value>) -> ControllerFuture {
|
||||
Box::pin(async move {
|
||||
use crate::openhuman::voice::hotkey::ActivationMode;
|
||||
|
||||
Reference in New Issue
Block a user