Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
34 commits
Select commit Hold shift + click to select a range
6e8ef4c
feat(voice): configure independent speech endpoints
johnmatthewtennant Sep 24, 2026
1ee0a32
fix(voice): guard dictation against custom realtime keys
johnmatthewtennant Sep 24, 2026
0b9c6dc
test(voice): cover independent endpoint settings in browser
johnmatthewtennant Sep 28, 2026
7efdf64
fix(voice): clarify per-endpoint key guidance
johnmatthewtennant Sep 28, 2026
3c98c06
Save each voice endpoint URL and key together
johnmatthewtennant Sep 28, 2026
40b8b8c
Clarify saved voice endpoint key masking
johnmatthewtennant Sep 28, 2026
b5d06ea
Keep spoken assistant replies in one transcript card
johnmatthewtennant Sep 28, 2026
df35a6d
fix(voice): secure endpoint URLs and reset saved overrides
johnmatthewtennant Sep 28, 2026
9572cf7
fix(voice): use the default OpenAI key for dictation
johnmatthewtennant Sep 28, 2026
6d47cb8
fix(voice): refresh endpoint key state across settings changes
johnmatthewtennant Sep 28, 2026
639eb8b
fix(voice): pass keychain account without needless borrow
johnmatthewtennant Sep 28, 2026
9632d10
Set Siri synthesis rate explicitly at 1x
johnmatthewtennant Sep 29, 2026
b59365e
Return accepted handoff tool output from Berd Call CLI
johnmatthewtennant Sep 29, 2026
9d586cb
fix(voice): serialize endpoint setting updates
johnmatthewtennant Sep 29, 2026
b145509
fix(voice): clarify Spanish endpoint key guidance
johnmatthewtennant Sep 29, 2026
6b9a036
refactor(voice): name fixed Realtime dictation credential
johnmatthewtennant Sep 29, 2026
a15d575
refactor(call): return recorded live event to handoff caller
johnmatthewtennant Sep 29, 2026
cf2f8d4
fix(call): keep endpoint parser formatting stable across platforms
johnmatthewtennant Sep 29, 2026
0d6003b
refactor(voice): share endpoint URL security policy
johnmatthewtennant Sep 29, 2026
8092e6e
refactor(voice): reuse settings-changed event name
johnmatthewtennant Sep 29, 2026
fbc98bd
refactor(voice): simplify key status and stabilize endpoint test
johnmatthewtennant Sep 29, 2026
10f931b
fix(call): order shared endpoint imports for Windows fmt
johnmatthewtennant Sep 29, 2026
0647279
refactor(voice): distinguish selected and default Realtime keys
johnmatthewtennant Sep 29, 2026
3ee4611
refactor(voice): name key-status freshness and update test fixture
johnmatthewtennant Sep 29, 2026
fb0c768
test(voice): describe endpoint save concurrency invariant
johnmatthewtennant Sep 29, 2026
f675b87
chore: retrigger automated review
johnmatthewtennant Sep 30, 2026
3ebd16c
fix(voice): reject key mutations for stale endpoint selections
johnmatthewtennant Oct 2, 2026
c2f14aa
test(voice): verify call-scoped endpoint routing with local services
johnmatthewtennant Oct 2, 2026
3db762d
Keep voice endpoint credentials paired during startup
johnmatthewtennant Oct 2, 2026
4763610
Preserve legacy environment-routed voice credentials
johnmatthewtennant Oct 5, 2026
f7f81f4
Refresh Realtime endpoint controls after voice reset
johnmatthewtennant Oct 5, 2026
be814e0
Move spoken reply card fix to its dedicated PR
johnmatthewtennant Oct 5, 2026
967fad5
Keep transient voice endpoint URLs out of restart preferences
johnmatthewtennant Oct 5, 2026
e2b182b
Preserve explicit default voice endpoints over legacy environment rou…
johnmatthewtennant Oct 5, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions playwright.config.ts
Original file line number Diff line number Diff line change
Expand Up @@ -85,6 +85,7 @@ export default defineConfig({
name: "voice-conversation",
testMatch: [
"**/voice-conversation.spec.ts",
"**/voice-endpoints.spec.ts",
"**/voice-settings-visual.spec.ts",
],
use: {
Expand Down
126 changes: 126 additions & 0 deletions scripts/verify-voice-endpoint-routing.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,126 @@
# /// script
# dependencies = ["websockets>=16,<17"]
# ///
"""Exercise berd-call URL overrides with loopback services and a PCM test host."""
import argparse
import asyncio
import json
import os
import socket
import struct
from pathlib import Path
from tempfile import TemporaryDirectory
from urllib.parse import urlsplit
from websockets.exceptions import ConnectionClosed
from websockets.asyncio.server import serve

async def main(binary):
observed = []
sandbox = TemporaryDirectory(prefix="berd-voice-routing-")
config = Path(sandbox.name) / "config" / "openai-voice-endpoints.json"
config.parent.mkdir()
saved_settings = b'{"stt":"wss://saved.example/stt","tts":"https://saved.example/tts","realtime":"wss://saved.example/realtime"}'
config.write_bytes(saved_settings)
async def websocket(ws):
observed.append(urlsplit(ws.request.path).path)
assert ws.request.headers["Authorization"] == "Bearer disposable-routing-test"
try:
async for raw in ws:
event = json.loads(raw)
if event.get('type') == 'session.update':
session = event['session']
await ws.send(json.dumps({'type': 'session.updated', 'session': {
'model': 'test-model', 'audio': session.get('audio', {})
}}))
except ConnectionClosed:
pass
async def synthesis(reader, writer):
headers = await reader.readuntil(b'\r\n\r\n')
length = next(int(line.split(b':', 1)[1]) for line in headers.split(b'\r\n') if line.lower().startswith(b'content-length:'))
body = json.loads(await reader.readexactly(length))
assert body['input'] == 'A lighthouse guides the boat home.'
observed.append(headers.split(b' ')[1].decode())
pcm = b'\x00\x00' * 2400
writer.write(b'HTTP/1.1 200 OK\r\nContent-Type: application/octet-stream\r\nContent-Length: ' + str(len(pcm)).encode() + b'\r\nConnection: close\r\n\r\n' + pcm)
await writer.drain()
writer.close()
async with serve(websocket, '127.0.0.1', 0) as ws_server:
http = await asyncio.start_server(synthesis, '127.0.0.1', 0)
ws_port = ws_server.sockets[0].getsockname()[1]
http_port = http.sockets[0].getsockname()[1]
async def run(options, speak=False, defaults=False):
child_audio, host_audio = socket.socketpair()
env = dict(os.environ, OPENAI_API_KEY='disposable-routing-test', OPENAI_REALTIME_MODEL='test-model', OPENAI_REALTIME_ENDPOINT='ws://127.0.0.1:1/unused', OPENAI_BASE_URL='http://127.0.0.1:1/unused')
env['GOOSE_PATH_ROOT'] = sandbox.name
if defaults:
env['OPENAI_REALTIME_ENDPOINT'] = f'ws://127.0.0.1:{ws_port}/default-stt'
env['OPENAI_BASE_URL'] = f'http://127.0.0.1:{http_port}/default'
process = await asyncio.create_subprocess_exec(str(binary), 'session', '--pcm-output-fd', str(child_audio.fileno()), *options, env=env, pass_fds=(child_audio.fileno(),), stdin=asyncio.subprocess.PIPE, stdout=asyncio.subprocess.PIPE, stderr=asyncio.subprocess.PIPE)
child_audio.close()
async def send(value):
payload = json.dumps(value).encode()
process.stdin.write(b'BV\x03\x01' + struct.pack('<I', len(payload)) + payload)
await process.stdin.drain()
async def pcm_host():
reader, writer = await asyncio.open_connection(sock=host_audio)
current = 0
try:
while True:
head = await reader.readexactly(8)
assert head[:3] == b'BA\x03'
payload = await reader.readexactly(struct.unpack('<I', head[4:])[0])
speech_id = struct.unpack('<Q', payload[:8])[0]
if head[3] == 1:
current = 0
await send({'type':'audio_begin_accepted','speech_id':speech_id})
elif head[3] == 2:
seq = struct.unpack('<Q',payload[8:16])[0]
current += (len(payload)-16)//4
await send({'type':'audio_chunk_accepted','speech_id':speech_id,'sequence':seq})
await send({'type':'audio_played','speech_id':speech_id,'played_frames':current})
elif head[3] == 3:
seq = struct.unpack('<Q',payload[8:16])[0]
await send({'type':'audio_drained','speech_id':speech_id,'sequence':seq,'played_frames':current})
elif head[3] == 4:
await send({'type':'audio_cancelled','speech_id':speech_id,'played_frames':current})
except asyncio.IncompleteReadError:
pass
finally:
writer.close()
audio_task = asyncio.create_task(pcm_host())
try:
await send({'type':'hello','id':1,'input_during_tts':'allow_barge_in'})
ready = json.loads(await asyncio.wait_for(process.stdout.readline(), 10))
assert ready['type'] == 'ready', ready
if speak:
await send({'type':'prepare_speak','id':2,'acknowledgement':None,'text':'A lighthouse guides the boat home.'})
while True:
event = json.loads(await asyncio.wait_for(process.stdout.readline(), 10))
if event['type'] == 'admitted':
await send({'type':'output_ready','id':2,'speech_id':event['speech_id']})
if event['type'] == 'speech_completed':
break
assert event['type'] not in ('fatal','speech_failed','not_admitted'), event
await send({'type':'shutdown'})
await asyncio.wait_for(process.wait(), 10)
assert process.returncode == 0, (await process.stderr.read()).decode()
finally:
if process.returncode is None:
process.kill()
await process.wait()
await audio_task
await run(['--tts-backend','openai','--stt-backend','openai','--stt-url',f'ws://127.0.0.1:{ws_port}/stt','--tts-url',f'http://127.0.0.1:{http_port}/tts'], True)
await run(['--mode','expert-spokesperson','--tts-backend','openai','--realtime-url',f'ws://127.0.0.1:{ws_port}/realtime'])
await run(['--tts-backend','openai','--stt-backend','openai'], True, True)
http.close()
await http.wait_closed()
assert sorted(observed) == ['/default-stt','/default/audio/speech','/realtime','/stt','/tts'], observed
assert config.read_bytes() == saved_settings
sandbox.cleanup()
print(json.dumps({'requests':observed,'speech':'completed','savedSettings':'unchanged','nextSession':'uses environment defaults','audio':'PCM test host, no microphone or speakers'}, indent=2))

if __name__ == '__main__':
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('--binary', type=Path, required=True)
args = parser.parse_args()
asyncio.run(main(args.binary.resolve()))
1 change: 1 addition & 0 deletions src-tauri/Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions src-tauri/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -97,6 +97,7 @@ windows-sys = { version = "0.59", features = [
] }

[target.'cfg(target_os = "macos")'.dependencies]
security-framework = "3.7"
block2 = "0.6"
coreaudio-rs = "0.14.2"
keyring = { version = "3.6.3", default-features = false, features = ["apple-native"] }
Expand Down
5 changes: 5 additions & 0 deletions src-tauri/crates/berd-call/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -111,8 +111,13 @@ berd-call session --tts-backend openai --rate 1.0
berd-call session --tts-backend pocket --model-dir /path/to/native-voice-v2 --voice george --rate 1.0
berd-call session --stt-backend parakeet --stt-model-dir /path/to/parakeet
berd-call session --stt-backend openai
berd-call session --tts-backend openai --mode expert-spokesperson --realtime-url ws://127.0.0.1:18870/v1/realtime
berd-call session --tts-backend openai --tts-url https://proxy.example/v1/audio/speech
berd-call session --stt-backend openai --stt-url wss://proxy.example/v1/realtime?intent=transcription
```

The three URL flags take full endpoints and override only their matching service for that call. Expert–Spokesperson uses `--realtime-url` for both input and output; `--stt-url` and `--tts-url` apply only to conventional mode. Without an override, each service uses its OpenAI endpoint (or its existing environment override). `OPENAI_API_KEY` supplies the CLI credential; the desktop app stores keys separately by endpoint URL in Keychain.

The default Siri backend still requires an exact installed voice name and
language. Missing or unavailable Siri voice configuration and an unavailable
current-locale macOS speech model fail startup with setup guidance. The session
Expand Down
9 changes: 6 additions & 3 deletions src-tauri/crates/berd-call/native/siri_tts_bridge.m
Original file line number Diff line number Diff line change
Expand Up @@ -201,6 +201,11 @@ - (void)finish:(NSError *)error {
}
if (completion) completion(error);
}
static void BerdSetSiriRequestRate(id request, float rate) {
if ([request respondsToSelector:@selector(setRate:)]) {
((void (*)(id, SEL, float))objc_msgSend)(request, @selector(setRate:), rate);
}
}
- (void)synthesizeText:(NSString *)text language:(NSString *)language
voiceName:(NSString *)voiceName rate:(float)rate
completion:(void (^)(NSError *))completion {
Expand All @@ -218,9 +223,7 @@ - (void)synthesizeText:(NSString *)text language:(NSString *)language
}
typedef id (*InitializeRequest)(id, SEL, id, id);
id request = ((InitializeRequest)objc_msgSend)([requestClass alloc], selector, text, voice);
if (rate != 1.0f && [request respondsToSelector:@selector(setRate:)]) {
((void (*)(id, SEL, float))objc_msgSend)(request, @selector(setRate:), rate);
}
BerdSetSiriRequestRate(request, rate);

NSXPCConnection *connection = [[NSXPCConnection alloc]
initWithMachServiceName:@"com.apple.sirittsd" options:0];
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,28 @@

#import "../siri_tts_bridge.m"

@interface BerdRateRecordingRequest : NSObject
@property(nonatomic, assign) float rate;
@property(nonatomic, assign) NSUInteger rateAssignments;
@end

@implementation BerdRateRecordingRequest
- (void)setRate:(float)rate {
_rate = rate;
_rateAssignments += 1;
}
@end

static BOOL BerdTestExplicitUnitRate(NSError **error) {
BerdRateRecordingRequest *request = [BerdRateRecordingRequest new];
BerdSetSiriRequestRate(request, 1.0f);
if (request.rateAssignments != 1 || request.rate != 1.0f) {
if (error) *error = BerdError(106, @"Siri rate 1.0 was not set explicitly.");
return NO;
}
return YES;
}

@interface BerdCapturedSiriPacket : NSObject
@property(nonatomic, strong) NSData *data;
@property(nonatomic, assign) AudioStreamBasicDescription format;
Expand Down Expand Up @@ -211,6 +233,10 @@ static BerdAudioComparison BerdCompareAudio(NSData *actualData, NSData *expected
int main(void) {
@autoreleasepool {
NSError *error = nil;
if (!BerdTestExplicitUnitRate(&error)) {
fprintf(stderr, "set Siri rate: %s\n", error.localizedDescription.UTF8String);
return 1;
}
if (!BerdTestPCMNormalization(&error)) {
fprintf(stderr, "normalize PCM: %s\n", error.localizedDescription.UTF8String);
return 1;
Expand Down
4 changes: 3 additions & 1 deletion src-tauri/crates/berd-call/src/cli_help.rs
Original file line number Diff line number Diff line change
Expand Up @@ -111,10 +111,12 @@ Each form also accepts:
[--rate FLOAT]
[--stt-backend macos|parakeet|openai] [--stt-model-dir PATH]
[--mode conventional|expert-spokesperson]
[--realtime-url WS_URL] [--stt-url WS_URL] [--tts-url HTTP_URL]
The host owns microphone capture, audio playback, transcript delivery, and
agent integration. See PROTOCOL.md for the framed stdin, stdout, and PCM
contracts."#;
contracts. Endpoint URL flags override only the matching service for this call
and default to OpenAI when no environment override is set."#;

const SYNTHESIZE_HELP: &str = r#"Render text through a configured TTS backend into a new WAV file.
Expand Down
60 changes: 60 additions & 0 deletions src-tauri/crates/berd-call/src/endpoint_url.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,60 @@
//! URL policy shared by saved voice endpoints and `berd-call` overrides.
use reqwest::Url;

#[derive(Clone, Copy)]
pub enum EndpointProtocol {
WebSocket,
Http,
}

pub fn is_allowed_endpoint_url(url: &Url, protocol: EndpointProtocol) -> bool {
let scheme_allowed = match protocol {
EndpointProtocol::WebSocket => matches!(url.scheme(), "ws" | "wss"),
EndpointProtocol::Http => matches!(url.scheme(), "http" | "https"),
};
let loopback = url.host_str().is_some_and(|host| {
host.eq_ignore_ascii_case("localhost")
|| host
.trim_matches(['[', ']'])
.parse::<std::net::IpAddr>()
.is_ok_and(|address| address.is_loopback())
});
scheme_allowed
&& url.host_str().is_some()
&& (!matches!(url.scheme(), "http" | "ws") || loopback)
&& url.username().is_empty()
&& url.password().is_none()
&& url.fragment().is_none()
}

#[cfg(test)]
mod tests {
use super::*;

#[test]
fn enforces_protocol_and_loopback_policy() {
assert!(is_allowed_endpoint_url(
&Url::parse("ws://[::1]:18870/realtime").unwrap(),
EndpointProtocol::WebSocket,
));
assert!(is_allowed_endpoint_url(
&Url::parse("http://localhost:18870/speech").unwrap(),
EndpointProtocol::Http,
));
for raw in [
"http://example.test/speech",
"http://user@example.test/speech",
"https://example.test/speech#fragment",
] {
assert!(!is_allowed_endpoint_url(
&Url::parse(raw).unwrap(),
EndpointProtocol::Http,
));
}
assert!(!is_allowed_endpoint_url(
&Url::parse("https://example.test/speech").unwrap(),
EndpointProtocol::WebSocket,
));
}
}
66 changes: 65 additions & 1 deletion src-tauri/crates/berd-call/src/host_session.rs
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,7 @@ use berd_call::PocketAudioPlayer;

use crate::codex::{self, CodexRecord, CodexRelay, CodexTarget};
use crate::host_control::{ControlServer, HostControl};
use crate::saved_settings;
use crate::session_audio::{
AUDIO_BEGIN_KIND, AUDIO_CANCEL_KIND, AUDIO_CHUNK_KIND, AUDIO_END_KIND,
AUDIO_FRAME_HEADER_BYTES, AUDIO_FRAME_MAGIC, AUDIO_FRAME_MARKER,
Expand Down Expand Up @@ -340,7 +341,7 @@ impl SessionActor {
) -> SessionStart {
SessionStart {
saved: self.saved.take().map(|(path, mut saved)| {
saved.arguments = restart.arguments[1..].to_vec();
saved.arguments = saved_settings::persistable_arguments(&restart.arguments[1..]);
saved.tts = None;
(path, saved)
}),
Expand Down Expand Up @@ -2687,6 +2688,69 @@ mod tests {
);
}

#[test]
fn restart_keeps_endpoint_overrides_out_of_saved_preferences() {
let directory = tempfile::tempdir().unwrap();
let path = directory.path().join("settings.json");
let mut child = Command::new("/bin/cat")
.stdin(Stdio::piped())
.stdout(Stdio::null())
.spawn()
.unwrap();
let writer = Arc::new(Mutex::new(child.stdin.take().unwrap()));
let (_events_tx, events) = mpsc::sync_channel(1);
let (_commands_tx, commands) = mpsc::sync_channel(1);
let (audio, _audio_rx) = mpsc::sync_channel(1);
let mut actor = test_actor(writer, events, commands, audio);
actor.saved = Some((
path.clone(),
crate::saved_settings::SavedSettings::default(),
));
let arguments = [
"session",
"--mode",
"chained",
"--realtime-url",
"ws://127.0.0.1:18870/realtime",
"--stt-backend",
"openai",
"--stt-url",
"ws://127.0.0.1:18870/stt",
"--tts-backend",
"openai",
"--tts-url",
"http://127.0.0.1:18870/tts",
]
.map(str::to_string)
.to_vec();
let (response, _reply) = mpsc::sync_channel(1);
let start = actor.restart_start(
PendingRestart {
arguments: arguments.clone(),
expert_spokesperson: false,
response,
},
InputDuringTtsPolicy::AllowBargeIn,
);
assert_eq!(start.arguments, arguments);
let (_, saved) = start.saved.unwrap();
crate::saved_settings::save(&path, &saved).unwrap();
assert_eq!(
crate::saved_settings::load(&path).unwrap().arguments,
[
"--mode",
"chained",
"--stt-backend",
"openai",
"--tts-backend",
"openai",
]
.map(str::to_string)
);
drop(actor);
assert!(child.wait().unwrap().success());
}

#[test]
fn restart_preserves_native_intent_before_child_acknowledgement() {
for requested in [true, false] {
Expand Down
1 change: 1 addition & 0 deletions src-tauri/crates/berd-call/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@ mod audio_output;
pub mod benchmark;
pub mod causal_inbox;
mod configured_tts;
pub mod endpoint_url;
pub mod expert_spokesperson;
pub mod input;
pub mod local_assets;
Expand Down
Loading
Loading