Skip to content

Commit 8bd5e70

Browse files
committed
feat(coding-agent): add Apple SpeechAnalyzer STT
1 parent caefa61 commit 8bd5e70

22 files changed

Lines changed: 1317 additions & 77 deletions

‎.github/workflows/bazel-cache-warm.yml‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -70,8 +70,8 @@ jobs:
7070
fail-fast: false
7171
matrix:
7272
include:
73-
- { os: macos-15-intel, target: darwin-x64-baseline, scope: release-darwin-x64 }
74-
- { os: macos-14, target: darwin-arm64, scope: release-darwin-arm64 }
73+
- { os: macos-26-intel, target: darwin-x64-baseline, scope: release-darwin-x64 }
74+
- { os: macos-26, target: darwin-arm64, scope: release-darwin-arm64 }
7575
runs-on: ${{ matrix.os }}
7676
steps:
7777
- uses: actions/checkout@v4

‎.github/workflows/ci.yml‎

Lines changed: 18 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -621,13 +621,13 @@ jobs:
621621
matrix:
622622
include:
623623
- {
624-
os: macos-15-intel,
624+
os: macos-26-intel,
625625
target_id: darwin-x64,
626626
binary_path: packages/coding-agent/binaries/omp-darwin-x64,
627627
native_targets: darwin-x64-baseline,
628628
}
629629
- {
630-
os: macos-14,
630+
os: macos-26,
631631
target_id: darwin-arm64,
632632
binary_path: packages/coding-agent/binaries/omp-darwin-arm64,
633633
native_targets: darwin-arm64,
@@ -693,6 +693,12 @@ jobs:
693693
with:
694694
name: omp-binary-${{ matrix.target_id }}
695695
path: ${{ matrix.binary_path }}
696+
- name: Upload Apple speech sidecar artifact
697+
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
698+
with:
699+
name: apple-speech-sidecar-${{ matrix.target_id }}
700+
path: packages/coding-agent/dist/omp-speech-analyzer-*
701+
if-no-files-found: error
696702

697703
# Publishes the five @oh-my-pi/pi-natives-<tag> leaf packages once
698704
# validation passes and every binary built. Runs on one linux runner: leaf
@@ -894,6 +900,16 @@ jobs:
894900
uses: ./.github/actions/native-artifacts
895901
with:
896902
targets: linux-x64-baseline linux-x64-modern
903+
- name: Install Apple speech sidecars
904+
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
905+
with:
906+
pattern: apple-speech-sidecar-*
907+
path: packages/coding-agent/dist
908+
merge-multiple: true
909+
- name: Verify Apple speech sidecars
910+
run: |
911+
test -s packages/coding-agent/dist/omp-speech-analyzer-arm64
912+
test -s packages/coding-agent/dist/omp-speech-analyzer-x64
897913
- name: Publish to npm
898914
env:
899915
# Fallback auth: setup-node wrote an .npmrc referencing

‎packages/coding-agent/CHANGELOG.md‎

Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -2,6 +2,10 @@
22

33
## [Unreleased]
44

5+
### Added
6+
7+
- Added opt-in Apple SpeechAnalyzer speech-to-text on macOS 26+, with live partials and system-managed locale assets.
8+
59
## [18.0.8] - 2026-08-27
610

711
### Added

‎packages/coding-agent/package.json‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -88,6 +88,7 @@
8888
"dist/cli.js",
8989
"dist/docs-index.generated.txt",
9090
"dist/CHANGELOG-*.md",
91+
"dist/omp-speech-analyzer-*",
9192
"dist/*.node",
9293
"dist/template-*.css",
9394
"dist/template-*.html",
Lines changed: 80 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,80 @@
1+
import * as fs from "node:fs/promises";
2+
import * as os from "node:os";
3+
import * as path from "node:path";
4+
import { $which } from "@oh-my-pi/pi-utils";
5+
6+
export type AppleSpeechArchitecture = "arm64" | "x64";
7+
8+
const packageDir = path.join(import.meta.dir, "..");
9+
const sourcePath = path.join(packageDir, "src", "stt", "speech-analyzer.swift");
10+
11+
/** Compile the SpeechAnalyzer helper for one Darwin release architecture. */
12+
export async function buildAppleSpeechSidecar(
13+
architecture: AppleSpeechArchitecture,
14+
outputPath: string,
15+
): Promise<void> {
16+
if (process.platform !== "darwin") {
17+
throw new Error("Apple SpeechAnalyzer sidecars can only be built on macOS.");
18+
}
19+
const swiftc = $which("swiftc");
20+
if (!swiftc) throw new Error("Swift compiler not found; Xcode 26 or newer is required.");
21+
await fs.mkdir(path.dirname(outputPath), { recursive: true });
22+
const targetArchitecture = architecture === "x64" ? "x86_64" : architecture;
23+
const proc = Bun.spawn(
24+
[
25+
swiftc,
26+
"-parse-as-library",
27+
"-O",
28+
"-target",
29+
`${targetArchitecture}-apple-macos26.0`,
30+
"-framework",
31+
"Speech",
32+
"-framework",
33+
"AVFAudio",
34+
"-framework",
35+
"CoreMedia",
36+
"-o",
37+
outputPath,
38+
sourcePath,
39+
],
40+
{ stdin: "ignore", stdout: "inherit", stderr: "pipe" },
41+
);
42+
const [exitCode, stderr] = await Promise.all([
43+
proc.exited,
44+
new Response(proc.stderr as ReadableStream<Uint8Array>).text(),
45+
]);
46+
if (exitCode !== 0) {
47+
throw new Error(`SpeechAnalyzer sidecar build failed: ${stderr.trim() || `swiftc exited ${exitCode}`}`);
48+
}
49+
await fs.chmod(outputPath, 0o755);
50+
51+
const codesign = $which("codesign");
52+
if (codesign) {
53+
const sign = Bun.spawn([codesign, "--force", "--sign", "-", outputPath], {
54+
stdin: "ignore",
55+
stdout: "ignore",
56+
stderr: "pipe",
57+
});
58+
const [signExitCode, signStderr] = await Promise.all([
59+
sign.exited,
60+
new Response(sign.stderr as ReadableStream<Uint8Array>).text(),
61+
]);
62+
if (signExitCode !== 0) {
63+
throw new Error(
64+
`SpeechAnalyzer sidecar signing failed: ${signStderr.trim() || `codesign exited ${signExitCode}`}`,
65+
);
66+
}
67+
}
68+
}
69+
70+
/** Build a temporary helper and return its bytes for Bun's compiled-binary embed. */
71+
export async function buildAppleSpeechSidecarBase64(architecture: AppleSpeechArchitecture): Promise<string> {
72+
const directory = await fs.mkdtemp(path.join(os.tmpdir(), "omp-speech-analyzer-build-"));
73+
const outputPath = path.join(directory, "omp-speech-analyzer");
74+
try {
75+
await buildAppleSpeechSidecar(architecture, outputPath);
76+
return Buffer.from(await Bun.file(outputPath).bytes()).toString("base64");
77+
} finally {
78+
await fs.rm(directory, { recursive: true, force: true });
79+
}
80+
}

‎packages/coding-agent/scripts/build-binary.ts‎

Lines changed: 11 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -2,6 +2,7 @@
22

33
import { createRequire } from "node:module";
44
import * as path from "node:path";
5+
import { buildAppleSpeechSidecarBase64 } from "./apple-speech-sidecar";
56
import { compileCodingAgent } from "./compile-binary";
67

78
const packageDir = path.join(import.meta.dir, "..");
@@ -88,12 +89,22 @@ async function main(): Promise<void> {
8889
["bun", "--cwd=../natives", "run", "gen:native"],
8990
crossBuild ? { ...Bun.env, TARGET_PLATFORM: crossBuild.platform, TARGET_ARCH: crossBuild.arch } : Bun.env,
9091
);
92+
const targetPlatform = crossBuild?.platform ?? process.platform;
93+
const targetArchitecture = crossBuild?.arch ?? process.arch;
94+
let appleSpeechSidecarBase64: string | undefined;
95+
if (targetPlatform === "darwin") {
96+
if (targetArchitecture !== "arm64" && targetArchitecture !== "x64") {
97+
throw new Error(`Unsupported Darwin architecture for SpeechAnalyzer: ${targetArchitecture}`);
98+
}
99+
appleSpeechSidecarBase64 = await buildAppleSpeechSidecarBase64(targetArchitecture);
100+
}
91101
try {
92102
await compileCodingAgent({
93103
repoRoot,
94104
entrypoint: path.join(packageDir, "src", "cli.ts"),
95105
outfile: outputPath,
96106
transformersVersion,
107+
appleSpeechSidecarBase64,
97108
target: crossBuild?.target,
98109
executablePath: Bun.env.BUN_COMPILE_EXECUTABLE_PATH || undefined,
99110
skipBuiltinCodesign: shouldAdhocSign,

‎packages/coding-agent/scripts/bundle-dist.ts‎

Lines changed: 9 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -3,6 +3,7 @@
33
import * as fs from "node:fs/promises";
44
import * as path from "node:path";
55
import { isEnoent } from "@oh-my-pi/pi-utils";
6+
import { buildAppleSpeechSidecar } from "./apple-speech-sidecar";
67
import { buildDocsIndexPayload } from "./generate-docs-index";
78

89
const packageDir = path.join(import.meta.dir, "..");
@@ -70,6 +71,7 @@ async function cleanBundleOutputs(): Promise<void> {
7071
entry === "cli.js" ||
7172
entry === "docs-index.generated.txt" ||
7273
entry.endsWith(".node") ||
74+
(process.platform === "darwin" && entry.startsWith("omp-speech-analyzer-")) ||
7375
entry.endsWith(".js.map") ||
7476
(entry.startsWith("CHANGELOG-") && entry.endsWith(".md")) ||
7577
legacyHtmlExportAssetPattern.test(entry),
@@ -81,6 +83,12 @@ async function cleanBundleOutputs(): Promise<void> {
8183
async function main(): Promise<void> {
8284
const start = Bun.nanoseconds();
8385
await cleanBundleOutputs();
86+
if (process.platform === "darwin") {
87+
await Promise.all([
88+
buildAppleSpeechSidecar("arm64", path.join(outDir, "omp-speech-analyzer-arm64")),
89+
buildAppleSpeechSidecar("x64", path.join(outDir, "omp-speech-analyzer-x64")),
90+
]);
91+
}
8492
// The npm bundle ships no stats dashboard sources, so embed the dashboard
8593
// archive the same way compiled binaries do (scripts/build-binary.ts). Reset
8694
// afterwards to keep the checked-in placeholder empty.
@@ -101,6 +109,7 @@ async function main(): Promise<void> {
101109
external: [...ALWAYS_EXTERNAL, ...RUNTIME_EXTERNAL],
102110
define: {
103111
"process.env.PI_BUNDLED": JSON.stringify("true"),
112+
"process.env.PI_APPLE_SPEECH_SIDECAR_BASE64": JSON.stringify(""),
104113
"process.env.PI_DOCS_EMBED": JSON.stringify(docsPayload.payload),
105114
},
106115
minify: {

‎packages/coding-agent/scripts/compile-binary.ts‎

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -14,6 +14,8 @@ export interface CodingAgentCompileOptions {
1414
readonly outfile: string;
1515
/** Concrete Transformers.js version baked into the tiny-model worker. */
1616
readonly transformersVersion: string;
17+
/** Base64-encoded native SpeechAnalyzer helper for the selected Darwin architecture. */
18+
readonly appleSpeechSidecarBase64?: string;
1719
/** Optional cross-compilation runtime target. */
1820
readonly target?: Bun.Build.CompileTarget;
1921
/** Optional unmodified Bun executable used as the standalone runtime template. */
@@ -41,6 +43,7 @@ export async function compileCodingAgent(options: CodingAgentCompileOptions): Pr
4143
define: {
4244
"process.env.PI_COMPILED": JSON.stringify("true"),
4345
"process.env.PI_TINY_TRANSFORMERS_VERSION": JSON.stringify(options.transformersVersion),
46+
"process.env.PI_APPLE_SPEECH_SIDECAR_BASE64": JSON.stringify(options.appleSpeechSidecarBase64 ?? ""),
4447
"process.env.PI_DOCS_EMBED": JSON.stringify((await buildDocsIndexPayload()).payload),
4548
},
4649
minify: {

‎packages/coding-agent/src/cli.ts‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -35,6 +35,7 @@ import type { WorkerInbound as JsWorkerInbound, WorkerOutbound as JsWorkerOutbou
3535
import { DAEMON_BROKER_WORKER_ARG } from "./launch/protocol";
3636
import { TERMINAL_OUTPUT_WORKER_ARG } from "./launch/terminal-output-worker-protocol";
3737
import { LSP_MUX_WORKER_ARG } from "./lsp/mux/protocol";
38+
import { smokeTestAppleSpeechSidecar } from "./stt/apple-speech-client";
3839
import rootLicense from "./tools/browser/relay/extension-assets/LICENSE.txt" with { type: "text" };
3940
import thirdPartyNotices from "./tools/browser/relay/extension-assets/THIRD-PARTY-NOTICES.txt" with { type: "text" };
4041
import { COMPUTER_WORKER_ARG } from "./tools/computer/protocol";
@@ -120,6 +121,7 @@ async function runSmokeTest(): Promise<void> {
120121

121122
await smokeTestTinyTitleWorker();
122123
await smokeTestSttWorker();
124+
await smokeTestAppleSpeechSidecar();
123125
await smokeTestJsEvalWorker();
124126
await smokeTestComputerWorker();
125127
await smokeTestTtsWorker();

‎packages/coding-agent/src/cli/setup-cli.ts‎

Lines changed: 35 additions & 10 deletions
Original file line numberDiff line numberDiff line change
@@ -9,8 +9,9 @@ import chalk from "@oh-my-pi/pi-utils/chalk";
99
import { Settings, settings } from "../config/settings";
1010
import { checkPythonKernelAvailability } from "../eval/py/kernel";
1111
import { theme } from "../modes/theme/theme";
12+
import { appleSpeechClient } from "../stt/apple-speech-client";
1213
import { downloadSttModel, isSttModelCached } from "../stt/downloader";
13-
import { isSttModelKey, STT_MODEL_OPTIONS } from "../stt/models";
14+
import { isSttModelKey, resolveSttModelSpec, STT_MODEL_OPTIONS } from "../stt/models";
1415
import { downloadTtsModel, isTtsLocalModelKey, isTtsModelCached, TTS_LOCAL_MODEL_OPTIONS } from "../tts";
1516
import { selectSetupModel } from "./setup-model-picker";
1617

@@ -147,9 +148,9 @@ async function handlePythonSetup(flags: { json?: boolean; check?: boolean }): Pr
147148
}
148149

149150
/**
150-
* One installable speech dependency. `isReady`/`status` are read-only probes;
151-
* `pick` (optional) lets an interactive user choose + persist a model; `ensure`
152-
* performs the download, streaming a normalized progress event.
151+
* One speech dependency. `isReady`/`status` are read-only probes; `pick`
152+
* (optional) lets an interactive user choose + persist an engine; `ensure`
153+
* prepares its application-managed model or system-managed locale asset.
153154
*/
154155
interface SpeechComponent {
155156
name: string;
@@ -163,10 +164,23 @@ function buildSpeechComponents(): SpeechComponent[] {
163164
return [
164165
{
165166
name: "Speech-to-Text model",
166-
isReady: () => isSttModelCached(settings.get("stt.modelName")),
167+
isReady: async () => {
168+
const spec = resolveSttModelSpec(settings.get("stt.modelName"));
169+
if (spec.engine === "speech-analyzer") {
170+
return (await appleSpeechClient.status(settings.get("stt.language"))).installed;
171+
}
172+
return await isSttModelCached(spec.key);
173+
},
167174
status: async () => {
168-
const key = settings.get("stt.modelName");
169-
return (await isSttModelCached(key)) ? key : `${key} — not downloaded`;
175+
const spec = resolveSttModelSpec(settings.get("stt.modelName"));
176+
if (spec.engine === "speech-analyzer") {
177+
const status = await appleSpeechClient.status(settings.get("stt.language"));
178+
if (status.installed) {
179+
return `${spec.key} — ${status.locale ?? "system locale"} (system-managed)`;
180+
}
181+
return `${spec.key} — ${status.error ?? "locale asset not prepared"}`;
182+
}
183+
return (await isSttModelCached(spec.key)) ? spec.key : `${spec.key} — not downloaded`;
170184
},
171185
pick: async () => {
172186
const chosen = await selectSetupModel(
@@ -181,10 +195,21 @@ function buildSpeechComponents(): SpeechComponent[] {
181195
}
182196
return true;
183197
},
184-
ensure: onProgress =>
185-
downloadSttModel(settings.get("stt.modelName"), progress =>
198+
ensure: async onProgress => {
199+
const spec = resolveSttModelSpec(settings.get("stt.modelName"));
200+
if (spec.engine === "speech-analyzer") {
201+
onProgress({ stage: "Preparing system-managed Apple speech recognition" });
202+
const status = await appleSpeechClient.prepare(settings.get("stt.language"));
203+
onProgress({
204+
stage: `Apple speech recognition ready${status.locale ? ` (${status.locale})` : ""}`,
205+
percent: 100,
206+
});
207+
return;
208+
}
209+
await downloadSttModel(spec.key, progress =>
186210
onProgress({ stage: `Downloading ${progress.label} model`, percent: progress.percent }),
187-
),
211+
);
212+
},
188213
},
189214
{
190215
name: "Text-to-Speech model",

0 commit comments

Comments
 (0)