From 14cdc309d87af12a9459e09ca94d071b62400331 Mon Sep 17 00:00:00 2001 From: nichtlegacy Date: Wed, 23 Sep 2026 21:19:53 +0000 Subject: [PATCH 1/2] fix(mobile): transcribe voice input in the user's language The app only ships an English localization, so on a German iPhone the app locale is en-DE and Apple's speech model transcribed German speech with the English model. Use the user's preferred languages instead and fall back through the list when one is unsupported. Word spacing on insertion now applies to every space-delimited language, not only English, so correctly detected languages keep it. --- .../src/native/voiceTranscription.ios.test.ts | 37 +++++++++++++++++++ .../src/native/voiceTranscription.ios.ts | 36 +++++++++++++++--- docs/user/composer.md | 3 ++ .../src/voice-input/controller.test.ts | 8 ++++ .../src/voice-input/controller.ts | 13 ++++--- 5 files changed, 87 insertions(+), 10 deletions(-) diff --git a/apps/mobile/src/native/voiceTranscription.ios.test.ts b/apps/mobile/src/native/voiceTranscription.ios.test.ts index b08e32cdb6f6..73f8ffbb2f72 100644 --- a/apps/mobile/src/native/voiceTranscription.ios.test.ts +++ b/apps/mobile/src/native/voiceTranscription.ios.test.ts @@ -8,8 +8,11 @@ const mocks = vi.hoisted(() => ({ prepare: vi.fn<(locale: string) => Promise>(), transcribe: vi.fn<(audio: ArrayBufferLike, locale: string) => Promise>(), readAudio: vi.fn<() => Promise>(), + getSetting: vi.fn<(key: string) => unknown>(), })); +vi.mock("react-native", () => ({ Settings: { get: mocks.getSetting } })); + vi.mock("@react-native-ai/apple/src/NativeAppleTranscription", () => ({ default: { isAvailable: mocks.isAvailable, @@ -74,6 +77,40 @@ describe("getLocalVoiceTranscriber", () => { expect(mocks.transcribe).toHaveBeenCalledWith(audio, "sv-SE"); }); + it("prefers the user's languages over the English-only app locale", async () => { + const resolvedOptions = Intl.DateTimeFormat().resolvedOptions(); + vi.spyOn(Intl.DateTimeFormat.prototype, "resolvedOptions").mockReturnValue({ + ...resolvedOptions, + locale: "en-DE", + }); + mocks.getSetting.mockReturnValue(["de-DE", "en-US"]); + mocks.prepare.mockResolvedValue("de_DE"); + + const prepared = await getLocalVoiceTranscriber()!.prepare({ + signal: new AbortController().signal, + }); + + expect(mocks.getSetting).toHaveBeenCalledWith("AppleLanguages"); + expect(mocks.prepare).toHaveBeenCalledWith("de-DE"); + expect(prepared.locale).toBe("de_DE"); + }); + + it("falls back to the next language when one is unsupported", async () => { + mocks.getSetting.mockReturnValue(["gsw-CH", "de-CH"]); + mocks.prepare.mockImplementation(async (locale) => { + if (locale === "gsw-CH") + throw Object.assign(new Error(), { code: "AppleTranscriptionUnsupportedLocale" }); + return "de_CH"; + }); + + const prepared = await getLocalVoiceTranscriber()!.prepare({ + signal: new AbortController().signal, + }); + + expect(mocks.prepare.mock.calls).toEqual([["gsw-CH"], ["de-CH"]]); + expect(prepared.locale).toBe("de_CH"); + }); + it("does not start native transcription after cancellation during a file read", async () => { const enteredRead = deferred(); const readResult = deferred(); diff --git a/apps/mobile/src/native/voiceTranscription.ios.ts b/apps/mobile/src/native/voiceTranscription.ios.ts index 216b9e958dd6..a8c3170eb3a6 100644 --- a/apps/mobile/src/native/voiceTranscription.ios.ts +++ b/apps/mobile/src/native/voiceTranscription.ios.ts @@ -1,5 +1,6 @@ import AppleTranscription from "@react-native-ai/apple/src/NativeAppleTranscription"; import { File } from "expo-file-system"; +import { Settings } from "react-native"; import { VoiceTranscriptionError, @@ -9,8 +10,17 @@ import { type VoiceTranscriptionOptions, } from "@t3tools/client-runtime/voice-input"; -function getDeviceLocale(): string { - return Intl.DateTimeFormat().resolvedOptions().locale; +/** + * The app only ships English, so its locale is e.g. "en-DE" on a German iPhone. + * Speech follows the user's preferred languages instead (`AppleLanguages` holds + * the same list as `Locale.preferredLanguages`), falling back to the app locale. + */ +function getPreferredLocales(): string[] { + const languages = Settings.get("AppleLanguages"); + const preferred = Array.isArray(languages) + ? languages.filter((language): language is string => typeof language === "string") + : []; + return [...new Set([...preferred, Intl.DateTimeFormat().resolvedOptions().locale])]; } function wrapError( @@ -34,12 +44,28 @@ function getNativeErrorCode(error: unknown): string | undefined { } export function getLocalVoiceTranscriber(): VoiceTranscriber | null { - const locale = getDeviceLocale(); - if (!AppleTranscription.isAvailable(locale)) return null; - return { prepare: (options) => prepareVoiceTranscription(locale, options) }; + const locales = getPreferredLocales(); + if (!AppleTranscription.isAvailable(locales[0]!)) return null; + return { prepare: (options) => prepareVoiceTranscription(locales, options) }; } async function prepareVoiceTranscription( + locales: readonly string[], + options: VoiceTranscriptionOptions, +): Promise { + for (const locale of locales.slice(0, -1)) { + try { + return await prepareLocale(locale, options); + } catch (error) { + if (!(error instanceof VoiceTranscriptionError && error.code === "unsupported-locale")) { + throw error; + } + } + } + return prepareLocale(locales.at(-1)!, options); +} + +async function prepareLocale( locale: string, { signal }: VoiceTranscriptionOptions, ): Promise { diff --git a/docs/user/composer.md b/docs/user/composer.md index 6e699d0669c1..633188bd483a 100644 --- a/docs/user/composer.md +++ b/docs/user/composer.md @@ -135,6 +135,9 @@ On supported iPhones with iOS 26 or later, use the composer's microphone to reco then confirm to transcribe. Text is inserted where your selection was when recording started, ready for you to review and edit before sending. +Transcription uses the first language in your iPhone's preferred language list +(Settings > General > Language & Region) that Apple's speech model supports. + The first use may download Apple's speech model and needs a network connection. Later transcription works offline for that language. Recordings can be up to five minutes long. Canceling, leaving the screen, or an audio interruption discards the diff --git a/packages/client-runtime/src/voice-input/controller.test.ts b/packages/client-runtime/src/voice-input/controller.test.ts index 5f26b882b694..9d803061aaef 100644 --- a/packages/client-runtime/src/voice-input/controller.test.ts +++ b/packages/client-runtime/src/voice-input/controller.test.ts @@ -134,6 +134,14 @@ describe("resolveTranscriptCommit", () => { }); }); + it("adds word spacing for other space-delimited languages", () => { + const atEnd = draft({ text: "Schön", selection: { start: 5, end: 5 } }); + expect(resolveTranscriptCommit(atEnd, atEnd, "Grüße", "de-DE")).toMatchObject({ + kind: "commit", + text: "Schön Grüße", + }); + }); + it("does not add English boundary spaces to CJK or selected inline text", () => { const cjk = draft({ text: "修正キャッシュ", selection: { start: 8, end: 8 } }); expect(resolveTranscriptCommit(cjk, cjk, "テストも", "ja-JP")).toMatchObject({ diff --git a/packages/client-runtime/src/voice-input/controller.ts b/packages/client-runtime/src/voice-input/controller.ts index cb284ad9f85b..39dd1916bf99 100644 --- a/packages/client-runtime/src/voice-input/controller.ts +++ b/packages/client-runtime/src/voice-input/controller.ts @@ -70,6 +70,9 @@ type TranscriptCommitResult = | { readonly kind: "stale" } | { readonly kind: "empty" }; +/** Languages written without spaces between words. */ +const WORD_SPACELESS_LANGUAGES = new Set(["ja", "zh", "yue", "th", "lo", "km", "my"]); + export function resolveTranscriptCommit( captured: VoiceDraftSnapshot, current: VoiceDraftSnapshot | null, @@ -91,19 +94,19 @@ export function resolveTranscriptCommit( } const isEmptySelection = captured.selection.start === captured.selection.end; - const normalizedLocale = locale.replaceAll("_", "-").toLowerCase(); - const usesEnglishSpacing = normalizedLocale === "en" || normalizedLocale.startsWith("en-"); + const language = locale.split(/[-_]/)[0]!.toLowerCase(); + const usesWordSpacing = !WORD_SPACELESS_LANGUAGES.has(language); let insertion = replacement; - if (isEmptySelection && usesEnglishSpacing) { + if (isEmptySelection && usesWordSpacing) { const left = captured.text[captured.selection.start - 1]; const right = captured.text[captured.selection.start]; const leftNeedsBoundary = left !== undefined && - /[A-Za-z0-9.!?,:;)\]}'"]/.test(left) && + /[\p{L}\p{N}.!?,:;)\]}'"]/u.test(left) && (right === undefined || /\s/.test(right)); const rightNeedsBoundary = right !== undefined && - /[A-Za-z0-9([{'"]/.test(right) && + /[\p{L}\p{N}([{'"]/u.test(right) && (left === undefined || /\s/.test(left)); insertion = `${leftNeedsBoundary ? " " : ""}${replacement}${rightNeedsBoundary ? " " : ""}`; } From 2d4d05c325d27837e354a97db05d7438122e6b60 Mon Sep 17 00:00:00 2001 From: nichtlegacy Date: Thu, 24 Sep 2026 08:16:00 +0000 Subject: [PATCH 2/2] fix(mobile): keep word spacing after combining marks Also note the English fallback in the voice input guide. --- docs/user/composer.md | 3 ++- packages/client-runtime/src/voice-input/controller.test.ts | 6 ++++++ packages/client-runtime/src/voice-input/controller.ts | 2 +- 3 files changed, 9 insertions(+), 2 deletions(-) diff --git a/docs/user/composer.md b/docs/user/composer.md index 633188bd483a..20c5668094ef 100644 --- a/docs/user/composer.md +++ b/docs/user/composer.md @@ -136,7 +136,8 @@ then confirm to transcribe. Text is inserted where your selection was when recording started, ready for you to review and edit before sending. Transcription uses the first language in your iPhone's preferred language list -(Settings > General > Language & Region) that Apple's speech model supports. +(Settings > General > Language & Region) that Apple's speech model supports, +or English if none is supported. The first use may download Apple's speech model and needs a network connection. Later transcription works offline for that language. Recordings can be up to five diff --git a/packages/client-runtime/src/voice-input/controller.test.ts b/packages/client-runtime/src/voice-input/controller.test.ts index 9d803061aaef..efef6b9f7e86 100644 --- a/packages/client-runtime/src/voice-input/controller.test.ts +++ b/packages/client-runtime/src/voice-input/controller.test.ts @@ -140,6 +140,12 @@ describe("resolveTranscriptCommit", () => { kind: "commit", text: "Schön Grüße", }); + + const combiningMark = draft({ text: "नमस्ते", selection: { start: 6, end: 6 } }); + expect(resolveTranscriptCommit(combiningMark, combiningMark, "दुनिया", "hi-IN")).toMatchObject({ + kind: "commit", + text: "नमस्ते दुनिया", + }); }); it("does not add English boundary spaces to CJK or selected inline text", () => { diff --git a/packages/client-runtime/src/voice-input/controller.ts b/packages/client-runtime/src/voice-input/controller.ts index 39dd1916bf99..8f794934f932 100644 --- a/packages/client-runtime/src/voice-input/controller.ts +++ b/packages/client-runtime/src/voice-input/controller.ts @@ -102,7 +102,7 @@ export function resolveTranscriptCommit( const right = captured.text[captured.selection.start]; const leftNeedsBoundary = left !== undefined && - /[\p{L}\p{N}.!?,:;)\]}'"]/u.test(left) && + /[\p{L}\p{M}\p{N}.!?,:;)\]}'"]/u.test(left) && (right === undefined || /\s/.test(right)); const rightNeedsBoundary = right !== undefined &&