Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
37 changes: 37 additions & 0 deletions apps/mobile/src/native/voiceTranscription.ios.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -8,8 +8,11 @@ const mocks = vi.hoisted(() => ({
prepare: vi.fn<(locale: string) => Promise<string>>(),
transcribe: vi.fn<(audio: ArrayBufferLike, locale: string) => Promise<TranscriptionResult>>(),
readAudio: vi.fn<() => Promise<ArrayBuffer>>(),
getSetting: vi.fn<(key: string) => unknown>(),
}));

vi.mock("react-native", () => ({ Settings: { get: mocks.getSetting } }));

vi.mock("@react-native-ai/apple/src/NativeAppleTranscription", () => ({
default: {
isAvailable: mocks.isAvailable,
Expand Down Expand Up @@ -74,6 +77,40 @@ describe("getLocalVoiceTranscriber", () => {
expect(mocks.transcribe).toHaveBeenCalledWith(audio, "sv-SE");
});

it("prefers the user's languages over the English-only app locale", async () => {
const resolvedOptions = Intl.DateTimeFormat().resolvedOptions();
vi.spyOn(Intl.DateTimeFormat.prototype, "resolvedOptions").mockReturnValue({
...resolvedOptions,
locale: "en-DE",
});
mocks.getSetting.mockReturnValue(["de-DE", "en-US"]);
mocks.prepare.mockResolvedValue("de_DE");

const prepared = await getLocalVoiceTranscriber()!.prepare({
signal: new AbortController().signal,
});

expect(mocks.getSetting).toHaveBeenCalledWith("AppleLanguages");
expect(mocks.prepare).toHaveBeenCalledWith("de-DE");
expect(prepared.locale).toBe("de_DE");
});

it("falls back to the next language when one is unsupported", async () => {
mocks.getSetting.mockReturnValue(["gsw-CH", "de-CH"]);
mocks.prepare.mockImplementation(async (locale) => {
if (locale === "gsw-CH")
throw Object.assign(new Error(), { code: "AppleTranscriptionUnsupportedLocale" });
return "de_CH";
});

const prepared = await getLocalVoiceTranscriber()!.prepare({
signal: new AbortController().signal,
});

expect(mocks.prepare.mock.calls).toEqual([["gsw-CH"], ["de-CH"]]);
expect(prepared.locale).toBe("de_CH");
});

it("does not start native transcription after cancellation during a file read", async () => {
const enteredRead = deferred<void>();
const readResult = deferred<ArrayBuffer>();
Expand Down
36 changes: 31 additions & 5 deletions apps/mobile/src/native/voiceTranscription.ios.ts
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
import AppleTranscription from "@react-native-ai/apple/src/NativeAppleTranscription";
import { File } from "expo-file-system";
import { Settings } from "react-native";

import {
VoiceTranscriptionError,
Expand All @@ -9,8 +10,17 @@ import {
type VoiceTranscriptionOptions,
} from "@t3tools/client-runtime/voice-input";

function getDeviceLocale(): string {
return Intl.DateTimeFormat().resolvedOptions().locale;
/**
* The app only ships English, so its locale is e.g. "en-DE" on a German iPhone.
* Speech follows the user's preferred languages instead (`AppleLanguages` holds
* the same list as `Locale.preferredLanguages`), falling back to the app locale.
*/
function getPreferredLocales(): string[] {
const languages = Settings.get("AppleLanguages");
const preferred = Array.isArray(languages)
? languages.filter((language): language is string => typeof language === "string")
: [];
return [...new Set([...preferred, Intl.DateTimeFormat().resolvedOptions().locale])];
}

function wrapError(
Expand All @@ -34,12 +44,28 @@ function getNativeErrorCode(error: unknown): string | undefined {
}

export function getLocalVoiceTranscriber(): VoiceTranscriber | null {
const locale = getDeviceLocale();
if (!AppleTranscription.isAvailable(locale)) return null;
return { prepare: (options) => prepareVoiceTranscription(locale, options) };
const locales = getPreferredLocales();
if (!AppleTranscription.isAvailable(locales[0]!)) return null;
return { prepare: (options) => prepareVoiceTranscription(locales, options) };
}

async function prepareVoiceTranscription(
locales: readonly string[],
options: VoiceTranscriptionOptions,
): Promise<PreparedVoiceTranscription> {
for (const locale of locales.slice(0, -1)) {
try {
return await prepareLocale(locale, options);
} catch (error) {
if (!(error instanceof VoiceTranscriptionError && error.code === "unsupported-locale")) {
throw error;
}
}
}
return prepareLocale(locales.at(-1)!, options);
}

async function prepareLocale(
locale: string,
{ signal }: VoiceTranscriptionOptions,
): Promise<PreparedVoiceTranscription> {
Expand Down
4 changes: 4 additions & 0 deletions docs/user/composer.md
Original file line number Diff line number Diff line change
Expand Up @@ -135,6 +135,10 @@ On supported iPhones with iOS 26 or later, use the composer's microphone to reco
then confirm to transcribe. Text is inserted where your selection was when
recording started, ready for you to review and edit before sending.

Transcription uses the first language in your iPhone's preferred language list
(Settings > General > Language & Region) that Apple's speech model supports,
or English if none is supported.

The first use may download Apple's speech model and needs a network connection.
Later transcription works offline for that language. Recordings can be up to five
minutes long. Canceling, leaving the screen, or an audio interruption discards the
Expand Down
14 changes: 14 additions & 0 deletions packages/client-runtime/src/voice-input/controller.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -134,6 +134,20 @@ describe("resolveTranscriptCommit", () => {
});
});

it("adds word spacing for other space-delimited languages", () => {
const atEnd = draft({ text: "Schön", selection: { start: 5, end: 5 } });
expect(resolveTranscriptCommit(atEnd, atEnd, "Grüße", "de-DE")).toMatchObject({
kind: "commit",
text: "Schön Grüße",
});

const combiningMark = draft({ text: "नमस्ते", selection: { start: 6, end: 6 } });
expect(resolveTranscriptCommit(combiningMark, combiningMark, "दुनिया", "hi-IN")).toMatchObject({
kind: "commit",
text: "नमस्ते दुनिया",
});
});

it("does not add English boundary spaces to CJK or selected inline text", () => {
const cjk = draft({ text: "修正キャッシュ", selection: { start: 8, end: 8 } });
expect(resolveTranscriptCommit(cjk, cjk, "テストも", "ja-JP")).toMatchObject({
Expand Down
13 changes: 8 additions & 5 deletions packages/client-runtime/src/voice-input/controller.ts
Original file line number Diff line number Diff line change
Expand Up @@ -70,6 +70,9 @@ type TranscriptCommitResult =
| { readonly kind: "stale" }
| { readonly kind: "empty" };

/** Languages written without spaces between words. */
const WORD_SPACELESS_LANGUAGES = new Set(["ja", "zh", "yue", "th", "lo", "km", "my"]);

export function resolveTranscriptCommit(
captured: VoiceDraftSnapshot,
current: VoiceDraftSnapshot | null,
Expand All @@ -91,19 +94,19 @@ export function resolveTranscriptCommit(
}

const isEmptySelection = captured.selection.start === captured.selection.end;
const normalizedLocale = locale.replaceAll("_", "-").toLowerCase();
const usesEnglishSpacing = normalizedLocale === "en" || normalizedLocale.startsWith("en-");
const language = locale.split(/[-_]/)[0]!.toLowerCase();
const usesWordSpacing = !WORD_SPACELESS_LANGUAGES.has(language);
let insertion = replacement;
if (isEmptySelection && usesEnglishSpacing) {
if (isEmptySelection && usesWordSpacing) {
const left = captured.text[captured.selection.start - 1];
const right = captured.text[captured.selection.start];
const leftNeedsBoundary =
left !== undefined &&
/[A-Za-z0-9.!?,:;)\]}'"]/.test(left) &&
/[\p{L}\p{M}\p{N}.!?,:;)\]}'"]/u.test(left) &&
(right === undefined || /\s/.test(right));
const rightNeedsBoundary =
right !== undefined &&
/[A-Za-z0-9([{'"]/.test(right) &&
/[\p{L}\p{N}([{'"]/u.test(right) &&
(left === undefined || /\s/.test(left));
insertion = `${leftNeedsBoundary ? " " : ""}${replacement}${rightNeedsBoundary ? " " : ""}`;
}
Expand Down
Loading