extract shared whisper transcribe helper, use in file-browser and pi stt routes
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -7,6 +7,7 @@ import * as errors from '@@/custom-errors';
|
|||||||
import { readTtsConfig } from '@@/api/server-settings/tts';
|
import { readTtsConfig } from '@@/api/server-settings/tts';
|
||||||
import { readSttConfig } from '@@/api/server-settings/stt';
|
import { readSttConfig } from '@@/api/server-settings/stt';
|
||||||
import { readOcrConfig } from '@@/api/server-settings/ocr';
|
import { readOcrConfig } from '@@/api/server-settings/ocr';
|
||||||
|
import { transcribeAudio } from '@@/api/stt/transcribe';
|
||||||
import { getUserSettings } from 'officerdb';
|
import { getUserSettings } from 'officerdb';
|
||||||
|
|
||||||
async function getUserTtsVoice(userId: number): Promise<string | null> {
|
async function getUserTtsVoice(userId: number): Promise<string | null> {
|
||||||
@@ -794,45 +795,17 @@ router.post('/transcribe', async (ctx) => {
|
|||||||
const whisperUrl = sttConfig?.url;
|
const whisperUrl = sttConfig?.url;
|
||||||
if (!whisperUrl) throw errors.BAD_REQUEST('Whisper not configured — set it up in Settings → Speech to Text');
|
if (!whisperUrl) throw errors.BAD_REQUEST('Whisper not configured — set it up in Settings → Speech to Text');
|
||||||
|
|
||||||
const audioFile = Bun.file(absPath);
|
|
||||||
|
|
||||||
// Step 1: Detect language
|
|
||||||
const detectForm = new FormData();
|
|
||||||
detectForm.append('file', audioFile);
|
|
||||||
detectForm.append('temperature', '0.0');
|
|
||||||
detectForm.append('response_format', 'verbose_json');
|
|
||||||
detectForm.append('detect_language', 'true');
|
|
||||||
|
|
||||||
const detectRes = await fetch(`${whisperUrl}/inference`, { method: 'POST', body: detectForm });
|
|
||||||
if (!detectRes.ok) throw errors.BAD_REQUEST('Language detection failed');
|
|
||||||
|
|
||||||
const detectJson = (await detectRes.json()) as { language?: string };
|
|
||||||
const detectedLang = detectJson.language ?? 'en';
|
|
||||||
|
|
||||||
// Step 2: Check user's spoken languages to decide if translation is needed
|
|
||||||
let shouldTranslate = false;
|
|
||||||
const settings = await getUserSettings(user.id);
|
const settings = await getUserSettings(user.id);
|
||||||
const spokenLanguages = (settings.languages as { spoken?: string[] })?.spoken ?? [];
|
const spokenLanguages = (settings.languages as { spoken?: string[] })?.spoken ?? [];
|
||||||
if (spokenLanguages.length > 0 && !spokenLanguages.includes(detectedLang)) {
|
|
||||||
shouldTranslate = true;
|
let text: string;
|
||||||
|
try {
|
||||||
|
const result = await transcribeAudio({ file: Bun.file(absPath), whisperUrl, spokenLanguages });
|
||||||
|
text = result.text;
|
||||||
|
} catch (err) {
|
||||||
|
throw errors.BAD_REQUEST(err instanceof Error ? err.message : 'Transcription failed');
|
||||||
}
|
}
|
||||||
|
|
||||||
// Step 3: Full transcription
|
|
||||||
const transcribeForm = new FormData();
|
|
||||||
transcribeForm.append('file', audioFile);
|
|
||||||
transcribeForm.append('temperature', '0.0');
|
|
||||||
transcribeForm.append('temperature_inc', '0.2');
|
|
||||||
transcribeForm.append('response_format', 'text');
|
|
||||||
transcribeForm.append('language', detectedLang);
|
|
||||||
if (shouldTranslate) {
|
|
||||||
transcribeForm.append('translate', 'true');
|
|
||||||
}
|
|
||||||
|
|
||||||
const res = await fetch(`${whisperUrl}/inference`, { method: 'POST', body: transcribeForm });
|
|
||||||
if (!res.ok) throw errors.BAD_REQUEST('Transcription request failed');
|
|
||||||
|
|
||||||
const text = (await res.text()).trim();
|
|
||||||
|
|
||||||
await mkdir(dirname(cacheAbs), { recursive: true });
|
await mkdir(dirname(cacheAbs), { recursive: true });
|
||||||
await Bun.write(cacheAbs, text);
|
await Bun.write(cacheAbs, text);
|
||||||
|
|
||||||
|
|||||||
@@ -6,6 +6,8 @@ import { readSttConfig } from '../server-settings/stt';
|
|||||||
import { listPiModels } from './list-models';
|
import { listPiModels } from './list-models';
|
||||||
import { getHomeDirForRole } from '../../data-path';
|
import { getHomeDirForRole } from '../../data-path';
|
||||||
import { logger } from './logger';
|
import { logger } from './logger';
|
||||||
|
import { transcribeAudio } from '../stt/transcribe';
|
||||||
|
import { getUserSettings } from 'officerdb';
|
||||||
|
|
||||||
export const piRestRouter = createRouter();
|
export const piRestRouter = createRouter();
|
||||||
|
|
||||||
@@ -72,30 +74,21 @@ piRestRouter.post('/pi/stt', async (ctx: Context) => {
|
|||||||
return ctx.json({ error: 'Whisper not configured — set it up in Settings → Speech to Text' }, 400);
|
return ctx.json({ error: 'Whisper not configured — set it up in Settings → Speech to Text' }, 400);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
const user = ctx.get('user');
|
||||||
const body = await ctx.req.parseBody();
|
const body = await ctx.req.parseBody();
|
||||||
const file = body['file'];
|
const file = body['file'];
|
||||||
if (!file || !(file instanceof File)) {
|
if (!file || !(file instanceof File)) {
|
||||||
return ctx.json({ error: 'file is required' }, 400);
|
return ctx.json({ error: 'file is required' }, 400);
|
||||||
}
|
}
|
||||||
|
|
||||||
const formData = new FormData();
|
const settings = (await getUserSettings(user.id)) as { languages?: { spoken?: string[] } };
|
||||||
formData.append('file', file, 'recording.wav');
|
const spokenLanguages = settings.languages?.spoken ?? [];
|
||||||
formData.append('temperature', String(body['temperature'] ?? '0.0'));
|
|
||||||
formData.append('temperature_inc', String(body['temperature_inc'] ?? '0.2'));
|
|
||||||
formData.append('response_format', String(body['response_format'] ?? 'json'));
|
|
||||||
|
|
||||||
try {
|
try {
|
||||||
const res = await fetch(`${sttConfig.url.replace(/\/+$/, '')}/inference`, {
|
const result = await transcribeAudio({ file, whisperUrl: sttConfig.url, spokenLanguages });
|
||||||
method: 'POST',
|
return ctx.json(result);
|
||||||
body: formData,
|
|
||||||
});
|
|
||||||
if (!res.ok) {
|
|
||||||
return ctx.json({ error: `Whisper returned ${res.status}` }, 502);
|
|
||||||
}
|
|
||||||
const json = await res.json();
|
|
||||||
return ctx.json(json);
|
|
||||||
} catch (err) {
|
} catch (err) {
|
||||||
logger.error('STT proxy failed', { error: String(err) });
|
logger.error('STT proxy failed', { error: String(err) });
|
||||||
return ctx.json({ error: 'Failed to reach Whisper server' }, 502);
|
return ctx.json({ error: err instanceof Error ? err.message : 'Failed to reach Whisper server' }, 502);
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
|
|||||||
@@ -0,0 +1,89 @@
|
|||||||
|
// Whisper transcription with language-aware translation.
|
||||||
|
//
|
||||||
|
// Strategy: detect the audio's language, then transcribe in that language.
|
||||||
|
// Only ask Whisper to translate (which is English-only) when the detected
|
||||||
|
// language is NOT one the user speaks — i.e. they wouldn't understand the
|
||||||
|
// raw transcript anyway.
|
||||||
|
|
||||||
|
type TranscribeArgs = {
|
||||||
|
file: File | Blob;
|
||||||
|
whisperUrl: string;
|
||||||
|
spokenLanguages: string[];
|
||||||
|
};
|
||||||
|
|
||||||
|
export type TranscribeResult = {
|
||||||
|
text: string;
|
||||||
|
detectedLanguage: string;
|
||||||
|
translated: boolean;
|
||||||
|
};
|
||||||
|
|
||||||
|
// whisper.cpp's "language" field uses full names ("english"); ISO codes live in
|
||||||
|
// `language_probabilities`. Map covers what the user can pick in Profile settings.
|
||||||
|
const LANG_NAME_TO_ISO: Record<string, string> = {
|
||||||
|
english: 'en',
|
||||||
|
portuguese: 'pt',
|
||||||
|
spanish: 'es',
|
||||||
|
french: 'fr',
|
||||||
|
german: 'de',
|
||||||
|
italian: 'it',
|
||||||
|
dutch: 'nl',
|
||||||
|
russian: 'ru',
|
||||||
|
chinese: 'zh',
|
||||||
|
japanese: 'ja',
|
||||||
|
korean: 'ko',
|
||||||
|
};
|
||||||
|
|
||||||
|
export async function transcribeAudio({ file, whisperUrl, spokenLanguages }: TranscribeArgs): Promise<TranscribeResult> {
|
||||||
|
const base = whisperUrl.replace(/\/+$/, '');
|
||||||
|
|
||||||
|
// Step 1: detect language. Bias toward the user's spoken languages — Whisper's
|
||||||
|
// raw auto-detect is unreliable on short clips and often picks English by default.
|
||||||
|
const detectForm = new FormData();
|
||||||
|
detectForm.append('file', file);
|
||||||
|
detectForm.append('temperature', '0.0');
|
||||||
|
detectForm.append('response_format', 'verbose_json');
|
||||||
|
detectForm.append('detect_language', 'true');
|
||||||
|
|
||||||
|
const detectRes = await fetch(`${base}/inference`, { method: 'POST', body: detectForm });
|
||||||
|
if (!detectRes.ok) throw new Error(`Language detection failed (${detectRes.status})`);
|
||||||
|
const detectJson = (await detectRes.json()) as {
|
||||||
|
language?: string;
|
||||||
|
language_probabilities?: Record<string, number>;
|
||||||
|
};
|
||||||
|
|
||||||
|
const probs = detectJson.language_probabilities;
|
||||||
|
let detectedLanguage: string;
|
||||||
|
if (probs && spokenLanguages.length > 0) {
|
||||||
|
// Pick the spoken language with the highest probability
|
||||||
|
const ranked = spokenLanguages
|
||||||
|
.map((code) => [code, probs[code] ?? 0] as const)
|
||||||
|
.sort((a, b) => b[1] - a[1]);
|
||||||
|
detectedLanguage = ranked[0]?.[0] ?? spokenLanguages[0]!;
|
||||||
|
} else if (probs) {
|
||||||
|
// No spoken-language hint: take the highest-probability ISO code
|
||||||
|
const top = Object.entries(probs).sort((a, b) => b[1] - a[1])[0];
|
||||||
|
detectedLanguage = top?.[0] ?? 'en';
|
||||||
|
} else {
|
||||||
|
// No probabilities field (non-whisper.cpp backend) — try to map the name
|
||||||
|
const raw = (detectJson.language ?? 'en').toLowerCase();
|
||||||
|
detectedLanguage = LANG_NAME_TO_ISO[raw] ?? raw;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Step 2: translate only when the user doesn't speak the detected language
|
||||||
|
const translated = spokenLanguages.length > 0 && !spokenLanguages.includes(detectedLanguage);
|
||||||
|
|
||||||
|
// Step 3: full transcription
|
||||||
|
const transcribeForm = new FormData();
|
||||||
|
transcribeForm.append('file', file);
|
||||||
|
transcribeForm.append('temperature', '0.0');
|
||||||
|
transcribeForm.append('temperature_inc', '0.2');
|
||||||
|
transcribeForm.append('response_format', 'text');
|
||||||
|
transcribeForm.append('language', detectedLanguage);
|
||||||
|
if (translated) transcribeForm.append('translate', 'true');
|
||||||
|
|
||||||
|
const res = await fetch(`${base}/inference`, { method: 'POST', body: transcribeForm });
|
||||||
|
if (!res.ok) throw new Error(`Transcription failed (${res.status})`);
|
||||||
|
const text = (await res.text()).trim();
|
||||||
|
|
||||||
|
return { text, detectedLanguage, translated };
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user