Tonemark is a SvelteKit PWA for transcribing YouTube videos, audio and video files, and microphone recordings using a local Whisper backend. Features: - Dark glassmorphic UI with electric-lime accent (5 switchable themes) - Rail nav (desktop) / tab bar (mobile) layout - Drop zone, YouTube URL input, and live audio recording inputs - Audio mode waveform cards (none / standard / aggressive / auto) - Real-time transcription progress with animated waveform - Job queue with SSE streaming updates - Push notifications on job completion - PWA with native SvelteKit service worker - SRT / TXT / MD / JSON transcript downloads Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>
This commit is contained in:
@@ -0,0 +1,153 @@
|
||||
import { execFile } from 'child_process';
|
||||
import { promisify } from 'util';
|
||||
import { existsSync } from 'fs';
|
||||
import { mkdir, unlink, rename } from 'fs/promises';
|
||||
import { join } from 'path';
|
||||
import type { AudioMode, AudioAnalysis } from '$lib/types.js';
|
||||
|
||||
const execFileAsync = promisify(execFile);
|
||||
|
||||
const TMP_DIR = join(process.env.DATA_DIR ?? '/tmp/.whisper-pwa', 'audio');
|
||||
|
||||
export async function ensureTmpDir() {
|
||||
if (!existsSync(TMP_DIR)) await mkdir(TMP_DIR, { recursive: true });
|
||||
}
|
||||
|
||||
export function tmpPath(jobId: string, suffix: string) {
|
||||
return join(TMP_DIR, `${jobId}${suffix}`);
|
||||
}
|
||||
|
||||
export async function cleanup(...paths: string[]) {
|
||||
await Promise.allSettled(paths.map((p) => unlink(p).catch(() => {})));
|
||||
}
|
||||
|
||||
/** Run ffmpeg volumedetect and return mean/max dB. */
|
||||
export async function analyzeVolume(inputPath: string): Promise<AudioAnalysis> {
|
||||
const { stderr } = await execFileAsync('ffmpeg', [
|
||||
'-i', inputPath,
|
||||
'-af', 'volumedetect',
|
||||
'-vn', '-sn', '-dn',
|
||||
'-f', 'null', '-'
|
||||
]);
|
||||
const meanMatch = stderr.match(/mean_volume:\s*([-\d.]+)\s*dB/);
|
||||
const maxMatch = stderr.match(/max_volume:\s*([-\d.]+)\s*dB/);
|
||||
return {
|
||||
meanVolume: meanMatch ? parseFloat(meanMatch[1]) : -99,
|
||||
maxVolume: maxMatch ? parseFloat(maxMatch[1]) : -99
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Detect leading silence duration (ms).
|
||||
* Only trims if silence begins at/near time 0 (< 0.5s).
|
||||
* Capped at 30s to prevent accidental over-trimming.
|
||||
*/
|
||||
async function detectLeadingSilenceMs(inputPath: string): Promise<number> {
|
||||
try {
|
||||
const { stderr } = await execFileAsync('ffmpeg', [
|
||||
'-i', inputPath,
|
||||
'-af', 'silencedetect=n=-40dB:d=0.1',
|
||||
'-vn', '-sn', '-dn',
|
||||
'-f', 'null', '-'
|
||||
]);
|
||||
const startMatch = stderr.match(/silence_start:\s*([\d.]+)/);
|
||||
const endMatch = stderr.match(/silence_end:\s*([\d.]+)/);
|
||||
// Only trim if silence genuinely starts at the very beginning of the file
|
||||
if (startMatch && endMatch && parseFloat(startMatch[1]) < 0.5) {
|
||||
return Math.min(Math.floor(parseFloat(endMatch[1]) * 1000), 30_000);
|
||||
}
|
||||
} catch {
|
||||
// ignore
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/** Build ffmpeg -af filter chain for the given mode and mean volume. */
|
||||
export function buildFilterChain(mode: AudioMode, meanVolume: number): string | null {
|
||||
const isQuiet = meanVolume < -30;
|
||||
|
||||
switch (mode) {
|
||||
case 'none':
|
||||
return null;
|
||||
|
||||
case 'standard':
|
||||
return 'highpass=f=80,lowpass=f=8000,loudnorm=I=-16:LRA=11:TP=-1.5';
|
||||
|
||||
case 'aggressive':
|
||||
return [
|
||||
'highpass=f=80',
|
||||
isQuiet ? 'volume=24dB,dynaudnorm=f=500:g=15' : null,
|
||||
'lowpass=f=8000',
|
||||
'afftdn=nf=-30',
|
||||
'agate=threshold=0.01:attack=5:release=50',
|
||||
'loudnorm=I=-16:LRA=11:TP=-1.5'
|
||||
]
|
||||
.filter(Boolean)
|
||||
.join(',');
|
||||
|
||||
case 'auto':
|
||||
default:
|
||||
if (isQuiet) {
|
||||
return [
|
||||
'highpass=f=80',
|
||||
'volume=24dB',
|
||||
'dynaudnorm=f=500:g=15',
|
||||
'lowpass=f=8000',
|
||||
'afftdn=nf=-25',
|
||||
'loudnorm=I=-16:LRA=11:TP=-1.5'
|
||||
].join(',');
|
||||
}
|
||||
return 'highpass=f=80,lowpass=f=8000,loudnorm=I=-16:LRA=11:TP=-1.5';
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Prepare audio for Whisper: convert to 16kHz mono WAV, trim leading silence,
|
||||
* apply the appropriate filter chain.
|
||||
* Returns path to the prepared WAV file.
|
||||
*/
|
||||
export async function prepareAudio(
|
||||
inputPath: string,
|
||||
jobId: string,
|
||||
mode: AudioMode
|
||||
): Promise<{ wavPath: string; analysis: AudioAnalysis }> {
|
||||
await ensureTmpDir();
|
||||
|
||||
// Step 1: analyse volume on the original file
|
||||
const analysis = await analyzeVolume(inputPath);
|
||||
|
||||
// Step 2: detect leading silence
|
||||
const silenceMs = await detectLeadingSilenceMs(inputPath);
|
||||
|
||||
const wavPath = tmpPath(jobId, '.wav');
|
||||
const filterChain = buildFilterChain(mode, analysis.meanVolume);
|
||||
|
||||
const args: string[] = ['-y'];
|
||||
|
||||
// Trim leading silence
|
||||
if (silenceMs > 0) {
|
||||
args.push('-ss', (silenceMs / 1000).toFixed(3));
|
||||
}
|
||||
|
||||
args.push('-i', inputPath, '-ar', '16000', '-ac', '1');
|
||||
|
||||
if (filterChain) {
|
||||
args.push('-af', filterChain);
|
||||
}
|
||||
|
||||
args.push('-c:a', 'pcm_s16le', wavPath);
|
||||
|
||||
await execFileAsync('ffmpeg', args);
|
||||
return { wavPath, analysis };
|
||||
}
|
||||
|
||||
/** Move a file to a new path (cross-device safe). */
|
||||
export async function moveFile(src: string, dest: string) {
|
||||
try {
|
||||
await rename(src, dest);
|
||||
} catch {
|
||||
const { copyFile } = await import('fs/promises');
|
||||
await copyFile(src, dest);
|
||||
await unlink(src);
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user