From 0f7d5e950c5325a77fe8e7b86eb7fa81f9a60285 Mon Sep 17 00:00:00 2001 From: Andrea Date: Sat, 26 Sep 2026 21:51:50 +0200 Subject: [PATCH] Enhance audio processing capabilities: add FFmpeg for audio transcoding, update database schema, and improve UI for audio file uploads --- Dockerfile | 5 ++- README.md | 8 +++-- app/main.py | 86 +++++++++++++++++++++++++++++++++++++++----- docker-compose.yml | 2 +- frontend/app.js | 37 ++++++++++++++++--- frontend/favicon.svg | 2 +- frontend/index.html | 7 +++- frontend/styles.css | 8 ++++- 8 files changed, 135 insertions(+), 20 deletions(-) diff --git a/Dockerfile b/Dockerfile index d7e9a80..fce1d8f 100644 --- a/Dockerfile +++ b/Dockerfile @@ -2,7 +2,10 @@ FROM python:3.12-slim WORKDIR /srv/faerro-kb COPY requirements.txt . -RUN pip install --no-cache-dir -r requirements.txt +RUN apt-get update \ + && apt-get install -y --no-install-recommends ffmpeg \ + && rm -rf /var/lib/apt/lists/* \ + && pip install --no-cache-dir -r requirements.txt COPY app ./app COPY frontend ./frontend diff --git a/README.md b/README.md index 50ec384..6ae19cb 100644 --- a/README.md +++ b/README.md @@ -27,21 +27,23 @@ In Safari on iPhone, open the HTTPS address and use **Share > Add to Home Screen ## Capture and offline behavior - Recording data and its timestamp are committed to IndexedDB before upload is attempted. +- Microphone recordings use the audio format supported by the browser (commonly M4A/AAC on iPhone Safari or Opus/WebM elsewhere). You can also add an audio file from the device; both sources remain queued locally until the server confirms storage. +- The server keeps each uploaded original and creates a mono PCM WAV copy at 16-bit / 22.05 kHz for transcription. FFmpeg performs this conversion locally in the container. After a non-empty transcript is stored, both audio files are deleted; failed or empty transcriptions keep the audio for recovery. - Uploads retry when the app opens, returns to the foreground, or the browser reports a connection. A recording is removed from the phone's queue only after the server confirms it was stored. - Service worker caching keeps the app shell available offline. Searching the server-side archive requires a connection. - iOS may suspend a PWA and can evict website data under storage pressure. Background uploads are not guaranteed; reopen the PWA while online to resume. Keep the phone powered and avoid clearing Safari website data for stronger practical retention. ## Local speech-to-text -The API can use a local `whisper.cpp` server when `WHISPER_SERVER_URL` is configured. Audio is sent only to that local endpoint; with the variable unset, uploads remain stored as `awaiting_transcription` and can still be searched after a transcript is edited or populated. +The API can use a local `whisper.cpp` server when `WHISPER_SERVER_URL` is configured. Audio is sent only to that local endpoint; with the variable unset, uploads remain stored as `awaiting_transcription` until a non-empty transcript is entered and saved. Saving that transcript deletes both server-side audio files. For NVIDIA GPU transcription, run a CUDA-enabled `whisper.cpp` server with a model on the same machine, then set `WHISPER_SERVER_URL` to its local inference endpoint, for example `http://whisper:8080/inference`. Use a model that fits the RTX 3070's VRAM; start with `small` or a quantized `medium` model and measure with your audio/language. Do not expose the transcription server outside the private Docker network. See the [whisper.cpp NVIDIA instructions](https://github.com/ggml-org/whisper.cpp#nvidia-gpu-support) for CUDA builds. -This version keeps raw audio and transcript separate. A future local correction model can propose punctuation and name fixes without replacing the original transcription. Proper-name matching and offline geolocation data are not enabled yet. +The transcript remains in SQLite after the audio files are deleted. A future local correction model can propose punctuation and name fixes without replacing the stored transcription. Proper-name matching and offline geolocation data are not enabled yet. ## Data and backups -Docker persists SQLite metadata in `./data/voice-kb.sqlite3` and original audio in `./data/audio/`. Back up the whole `./data` directory while the service is stopped or use a SQLite-aware backup procedure. Model files should also be stored locally and backed up separately if needed. +Docker persists SQLite metadata in `./data/voice-kb.sqlite3` and audio awaiting transcription in `./data/audio/`. After a non-empty transcript is saved, its audio files are removed. Back up the whole `./data` directory while the service is stopped or use a SQLite-aware backup procedure. Model files should also be stored locally and backed up separately if needed. ## Development diff --git a/app/main.py b/app/main.py index da355f2..689ea67 100644 --- a/app/main.py +++ b/app/main.py @@ -1,6 +1,9 @@ +import asyncio import os import re import sqlite3 +import subprocess +import tempfile import uuid from datetime import datetime, timezone from pathlib import Path @@ -44,10 +47,26 @@ def initialize_db() -> None: """CREATE VIRTUAL TABLE IF NOT EXISTS note_search USING fts5(note_id UNINDEXED, transcript)""" ) + columns = {row[1] for row in connection.execute("PRAGMA table_info(notes)")} + if "original_audio_path" not in columns: + connection.execute("ALTER TABLE notes ADD COLUMN original_audio_path TEXT") initialize_db() +def transcode_audio(source_path: Path, output_path: Path) -> None: + subprocess.run( + [ + "ffmpeg", "-nostdin", "-hide_banner", "-loglevel", "error", "-y", + "-i", str(source_path), "-vn", "-ac", "1", "-ar", "22050", + "-c:a", "pcm_s16le", str(output_path), + ], + check=True, + stdout=subprocess.DEVNULL, + stderr=subprocess.PIPE, + timeout=300, + ) + def update_transcript(note_id: str, transcript: str, status: str) -> None: with connect_db() as connection: @@ -63,6 +82,23 @@ def update_transcript(note_id: str, transcript: str, status: str) -> None: ) +def remove_transcribed_audio(note_id: str) -> None: + with connect_db() as connection: + note = connection.execute( + "SELECT audio_path, original_audio_path FROM notes WHERE id = ?", (note_id,) + ).fetchone() + if not note: + return + for audio_path in (note["audio_path"], note["original_audio_path"]): + if audio_path: + Path(audio_path).unlink(missing_ok=True) + with connect_db() as connection: + connection.execute( + "UPDATE notes SET audio_path = '', original_audio_path = NULL WHERE id = ?", + (note_id,), + ) + + async def transcribe_note(note_id: str, audio_path: Path) -> None: if not WHISPER_SERVER_URL: return @@ -76,9 +112,12 @@ async def transcribe_note(note_id: str, audio_path: Path) -> None: ) response.raise_for_status() transcript = response.json().get("text", "").strip() - update_transcript(note_id, transcript, "transcribed" if transcript else "empty_transcript") except Exception: update_transcript(note_id, "", "transcription_failed") + return + update_transcript(note_id, transcript, "transcribed" if transcript else "empty_transcript") + if transcript: + remove_transcribed_audio(note_id) @app.get("/api/health") @@ -102,15 +141,39 @@ async def create_note( if existing: return {"id": note_id, "status": existing["status"]} suffix = Path(audio.filename or "recording.webm").suffix.lower() - if not suffix or len(suffix) > 10: + if not re.fullmatch(r"\.[a-z0-9]{1,9}", suffix): suffix = ".audio" - audio_path = AUDIO_DIR / f"{note_id}{suffix}" + original_audio_path = AUDIO_DIR / f"{note_id}-original{suffix}" + audio_path = AUDIO_DIR / f"{note_id}.wav" + temporary_output = None try: - with audio_path.open("wb") as destination: + with original_audio_path.open("wb") as destination: while chunk := await audio.read(1024 * 1024): destination.write(chunk) + with tempfile.NamedTemporaryFile(suffix=".wav", dir=AUDIO_DIR, delete=False) as temporary_file: + temporary_output = Path(temporary_file.name) + await asyncio.to_thread(transcode_audio, original_audio_path, temporary_output) + temporary_output.replace(audio_path) + except FileNotFoundError as error: + original_audio_path.unlink(missing_ok=True) + if temporary_output: + temporary_output.unlink(missing_ok=True) + raise HTTPException(status_code=500, detail="Audio conversion is unavailable") from error + except subprocess.CalledProcessError as error: + original_audio_path.unlink(missing_ok=True) + if temporary_output: + temporary_output.unlink(missing_ok=True) + raise HTTPException(status_code=400, detail="Unsupported or invalid audio file") from error + except subprocess.TimeoutExpired as error: + original_audio_path.unlink(missing_ok=True) + if temporary_output: + temporary_output.unlink(missing_ok=True) + raise HTTPException(status_code=422, detail="Audio conversion timed out") from error except Exception as error: + original_audio_path.unlink(missing_ok=True) audio_path.unlink(missing_ok=True) + if temporary_output: + temporary_output.unlink(missing_ok=True) raise HTTPException(status_code=500, detail="Could not store audio") from error finally: await audio.close() @@ -126,8 +189,10 @@ async def create_note( status = "awaiting_transcription" if not WHISPER_SERVER_URL else "transcribing" with connect_db() as connection: connection.execute( - "INSERT OR IGNORE INTO notes (id, created_at, filename, audio_path, status) VALUES (?, ?, ?, ?, ?)", - (note_id, created_at, audio.filename or "recording.audio", str(audio_path), status), + """INSERT OR IGNORE INTO notes + (id, created_at, filename, audio_path, original_audio_path, status) + VALUES (?, ?, ?, ?, ?, ?)""", + (note_id, created_at, audio.filename or "recording.audio", str(audio_path), str(original_audio_path), status), ) if WHISPER_SERVER_URL: background_tasks.add_task(transcribe_note, note_id, audio_path) @@ -173,13 +238,16 @@ def list_notes(q: str = "", status: str = "") -> list[dict[str, str]]: def delete_note(note_id: str) -> dict[str, str]: with connect_db() as connection: note = connection.execute( - "SELECT audio_path FROM notes WHERE id = ?", (note_id,) + "SELECT audio_path, original_audio_path FROM notes WHERE id = ?", (note_id,) ).fetchone() if not note: raise HTTPException(status_code=404, detail="Note not found") connection.execute("DELETE FROM note_search WHERE note_id = ?", (note_id,)) connection.execute("DELETE FROM notes WHERE id = ?", (note_id,)) - Path(note["audio_path"]).unlink(missing_ok=True) + if note["audio_path"]: + Path(note["audio_path"]).unlink(missing_ok=True) + if note["original_audio_path"]: + Path(note["original_audio_path"]).unlink(missing_ok=True) return {"id": note_id, "status": "deleted"} @@ -191,6 +259,8 @@ def save_transcript(note_id: str, payload: dict[str, str]) -> dict[str, str]: if not exists: raise HTTPException(status_code=404, detail="Note not found") update_transcript(note_id, transcript, "transcribed" if transcript else "awaiting_transcription") + if transcript: + remove_transcribed_audio(note_id) return {"id": note_id, "status": "transcribed" if transcript else "awaiting_transcription"} diff --git a/docker-compose.yml b/docker-compose.yml index f6e078a..dcd8997 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -1,6 +1,6 @@ services: faerro-kb: - image: gitea.faerro.it/andrea/faerro-kb:0.0.1 + image: gitea.faerro.it/andrea/faerro-kb:latest ports: - "${HTTP_PORT:-80}:80" environment: diff --git a/frontend/app.js b/frontend/app.js index 1764a4d..9d91a86 100644 --- a/frontend/app.js +++ b/frontend/app.js @@ -2,7 +2,7 @@ const DB_NAME = 'faerro-kb-local'; const STORE_NAME = 'recordings'; const MAX_RECORDING_MS = 30 * 60 * 1000; const elements = Object.fromEntries([ - 'record-button', 'record-title', 'record-help', 'record-time', 'record-meter-fill', 'audio-level', 'audio-level-fill', 'capture-message', + 'record-button', 'record-title', 'record-help', 'record-time', 'record-meter-fill', 'audio-level', 'audio-level-fill', 'capture-message', 'audio-file-input', 'queue-list', 'queue-empty', 'queue-count', 'sync-button', 'archive-list', 'archive-empty', 'network-state', 'storage-status', ].map((id) => [id, document.getElementById(id)])); @@ -53,7 +53,9 @@ function updateNetwork() { const online = navigator.onLine; elements['network-state'].classList.toggle('online', online); elements['network-state'].innerHTML = `${online ? 'Connected' : 'Offline'}`; - elements['storage-status'].textContent = online ? 'Pending audio stays here until server confirmation.' : 'Offline: new audio is saved on this device.'; + if (elements['storage-status']) { + elements['storage-status'].textContent = online ? 'Pending audio stays here until server confirmation.' : 'Offline: new audio is saved on this device.'; + } } function formatDuration(milliseconds) { @@ -170,7 +172,7 @@ async function saveRecording() { setMessage('No audio was captured. Please try again.', true); return; } - const extension = blob.type.includes('webm') ? 'webm' : blob.type.includes('ogg') ? 'ogg' : 'm4a'; + const extension = blob.type.includes('webm') ? 'webm' : blob.type.includes('ogg') ? 'ogg' : blob.type.includes('mp4') ? 'm4a' : 'audio'; const createdAt = new Date(recordingStartedAt).toISOString(); const recording = { id: crypto.randomUUID(), @@ -190,6 +192,27 @@ async function saveRecording() { } } +async function queueAudioFile(file) { + const createdAtMs = Date.now(); + const createdAt = new Date(createdAtMs).toISOString(); + const recording = { + id: crypto.randomUUID(), + blob: file, + createdAt, + filename: file.name, + durationMs: 0, + createdAtMs, + }; + try { + await putQueued(recording); + setMessage(navigator.onLine ? 'Audio saved on device. Uploading now…' : 'Audio saved on device. It will upload when you reconnect.'); + await renderQueue(); + await syncQueue(); + } catch { + setMessage('Could not save this file in browser storage. Check available device storage and try again.', true); + } +} + async function syncQueue() { if (syncing || !navigator.onLine) return; syncing = true; @@ -247,7 +270,7 @@ async function renderQueue() { status.textContent = navigator.onLine ? 'WAITING TO SYNC' : 'SAVED OFFLINE'; const meta = document.createElement('div'); meta.className = 'item-meta'; - meta.textContent = `${prettyDate(recording.createdAt)} · ${formatDuration(recording.durationMs)}`; + meta.textContent = `${prettyDate(recording.createdAt)}${recording.durationMs ? ` · ${formatDuration(recording.durationMs)}` : ''}`; const remove = createDeleteButton(`Delete queued note ${recording.filename}`); remove.disabled = syncing; remove.addEventListener('click', async () => { @@ -355,6 +378,12 @@ async function loadAwaitingNotes() { } elements['record-button'].addEventListener('click', () => mediaRecorder?.state === 'recording' ? stopRecording() : startRecording()); +elements['audio-file-input'].addEventListener('change', async (event) => { + const [file] = event.target.files; + if (file) await queueAudioFile(file); + event.target.value = ''; +}); +elements['sync-button'].addEventListener('click', syncQueue); elements['sync-button'].addEventListener('click', syncQueue); window.addEventListener('online', () => { updateNetwork(); syncQueue(); loadAwaitingNotes(); }); window.addEventListener('offline', updateNetwork); diff --git a/frontend/favicon.svg b/frontend/favicon.svg index 38364cb..f1caf0a 100644 --- a/frontend/favicon.svg +++ b/frontend/favicon.svg @@ -1,4 +1,4 @@ - KB + KB \ No newline at end of file diff --git a/frontend/index.html b/frontend/index.html index 3c71a94..3141007 100644 --- a/frontend/index.html +++ b/frontend/index.html @@ -13,7 +13,7 @@
- VFaerro KB + KBFaerro KB Checking
@@ -32,6 +32,11 @@ +
+ + + Files stay on this device until the server confirms storage. +
diff --git a/frontend/styles.css b/frontend/styles.css index aa1ab77..033b939 100644 --- a/frontend/styles.css +++ b/frontend/styles.css @@ -18,7 +18,7 @@ body { margin: 0; background: var(--paper); color: var(--ink); font-family: var( .topbar { height: 68px; padding: 0 max(24px, calc((100vw - 1040px) / 2)); border-bottom: 1px solid var(--line); display: flex; align-items: center; justify-content: space-between; } .brand { display: flex; align-items: center; gap: 10px; text-decoration: none; color: var(--ink); font-size: 15px; font-weight: 600; } .brand b { color: var(--moss); } -.brand-mark { width: 28px; height: 28px; border-radius: 50%; background: var(--moss-dark); color: #fff; display: grid; place-items: center; font: 500 15px var(--serif); } +.brand-mark { width: 28px; height: 28px; flex: 0 0 auto; border-radius: 50%; background: var(--moss-dark); color: #fff; display: grid; place-items: center; font: 500 15px var(--serif); } .network-state { font: 11px var(--mono); color: var(--muted); display: flex; align-items: center; gap: 8px; } .network-state i { width: 7px; height: 7px; border-radius: 50%; background: var(--coral); } .network-state.online i { background: #57986b; } @@ -44,6 +44,11 @@ h1 em { color: var(--moss); font-weight: 500; } .record-meter { height: 2px; max-width: 590px; margin-inline: auto; background: #e4e3dc; } .record-meter span { display: block; height: 100%; width: 0; background: var(--coral); transition: width .25s linear; } .capture-message { height: 20px; max-width: 590px; margin-inline: auto; padding-top: 6px; color: var(--muted); font-size: 11px; text-align: center; } +.audio-upload-row { max-width: 590px; margin: 14px auto 0; display: flex; align-items: center; justify-content: center; gap: 12px; } +.audio-upload-button { display: inline-flex; min-height: 36px; align-items: center; gap: 7px; border: 1px solid var(--line); padding: 0 12px; color: var(--moss-dark); background: #fffefa; font-size: 12px; cursor: pointer; } +.audio-upload-button:hover { border-color: var(--moss); } +#audio-file-input { position: absolute; width: 1px; height: 1px; overflow: hidden; clip: rect(0, 0, 0, 0); white-space: nowrap; clip-path: inset(50%); } +.audio-upload-row > span { color: var(--muted); font-size: 10px; } .queue-section, .archive-section { padding: 30px 0 34px; border-bottom: 1px solid var(--line); } .section-heading { position: relative; display: flex; align-items: center; justify-content: center; gap: 18px; text-align: center; } .section-heading .section-kicker { justify-content: center; } @@ -80,6 +85,7 @@ footer { max-width: 1040px; margin: auto; padding: 18px 24px calc(18px + env(saf .record-button { width: 46px; height: 46px; } .record-copy strong { font-size: 12px; } .record-copy span { font-size: 10px; } + .audio-upload-row { align-items: flex-start; flex-direction: column; } h2 { font-size: 23px; } .archive-date { width: 60px; font-size: 9px; } footer { padding-left: 18px; padding-right: 18px; font-size: 8px; }