diff --git a/Dockerfile b/Dockerfile index d7e9a80..fce1d8f 100644 --- a/Dockerfile +++ b/Dockerfile @@ -2,7 +2,10 @@ FROM python:3.12-slim WORKDIR /srv/faerro-kb COPY requirements.txt . -RUN pip install --no-cache-dir -r requirements.txt +RUN apt-get update \ + && apt-get install -y --no-install-recommends ffmpeg \ + && rm -rf /var/lib/apt/lists/* \ + && pip install --no-cache-dir -r requirements.txt COPY app ./app COPY frontend ./frontend diff --git a/README.md b/README.md index 50ec384..6ae19cb 100644 --- a/README.md +++ b/README.md @@ -27,21 +27,23 @@ In Safari on iPhone, open the HTTPS address and use **Share > Add to Home Screen ## Capture and offline behavior - Recording data and its timestamp are committed to IndexedDB before upload is attempted. +- Microphone recordings use the audio format supported by the browser (commonly M4A/AAC on iPhone Safari or Opus/WebM elsewhere). You can also add an audio file from the device; both sources remain queued locally until the server confirms storage. +- The server keeps each uploaded original and creates a mono PCM WAV copy at 16-bit / 22.05 kHz for transcription. FFmpeg performs this conversion locally in the container. After a non-empty transcript is stored, both audio files are deleted; failed or empty transcriptions keep the audio for recovery. - Uploads retry when the app opens, returns to the foreground, or the browser reports a connection. A recording is removed from the phone's queue only after the server confirms it was stored. - Service worker caching keeps the app shell available offline. Searching the server-side archive requires a connection. - iOS may suspend a PWA and can evict website data under storage pressure. Background uploads are not guaranteed; reopen the PWA while online to resume. Keep the phone powered and avoid clearing Safari website data for stronger practical retention. ## Local speech-to-text -The API can use a local `whisper.cpp` server when `WHISPER_SERVER_URL` is configured. Audio is sent only to that local endpoint; with the variable unset, uploads remain stored as `awaiting_transcription` and can still be searched after a transcript is edited or populated. +The API can use a local `whisper.cpp` server when `WHISPER_SERVER_URL` is configured. Audio is sent only to that local endpoint; with the variable unset, uploads remain stored as `awaiting_transcription` until a non-empty transcript is entered and saved. Saving that transcript deletes both server-side audio files. For NVIDIA GPU transcription, run a CUDA-enabled `whisper.cpp` server with a model on the same machine, then set `WHISPER_SERVER_URL` to its local inference endpoint, for example `http://whisper:8080/inference`. Use a model that fits the RTX 3070's VRAM; start with `small` or a quantized `medium` model and measure with your audio/language. Do not expose the transcription server outside the private Docker network. See the [whisper.cpp NVIDIA instructions](https://github.com/ggml-org/whisper.cpp#nvidia-gpu-support) for CUDA builds. -This version keeps raw audio and transcript separate. A future local correction model can propose punctuation and name fixes without replacing the original transcription. Proper-name matching and offline geolocation data are not enabled yet. +The transcript remains in SQLite after the audio files are deleted. A future local correction model can propose punctuation and name fixes without replacing the stored transcription. Proper-name matching and offline geolocation data are not enabled yet. ## Data and backups -Docker persists SQLite metadata in `./data/voice-kb.sqlite3` and original audio in `./data/audio/`. Back up the whole `./data` directory while the service is stopped or use a SQLite-aware backup procedure. Model files should also be stored locally and backed up separately if needed. +Docker persists SQLite metadata in `./data/voice-kb.sqlite3` and audio awaiting transcription in `./data/audio/`. After a non-empty transcript is saved, its audio files are removed. Back up the whole `./data` directory while the service is stopped or use a SQLite-aware backup procedure. Model files should also be stored locally and backed up separately if needed. ## Development diff --git a/app/main.py b/app/main.py index da355f2..689ea67 100644 --- a/app/main.py +++ b/app/main.py @@ -1,6 +1,9 @@ +import asyncio import os import re import sqlite3 +import subprocess +import tempfile import uuid from datetime import datetime, timezone from pathlib import Path @@ -44,10 +47,26 @@ def initialize_db() -> None: """CREATE VIRTUAL TABLE IF NOT EXISTS note_search USING fts5(note_id UNINDEXED, transcript)""" ) + columns = {row[1] for row in connection.execute("PRAGMA table_info(notes)")} + if "original_audio_path" not in columns: + connection.execute("ALTER TABLE notes ADD COLUMN original_audio_path TEXT") initialize_db() +def transcode_audio(source_path: Path, output_path: Path) -> None: + subprocess.run( + [ + "ffmpeg", "-nostdin", "-hide_banner", "-loglevel", "error", "-y", + "-i", str(source_path), "-vn", "-ac", "1", "-ar", "22050", + "-c:a", "pcm_s16le", str(output_path), + ], + check=True, + stdout=subprocess.DEVNULL, + stderr=subprocess.PIPE, + timeout=300, + ) + def update_transcript(note_id: str, transcript: str, status: str) -> None: with connect_db() as connection: @@ -63,6 +82,23 @@ def update_transcript(note_id: str, transcript: str, status: str) -> None: ) +def remove_transcribed_audio(note_id: str) -> None: + with connect_db() as connection: + note = connection.execute( + "SELECT audio_path, original_audio_path FROM notes WHERE id = ?", (note_id,) + ).fetchone() + if not note: + return + for audio_path in (note["audio_path"], note["original_audio_path"]): + if audio_path: + Path(audio_path).unlink(missing_ok=True) + with connect_db() as connection: + connection.execute( + "UPDATE notes SET audio_path = '', original_audio_path = NULL WHERE id = ?", + (note_id,), + ) + + async def transcribe_note(note_id: str, audio_path: Path) -> None: if not WHISPER_SERVER_URL: return @@ -76,9 +112,12 @@ async def transcribe_note(note_id: str, audio_path: Path) -> None: ) response.raise_for_status() transcript = response.json().get("text", "").strip() - update_transcript(note_id, transcript, "transcribed" if transcript else "empty_transcript") except Exception: update_transcript(note_id, "", "transcription_failed") + return + update_transcript(note_id, transcript, "transcribed" if transcript else "empty_transcript") + if transcript: + remove_transcribed_audio(note_id) @app.get("/api/health") @@ -102,15 +141,39 @@ async def create_note( if existing: return {"id": note_id, "status": existing["status"]} suffix = Path(audio.filename or "recording.webm").suffix.lower() - if not suffix or len(suffix) > 10: + if not re.fullmatch(r"\.[a-z0-9]{1,9}", suffix): suffix = ".audio" - audio_path = AUDIO_DIR / f"{note_id}{suffix}" + original_audio_path = AUDIO_DIR / f"{note_id}-original{suffix}" + audio_path = AUDIO_DIR / f"{note_id}.wav" + temporary_output = None try: - with audio_path.open("wb") as destination: + with original_audio_path.open("wb") as destination: while chunk := await audio.read(1024 * 1024): destination.write(chunk) + with tempfile.NamedTemporaryFile(suffix=".wav", dir=AUDIO_DIR, delete=False) as temporary_file: + temporary_output = Path(temporary_file.name) + await asyncio.to_thread(transcode_audio, original_audio_path, temporary_output) + temporary_output.replace(audio_path) + except FileNotFoundError as error: + original_audio_path.unlink(missing_ok=True) + if temporary_output: + temporary_output.unlink(missing_ok=True) + raise HTTPException(status_code=500, detail="Audio conversion is unavailable") from error + except subprocess.CalledProcessError as error: + original_audio_path.unlink(missing_ok=True) + if temporary_output: + temporary_output.unlink(missing_ok=True) + raise HTTPException(status_code=400, detail="Unsupported or invalid audio file") from error + except subprocess.TimeoutExpired as error: + original_audio_path.unlink(missing_ok=True) + if temporary_output: + temporary_output.unlink(missing_ok=True) + raise HTTPException(status_code=422, detail="Audio conversion timed out") from error except Exception as error: + original_audio_path.unlink(missing_ok=True) audio_path.unlink(missing_ok=True) + if temporary_output: + temporary_output.unlink(missing_ok=True) raise HTTPException(status_code=500, detail="Could not store audio") from error finally: await audio.close() @@ -126,8 +189,10 @@ async def create_note( status = "awaiting_transcription" if not WHISPER_SERVER_URL else "transcribing" with connect_db() as connection: connection.execute( - "INSERT OR IGNORE INTO notes (id, created_at, filename, audio_path, status) VALUES (?, ?, ?, ?, ?)", - (note_id, created_at, audio.filename or "recording.audio", str(audio_path), status), + """INSERT OR IGNORE INTO notes + (id, created_at, filename, audio_path, original_audio_path, status) + VALUES (?, ?, ?, ?, ?, ?)""", + (note_id, created_at, audio.filename or "recording.audio", str(audio_path), str(original_audio_path), status), ) if WHISPER_SERVER_URL: background_tasks.add_task(transcribe_note, note_id, audio_path) @@ -173,13 +238,16 @@ def list_notes(q: str = "", status: str = "") -> list[dict[str, str]]: def delete_note(note_id: str) -> dict[str, str]: with connect_db() as connection: note = connection.execute( - "SELECT audio_path FROM notes WHERE id = ?", (note_id,) + "SELECT audio_path, original_audio_path FROM notes WHERE id = ?", (note_id,) ).fetchone() if not note: raise HTTPException(status_code=404, detail="Note not found") connection.execute("DELETE FROM note_search WHERE note_id = ?", (note_id,)) connection.execute("DELETE FROM notes WHERE id = ?", (note_id,)) - Path(note["audio_path"]).unlink(missing_ok=True) + if note["audio_path"]: + Path(note["audio_path"]).unlink(missing_ok=True) + if note["original_audio_path"]: + Path(note["original_audio_path"]).unlink(missing_ok=True) return {"id": note_id, "status": "deleted"} @@ -191,6 +259,8 @@ def save_transcript(note_id: str, payload: dict[str, str]) -> dict[str, str]: if not exists: raise HTTPException(status_code=404, detail="Note not found") update_transcript(note_id, transcript, "transcribed" if transcript else "awaiting_transcription") + if transcript: + remove_transcribed_audio(note_id) return {"id": note_id, "status": "transcribed" if transcript else "awaiting_transcription"} diff --git a/docker-compose.yml b/docker-compose.yml index f6e078a..dcd8997 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -1,6 +1,6 @@ services: faerro-kb: - image: gitea.faerro.it/andrea/faerro-kb:0.0.1 + image: gitea.faerro.it/andrea/faerro-kb:latest ports: - "${HTTP_PORT:-80}:80" environment: diff --git a/frontend/app.js b/frontend/app.js index 1764a4d..9d91a86 100644 --- a/frontend/app.js +++ b/frontend/app.js @@ -2,7 +2,7 @@ const DB_NAME = 'faerro-kb-local'; const STORE_NAME = 'recordings'; const MAX_RECORDING_MS = 30 * 60 * 1000; const elements = Object.fromEntries([ - 'record-button', 'record-title', 'record-help', 'record-time', 'record-meter-fill', 'audio-level', 'audio-level-fill', 'capture-message', + 'record-button', 'record-title', 'record-help', 'record-time', 'record-meter-fill', 'audio-level', 'audio-level-fill', 'capture-message', 'audio-file-input', 'queue-list', 'queue-empty', 'queue-count', 'sync-button', 'archive-list', 'archive-empty', 'network-state', 'storage-status', ].map((id) => [id, document.getElementById(id)])); @@ -53,7 +53,9 @@ function updateNetwork() { const online = navigator.onLine; elements['network-state'].classList.toggle('online', online); elements['network-state'].innerHTML = `${online ? 'Connected' : 'Offline'}`; - elements['storage-status'].textContent = online ? 'Pending audio stays here until server confirmation.' : 'Offline: new audio is saved on this device.'; + if (elements['storage-status']) { + elements['storage-status'].textContent = online ? 'Pending audio stays here until server confirmation.' : 'Offline: new audio is saved on this device.'; + } } function formatDuration(milliseconds) { @@ -170,7 +172,7 @@ async function saveRecording() { setMessage('No audio was captured. Please try again.', true); return; } - const extension = blob.type.includes('webm') ? 'webm' : blob.type.includes('ogg') ? 'ogg' : 'm4a'; + const extension = blob.type.includes('webm') ? 'webm' : blob.type.includes('ogg') ? 'ogg' : blob.type.includes('mp4') ? 'm4a' : 'audio'; const createdAt = new Date(recordingStartedAt).toISOString(); const recording = { id: crypto.randomUUID(), @@ -190,6 +192,27 @@ async function saveRecording() { } } +async function queueAudioFile(file) { + const createdAtMs = Date.now(); + const createdAt = new Date(createdAtMs).toISOString(); + const recording = { + id: crypto.randomUUID(), + blob: file, + createdAt, + filename: file.name, + durationMs: 0, + createdAtMs, + }; + try { + await putQueued(recording); + setMessage(navigator.onLine ? 'Audio saved on device. Uploading now…' : 'Audio saved on device. It will upload when you reconnect.'); + await renderQueue(); + await syncQueue(); + } catch { + setMessage('Could not save this file in browser storage. Check available device storage and try again.', true); + } +} + async function syncQueue() { if (syncing || !navigator.onLine) return; syncing = true; @@ -247,7 +270,7 @@ async function renderQueue() { status.textContent = navigator.onLine ? 'WAITING TO SYNC' : 'SAVED OFFLINE'; const meta = document.createElement('div'); meta.className = 'item-meta'; - meta.textContent = `${prettyDate(recording.createdAt)} · ${formatDuration(recording.durationMs)}`; + meta.textContent = `${prettyDate(recording.createdAt)}${recording.durationMs ? ` · ${formatDuration(recording.durationMs)}` : ''}`; const remove = createDeleteButton(`Delete queued note ${recording.filename}`); remove.disabled = syncing; remove.addEventListener('click', async () => { @@ -355,6 +378,12 @@ async function loadAwaitingNotes() { } elements['record-button'].addEventListener('click', () => mediaRecorder?.state === 'recording' ? stopRecording() : startRecording()); +elements['audio-file-input'].addEventListener('change', async (event) => { + const [file] = event.target.files; + if (file) await queueAudioFile(file); + event.target.value = ''; +}); +elements['sync-button'].addEventListener('click', syncQueue); elements['sync-button'].addEventListener('click', syncQueue); window.addEventListener('online', () => { updateNetwork(); syncQueue(); loadAwaitingNotes(); }); window.addEventListener('offline', updateNetwork); diff --git a/frontend/favicon.svg b/frontend/favicon.svg index 38364cb..f1caf0a 100644 --- a/frontend/favicon.svg +++ b/frontend/favicon.svg @@ -1,4 +1,4 @@ \ No newline at end of file diff --git a/frontend/index.html b/frontend/index.html index 3c71a94..3141007 100644 --- a/frontend/index.html +++ b/frontend/index.html @@ -13,7 +13,7 @@