diff --git a/Dockerfile b/Dockerfile index fce1d8f..633fc4e 100644 --- a/Dockerfile +++ b/Dockerfile @@ -8,6 +8,7 @@ RUN apt-get update \ && pip install --no-cache-dir -r requirements.txt COPY app ./app COPY frontend ./frontend +COPY ["audio test.wav", "./audio test.wav"] ENV DATA_DIR=/data EXPOSE 80 diff --git a/README.md b/README.md index 9e3e7cb..a0ffade 100644 --- a/README.md +++ b/README.md @@ -38,7 +38,7 @@ In Safari on iPhone, open the HTTPS address and use **Share > Add to Home Screen ## Local speech-to-text -The API can use a local `onerahmet/openai-whisper-asr-webservice` server when `WHISPER_SERVER_URL` is configured. Set it to the service's `/asr` endpoint, for example `http://whisper:9000/asr`. Faerro KB sends the converted WAV as the `audio_file` multipart field with `task=transcribe` and `output=json`; audio is sent only to that local endpoint. With the variable unset, uploads remain stored as `awaiting_transcription` until a non-empty transcript is entered and saved. Saving that transcript deletes both server-side audio files. +The API can use a local `onerahmet/openai-whisper-asr-webservice` server when `WHISPER_SERVER_URL` is configured. Set it to the service's `/asr` endpoint, including its listening port, for example `http://whisper:9000/asr` or `http://192.168.1.102:9000/asr`. Faerro KB sends the converted WAV as the `audio_file` multipart field with `task=transcribe` and `output=json`; audio is sent only to that local endpoint. At startup, the API also sends the bundled `audio test.wav` to the endpoint as a connection test before resuming queued jobs. Logs report the HTTP result, detected language, segment count, and elapsed time, but never log the transcript. With the variable unset, uploads remain stored as `awaiting_transcription` until a non-empty transcript is entered and saved. When the API starts with a successful Whisper test, it resumes awaiting, interrupted, and failed transcription jobs whose audio is still stored. Restart/recreate the API container after changing this environment variable; configuration is read at startup. Saving a non-empty transcript deletes both server-side audio files. Run the ASR web service with a model that fits the available hardware, then set `WHISPER_SERVER_URL` to its `/asr` endpoint. Keep the transcription server private to the Docker network and do not expose it publicly. diff --git a/app/main.py b/app/main.py index 71077dd..932db23 100644 --- a/app/main.py +++ b/app/main.py @@ -1,9 +1,11 @@ import asyncio +import logging import os import re import sqlite3 import subprocess import tempfile +import time import uuid from datetime import datetime, timezone from pathlib import Path @@ -18,6 +20,10 @@ AUDIO_DIR = DATA_DIR / "audio" DB_PATH = DATA_DIR / "faerro-kb.sqlite3" WHISPER_SERVER_URL = os.environ.get("WHISPER_SERVER_URL", "").strip() FRONTEND_DIR = Path(__file__).resolve().parent.parent / "frontend" +WHISPER_TEST_AUDIO_PATH = Path(__file__).resolve().parent.parent / "audio test.wav" +TRANSCRIPTION_TASKS: set[asyncio.Task[None]] = set() +logger = logging.getLogger(__name__) +logging.getLogger("httpx").setLevel(logging.WARNING) DATA_DIR.mkdir(parents=True, exist_ok=True) AUDIO_DIR.mkdir(parents=True, exist_ok=True) @@ -120,6 +126,85 @@ async def transcribe_note(note_id: str, audio_path: Path) -> None: remove_transcribed_audio(note_id) +async def test_whisper_connection() -> bool: + if not WHISPER_SERVER_URL: + logger.warning("Whisper startup test skipped: WHISPER_SERVER_URL is not configured") + return False + if not WHISPER_TEST_AUDIO_PATH.is_file(): + logger.error("Whisper startup test failed: bundled test audio is missing") + return False + + started_at = time.monotonic() + try: + async with httpx.AsyncClient(timeout=httpx.Timeout(120, connect=5)) as client: + with WHISPER_TEST_AUDIO_PATH.open("rb") as audio_file: + response = await client.post( + WHISPER_SERVER_URL, + params={"encode": "true", "task": "transcribe", "output": "json"}, + files={"audio_file": (WHISPER_TEST_AUDIO_PATH.name, audio_file, "audio/wav")}, + ) + response.raise_for_status() + result = response.json() + transcript = result.get("text") if isinstance(result, dict) else None + if not isinstance(transcript, str) or not transcript.strip(): + logger.error( + "Whisper startup test failed: HTTP %s returned no transcript (%.1fs)", + response.status_code, + time.monotonic() - started_at, + ) + return False + language = result.get("language", "unknown") + if not isinstance(language, str) or not re.fullmatch(r"[A-Za-z-]{1,12}", language): + language = "unknown" + segments = result.get("segments", []) + segment_count = len(segments) if isinstance(segments, list) else 0 + logger.info( + "Whisper startup test passed: HTTP %s, language=%s, segments=%d, elapsed=%.1fs", + response.status_code, + language, + segment_count, + time.monotonic() - started_at, + ) + return True + except httpx.HTTPStatusError as error: + logger.error("Whisper startup test failed: HTTP %s", error.response.status_code) + except Exception as error: + logger.error("Whisper startup test failed: %s", type(error).__name__) + return False + + +@app.on_event("startup") +async def resume_pending_transcriptions() -> None: + with connect_db() as connection: + connection.execute( + "UPDATE notes SET status = 'awaiting_transcription' WHERE status = 'transcribing'" + ) + if not await test_whisper_connection(): + return + with connect_db() as connection: + notes = connection.execute( + """SELECT id, audio_path FROM notes + WHERE status IN ('awaiting_transcription', 'transcribing', 'transcription_failed')""" + ).fetchall() + for note in notes: + audio_path = Path(note["audio_path"]) + if audio_path.is_file(): + with connect_db() as connection: + connection.execute( + "UPDATE notes SET status = 'transcribing' WHERE id = ?", + (note["id"],), + ) + task = asyncio.create_task(transcribe_note(note["id"], audio_path)) + TRANSCRIPTION_TASKS.add(task) + task.add_done_callback(TRANSCRIPTION_TASKS.discard) + else: + with connect_db() as connection: + connection.execute( + "UPDATE notes SET status = 'transcription_failed' WHERE id = ?", + (note["id"],), + ) + + @app.get("/api/health") def health() -> dict[str, str]: return {"status": "ok"} diff --git a/audio test.wav b/audio test.wav new file mode 100644 index 0000000..41ec5b7 Binary files /dev/null and b/audio test.wav differ diff --git a/frontend/app.js b/frontend/app.js index d63a38a..c2213f8 100644 --- a/frontend/app.js +++ b/frontend/app.js @@ -19,6 +19,7 @@ let recordingStartedAt = 0; let recordingTimer; let syncing = false; let draftSaveTimer; +let transcriptionPoll; function openDatabase() { if (databasePromise) return databasePromise; @@ -421,6 +422,16 @@ function renderArchive(notes) { elements['archive-empty'].hidden = notes.length > 0; if (!navigator.onLine) elements['archive-empty'].textContent = 'Reconnect to view saved notes.'; else elements['archive-empty'].textContent = 'No saved notes yet.'; + if (notes.some((note) => note.status === 'transcribing')) { + if (!transcriptionPoll) { + transcriptionPoll = window.setInterval(() => { + if (navigator.onLine && document.visibilityState === 'visible') loadAwaitingNotes(); + }, 5000); + } + } else { + clearInterval(transcriptionPoll); + transcriptionPoll = undefined; + } for (const note of notes) { const item = document.createElement('article'); item.className = 'archive-item'; @@ -458,7 +469,20 @@ function renderArchive(notes) { }); const status = document.createElement('div'); status.className = 'archive-status'; - status.textContent = isWrittenNote ? 'WRITTEN NOTE' : note.status === 'transcribing' ? 'TRANSCRIBING' : note.status === 'transcribed' ? 'TRANSCRIPT SAVED' : 'AWAITING TRANSCRIPTION'; + status.textContent = isWrittenNote ? 'WRITTEN NOTE' : note.status === 'transcribing' ? 'TRANSCRIBING' : note.status === 'transcribed' ? 'TRANSCRIPT SAVED' : note.status === 'transcription_failed' ? 'TRANSCRIPTION FAILED; AUDIO RETAINED' : note.status === 'empty_transcript' ? 'NO SPEECH DETECTED; AUDIO RETAINED' : 'AWAITING TRANSCRIPTION'; + if (note.status === 'transcribing') { + const progress = document.createElement('div'); + progress.className = 'transcription-progress'; + progress.setAttribute('role', 'progressbar'); + progress.setAttribute('aria-label', 'Transcription in progress'); + progress.setAttribute('aria-valuemin', '0'); + progress.setAttribute('aria-valuemax', '100'); + progress.setAttribute('aria-valuetext', 'In progress'); + progress.append(document.createElement('span')); + body.append(name, transcript, save, status, progress); + } else { + body.append(name, transcript, save, status); + } const remove = createDeleteButton(`Delete note ${note.filename}`); remove.addEventListener('click', async () => { remove.disabled = true; @@ -471,7 +495,6 @@ function renderArchive(notes) { setMessage('Could not delete this note. Please try again.', true); } }); - body.append(name, transcript, save, status); item.append(date, body, remove); elements['archive-list'].append(item); } @@ -486,6 +509,9 @@ async function loadAwaitingNotes() { const responses = await Promise.all([ fetch('/api/notes?status=awaiting_transcription'), fetch('/api/notes?status=written'), + fetch('/api/notes?status=transcribing'), + fetch('/api/notes?status=transcription_failed'), + fetch('/api/notes?status=empty_transcript'), ]); if (responses.some((response) => !response.ok)) throw new Error('Archive unavailable'); const notes = (await Promise.all(responses.map((response) => response.json()))).flat(); diff --git a/frontend/styles.css b/frontend/styles.css index 3e75bcc..3796afc 100644 --- a/frontend/styles.css +++ b/frontend/styles.css @@ -87,8 +87,11 @@ h2 { margin: 8px 0 0; font: 500 25px/1.1 var(--serif); } .archive-transcript:focus { border-bottom-color: var(--moss); } .save-transcript { margin-top: 7px; border: 0; padding: 0; color: var(--moss-dark); background: transparent; font: 11px var(--mono); cursor: pointer; } .archive-status { color: var(--muted); font: 10px var(--mono); margin-top: 7px; } +.transcription-progress { width: min(240px, 100%); height: 3px; margin-top: 9px; overflow: hidden; background: #e4e3dc; } +.transcription-progress span { display: block; width: 34%; height: 100%; background: var(--moss); animation: transcription-progress 1.4s ease-in-out infinite; } footer { max-width: 1040px; margin: auto; padding: 18px 24px calc(18px + env(safe-area-inset-bottom)); display: flex; justify-content: space-between; gap: 20px; color: var(--muted); font: 9px var(--mono); } @keyframes reveal { from { opacity: 0; transform: translateY(5px); } to { opacity: 1; transform: translateY(0); } } +@keyframes transcription-progress { from { transform: translateX(-110%); } to { transform: translateX(300%); } } @media (max-width: 600px) { .topbar { height: 58px; padding: 0 18px; } main { padding: 0 18px; }