From 0fb66f2ceedaec0811dac4901a4a6686afeaa85a Mon Sep 17 00:00:00 2001 From: Andrea Date: Mon, 28 Sep 2026 14:05:39 +0200 Subject: [PATCH] Update README and API integration: specify local ASR server details and adjust audio upload parameters --- README.md | 4 ++-- app/main.py | 4 ++-- frontend/styles.css | 2 +- 3 files changed, 5 insertions(+), 5 deletions(-) diff --git a/README.md b/README.md index e56e942..9e3e7cb 100644 --- a/README.md +++ b/README.md @@ -38,9 +38,9 @@ In Safari on iPhone, open the HTTPS address and use **Share > Add to Home Screen ## Local speech-to-text -The API can use a local `whisper.cpp` server when `WHISPER_SERVER_URL` is configured. Audio is sent only to that local endpoint; with the variable unset, uploads remain stored as `awaiting_transcription` until a non-empty transcript is entered and saved. Saving that transcript deletes both server-side audio files. +The API can use a local `onerahmet/openai-whisper-asr-webservice` server when `WHISPER_SERVER_URL` is configured. Set it to the service's `/asr` endpoint, for example `http://whisper:9000/asr`. Faerro KB sends the converted WAV as the `audio_file` multipart field with `task=transcribe` and `output=json`; audio is sent only to that local endpoint. With the variable unset, uploads remain stored as `awaiting_transcription` until a non-empty transcript is entered and saved. Saving that transcript deletes both server-side audio files. -For NVIDIA GPU transcription, run a CUDA-enabled `whisper.cpp` server with a model on the same machine, then set `WHISPER_SERVER_URL` to its local inference endpoint, for example `http://whisper:8080/inference`. Use a model that fits the RTX 3070's VRAM; start with `small` or a quantized `medium` model and measure with your audio/language. Do not expose the transcription server outside the private Docker network. See the [whisper.cpp NVIDIA instructions](https://github.com/ggml-org/whisper.cpp#nvidia-gpu-support) for CUDA builds. +Run the ASR web service with a model that fits the available hardware, then set `WHISPER_SERVER_URL` to its `/asr` endpoint. Keep the transcription server private to the Docker network and do not expose it publicly. The transcript remains in SQLite after the audio files are deleted. A future local correction model can propose punctuation and name fixes without replacing the stored transcription. Proper-name matching and offline geolocation data are not enabled yet. diff --git a/app/main.py b/app/main.py index 696c16c..71077dd 100644 --- a/app/main.py +++ b/app/main.py @@ -107,8 +107,8 @@ async def transcribe_note(note_id: str, audio_path: Path) -> None: with audio_path.open("rb") as audio_file: response = await client.post( WHISPER_SERVER_URL, - files={"file": (audio_path.name, audio_file, "application/octet-stream")}, - data={"response_format": "json"}, + params={"task": "transcribe", "output": "json"}, + files={"audio_file": (audio_path.name, audio_file, "application/octet-stream")}, ) response.raise_for_status() transcript = response.json().get("text", "").strip() diff --git a/frontend/styles.css b/frontend/styles.css index c88ab33..3e75bcc 100644 --- a/frontend/styles.css +++ b/frontend/styles.css @@ -57,7 +57,7 @@ h1 em { color: var(--moss); font-weight: 500; } .capture-message { min-height: 20px; max-width: 590px; margin: 8px auto 0; padding-top: 6px; color: var(--muted); font-size: 11px; text-align: center; } .capture-message.upload-confirmed { display: flex; align-items: center; justify-content: center; gap: 8px; width: 100%; color: forestgreen; font-size: 16px; } .capture-message.upload-confirmed::after { content: '✓'; color: forestgreen; font-size: 1.05em; line-height: 1; } -.audio-upload-row { max-width: 590px; margin: 14px auto 6px; display: flex; align-items: center; justify-content: center; gap: 12px; } +.audio-upload-row { max-width: 590px; margin: 10px auto 10 px; display: flex; align-items: center; justify-content: center; gap: 12px; } .audio-upload-button { display: inline-flex; min-height: 36px; align-items: center; gap: 7px; border: 1px solid var(--line); padding: 0 12px; color: var(--moss-dark); background: #fffefa; font-size: 12px; cursor: pointer; } .audio-upload-button:hover { border-color: var(--moss); } #audio-file-input { position: absolute; width: 1px; height: 1px; overflow: hidden; clip: rect(0, 0, 0, 0); white-space: nowrap; clip-path: inset(50%); }