From 582d507c36579f9274b5f9bf16f9605c2724c60d Mon Sep 17 00:00:00 2001 From: Shaurya Saria Date: Wed, 19 Aug 2026 18:48:43 +0530 Subject: [PATCH 1/7] docs: add canonical README filename --- README.md | 316 ++++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 316 insertions(+) create mode 100644 README.md diff --git a/README.md b/README.md new file mode 100644 index 0000000..0ec330c --- /dev/null +++ b/README.md @@ -0,0 +1,316 @@ +
+ +[![CI](https://github.com/icecold009/audio-recognition/actions/workflows/ci.yml/badge.svg)](https://github.com/icecold009/audio-recognition/actions/workflows/ci.yml) +
+ +**Identify any song from your microphone or an audio file.** +Validated audio pipeline · Multi-backend matching · Flask web UI · Terminal output + + +
+ +*** +
+

Audio Recognition

+

Identify songs from your microphone or an audio file with validated multi-backend matching

+
+ +## Overview +DIY Shazam captures audio from the CLI microphone/file path or the Flask browser UI, normalizes it through one bounded audio pipeline, and identifies tracks using RapidAPI/Shazam, AcoustID, AudD, or local spectrogram peaks and constellation hash pairs. FFT output is a diagnostic visualization only; it is not the recognition algorithm. Flask serves the complete browser UI and JSON API from one origin. + +## Performance + +| Metric | Result | +|--------|--------| +| Average recognition time (RapidAPI backend) | See the imported benchmark report below | +| Average recognition time (AudD backend) | See the imported benchmark report below | +| Test set accuracy | See the imported benchmark report below | +| Minimum audio duration | 1 s by default (configurable) | +| Maximum audio duration | 30 s by default (configurable) | +| Maximum upload size | 10 MiB by default (configurable) | +| Platforms tested | Not established by this branch's validation | + + +No complete real-world benchmark has been imported. Run the documented evaluation only after assembling a legally reusable corpus and supplying the operator metadata and provider configuration. + + +## Architecture +```mermaid +flowchart LR + A[CLI / Flask Browser UI] --> B[Audio Input\n(mic or upload)] + B --> C[Validate and normalize\nmono float32 / internal rate] + C --> D[FFT diagnostic only] + C --> E[Write 16-bit PCM WAV temp] + E --> F{Matcher Backends} + F -->|RapidAPI| G[Shazam] + F -->|AcoustID| H[AcoustID] + F -->|AudD| I[AudD] + F -->|Local hashes| J[Peak/hash index] + G & H & I & J --> K[Normalized Result] + K --> L[Display (CLI) / JSON (Web)] + classDef blue fill:#ffffff,stroke:#1E90FF,stroke-width:2px,color:#1E90FF; + class A,B,C,D,E,F,G,H,I,J,K,L blue; +``` +Theme: black / white / blue — white nodes with a professional DodgerBlue accent (#1E90FF). The browser UI is served directly by Flask; there is no separate browser bundle. + +## Quickstart +### Windows PowerShell + +1) Create and activate a venv: +```powershell +python -m venv .venv +.\.venv\Scripts\Activate.ps1 +``` +2) Install runtime dependencies and copy configuration: +```powershell +python -m pip install -r requirements.txt +Copy-Item .env.example .env +``` +3) Add provider values to `.env` if recognition is needed. Verify host tools when using +non-WAV uploads or AcoustID: +```powershell +ffmpeg -version +fpcalc -version +``` +4) Run the Flask development server: +```powershell +python web/app.py +# open http://127.0.0.1:5000 +``` +5) Run CLI when terminal recognition is needed: +```powershell +python main.py +``` + +### macOS/Linux + +```bash +python3 -m venv .venv +. .venv/bin/activate +python -m pip install -r requirements.txt +cp .env.example .env +python web/app.py +``` + +Install host tools outside Docker when needed: macOS uses `brew install ffmpeg chromaprint`; +Debian/Ubuntu uses `sudo apt-get install ffmpeg libchromaprint-tools`. These commands are +only for local host setup; the production image installs the same runtime tools itself. + +Run the CLI with `python main.py`. On macOS/Linux, a production-style local WSGI process is: + +```bash +APP_ENV=production PORT=8000 gunicorn --config gunicorn.conf.py web.app:app +``` + +Production mode requires the server-only Supabase quota configuration and at least one +recognition backend. It fails closed with HTTP 503 when those requirements are unavailable. + +### Docker + +The reproducible production image installs Python, Gunicorn, FFmpeg, Chromaprint/fpcalc, +audio libraries, and curl for health checks. It runs as UID/GID `10001`, uses Gunicorn, and +reads the platform-provided `PORT`. + +```powershell +Copy-Item .env.example .env +docker compose up --build +``` + +The compose service uses Gunicorn with a read-only root filesystem and a bounded 64 MiB +`/tmp` tmpfs. Its default `APP_ENV=development` keeps the local quickstart usable without +Supabase; configure `.env` and set `APP_ENV=production` when testing the fail-closed +production path. A direct production start is: + +```powershell +docker build -t audio-recognition . +docker run --rm -p 5000:5000 --env-file .env audio-recognition +``` + +The same commands work from macOS/Linux after replacing `Copy-Item` with `cp`. + +The canonical Flask development command remains: + +```powershell +python web/app.py +# open http://127.0.0.1:5000 +``` + +## Project Structure +Core source modules now live under `shazam_project/`: + +- `shazam_project/config.py` +- `shazam_project/recorder.py` +- `shazam_project/fft_analyze.py` +- `shazam_project/matcher.py` +- `shazam_project/display.py` + +Entrypoints remain: + +- `main.py` (CLI) +- `web/app.py` (canonical Flask browser app and API) +- `web/templates/index.html` and `web/static/` (same-origin browser assets) + +## Configuration +Supported env vars (see `shazam_project.config.load_config()`): `AUDD_API_TOKEN`, `ACOUSTID_API_KEY`, `FP_CALC_PATH`, `RAPIDAPI_KEY`, and optional `LOCAL_FINGERPRINT_INDEX` (with `FINGERPRINT_INDEX_PATH` accepted as a legacy alias). The shared audio contract is controlled by `INTERNAL_SAMPLE_RATE`, `MIN_AUDIO_SECONDS`, `MAX_AUDIO_SECONDS`, `MAX_UPLOAD_BYTES`, and `FFMPEG_TIMEOUT_SECONDS`; provider WAVs are always fixed 16-bit PCM. Matcher order is RapidAPI → AcoustID → AudD → local fingerprint index. + +## Web UI +`python web/app.py` serves `/`, `/static/*`, `/api/match`, and `/api/status` from the same origin. CLI file mode accepts WAV/PCM files. Web uploads support WAV, MP3, M4A, AAC, OGG, FLAC, and WEBM; non-WAV web uploads require FFmpeg on `PATH` and are converted before decoding. The browser also supports microphone recording, manual stop, waveform visualization, loading/error/no-match states, light/dark theme persistence, and session-only recognition history. + +Every input is downmixed to mono float32 samples in `[-1, 1]` and resampled to 44,100 Hz by default. Provider adapters receive temporary mono 16-bit PCM WAV files. Inputs shorter than 1 second, longer than 30 seconds, or larger than 10 MiB are rejected by default; all three limits are configurable. + +The deployment endpoints are: + +- `/healthz` is a dependency-free process liveness check and returns HTTP 200 when Flask is serving. +- `/readyz` checks production configuration, writable temporary storage, FFmpeg, fpcalc when + AcoustID is enabled, Supabase quota availability, and at least one recognition backend. It + returns HTTP 503 with stable check names when not ready; it never returns secrets, paths, + exception text, or database details. +- `/api/status` reports non-secret backend/tool flags, quota mode, limits, and audio settings. + +Gunicorn defaults to two workers, a 45-second request timeout, a 15-second graceful shutdown +window, and a five-second keep-alive. Provider requests and FFmpeg conversion remain bounded +by their existing 15-second timeouts. `MAX_UPLOAD_BYTES`, duration limits, Flask's request +limit, and temporary-file cleanup bound upload and conversion resource use. + +## Testing +Run the Python tests and coverage locally: +```powershell +python -m pytest -q +python -m coverage run --branch --source=shazam_project,web,scripts -m pytest -q +python -m coverage report --fail-under=70 --show-missing +ruff format --check . +ruff check . +pip-audit -r requirements.txt -r requirements-dev.txt --progress-spinner off +``` +`tests/` covers configuration loading and backend combinations, mocked microphone failures and cleanup, normalized audio, mocked provider flows and fallback, Flask routes, safe incomplete metadata rendering contracts, rate limits, FFmpeg failures, Supabase failures, and a generated WAV end-to-end test. The generated WAV is created in memory and contains no third-party recording. + +GitHub Actions runs pytest on every push and pull request across Python 3.10, 3.11, and 3.12, Ruff formatting and lint, a Python 3.12 branch-coverage gate at 70%, `pip-audit`, and Gitleaks secret scanning. The coverage job publishes `coverage.xml` as an artifact. Provider network calls, Supabase credentials, `fpcalc`, FFmpeg, and microphone hardware are mocked or tested through stable failure paths in CI; they are excluded from the release gate because they require external credentials or host devices. + +Direct Python dependencies use reviewed major-compatible ranges in `requirements.txt` and `requirements-dev.txt`. To update one, review its release notes and Python 3.10–3.12 compatibility, edit its range, install from both requirement files, then run the full pytest, coverage, Ruff, compile, diff, and `pip-audit` checks. Do not add credentials or resolve updates from a developer's private environment. + +For the real-world comparison, see [`evaluation/README.md`](evaluation/README.md). It validates a source catalog, records resumable speaker-to-microphone clips at 4, 8, and 15 seconds, caches deterministic backend results without credentials, builds the local landmark-hash index from clean source tracks, and compares the local backend against all three provider backends. + +## Limitations + +Recognition is not guaranteed outside the happy path. The main failure modes are: + +- **Background noise and recording quality:** speech, room echo, speaker distortion, very low volume, clipping, or music mixed with other sounds can hide the spectral peaks used by fingerprinting. +- **Catalog coverage:** a provider can only return tracks in its database, while the local matcher can only identify tracks present in its local fingerprint index. A `no_match` result does not prove that the audio is invalid. +- **Language and regional catalog differences:** the fingerprinting itself is not English-specific, but provider metadata and catalog coverage vary by language, region, release, and recording availability. +- **Live, cover, remix, and alternate versions:** crowd noise, changed instrumentation, tempo or pitch, medleys, and different arrangements may fail to match or may be returned as the closest studio recording rather than the exact performance. + +The evaluation dataset is designed to measure these cases separately. Until that dataset is recorded and run through all configured backends, the README does not claim a general accuracy percentage. + +## Production rate limits + +Production quota enforcement uses the exposed-but-restricted `public.check_api_quota` preflight and `public.consume_api_quota` RPCs created by [`supabase/migrations/20260801145213_production_rate_limits.sql`](supabase/migrations/20260801145213_production_rate_limits.sql). The preflight is read-only and runs before upload saving; the final operation locks one HMAC-keyed usage row and checks cooldown, daily, and monthly limits before incrementing both counters atomically after valid audio decoding. Both functions are `SECURITY INVOKER`, use an explicit safe search path, and are executable only by `service_role`. The `public.api_usage` table has RLS enabled, no public policies, and no grants to `anon` or `authenticated`; the private schema is not exposed. + +Required server-only configuration: + +- `SUPABASE_URL` +- `SUPABASE_SERVICE_ROLE_KEY` — never put this value in JavaScript, HTML, API responses, logs, or screenshots. +- `CLIENT_ID_HMAC_SECRET` — a separate secret used to derive the stored client identifier; raw IP addresses are never stored. + +The local development quickstart uses `APP_ENV=development`, so local Flask matching works without Supabase and uses a bounded, expiring in-memory limiter. To exercise production behavior, set `APP_ENV=production` and provide all three server-only values above; missing configuration or a quota-service failure returns HTTP 503 rather than assuming zero usage. `APP_ENV=production` should be configured separately in the deployment environment, never copied blindly into a local `.env`. + +The migration workflow is: + +```powershell +supabase start +supabase db reset +supabase db advisors --local --type all --fail-on warn +supabase migration list --local +``` + +For a disposable linked development project, verify with `supabase link --project-ref `, `supabase db push --dry-run`, `supabase db push`, `supabase db advisors --linked --type all --fail-on warn`, and `supabase migration list --linked`. Never run `supabase db reset --linked` against production. The migration was created with `supabase migration new production_rate_limits`. + +`/api/status` reports the quota mode, configured daily/monthly limits, cooldown, and whether production-grade quotas are enabled; it never reports client hashes or usage rows. + +`INTERNAL_API_SECRET`, when configured, prevents the current browser UI from calling `/api/match` unless a deliberate server-side authentication design supplies `X-API-Secret`; the secret is never placed in JavaScript. An allowed `Origin` or `Referer` can never authenticate a request. Forwarded client addresses are ignored unless both `TRUSTED_PROXY_COUNT` and an allowlisted `TRUSTED_PROXY_IPS` chain are configured; every trusted proxy hop is validated when more than one hop is configured. Flask debug mode is enabled only when `APP_ENV=development`. + +## Notes & Tips +- Record in a quiet space and keep the mic near the audio source. +- For AcoustID, install Chromaprint (`fpcalc`): macOS `brew install chromaprint`, Debian/Ubuntu `apt install libchromaprint-tools`. +- FFmpeg is required for non-WAV web uploads and is not required for WAV uploads. + +## Contributing +Pull requests are welcome. For major changes, open an issue first. +Run `python -m pytest -q` and `python -m coverage report` before submitting. + +*** + +## Example Output + +``` +Listen via microphone or load a file? (mic/file): mic + +Recording for 8 seconds... +Recognition uses the normalized recording and configured matcher backends. + +Song: Blinding Lights +Artist: The Weeknd + +[Album art opens in image viewer] +``` + +FFT spectrum for a sample clip: + +![FFT spectrum output](docs/screenshots/fft-output.png) + +This image is diagnostic output from `shazam_project.fft_analyze.analyze_audio`; the canonical browser page is served by `python web/app.py` at `/`. + +*** + +## Web API Reference + +| Endpoint | Method | Description | +|----------------|--------|-------------------------------------------------------| +| `/api/match` | POST | Upload an audio file for recognition. Returns JSON. | +| `/api/status` | GET | Reports configured backends, ffmpeg, fpcalc status. | +| `/healthz` | GET | Dependency-free process liveness check. | +| `/readyz` | GET | Configuration and dependency readiness check. | + +**Example — cURL:** +```bash +curl -X POST http://localhost:5000/api/match \ + -F "file=@song.wav" +``` + +**Example — Response:** +```json +{ + "status": "matched", + "title": "Blinding Lights", + "artist": "The Weeknd", + "album": "After Hours", + "image": "https://..." +} +``` + +Public `status` is one of: `matched` · `no_match` · `not_configured` · `invalid_audio` · `rate_limited` · `error`. Provider attempts contain only backend/status/error codes and generic safe messages; raw provider payloads, credentials, local paths, and stack traces are not public response data. + +*** + +## Notes + +- Record in a **quiet environment** for best accuracy +- CLI mic mode requires a working input device; the web UI uses browser microphone +- File mode (CLI) accepts WAV/PCM only; the web UI supports the documented formats and requires FFmpeg for non-WAV uploads. +- Windows, macOS, and Linux support has not been independently verified by this branch's evidence. + +*** + +## Roadmap + +- [x] Add CI — run pytest and coverage on every push and pull request +- [ ] CLI flags: `--mode`, `--duration`, `--file` for unattended/scripted use +- [ ] Local match history saved as JSON +- [ ] `--no-open-image` flag for headless environments +- [ ] Structured logging in `shazam_project/matcher.py` for easier debugging +- [x] Flask integration tests for the browser entry point, static assets, API status, and upload outcomes + +*** + +## License + +[MIT](LICENSE) — free to use and modify. From 7cca1b59babb29ac8c31479ce5055cb31f59caa5 Mon Sep 17 00:00:00 2001 From: Shaurya Saria Date: Wed, 19 Aug 2026 18:49:11 +0530 Subject: [PATCH 2/7] docs: remove lowercase README filename --- readme.md | 316 ------------------------------------------------------ 1 file changed, 316 deletions(-) delete mode 100644 readme.md diff --git a/readme.md b/readme.md deleted file mode 100644 index 0ec330c..0000000 --- a/readme.md +++ /dev/null @@ -1,316 +0,0 @@ -
- -[![CI](https://github.com/icecold009/audio-recognition/actions/workflows/ci.yml/badge.svg)](https://github.com/icecold009/audio-recognition/actions/workflows/ci.yml) -
- -**Identify any song from your microphone or an audio file.** -Validated audio pipeline · Multi-backend matching · Flask web UI · Terminal output - - -
- -*** -
-

Audio Recognition

-

Identify songs from your microphone or an audio file with validated multi-backend matching

-
- -## Overview -DIY Shazam captures audio from the CLI microphone/file path or the Flask browser UI, normalizes it through one bounded audio pipeline, and identifies tracks using RapidAPI/Shazam, AcoustID, AudD, or local spectrogram peaks and constellation hash pairs. FFT output is a diagnostic visualization only; it is not the recognition algorithm. Flask serves the complete browser UI and JSON API from one origin. - -## Performance - -| Metric | Result | -|--------|--------| -| Average recognition time (RapidAPI backend) | See the imported benchmark report below | -| Average recognition time (AudD backend) | See the imported benchmark report below | -| Test set accuracy | See the imported benchmark report below | -| Minimum audio duration | 1 s by default (configurable) | -| Maximum audio duration | 30 s by default (configurable) | -| Maximum upload size | 10 MiB by default (configurable) | -| Platforms tested | Not established by this branch's validation | - - -No complete real-world benchmark has been imported. Run the documented evaluation only after assembling a legally reusable corpus and supplying the operator metadata and provider configuration. - - -## Architecture -```mermaid -flowchart LR - A[CLI / Flask Browser UI] --> B[Audio Input\n(mic or upload)] - B --> C[Validate and normalize\nmono float32 / internal rate] - C --> D[FFT diagnostic only] - C --> E[Write 16-bit PCM WAV temp] - E --> F{Matcher Backends} - F -->|RapidAPI| G[Shazam] - F -->|AcoustID| H[AcoustID] - F -->|AudD| I[AudD] - F -->|Local hashes| J[Peak/hash index] - G & H & I & J --> K[Normalized Result] - K --> L[Display (CLI) / JSON (Web)] - classDef blue fill:#ffffff,stroke:#1E90FF,stroke-width:2px,color:#1E90FF; - class A,B,C,D,E,F,G,H,I,J,K,L blue; -``` -Theme: black / white / blue — white nodes with a professional DodgerBlue accent (#1E90FF). The browser UI is served directly by Flask; there is no separate browser bundle. - -## Quickstart -### Windows PowerShell - -1) Create and activate a venv: -```powershell -python -m venv .venv -.\.venv\Scripts\Activate.ps1 -``` -2) Install runtime dependencies and copy configuration: -```powershell -python -m pip install -r requirements.txt -Copy-Item .env.example .env -``` -3) Add provider values to `.env` if recognition is needed. Verify host tools when using -non-WAV uploads or AcoustID: -```powershell -ffmpeg -version -fpcalc -version -``` -4) Run the Flask development server: -```powershell -python web/app.py -# open http://127.0.0.1:5000 -``` -5) Run CLI when terminal recognition is needed: -```powershell -python main.py -``` - -### macOS/Linux - -```bash -python3 -m venv .venv -. .venv/bin/activate -python -m pip install -r requirements.txt -cp .env.example .env -python web/app.py -``` - -Install host tools outside Docker when needed: macOS uses `brew install ffmpeg chromaprint`; -Debian/Ubuntu uses `sudo apt-get install ffmpeg libchromaprint-tools`. These commands are -only for local host setup; the production image installs the same runtime tools itself. - -Run the CLI with `python main.py`. On macOS/Linux, a production-style local WSGI process is: - -```bash -APP_ENV=production PORT=8000 gunicorn --config gunicorn.conf.py web.app:app -``` - -Production mode requires the server-only Supabase quota configuration and at least one -recognition backend. It fails closed with HTTP 503 when those requirements are unavailable. - -### Docker - -The reproducible production image installs Python, Gunicorn, FFmpeg, Chromaprint/fpcalc, -audio libraries, and curl for health checks. It runs as UID/GID `10001`, uses Gunicorn, and -reads the platform-provided `PORT`. - -```powershell -Copy-Item .env.example .env -docker compose up --build -``` - -The compose service uses Gunicorn with a read-only root filesystem and a bounded 64 MiB -`/tmp` tmpfs. Its default `APP_ENV=development` keeps the local quickstart usable without -Supabase; configure `.env` and set `APP_ENV=production` when testing the fail-closed -production path. A direct production start is: - -```powershell -docker build -t audio-recognition . -docker run --rm -p 5000:5000 --env-file .env audio-recognition -``` - -The same commands work from macOS/Linux after replacing `Copy-Item` with `cp`. - -The canonical Flask development command remains: - -```powershell -python web/app.py -# open http://127.0.0.1:5000 -``` - -## Project Structure -Core source modules now live under `shazam_project/`: - -- `shazam_project/config.py` -- `shazam_project/recorder.py` -- `shazam_project/fft_analyze.py` -- `shazam_project/matcher.py` -- `shazam_project/display.py` - -Entrypoints remain: - -- `main.py` (CLI) -- `web/app.py` (canonical Flask browser app and API) -- `web/templates/index.html` and `web/static/` (same-origin browser assets) - -## Configuration -Supported env vars (see `shazam_project.config.load_config()`): `AUDD_API_TOKEN`, `ACOUSTID_API_KEY`, `FP_CALC_PATH`, `RAPIDAPI_KEY`, and optional `LOCAL_FINGERPRINT_INDEX` (with `FINGERPRINT_INDEX_PATH` accepted as a legacy alias). The shared audio contract is controlled by `INTERNAL_SAMPLE_RATE`, `MIN_AUDIO_SECONDS`, `MAX_AUDIO_SECONDS`, `MAX_UPLOAD_BYTES`, and `FFMPEG_TIMEOUT_SECONDS`; provider WAVs are always fixed 16-bit PCM. Matcher order is RapidAPI → AcoustID → AudD → local fingerprint index. - -## Web UI -`python web/app.py` serves `/`, `/static/*`, `/api/match`, and `/api/status` from the same origin. CLI file mode accepts WAV/PCM files. Web uploads support WAV, MP3, M4A, AAC, OGG, FLAC, and WEBM; non-WAV web uploads require FFmpeg on `PATH` and are converted before decoding. The browser also supports microphone recording, manual stop, waveform visualization, loading/error/no-match states, light/dark theme persistence, and session-only recognition history. - -Every input is downmixed to mono float32 samples in `[-1, 1]` and resampled to 44,100 Hz by default. Provider adapters receive temporary mono 16-bit PCM WAV files. Inputs shorter than 1 second, longer than 30 seconds, or larger than 10 MiB are rejected by default; all three limits are configurable. - -The deployment endpoints are: - -- `/healthz` is a dependency-free process liveness check and returns HTTP 200 when Flask is serving. -- `/readyz` checks production configuration, writable temporary storage, FFmpeg, fpcalc when - AcoustID is enabled, Supabase quota availability, and at least one recognition backend. It - returns HTTP 503 with stable check names when not ready; it never returns secrets, paths, - exception text, or database details. -- `/api/status` reports non-secret backend/tool flags, quota mode, limits, and audio settings. - -Gunicorn defaults to two workers, a 45-second request timeout, a 15-second graceful shutdown -window, and a five-second keep-alive. Provider requests and FFmpeg conversion remain bounded -by their existing 15-second timeouts. `MAX_UPLOAD_BYTES`, duration limits, Flask's request -limit, and temporary-file cleanup bound upload and conversion resource use. - -## Testing -Run the Python tests and coverage locally: -```powershell -python -m pytest -q -python -m coverage run --branch --source=shazam_project,web,scripts -m pytest -q -python -m coverage report --fail-under=70 --show-missing -ruff format --check . -ruff check . -pip-audit -r requirements.txt -r requirements-dev.txt --progress-spinner off -``` -`tests/` covers configuration loading and backend combinations, mocked microphone failures and cleanup, normalized audio, mocked provider flows and fallback, Flask routes, safe incomplete metadata rendering contracts, rate limits, FFmpeg failures, Supabase failures, and a generated WAV end-to-end test. The generated WAV is created in memory and contains no third-party recording. - -GitHub Actions runs pytest on every push and pull request across Python 3.10, 3.11, and 3.12, Ruff formatting and lint, a Python 3.12 branch-coverage gate at 70%, `pip-audit`, and Gitleaks secret scanning. The coverage job publishes `coverage.xml` as an artifact. Provider network calls, Supabase credentials, `fpcalc`, FFmpeg, and microphone hardware are mocked or tested through stable failure paths in CI; they are excluded from the release gate because they require external credentials or host devices. - -Direct Python dependencies use reviewed major-compatible ranges in `requirements.txt` and `requirements-dev.txt`. To update one, review its release notes and Python 3.10–3.12 compatibility, edit its range, install from both requirement files, then run the full pytest, coverage, Ruff, compile, diff, and `pip-audit` checks. Do not add credentials or resolve updates from a developer's private environment. - -For the real-world comparison, see [`evaluation/README.md`](evaluation/README.md). It validates a source catalog, records resumable speaker-to-microphone clips at 4, 8, and 15 seconds, caches deterministic backend results without credentials, builds the local landmark-hash index from clean source tracks, and compares the local backend against all three provider backends. - -## Limitations - -Recognition is not guaranteed outside the happy path. The main failure modes are: - -- **Background noise and recording quality:** speech, room echo, speaker distortion, very low volume, clipping, or music mixed with other sounds can hide the spectral peaks used by fingerprinting. -- **Catalog coverage:** a provider can only return tracks in its database, while the local matcher can only identify tracks present in its local fingerprint index. A `no_match` result does not prove that the audio is invalid. -- **Language and regional catalog differences:** the fingerprinting itself is not English-specific, but provider metadata and catalog coverage vary by language, region, release, and recording availability. -- **Live, cover, remix, and alternate versions:** crowd noise, changed instrumentation, tempo or pitch, medleys, and different arrangements may fail to match or may be returned as the closest studio recording rather than the exact performance. - -The evaluation dataset is designed to measure these cases separately. Until that dataset is recorded and run through all configured backends, the README does not claim a general accuracy percentage. - -## Production rate limits - -Production quota enforcement uses the exposed-but-restricted `public.check_api_quota` preflight and `public.consume_api_quota` RPCs created by [`supabase/migrations/20260801145213_production_rate_limits.sql`](supabase/migrations/20260801145213_production_rate_limits.sql). The preflight is read-only and runs before upload saving; the final operation locks one HMAC-keyed usage row and checks cooldown, daily, and monthly limits before incrementing both counters atomically after valid audio decoding. Both functions are `SECURITY INVOKER`, use an explicit safe search path, and are executable only by `service_role`. The `public.api_usage` table has RLS enabled, no public policies, and no grants to `anon` or `authenticated`; the private schema is not exposed. - -Required server-only configuration: - -- `SUPABASE_URL` -- `SUPABASE_SERVICE_ROLE_KEY` — never put this value in JavaScript, HTML, API responses, logs, or screenshots. -- `CLIENT_ID_HMAC_SECRET` — a separate secret used to derive the stored client identifier; raw IP addresses are never stored. - -The local development quickstart uses `APP_ENV=development`, so local Flask matching works without Supabase and uses a bounded, expiring in-memory limiter. To exercise production behavior, set `APP_ENV=production` and provide all three server-only values above; missing configuration or a quota-service failure returns HTTP 503 rather than assuming zero usage. `APP_ENV=production` should be configured separately in the deployment environment, never copied blindly into a local `.env`. - -The migration workflow is: - -```powershell -supabase start -supabase db reset -supabase db advisors --local --type all --fail-on warn -supabase migration list --local -``` - -For a disposable linked development project, verify with `supabase link --project-ref `, `supabase db push --dry-run`, `supabase db push`, `supabase db advisors --linked --type all --fail-on warn`, and `supabase migration list --linked`. Never run `supabase db reset --linked` against production. The migration was created with `supabase migration new production_rate_limits`. - -`/api/status` reports the quota mode, configured daily/monthly limits, cooldown, and whether production-grade quotas are enabled; it never reports client hashes or usage rows. - -`INTERNAL_API_SECRET`, when configured, prevents the current browser UI from calling `/api/match` unless a deliberate server-side authentication design supplies `X-API-Secret`; the secret is never placed in JavaScript. An allowed `Origin` or `Referer` can never authenticate a request. Forwarded client addresses are ignored unless both `TRUSTED_PROXY_COUNT` and an allowlisted `TRUSTED_PROXY_IPS` chain are configured; every trusted proxy hop is validated when more than one hop is configured. Flask debug mode is enabled only when `APP_ENV=development`. - -## Notes & Tips -- Record in a quiet space and keep the mic near the audio source. -- For AcoustID, install Chromaprint (`fpcalc`): macOS `brew install chromaprint`, Debian/Ubuntu `apt install libchromaprint-tools`. -- FFmpeg is required for non-WAV web uploads and is not required for WAV uploads. - -## Contributing -Pull requests are welcome. For major changes, open an issue first. -Run `python -m pytest -q` and `python -m coverage report` before submitting. - -*** - -## Example Output - -``` -Listen via microphone or load a file? (mic/file): mic - -Recording for 8 seconds... -Recognition uses the normalized recording and configured matcher backends. - -Song: Blinding Lights -Artist: The Weeknd - -[Album art opens in image viewer] -``` - -FFT spectrum for a sample clip: - -![FFT spectrum output](docs/screenshots/fft-output.png) - -This image is diagnostic output from `shazam_project.fft_analyze.analyze_audio`; the canonical browser page is served by `python web/app.py` at `/`. - -*** - -## Web API Reference - -| Endpoint | Method | Description | -|----------------|--------|-------------------------------------------------------| -| `/api/match` | POST | Upload an audio file for recognition. Returns JSON. | -| `/api/status` | GET | Reports configured backends, ffmpeg, fpcalc status. | -| `/healthz` | GET | Dependency-free process liveness check. | -| `/readyz` | GET | Configuration and dependency readiness check. | - -**Example — cURL:** -```bash -curl -X POST http://localhost:5000/api/match \ - -F "file=@song.wav" -``` - -**Example — Response:** -```json -{ - "status": "matched", - "title": "Blinding Lights", - "artist": "The Weeknd", - "album": "After Hours", - "image": "https://..." -} -``` - -Public `status` is one of: `matched` · `no_match` · `not_configured` · `invalid_audio` · `rate_limited` · `error`. Provider attempts contain only backend/status/error codes and generic safe messages; raw provider payloads, credentials, local paths, and stack traces are not public response data. - -*** - -## Notes - -- Record in a **quiet environment** for best accuracy -- CLI mic mode requires a working input device; the web UI uses browser microphone -- File mode (CLI) accepts WAV/PCM only; the web UI supports the documented formats and requires FFmpeg for non-WAV uploads. -- Windows, macOS, and Linux support has not been independently verified by this branch's evidence. - -*** - -## Roadmap - -- [x] Add CI — run pytest and coverage on every push and pull request -- [ ] CLI flags: `--mode`, `--duration`, `--file` for unattended/scripted use -- [ ] Local match history saved as JSON -- [ ] `--no-open-image` flag for headless environments -- [ ] Structured logging in `shazam_project/matcher.py` for easier debugging -- [x] Flask integration tests for the browser entry point, static assets, API status, and upload outcomes - -*** - -## License - -[MIT](LICENSE) — free to use and modify. From a1195b1e930ccf7eba1917953ca2e5b8806e5b71 Mon Sep 17 00:00:00 2001 From: Shaurya Saria Date: Wed, 19 Aug 2026 18:50:03 +0530 Subject: [PATCH 3/7] docs: normalize README references --- Dockerfile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Dockerfile b/Dockerfile index e076ee8..0d854c7 100644 --- a/Dockerfile +++ b/Dockerfile @@ -26,7 +26,7 @@ COPY requirements.txt ./ RUN python -m pip install --upgrade pip \ && python -m pip install -r requirements.txt -COPY gunicorn.conf.py main.py readme.md ./ +COPY gunicorn.conf.py main.py README.md ./ COPY shazam_project ./shazam_project COPY web ./web COPY scripts ./scripts From fa6a0bc877c0d8078faac6bf2e5201a6dd925d7f Mon Sep 17 00:00:00 2001 From: Shaurya Saria Date: Wed, 19 Aug 2026 18:50:08 +0530 Subject: [PATCH 4/7] docs: normalize README references --- evaluation/README.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/evaluation/README.md b/evaluation/README.md index f020630..f49ebbd 100644 --- a/evaluation/README.md +++ b/evaluation/README.md @@ -130,7 +130,7 @@ PowerShell: ```powershell .\.venv\Scripts\python.exe scripts/update_readme.py ` --results evaluation/results/benchmark.json ` - --readme readme.md + --readme README.md ``` Bash: @@ -138,7 +138,7 @@ Bash: ```bash .venv/bin/python scripts/update_readme.py \ --results evaluation/results/benchmark.json \ - --readme readme.md + --readme README.md ``` Review the diff, JSON, and Markdown report before committing. The command must not be used for partial, synthetic, credential-incomplete, or provider-free runs. From beda367fe5fa7bc0fa65578c4cd3d31c7d37bc7a Mon Sep 17 00:00:00 2001 From: Shaurya Saria Date: Wed, 19 Aug 2026 18:50:14 +0530 Subject: [PATCH 5/7] docs: normalize README references --- TODO.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/TODO.md b/TODO.md index 2347b48..6ac1dea 100644 --- a/TODO.md +++ b/TODO.md @@ -128,7 +128,7 @@ Complete these main tasks in order. A main task may be ticked only after its sub - [ ] Record dataset size, clip length, hardware, operating system, network region, date, provider plan, and warm/cold conditions. - [ ] Add confidence intervals or an explicit limitation explaining why the benchmark is too small for them. - [x] Implement a fourth local constellation-map/hash-pair backend over a small library, with an index builder and benchmark integration. Real-world accuracy and speed comparison remain open until the corpus is recorded. -- [ ] Replace `Test set accuracy | X / Y songs matched correctly` in `readme.md` with generated, reviewable results. +- [ ] Replace `Test set accuracy | X / Y songs matched correctly` in `README.md` with generated, reviewable results. - [ ] Remove or substantiate the existing `~2.1 s` and `~3.4 s` performance values. ### Matcher correctness From 1089259592f439ae669bedc7bfb446c7d92a2b77 Mon Sep 17 00:00:00 2001 From: Shaurya Saria Date: Wed, 19 Aug 2026 18:50:22 +0530 Subject: [PATCH 6/7] docs: normalize README references --- scripts/update_readme.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/scripts/update_readme.py b/scripts/update_readme.py index 02b39dc..1a358d0 100644 --- a/scripts/update_readme.py +++ b/scripts/update_readme.py @@ -143,10 +143,10 @@ def update_readme(readme_path: Path, results: dict[str, Any]) -> None: def main() -> int: parser = argparse.ArgumentParser( - description="Import complete benchmark results into readme.md." + description="Import complete benchmark results into README.md." ) parser.add_argument("--results", type=Path, required=True) - parser.add_argument("--readme", type=Path, default=Path("readme.md")) + parser.add_argument("--readme", type=Path, default=Path("README.md")) args = parser.parse_args() try: results = json.loads(args.results.read_text(encoding="utf-8")) From 0fa507958e74cc40e696f22c7d4715772f51bbd1 Mon Sep 17 00:00:00 2001 From: Shaurya Saria Date: Wed, 19 Aug 2026 18:50:28 +0530 Subject: [PATCH 7/7] docs: normalize README references --- tests/test_reproducible_benchmark.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/tests/test_reproducible_benchmark.py b/tests/test_reproducible_benchmark.py index 18583b9..e6764ee 100644 --- a/tests/test_reproducible_benchmark.py +++ b/tests/test_reproducible_benchmark.py @@ -554,7 +554,7 @@ def forbidden_provider(*_args, **_kwargs): assert summary["missing_inputs"] == 1 assert results["records"][0]["error_code"] == "missing_clip" assert results["records"][0]["status"] == "invalid_audio" - readme = tmp_path / "readme.md" + readme = tmp_path / "README.md" readme.write_text( "before\n\nold\n\n", encoding="utf-8", @@ -800,7 +800,7 @@ def _replace_records(results: dict, records: list[dict], clip_count: int) -> dic def test_readme_update_refuses_incomplete_results_without_writing(tmp_path): - readme = tmp_path / "readme.md" + readme = tmp_path / "README.md" original = "before\n\nold\n\n" readme.write_text(original, encoding="utf-8") incomplete = {"metadata": {"complete": False}, "backend_summary": {}, "clip_count": 1} @@ -832,7 +832,7 @@ def test_validator_rejects_inconsistent_denominator_even_when_metadata_claims_co def test_readme_update_imports_only_generated_complete_metrics(tmp_path): - readme = tmp_path / "readme.md" + readme = tmp_path / "README.md" readme.write_text( "before\n\nold\n\n", encoding="utf-8",