From dc982909fcda709604cdb8d5dd731fc31fe15867 Mon Sep 17 00:00:00 2001 From: Steve Seguin Date: Mon, 7 Sep 2026 18:00:22 -0400 Subject: [PATCH] Release abandoned capture sessions and test browser recovery in CI --- .github/workflows/ci.yml | 5 ++ API.md | 6 +- CONTRIBUTING.md | 5 ++ OPERATIONS.md | 10 ++- scripts/browser_capture_https.py | 9 ++- scripts/browser_discard.py | 134 +++++++++++++++++++++++++++++++ server.py | 12 ++- static/app.js | 10 ++- tests/test_api_compat.py | 36 +++++++++ 9 files changed, 217 insertions(+), 10 deletions(-) create mode 100644 scripts/browser_discard.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 51c7b35..abeb2b6 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -29,6 +29,11 @@ jobs: run: node --test tests/audio-buffer.test.cjs tests/local-connection.test.cjs tests/relay-config.test.cjs tests/relay-publisher.test.cjs - name: Documentation and source checks run: python scripts/check_release.py + - name: Browser discard and admission recovery + run: | + python -m pip install playwright==1.62.0 + python -m playwright install --with-deps chromium + python scripts/browser_discard.py --output /tmp/caption-discard.json - name: Build source archive run: python scripts/package_release.py - name: Validate Compose configurations diff --git a/API.md b/API.md index 942cc2f..9bba041 100644 --- a/API.md +++ b/API.md @@ -29,8 +29,10 @@ before sending it. Uploads have a ten-second deadline and bounded admission. The adapter uses the native scheduler and retry cache. Optional `X-Stream-ID` and `X-Request-ID` retain native retry identity; keep them stable across retries and -use distinct stream IDs per producer. Without a stream ID, successful calls use -temporary independent sessions that are closed afterward. Automatic SDK retries +use distinct stream IDs per producer. Without a stream ID, calls use temporary +independent sessions that close afterward. If the client disconnects during +inference, the temporary session closes when its worker finishes; the running +worker keeps its admission slot until then. Automatic SDK retries without both IDs do not promise idempotency. Native `detail` errors and HTTP status codes are retained. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 8b53095..8831f09 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -76,6 +76,7 @@ records the checked pages, viewport sizes and limits of these checks. # Windows; use .venv/bin/python on Linux. Requires development dependencies. .venv\Scripts\python.exe scripts/review_pages.py --checkout samples/captionninja --output samples/visual-review/ui .venv\Scripts\python.exe scripts/review_guides.py --output samples/visual-review/guides +.venv\Scripts\python.exe scripts/browser_discard.py --output samples/discard-review/result.json ``` Both commands expect a captionninja checkout at `samples/captionninja`; the UI @@ -84,6 +85,10 @@ relay traffic are mocked. Guide rendering requires authenticated `gh` access; it submits public Markdown to GitHub's renderer and uses a local reading stylesheet. It does not reproduce GitHub's surrounding interface. Refresh the onboarding screenshot with `scripts/capture_onboarding.py` after capture changes. +The discard probe uses its own temporary synthetic tone, a real loopback HTTP +scheduler and an injected engine failure. It checks retained audio and admission +release on both capture pages without downloading a model or recording a microphone. +Linux CI runs it automatically alongside the protocol tests. ## Changes worth testing diff --git a/OPERATIONS.md b/OPERATIONS.md index 5ecf73a..377d83c 100644 --- a/OPERATIONS.md +++ b/OPERATIONS.md @@ -63,7 +63,10 @@ or browser audio suspension stops capture and drains already captured speech. competing CPU load before starting again. Reduce the active stream count or use a faster model. - **Relay disconnected:** caption.ninja output queues at most 100 captions. Queue length and dropped messages are visible. Local text remains downloadable. - External relay delivery has no acknowledgement or guaranteed replay. + The public relay has no acknowledgement or guaranteed replay. The optional + [private relay](https://github.com/steveseguin/captionninja/blob/master/relay/README.md) + acknowledges delivery and replays bounded in-memory history. Expired history + and server restarts can still cause visible gaps; recovery is not durable. - **Wrong words at a boundary:** check the recording and sensitivity; use human review. This is still automatic recognition, not an exact transcription promise. - **Port already used:** stop the previous instance or choose `--port` / CAPTION_PORT. @@ -100,8 +103,9 @@ are pending, but a crashed browser cannot preserve in-memory work. An optional shared service token and request metadata logs are described in [deployment profiles](docs/DEPLOYMENT-PROFILES.md). Individual accounts, tenant isolation and TLS termination are not supplied. -For remote microphones, keep the inference host on loopback and use SSH. Room -names are relay access secrets, not encryption. Do not reuse real event rooms in +For remote microphones, keep the inference host on loopback and use SSH. Public +relay room names act as access secrets; the private relay uses separate viewing +and publishing tokens. Neither replaces encrypted transport. Do not reuse real event rooms in tests. Windows CPU development tests are recorded separately from the published Linux CPU release. Native Windows and WSL NVIDIA inference are now tested on a TITAN RTX; see the [GPU sustained report](evidence/gpu-sustained/report.md) for diff --git a/scripts/browser_capture_https.py b/scripts/browser_capture_https.py index 66cce31..b305605 100644 --- a/scripts/browser_capture_https.py +++ b/scripts/browser_capture_https.py @@ -2,6 +2,7 @@ import argparse import base64 import json +import mimetypes from pathlib import Path import subprocess import sys @@ -52,7 +53,8 @@ try: base=(ROOT/'samples/captionninja').resolve() if source.is_relative_to(base) and source.is_file(): cdp.send('Fetch.fulfillRequest',{'requestId':request_id,'responseCode':200, - 'responseHeaders':[{'name':'Content-Type','value':'text/html' if source.suffix=='.html' else 'application/javascript'}], + 'responseHeaders':[{'name':'Content-Type','value':mimetypes.guess_type(source)[0] or 'application/octet-stream'}, + {'name':'X-Content-Type-Options','value':'nosniff'}], 'body':base64.b64encode(source.read_bytes()).decode('ascii')}); return cdp.send('Fetch.failRequest',{'requestId':request_id,'errorReason':'BlockedByClient'}) # Probe direct Fetch events without relying on Playwright's usual @@ -66,7 +68,12 @@ try: ready=not page.locator('#start').is_disabled() scenario={'permission':permission,'ready':ready,'request_failures':failures,'console':console, 'intercepted_paths':intercepted,'secure_context':page.evaluate('isSecureContext')} + scenario['stylesheet_loaded']=page.evaluate('''() => [...document.styleSheets].some(sheet => { + try { return new URL(sheet.href).pathname.endsWith('/capture.css') && sheet.cssRules.length > 0; } + catch (_) { return false; } + })''') report['scenarios'].append(scenario) + assert scenario['stylesheet_loaded'], 'Capture stylesheet did not load in HTTPS preview' if ready: page.click('#start') try: diff --git a/scripts/browser_discard.py b/scripts/browser_discard.py new file mode 100644 index 0000000..76b5ed3 --- /dev/null +++ b/scripts/browser_discard.py @@ -0,0 +1,134 @@ +"""Check that discarding failed capture releases a real server admission slot. + +Uses a synthetic microphone, the actual HTTP scheduler and a failing fake engine. +No real model, physical recording or external caption relay is used. +""" +import argparse +import json +from pathlib import Path +import sys +import tempfile +import threading +import urllib.error +import urllib.request +import wave + +import numpy as np +import uvicorn +from playwright.sync_api import sync_playwright + +from browser_support import browser_options, authenticate +from service_test_support import require_free_port + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT)) +from server import create_app + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('--output', type=Path, required=True) + args = parser.parse_args() + require_free_port(8791) + args.output.parent.mkdir(parents=True, exist_ok=True) + with tempfile.NamedTemporaryFile(suffix='.wav', delete=False) as temporary: + fixture = Path(temporary.name) + with wave.open(str(fixture), 'wb') as audio: + audio.setparams((1, 2, 16000, 0, 'NONE', 'not compressed')) + tone = .05 * np.sin(2 * np.pi * 220 * np.arange(96000) / 16000) + audio.writeframes((tone * 32767).astype(' failed && !processing && !stopping', timeout=30000) + row = {'path': path, 'retained_before_discard': page.evaluate( + '() => !!pending && buffer.length > 0'), 'browser_errors': errors} + report['pages'].append(row) + assert row['retained_before_discard'] + assert page.locator('#status').inner_text() == 'Stopped · pending audio retained' + audio = np.full(16000, .02, dtype=' !failed && !pending && !processing && buffer.length === 0') + row['sessions_after_discard'] = request('/health')[1]['sessions'] + row['other_producer_after_discard'] = request('/transcribe?language=en', audio)[0] + assert row['sessions_after_discard'] == 0, 'Discard left the service admission slot occupied' + assert row['other_producer_after_discard'] == 200 + assert not errors + page.screenshot(path=str(args.output.with_name(f'discard-{slug}-cleared.png')), full_page=True) + request('/streams/another-producer', method='DELETE') + context.close() + browser.close() + report['passed'] = True + except Exception as error: + report['error'] = str(error) + raise + finally: + service.should_exit = True + thread.join(timeout=10) + fixture.unlink(missing_ok=True) + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(json.dumps(report, indent=2) + '\n', encoding='utf-8', newline='\n') + print('Discard releases admission capacity on both capture pages') + + +if __name__ == '__main__': + main() diff --git a/server.py b/server.py index 0e8145c..b8782e8 100644 --- a/server.py +++ b/server.py @@ -167,6 +167,8 @@ def create_app(engine, max_streams=12, queue_timeout=20, api_key=None, log_reque def release(stream_id): streams[stream_id]["busy"] = False streams[stream_id]["seen"] = time.monotonic() + if streams[stream_id].get("close_when_idle"): + del streams[stream_id] running.pop(stream_id, None) state["busy_since"] = min(running.values(), default=None) @@ -401,8 +403,14 @@ def create_app(engine, max_streams=12, queue_timeout=20, api_key=None, log_reque from api_compat import register_audio_api def close_ephemeral(stream_id): session = streams.get(stream_id) - if session and not session['busy']: - streams.pop(stream_id, None) + if session: + if session['busy']: + # A disconnected WAV client has no stream ID to reuse. Keep + # admission locked until its shielded worker finishes, then + # release the otherwise abandoned session immediately. + session['close_when_idle'] = True + else: + streams.pop(stream_id, None) register_audio_api(app, transcribe, close_ephemeral, max_streams, getattr(engine, 'model_name', 'local')) app.mount("/static", StaticFiles(directory=ROOT / "static"), name="static") return app diff --git a/static/app.js b/static/app.js index 0e1ae96..30919fc 100644 --- a/static/app.js +++ b/static/app.js @@ -159,7 +159,8 @@ async function drain() { } } catch (error) { failed = true; - fail(`${error.message}. Capture stopped. Pending audio is retained: retry it or discard it below.`); + $('status').textContent = 'Stopped · pending audio retained'; + fail(`${error.message}. Capture stopped. Audio is retained: use Retry pending audio or Discard pending audio.`); await stop(); } finally { if (!running && !stopping && !failed && !pending) { @@ -199,7 +200,12 @@ async function stop() { } $('stop').onclick = stop; $('retry').onclick = async () => { failed = false; fail(''); await drain(); }; -$('discard').onclick = () => { buffer.reset(); pending = null; failed = false; fail('Pending audio discarded.'); controls(); }; +$('discard').onclick = async () => { + buffer.reset(); pending = null; failed = false; fail('Pending audio discarded.'); + // The empty drain closes our idle server session, just like a successful Stop. + await drain(); + if (!failed) $('status').textContent = 'Stopped'; +}; $('start').onclick = async () => { if (starting || running || processing || failed) return; starting = true; controls(); fail(''); diff --git a/tests/test_api_compat.py b/tests/test_api_compat.py index e580657..55695c6 100644 --- a/tests/test_api_compat.py +++ b/tests/test_api_compat.py @@ -132,3 +132,39 @@ def test_adapter_disconnect_keeps_worker_and_native_retry_identity(): assert (await call()).json()=={'text':'Bienvenue.'} assert engine.calls==1 asyncio.run(run()) + + +def test_disconnected_ephemeral_upload_releases_session_after_worker_finishes(): + entered = threading.Event() + release = threading.Event() + + class Blocked(Engine): + def transcribe(self, audio, language, task='transcribe'): + entered.set() + assert release.wait(5) + return 'Finished.', 'en' + + async def run(): + app = create_app(Blocked(), max_streams=1) + async with httpx.AsyncClient(transport=httpx.ASGITransport(app=app), base_url='http://localhost') as client: + async def upload(): + return await client.post('/v1/audio/transcriptions', files={'file': ('test.wav', wav())}, + data={'model': 'small', 'language': 'en'}) + pending = asyncio.create_task(upload()) + assert await asyncio.to_thread(entered.wait, 3) + try: + pending.cancel() + with pytest.raises(asyncio.CancelledError): + await pending + health = (await client.get('/health')).json() + assert health['running'] == health['sessions'] == 1 + assert (await upload()).status_code == 429 + finally: + release.set() + async with asyncio.timeout(3): + while (await client.get('/health')).json()['running']: + await asyncio.sleep(.01) + assert (await client.get('/health')).json()['sessions'] == 0 + assert (await upload()).status_code == 200 + assert (await client.get('/health')).json()['sessions'] == 0 + asyncio.run(run())