diff --git a/.github/workflows/check.yml b/.github/workflows/check.yml index a064ec36..963f933a 100644 --- a/.github/workflows/check.yml +++ b/.github/workflows/check.yml @@ -850,6 +850,11 @@ jobs: code=$(status "$probe") [ "$code" = 200 ] || { echo "embedded $probe answered $code, not 200" >&2; exit 1; } done + voice_type=$(curl -fsS -o "$empty/test-voice.wav" -w '%{content_type}' \ + http://127.0.0.1:18999/audio/test-voice.wav) + [ "$voice_type" = audio/wav ] \ + || { echo "embedded test voice has type $voice_type, not audio/wav" >&2; exit 1; } + cmp web/audio/test-voice.wav "$empty/test-voice.wav" [ "$(status /.git/config)" = 404 ] \ || { echo "a dotfile was served" >&2; exit 1; } diff --git a/scripts/browser-check.cjs b/scripts/browser-check.cjs index 38d64b5a..8abfedd6 100644 --- a/scripts/browser-check.cjs +++ b/scripts/browser-check.cjs @@ -521,10 +521,7 @@ async function clearMediaGate(page) { await gate.waitFor({ state: "visible", timeout: 30000 }); await page.getByRole("button", { name: "Play test tone" }).click(); await page.getByRole("button", { name: "I heard it" }).click(); - if ( - (await page.locator("#audio-step-output p").innerText()) !== - "Output — Confirmed" - ) { + if ((await page.locator("#audio-output-state").innerText()) !== "Confirmed") { throw new Error("output confirmation did not render immediately"); } const join = page.getByRole("button", { name: "Start interview" }); diff --git a/src/web/assets.rs b/src/web/assets.rs index ae5b1600..f5294813 100644 --- a/src/web/assets.rs +++ b/src/web/assets.rs @@ -498,6 +498,7 @@ pub(crate) fn content_type(path: &Path) -> Option { // served as anything but JavaScript is refused, not sniffed. Some("js" | "mjs") => "text/javascript; charset=utf-8", Some("json") => "application/json", + Some("wav") => "audio/wav", // `WebAssembly.instantiateStreaming` refuses anything that is not // `application/wasm`, and a body with no type at all is a coin toss for diff --git a/tests/browser/preflight-speech.test.js b/tests/browser/preflight-speech.test.js new file mode 100644 index 00000000..8e8bfdd8 --- /dev/null +++ b/tests/browser/preflight-speech.test.js @@ -0,0 +1,243 @@ +// Browser playback cannot prove physical audibility through system effects. +import { after, before, test } from "node:test"; +import assert from "node:assert/strict"; +import { DEFAULT_RUNTIME_CONFIG, startStaticServer } from "./source.js"; + +let browser; +let server; +let base; +before(async () => { + try { + const { chromium } = await import("playwright"); + browser = await chromium.launch({ + args: [ + "--use-fake-device-for-media-stream", + "--use-fake-ui-for-media-stream", + ], + }); + } catch (error) { + if (process.env.CI) throw error; + return; + } + ({ server, base } = await startStaticServer({ + runtimeConfig: DEFAULT_RUNTIME_CONFIG, + })); +}); +after(async () => { + await browser?.close(); + await new Promise((closed) => (server ? server.close(closed) : closed())); +}); + +async function preflight(t) { + const page = await browser.newPage(); + t.after(() => page.close()); + await page.addInitScript(() => { + const capture = navigator.mediaDevices.getUserMedia.bind( + navigator.mediaDevices, + ); + navigator.mediaDevices.getUserMedia = async (constraints) => { + const stream = await capture(constraints); + if (constraints.audio) window.testMicrophone = stream.getAudioTracks()[0]; + return stream; + }; + }); + await page.goto(`${base}/interview.html?problem=chargeback-pair-match`); + await page.waitForFunction( + () => window.testMicrophone?.readyState === "live", + ); + return page; +} + +test("preflight speech replays during capture and still requires the candidate's confirmation", async (t) => { + if (!browser) return t.skip("playwright chromium unavailable"); + const page = await preflight(t); + const response = await page.request.get(`${base}/audio/test-voice.wav`); + assert.equal(response.status(), 200); + assert.match(response.headers()["content-type"], /^audio\/wav/); + const signal = await page.evaluate(async () => { + const response = await fetch("./audio/test-voice.wav"); + const buffer = await new OfflineAudioContext(1, 1, 48000).decodeAudioData( + await response.arrayBuffer(), + ); + let peak = 0; + for (const value of buffer.getChannelData(0)) + peak = Math.max(peak, Math.abs(value)); + return { peak, duration: buffer.duration }; + }); + assert.ok(signal.peak > 0.01, "the shipped speech sample contains a signal"); + assert.ok(signal.duration > 1 && signal.duration < 10); + for (let count = 0; count < 3; count++) { + await page.locator("#audio-test-speech").click(); + await page.waitForFunction( + () => document.querySelector("#audio-test-voice").ended, + ); + assert.equal( + await page.evaluate(() => window.testMicrophone.readyState), + "live", + ); + assert.equal( + await page + .locator("#audio-step-output") + .evaluate((node) => node.classList.contains("done")), + false, + ); + assert.equal(await page.locator("#audio-check-join").isDisabled(), true); + } + await page.locator("#audio-heard").click(); + await page.waitForFunction(() => + document.querySelector("#audio-step-output").classList.contains("done"), + ); +}); + +test("preflight speech reports a rejected play and lets the candidate retry", async (t) => { + if (!browser) return t.skip("playwright chromium unavailable"); + const page = await preflight(t); + await page.evaluate(() => { + document.querySelector("#audio-test-voice").play = () => + Promise.reject(new DOMException("Autoplay blocked", "NotAllowedError")); + }); + await page.locator("#audio-test-speech").click(); + await page.waitForFunction(() => + document + .querySelector("#audio-check-status") + .textContent.includes("Could not play"), + ); + assert.equal(await page.locator("#audio-test-speech").isEnabled(), true); + assert.equal(await page.locator("#audio-heard").textContent(), "I heard it"); + await page.evaluate(() => { + delete document.querySelector("#audio-test-voice").play; + }); + await page.locator("#audio-test-speech").click(); + await page.waitForFunction( + () => document.querySelector("#audio-test-voice").ended, + ); + assert.equal( + await page.locator("#audio-test-voice").evaluate((audio) => audio.error), + null, + ); +}); + +test("preflight retries a failed speech load without confirming output", async (t) => { + if (!browser) return t.skip("playwright chromium unavailable"); + const page = await preflight(t); + let requests = 0; + await page.route("**/audio/test-voice.wav", (route) => { + requests++; + return requests === 1 + ? route.fulfill({ status: 404, body: "not found" }) + : route.continue(); + }); + await page.locator("#audio-test-speech").click(); + await page.waitForFunction( + () => document.querySelector("#audio-test-voice").error?.code === 4, + ); + await page.waitForFunction(() => + document + .querySelector("#audio-check-status") + .textContent.includes("Could not"), + ); + assert.equal(await page.locator("#audio-test-speech").isEnabled(), true); + assert.equal(requests, 1); + await page.locator("#audio-test-speech").click(); + await page.waitForFunction( + () => document.querySelector("#audio-test-voice").ended, + ); + assert.ok(requests > 1, "retry fetches the speech asset again"); + assert.equal( + await page.locator("#audio-test-voice").evaluate((audio) => audio.error), + null, + ); + assert.equal( + await page + .locator("#audio-step-output") + .evaluate((node) => node.classList.contains("done")), + false, + ); +}); + +test("switching to the tone cancels pending speech without reporting a playback failure", async (t) => { + if (!browser) return t.skip("playwright chromium unavailable"); + const page = await preflight(t); + let releaseSample; + const heldSample = new Promise((resolve) => { + releaseSample = resolve; + }); + await page.route("**/audio/test-voice.wav", async (route) => { + await heldSample; + await route.continue(); + }); + try { + await page.locator("#audio-test-speech").click(); + assert.equal(await page.locator("#audio-test-speech").isDisabled(), true); + await page.locator("#audio-test-tone").click(); + await page.waitForFunction( + () => !document.querySelector("#audio-test-speech").disabled, + ); + assert.equal( + await page.locator("#audio-test-voice").evaluate((audio) => audio.paused), + true, + ); + assert.doesNotMatch( + await page.locator("#audio-check-status").textContent(), + /Could not/, + ); + } finally { + releaseSample(); + } +}); + +test("leaving the preflight stops speech before navigation", async (t) => { + if (!browser) return t.skip("playwright chromium unavailable"); + const page = await preflight(t); + let releaseLobby; + const heldLobby = new Promise((resolve) => { + releaseLobby = resolve; + }); + await page.route(`${base}/`, async (route) => { + await heldLobby; + await route.continue(); + }); + try { + await page.locator("#audio-test-speech").click(); + await page.waitForFunction( + () => !document.querySelector("#audio-test-voice").paused, + ); + const stopped = await page.evaluate(() => { + document.querySelector("#audio-check-leave").click(); + const audio = document.querySelector("#audio-test-voice"); + return { paused: audio.paused, time: audio.currentTime }; + }); + assert.deepEqual(stopped, { paused: true, time: 0 }); + } finally { + releaseLobby(); + } + await page.waitForURL(`${base}/`); +}); + +test("starting the interview stops preflight speech while keeping the microphone", async (t) => { + if (!browser) return t.skip("playwright chromium unavailable"); + const page = await preflight(t); + await page.route("**/api/token", (route) => + route.fulfill({ status: 503, body: "local test" }), + ); + await page.locator("#camera-skip").click(); + await page.locator("#audio-heard").click(); + await page.waitForFunction( + () => !document.querySelector("#audio-check-join").disabled, + ); + await page.locator("#audio-test-speech").click(); + await page.waitForFunction( + () => !document.querySelector("#audio-test-voice").paused, + ); + const stopped = await page.evaluate(() => { + document.querySelector("#audio-check-join").click(); + const audio = document.querySelector("#audio-test-voice"); + return { + paused: audio.paused, + time: audio.currentTime, + microphone: window.testMicrophone.readyState, + }; + }); + assert.deepEqual(stopped, { paused: true, time: 0, microphone: "live" }); + assert.equal(await page.locator("#audio-check").isHidden(), true); +}); diff --git a/tests/browser/source.js b/tests/browser/source.js index a734abbe..0b5650f6 100644 --- a/tests/browser/source.js +++ b/tests/browser/source.js @@ -220,6 +220,7 @@ const STATIC_CONTENT_TYPES = { ".html": "text/html", ".css": "text/css", ".wasm": "application/wasm", + ".wav": "audio/wav", }; /// A static file server over `web/`, the way the browser sees it once diff --git a/tests/unit/web/assets.rs b/tests/unit/web/assets.rs index 9974a66c..0a06bef5 100644 --- a/tests/unit/web/assets.rs +++ b/tests/unit/web/assets.rs @@ -5,6 +5,20 @@ use super::{EmbeddedWeb, embedded_etag, is_refused_segment, static_candidates}; +#[test] +fn test_voice_is_served_as_wave_audio() { + assert_eq!( + super::content_type(std::path::Path::new("audio/test-voice.wav")) + .unwrap() + .to_str() + .unwrap(), + "audio/wav" + ); + let sample = EmbeddedWeb::get("audio/test-voice.wav").expect("shipped test voice"); + assert!(sample.data.starts_with(b"RIFF")); + assert_eq!(&sample.data[8..12], b"WAVE"); +} + /// Normalization is tested here rather than through a served request /// because a served request cannot see it. `EmbeddedWeb::get` matches keys /// exactly only in release; a debug or test build reads the same names off diff --git a/web/audio-check.js b/web/audio-check.js index 1c6b4870..eed1e89c 100644 --- a/web/audio-check.js +++ b/web/audio-check.js @@ -124,7 +124,7 @@ export function mediaReadiness({ steps, ready: false, blocker: "output", - message: "Play the test tone and confirm you heard it.", + message: "Play the test tone or voice and confirm you heard it.", }; } return { diff --git a/web/audio/README.md b/web/audio/README.md new file mode 100644 index 00000000..b243840f --- /dev/null +++ b/web/audio/README.md @@ -0,0 +1,22 @@ +# Test voice + +`test-voice.wav` was generated with [TTSMaker](https://ttsmaker.com/) on +2026-10-08, using voice 148, Alayna (US English, female). The downloaded MP3 +was converted to 24,000 Hz, 16-bit mono PCM WAV. + +Spoken text: "If you can hear this voice, your audio is working." + +## Audio usage grant + +The repository's MIT license covers the code. This generated audio is provided +under [TTSMaker's commercial license terms](https://ttsmaker.com/copyright_and_commercial_license_terms). +The redistribution grant below was recorded on 2026-10-09: + +> Users may freely use, modify, reproduce, distribute, and display the audio +> content for both personal and commercial purposes + +TTSMaker grants a non-exclusive, irrevocable, worldwide, perpetual license to +generated audio, with no extra fee or permission needed for those uses. This +is a usage license; it does not transfer copyright ownership or rights to the +underlying technology or voice models. Use must comply with local law and +respect third-party rights. diff --git a/web/audio/test-voice.wav b/web/audio/test-voice.wav new file mode 100644 index 00000000..f5fb3470 Binary files /dev/null and b/web/audio/test-voice.wav differ diff --git a/web/interview.html b/web/interview.html index 914d4234..b6326318 100644 --- a/web/interview.html +++ b/web/interview.html @@ -347,13 +347,16 @@

Media preflight

Output — play a short tone.play a test sound and confirm you heard it.

+
+ +

+ If the tone is silent, try the test voice. Playback noise + reduction or audio enhancements may filter test sounds. If both + are silent, check your output device and volume, or try turning + those effects off in your audio settings. +

diff --git a/web/interview.js b/web/interview.js index 5935ee77..ad2a7b4c 100644 --- a/web/interview.js +++ b/web/interview.js @@ -505,6 +505,8 @@ const nodes = { report: document.querySelector("#report-modal"), audioCheck: document.querySelector("#audio-check"), audioTestTone: document.querySelector("#audio-test-tone"), + audioTestSpeech: document.querySelector("#audio-test-speech"), + audioTestVoice: document.querySelector("#audio-test-voice"), audioHeard: document.querySelector("#audio-heard"), audioMeterFill: document.querySelector("#audio-meter-fill"), audioStatus: document.querySelector("#audio-check-status"), @@ -966,7 +968,7 @@ function paintPreflight(state, hint) { nodes.audioStatus.textContent = hint || state.message; nodes.audioOutputState.textContent = state.steps.output ? "Confirmed" - : "play a short tone."; + : "play a test sound and confirm you heard it."; nodes.cameraState.textContent = state.cameraSkipped ? `Not used (${state.cameraSkipReason}).` : state.steps.camera @@ -988,8 +990,8 @@ function paintPreflight(state, hint) { } /// Resolves once the candidate has proven output, microphone, and camera. -/// The tone doubles as the user gesture browsers require before -/// any audio plays, so confirming it also unblocks the interviewer's voice. +/// Confirming output supplies the user gesture browsers require before +/// the interviewer's voice plays, whether the candidate heard tone or speech. function runAudioCheck() { return new Promise((resolve) => { const mediaDevices = navigator.mediaDevices; @@ -1097,6 +1099,8 @@ function runAudioCheck() { // so a request still pending when this runs never lights the camera. const stopChecks = () => { finished = true; + nodes.audioTestVoice.pause(); + nodes.audioTestVoice.currentTime = 0; meter.stop(); pool.cancelRetry(); faceCheck.close(); @@ -1138,6 +1142,7 @@ function runAudioCheck() { }; nodes.audioTestTone.addEventListener("click", async () => { + nodes.audioTestVoice.pause(); if (!(await unblockOutput())) { showHint("The browser is still blocking audio. Click the tone again."); return; @@ -1154,6 +1159,37 @@ function runAudioCheck() { refresh(); }); + // Playback-side noise reduction can remove a pure tone while preserving + // speech. A local sample needs no interview session or synthesis service, + // and playing it says nothing about whether the candidate actually heard it. + nodes.audioTestSpeech.addEventListener("click", async () => { + nodes.audioTestSpeech.disabled = true; + try { + nodes.audioTestVoice.pause(); + nodes.audioTestVoice.currentTime = 0; + if (nodes.audioTestVoice.error) nodes.audioTestVoice.load(); + await nodes.audioTestVoice.play(); + if (finished) return; + nodes.audioTestSpeech.textContent = "Play speech again"; + showHint('If you hear the test voice, click "I heard it".'); + } catch (error) { + if (!finished && error?.name !== "AbortError") { + showHint( + "Could not play the test voice. Try again and check your audio settings.", + ); + } + } finally { + if (!finished) nodes.audioTestSpeech.disabled = false; + } + }); + nodes.audioTestVoice.addEventListener("error", () => { + if (!finished) { + showHint( + "Could not load the test voice. Try again and check your connection.", + ); + } + }); + // Bound to both events because a mouse click fires each of them and // pointerdown is the earliest gesture available to unblock audio, so the // second call is expected and returns immediately.