PJSUA2 media plane — audio finally reaches the classifier #8

Merged
r merged 5 commits from feat/pjsua2-media-tap into main 2026-07-29 21:34:10 +00:00
2 changed files with 158 additions and 12 deletions
Showing only changes of commit 92c45e9c4d - Show all commits

View File

@@ -12,7 +12,6 @@ finds `lab-music.sln`.
python generate.py [outdir] python generate.py [outdir]
""" """
import struct
import sys import sys
from pathlib import Path from pathlib import Path
@@ -29,13 +28,14 @@ def _write_sln(path: Path, samples: np.ndarray) -> None:
print(f" {path.name}: {len(pcm) / RATE:.1f}s ({path.stat().st_size} bytes)") print(f" {path.name}: {len(pcm) / RATE:.1f}s ({path.stat().st_size} bytes)")
def make_music(seconds: float = 30.0) -> np.ndarray: def make_music(seconds: float = 30.0, seed: int = 7) -> np.ndarray:
"""Sustained multi-harmonic tones — what the classifier must call MUSIC. """Sustained multi-harmonic tones — what the classifier must call MUSIC.
A chord progression with stable pitch and strong harmonic structure. The A chord progression with stable pitch and strong harmonic structure. The
steady spectrum across a long window is what distinguishes music from steady spectrum across a long window is what distinguishes music from
speech; this deliberately has no pauses. speech; this deliberately has no pauses.
""" """
rng = np.random.default_rng(seed)
t = np.linspace(0, seconds, int(RATE * seconds), endpoint=False) t = np.linspace(0, seconds, int(RATE * seconds), endpoint=False)
# A-minor-ish progression, one chord per 2s bar. # A-minor-ish progression, one chord per 2s bar.
chords = [(220.0, 261.6, 329.6), (196.0, 246.9, 293.7), chords = [(220.0, 261.6, 329.6), (196.0, 246.9, 293.7),
@@ -48,12 +48,21 @@ def make_music(seconds: float = 30.0) -> np.ndarray:
break break
mask = (t >= start) & (t < end) mask = (t >= start) & (t < end)
for j, freq in enumerate(chord): for j, freq in enumerate(chord):
# Fundamental plus two harmonics, decaying — a plucked-string feel. # Fundamental plus four harmonics, decaying — a plucked-string
for h, amp in ((1, 0.30), (2, 0.12), (3, 0.05)): # feel. Enough harmonics to keep spectral flatness inside the
# music score's 0.05-0.4 band: with only three, some windows fall
# *below* 0.05 (too pure to read as music) and score as speech.
for h, amp in ((1, 0.30), (2, 0.12), (3, 0.05), (4, 0.03), (5, 0.02)):
out[mask] += amp / (j + 1) * np.sin(2 * np.pi * freq * h * t[mask]) out[mask] += amp / (j + 1) * np.sin(2 * np.pi * freq * h * t[mask])
# Gentle per-bar envelope so bars are distinguishable but never silent. # Gentle per-bar envelope so bars are distinguishable but never silent.
env = 0.8 + 0.2 * np.sin(2 * np.pi * (t[mask] - start) / bar) env = 0.8 + 0.2 * np.sin(2 * np.pi * (t[mask] - start) / bar)
out[mask] *= env out[mask] *= env
# Recording-style noise floor. Windows straddling a chord change have a
# momentarily sparse spectrum and land just *under* the music score's
# 0.05 flatness floor, scoring as speech. This is well below the level
# that would disturb tonality — every real recording has one.
out += rng.normal(0, 0.004, len(out))
return out * 0.45 return out * 0.45
@@ -74,16 +83,44 @@ def make_speech(seconds: float = 8.0, seed: int = 1337) -> np.ndarray:
mask = (t >= pos) & (t < pos + syl) mask = (t >= pos) & (t < pos + syl)
if mask.any(): if mask.any():
local = t[mask] - pos local = t[mask] - pos
f0 = rng.uniform(95, 165) # fundamental — adult speaking range frac = local / syl
# Two formants, swept slightly across the syllable. The ranges
# deliberately avoid the DTMF bands (rows 697-941, columns # Pitch CONTOUR, not a constant. This is the single feature that
# 1209-1633): a formant pair landing on both trips the Goertzel # separates this fixture from music. `_detect_tonality` looks for
# detector and the whole utterance is classified as a keypress. # an autocorrelation peak > 0.5 in the 50-1000 Hz lag range; a
f1 = rng.uniform(300, 620) + rng.uniform(-40, 40) * local / syl # fixed f0 is perfectly periodic there, scores is_tonal=True, and
f2 = rng.uniform(1750, 2600) + rng.uniform(-120, 120) * local / syl # hands the music score a free 0.3 that speech cannot outrun.
sig = (0.50 * np.sin(2 * np.pi * f0 * local) # Real voices glide and jitter, so the periodicity never locks.
f0_start = rng.uniform(95, 165)
f0_end = f0_start * rng.uniform(0.72, 1.38) # rise or fall
f0 = f0_start + (f0_end - f0_start) * frac
# Cycle-to-cycle jitter on top of the glide (~2% is human).
f0 *= 1.0 + 0.02 * rng.standard_normal(len(local))
# Integrate frequency to phase — with a varying f0, `2*pi*f*t`
# would be wrong (that is a chirp only if f is the *instantaneous*
# rate, which it is not once f0 itself moves).
ph0 = 2 * np.pi * np.cumsum(f0) / RATE
# Two formants, swept across the syllable. The ranges deliberately
# avoid the DTMF bands (rows 697-941, columns 1209-1633): a formant
# pair landing on both trips the Goertzel detector and the whole
# utterance is classified as a keypress.
f1 = rng.uniform(300, 620) + rng.uniform(-40, 40) * frac
f2 = rng.uniform(1750, 2600) + rng.uniform(-120, 120) * frac
sig = (0.50 * np.sin(ph0)
+ 0.30 * np.sin(2 * np.pi * f1 * local) + 0.30 * np.sin(2 * np.pi * f1 * local)
+ 0.18 * np.sin(2 * np.pi * f2 * local)) + 0.18 * np.sin(2 * np.pi * f2 * local))
# Aspiration noise — HIGH-PASSED, not broadband. Real speech noise
# sits above the formants; flat noise puts energy in every
# Goertzel bin, so the strongest DTMF row and column both clear
# the detector's `total_power * 0.1` threshold and every syllable
# reads as a keypress. A first-difference filter (y[n]-y[n-1]) is
# a cheap +6dB/octave tilt that leaves the 697-1633 Hz DTMF bands
# comparatively empty. The 0.09 level is chosen for margin: it puts
# spectral flatness at ~0.46, mid-way through the 0.1-0.5 band the
# speech score rewards, rather than on either edge.
noise = rng.standard_normal(len(local) + 1)
sig += 0.09 * np.diff(noise)
# Raised-cosine envelope: no clicks at syllable edges. # Raised-cosine envelope: no clicks at syllable edges.
sig *= np.sin(np.pi * local / syl) ** 0.6 sig *= np.sin(np.pi * local / syl) ** 0.6
out[mask] += sig out[mask] += sig

109
tests/test_lab_fixtures.py Normal file
View File

@@ -0,0 +1,109 @@
"""
Lab audio fixtures — the classifier must agree with what each one claims to be.
These guard the *fixtures*, not the classifier. `tests/lab/sounds/generate.py`
synthesises music/speech/silence that the Asterisk lab plays down a real call;
if a fixture drifts into the wrong class, every lab result built on it is
quietly meaningless — a hold-music scenario that never classifies as music
proves nothing about the hold slayer.
The first version of these fixtures passed on the opening 3s window and drifted
to MUSIC after, which a single-window check would not have caught. Hence the
sweep across every window.
Skipped when the fixtures have not been generated: they are gitignored (~680K,
reproducible from a fixed seed), so a fresh checkout has none until
`python tests/lab/sounds/generate.py` runs.
"""
import subprocess
import sys
from pathlib import Path
import numpy as np
import pytest
from config import Settings
from models.call import AudioClassification
from services.audio_classifier import SAMPLE_RATE, AudioClassifier
SOUNDS_DIR = Path(__file__).parent / "lab" / "sounds"
GENERATOR = SOUNDS_DIR / "generate.py"
# The lab writes 8 kHz .sln; the classifier works at 16 kHz.
LAB_RATE = 8000
WINDOW_SAMPLES = SAMPLE_RATE * 3 # classifier's 3s analysis window
FIXTURES = [
("lab-music.sln", AudioClassification.MUSIC),
("lab-speech.sln", AudioClassification.LIVE_HUMAN),
("lab-silence.sln", AudioClassification.SILENCE),
]
def _load_16k(path: Path) -> np.ndarray:
"""Load an 8 kHz .sln and upsample to the classifier's 16 kHz."""
return np.repeat(np.fromfile(path, dtype="<i2"), SAMPLE_RATE // LAB_RATE)
def _windows(samples: np.ndarray, step: int):
"""Yield successive analysis windows; at least one, even for short files."""
end = max(1, len(samples) - WINDOW_SAMPLES)
for offset in range(0, end, step):
yield samples[offset : offset + WINDOW_SAMPLES].astype("<i2").tobytes()
@pytest.fixture(scope="module")
def classifier():
return AudioClassifier(settings=Settings().classifier)
@pytest.mark.parametrize("filename,expected", FIXTURES)
def test_fixture_classifies_correctly_in_every_window(filename, expected, classifier):
"""Every window must classify correctly — not just the first.
Stepped at half the window length so windows overlap: a fixture that only
works on aligned boundaries would still be a trap in a live call, where
the window has no relationship to where the audio started.
"""
path = SOUNDS_DIR / filename
if not path.exists():
pytest.skip(f"{filename} not generated — run {GENERATOR}")
samples = _load_16k(path)
results = [
classifier.classify_chunk(w).audio_type
for w in _windows(samples, step=WINDOW_SAMPLES // 2)
]
wrong = [(i, r.value) for i, r in enumerate(results) if r is not expected]
assert not wrong, (
f"{filename} must classify as {expected.value} in all "
f"{len(results)} windows; wrong: {wrong}"
)
def test_generator_is_deterministic(tmp_path):
"""Same bytes on every run — the whole point of synthesising them.
Real hold music varies per call, so a classifier regression on the PSTN is
indistinguishable from noise. Fixed-seed audio makes the answer binary.
"""
if not GENERATOR.exists():
pytest.skip("generator not present")
def run(target: Path) -> dict[str, bytes]:
subprocess.run(
[sys.executable, str(GENERATOR), str(target)],
check=True,
capture_output=True,
)
return {p.name: p.read_bytes() for p in sorted(target.glob("*.sln"))}
first = run(tmp_path / "a")
second = run(tmp_path / "b")
assert first, "generator produced no .sln files"
assert first.keys() == second.keys()
for name in first:
assert first[name] == second[name], f"{name} differs between runs"