{
  "measurement": "Diarization error rate measured at every hand-off between the live diarizer's raw output and the transcript shown to the user",
  "measured_on": "2026-06-09",
  "app_repository_evidence": "artifacts/agent-state/current-boundary-v6.md:43-52 (app repository)",
  "run_id": "boundary_v6_013_boundary_finalizer_rerun_20260609",
  "fixture": "librispeech_clean_4spk_roundtable_b \u2014 one synthetic four-speaker round-table case, 8 reference turns, 202 reference words",
  "run_totals": {
    "app_wer": 0.2574,
    "der": 0.2284,
    "wder": 0.4406
  },
  "metric": "DER \u2014 share of audio time carrying the wrong speaker label. Lower is better.",
  "limitations": [
    "One fixture, not a corpus average. The same run scored 1 failed case and 9 not-run cases of a 10-case suite.",
    "The fixture is built from LibriSpeech, which is in the diarization model's training data: it is a regression floor, not a quality claim.",
    "Measured on the maintainer's Apple silicon Mac through the eval harness, not on an iPhone."
  ],
  "stages": [
    {
      "stage_order": 1,
      "stage": "Raw live diarizer output, collapsed",
      "der": 0.1941
    },
    {
      "stage_order": 2,
      "stage": "After the stability gate (confirmed speaker events)",
      "der": 0.1505
    },
    {
      "stage_order": 3,
      "stage": "Word-ownership policy decisions",
      "der": 0.4074
    },
    {
      "stage_order": 4,
      "stage": "Live committed words",
      "der": 0.2144
    },
    {
      "stage_order": 5,
      "stage": "End-of-session words",
      "der": 0.1941
    },
    {
      "stage_order": 6,
      "stage": "Offline effective refinement windows",
      "der": 0.1556
    },
    {
      "stage_order": 7,
      "stage": "Refined transcript words",
      "der": 0.2622
    },
    {
      "stage_order": 8,
      "stage": "Displayed transcript words",
      "der": 0.2284
    }
  ]
}
