{
  "benchmark": "Apollo 13 radio communications transcription benchmark",
  "evaluated_at": "2026-05-19",
  "published_at": "2026-07-10",
  "source": {
    "publisher": "NASA",
    "duration_seconds": 178.660476,
    "codec": "Vorbis",
    "sample_rate_hz": 44100,
    "channels": 2,
    "audio_sha256": "fd0e3d47739d553bb0cf53fda782ddc3b410f1ae2b04323fb519a0d75a62308c",
    "reference_sha256": "a2d03718b6244c79a6bd67851d3dba8165cb0ddb14df750cb99adde849758246",
    "mission_context_url": "https://www.nasa.gov/history/houston-weve-had-a-problem/",
    "media_guidelines_url": "https://www.nasa.gov/nasa-brand-center/images-and-media/"
  },
  "configuration": {
    "asr_model": "Whisper large-v3",
    "language": "en",
    "speaker_method": "Fast ECAPA",
    "word_timestamps": true,
    "vad_filter_requested": true,
    "sparse_vad_fallback_used": true,
    "reference_words": 391,
    "predicted_words": 350,
    "matched_words_for_role_scoring": 281
  },
  "metrics": {
    "text_wer_definition": "Word error rate after normalizing case and punctuation. Lower is better.",
    "speaker_wer_definition": "Word error rate where both the aligned word and mapped speaker role must match. Lower is better.",
    "role_accuracy_definition": "Share of exactly aligned words assigned to the correct mapped speaker role. Higher is better.",
    "speaker_mapping_method": "The scorer tests every mapping from anonymous output speaker IDs to HOUSTON or SPACECRAFT and reports the mapping with the lowest speaker-attributed WER. This reference-assisted mapping is evaluation-only."
  },
  "artifacts": {
    "plain_output_sha256": "f70d0a195de8eae8433f81195bc13a29d3635fa9d98221988c84842e90c35d4b",
    "production_speaker_output_sha256": "e1e60b7628b75f72be796ed83d5b75ee49a3a622a206f5cfcd780a28318d89b3",
    "direct_no_vad_two_speakers_output_sha256": "e1e60b7628b75f72be796ed83d5b75ee49a3a622a206f5cfcd780a28318d89b3",
    "forced_three_speakers_output_sha256": "8f95510cd4650c35b1cdd7a9d01578587ce4a01e6ce31b2329c8ce0b40b8e4b8",
    "no_word_timestamps_output_sha256": "6cba0fbd3e4e6ee0f61ee97fdf8e4ac212bb90666a27db550d867fcc2925525c",
    "production_json_sha256": "1b7f9759c4f7614dd79e7297cf9e030b6980495d774cb822ec4f71717d2b4f4e",
    "direct_no_vad_two_speakers_json_sha256": "62c208d5facd2e8bf0bde1024c199813feccde88a66acf7260d9dd147955fd51",
    "forced_three_speakers_json_sha256": "cda1c22f2faebce42bbbacd71c711bb47f989256b414a4625bb109e5f0b51a32",
    "no_word_timestamps_json_sha256": "74a04e75b10877d05c58391f321a03f4f11f795f45b738b69692bdd50473c61b",
    "scorer_sha256": "388adc3961e22aa8930c73b059d7035fc9f0de45cc3b6f94eb5e15857409d0b3",
    "speaker_uncertainty_evaluation_sha256": "7fee2f60ce1d1bfe459362fe997c2852691e00a2a76aa901224bf61b7431fa44"
  },
  "results": [
    {
      "variant": "production_style_sparse_vad_fallback",
      "output_file": "production-output.json",
      "text_wer": 0.2966751918,
      "speaker_attributed_wer": 0.3094629156,
      "matched_word_role_accuracy": 0.9822064057,
      "gpu_worker_seconds": 24.722,
      "generic": true
    },
    {
      "variant": "direct_no_vad_two_speakers",
      "output_file": "direct-no-vad-two-speakers-output.json",
      "text_wer": 0.2966751918,
      "speaker_attributed_wer": 0.3094629156,
      "matched_word_role_accuracy": 0.9822064057,
      "gpu_worker_seconds": 15.376,
      "generic": false,
      "note": "Same score on this continuous radio sample, but not a replacement for VAD on arbitrary audio."
    },
    {
      "variant": "force_three_speakers",
      "output_file": "forced-three-speakers-output.json",
      "text_wer": 0.2966751918,
      "speaker_attributed_wer": 0.3324808184,
      "matched_word_role_accuracy": 0.9501779359,
      "gpu_worker_seconds": 16.714,
      "generic": false
    },
    {
      "variant": "without_word_timestamps",
      "output_file": "no-word-timestamps-output.json",
      "text_wer": 0.2685421995,
      "speaker_attributed_wer": 0.4757033248,
      "matched_word_role_accuracy": 0.7226027397,
      "gpu_worker_seconds": 15.261,
      "generic": false,
      "note": "Plain text improved, but speaker attribution became substantially worse."
    }
  ],
  "speaker_uncertainty_review": {
    "words_with_uncertainty_fields": 350,
    "words_flagged": 8,
    "precision": 0.75,
    "recall": 0.75,
    "f1": 0.75,
    "note": "Useful as a human-review signal, not as an automatic speaker correction."
  },
  "limitations": [
    "This is one English, two-role, noisy radio sample and is not a general accuracy claim.",
    "Speaker roles were collapsed to HOUSTON and SPACECRAFT for evaluation.",
    "GPU worker seconds are wall-clock runtime proxies, not billing units.",
    "The reference transcript and speaker mapping include human editorial judgment."
  ]
}
