{
  "schema": 1,
  "report_id": "das-bba-pilot-20260915",
  "title": "DavidAgents Big Bench Audio engineering pilot",
  "publication_scope": "Self-run balanced pilot; additive post-run engineering rubric, not official Big Bench Audio accuracy or an independent certification.",
  "dataset": {
    "dataset": "ArtificialAnalysis/big_bench_audio",
    "revision": "af7bb9c25b015792583ca4da3ee27ec62cb79fe6",
    "license": "MIT",
    "dataset_url": "https://huggingface.co/datasets/ArtificialAnalysis/big_bench_audio/tree/af7bb9c25b015792583ca4da3ee27ec62cb79fe6",
    "metadata_sha256": "102a7de908f466f6c415042033e11dc8cfcd73e4687d3a00d0fe5b25b0517657",
    "selection_seed": "das-bba-pilot-v1",
    "selection_method": "sort SHA256(seed:id) independently per category; interleave categories",
    "selected_count": 8,
    "full_dataset_count": 1000,
    "per_category": 2,
    "scope": "balanced_pilot",
    "manifest_sha256": "dbe963a384577bc62856937f2644a524d025b9c2fde1f5aab8e7d9b95c5f381a"
  },
  "configuration": {
    "speech_model": "gpt-live-1",
    "voice": "marin",
    "support_provider": "Cerebras",
    "support_model": "qwen-3.8-27b",
    "support_runtime": "Claude Code",
    "support_roles": [
      "delegated reasoning",
      "proactive observer"
    ],
    "tools_enabled": false,
    "thinking_enabled": false,
    "harness_effort": "medium",
    "provider_reasoning": "none",
    "fallbacks_enabled": false,
    "support_completion_limit_per_role_per_case": 6,
    "observer_policy": {
      "interval_seconds": 8,
      "coalesce_seconds": 1.2,
      "max_coalesce_seconds": 4,
      "max_note_age_seconds": 20
    },
    "prompt_sha256": {
      "foreground": "92181fd1c183d38e5b3175b6a785f120b5cc60eaf6eb4ae802d49ae62f253ae3",
      "delegate": "27debc65603cdfc9f3b0587794f3061f3beb19dbacd30fdf780347524a4a7898",
      "observer": "58f8e8c8fc9436ca0ff6f99ee713c5fed5bcfc40c238826b306b68cded043b47"
    },
    "answer_seconds": 45,
    "settle_seconds": 3,
    "input": "Original dataset audio decoded to 24 kHz mono PCM. No original question text, case ID, category or ground truth supplied to speech or support models.",
    "transport": "Direct server audio; no phone carrier or client device in the measurement."
  },
  "review": {
    "reviewer": "Codex Astra artifact review",
    "basis": "Semantic review of hash-verified captured output through cached local small.en ASR: beam 1 plus a second beam 5 transcription of audible regions, cross-checked with stream captions. Both ASR passes used the same model. No independent human listening, paid judge or new voice runs.",
    "rubric_timing": "Defined during post-run review; not the original automatic scorer or the Artificial Analysis grader.",
    "rubric": "Additive post-run review: first and final answers match ground truth, no spoken contradiction, and original execution protocol completed",
    "early_answer_policy": "Two early answers are separately flagged and remain headline passes under this rubric. The rubric does not require waiting until the entire input has finished.",
    "original_grades_retained": true
  },
  "headline": {
    "passed": 6,
    "selected": 8,
    "percent": 75.0,
    "full_dataset_score": null,
    "official_leaderboard_score": false
  },
  "diagnostics": {
    "first_answer_correct": 7,
    "final_answer_correct": 8,
    "original_protocol_completed": 7,
    "contradictory_then_corrected": 1,
    "answers_before_input_finished": 2,
    "selected_denominator_for_every_count": 8,
    "note": "Final-answer correctness is diagnostic; it does not repair protocol failures or erase earlier wrong answers"
  },
  "original_automatic_summary": {
    "correct": 1,
    "incorrect": 1,
    "needs_review": 6,
    "missing": 0,
    "complete": false,
    "accuracy_percent": null,
    "attempted": 8,
    "finished": true
  },
  "timing": {
    "input_end_to_final_answer_completion_seconds": {
      "n": 8,
      "min": 0.634,
      "median": 1.9729,
      "max": 12.8283
    },
    "original_next_audio_after_input_ms": {
      "n": 8,
      "min": 1207.4191321339463,
      "median": 1451.070750500076,
      "max": 1935.1825840026145
    },
    "limitations": "Server playback only. Offline ASR alignment is approximate. Next audio may be filler or repeated speech and omits early speech; no PSTN, p95 or SLA claim."
  },
  "cost": {
    "gpt_live_provider_usage_seconds": 238.0,
    "gpt_live_usd_at_published_rate": 0.19833333333333333,
    "gpt_live_rate_usd_per_minute": 0.05,
    "support_usage_tokens": {
      "input_tokens": 25192,
      "output_tokens": 2365,
      "cache_read_input_tokens": 0,
      "cache_creation_input_tokens": 0
    },
    "support_completions": {
      "delegate": 4,
      "observer": 26
    },
    "support_usd_at_gateway_estimate": 0.029922,
    "combined_modeled_usd": 0.22825533333333334,
    "support_rate_basis": "Conservative gateway estimates $1/M input and $2/M output, not a provider invoice"
  },
  "cost_scope": "Modeled voice and support inference cost only. Excludes carrier, local ASR/hosting and other infrastructure costs. Not a provider invoice or a per-minute customer price.",
  "observer": {
    "reviews": 25,
    "notes": 3,
    "stale": 12,
    "duplicates": 0,
    "rejected": 3,
    "errors": 1
  },
  "cases": [
    {
      "id": 101,
      "category": "formal_fallacies",
      "official_answer": "invalid",
      "received_audio_transcription": "Alright, let me think about that. No. It's invalid. invalid.",
      "first_answer": "invalid",
      "first_answer_correct": true,
      "final_answer": "invalid",
      "final_answer_correct": true,
      "contradictory_spoken_answer": false,
      "original_protocol_success": true,
      "headline_pass": true,
      "original_error": null,
      "original_grade": {
        "verdict": "needs_review",
        "reason": "not_an_unambiguous_short_answer",
        "sha256": "2480aebe638bf3ef6e4676d53488e52e48957b610ab4f46410167f179f3f093f"
      },
      "review_rationale": "The received audio says No, clarifies invalid, and repeats invalid after the input ends. No contradictory answer; early answer is separately flagged.",
      "first_answer_before_input_finished": true,
      "input_end_to_first_answer_completion_seconds": -9.186,
      "input_end_to_final_answer_completion_seconds": 0.634,
      "timing_basis": "Secondary offline ASR word endpoints minus actual final input-frame endpoint; negative means early speech",
      "original_next_audio_after_input_ms": 1308.0592549126563,
      "provider_usage_seconds": 43.0,
      "evidence_hashes": {
        "record_sha256": "49052cdfddf60219374c86a4c1d67d5b04e6231633de9750f295b6163dbbe4a4",
        "output_pcm_sha256": "b653b9135cbf34b03925a82992a614c272a776c81ee72de25e4d342b81b49cee",
        "original_audio_sha256": "afb7c9eb1bc525c97047b71defee49fdbb0fef2460d02c4c1dce9fb6addd103e"
      }
    },
    {
      "id": 330,
      "category": "navigate",
      "official_answer": "Yes",
      "received_audio_transcription": "Yes.",
      "first_answer": "yes",
      "first_answer_correct": true,
      "final_answer": "yes",
      "final_answer_correct": true,
      "contradictory_spoken_answer": false,
      "original_protocol_success": true,
      "headline_pass": true,
      "original_error": null,
      "original_grade": {
        "verdict": "correct",
        "reason": "strict_spoken_answer",
        "sha256": "e0a4ea0fafbd6155444c32235c4bb8cba7b26b9fc2bad9e2638e272f3005d55c"
      },
      "review_rationale": "The received answer is Yes, matching the original answer key.",
      "first_answer_before_input_finished": false,
      "input_end_to_first_answer_completion_seconds": 1.3684,
      "input_end_to_final_answer_completion_seconds": 1.3684,
      "timing_basis": "Secondary offline ASR word endpoints minus actual final input-frame endpoint; negative means early speech",
      "original_next_audio_after_input_ms": 1529.6480870060627,
      "provider_usage_seconds": 24.0,
      "evidence_hashes": {
        "record_sha256": "d17a1581fcb297b28b50b7bac221872ecc333c7218827f4e272c2aac078a623c",
        "output_pcm_sha256": "e13b8a1abc63a67200af9249c141aa718b7074712f060ca5f8ad2127667091e4",
        "original_audio_sha256": "88152db87c14071b8dc8a17955141c0ab4b1f5649b5d04ee81f126517e3a0550"
      }
    },
    {
      "id": 562,
      "category": "object_counting",
      "official_answer": "9",
      "received_audio_transcription": "Hmm, you have nine fruits.",
      "first_answer": "9",
      "first_answer_correct": true,
      "final_answer": "9",
      "final_answer_correct": true,
      "contradictory_spoken_answer": false,
      "original_protocol_success": true,
      "headline_pass": true,
      "original_error": null,
      "original_grade": {
        "verdict": "needs_review",
        "reason": "not_an_unambiguous_short_answer",
        "sha256": "e9013b1c298e1bc0a57e3e9a93e2e7b4d787d4e49c12dd56a8ce99a482162f53"
      },
      "review_rationale": "The received answer says nine fruits, matching 9. Filler and a complete sentence caused the strict initial scorer to defer review.",
      "first_answer_before_input_finished": false,
      "input_end_to_first_answer_completion_seconds": 3.5806,
      "input_end_to_final_answer_completion_seconds": 3.5806,
      "timing_basis": "Secondary offline ASR word endpoints minus actual final input-frame endpoint; negative means early speech",
      "original_next_audio_after_input_ms": 1523.2973869610582,
      "provider_usage_seconds": 15.0,
      "evidence_hashes": {
        "record_sha256": "470d6c5a50ca35b18cafee4df8e6b3106c3a10885417bdd2014ead050e51077c",
        "output_pcm_sha256": "bf15f395dbb681e45b50c00fac0bce04836858235981a042c9afd52c858872fd",
        "original_audio_sha256": "e0775738c7f2c6424added7bb950857dcf348530049b7418258e4f83d8b7411e"
      }
    },
    {
      "id": 883,
      "category": "web_of_lies",
      "official_answer": "No",
      "received_audio_transcription": "Hmm. Yes, Inga tells the truth. Sorry, Inga does not tell the truth.",
      "first_answer": "yes",
      "first_answer_correct": false,
      "final_answer": "no",
      "final_answer_correct": true,
      "contradictory_spoken_answer": true,
      "original_protocol_success": true,
      "headline_pass": false,
      "original_error": null,
      "original_grade": {
        "verdict": "needs_review",
        "reason": "not_an_unambiguous_short_answer",
        "sha256": "4c67c9fa33dd90d2bb0863415e4e20e101056536a13ea5940be2046a731f3b51"
      },
      "review_rationale": "The received audio first states that Inga tells the truth, then explicitly corrects to not truthful. Final answer matches No, but the earlier contradiction fails the conservative headline. A corrective observer note is recorded in the same interval.",
      "first_answer_before_input_finished": false,
      "input_end_to_first_answer_completion_seconds": 4.2083,
      "input_end_to_final_answer_completion_seconds": 12.8283,
      "timing_basis": "Secondary offline ASR word endpoints minus actual final input-frame endpoint; negative means early speech",
      "original_next_audio_after_input_ms": 1935.1825840026145,
      "provider_usage_seconds": 29.0,
      "evidence_hashes": {
        "record_sha256": "9180aadd30a688b1953d0cfc52cac5db4b347e38536d3d7d84ef39fc17be1f12",
        "output_pcm_sha256": "77e9db915a16d9d1124b89d5b3740d0a8483a7d6d91ab5035c094fed28a6fdf4",
        "original_audio_sha256": "0bb650e25506c2eeb6e7bfb9f249c26c5d9b93aa85cd5dae4ba8e00e61fbe688"
      }
    },
    {
      "id": 49,
      "category": "formal_fallacies",
      "official_answer": "invalid",
      "received_audio_transcription": "Okay, checking whether the conclusion follows from the premises. invalid",
      "first_answer": "invalid",
      "first_answer_correct": true,
      "final_answer": "invalid",
      "final_answer_correct": true,
      "contradictory_spoken_answer": false,
      "original_protocol_success": true,
      "headline_pass": true,
      "original_error": null,
      "original_grade": {
        "verdict": "needs_review",
        "reason": "not_an_unambiguous_short_answer",
        "sha256": "914d5111e1c1bac01f8dc9cefa7f071205379fe1ad089d2b81f1e62f8afd02ac"
      },
      "review_rationale": "After an acknowledgement, the received answer is invalid, matching the original key. Acknowledgement text made the initial exact scorer defer review.",
      "first_answer_before_input_finished": false,
      "input_end_to_first_answer_completion_seconds": 5.8472,
      "input_end_to_final_answer_completion_seconds": 5.8472,
      "timing_basis": "Secondary offline ASR word endpoints minus actual final input-frame endpoint; negative means early speech",
      "original_next_audio_after_input_ms": 1502.5732899270956,
      "provider_usage_seconds": 34.0,
      "evidence_hashes": {
        "record_sha256": "df9687c9dfd3f5fc207311395d87fab3a79162bcad063ccfd8aaef9327a6e141",
        "output_pcm_sha256": "44d542fe85d46d0a88b389ef9d3ab1c9218205adeab565055f279af15610b8cf",
        "original_audio_sha256": "8e090bea698c061d61c5699a0418147149ea23df1873db05c0807646efd41611"
      }
    },
    {
      "id": 295,
      "category": "navigate",
      "official_answer": "Yes",
      "received_audio_transcription": "Yes. Yes.",
      "first_answer": "yes",
      "first_answer_correct": true,
      "final_answer": "yes",
      "final_answer_correct": true,
      "contradictory_spoken_answer": false,
      "original_protocol_success": true,
      "headline_pass": true,
      "original_error": null,
      "original_grade": {
        "verdict": "needs_review",
        "reason": "not_an_unambiguous_short_answer",
        "sha256": "5bdcdf87f26987cce0cb90e0e0816dc73d655c7debf03aedd31fe553540296ed"
      },
      "review_rationale": "The received audio repeats Yes without contradiction. The first Yes begins before the original input finishes, separately flagged.",
      "first_answer_before_input_finished": true,
      "input_end_to_first_answer_completion_seconds": -1.3464,
      "input_end_to_final_answer_completion_seconds": 0.9136,
      "timing_basis": "Secondary offline ASR word endpoints minus actual final input-frame endpoint; negative means early speech",
      "original_next_audio_after_input_ms": 1399.5682110730563,
      "provider_usage_seconds": 19.0,
      "evidence_hashes": {
        "record_sha256": "63f49a03911a8b01b2f8e76238ddba856c3e7f17a2ef36937444f37a7f3ce1ed",
        "output_pcm_sha256": "06ad476786bb9b1f5b2d0a162120fc52afcad9aa50679a7514045f34603887e1",
        "original_audio_sha256": "ed6b5b3867e533cb6d2f705e8902cc132ef3a7fe6be6858f59319a6785ec0254"
      }
    },
    {
      "id": 676,
      "category": "object_counting",
      "official_answer": "7",
      "received_audio_transcription": "You have seven objects.",
      "first_answer": "7",
      "first_answer_correct": true,
      "final_answer": "7",
      "final_answer_correct": true,
      "contradictory_spoken_answer": false,
      "original_protocol_success": false,
      "headline_pass": false,
      "original_error": "GptLiveError",
      "original_grade": {
        "verdict": "incorrect",
        "reason": "incomplete_or_error",
        "sha256": "c019107177cca663df1d2a8c4fee59e6389d4701a7280ea2f8070a8b8cd96ef6"
      },
      "review_rationale": "The received answer says seven objects, matching 7. Original GptLiveError remains a failed protocol: the completion predicate did not recognize this five-word number-word answer; the session hit its duration limit and a later input frame was rejected.",
      "first_answer_before_input_finished": false,
      "input_end_to_first_answer_completion_seconds": 1.9128,
      "input_end_to_final_answer_completion_seconds": 1.9128,
      "timing_basis": "Secondary offline ASR word endpoints minus actual final input-frame endpoint; negative means early speech",
      "original_next_audio_after_input_ms": 1207.4191321339463,
      "provider_usage_seconds": 51.0,
      "evidence_hashes": {
        "record_sha256": "ae5dd724c527f9d8c438b4df2172812c2f7e4fea55f95833a6c866b51c43c4eb",
        "output_pcm_sha256": "30f45d09b2360fa2a1c53bdc6902a3946279feed5eb354f3035ab3c70e2278dc",
        "original_audio_sha256": "b916c30c05b35278e548a73c286dd0d026888ff0b748b80adf687ce722ad4a72"
      }
    },
    {
      "id": 852,
      "category": "web_of_lies",
      "official_answer": "Yes",
      "received_audio_transcription": "Yes, Delbert tells the truth.",
      "first_answer": "yes",
      "first_answer_correct": true,
      "final_answer": "yes",
      "final_answer_correct": true,
      "contradictory_spoken_answer": false,
      "original_protocol_success": true,
      "headline_pass": true,
      "original_error": null,
      "original_grade": {
        "verdict": "needs_review",
        "reason": "not_an_unambiguous_short_answer",
        "sha256": "63ffeb70751727eae7356040f58e2e209aa231bc51e10c1224f539f5771ac766"
      },
      "review_rationale": "The received answer says Yes, Delbert tells the truth, matching Yes. An uncertain observer note was delivered afterwards but the spoken answer was not changed.",
      "first_answer_before_input_finished": false,
      "input_end_to_first_answer_completion_seconds": 2.033,
      "input_end_to_final_answer_completion_seconds": 2.033,
      "timing_basis": "Secondary offline ASR word endpoints minus actual final input-frame endpoint; negative means early speech",
      "original_next_audio_after_input_ms": 1382.8573499526833,
      "provider_usage_seconds": 23.0,
      "evidence_hashes": {
        "record_sha256": "2f6ccf6eb0b54b151e7cad0b20e8d76235cc8a96fce06a7b60ff56e24040bef9",
        "output_pcm_sha256": "9b43853ea0e27e789ee407005a85898e8e8263c5f444f1ad427c7378838dc1b6",
        "original_audio_sha256": "a9799b991db2d23a3180c3f8ee765b8ec1e3203de63da750c2c123978ffe07c4"
      }
    }
  ],
  "comparability": "Eight frozen cases out of 1,000 with a post-run engineering rubric. The sample and scoring differ from published Artificial Analysis and vendor benchmark results; this is unsuitable for direct leaderboard comparison.",
  "sources": [
    {
      "title": "Original Big Bench Audio dataset",
      "url": "https://huggingface.co/datasets/ArtificialAnalysis/big_bench_audio"
    },
    {
      "title": "Artificial Analysis speech benchmarking methodology",
      "url": "https://artificialanalysis.ai/methodology/speech-to-speech-benchmarking"
    },
    {
      "title": "OpenAI GPT-Live 1 model and pricing",
      "url": "https://developers.openai.com/api/docs/models/gpt-live-1"
    }
  ],
  "artifact_hashes": {
    "original_summary_sha256": "076d4296cdfaa28b5a8d46dbdb01a40a3418a29df36c418850179c1f1cc1aa2f",
    "decisions_sha256": "2137c41318fb84de294b6f42ec4959aa7dc41d6dd9c6b113eab7b0ac014e0272",
    "secondary_asr_sha256": "aa529bed79260bfe7b550b0a084dda5d801582341eb7024cefbb4a4cb0a85634"
  },
  "evidence_publication": "SHA-256 values identify retained source records and audio. Raw WAVs and execution records are not linked in this public release."
}
