music/benchmark/EAR_GROUND_TRUTH.json
{
"_doc": "THE EAR GROUND-TRUTH CONTROL -- every Josh ruling with a judged record, his own words, and the battery's reading of it under the old math and the new one.",
"_honest_tier": "A CONTROL, NOT A SCORE. It proves the reading does not contradict a ruling. It never claims the reading is right about anything he has not ruled on.",
"version": "ear-ground-truth/1.0.0",
"fixtures": {
"schema": "ear-ground-truth/1.0.0",
"version": "ear-ground-truth/1.0.0",
"rulings": [
{
"track_id": "BENCH_R1_ELEVENLABS_EDIT_V2_1",
"grade": "up",
"evidence_tier": "QUOTED",
"verbatim": "Variation 2 are my favorite on all 4, save these... they're the start of our new chapter 2 sound track.",
"nuance_verbatim": null,
"cite": "docs/review_candidates.json :: candidates[] where id == 'CH02_SOUNDTRACK_T1'",
"cite_file": "docs/review_candidates.json",
"quote_check": "VACUOUS BY CONSTRUCTION, AND SKIPPED RATHER THAN COUNTED. This row's verbatim was READ OUT OF docs/review_candidates.json a microsecond earlier, so grepping the same file for it can only ever succeed -- it verifies a string against itself and proves nothing about drift. The teeth in G1 belong to the DECLARED_DOWN probes alone, which cite a DIFFERENT file than the one they are written in and therefore can actually fail. Counting this row as a passed check would inflate the arm's denominator with checks that cannot fail, which is the same defect class as a search that cannot match reporting zero.",
"display_name": "Zither and brass, variation two",
"cue_class": "EXPLORATION",
"sha256": "b6f96cdb5bb85257e9e453a36218b682861aab5d46e35cffc3eaa59710db94c3",
"record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_EDIT_V2_1.json",
"reason_axes": [],
"reason_note": "a graded-UP ruling names no defect; the reason arm applies to rejections only",
"emitted_from_register": true
},
{
"track_id": "BENCH_R1_ELEVENLABS_EDIT_V2_2",
"grade": "up",
"evidence_tier": "QUOTED",
"verbatim": "Variation 2 are my favorite on all 4, save these... they're the start of our new chapter 2 sound track.",
"nuance_verbatim": null,
"cite": "docs/review_candidates.json :: candidates[] where id == 'CH02_SOUNDTRACK_T2'",
"cite_file": "docs/review_candidates.json",
"quote_check": "VACUOUS BY CONSTRUCTION, AND SKIPPED RATHER THAN COUNTED. This row's verbatim was READ OUT OF docs/review_candidates.json a microsecond earlier, so grepping the same file for it can only ever succeed -- it verifies a string against itself and proves nothing about drift. The teeth in G1 belong to the DECLARED_DOWN probes alone, which cite a DIFFERENT file than the one they are written in and therefore can actually fail. Counting this row as a passed check would inflate the arm's denominator with checks that cannot fail, which is the same defect class as a search that cannot match reporting zero.",
"display_name": "Surging groove, variation two",
"cue_class": "EXPLORATION",
"sha256": "0edb02ea96e9c70d2e5dc260cdf4bbc2960bad334e8471a4e979dc95df9d54ee",
"record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_EDIT_V2_2.json",
"reason_axes": [],
"reason_note": "a graded-UP ruling names no defect; the reason arm applies to rejections only",
"emitted_from_register": true
},
{
"track_id": "BENCH_R1_ELEVENLABS_EDIT_V2_3",
"grade": "up",
"evidence_tier": "QUOTED",
"verbatim": "Variation 2 are my favorite on all 4, save these... they're the start of our new chapter 2 sound track.",
"nuance_verbatim": null,
"cite": "docs/review_candidates.json :: candidates[] where id == 'CH02_SOUNDTRACK_T3'",
"cite_file": "docs/review_candidates.json",
"quote_check": "VACUOUS BY CONSTRUCTION, AND SKIPPED RATHER THAN COUNTED. This row's verbatim was READ OUT OF docs/review_candidates.json a microsecond earlier, so grepping the same file for it can only ever succeed -- it verifies a string against itself and proves nothing about drift. The teeth in G1 belong to the DECLARED_DOWN probes alone, which cite a DIFFERENT file than the one they are written in and therefore can actually fail. Counting this row as a passed check would inflate the arm's denominator with checks that cannot fail, which is the same defect class as a search that cannot match reporting zero.",
"display_name": "Bridge groove, variation two",
"cue_class": "EXPLORATION",
"sha256": "a60101ffe195603f60c6750d3ea97acade2dd0715952e87c838d1fd5a84a359a",
"record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_EDIT_V2_3.json",
"reason_axes": [],
"reason_note": "a graded-UP ruling names no defect; the reason arm applies to rejections only",
"emitted_from_register": true
},
{
"track_id": "BENCH_R1_ELEVENLABS_EDIT_V2_4",
"grade": "up",
"evidence_tier": "QUOTED",
"verbatim": "Variation 2 are my favorite on all 4, save these... they're the start of our new chapter 2 sound track.",
"nuance_verbatim": null,
"cite": "docs/review_candidates.json :: candidates[] where id == 'CH02_SOUNDTRACK_T4'",
"cite_file": "docs/review_candidates.json",
"quote_check": "VACUOUS BY CONSTRUCTION, AND SKIPPED RATHER THAN COUNTED. This row's verbatim was READ OUT OF docs/review_candidates.json a microsecond earlier, so grepping the same file for it can only ever succeed -- it verifies a string against itself and proves nothing about drift. The teeth in G1 belong to the DECLARED_DOWN probes alone, which cite a DIFFERENT file than the one they are written in and therefore can actually fail. Counting this row as a passed check would inflate the arm's denominator with checks that cannot fail, which is the same defect class as a search that cannot match reporting zero.",
"display_name": "Sasando festival, variation two",
"cue_class": "EXPLORATION",
"sha256": "a8a2f7b6a732073a48f311072d8ecb2e47997963a200e55dbfc7db76f9fd6607",
"record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_EDIT_V2_4.json",
"reason_axes": [],
"reason_note": "a graded-UP ruling names no defect; the reason arm applies to rejections only",
"emitted_from_register": true
},
{
"track_id": "BENCH_R1_ELEVENLABS_EDIT_SERENE_V2",
"grade": "up",
"evidence_tier": "QUOTED",
"verbatim": "You can use both the serene ones, variation 2 for hearth of Flores, push them to website",
"nuance_verbatim": "The serene pair are good and I do like them but very slow and kind of boring although that's how serene and peaceful is I guess",
"cite": "docs/review_candidates.json :: candidates[] where id == 'CH02_SOUNDTRACK_T5'",
"cite_file": "docs/review_candidates.json",
"quote_check": "VACUOUS BY CONSTRUCTION, AND SKIPPED RATHER THAN COUNTED. This row's verbatim was READ OUT OF docs/review_candidates.json a microsecond earlier, so grepping the same file for it can only ever succeed -- it verifies a string against itself and proves nothing about drift. The teeth in G1 belong to the DECLARED_DOWN probes alone, which cite a DIFFERENT file than the one they are written in and therefore can actually fail. Counting this row as a passed check would inflate the arm's denominator with checks that cannot fail, which is the same defect class as a search that cannot match reporting zero.",
"display_name": "Hearth of Flores, variation two",
"cue_class": "SERENE",
"sha256": "2a58d047ba7df78ee7539bd140f98e8277c593354aebc8922548e375d1b0e2e1",
"record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_EDIT_SERENE_V2.json",
"reason_axes": [],
"reason_note": "a graded-UP ruling names no defect; the reason arm applies to rejections only",
"emitted_from_register": true
},
{
"track_id": "BENCH_R1_ELEVENLABS_TP_S1_v1_2",
"grade": "up",
"evidence_tier": "QUOTED",
"verbatim": "You can use both the serene ones, variation 2 for hearth of Flores, push them to website",
"nuance_verbatim": "The serene pair are good and I do like them but very slow and kind of boring although that's how serene and peaceful is I guess",
"cite": "docs/review_candidates.json :: candidates[] where id == 'CH02_SOUNDTRACK_T6'",
"cite_file": "docs/review_candidates.json",
"quote_check": "VACUOUS BY CONSTRUCTION, AND SKIPPED RATHER THAN COUNTED. This row's verbatim was READ OUT OF docs/review_candidates.json a microsecond earlier, so grepping the same file for it can only ever succeed -- it verifies a string against itself and proves nothing about drift. The teeth in G1 belong to the DECLARED_DOWN probes alone, which cite a DIFFERENT file than the one they are written in and therefore can actually fail. Counting this row as a passed check would inflate the arm's denominator with checks that cannot fail, which is the same defect class as a search that cannot match reporting zero.",
"display_name": "Highland Hearth",
"cue_class": "SERENE",
"sha256": "20f024f70ccb3f8445f6d6d626d6712fb60c235ac48e68a980ff844086d496ca",
"record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_S1_v1_2.json",
"reason_axes": [],
"reason_note": "a graded-UP ruling names no defect; the reason arm applies to rejections only",
"emitted_from_register": true
},
{
"track_id": "BENCH_R1_ELEVENLABS_EDIT_BATTLE_V2",
"grade": "up",
"evidence_tier": "QUOTED",
"verbatim": "I like this caci one attached and think it's good to push",
"nuance_verbatim": null,
"cite": "docs/review_candidates.json :: candidates[] where id == 'CH02_SOUNDTRACK_T7'",
"cite_file": "docs/review_candidates.json",
"quote_check": "VACUOUS BY CONSTRUCTION, AND SKIPPED RATHER THAN COUNTED. This row's verbatim was READ OUT OF docs/review_candidates.json a microsecond earlier, so grepping the same file for it can only ever succeed -- it verifies a string against itself and proves nothing about drift. The teeth in G1 belong to the DECLARED_DOWN probes alone, which cite a DIFFERENT file than the one they are written in and therefore can actually fail. Counting this row as a passed check would inflate the arm's denominator with checks that cannot fail, which is the same defect class as a search that cannot match reporting zero.",
"display_name": "Caci highlands duel, variation two",
"cue_class": "BATTLE",
"sha256": "0ac2c601af7bb92de000e8ed76104dee9bed26a7734e8bd90c6c3379a2fe3494",
"record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_EDIT_BATTLE_V2.json",
"reason_axes": [],
"reason_note": "a graded-UP ruling names no defect; the reason arm applies to rejections only",
"emitted_from_register": true
},
{
"track_id": "BENCH_R1_ELEVENLABS_TP_B2_HYBRID_1",
"grade": "up",
"evidence_tier": "QUOTED",
"verbatim": "No complaints on highland fury. I think that one's pretty good for a battle song",
"nuance_verbatim": null,
"cite": "docs/review_candidates.json :: candidates[] where id == 'CH02_SOUNDTRACK_T8'",
"cite_file": "docs/review_candidates.json",
"quote_check": "VACUOUS BY CONSTRUCTION, AND SKIPPED RATHER THAN COUNTED. This row's verbatim was READ OUT OF docs/review_candidates.json a microsecond earlier, so grepping the same file for it can only ever succeed -- it verifies a string against itself and proves nothing about drift. The teeth in G1 belong to the DECLARED_DOWN probes alone, which cite a DIFFERENT file than the one they are written in and therefore can actually fail. Counting this row as a passed check would inflate the arm's denominator with checks that cannot fail, which is the same defect class as a search that cannot match reporting zero.",
"display_name": "Highland fury",
"cue_class": "BATTLE",
"sha256": "623373c7197f834a9eedad4cdbd0bd4574282e6a8c1f0991b5e61ba7bb10066e",
"record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_B2_HYBRID_1.json",
"reason_axes": [],
"reason_note": "a graded-UP ruling names no defect; the reason arm applies to rejections only",
"emitted_from_register": true
},
{
"track_id": "BENCH_R1_ELEVENLABS_EDIT_T2_MELODY_1",
"grade": "down",
"evidence_tier": "QUOTED",
"verbatim": "I'm thinking about going back to the first one we said we loved. These new ones aren't hitting.",
"quote_probe": "These new ones aren't hitting",
"cite": "docs/spine/DECISIONS_PENDING_JOSH.md (T2 SURGING GROOVE block)",
"cite_file": "docs/spine/DECISIONS_PENDING_JOSH.md",
"reason_axes": [],
"reason_note": "A PREFERENCE, not a named defect: he restored the earlier take without naming an axis. The reason-agreement arm therefore SKIPS this row and records the skip -- inventing an axis for it would be the curve fit WO-12 forbids. The ordering arms still bind, and this is the pair the order was minted on.",
"pair_with": "BENCH_R1_ELEVENLABS_EDIT_V2_2",
"pair_note": "the explicit kept-vs-rejected pair; the kept take is the restored 0edb02ea and the rejected one is the melody rework of it",
"record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_EDIT_T2_MELODY_1.json",
"emitted_from_register": false,
"quote_check": "PROBED: \"These new ones aren't hitting\" is still findable in docs/spine/DECISIONS_PENDING_JOSH.md"
},
{
"track_id": "BENCH_R1_INHOUSE_BASELINE_R4",
"grade": "down",
"evidence_tier": "QUOTED",
"verbatim": "Round 4 is even worse than round 2 and is not shippable. The melody sucks. I hate the instruments you are using. I hate how slow the melody is and there is no variation or changes or movement throughout the songs. Everything sucks.",
"quote_probe": "no variation or changes or movement",
"cite": "docs/review_candidates.json (PASS7_FLORES_* rows, graded 2026-08-09)",
"cite_file": "docs/review_candidates.json",
"reason_axes": [
"structural_yield",
"dev_ops_per_min",
"dev_kinds"
],
"reason_note": "\"no variation or changes or movement\" is the motivic-development and turn-pacing clause in his own words -- the one down ruling on file that names a defect an instrument can see.",
"record": "build/audio/benchmark/records/BENCH_R1_INHOUSE_BASELINE_R4.json",
"emitted_from_register": false,
"quote_check": "PROBED: 'no variation or changes or movement' is still findable in docs/review_candidates.json"
},
{
"track_id": "BENCH_R1_ELEVENLABS_TP_E1_v5_1",
"grade": "down",
"evidence_tier": "ATTRIBUTED",
"verbatim": "whole thing not good",
"quote_probe": "whole thing not good",
"cite": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_E1_v5_1.json :: measurements.floor_battery.results.cycle_law.title -- NOT the record's top-level `title`, which is None on every row",
"cite_file": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_E1_v5_1.json",
"reason_axes": [],
"reason_note": "a lane's paraphrase inside a record title, not a quoted sentence.",
"record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_E1_v5_1.json",
"emitted_from_register": false,
"quote_check": "PROBED: 'whole thing not good' is still findable in build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_E1_v5_1.json"
},
{
"track_id": "BENCH_R1_ELEVENLABS_TP_E1_v5_2",
"grade": "down",
"evidence_tier": "ATTRIBUTED",
"verbatim": "whole thing not good",
"quote_probe": "whole thing not good",
"cite": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_E1_v5_2.json :: measurements.floor_battery.results.cycle_law.title -- NOT the record's top-level `title`, which is None on every row",
"cite_file": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_E1_v5_2.json",
"reason_axes": [],
"reason_note": "a lane's paraphrase inside a record title, not a quoted sentence.",
"record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_E1_v5_2.json",
"emitted_from_register": false,
"quote_check": "PROBED: 'whole thing not good' is still findable in build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_E1_v5_2.json"
},
{
"track_id": "BENCH_R1_ELEVENLABS_TP_B1_v1_2",
"grade": "down",
"evidence_tier": "ATTRIBUTED",
"verbatim": "repetitive, failed",
"quote_probe": "repetitive, failed",
"cite": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_B1_v1_2.json :: measurements.floor_battery.results.cycle_law.title -- NOT the record's top-level `title`, which is None on every row",
"cite_file": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_B1_v1_2.json",
"reason_axes": [
"structural_yield"
],
"reason_note": "\"repetitive\" is the turn-pacing clause, at the attributed tier only.",
"record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_B1_v1_2.json",
"emitted_from_register": false,
"quote_check": "PROBED: 'repetitive, failed' is still findable in build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_B1_v1_2.json"
}
],
"owed": [
{
"ruling": "The Komodo Tyrant -- \"a minute of literally nothing going on and the dead silence on the bridge sucks\"",
"cite": "docs/spine/DECISIONS_PENDING_JOSH.md (2026-08-09/10 late sitting)",
"why_it_cannot_enter": "no benchmark record on disk",
"how_to_close": "download the WAV, run benchmark_judge --audio, then ear_axes dead_space -- this is the exact ruling the interior-dead-space finding was built for, and it is the only ruling on file that names that defect"
},
{
"ruling": "The Komodo's Fury v2 -- \"the chanting doesn't even take a breath and again it's not complex at all just a bunch of banging\"",
"cite": "docs/spine/DECISIONS_PENDING_JOSH.md (2026-08-09/10 late sitting)",
"why_it_cannot_enter": "no benchmark record on disk AND no instrument reads vocal rest structure -- even with a record, the named defect is unmeasured",
"how_to_close": "a vocal-rest instrument would be new work with its own theory ground; it is not in this build and is not claimed"
},
{
"ruling": "\"Highlands of Flores\" 4:00 hybrid -- \"from the build to the climax it is literally all the same thing\"",
"cite": "docs/spine/DECISIONS_PENDING_JOSH.md (2026-08-09/10 late sitting)",
"why_it_cannot_enter": "IDENTITY UNRESOLVED. TP_E1_v6_2 is a 4:00 'Highlands of Flores' and would fit, but the same document grades the v6 pair UP and calls the retired take a HYBRID. Binding them would be a guess, and a guessed fixture row is worse than a missing one",
"how_to_close": "the director names the sha256 of the retired take, or the row stays out"
},
{
"ruling": "Rounds 1-3 -- \"a bunch of noise... these main melodies really suck... Never any harmonies added on\"; the round-3 wall read 57 of 57",
"cite": "docs/review_candidates.json (PASS5_FLORES_* rows, graded 2026-08-08)",
"why_it_cannot_enter": "no benchmark records; those rounds predate the judge",
"how_to_close": "judge the round-3 renders if they are still on the box. This is the FOUNDING case of the inversion this control exists to catch"
}
],
"available": true,
"reason": "",
"quote_verification": {
"probes_with_teeth": 5,
"vacuous_and_skipped": 8,
"why": "VACUOUS BY CONSTRUCTION, AND SKIPPED RATHER THAN COUNTED. This row's verbatim was READ OUT OF docs/review_candidates.json a microsecond earlier, so grepping the same file for it can only ever succeed -- it verifies a string against itself and proves nothing about drift. The teeth in G1 belong to the DECLARED_DOWN probes alone, which cite a DIFFERENT file than the one they are written in and therefore can actually fail. Counting this row as a passed check would inflate the arm's denominator with checks that cannot fail, which is the same defect class as a search that cannot match reporting zero.",
"what_G1_actually_proves": "that 5 DECLARED_DOWN quote(s) are still findable in the OTHER file each one cites. It proves nothing about the 8 register-emitted row(s), whose text came out of the file the probe would search."
},
"harvested_down_from_register": [],
"counts": {
"up": 8,
"down": 5,
"quoted": 10,
"attributed": 3,
"harvested_down_from_register": 0,
"owed_no_record": 4
}
},
"readings": {
"old_wall_fraction": {
"BENCH_R1_ELEVENLABS_EDIT_V2_1": 0.7778,
"BENCH_R1_ELEVENLABS_EDIT_V2_2": 0.25,
"BENCH_R1_ELEVENLABS_EDIT_V2_3": 1.0,
"BENCH_R1_ELEVENLABS_EDIT_V2_4": 0.7778,
"BENCH_R1_ELEVENLABS_EDIT_SERENE_V2": 0.75,
"BENCH_R1_ELEVENLABS_TP_S1_v1_2": 0.8889,
"BENCH_R1_ELEVENLABS_EDIT_BATTLE_V2": 0.625,
"BENCH_R1_ELEVENLABS_TP_B2_HYBRID_1": 0.625,
"BENCH_R1_ELEVENLABS_EDIT_T2_MELODY_1": 0.7778,
"BENCH_R1_INHOUSE_BASELINE_R4": 0.6667,
"BENCH_R1_ELEVENLABS_TP_E1_v5_1": 0.5556,
"BENCH_R1_ELEVENLABS_TP_E1_v5_2": 0.5556,
"BENCH_R1_ELEVENLABS_TP_B1_v1_2": 0.5714
},
"character_reading": {
"BENCH_R1_ELEVENLABS_EDIT_V2_1": 0.6098,
"BENCH_R1_ELEVENLABS_EDIT_V2_2": 0.7604,
"BENCH_R1_ELEVENLABS_EDIT_V2_3": 0.4896,
"BENCH_R1_ELEVENLABS_EDIT_V2_4": 0.3325,
"BENCH_R1_ELEVENLABS_EDIT_SERENE_V2": 0.7284,
"BENCH_R1_ELEVENLABS_TP_S1_v1_2": 0.5088,
"BENCH_R1_ELEVENLABS_EDIT_BATTLE_V2": 0.512,
"BENCH_R1_ELEVENLABS_TP_B2_HYBRID_1": 0.5537,
"BENCH_R1_ELEVENLABS_EDIT_T2_MELODY_1": 0.4111,
"BENCH_R1_INHOUSE_BASELINE_R4": 0.4335,
"BENCH_R1_ELEVENLABS_TP_E1_v5_1": 0.4215,
"BENCH_R1_ELEVENLABS_TP_E1_v5_2": 0.4215,
"BENCH_R1_ELEVENLABS_TP_B1_v1_2": 0.2564
},
"weak_axes_per_track": {
"BENCH_R1_ELEVENLABS_EDIT_V2_1": [
"orchestrational_dialogue"
],
"BENCH_R1_ELEVENLABS_EDIT_V2_2": [
"boundaries_per_min",
"loudness_range_lu"
],
"BENCH_R1_ELEVENLABS_EDIT_V2_3": [
"dev_kinds",
"dev_ops_per_min",
"orchestrational_dialogue"
],
"BENCH_R1_ELEVENLABS_EDIT_V2_4": [
"dev_kinds",
"dev_ops_per_min",
"orchestrational_dialogue",
"structural_yield"
],
"BENCH_R1_ELEVENLABS_EDIT_SERENE_V2": [
"counter_melody_presence"
],
"BENCH_R1_ELEVENLABS_TP_S1_v1_2": [
"dev_ops_per_min",
"loudness_range_lu"
],
"BENCH_R1_ELEVENLABS_EDIT_BATTLE_V2": [
"counter_melody_presence",
"dynamic_range_db"
],
"BENCH_R1_ELEVENLABS_TP_B2_HYBRID_1": [
"counter_melody_presence",
"dynamic_range_db",
"loudness_range_lu"
],
"BENCH_R1_ELEVENLABS_EDIT_T2_MELODY_1": [
"dynamic_range_db",
"loudness_range_lu"
],
"BENCH_R1_INHOUSE_BASELINE_R4": [
"boundaries_per_min",
"orchestrational_dialogue",
"structural_yield"
],
"BENCH_R1_ELEVENLABS_TP_E1_v5_1": [
"boundaries_per_min",
"structural_yield"
],
"BENCH_R1_ELEVENLABS_TP_E1_v5_2": [
"boundaries_per_min",
"structural_yield"
],
"BENCH_R1_ELEVENLABS_TP_B1_v1_2": [
"counter_melody_presence",
"dev_ops_per_min",
"dynamic_range_db"
]
},
"measurable_axes_per_track": {
"BENCH_R1_ELEVENLABS_EDIT_V2_1": [
"boundaries_per_min",
"counter_melody_presence",
"dev_kinds",
"dev_ops_per_min",
"dynamic_range_db",
"loudness_range_lu",
"orchestrational_dialogue",
"structural_yield",
"syncopation_audio"
],
"BENCH_R1_ELEVENLABS_EDIT_V2_2": [
"boundaries_per_min",
"counter_melody_presence",
"dev_kinds",
"dev_ops_per_min",
"dynamic_range_db",
"loudness_range_lu",
"orchestrational_dialogue",
"structural_yield",
"syncopation_audio"
],
"BENCH_R1_ELEVENLABS_EDIT_V2_3": [
"boundaries_per_min",
"counter_melody_presence",
"dev_kinds",
"dev_ops_per_min",
"dynamic_range_db",
"loudness_range_lu",
"orchestrational_dialogue",
"structural_yield",
"syncopation_audio"
],
"BENCH_R1_ELEVENLABS_EDIT_V2_4": [
"boundaries_per_min",
"counter_melody_presence",
"dev_kinds",
"dev_ops_per_min",
"dynamic_range_db",
"loudness_range_lu",
"orchestrational_dialogue",
"structural_yield",
"syncopation_audio"
],
"BENCH_R1_ELEVENLABS_EDIT_SERENE_V2": [
"boundaries_per_min",
"counter_melody_presence",
"dev_kinds",
"dev_ops_per_min",
"dynamic_range_db",
"loudness_range_lu",
"orchestrational_dialogue",
"structural_yield",
"syncopation_audio"
],
"BENCH_R1_ELEVENLABS_TP_S1_v1_2": [
"boundaries_per_min",
"counter_melody_presence",
"dev_kinds",
"dev_ops_per_min",
"dynamic_range_db",
"loudness_range_lu",
"orchestrational_dialogue",
"structural_yield",
"syncopation_audio"
],
"BENCH_R1_ELEVENLABS_EDIT_BATTLE_V2": [
"boundaries_per_min",
"counter_melody_presence",
"dev_kinds",
"dev_ops_per_min",
"dynamic_range_db",
"loudness_range_lu",
"orchestrational_dialogue",
"structural_yield",
"syncopation_audio"
],
"BENCH_R1_ELEVENLABS_TP_B2_HYBRID_1": [
"boundaries_per_min",
"counter_melody_presence",
"dev_kinds",
"dev_ops_per_min",
"dynamic_range_db",
"loudness_range_lu",
"orchestrational_dialogue",
"structural_yield",
"syncopation_audio"
],
"BENCH_R1_ELEVENLABS_EDIT_T2_MELODY_1": [
"boundaries_per_min",
"counter_melody_presence",
"dev_kinds",
"dev_ops_per_min",
"dynamic_range_db",
"loudness_range_lu",
"orchestrational_dialogue",
"structural_yield",
"syncopation_audio"
],
"BENCH_R1_INHOUSE_BASELINE_R4": [
"boundaries_per_min",
"counter_melody_presence",
"dev_kinds",
"dev_ops_per_min",
"dynamic_range_db",
"loudness_range_lu",
"orchestrational_dialogue",
"structural_yield",
"syncopation_audio"
],
"BENCH_R1_ELEVENLABS_TP_E1_v5_1": [
"boundaries_per_min",
"counter_melody_presence",
"dev_kinds",
"dev_ops_per_min",
"dynamic_range_db",
"loudness_range_lu",
"orchestrational_dialogue",
"structural_yield",
"syncopation_audio"
],
"BENCH_R1_ELEVENLABS_TP_E1_v5_2": [
"boundaries_per_min",
"counter_melody_presence",
"dev_kinds",
"dev_ops_per_min",
"dynamic_range_db",
"loudness_range_lu",
"orchestrational_dialogue",
"structural_yield",
"syncopation_audio"
],
"BENCH_R1_ELEVENLABS_TP_B1_v1_2": [
"counter_melody_presence",
"dev_kinds",
"dev_ops_per_min",
"dynamic_range_db",
"loudness_range_lu",
"orchestrational_dialogue",
"syncopation_audio"
]
}
},
"arms": {
"old_math_MUST_FAIL": {
"label": "OLD",
"green": false,
"arms": [
{
"arm": "A1 no graded-down row reads above the MEDIAN graded-up row",
"green": false,
"detail": "median graded-up reading 0.7639; violations ['ABS_EDIT_T2_MELODY_1']",
"why_the_median": "four of the eight graded-up rows share ONE sentence, so they are one grading event, not four independent rulings. A strict every-up-above-every-down arm would be a claim the evidence cannot carry; the median arm is what 8 rows over ~4 events supports, and the exceptions are printed.",
"strict_variant": {
"every_up_above_every_down": false,
"exceptions": [
"ABS_EDIT_T2_MELODY_1 reads above ABS_EDIT_V2_1",
"ABS_EDIT_T2_MELODY_1 reads above ABS_EDIT_V2_2",
"_BASELINE_R4 reads above ABS_EDIT_V2_2",
"ABS_TP_E1_v5_1 reads above ABS_EDIT_V2_2",
"ABS_TP_E1_v5_2 reads above ABS_EDIT_V2_2",
"ABS_TP_B1_v1_2 reads above ABS_EDIT_V2_2",
"ABS_EDIT_T2_MELODY_1 reads above ABS_EDIT_V2_4",
"ABS_EDIT_T2_MELODY_1 reads above ABS_EDIT_SERENE_V2",
"ABS_EDIT_T2_MELODY_1 reads above ABS_EDIT_BATTLE_V2",
"_BASELINE_R4 reads above ABS_EDIT_BATTLE_V2",
"ABS_EDIT_T2_MELODY_1 reads above ABS_TP_B2_HYBRID_1",
"_BASELINE_R4 reads above ABS_TP_B2_HYBRID_1"
],
"status": "REPORTED, NOT ARMED -- see why_the_median"
},
"auc": 0.725
},
{
"arm": "A2 on every declared kept-vs-rejected pair, the kept take reads above the rejected",
"green": false,
"detail": "1 declared pair(s); ABS_EDIT_V2_2 (0.25) does NOT read above ABS_EDIT_T2_MELODY_1 (0.7778)",
"pairs": [
"ABS_EDIT_V2_2 over ABS_EDIT_T2_MELODY_1"
]
},
{
"arm": "A3 a rejected track's weak axes include one that AGREES with his stated reason",
"green": false,
"detail": "agreed: none; contradicted: [\"_BASELINE_R4: wanted one of ['structural_yield', 'dev_ops_per_min', 'dev_kinds'], the reading flagged ['dynamic_arc', 'groove_instrument', 'melodic_intelligence']\", \"ABS_TP_B1_v1_2: wanted one of ['structural_yield'], the reading flagged ['cycle_law', 'groove_instrument', 'melody_bands']\"]; unmeasurable: none",
"skipped": [
"ABS_EDIT_T2_MELODY_1: A PREFERENCE, not a named defect: he restored the earlier take without naming an axis. The reason-agreement arm therefore SKIPS this row and records the skip -- inventing an axis for it would be the curve fit WO-12 forbids. The ordering arms still bind, and this is the pair the order was minted on.",
"ABS_TP_E1_v5_1: a lane's paraphrase inside a record title, not a quoted sentence.",
"ABS_TP_E1_v5_2: a lane's paraphrase inside a record title, not a quoted sentence."
],
"vacuity_guard": "RED because no down ruling could be checked at all",
"unmeasurable_note": "TP_B1_v1_2 is the case worth reading twice. Its attributed reason is 'repetitive', and `structural_yield` -- the axis for exactly that -- is UNAVAILABLE on it, because the card carries only three novelty boundaries and floor_instruments refuses the ratio below four (its own C4 thin-card control). The refusal is correct and its side effect is not: the rows most likely to be structurally repetitive are the rows the yield axis declines to score. That is a hole in the instrument, recorded here rather than papered over by loosening the arm.",
"skip_note": "A ruling that names no defect ('these new ones aren't hitting', 'whole thing not good') cannot have a reason arm without one being invented for it. The skips are the honest denominator: only ONE down ruling with a record names a defect an instrument can see, and the two that name defects most precisely (the Komodo pair) have no record at all."
}
],
"n_up": 8,
"n_down": 5
},
"new_math": {
"label": "NEW",
"green": true,
"arms": [
{
"arm": "A1 no graded-down row reads above the MEDIAN graded-up row",
"green": true,
"detail": "median graded-up reading 0.5329; violations none",
"why_the_median": "four of the eight graded-up rows share ONE sentence, so they are one grading event, not four independent rulings. A strict every-up-above-every-down arm would be a claim the evidence cannot carry; the median arm is what 8 rows over ~4 events supports, and the exceptions are printed.",
"strict_variant": {
"every_up_above_every_down": false,
"exceptions": [
"ABS_EDIT_T2_MELODY_1 reads above ABS_EDIT_V2_4",
"_BASELINE_R4 reads above ABS_EDIT_V2_4",
"ABS_TP_E1_v5_1 reads above ABS_EDIT_V2_4",
"ABS_TP_E1_v5_2 reads above ABS_EDIT_V2_4"
],
"status": "REPORTED, NOT ARMED -- see why_the_median"
},
"auc": 0.9
},
{
"arm": "A2 on every declared kept-vs-rejected pair, the kept take reads above the rejected",
"green": true,
"detail": "1 declared pair(s); all ordered as Josh ordered them",
"pairs": [
"ABS_EDIT_V2_2 over ABS_EDIT_T2_MELODY_1"
]
},
{
"arm": "A3 a rejected track's weak axes include one that AGREES with his stated reason",
"green": true,
"detail": "agreed: [\"_BASELINE_R4: ['structural_yield']\"]; contradicted: none; unmeasurable: [\"ABS_TP_B1_v1_2: the reading carries none of ['structural_yield'] on this row (available: ['counter_melody_presence', 'dev_kinds', 'dev_ops_per_min', 'dynamic_range_db', 'loudness_range_lu', 'orchestrational_dialogue', 'syncopation_audio'])\"]",
"skipped": [
"ABS_EDIT_T2_MELODY_1: A PREFERENCE, not a named defect: he restored the earlier take without naming an axis. The reason-agreement arm therefore SKIPS this row and records the skip -- inventing an axis for it would be the curve fit WO-12 forbids. The ordering arms still bind, and this is the pair the order was minted on.",
"ABS_TP_E1_v5_1: a lane's paraphrase inside a record title, not a quoted sentence.",
"ABS_TP_E1_v5_2: a lane's paraphrase inside a record title, not a quoted sentence."
],
"vacuity_guard": "1 ruling(s) actually checked",
"unmeasurable_note": "TP_B1_v1_2 is the case worth reading twice. Its attributed reason is 'repetitive', and `structural_yield` -- the axis for exactly that -- is UNAVAILABLE on it, because the card carries only three novelty boundaries and floor_instruments refuses the ratio below four (its own C4 thin-card control). The refusal is correct and its side effect is not: the rows most likely to be structurally repetitive are the rows the yield axis declines to score. That is a hole in the instrument, recorded here rather than papered over by loosening the arm.",
"skip_note": "A ruling that names no defect ('these new ones aren't hitting', 'whole thing not good') cannot have a reason arm without one being invented for it. The skips are the honest denominator: only ONE down ruling with a record names a defect an instrument can see, and the two that name defects most precisely (the Komodo pair) have no record at all."
}
],
"n_up": 8,
"n_down": 5
}
},
"leave_one_term_out": {
"BENCH_R1_ELEVENLABS_EDIT_V2_1": {
"without_turn_pacing": 0.5769,
"without_motivic_dev": 0.5822,
"without_dynamic_breath": 0.6207,
"without_voice_dialogue": 0.6592
},
"BENCH_R1_ELEVENLABS_EDIT_V2_2": {
"without_turn_pacing": 0.8333,
"without_motivic_dev": 0.7062,
"without_dynamic_breath": 0.6934,
"without_voice_dialogue": 0.8087
},
"BENCH_R1_ELEVENLABS_EDIT_V2_3": {
"without_turn_pacing": 0.3333,
"without_motivic_dev": 0.6271,
"without_dynamic_breath": 0.5374,
"without_voice_dialogue": 0.4605
},
"BENCH_R1_ELEVENLABS_EDIT_V2_4": {
"without_turn_pacing": 0.3462,
"without_motivic_dev": 0.4178,
"without_dynamic_breath": 0.2767,
"without_voice_dialogue": 0.2895
},
"BENCH_R1_ELEVENLABS_EDIT_SERENE_V2": {
"without_turn_pacing": 0.6795,
"without_motivic_dev": 0.7532,
"without_dynamic_breath": 0.6763,
"without_voice_dialogue": 0.8045
},
"BENCH_R1_ELEVENLABS_TP_S1_v1_2": {
"without_turn_pacing": 0.5256,
"without_motivic_dev": 0.5758,
"without_dynamic_breath": 0.5374,
"without_voice_dialogue": 0.3964
},
"BENCH_R1_ELEVENLABS_EDIT_BATTLE_V2": {
"without_turn_pacing": 0.4743,
"without_motivic_dev": 0.3878,
"without_dynamic_breath": 0.5929,
"without_voice_dialogue": 0.5929
},
"BENCH_R1_ELEVENLABS_TP_B2_HYBRID_1": {
"without_turn_pacing": 0.4744,
"without_motivic_dev": 0.4818,
"without_dynamic_breath": 0.6998,
"without_voice_dialogue": 0.5588
},
"BENCH_R1_ELEVENLABS_EDIT_T2_MELODY_1": {
"without_turn_pacing": 0.4231,
"without_motivic_dev": 0.3942,
"without_dynamic_breath": 0.484,
"without_voice_dialogue": 0.343
},
"BENCH_R1_INHOUSE_BASELINE_R4": {
"without_turn_pacing": 0.5641,
"without_motivic_dev": 0.4113,
"without_dynamic_breath": 0.3088,
"without_voice_dialogue": 0.4498
},
"BENCH_R1_ELEVENLABS_TP_E1_v5_1": {
"without_turn_pacing": 0.5064,
"without_motivic_dev": 0.4274,
"without_dynamic_breath": 0.3312,
"without_voice_dialogue": 0.4209
},
"BENCH_R1_ELEVENLABS_TP_E1_v5_2": {
"without_turn_pacing": 0.5064,
"without_motivic_dev": 0.4274,
"without_dynamic_breath": 0.3312,
"without_voice_dialogue": 0.4209
},
"BENCH_R1_ELEVENLABS_TP_B1_v1_2": {
"without_turn_pacing": 0.2564,
"without_motivic_dev": 0.2115,
"without_dynamic_breath": 0.3654,
"without_voice_dialogue": 0.1923
}
}
}