EAR_GROUND_TRUTH.json

music/benchmark/EAR_GROUND_TRUTH.json

{
 "_doc": "THE EAR GROUND-TRUTH CONTROL -- every Josh ruling with a judged record, his own words, and the battery's reading of it under the old math and the new one.",
 "_honest_tier": "A CONTROL, NOT A SCORE. It proves the reading does not contradict a ruling. It never claims the reading is right about anything he has not ruled on.",
 "version": "ear-ground-truth/1.0.0",
 "fixtures": {
  "schema": "ear-ground-truth/1.0.0",
  "version": "ear-ground-truth/1.0.0",
  "rulings": [
   {
    "track_id": "BENCH_R1_ELEVENLABS_EDIT_V2_1",
    "grade": "up",
    "evidence_tier": "QUOTED",
    "verbatim": "Variation 2 are my favorite on all 4, save these... they're the start of our new chapter 2 sound track.",
    "nuance_verbatim": null,
    "cite": "docs/review_candidates.json :: candidates[] where id == 'CH02_SOUNDTRACK_T1'",
    "cite_file": "docs/review_candidates.json",
    "quote_check": "VACUOUS BY CONSTRUCTION, AND SKIPPED RATHER THAN COUNTED. This row's verbatim was READ OUT OF docs/review_candidates.json a microsecond earlier, so grepping the same file for it can only ever succeed -- it verifies a string against itself and proves nothing about drift. The teeth in G1 belong to the DECLARED_DOWN probes alone, which cite a DIFFERENT file than the one they are written in and therefore can actually fail. Counting this row as a passed check would inflate the arm's denominator with checks that cannot fail, which is the same defect class as a search that cannot match reporting zero.",
    "display_name": "Zither and brass, variation two",
    "cue_class": "EXPLORATION",
    "sha256": "b6f96cdb5bb85257e9e453a36218b682861aab5d46e35cffc3eaa59710db94c3",
    "record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_EDIT_V2_1.json",
    "reason_axes": [],
    "reason_note": "a graded-UP ruling names no defect; the reason arm applies to rejections only",
    "emitted_from_register": true
   },
   {
    "track_id": "BENCH_R1_ELEVENLABS_EDIT_V2_2",
    "grade": "up",
    "evidence_tier": "QUOTED",
    "verbatim": "Variation 2 are my favorite on all 4, save these... they're the start of our new chapter 2 sound track.",
    "nuance_verbatim": null,
    "cite": "docs/review_candidates.json :: candidates[] where id == 'CH02_SOUNDTRACK_T2'",
    "cite_file": "docs/review_candidates.json",
    "quote_check": "VACUOUS BY CONSTRUCTION, AND SKIPPED RATHER THAN COUNTED. This row's verbatim was READ OUT OF docs/review_candidates.json a microsecond earlier, so grepping the same file for it can only ever succeed -- it verifies a string against itself and proves nothing about drift. The teeth in G1 belong to the DECLARED_DOWN probes alone, which cite a DIFFERENT file than the one they are written in and therefore can actually fail. Counting this row as a passed check would inflate the arm's denominator with checks that cannot fail, which is the same defect class as a search that cannot match reporting zero.",
    "display_name": "Surging groove, variation two",
    "cue_class": "EXPLORATION",
    "sha256": "0edb02ea96e9c70d2e5dc260cdf4bbc2960bad334e8471a4e979dc95df9d54ee",
    "record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_EDIT_V2_2.json",
    "reason_axes": [],
    "reason_note": "a graded-UP ruling names no defect; the reason arm applies to rejections only",
    "emitted_from_register": true
   },
   {
    "track_id": "BENCH_R1_ELEVENLABS_EDIT_V2_3",
    "grade": "up",
    "evidence_tier": "QUOTED",
    "verbatim": "Variation 2 are my favorite on all 4, save these... they're the start of our new chapter 2 sound track.",
    "nuance_verbatim": null,
    "cite": "docs/review_candidates.json :: candidates[] where id == 'CH02_SOUNDTRACK_T3'",
    "cite_file": "docs/review_candidates.json",
    "quote_check": "VACUOUS BY CONSTRUCTION, AND SKIPPED RATHER THAN COUNTED. This row's verbatim was READ OUT OF docs/review_candidates.json a microsecond earlier, so grepping the same file for it can only ever succeed -- it verifies a string against itself and proves nothing about drift. The teeth in G1 belong to the DECLARED_DOWN probes alone, which cite a DIFFERENT file than the one they are written in and therefore can actually fail. Counting this row as a passed check would inflate the arm's denominator with checks that cannot fail, which is the same defect class as a search that cannot match reporting zero.",
    "display_name": "Bridge groove, variation two",
    "cue_class": "EXPLORATION",
    "sha256": "a60101ffe195603f60c6750d3ea97acade2dd0715952e87c838d1fd5a84a359a",
    "record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_EDIT_V2_3.json",
    "reason_axes": [],
    "reason_note": "a graded-UP ruling names no defect; the reason arm applies to rejections only",
    "emitted_from_register": true
   },
   {
    "track_id": "BENCH_R1_ELEVENLABS_EDIT_V2_4",
    "grade": "up",
    "evidence_tier": "QUOTED",
    "verbatim": "Variation 2 are my favorite on all 4, save these... they're the start of our new chapter 2 sound track.",
    "nuance_verbatim": null,
    "cite": "docs/review_candidates.json :: candidates[] where id == 'CH02_SOUNDTRACK_T4'",
    "cite_file": "docs/review_candidates.json",
    "quote_check": "VACUOUS BY CONSTRUCTION, AND SKIPPED RATHER THAN COUNTED. This row's verbatim was READ OUT OF docs/review_candidates.json a microsecond earlier, so grepping the same file for it can only ever succeed -- it verifies a string against itself and proves nothing about drift. The teeth in G1 belong to the DECLARED_DOWN probes alone, which cite a DIFFERENT file than the one they are written in and therefore can actually fail. Counting this row as a passed check would inflate the arm's denominator with checks that cannot fail, which is the same defect class as a search that cannot match reporting zero.",
    "display_name": "Sasando festival, variation two",
    "cue_class": "EXPLORATION",
    "sha256": "a8a2f7b6a732073a48f311072d8ecb2e47997963a200e55dbfc7db76f9fd6607",
    "record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_EDIT_V2_4.json",
    "reason_axes": [],
    "reason_note": "a graded-UP ruling names no defect; the reason arm applies to rejections only",
    "emitted_from_register": true
   },
   {
    "track_id": "BENCH_R1_ELEVENLABS_EDIT_SERENE_V2",
    "grade": "up",
    "evidence_tier": "QUOTED",
    "verbatim": "You can use both the serene ones, variation 2 for hearth of Flores, push them to website",
    "nuance_verbatim": "The serene pair are good and I do like them but very slow and kind of boring although that's how serene and peaceful is I guess",
    "cite": "docs/review_candidates.json :: candidates[] where id == 'CH02_SOUNDTRACK_T5'",
    "cite_file": "docs/review_candidates.json",
    "quote_check": "VACUOUS BY CONSTRUCTION, AND SKIPPED RATHER THAN COUNTED. This row's verbatim was READ OUT OF docs/review_candidates.json a microsecond earlier, so grepping the same file for it can only ever succeed -- it verifies a string against itself and proves nothing about drift. The teeth in G1 belong to the DECLARED_DOWN probes alone, which cite a DIFFERENT file than the one they are written in and therefore can actually fail. Counting this row as a passed check would inflate the arm's denominator with checks that cannot fail, which is the same defect class as a search that cannot match reporting zero.",
    "display_name": "Hearth of Flores, variation two",
    "cue_class": "SERENE",
    "sha256": "2a58d047ba7df78ee7539bd140f98e8277c593354aebc8922548e375d1b0e2e1",
    "record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_EDIT_SERENE_V2.json",
    "reason_axes": [],
    "reason_note": "a graded-UP ruling names no defect; the reason arm applies to rejections only",
    "emitted_from_register": true
   },
   {
    "track_id": "BENCH_R1_ELEVENLABS_TP_S1_v1_2",
    "grade": "up",
    "evidence_tier": "QUOTED",
    "verbatim": "You can use both the serene ones, variation 2 for hearth of Flores, push them to website",
    "nuance_verbatim": "The serene pair are good and I do like them but very slow and kind of boring although that's how serene and peaceful is I guess",
    "cite": "docs/review_candidates.json :: candidates[] where id == 'CH02_SOUNDTRACK_T6'",
    "cite_file": "docs/review_candidates.json",
    "quote_check": "VACUOUS BY CONSTRUCTION, AND SKIPPED RATHER THAN COUNTED. This row's verbatim was READ OUT OF docs/review_candidates.json a microsecond earlier, so grepping the same file for it can only ever succeed -- it verifies a string against itself and proves nothing about drift. The teeth in G1 belong to the DECLARED_DOWN probes alone, which cite a DIFFERENT file than the one they are written in and therefore can actually fail. Counting this row as a passed check would inflate the arm's denominator with checks that cannot fail, which is the same defect class as a search that cannot match reporting zero.",
    "display_name": "Highland Hearth",
    "cue_class": "SERENE",
    "sha256": "20f024f70ccb3f8445f6d6d626d6712fb60c235ac48e68a980ff844086d496ca",
    "record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_S1_v1_2.json",
    "reason_axes": [],
    "reason_note": "a graded-UP ruling names no defect; the reason arm applies to rejections only",
    "emitted_from_register": true
   },
   {
    "track_id": "BENCH_R1_ELEVENLABS_EDIT_BATTLE_V2",
    "grade": "up",
    "evidence_tier": "QUOTED",
    "verbatim": "I like this caci one attached and think it's good to push",
    "nuance_verbatim": null,
    "cite": "docs/review_candidates.json :: candidates[] where id == 'CH02_SOUNDTRACK_T7'",
    "cite_file": "docs/review_candidates.json",
    "quote_check": "VACUOUS BY CONSTRUCTION, AND SKIPPED RATHER THAN COUNTED. This row's verbatim was READ OUT OF docs/review_candidates.json a microsecond earlier, so grepping the same file for it can only ever succeed -- it verifies a string against itself and proves nothing about drift. The teeth in G1 belong to the DECLARED_DOWN probes alone, which cite a DIFFERENT file than the one they are written in and therefore can actually fail. Counting this row as a passed check would inflate the arm's denominator with checks that cannot fail, which is the same defect class as a search that cannot match reporting zero.",
    "display_name": "Caci highlands duel, variation two",
    "cue_class": "BATTLE",
    "sha256": "0ac2c601af7bb92de000e8ed76104dee9bed26a7734e8bd90c6c3379a2fe3494",
    "record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_EDIT_BATTLE_V2.json",
    "reason_axes": [],
    "reason_note": "a graded-UP ruling names no defect; the reason arm applies to rejections only",
    "emitted_from_register": true
   },
   {
    "track_id": "BENCH_R1_ELEVENLABS_TP_B2_HYBRID_1",
    "grade": "up",
    "evidence_tier": "QUOTED",
    "verbatim": "No complaints on highland fury. I think that one's pretty good for a battle song",
    "nuance_verbatim": null,
    "cite": "docs/review_candidates.json :: candidates[] where id == 'CH02_SOUNDTRACK_T8'",
    "cite_file": "docs/review_candidates.json",
    "quote_check": "VACUOUS BY CONSTRUCTION, AND SKIPPED RATHER THAN COUNTED. This row's verbatim was READ OUT OF docs/review_candidates.json a microsecond earlier, so grepping the same file for it can only ever succeed -- it verifies a string against itself and proves nothing about drift. The teeth in G1 belong to the DECLARED_DOWN probes alone, which cite a DIFFERENT file than the one they are written in and therefore can actually fail. Counting this row as a passed check would inflate the arm's denominator with checks that cannot fail, which is the same defect class as a search that cannot match reporting zero.",
    "display_name": "Highland fury",
    "cue_class": "BATTLE",
    "sha256": "623373c7197f834a9eedad4cdbd0bd4574282e6a8c1f0991b5e61ba7bb10066e",
    "record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_B2_HYBRID_1.json",
    "reason_axes": [],
    "reason_note": "a graded-UP ruling names no defect; the reason arm applies to rejections only",
    "emitted_from_register": true
   },
   {
    "track_id": "BENCH_R1_ELEVENLABS_EDIT_T2_MELODY_1",
    "grade": "down",
    "evidence_tier": "QUOTED",
    "verbatim": "I'm thinking about going back to the first one we said we loved. These new ones aren't hitting.",
    "quote_probe": "These new ones aren't hitting",
    "cite": "docs/spine/DECISIONS_PENDING_JOSH.md (T2 SURGING GROOVE block)",
    "cite_file": "docs/spine/DECISIONS_PENDING_JOSH.md",
    "reason_axes": [],
    "reason_note": "A PREFERENCE, not a named defect: he restored the earlier take without naming an axis. The reason-agreement arm therefore SKIPS this row and records the skip -- inventing an axis for it would be the curve fit WO-12 forbids. The ordering arms still bind, and this is the pair the order was minted on.",
    "pair_with": "BENCH_R1_ELEVENLABS_EDIT_V2_2",
    "pair_note": "the explicit kept-vs-rejected pair; the kept take is the restored 0edb02ea and the rejected one is the melody rework of it",
    "record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_EDIT_T2_MELODY_1.json",
    "emitted_from_register": false,
    "quote_check": "PROBED: \"These new ones aren't hitting\" is still findable in docs/spine/DECISIONS_PENDING_JOSH.md"
   },
   {
    "track_id": "BENCH_R1_INHOUSE_BASELINE_R4",
    "grade": "down",
    "evidence_tier": "QUOTED",
    "verbatim": "Round 4 is even worse than round 2 and is not shippable. The melody sucks. I hate the instruments you are using. I hate how slow the melody is and there is no variation or changes or movement throughout the songs. Everything sucks.",
    "quote_probe": "no variation or changes or movement",
    "cite": "docs/review_candidates.json (PASS7_FLORES_* rows, graded 2026-08-09)",
    "cite_file": "docs/review_candidates.json",
    "reason_axes": [
     "structural_yield",
     "dev_ops_per_min",
     "dev_kinds"
    ],
    "reason_note": "\"no variation or changes or movement\" is the motivic-development and turn-pacing clause in his own words -- the one down ruling on file that names a defect an instrument can see.",
    "record": "build/audio/benchmark/records/BENCH_R1_INHOUSE_BASELINE_R4.json",
    "emitted_from_register": false,
    "quote_check": "PROBED: 'no variation or changes or movement' is still findable in docs/review_candidates.json"
   },
   {
    "track_id": "BENCH_R1_ELEVENLABS_TP_E1_v5_1",
    "grade": "down",
    "evidence_tier": "ATTRIBUTED",
    "verbatim": "whole thing not good",
    "quote_probe": "whole thing not good",
    "cite": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_E1_v5_1.json :: measurements.floor_battery.results.cycle_law.title -- NOT the record's top-level `title`, which is None on every row",
    "cite_file": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_E1_v5_1.json",
    "reason_axes": [],
    "reason_note": "a lane's paraphrase inside a record title, not a quoted sentence.",
    "record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_E1_v5_1.json",
    "emitted_from_register": false,
    "quote_check": "PROBED: 'whole thing not good' is still findable in build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_E1_v5_1.json"
   },
   {
    "track_id": "BENCH_R1_ELEVENLABS_TP_E1_v5_2",
    "grade": "down",
    "evidence_tier": "ATTRIBUTED",
    "verbatim": "whole thing not good",
    "quote_probe": "whole thing not good",
    "cite": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_E1_v5_2.json :: measurements.floor_battery.results.cycle_law.title -- NOT the record's top-level `title`, which is None on every row",
    "cite_file": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_E1_v5_2.json",
    "reason_axes": [],
    "reason_note": "a lane's paraphrase inside a record title, not a quoted sentence.",
    "record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_E1_v5_2.json",
    "emitted_from_register": false,
    "quote_check": "PROBED: 'whole thing not good' is still findable in build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_E1_v5_2.json"
   },
   {
    "track_id": "BENCH_R1_ELEVENLABS_TP_B1_v1_2",
    "grade": "down",
    "evidence_tier": "ATTRIBUTED",
    "verbatim": "repetitive, failed",
    "quote_probe": "repetitive, failed",
    "cite": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_B1_v1_2.json :: measurements.floor_battery.results.cycle_law.title -- NOT the record's top-level `title`, which is None on every row",
    "cite_file": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_B1_v1_2.json",
    "reason_axes": [
     "structural_yield"
    ],
    "reason_note": "\"repetitive\" is the turn-pacing clause, at the attributed tier only.",
    "record": "build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_B1_v1_2.json",
    "emitted_from_register": false,
    "quote_check": "PROBED: 'repetitive, failed' is still findable in build/audio/benchmark/records/BENCH_R1_ELEVENLABS_TP_B1_v1_2.json"
   }
  ],
  "owed": [
   {
    "ruling": "The Komodo Tyrant -- \"a minute of literally nothing going on and the dead silence on the bridge sucks\"",
    "cite": "docs/spine/DECISIONS_PENDING_JOSH.md (2026-08-09/10 late sitting)",
    "why_it_cannot_enter": "no benchmark record on disk",
    "how_to_close": "download the WAV, run benchmark_judge --audio, then ear_axes dead_space -- this is the exact ruling the interior-dead-space finding was built for, and it is the only ruling on file that names that defect"
   },
   {
    "ruling": "The Komodo's Fury v2 -- \"the chanting doesn't even take a breath and again it's not complex at all just a bunch of banging\"",
    "cite": "docs/spine/DECISIONS_PENDING_JOSH.md (2026-08-09/10 late sitting)",
    "why_it_cannot_enter": "no benchmark record on disk AND no instrument reads vocal rest structure -- even with a record, the named defect is unmeasured",
    "how_to_close": "a vocal-rest instrument would be new work with its own theory ground; it is not in this build and is not claimed"
   },
   {
    "ruling": "\"Highlands of Flores\" 4:00 hybrid -- \"from the build to the climax it is literally all the same thing\"",
    "cite": "docs/spine/DECISIONS_PENDING_JOSH.md (2026-08-09/10 late sitting)",
    "why_it_cannot_enter": "IDENTITY UNRESOLVED. TP_E1_v6_2 is a 4:00 'Highlands of Flores' and would fit, but the same document grades the v6 pair UP and calls the retired take a HYBRID. Binding them would be a guess, and a guessed fixture row is worse than a missing one",
    "how_to_close": "the director names the sha256 of the retired take, or the row stays out"
   },
   {
    "ruling": "Rounds 1-3 -- \"a bunch of noise... these main melodies really suck... Never any harmonies added on\"; the round-3 wall read 57 of 57",
    "cite": "docs/review_candidates.json (PASS5_FLORES_* rows, graded 2026-08-08)",
    "why_it_cannot_enter": "no benchmark records; those rounds predate the judge",
    "how_to_close": "judge the round-3 renders if they are still on the box. This is the FOUNDING case of the inversion this control exists to catch"
   }
  ],
  "available": true,
  "reason": "",
  "quote_verification": {
   "probes_with_teeth": 5,
   "vacuous_and_skipped": 8,
   "why": "VACUOUS BY CONSTRUCTION, AND SKIPPED RATHER THAN COUNTED. This row's verbatim was READ OUT OF docs/review_candidates.json a microsecond earlier, so grepping the same file for it can only ever succeed -- it verifies a string against itself and proves nothing about drift. The teeth in G1 belong to the DECLARED_DOWN probes alone, which cite a DIFFERENT file than the one they are written in and therefore can actually fail. Counting this row as a passed check would inflate the arm's denominator with checks that cannot fail, which is the same defect class as a search that cannot match reporting zero.",
   "what_G1_actually_proves": "that 5 DECLARED_DOWN quote(s) are still findable in the OTHER file each one cites. It proves nothing about the 8 register-emitted row(s), whose text came out of the file the probe would search."
  },
  "harvested_down_from_register": [],
  "counts": {
   "up": 8,
   "down": 5,
   "quoted": 10,
   "attributed": 3,
   "harvested_down_from_register": 0,
   "owed_no_record": 4
  }
 },
 "readings": {
  "old_wall_fraction": {
   "BENCH_R1_ELEVENLABS_EDIT_V2_1": 0.7778,
   "BENCH_R1_ELEVENLABS_EDIT_V2_2": 0.25,
   "BENCH_R1_ELEVENLABS_EDIT_V2_3": 1.0,
   "BENCH_R1_ELEVENLABS_EDIT_V2_4": 0.7778,
   "BENCH_R1_ELEVENLABS_EDIT_SERENE_V2": 0.75,
   "BENCH_R1_ELEVENLABS_TP_S1_v1_2": 0.8889,
   "BENCH_R1_ELEVENLABS_EDIT_BATTLE_V2": 0.625,
   "BENCH_R1_ELEVENLABS_TP_B2_HYBRID_1": 0.625,
   "BENCH_R1_ELEVENLABS_EDIT_T2_MELODY_1": 0.7778,
   "BENCH_R1_INHOUSE_BASELINE_R4": 0.6667,
   "BENCH_R1_ELEVENLABS_TP_E1_v5_1": 0.5556,
   "BENCH_R1_ELEVENLABS_TP_E1_v5_2": 0.5556,
   "BENCH_R1_ELEVENLABS_TP_B1_v1_2": 0.5714
  },
  "character_reading": {
   "BENCH_R1_ELEVENLABS_EDIT_V2_1": 0.6098,
   "BENCH_R1_ELEVENLABS_EDIT_V2_2": 0.7604,
   "BENCH_R1_ELEVENLABS_EDIT_V2_3": 0.4896,
   "BENCH_R1_ELEVENLABS_EDIT_V2_4": 0.3325,
   "BENCH_R1_ELEVENLABS_EDIT_SERENE_V2": 0.7284,
   "BENCH_R1_ELEVENLABS_TP_S1_v1_2": 0.5088,
   "BENCH_R1_ELEVENLABS_EDIT_BATTLE_V2": 0.512,
   "BENCH_R1_ELEVENLABS_TP_B2_HYBRID_1": 0.5537,
   "BENCH_R1_ELEVENLABS_EDIT_T2_MELODY_1": 0.4111,
   "BENCH_R1_INHOUSE_BASELINE_R4": 0.4335,
   "BENCH_R1_ELEVENLABS_TP_E1_v5_1": 0.4215,
   "BENCH_R1_ELEVENLABS_TP_E1_v5_2": 0.4215,
   "BENCH_R1_ELEVENLABS_TP_B1_v1_2": 0.2564
  },
  "weak_axes_per_track": {
   "BENCH_R1_ELEVENLABS_EDIT_V2_1": [
    "orchestrational_dialogue"
   ],
   "BENCH_R1_ELEVENLABS_EDIT_V2_2": [
    "boundaries_per_min",
    "loudness_range_lu"
   ],
   "BENCH_R1_ELEVENLABS_EDIT_V2_3": [
    "dev_kinds",
    "dev_ops_per_min",
    "orchestrational_dialogue"
   ],
   "BENCH_R1_ELEVENLABS_EDIT_V2_4": [
    "dev_kinds",
    "dev_ops_per_min",
    "orchestrational_dialogue",
    "structural_yield"
   ],
   "BENCH_R1_ELEVENLABS_EDIT_SERENE_V2": [
    "counter_melody_presence"
   ],
   "BENCH_R1_ELEVENLABS_TP_S1_v1_2": [
    "dev_ops_per_min",
    "loudness_range_lu"
   ],
   "BENCH_R1_ELEVENLABS_EDIT_BATTLE_V2": [
    "counter_melody_presence",
    "dynamic_range_db"
   ],
   "BENCH_R1_ELEVENLABS_TP_B2_HYBRID_1": [
    "counter_melody_presence",
    "dynamic_range_db",
    "loudness_range_lu"
   ],
   "BENCH_R1_ELEVENLABS_EDIT_T2_MELODY_1": [
    "dynamic_range_db",
    "loudness_range_lu"
   ],
   "BENCH_R1_INHOUSE_BASELINE_R4": [
    "boundaries_per_min",
    "orchestrational_dialogue",
    "structural_yield"
   ],
   "BENCH_R1_ELEVENLABS_TP_E1_v5_1": [
    "boundaries_per_min",
    "structural_yield"
   ],
   "BENCH_R1_ELEVENLABS_TP_E1_v5_2": [
    "boundaries_per_min",
    "structural_yield"
   ],
   "BENCH_R1_ELEVENLABS_TP_B1_v1_2": [
    "counter_melody_presence",
    "dev_ops_per_min",
    "dynamic_range_db"
   ]
  },
  "measurable_axes_per_track": {
   "BENCH_R1_ELEVENLABS_EDIT_V2_1": [
    "boundaries_per_min",
    "counter_melody_presence",
    "dev_kinds",
    "dev_ops_per_min",
    "dynamic_range_db",
    "loudness_range_lu",
    "orchestrational_dialogue",
    "structural_yield",
    "syncopation_audio"
   ],
   "BENCH_R1_ELEVENLABS_EDIT_V2_2": [
    "boundaries_per_min",
    "counter_melody_presence",
    "dev_kinds",
    "dev_ops_per_min",
    "dynamic_range_db",
    "loudness_range_lu",
    "orchestrational_dialogue",
    "structural_yield",
    "syncopation_audio"
   ],
   "BENCH_R1_ELEVENLABS_EDIT_V2_3": [
    "boundaries_per_min",
    "counter_melody_presence",
    "dev_kinds",
    "dev_ops_per_min",
    "dynamic_range_db",
    "loudness_range_lu",
    "orchestrational_dialogue",
    "structural_yield",
    "syncopation_audio"
   ],
   "BENCH_R1_ELEVENLABS_EDIT_V2_4": [
    "boundaries_per_min",
    "counter_melody_presence",
    "dev_kinds",
    "dev_ops_per_min",
    "dynamic_range_db",
    "loudness_range_lu",
    "orchestrational_dialogue",
    "structural_yield",
    "syncopation_audio"
   ],
   "BENCH_R1_ELEVENLABS_EDIT_SERENE_V2": [
    "boundaries_per_min",
    "counter_melody_presence",
    "dev_kinds",
    "dev_ops_per_min",
    "dynamic_range_db",
    "loudness_range_lu",
    "orchestrational_dialogue",
    "structural_yield",
    "syncopation_audio"
   ],
   "BENCH_R1_ELEVENLABS_TP_S1_v1_2": [
    "boundaries_per_min",
    "counter_melody_presence",
    "dev_kinds",
    "dev_ops_per_min",
    "dynamic_range_db",
    "loudness_range_lu",
    "orchestrational_dialogue",
    "structural_yield",
    "syncopation_audio"
   ],
   "BENCH_R1_ELEVENLABS_EDIT_BATTLE_V2": [
    "boundaries_per_min",
    "counter_melody_presence",
    "dev_kinds",
    "dev_ops_per_min",
    "dynamic_range_db",
    "loudness_range_lu",
    "orchestrational_dialogue",
    "structural_yield",
    "syncopation_audio"
   ],
   "BENCH_R1_ELEVENLABS_TP_B2_HYBRID_1": [
    "boundaries_per_min",
    "counter_melody_presence",
    "dev_kinds",
    "dev_ops_per_min",
    "dynamic_range_db",
    "loudness_range_lu",
    "orchestrational_dialogue",
    "structural_yield",
    "syncopation_audio"
   ],
   "BENCH_R1_ELEVENLABS_EDIT_T2_MELODY_1": [
    "boundaries_per_min",
    "counter_melody_presence",
    "dev_kinds",
    "dev_ops_per_min",
    "dynamic_range_db",
    "loudness_range_lu",
    "orchestrational_dialogue",
    "structural_yield",
    "syncopation_audio"
   ],
   "BENCH_R1_INHOUSE_BASELINE_R4": [
    "boundaries_per_min",
    "counter_melody_presence",
    "dev_kinds",
    "dev_ops_per_min",
    "dynamic_range_db",
    "loudness_range_lu",
    "orchestrational_dialogue",
    "structural_yield",
    "syncopation_audio"
   ],
   "BENCH_R1_ELEVENLABS_TP_E1_v5_1": [
    "boundaries_per_min",
    "counter_melody_presence",
    "dev_kinds",
    "dev_ops_per_min",
    "dynamic_range_db",
    "loudness_range_lu",
    "orchestrational_dialogue",
    "structural_yield",
    "syncopation_audio"
   ],
   "BENCH_R1_ELEVENLABS_TP_E1_v5_2": [
    "boundaries_per_min",
    "counter_melody_presence",
    "dev_kinds",
    "dev_ops_per_min",
    "dynamic_range_db",
    "loudness_range_lu",
    "orchestrational_dialogue",
    "structural_yield",
    "syncopation_audio"
   ],
   "BENCH_R1_ELEVENLABS_TP_B1_v1_2": [
    "counter_melody_presence",
    "dev_kinds",
    "dev_ops_per_min",
    "dynamic_range_db",
    "loudness_range_lu",
    "orchestrational_dialogue",
    "syncopation_audio"
   ]
  }
 },
 "arms": {
  "old_math_MUST_FAIL": {
   "label": "OLD",
   "green": false,
   "arms": [
    {
     "arm": "A1 no graded-down row reads above the MEDIAN graded-up row",
     "green": false,
     "detail": "median graded-up reading 0.7639; violations ['ABS_EDIT_T2_MELODY_1']",
     "why_the_median": "four of the eight graded-up rows share ONE sentence, so they are one grading event, not four independent rulings. A strict every-up-above-every-down arm would be a claim the evidence cannot carry; the median arm is what 8 rows over ~4 events supports, and the exceptions are printed.",
     "strict_variant": {
      "every_up_above_every_down": false,
      "exceptions": [
       "ABS_EDIT_T2_MELODY_1 reads above ABS_EDIT_V2_1",
       "ABS_EDIT_T2_MELODY_1 reads above ABS_EDIT_V2_2",
       "_BASELINE_R4 reads above ABS_EDIT_V2_2",
       "ABS_TP_E1_v5_1 reads above ABS_EDIT_V2_2",
       "ABS_TP_E1_v5_2 reads above ABS_EDIT_V2_2",
       "ABS_TP_B1_v1_2 reads above ABS_EDIT_V2_2",
       "ABS_EDIT_T2_MELODY_1 reads above ABS_EDIT_V2_4",
       "ABS_EDIT_T2_MELODY_1 reads above ABS_EDIT_SERENE_V2",
       "ABS_EDIT_T2_MELODY_1 reads above ABS_EDIT_BATTLE_V2",
       "_BASELINE_R4 reads above ABS_EDIT_BATTLE_V2",
       "ABS_EDIT_T2_MELODY_1 reads above ABS_TP_B2_HYBRID_1",
       "_BASELINE_R4 reads above ABS_TP_B2_HYBRID_1"
      ],
      "status": "REPORTED, NOT ARMED -- see why_the_median"
     },
     "auc": 0.725
    },
    {
     "arm": "A2 on every declared kept-vs-rejected pair, the kept take reads above the rejected",
     "green": false,
     "detail": "1 declared pair(s); ABS_EDIT_V2_2 (0.25) does NOT read above ABS_EDIT_T2_MELODY_1 (0.7778)",
     "pairs": [
      "ABS_EDIT_V2_2 over ABS_EDIT_T2_MELODY_1"
     ]
    },
    {
     "arm": "A3 a rejected track's weak axes include one that AGREES with his stated reason",
     "green": false,
     "detail": "agreed: none; contradicted: [\"_BASELINE_R4: wanted one of ['structural_yield', 'dev_ops_per_min', 'dev_kinds'], the reading flagged ['dynamic_arc', 'groove_instrument', 'melodic_intelligence']\", \"ABS_TP_B1_v1_2: wanted one of ['structural_yield'], the reading flagged ['cycle_law', 'groove_instrument', 'melody_bands']\"]; unmeasurable: none",
     "skipped": [
      "ABS_EDIT_T2_MELODY_1: A PREFERENCE, not a named defect: he restored the earlier take without naming an axis. The reason-agreement arm therefore SKIPS this row and records the skip -- inventing an axis for it would be the curve fit WO-12 forbids. The ordering arms still bind, and this is the pair the order was minted on.",
      "ABS_TP_E1_v5_1: a lane's paraphrase inside a record title, not a quoted sentence.",
      "ABS_TP_E1_v5_2: a lane's paraphrase inside a record title, not a quoted sentence."
     ],
     "vacuity_guard": "RED because no down ruling could be checked at all",
     "unmeasurable_note": "TP_B1_v1_2 is the case worth reading twice. Its attributed reason is 'repetitive', and `structural_yield` -- the axis for exactly that -- is UNAVAILABLE on it, because the card carries only three novelty boundaries and floor_instruments refuses the ratio below four (its own C4 thin-card control). The refusal is correct and its side effect is not: the rows most likely to be structurally repetitive are the rows the yield axis declines to score. That is a hole in the instrument, recorded here rather than papered over by loosening the arm.",
     "skip_note": "A ruling that names no defect ('these new ones aren't hitting', 'whole thing not good') cannot have a reason arm without one being invented for it. The skips are the honest denominator: only ONE down ruling with a record names a defect an instrument can see, and the two that name defects most precisely (the Komodo pair) have no record at all."
    }
   ],
   "n_up": 8,
   "n_down": 5
  },
  "new_math": {
   "label": "NEW",
   "green": true,
   "arms": [
    {
     "arm": "A1 no graded-down row reads above the MEDIAN graded-up row",
     "green": true,
     "detail": "median graded-up reading 0.5329; violations none",
     "why_the_median": "four of the eight graded-up rows share ONE sentence, so they are one grading event, not four independent rulings. A strict every-up-above-every-down arm would be a claim the evidence cannot carry; the median arm is what 8 rows over ~4 events supports, and the exceptions are printed.",
     "strict_variant": {
      "every_up_above_every_down": false,
      "exceptions": [
       "ABS_EDIT_T2_MELODY_1 reads above ABS_EDIT_V2_4",
       "_BASELINE_R4 reads above ABS_EDIT_V2_4",
       "ABS_TP_E1_v5_1 reads above ABS_EDIT_V2_4",
       "ABS_TP_E1_v5_2 reads above ABS_EDIT_V2_4"
      ],
      "status": "REPORTED, NOT ARMED -- see why_the_median"
     },
     "auc": 0.9
    },
    {
     "arm": "A2 on every declared kept-vs-rejected pair, the kept take reads above the rejected",
     "green": true,
     "detail": "1 declared pair(s); all ordered as Josh ordered them",
     "pairs": [
      "ABS_EDIT_V2_2 over ABS_EDIT_T2_MELODY_1"
     ]
    },
    {
     "arm": "A3 a rejected track's weak axes include one that AGREES with his stated reason",
     "green": true,
     "detail": "agreed: [\"_BASELINE_R4: ['structural_yield']\"]; contradicted: none; unmeasurable: [\"ABS_TP_B1_v1_2: the reading carries none of ['structural_yield'] on this row (available: ['counter_melody_presence', 'dev_kinds', 'dev_ops_per_min', 'dynamic_range_db', 'loudness_range_lu', 'orchestrational_dialogue', 'syncopation_audio'])\"]",
     "skipped": [
      "ABS_EDIT_T2_MELODY_1: A PREFERENCE, not a named defect: he restored the earlier take without naming an axis. The reason-agreement arm therefore SKIPS this row and records the skip -- inventing an axis for it would be the curve fit WO-12 forbids. The ordering arms still bind, and this is the pair the order was minted on.",
      "ABS_TP_E1_v5_1: a lane's paraphrase inside a record title, not a quoted sentence.",
      "ABS_TP_E1_v5_2: a lane's paraphrase inside a record title, not a quoted sentence."
     ],
     "vacuity_guard": "1 ruling(s) actually checked",
     "unmeasurable_note": "TP_B1_v1_2 is the case worth reading twice. Its attributed reason is 'repetitive', and `structural_yield` -- the axis for exactly that -- is UNAVAILABLE on it, because the card carries only three novelty boundaries and floor_instruments refuses the ratio below four (its own C4 thin-card control). The refusal is correct and its side effect is not: the rows most likely to be structurally repetitive are the rows the yield axis declines to score. That is a hole in the instrument, recorded here rather than papered over by loosening the arm.",
     "skip_note": "A ruling that names no defect ('these new ones aren't hitting', 'whole thing not good') cannot have a reason arm without one being invented for it. The skips are the honest denominator: only ONE down ruling with a record names a defect an instrument can see, and the two that name defects most precisely (the Komodo pair) have no record at all."
    }
   ],
   "n_up": 8,
   "n_down": 5
  }
 },
 "leave_one_term_out": {
  "BENCH_R1_ELEVENLABS_EDIT_V2_1": {
   "without_turn_pacing": 0.5769,
   "without_motivic_dev": 0.5822,
   "without_dynamic_breath": 0.6207,
   "without_voice_dialogue": 0.6592
  },
  "BENCH_R1_ELEVENLABS_EDIT_V2_2": {
   "without_turn_pacing": 0.8333,
   "without_motivic_dev": 0.7062,
   "without_dynamic_breath": 0.6934,
   "without_voice_dialogue": 0.8087
  },
  "BENCH_R1_ELEVENLABS_EDIT_V2_3": {
   "without_turn_pacing": 0.3333,
   "without_motivic_dev": 0.6271,
   "without_dynamic_breath": 0.5374,
   "without_voice_dialogue": 0.4605
  },
  "BENCH_R1_ELEVENLABS_EDIT_V2_4": {
   "without_turn_pacing": 0.3462,
   "without_motivic_dev": 0.4178,
   "without_dynamic_breath": 0.2767,
   "without_voice_dialogue": 0.2895
  },
  "BENCH_R1_ELEVENLABS_EDIT_SERENE_V2": {
   "without_turn_pacing": 0.6795,
   "without_motivic_dev": 0.7532,
   "without_dynamic_breath": 0.6763,
   "without_voice_dialogue": 0.8045
  },
  "BENCH_R1_ELEVENLABS_TP_S1_v1_2": {
   "without_turn_pacing": 0.5256,
   "without_motivic_dev": 0.5758,
   "without_dynamic_breath": 0.5374,
   "without_voice_dialogue": 0.3964
  },
  "BENCH_R1_ELEVENLABS_EDIT_BATTLE_V2": {
   "without_turn_pacing": 0.4743,
   "without_motivic_dev": 0.3878,
   "without_dynamic_breath": 0.5929,
   "without_voice_dialogue": 0.5929
  },
  "BENCH_R1_ELEVENLABS_TP_B2_HYBRID_1": {
   "without_turn_pacing": 0.4744,
   "without_motivic_dev": 0.4818,
   "without_dynamic_breath": 0.6998,
   "without_voice_dialogue": 0.5588
  },
  "BENCH_R1_ELEVENLABS_EDIT_T2_MELODY_1": {
   "without_turn_pacing": 0.4231,
   "without_motivic_dev": 0.3942,
   "without_dynamic_breath": 0.484,
   "without_voice_dialogue": 0.343
  },
  "BENCH_R1_INHOUSE_BASELINE_R4": {
   "without_turn_pacing": 0.5641,
   "without_motivic_dev": 0.4113,
   "without_dynamic_breath": 0.3088,
   "without_voice_dialogue": 0.4498
  },
  "BENCH_R1_ELEVENLABS_TP_E1_v5_1": {
   "without_turn_pacing": 0.5064,
   "without_motivic_dev": 0.4274,
   "without_dynamic_breath": 0.3312,
   "without_voice_dialogue": 0.4209
  },
  "BENCH_R1_ELEVENLABS_TP_E1_v5_2": {
   "without_turn_pacing": 0.5064,
   "without_motivic_dev": 0.4274,
   "without_dynamic_breath": 0.3312,
   "without_voice_dialogue": 0.4209
  },
  "BENCH_R1_ELEVENLABS_TP_B1_v1_2": {
   "without_turn_pacing": 0.2564,
   "without_motivic_dev": 0.2115,
   "without_dynamic_breath": 0.3654,
   "without_voice_dialogue": 0.1923
  }
 }
}

Generated by harness/site/structure_site.py — the URL path is the repo path. review root