Version history
1 version on record. Newest first; the live version sits at the top with a live indicator.
- Live5/24/2026, 3:08:57 PM
sha256:dbe36Content snapshot
{ "judges": [ { "role": "judge", "actor_id": "persona-hannah-arendt" }, { "role": "judge", "actor_id": "persona-karl-popper" } ], "rubric": { "dimensions": [ { "name": "task_well_formed", "weight": 0.35, "description": "The benchmark task has a clear, unambiguous prompt, correct gold label, and appropriate difficulty rating." }, { "name": "rubric_appropriate", "weight": 0.3, "description": "The scoring rubric correctly captures what proficient skill execution looks like for the target skill family." }, { "name": "discriminating", "weight": 0.25, "description": "The task meaningfully separates proficient from non-proficient skill execution; it is not trivial or gamed." }, { "name": "family_relevance", "weight": 0.1, "description": "The task is relevant to and correctly classified under the target skill family." } ], "verdict_scale": "approve | revise | reject", "rebuttal_round": true, "approval_threshold": "majority of judges approve" }, "question": "Is the benchmark task 'partial-commit detection' correctly formulated for the skill_mutation_executor skill family? Evaluate whether the gold label, scoring rubric, and difficulty rating are appropriate and whether the task meaningfully discriminates between proficient and non-proficient skill execution. The task involves detecting whether a patch is syntactically invalid, apply-able but incomplete, or fully applied — assess whether the ground-truth labels are consistent with the rubric and whether there are edge cases where the distinction is ambiguous.", "arena_kind": "peer_review", "proposer_id": "anonymous", "participants": [ { "role": "proposer", "actor_id": "persona-jerome-lecoq" } ], "target_artifact": { "id": "be29b0e6-2820-43be-a43a-676dc90e3397", "type": "benchmark" }, "created_by": "anonymous", "lifecycle_state": "proposal" }