Created
July 16, 2026 07:51
-
-
Save lagz0ne/fddb86de7ff400f9aa21a09018ce5e1e to your computer and use it in GitHub Desktop.
C3 held-out paired evaluation v15 preregistration
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| { | |
| "protocol": "held-out-paired-c3-v15", | |
| "registered_at_utc": "2026-07-16T07:50:56Z", | |
| "status": "registered_after_case-loader preflight and before any v15 arm or judge inference", | |
| "supersedes": { | |
| "protocols": ["held-out-paired-c3-v12", "held-out-paired-c3-v13", "held-out-paired-c3-v14"], | |
| "reason": "V12 and v13 stopped before inference on harness schema validation. V14 produced no answer because its first control arm was killed after crossing an unrealistically low six-tool-call wall. V15 predeclares ten tool calls and 300000 tokens without changing cases, rubrics, conditions, or analysis.", | |
| "v12_v13_inference_call_count": 0, | |
| "v14_completed_answer_count": 0, | |
| "v14_protocol_deviation_count": 3, | |
| "v14_conservative_external_reserve_usd": 0.4 | |
| }, | |
| "relationship_to_prior_results": "Protocols v8 through v14 remain negative or infrastructure evidence and are not pooled into this confirmatory result.", | |
| "objective": { | |
| "conceptual_case_family_count": 3, | |
| "randomized_repeats_per_condition": 3, | |
| "valid_pair_count": 9, | |
| "valid_arm_count": 18, | |
| "agent_model": "gpt-5.6-sol", | |
| "agent_effort": "low", | |
| "raw_pair_preference_agreement_target": 0.8 | |
| }, | |
| "held_out_design": { | |
| "families": [ | |
| "authorization policy change impact", | |
| "attachment lifecycle change impact", | |
| "shared UI contract change impact" | |
| ], | |
| "excluded_observed_family_count": 3, | |
| "private_case_commitments_sha256": [ | |
| "fa9f19dab5d58b36bcd6590bbd285a23ccbf06797e4d6a9c559425833505b6fa", | |
| "278a88ae3696a860aac22d87e0d56606ee1b4579d468ee36dd5ee217cb9b6c7a", | |
| "c604a7f6c4636157f325fce9a3295a1bddae8112975ec3028caac2affe6f5dc8" | |
| ], | |
| "private_protocol_commitment_sha256": "bcbe68ef0a9143138424747d1312e2ce51d3fad2dcb61b42897dc70af15ff9ea", | |
| "project_specific_public_field_count": 0 | |
| }, | |
| "controls": { | |
| "standardized_baseline_generation_count": 1, | |
| "standardized_baseline_instruction_sha256": "c67682acb11535517ba2c70c08eb2de907c506dabc3e2f2a5c69e0f3fafd7ac9", | |
| "treatment_instruction_sha256": "86dafd2a6d2c1766276aa5b05ea285cce7218bf9719564be8b6a6a1c2d99ec8b", | |
| "frozen_runtime_sha256": "e346474c62fa093daac590734c6adcb28e1dbf7164500952003db48502a6901b", | |
| "condition_order_seed_commitment": "seeds 2101 through 2109", | |
| "max_tool_calls_per_arm": 10, | |
| "max_total_tokens_per_arm": 300000, | |
| "max_cost_usd_per_arm": 0.4, | |
| "timeout_seconds_per_arm": 600 | |
| }, | |
| "judging": { | |
| "judge_count": 2, | |
| "distinct_provider_count": 2, | |
| "condition_blind": true, | |
| "opposite_alias_randomization": true, | |
| "claim_labels": ["supported", "missing", "contradicted"], | |
| "score_formula": "2 * supported gold claims - 3 * contradicted gold claims - 4 * supported forbidden claims", | |
| "pick_rule": "higher score, then fewer supported forbidden claims, then higher change-usefulness score, otherwise tie", | |
| "agreement_metric": "same condition-level pair preference after private alias remapping", | |
| "agreement_target": 0.8, | |
| "adjudication": "Every preference, claim-label, or material usefulness disagreement receives condition-blind rule-consistent adjudication." | |
| }, | |
| "analysis": { | |
| "direction": "with skill minus without skill", | |
| "bootstrap_seed": 20260717, | |
| "bootstrap_resamples": 100000, | |
| "confidence_level": 0.95, | |
| "metrics": ["quality", "correctness", "elapsed time", "tokens", "cost"] | |
| }, | |
| "anti_goals": { | |
| "v15_total_cost_usd_max": 10.0, | |
| "v15_valid_arm_protocol_deviation_count_max": 0, | |
| "v15_study_protocol_deviation_count_max": 0, | |
| "public_project_specific_field_count_max": 0, | |
| "anti_goal_bypass_or_dishonesty_count_max": 0 | |
| }, | |
| "stopping_rule": "No paper-grade claim unless every v15 objective item and anti-goal passes from replayable evidence. Failed or invalid v15 inference attempts remain in the v15 cost and protocol ledger." | |
| } |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment