{"architectures":{"contracted_react":"Clear tool contract, all tools visible, complete trace; the model chooses planning, recovery, and stop.","goal_graph_hybrid":"Explicit goal/dependency state, all currently legal frontier actions visible, evidence ledger, one-attempt failed-route recovery, and deterministic bounded parallel branches for independent decomposition goals."},"batch_id":"2026-07-22-pilot-b-deepseek-v4-flash-a","batch_mode":"formal","budget":{"max_concurrent_branches":3,"max_model_calls":18,"max_tool_calls":24,"max_total_tokens":30000,"timeout_seconds":240},"controlled_constants":["provider_client","requested_model","task_instances","tool_names","fixture","deterministic_outcome_verifier","aggregate_run_budget","json_action_protocol"],"execution_policy":{"pair_exclusion":"If either cell is provider- or harness-excluded, retain both records and exclude the pair from primary arm estimates.","planned_attempts_per_cell":1,"replacement_runs":false,"silent_reruns":false,"smoke_isolation":"Smoke batches use a distinct private root and are never pooled with formal batches."},"fixture":{"canonical_sha256":"48e4786c90e2c07585ab88970364710323a6a8d2401ee1987e66b52fd2ccfb25","instance_count":12,"schema_version":"1.0"},"matrix":{"architectures":["contracted_react","goal_graph_hybrid"],"families":["alternative_path","open_evidence","parallel_decomposition"],"instances_per_family":4,"planned_cells":24},"metric_definitions":{"critical_path_rounds":"One-based aggregate model round of the first accepted submission; concurrent delegated calls share one round, and runs without a submission report all consumed aggregate rounds.","path_switch":"True only when a successful fixture route follows an attempted fixture route whose frozen outcome is an error.","utility":"60 points for exact answer, 30 times required-fact recall, 10 for zero false evidence, minus 5 per merge omission or conflict; clamped to 0-100."},"metrics":["verified_success","utility","required_fact_recall","false_evidence_count","path_switch","repeated_action_count","unrelated_action_count","model_calls","tool_calls","prompt_tokens","completion_tokens","total_tokens","elapsed_seconds","critical_path_rounds","merge_omissions","merge_conflicts","controller_rejection_count"],"model_identity":{"expected_model_id":"deepseek-v4-flash","observed_model_ids":["deepseek-v4-flash"],"preflight_performed":true,"requested_model_id":"deepseek-v4-flash"},"prompt_hashes":{"architecture_instruction_sha256":{"contracted_react":"f11c06bcaefd6d6b5e7abb513ee51a0285139d74f568accd4a7b74dd82bda3ea","goal_graph_hybrid":"d116405324c3d29651d4f7ce019ac5c3aee6716bddaf5e28be7c1f240a4d42a4"},"shared_action_protocol_sha256":"5d7b0d0de7b513772ad25587643adf5d052487a2ed01efdbda2e0af3b6aa0139","shared_tool_contract_sha256":"10167d6499052e13792904a242d0f1b64de03a0ba5e32e1e7296efa7fdd8410a"},"provider_request":{"action_protocol":"json_action_v1","parameters":{"max_tokens":1024,"response_format":{"type":"json_object"},"stream":false,"temperature":0,"thinking":{"type":"disabled"}},"protocol":"openai_chat_completions"},"research_question":"Compare two minimal agent-control architectures while holding provider, model, task instances, tools, fixtures, verifier, and aggregate budgets fixed.","review":{"private_action_arguments_included":false,"private_artifacts_included":false,"projection":"allowlist","raw_provider_bodies_included":false,"raw_provider_messages_included":false,"reviewed":true},"schema_version":"1.0","study":"pilot-b","tool_names":["list_evidence","read_evidence","submit_result","finish"],"verifier":{"deterministic":true,"outcome_only":true,"prescribed_route_checked":false}}
