Skip to content

Commit 5606943

Browse files
kaoru0822-kitaujicursoragent
authored andcommitted
eval: freeze mixed-difficulty trajectory study
1 parent f9e197e commit 5606943

252 files changed

Lines changed: 32368 additions & 0 deletions

File tree

Some content is hidden

Large Commits have some content hidden by default. Use the searchbox below for content that may be hidden.
Lines changed: 114 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,114 @@
1+
{
2+
"agent_instructions": "Complete the mathematical task in this isolated directory. Read instruction.md, input.json, and submission_schema.json. Harbor's /app/submission.json is mapped to ./submission.json and environment/submission_schema.json is mapped to ./submission_schema.json; remain inside this workspace. Write submission.json and every required evidence file, using any mathematical method and any available Jacobian capability when useful. Do not use the web, inspect paths outside this directory, or look for verifier or solution material. Tool calls, tokens, log length, retries, and repeated actions receive no reward. The Jacobian server requires concise external reasoning records: one PLAN before work, BEFORE_TOOL and AFTER_TOOL around each math.run, and one FINAL audit before answering. Do not infer a mathematical conclusion from timeout, incomplete search, provider status, or one-sided evidence.",
3+
"calibration_sources": [
4+
{
5+
"calibration_id": "trajectory-value-calibration-codex-v1",
6+
"manifest_digest": "sha256:a0312ea7a7b9ebaf1c1c5120d421c96037b712c528e65adc7d0d8fef74adf74f",
7+
"manifest_path": "benchmarks/studies/trajectory-value-calibration-codex-v1/manifest.json",
8+
"summary_digest": "sha256:b78411cf0640f7c368e3d0cefc0314fe5297b1ca7f097191aec3b95a1396720d",
9+
"summary_path": "benchmarks/studies/trajectory-value-calibration-codex-v1/summary.json"
10+
},
11+
{
12+
"calibration_id": "trajectory-value-calibration-extension-codex-v1",
13+
"manifest_digest": "sha256:b0dff703db25237711d2aea33c5291b6a9275ad85f55a6e67243acfe9c686570",
14+
"manifest_path": "benchmarks/studies/trajectory-value-calibration-extension-codex-v1/manifest.json",
15+
"summary_digest": "sha256:a86ff39be7bfb50428dade33fc2127c9118b8e4eb12fbfd58a2a705a9ce32e1c",
16+
"summary_path": "benchmarks/studies/trajectory-value-calibration-extension-codex-v1/summary.json"
17+
}
18+
],
19+
"estimator_comparison": [
20+
"GROUP_ROLLOUT",
21+
"NUMCA_NUMERICAL",
22+
"REASONING_TEXT",
23+
"JACOBIAN_TYPED_EXACT",
24+
"ABSTRACT_VALUE_STATE",
25+
"ABSTRACT_VALUE_STATE_TEXT"
26+
],
27+
"exact_resume_supported": false,
28+
"h3_warning_rule": "any-strictly-negative-preterminal-selected-state-value-delta",
29+
"intermediate_value_surrogate": "leave-one-trajectory-out-success-frequency-among-compatible-cross-rollout-states",
30+
"label_status_at_freeze": "main-labels-not-collected",
31+
"model": {
32+
"codex_cli_version": "codex-cli 0.147.0",
33+
"model_id": "gpt-5.4-mini",
34+
"reasoning_effort": "medium"
35+
},
36+
"reasoning_log_mode": "REQUIRED",
37+
"repetitions_per_task": 8,
38+
"retries_for_wrong_answers": 0,
39+
"sandbox": "workspace-write",
40+
"schema_version": "1",
41+
"scorer_intervention": false,
42+
"selection_policy": {
43+
"calibration_rule": {
44+
"maximum_selected_tasks": 4,
45+
"maximum_success_rate_millionths": 800000,
46+
"minimum_labelled_rollouts": 2,
47+
"minimum_success_rate_millionths": 200000,
48+
"ordering": "candidate-order",
49+
"uncertainty": "wilson-95"
50+
},
51+
"maximum_main_tasks": 4,
52+
"minimum_main_tasks": 2,
53+
"source_combination": "source-order-then-candidate-order"
54+
},
55+
"study_id": "trajectory-value-mixed-codex-v1",
56+
"tasks": [
57+
{
58+
"accepted": 1,
59+
"calibration_id": "trajectory-value-calibration-codex-v1",
60+
"calibration_result_digest": "sha256:b1ac5a2b155580127cede1f1f6d08598ef2603b8986557c239b2b0aa59e3ccd2",
61+
"calibration_tags": [
62+
"artifact-binding",
63+
"capability-routing"
64+
],
65+
"dataset_id": "mathematical-benchmarks-v1",
66+
"labelled": 2,
67+
"rejected": 1,
68+
"success_rate_millionths": 500000,
69+
"task_contract_digest": "sha256:a635b8446904af69f86704e4d1eec8a7c2bdc195820596de6cdd46a813fbe334",
70+
"task_family": "graph-artifact-composition",
71+
"task_group": "graph-artifact-composition",
72+
"task_id": "graph-artifact-composition"
73+
},
74+
{
75+
"accepted": 1,
76+
"calibration_id": "trajectory-value-calibration-extension-codex-v1",
77+
"calibration_result_digest": "sha256:9712f23507ad8b12f990a72876c1755bdeeb9db9ff4a14edcb43802ec1058d74",
78+
"calibration_tags": [
79+
"scope-assurance",
80+
"candidate-checker-repair"
81+
],
82+
"dataset_id": "mathematical-benchmarks-v1",
83+
"labelled": 2,
84+
"rejected": 1,
85+
"success_rate_millionths": 500000,
86+
"task_contract_digest": "sha256:38936ad97260ff38385fc57ac457aabe772a99606e66b50b64b951f31139577a",
87+
"task_family": "apollonius-proof-gap-repair",
88+
"task_group": "apollonius-gap-repair",
89+
"task_id": "apollonius-gap-repair"
90+
},
91+
{
92+
"accepted": 1,
93+
"calibration_id": "trajectory-value-calibration-extension-codex-v1",
94+
"calibration_result_digest": "sha256:d009398468dc8a2f9b955a69f430d33a518259ef9f49b601195fa1d0a60ad43f",
95+
"calibration_tags": [
96+
"artifact-binding",
97+
"scope-assurance"
98+
],
99+
"dataset_id": "mathematical-benchmarks-v1",
100+
"labelled": 2,
101+
"rejected": 1,
102+
"success_rate_millionths": 500000,
103+
"task_contract_digest": "sha256:7006391aacfe6a49ef26b37f9577476aa1cc470e431fdcc53944834453ada975",
104+
"task_family": "projective-plane-homology-lattice",
105+
"task_group": "rp2-homology-lattice",
106+
"task_id": "rp2-homology-lattice"
107+
}
108+
],
109+
"terminal_reward": "clean-room-verifier-acceptance-only",
110+
"timeout_seconds": 420,
111+
"tool_mode": "direct",
112+
"training_performed": false,
113+
"web_search": "disabled"
114+
}

0 commit comments

Comments
 (0)