forked from morluto/jacobian
-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathtrajectory-value-mixed-study-v1.json
More file actions
114 lines (114 loc) · 5.32 KB
/
Copy pathtrajectory-value-mixed-study-v1.json
File metadata and controls
114 lines (114 loc) · 5.32 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
{
"agent_instructions": "Complete the mathematical task in this isolated directory. Read instruction.md, input.json, and submission_schema.json. Harbor's /app/submission.json is mapped to ./submission.json and environment/submission_schema.json is mapped to ./submission_schema.json; remain inside this workspace. Write submission.json and every required evidence file, using any mathematical method and any available Jacobian capability when useful. Do not use the web, inspect paths outside this directory, or look for verifier or solution material. Tool calls, tokens, log length, retries, and repeated actions receive no reward. The Jacobian server requires concise external reasoning records: one PLAN before work, BEFORE_TOOL and AFTER_TOOL around each math.run, and one FINAL audit before answering. Do not infer a mathematical conclusion from timeout, incomplete search, provider status, or one-sided evidence.",
"calibration_sources": [
{
"calibration_id": "trajectory-value-calibration-codex-v1",
"manifest_digest": "sha256:a0312ea7a7b9ebaf1c1c5120d421c96037b712c528e65adc7d0d8fef74adf74f",
"manifest_path": "benchmarks/studies/trajectory-value-calibration-codex-v1/manifest.json",
"summary_digest": "sha256:b78411cf0640f7c368e3d0cefc0314fe5297b1ca7f097191aec3b95a1396720d",
"summary_path": "benchmarks/studies/trajectory-value-calibration-codex-v1/summary.json"
},
{
"calibration_id": "trajectory-value-calibration-extension-codex-v1",
"manifest_digest": "sha256:b0dff703db25237711d2aea33c5291b6a9275ad85f55a6e67243acfe9c686570",
"manifest_path": "benchmarks/studies/trajectory-value-calibration-extension-codex-v1/manifest.json",
"summary_digest": "sha256:a86ff39be7bfb50428dade33fc2127c9118b8e4eb12fbfd58a2a705a9ce32e1c",
"summary_path": "benchmarks/studies/trajectory-value-calibration-extension-codex-v1/summary.json"
}
],
"estimator_comparison": [
"GROUP_ROLLOUT",
"NUMCA_NUMERICAL",
"REASONING_TEXT",
"JACOBIAN_TYPED_EXACT",
"ABSTRACT_VALUE_STATE",
"ABSTRACT_VALUE_STATE_TEXT"
],
"exact_resume_supported": false,
"h3_warning_rule": "any-strictly-negative-preterminal-selected-state-value-delta",
"intermediate_value_surrogate": "leave-one-trajectory-out-success-frequency-among-compatible-cross-rollout-states",
"label_status_at_freeze": "main-labels-not-collected",
"model": {
"codex_cli_version": "codex-cli 0.147.0",
"model_id": "gpt-5.4-mini",
"reasoning_effort": "medium"
},
"reasoning_log_mode": "REQUIRED",
"repetitions_per_task": 8,
"retries_for_wrong_answers": 0,
"sandbox": "workspace-write",
"schema_version": "1",
"scorer_intervention": false,
"selection_policy": {
"calibration_rule": {
"maximum_selected_tasks": 4,
"maximum_success_rate_millionths": 800000,
"minimum_labelled_rollouts": 2,
"minimum_success_rate_millionths": 200000,
"ordering": "candidate-order",
"uncertainty": "wilson-95"
},
"maximum_main_tasks": 4,
"minimum_main_tasks": 2,
"source_combination": "source-order-then-candidate-order"
},
"study_id": "trajectory-value-mixed-codex-v1",
"tasks": [
{
"accepted": 1,
"calibration_id": "trajectory-value-calibration-codex-v1",
"calibration_result_digest": "sha256:b1ac5a2b155580127cede1f1f6d08598ef2603b8986557c239b2b0aa59e3ccd2",
"calibration_tags": [
"artifact-binding",
"capability-routing"
],
"dataset_id": "mathematical-benchmarks-v1",
"labelled": 2,
"rejected": 1,
"success_rate_millionths": 500000,
"task_contract_digest": "sha256:a635b8446904af69f86704e4d1eec8a7c2bdc195820596de6cdd46a813fbe334",
"task_family": "graph-artifact-composition",
"task_group": "graph-artifact-composition",
"task_id": "graph-artifact-composition"
},
{
"accepted": 1,
"calibration_id": "trajectory-value-calibration-extension-codex-v1",
"calibration_result_digest": "sha256:9712f23507ad8b12f990a72876c1755bdeeb9db9ff4a14edcb43802ec1058d74",
"calibration_tags": [
"scope-assurance",
"candidate-checker-repair"
],
"dataset_id": "mathematical-benchmarks-v1",
"labelled": 2,
"rejected": 1,
"success_rate_millionths": 500000,
"task_contract_digest": "sha256:38936ad97260ff38385fc57ac457aabe772a99606e66b50b64b951f31139577a",
"task_family": "apollonius-proof-gap-repair",
"task_group": "apollonius-gap-repair",
"task_id": "apollonius-gap-repair"
},
{
"accepted": 1,
"calibration_id": "trajectory-value-calibration-extension-codex-v1",
"calibration_result_digest": "sha256:d009398468dc8a2f9b955a69f430d33a518259ef9f49b601195fa1d0a60ad43f",
"calibration_tags": [
"artifact-binding",
"scope-assurance"
],
"dataset_id": "mathematical-benchmarks-v1",
"labelled": 2,
"rejected": 1,
"success_rate_millionths": 500000,
"task_contract_digest": "sha256:7006391aacfe6a49ef26b37f9577476aa1cc470e431fdcc53944834453ada975",
"task_family": "projective-plane-homology-lattice",
"task_group": "rp2-homology-lattice",
"task_id": "rp2-homology-lattice"
}
],
"terminal_reward": "clean-room-verifier-acceptance-only",
"timeout_seconds": 420,
"tool_mode": "direct",
"training_performed": false,
"web_search": "disabled"
}