-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathr2l-e-0001-lightman-prm.json
More file actions
196 lines (196 loc) · 6.66 KB
/
Copy pathr2l-e-0001-lightman-prm.json
File metadata and controls
196 lines (196 loc) · 6.66 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
{
"schema_version": "1.0.0",
"record_id": "r2l-e-0001-lightman-prm",
"record_status": "draft",
"disclaimer": "DRAFT SOURCE EXTRACTION: values are transcribed from the cited source and remain ineligible for synthesis until two named authors independently verify them.",
"provenance": {
"evidence_class": "reported_by_source",
"source": {
"citation_key": "lightman2023let",
"title": "Let's Verify Step by Step",
"persistent_identifier": {
"type": "arxiv",
"value": "2305.20050",
"url": "https://arxiv.org/abs/2305.20050"
},
"publication_version": "source version cited by the manuscript",
"publication_date": "2023",
"accessed_at": "2026-07-23"
},
"extraction": {
"extracted_at": "2026-07-23",
"extracted_by": "Codex-assisted manuscript revision",
"verification_status": "unverified",
"verification_checks": []
},
"synthesis_eligibility": {
"status": "pending",
"reason": "Pending independent verification by two named manuscript authors."
},
"evidence_details": {
"input_record_ids": [],
"formula": null,
"normalization": null,
"uncertainty_procedure": null,
"assumptions": [],
"theorem_locator": null,
"supporting_record_ids": [],
"synthesis_rationale": null
},
"change_log": [
{
"changed_at": "2026-07-23",
"changed_by": "Codex-assisted manuscript revision",
"reason": "Created a source-reported draft for a manuscript quantitative claim."
}
]
},
"system": {
"name": "Process-supervised reward model",
"version_or_checkpoint": "source-reported paper configuration",
"lineage": "rl_for_deliberation",
"implementation_reference": null
},
"task": {
"domain": "language model reasoning or tool use",
"name": "MATH",
"version": "source-reported version",
"split": "representative subset of 500 MATH test problems",
"environment_or_dataset": "MATH",
"unit_of_analysis": "one evaluation item",
"evaluation_protocol": "Best-of-1,860 selection using the process reward model."
},
"configuration": {
"configuration_id": "lightman_prm_best_of_1860",
"training_summary": "The source trains a reward model from human process labels; this record does not treat the generator as RL-trained.",
"inference_summary": "Select the highest-scored candidate among 1,860 generated solutions.",
"parameters": []
},
"r2l_interface": {
"reasoning_object": {
"type": "generated mathematical solution with step boundaries",
"description": "generated mathematical solution with step boundaries",
"origin": "learned",
"explicit": true
},
"functional_coupling": {
"operations": [
"generated",
"verified",
"consumed"
],
"affects": [
"policy_optimization",
"action_selection"
]
},
"learning_signals": [
"human process labels"
],
"integration_points": [
"reward",
"search"
],
"grounding": {
"target": "The task inputs, executable checks, or reference answers declared by the source.",
"mechanism": "The source-specific reward or verifier connects the explicit trace to task feedback.",
"diagnostic_ids": []
}
},
"resources": [
{
"resource_id": "candidate_solutions",
"type": "samples",
"phase": "inference",
"reporting_status": "reported",
"quantity": 1860,
"unit": "candidate solutions per problem",
"accounting_scope": "Candidate pool used for each best-of-1,860 selection.",
"notes": "Source-local selection budget."
}
],
"controls": {
"comparators": [
{
"comparator_id": "outcome_reward_model",
"name": "Outcome-supervised reward model",
"configuration_summary": "Outcome-supervised model evaluated with the same selection protocol.",
"resource_matching": [
{
"resource_id": "candidate_solutions",
"status": "not_reported",
"explanation": "Resource parity requires confirmation during human verification."
}
]
}
],
"no_comparator_reason": null,
"ablations": [],
"leakage_and_contamination_checks": [
"Use only source-reported contamination checks; none are inferred in this extraction."
],
"privileged_information_checks": [
"Verifier, reference-answer, and tool access must be checked during human verification."
]
},
"outcomes": [
{
"measurement_id": "math_best_of_1860",
"metric": "MATH solved percent",
"evaluation_layer": "outcome",
"population_or_split": "representative subset of 500 MATH test problems",
"estimate": {
"kind": "numeric",
"value": 78.2,
"unit": "percent",
"denominator": {
"status": "reported",
"value": 500,
"description": "500 evaluated cases"
},
"aggregation": "Source-reported aggregate for the stated configuration."
},
"uncertainty": {
"status": "not_reported",
"description": "The cited result does not report an uncertainty interval for this value."
},
"comparator_id": "outcome_reward_model",
"source_locator": "Lightman et al., Figure 3, Best-of-1860 test performance, product reduction with neutral labels treated as positive",
"permitted_interpretation": "The PRM-selected solution is correct for 78.2 percent of the stated 500-problem subset under best-of-1860 selection."
}
],
"diagnostics": [],
"claim_scope": {
"supported_interpretations": [
"Process-supervised selection reaches the reported source-local result under the stated sampling budget."
],
"unsupported_inferences": [
"The record does not establish universal superiority of process supervision or an RL-trained generator."
]
},
"limitations": [
{
"limitation_id": "lightman_component_boundary",
"source": "extractor_identified",
"category": "external_validity",
"description": "The result evaluates verifier-based selection and does not show that the generator was trained with RL.",
"affected_measurement_ids": [
"math_best_of_1860"
]
},
{
"limitation_id": "lightman_uncertainty",
"source": "extractor_identified",
"category": "missing_uncertainty",
"description": "No uncertainty interval is retained for the reported percentage.",
"affected_measurement_ids": [
"math_best_of_1860"
]
}
],
"extensions": {
"review_study_id": "R2L-S-0023",
"evidence_record_id": "R2L-E-0001",
"claim_direction": "positive_local"
}
}