-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathr2l-e-0004-yue-rlvr-boundary.json
More file actions
186 lines (186 loc) · 6.58 KB
/
Copy pathr2l-e-0004-yue-rlvr-boundary.json
File metadata and controls
186 lines (186 loc) · 6.58 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
{
"schema_version": "1.0.0",
"record_id": "r2l-e-0004-yue-rlvr-boundary",
"record_status": "draft",
"disclaimer": "DRAFT SOURCE EXTRACTION: values are transcribed from the cited source and remain ineligible for synthesis until two named authors independently verify them.",
"provenance": {
"evidence_class": "reported_by_source",
"source": {
"citation_key": "yue2025rlcapacity",
"title": "Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?",
"persistent_identifier": {
"type": "arxiv",
"value": "2504.13837",
"url": "https://arxiv.org/abs/2504.13837"
},
"publication_version": "source version cited by the manuscript",
"publication_date": "2025",
"accessed_at": "2026-07-23"
},
"extraction": {
"extracted_at": "2026-07-23",
"extracted_by": "Codex-assisted manuscript revision",
"verification_status": "unverified",
"verification_checks": []
},
"synthesis_eligibility": {
"status": "pending",
"reason": "Pending independent verification by two named manuscript authors."
},
"evidence_details": {
"input_record_ids": [],
"formula": null,
"normalization": null,
"uncertainty_procedure": null,
"assumptions": [],
"theorem_locator": null,
"supporting_record_ids": [],
"synthesis_rationale": null
},
"change_log": [
{
"changed_at": "2026-07-23",
"changed_by": "Codex-assisted manuscript revision",
"reason": "Created a source-reported draft for a manuscript quantitative claim."
}
]
},
"system": {
"name": "RLVR-trained models across source-reported model families",
"version_or_checkpoint": "source-reported paper configuration",
"lineage": "rl_for_deliberation",
"implementation_reference": null
},
"task": {
"domain": "language model reasoning or tool use",
"name": "language and visual reasoning benchmarks",
"version": "source-reported version",
"split": "source-reported benchmark and model-family combinations",
"environment_or_dataset": "language and visual reasoning benchmarks",
"unit_of_analysis": "one evaluation item",
"evaluation_protocol": "Compare base and RLVR-trained pass@k curves at small and large k."
},
"configuration": {
"configuration_id": "yue_rlvr_large_k_comparison",
"training_summary": "Source-reported RLVR configurations compared with their corresponding base models.",
"inference_summary": "Repeated sampling at increasing k to estimate pass@k and the sampled capability boundary.",
"parameters": []
},
"r2l_interface": {
"reasoning_object": {
"type": "generated reasoning trajectories",
"description": "generated reasoning trajectories",
"origin": "learned",
"explicit": true
},
"functional_coupling": {
"operations": [
"generated",
"verified",
"consumed"
],
"affects": [
"policy_optimization",
"action_selection"
]
},
"learning_signals": [
"verifiable outcome rewards"
],
"integration_points": [
"reward"
],
"grounding": {
"target": "The task inputs, executable checks, or reference answers declared by the source.",
"mechanism": "The source-specific reward or verifier connects the explicit trace to task feedback.",
"diagnostic_ids": []
}
},
"resources": [
{
"resource_id": "samples_per_problem",
"type": "samples",
"phase": "evaluation",
"reporting_status": "not_reported",
"quantity": null,
"unit": null,
"accounting_scope": "Pass@k sampling budget",
"notes": "The manuscript extraction does not assert this quantity; verify it from the source."
}
],
"controls": {
"comparators": [
{
"comparator_id": "base_model",
"name": "Corresponding base model",
"configuration_summary": "Base checkpoint sampled under the source's comparison protocol.",
"resource_matching": [
{
"resource_id": "samples_per_problem",
"status": "not_reported",
"explanation": "Resource parity requires confirmation during human verification."
}
]
}
],
"no_comparator_reason": null,
"ablations": [],
"leakage_and_contamination_checks": [
"Use only source-reported contamination checks; none are inferred in this extraction."
],
"privileged_information_checks": [
"Verifier, reference-answer, and tool access must be checked during human verification."
]
},
"outcomes": [
{
"measurement_id": "large_k_boundary",
"metric": "pass@k capability-boundary comparison",
"evaluation_layer": "outcome",
"population_or_split": "multiple language and visual reasoning benchmarks",
"estimate": {
"kind": "text",
"value": "At large k, base models match or exceed RLVR variants in the reported comparisons.",
"unit": null,
"denominator": {
"status": "not_reported",
"value": null,
"description": "The source-local extraction does not assert a denominator."
},
"aggregation": "Source-reported aggregate for the stated configuration."
},
"uncertainty": {
"status": "not_reported",
"description": "The cited result does not report an uncertainty interval for this value."
},
"comparator_id": "base_model",
"source_locator": "Yue et al., abstract, Figure 1, and Sections 3--4 on pass@k across base and RLVR-trained models",
"permitted_interpretation": "The study reports improved low-k sampling efficiency without uniform expansion of the large-k sampled capability boundary."
}
],
"diagnostics": [],
"claim_scope": {
"supported_interpretations": [
"The reported comparisons challenge a universal claim that RLVR expands the base model's sampled reasoning boundary."
],
"unsupported_inferences": [
"The record does not show that RLVR has no value for pass@1, efficiency, or other model families."
]
},
"limitations": [
{
"limitation_id": "yue_boundary_proxy",
"source": "extractor_identified",
"category": "external_validity",
"description": "Large-k pass@k is an empirical proxy for sampled capability, not proof of the model's complete reasoning support.",
"affected_measurement_ids": [
"large_k_boundary"
]
}
],
"extensions": {
"review_study_id": "R2L-S-0060",
"evidence_record_id": "R2L-E-0004",
"claim_direction": "counterevidence"
}
}