-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathr2l-e-0002a-deepseekmath-top1.json
More file actions
174 lines (174 loc) · 5.9 KB
/
Copy pathr2l-e-0002a-deepseekmath-top1.json
File metadata and controls
174 lines (174 loc) · 5.9 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
{
"schema_version": "1.0.0",
"record_id": "r2l-e-0002a-deepseekmath-top1",
"record_status": "draft",
"disclaimer": "DRAFT SOURCE EXTRACTION: values are transcribed from the cited source and remain ineligible for synthesis until two named authors independently verify them.",
"provenance": {
"evidence_class": "reported_by_source",
"source": {
"citation_key": "shao2024deepseekmath",
"title": "DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models",
"persistent_identifier": {
"type": "arxiv",
"value": "2402.03300",
"url": "https://arxiv.org/abs/2402.03300"
},
"publication_version": "source version cited by the manuscript",
"publication_date": "2023",
"accessed_at": "2026-07-23"
},
"extraction": {
"extracted_at": "2026-07-23",
"extracted_by": "Codex-assisted manuscript revision",
"verification_status": "unverified",
"verification_checks": []
},
"synthesis_eligibility": {
"status": "pending",
"reason": "Pending independent verification by two named manuscript authors."
},
"evidence_details": {
"input_record_ids": [],
"formula": null,
"normalization": null,
"uncertainty_procedure": null,
"assumptions": [],
"theorem_locator": null,
"supporting_record_ids": [],
"synthesis_rationale": null
},
"change_log": [
{
"changed_at": "2026-07-23",
"changed_by": "Codex-assisted manuscript revision",
"reason": "Created a source-reported draft for a manuscript quantitative claim."
}
]
},
"system": {
"name": "DeepSeekMath-RL 7B",
"version_or_checkpoint": "source-reported paper configuration",
"lineage": "rl_for_deliberation",
"implementation_reference": null
},
"task": {
"domain": "language model reasoning or tool use",
"name": "MATH",
"version": "source-reported version",
"split": "competition-level MATH evaluation",
"environment_or_dataset": "MATH",
"unit_of_analysis": "one evaluation item",
"evaluation_protocol": "Top-1 exact-answer evaluation without external tools or voting."
},
"configuration": {
"configuration_id": "deepseekmath_rl_math_top1",
"training_summary": "GRPO outcome-supervision stage applied to DeepSeekMath-Instruct 7B.",
"inference_summary": "One source-reported answer without tools or voting.",
"parameters": []
},
"r2l_interface": {
"reasoning_object": {
"type": "generated mathematical reasoning trace",
"description": "generated mathematical reasoning trace",
"origin": "learned",
"explicit": true
},
"functional_coupling": {
"operations": [
"generated",
"verified",
"consumed"
],
"affects": [
"policy_optimization",
"action_selection"
]
},
"learning_signals": [
"rule-based correctness reward",
"GRPO group-relative advantage"
],
"integration_points": [
"reward"
],
"grounding": {
"target": "The task inputs, executable checks, or reference answers declared by the source.",
"mechanism": "The source-specific reward or verifier connects the explicit trace to task feedback.",
"diagnostic_ids": []
}
},
"resources": [
{
"resource_id": "evaluation_items",
"type": "other",
"phase": "evaluation",
"reporting_status": "not_reported",
"quantity": null,
"unit": null,
"accounting_scope": "MATH evaluation items",
"notes": "The manuscript extraction does not assert this quantity; verify it from the source."
}
],
"controls": {
"comparators": [],
"no_comparator_reason": "This record preserves an absolute source-reported result; no comparator value is asserted.",
"ablations": [],
"leakage_and_contamination_checks": [
"Use only source-reported contamination checks; none are inferred in this extraction."
],
"privileged_information_checks": [
"Verifier, reference-answer, and tool access must be checked during human verification."
]
},
"outcomes": [
{
"measurement_id": "math_top1",
"metric": "MATH top-1 accuracy percent",
"evaluation_layer": "outcome",
"population_or_split": "competition-level MATH evaluation",
"estimate": {
"kind": "numeric",
"value": 51.7,
"unit": "percent",
"denominator": {
"status": "not_reported",
"value": null,
"description": "The source-local extraction does not assert a denominator."
},
"aggregation": "Source-reported aggregate for the stated configuration."
},
"uncertainty": {
"status": "not_reported",
"description": "The cited result does not report an uncertainty interval for this value."
},
"comparator_id": null,
"source_locator": "Shao et al., abstract and Figure 1, DeepSeekMath-RL 7B top-1 MATH result without external tools or voting",
"permitted_interpretation": "DeepSeekMath-RL 7B attains the stated top-1 result for the source's MATH evaluation."
}
],
"diagnostics": [],
"claim_scope": {
"supported_interpretations": [
"The source reports 51.7 percent top-1 MATH accuracy for DeepSeekMath-RL 7B."
],
"unsupported_inferences": [
"The value cannot be compared with multi-sample selection or other resource budgets as if protocols were identical."
]
},
"limitations": [
{
"limitation_id": "deep_top_denominator",
"source": "extractor_identified",
"category": "missing_uncertainty",
"description": "The denominator and uncertainty interval require confirmation from the evaluation protocol.",
"affected_measurement_ids": [
"math_top1"
]
}
],
"extensions": {
"review_study_id": "R2L-S-0045",
"evidence_record_id": "R2L-E-0002a",
"claim_direction": "positive_local"
}
}