-
Notifications
You must be signed in to change notification settings - Fork 11
Expand file tree
/
Copy pathdocker-validation.json
More file actions
140 lines (140 loc) · 6.4 KB
/
Copy pathdocker-validation.json
File metadata and controls
140 lines (140 loc) · 6.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
{
"purpose": "Fresh-container capacity and regression checks, not replacement headline benchmarks",
"date": "2026-09-16",
"hardware": {
"gpus": "2x RTX 3090 24 GB",
"ram": "128 GB, 8-channel DDR4-3200"
},
"runtime": {
"build_source_commit": "c2529343cbee0ac9a897ae5a38baf3982ba5220f",
"deployment": "Public Dockerfile built from an empty Docker image cache; public docker_serve.sh; read-only checkpoint mount; no native runtime mounts",
"base_image_digest": "sha256:fc120ece0a388cc0aa1caad4a9f1cd92113484ab7ec2fd0efadd62585be05bf8",
"local_image_index_digest": "sha256:feedb4420154bda05f69cc39f55633e7e7f0b317c5a8b20675c2b5a3c5e28221",
"local_image_amd64_manifest_digest": "sha256:98f6c694c8781fe9b9a5940fd3976ab82ef6280f931fc6fe2d5b4f77ddbb656f",
"checkpoint_revision": "ef554143369a706525336f6b42a09094835dc077",
"checkpoint_check": "Target/MTP indices, shard headers, tensor counts and payload byte totals passed; tensor payload hashes were not repeated in this run",
"os": "Ubuntu 26.04 LTS",
"driver": "595.84",
"docker": "29.1.3",
"nvidia_container_toolkit": "1.20.0",
"torch": "2.13.0+cu130",
"cuda": "13.0",
"vllm": "0.1.dev20073+g8e685d198",
"humming_kernels": "0.1.12",
"overlay_files_verified_inside_image": 29,
"overlay_manifest_sha256": "55c9bde05764d2156e6167095a76a07c60f02d7baef11ec01e78e81cfbaf2921",
"gpu_free_tests_passed_inside_image": 32,
"cuda_peer_access": "Both directions passed capability checks and tensor-copy equality checks inside the image",
"custom_all_reduce": false,
"allocator": "expandable_segments:True",
"kv_dtype": "bfloat16",
"qsa": "approximate, with the 64 MiB workspace fix",
"max_parallel_loading_workers": "1 passed by launcher; ignored by pinned runtime",
"swap": "8 GiB original plus 56 GiB temporary loading swap; temporary file removed after each server"
},
"two_clients": {
"settings": {
"MAX_NUM_SEQS": 2,
"MAX_MODEL_LEN": 131072,
"MAX_NUM_BATCHED_TOKENS": 2048,
"KV_CACHE_MEMORY_BYTES": 4697620480,
"VLLM_WNA16_STATIC_HOT_CACHE_SIZE": 80,
"MTP_DEPTH": 3
},
"gpu_kv_tokens": 263416,
"startup_seconds": 474.078613759,
"graph_capture_sizes": "Default, not manually restricted",
"request_recipe": "repo-chat; second client rotates the public source corpus by half its characters; unique cache salt per request; greedy forced-length outputs",
"warmup_policy": "One short pair followed by one full-context pair; no repeated statistical runs",
"short": {
"requests": 2,
"input_tokens_each": 1024,
"output_tokens_each": 2048,
"exact_counts_passed": true,
"ttft_seconds": [5.595631230, 5.595673189],
"individual_post_first_chunk_tps": [45.838999912, 49.348504339],
"overlapping_decode_seconds": 41.480472803,
"overlapping_output_tokens": [1812, 2047],
"overlapping_aggregate_tps": 93.031726478
},
"full128k": {
"requests": 2,
"input_tokens_each": 129024,
"output_tokens_each": 2048,
"exact_counts_passed": true,
"wall_seconds": 404.180576801,
"ttft_seconds": [180.075548741, 356.006889247],
"individual_post_first_chunk_tps_including_other_client_prefill": [9.283074617, 42.509164195],
"overlapping_decode_seconds": 44.583164454,
"overlapping_output_tokens": [1830, 1821],
"overlapping_aggregate_tps": 81.891898988,
"peak_kv_usage_fraction": 0.9708737864,
"peak_running_requests": 2,
"preemptions": 0
},
"isolation_checks": {
"sequential": "2/2 passed",
"concurrent": "2/2 passed"
},
"mtp_counters_whole_suite": {
"drafted_tokens": 9474,
"accepted_tokens": 5141
},
"allocation_retries_loading": 2,
"allocation_retries_inference": 0,
"peaks": {
"cpu_c": 82.25,
"gpu_c": [70, 75],
"gpu_w": [283.06, 294.54],
"gpu_mib": [23211, 23643],
"min_host_available_gib": 14.195163727,
"max_host_swap_used_gib": 52.853134155
},
"cleanup": "Server stopped normally; temporary swap removed; original swap retained"
},
"single_request_hot84": {
"settings": {
"MAX_NUM_SEQS": 1,
"MAX_MODEL_LEN": 262144,
"MAX_NUM_BATCHED_TOKENS": 4096,
"KV_CACHE_MEMORY_BYTES": 4429185024,
"VLLM_WNA16_STATIC_HOT_CACHE_SIZE": 84,
"MTP_DEPTH": 3
},
"settings_note": "Image defaults, with no hot-cache or capacity overrides",
"gpu_kv_tokens": 276313,
"startup_seconds": 467.701298013,
"request_recipe": "repo-chat, unique cache salt per request, greedy forced-length outputs",
"probe_order": [[262016, 128], [1024, 128], [128, 1024]],
"first_user_request_is_full_context": true,
"all_exact_counts_passed": true,
"ttft_seconds_in_probe_order": [229.853154168, 2.534141732, 1.864057507],
"request_duration_seconds_in_probe_order": [231.622007849, 4.137608428, 16.643088296],
"post_first_chunk_tps_in_probe_order": [71.803638258, 79.210467235, 69.220256268],
"preemptions": 0,
"mtp_counters_whole_suite": {
"drafted_tokens": 1527,
"accepted_tokens": 768
},
"allocation_retries_loading": 2,
"allocation_retries_inference": 0,
"peaks": {
"cpu_c": 82.375,
"gpu_c": [70, 74],
"gpu_w": [285.17, 292.28],
"gpu_mib": [23955, 23953],
"min_host_available_gib": 24.039257050,
"max_host_swap_used_gib": 53.309982300
},
"cleanup": "Server stopped normally; temporary swap removed; original swap retained",
"limits": "Fresh server, not cold filesystem. First-use JIT warnings occurred during inference. No explicit warmup or repeats; short outputs are capacity evidence, not sustained-decode benchmarks."
},
"limits": [
"One server launch per profile and one run per shape on one host; no universal driver or display-memory guarantee.",
"Decode overlap is the intersection of both streams' generation intervals; do not sum rates whose intervals differ.",
"The first long client's generation overlaps the second client's prefill. Its whole-generation rate is not isolated decode throughput.",
"Forced output lengths and small isolation checks do not constitute a model-quality evaluation.",
"Fresh container startup includes kernel compilation. The existing checkpoint and host filesystem cache were reused.",
"Host swap was used and sample vmstat checks observed page-ins; these are not fully RAM-resident performance claims."
]
}