Add H100 validation artifacts
Browse files- .gitattributes +1 -0
- validation/h100_20260420/artifacts/hunyuanvideo_bf16_544x960_17f_8steps.mp4 +0 -0
- validation/h100_20260420/artifacts/hunyuanvideo_bf16_544x960_17f_8steps_perf.json +87 -0
- validation/h100_20260420/artifacts/hunyuanvideo_bf16_fp8_contact_sheet.png +3 -0
- validation/h100_20260420/artifacts/hunyuanvideo_fp8_544x960_17f_8steps.mp4 +0 -0
- validation/h100_20260420/artifacts/hunyuanvideo_fp8_544x960_17f_8steps_perf.json +87 -0
- validation/h100_20260420/benchmark/bf16_offline_throughput.jsonl +1 -0
- validation/h100_20260420/benchmark/fp8_offline_throughput.jsonl +1 -0
- validation/h100_20260420/benchmark/fp8_offline_throughput_transformer_path.jsonl +1 -0
- validation/h100_20260420/logs/bench_bf16.log +229 -0
- validation/h100_20260420/logs/bench_fp8_transformer_path.log +231 -0
- validation/h100_20260420/logs/convert_sglang_fp8.log +8 -0
- validation/h100_20260420/logs/modelopt_quantize_fp8.log +0 -0
- validation/h100_20260420/logs/profile_bf16.log +90 -0
- validation/h100_20260420/logs/profile_fp8.log +92 -0
- validation/h100_20260420/profiler/5b26da87-550f-4d35-8cb8-bf9aa9c535d0-5_steps-global-rank0.trace.json.gz +3 -0
- validation/h100_20260420/profiler/63e4807c-f120-4fdc-b5b0-c1ab96522a7e-5_steps-global-rank0.trace.json.gz +3 -0
- validation/h100_20260420/profiler/kernel_summary.md +53 -0
- validation/h100_20260420/result_summary.md +53 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
validation/h100_20260420/artifacts/hunyuanvideo_bf16_fp8_contact_sheet.png filter=lfs diff=lfs merge=lfs -text
|
validation/h100_20260420/artifacts/hunyuanvideo_bf16_544x960_17f_8steps.mp4
ADDED
|
Binary file (39.3 kB). View file
|
|
|
validation/h100_20260420/artifacts/hunyuanvideo_bf16_544x960_17f_8steps_perf.json
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"timestamp": "2026-04-20T02:24:39.346613+00:00",
|
| 3 |
+
"request_id": "f14e3fae-01d7-4d44-bc16-ac805cdc4aa9",
|
| 4 |
+
"commit_hash": "N/A",
|
| 5 |
+
"tag": "cli_generate",
|
| 6 |
+
"total_duration_ms": 10885.856637032703,
|
| 7 |
+
"steps": [
|
| 8 |
+
{
|
| 9 |
+
"name": "InputValidationStage",
|
| 10 |
+
"duration_ms": 0.039525795727968216
|
| 11 |
+
},
|
| 12 |
+
{
|
| 13 |
+
"name": "TextEncodingStage",
|
| 14 |
+
"duration_ms": 952.3325660265982
|
| 15 |
+
},
|
| 16 |
+
{
|
| 17 |
+
"name": "TimestepPreparationStage",
|
| 18 |
+
"duration_ms": 0.37180911749601364
|
| 19 |
+
},
|
| 20 |
+
{
|
| 21 |
+
"name": "LatentPreparationStage",
|
| 22 |
+
"duration_ms": 0.8033178746700287
|
| 23 |
+
},
|
| 24 |
+
{
|
| 25 |
+
"name": "DenoisingStage",
|
| 26 |
+
"duration_ms": 4592.238221084699
|
| 27 |
+
},
|
| 28 |
+
{
|
| 29 |
+
"name": "DecodingStage",
|
| 30 |
+
"duration_ms": 5328.874800121412
|
| 31 |
+
}
|
| 32 |
+
],
|
| 33 |
+
"denoise_steps_ms": [
|
| 34 |
+
{
|
| 35 |
+
"step": 0,
|
| 36 |
+
"duration_ms": 1984.2948359437287
|
| 37 |
+
},
|
| 38 |
+
{
|
| 39 |
+
"step": 1,
|
| 40 |
+
"duration_ms": 104.41974294371903
|
| 41 |
+
},
|
| 42 |
+
{
|
| 43 |
+
"step": 2,
|
| 44 |
+
"duration_ms": 412.1756220702082
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
"step": 3,
|
| 48 |
+
"duration_ms": 426.5072650741786
|
| 49 |
+
},
|
| 50 |
+
{
|
| 51 |
+
"step": 4,
|
| 52 |
+
"duration_ms": 406.0544071253389
|
| 53 |
+
},
|
| 54 |
+
{
|
| 55 |
+
"step": 5,
|
| 56 |
+
"duration_ms": 423.0810720473528
|
| 57 |
+
},
|
| 58 |
+
{
|
| 59 |
+
"step": 6,
|
| 60 |
+
"duration_ms": 414.66441797092557
|
| 61 |
+
},
|
| 62 |
+
{
|
| 63 |
+
"step": 7,
|
| 64 |
+
"duration_ms": 418.7105300370604
|
| 65 |
+
}
|
| 66 |
+
],
|
| 67 |
+
"memory_checkpoints": {
|
| 68 |
+
"before_forward": {
|
| 69 |
+
"allocated_mb": 24456.18,
|
| 70 |
+
"reserved_mb": 24796.0,
|
| 71 |
+
"peak_allocated_mb": 24456.18,
|
| 72 |
+
"peak_reserved_mb": 24796.0
|
| 73 |
+
},
|
| 74 |
+
"after_forward": {
|
| 75 |
+
"allocated_mb": 24542.47,
|
| 76 |
+
"reserved_mb": 45960.0,
|
| 77 |
+
"peak_allocated_mb": 44298.8,
|
| 78 |
+
"peak_reserved_mb": 45960.0
|
| 79 |
+
}
|
| 80 |
+
},
|
| 81 |
+
"meta": {
|
| 82 |
+
"prompt": [
|
| 83 |
+
"A cinematic shot of a red sports car driving through rain at night, reflections on wet streets, smooth camera movement."
|
| 84 |
+
],
|
| 85 |
+
"model": "/root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773"
|
| 86 |
+
}
|
| 87 |
+
}
|
validation/h100_20260420/artifacts/hunyuanvideo_bf16_fp8_contact_sheet.png
ADDED
|
Git LFS Details
|
validation/h100_20260420/artifacts/hunyuanvideo_fp8_544x960_17f_8steps.mp4
ADDED
|
Binary file (37.2 kB). View file
|
|
|
validation/h100_20260420/artifacts/hunyuanvideo_fp8_544x960_17f_8steps_perf.json
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"timestamp": "2026-04-20T02:25:58.322820+00:00",
|
| 3 |
+
"request_id": "221d84f7-7f54-4343-9d8a-13ca3fa105dd",
|
| 4 |
+
"commit_hash": "N/A",
|
| 5 |
+
"tag": "cli_generate",
|
| 6 |
+
"total_duration_ms": 10391.746508190408,
|
| 7 |
+
"steps": [
|
| 8 |
+
{
|
| 9 |
+
"name": "InputValidationStage",
|
| 10 |
+
"duration_ms": 0.036803074181079865
|
| 11 |
+
},
|
| 12 |
+
{
|
| 13 |
+
"name": "TextEncodingStage",
|
| 14 |
+
"duration_ms": 973.2257057912648
|
| 15 |
+
},
|
| 16 |
+
{
|
| 17 |
+
"name": "TimestepPreparationStage",
|
| 18 |
+
"duration_ms": 0.33611198887228966
|
| 19 |
+
},
|
| 20 |
+
{
|
| 21 |
+
"name": "LatentPreparationStage",
|
| 22 |
+
"duration_ms": 0.7671341300010681
|
| 23 |
+
},
|
| 24 |
+
{
|
| 25 |
+
"name": "DenoisingStage",
|
| 26 |
+
"duration_ms": 4008.4011738654226
|
| 27 |
+
},
|
| 28 |
+
{
|
| 29 |
+
"name": "DecodingStage",
|
| 30 |
+
"duration_ms": 5397.510214010254
|
| 31 |
+
}
|
| 32 |
+
],
|
| 33 |
+
"denoise_steps_ms": [
|
| 34 |
+
{
|
| 35 |
+
"step": 0,
|
| 36 |
+
"duration_ms": 1814.3004791345447
|
| 37 |
+
},
|
| 38 |
+
{
|
| 39 |
+
"step": 1,
|
| 40 |
+
"duration_ms": 89.51880293898284
|
| 41 |
+
},
|
| 42 |
+
{
|
| 43 |
+
"step": 2,
|
| 44 |
+
"duration_ms": 345.99901596084237
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
"step": 3,
|
| 48 |
+
"duration_ms": 357.83073981292546
|
| 49 |
+
},
|
| 50 |
+
{
|
| 51 |
+
"step": 4,
|
| 52 |
+
"duration_ms": 347.1649489365518
|
| 53 |
+
},
|
| 54 |
+
{
|
| 55 |
+
"step": 5,
|
| 56 |
+
"duration_ms": 348.6302839592099
|
| 57 |
+
},
|
| 58 |
+
{
|
| 59 |
+
"step": 6,
|
| 60 |
+
"duration_ms": 355.295421089977
|
| 61 |
+
},
|
| 62 |
+
{
|
| 63 |
+
"step": 7,
|
| 64 |
+
"duration_ms": 347.20492409542203
|
| 65 |
+
}
|
| 66 |
+
],
|
| 67 |
+
"memory_checkpoints": {
|
| 68 |
+
"before_forward": {
|
| 69 |
+
"allocated_mb": 15906.27,
|
| 70 |
+
"reserved_mb": 15964.0,
|
| 71 |
+
"peak_allocated_mb": 15906.27,
|
| 72 |
+
"peak_reserved_mb": 15964.0
|
| 73 |
+
},
|
| 74 |
+
"after_forward": {
|
| 75 |
+
"allocated_mb": 15992.56,
|
| 76 |
+
"reserved_mb": 51272.0,
|
| 77 |
+
"peak_allocated_mb": 35748.73,
|
| 78 |
+
"peak_reserved_mb": 51272.0
|
| 79 |
+
}
|
| 80 |
+
},
|
| 81 |
+
"meta": {
|
| 82 |
+
"prompt": [
|
| 83 |
+
"A cinematic shot of a red sports car driving through rain at night, reflections on wet streets, smooth camera movement."
|
| 84 |
+
],
|
| 85 |
+
"model": "/root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773"
|
| 86 |
+
}
|
| 87 |
+
}
|
validation/h100_20260420/benchmark/bf16_offline_throughput.jsonl
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"metadata": {"timestamp": "2026-04-20T02:27:47", "model_path": "/root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773", "task_type": "unknown", "backend": "engine"}, "configuration": {"num_inference_steps": 8, "guidance_scale": 1.0, "seed": 0, "batch_size": 1, "num_prompts": 3, "resolution": "960x544x17", "dataset": "random"}, "results": {"num_requests": 3, "successful_requests": 3, "failed_requests": 0, "total_duration_seconds": 26.11314248898998, "total_frames_generated": 3, "total_pixels_generated": 26634240, "images_per_second": 0.11488467928610595, "frames_per_second": 0.11488467928610595, "megapixels_per_second": 1.0199553734763915, "requests_per_second": 0.11488467928610595, "latency_per_request_seconds": 8.704380829663327, "peak_memory_mb": 0.0}}
|
validation/h100_20260420/benchmark/fp8_offline_throughput.jsonl
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"metadata": {"timestamp": "2026-04-20T02:31:53", "model_path": "/data/bbuf/hunyuanvideo_fp8_20260420/sglang_fp8_pipeline", "task_type": "unknown", "backend": "engine"}, "configuration": {"num_inference_steps": 8, "guidance_scale": 1.0, "seed": 0, "batch_size": 1, "num_prompts": 3, "resolution": "960x544x17", "dataset": "random"}, "results": {"num_requests": 3, "successful_requests": 3, "failed_requests": 0, "total_duration_seconds": 24.55099329398945, "total_frames_generated": 3, "total_pixels_generated": 26634240, "images_per_second": 0.12219464866761448, "frames_per_second": 0.12219464866761448, "megapixels_per_second": 1.0848538664429748, "requests_per_second": 0.12219464866761448, "latency_per_request_seconds": 8.183664431329817, "peak_memory_mb": 0.0}}
|
validation/h100_20260420/benchmark/fp8_offline_throughput_transformer_path.jsonl
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"metadata": {"timestamp": "2026-04-20T02:50:19", "model_path": "/root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773", "task_type": "unknown", "backend": "engine"}, "configuration": {"num_inference_steps": 8, "guidance_scale": 1.0, "seed": 0, "batch_size": 1, "num_prompts": 3, "resolution": "960x544x17", "dataset": "random"}, "results": {"num_requests": 3, "successful_requests": 3, "failed_requests": 0, "total_duration_seconds": 24.901405425043777, "total_frames_generated": 3, "total_pixels_generated": 26634240, "images_per_second": 0.12047512776057402, "frames_per_second": 0.12047512776057402, "megapixels_per_second": 1.069587822268597, "requests_per_second": 0.12047512776057402, "latency_per_request_seconds": 8.300468475014592, "peak_memory_mb": 0.0}}
|
validation/h100_20260420/logs/bench_bf16.log
ADDED
|
@@ -0,0 +1,229 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 1 |
12%|█▎ | 1/8 [00:02<00:14, 2.06s/it]
|
| 2 |
25%|██▌ | 2/8 [00:02<00:05, 1.10it/s]
|
| 3 |
38%|███▊ | 3/8 [00:02<00:03, 1.47it/s]
|
| 4 |
50%|█████ | 4/8 [00:03<00:02, 1.72it/s]
|
| 5 |
62%|██████▎ | 5/8 [00:03<00:01, 1.93it/s]
|
| 6 |
75%|███████▌ | 6/8 [00:03<00:00, 2.05it/s]
|
| 7 |
88%|████████▊ | 7/8 [00:04<00:00, 2.16it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 8 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 9 |
12%|█▎ | 1/8 [00:00<00:00, 9.62it/s]
|
| 10 |
25%|██▌ | 2/8 [00:00<00:01, 3.56it/s]
|
| 11 |
38%|███▊ | 3/8 [00:00<00:01, 2.90it/s]
|
| 12 |
50%|█████ | 4/8 [00:01<00:01, 2.68it/s]
|
| 13 |
62%|██████▎ | 5/8 [00:01<00:01, 2.58it/s]
|
| 14 |
75%|███████▌ | 6/8 [00:02<00:00, 2.49it/s]
|
| 15 |
88%|████████▊ | 7/8 [00:02<00:00, 2.47it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 16 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 17 |
12%|█▎ | 1/8 [00:00<00:00, 9.66it/s]
|
| 18 |
25%|██▌ | 2/8 [00:00<00:01, 3.54it/s]
|
| 19 |
38%|███▊ | 3/8 [00:00<00:01, 2.89it/s]
|
| 20 |
50%|█████ | 4/8 [00:01<00:01, 2.67it/s]
|
| 21 |
62%|██████▎ | 5/8 [00:01<00:01, 2.57it/s]
|
| 22 |
75%|███████▌ | 6/8 [00:02<00:00, 2.47it/s]
|
| 23 |
88%|████████▊ | 7/8 [00:02<00:00, 2.46it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 24 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 25 |
12%|█▎ | 1/8 [00:00<00:00, 9.68it/s]
|
| 26 |
25%|██▌ | 2/8 [00:00<00:01, 3.55it/s]
|
| 27 |
38%|███▊ | 3/8 [00:00<00:01, 2.88it/s]
|
| 28 |
50%|█████ | 4/8 [00:01<00:01, 2.66it/s]
|
| 29 |
62%|██████▎ | 5/8 [00:01<00:01, 2.57it/s]
|
| 30 |
75%|███████▌ | 6/8 [00:02<00:00, 2.48it/s]
|
| 31 |
88%|████████▊ | 7/8 [00:02<00:00, 2.46it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
No platform detected. Using base SRTPlatform with defaults.
|
| 2 |
+
[04-20 02:26:33] server_args: {"model_path": "/root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773", "model_id": null, "backend": "sglang", "attention_backend": null, "attention_backend_config": {}, "cache_dit_config": null, "nccl_port": null, "trust_remote_code": false, "revision": null, "num_gpus": 1, "tp_size": 1, "sp_degree": 1, "ulysses_degree": 1, "ring_degree": 1, "dp_size": 1, "dp_degree": 1, "enable_cfg_parallel": false, "hsdp_replicate_dim": 1, "hsdp_shard_dim": 1, "dist_timeout": 3600, "pipeline_class_name": null, "lora_path": null, "lora_nickname": "default", "lora_scale": 1.0, "lora_weight_name": null, "component_paths": {}, "transformer_weights_path": null, "lora_target_modules": null, "dit_cpu_offload": false, "dit_layerwise_offload": false, "dit_offload_prefetch_size": 0.0, "text_encoder_cpu_offload": true, "image_encoder_cpu_offload": true, "vae_cpu_offload": true, "use_fsdp_inference": false, "pin_cpu_memory": true, "ltx2_two_stage_device_mode": null, "comfyui_mode": false, "enable_torch_compile": false, "warmup": false, "warmup_resolutions": null, "warmup_steps": 1, "disable_autocast": false, "master_port": 30005, "host": "127.0.0.1", "port": 30000, "webui": false, "webui_port": 12312, "scheduler_port": 5644, "strict_ports": false, "output_path": "outputs/", "input_save_path": "inputs/uploads", "prompt_file_path": null, "model_paths": {}, "model_loaded": {"transformer": true, "vae": true, "video_vae": true, "audio_vae": true, "video_dit": true, "audio_dit": true, "dual_tower_bridge": true}, "boundary_ratio": null, "base_gpu_id": 0, "disagg_role": "monolithic", "disagg_timeout": 600, "disagg_dispatch_policy": "round_robin", "disagg_mode": false, "disagg_server_addr": null, "encoder_urls": null, "denoiser_urls": null, "decoder_urls": null, "encoder_tp": null, "denoiser_tp": null, "denoiser_sp": null, "denoiser_ulysses": null, "denoiser_ring": null, "decoder_tp": null, "disagg_transfer_pool_size": 268435456, "disagg_p2p_hostname": "127.0.0.1", "disagg_ib_device": null, "pool_work_endpoint": null, "pool_result_endpoint": null, "log_level": "info", "uvicorn_access_log_exclude_prefixes": []}
|
| 3 |
+
[04-20 02:26:33] Starting offline throughput benchmark...
|
| 4 |
+
[04-20 02:26:33] Initializing engine...
|
| 5 |
+
[04-20 02:26:33] Local mode: True
|
| 6 |
+
[04-20 02:26:33] Starting server...
|
| 7 |
+
No platform detected. Using base SRTPlatform with defaults.
|
| 8 |
+
[04-20 02:26:40] Scheduler bind at endpoint: tcp://127.0.0.1:5644
|
| 9 |
+
[04-20 02:26:40] Initializing distributed environment with world_size=1, device=cuda:0, timeout=3600
|
| 10 |
+
[04-20 02:26:40] Setting distributed timeout to 3600 seconds
|
| 11 |
+
[04-20 02:26:41] No pipeline_class_name specified, using model_index.json
|
| 12 |
+
[04-20 02:26:41] Diffusers version: 0.32.0.dev0
|
| 13 |
+
[04-20 02:26:41] Using pipeline from model_index.json: HunyuanVideoPipeline
|
| 14 |
+
[04-20 02:26:41] Loading pipeline modules...
|
| 15 |
+
[04-20 02:26:41] Model already exists locally and is complete
|
| 16 |
+
[04-20 02:26:41] Model path: /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773
|
| 17 |
+
[04-20 02:26:41] Diffusers version: 0.32.0.dev0
|
| 18 |
+
[04-20 02:26:41] Loading pipeline modules from config: {'_class_name': 'HunyuanVideoPipeline', '_diffusers_version': '0.32.0.dev0', 'scheduler': ['diffusers', 'FlowMatchEulerDiscreteScheduler'], 'text_encoder': ['transformers', 'LlamaModel'], 'text_encoder_2': ['transformers', 'CLIPTextModel'], 'tokenizer': ['transformers', 'LlamaTokenizerFast'], 'tokenizer_2': ['transformers', 'CLIPTokenizer'], 'transformer': ['diffusers', 'HunyuanVideoTransformer3DModel'], 'vae': ['diffusers', 'AutoencoderKLHunyuanVideo']}
|
| 19 |
+
[04-20 02:26:41] Loading required components: ['text_encoder', 'text_encoder_2', 'tokenizer', 'tokenizer_2', 'vae', 'transformer', 'scheduler']
|
| 20 |
+
|
| 21 |
+
[04-20 02:26:41] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
|
| 22 |
+
[04-20 02:26:43] [RunAI Streamer] Overall time to stream 14.0 GiB of all files to cpu: 2.07s, 6.8 GiB/s
|
| 23 |
+
[04-20 02:27:01] Loaded text_encoder: FSDPLlamaModel (sgl-diffusion version). model size: 13.98 GB, consumed GPU mem: 0.81 GB, avail GPU mem: 73.04 GB
|
| 24 |
+
|
| 25 |
+
[04-20 02:27:02] Using Torch SDPA backend
|
| 26 |
+
[04-20 02:27:02] [RunAI Streamer] Overall time to stream 234.7 MiB of all files to cpu: 0.03s, 7.7 GiB/s
|
| 27 |
+
[04-20 02:27:03] Loaded text_encoder_2: FSDPCLIPTextModel (sgl-diffusion version). model size: 0.23 GB, consumed GPU mem: 0.76 GB, avail GPU mem: 72.27 GB
|
| 28 |
+
|
| 29 |
+
[04-20 02:27:04] Loaded tokenizer: LlamaTokenizer (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
|
| 30 |
+
|
| 31 |
+
[04-20 02:27:04] Loaded tokenizer_2: CLIPTokenizer (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
|
| 32 |
+
[04-20 02:27:04] Loading vae from /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773/vae. avail mem: 72.27 GB
|
| 33 |
+
[93m[04-20 02:27:04] VAE unexpected keys: ['encoder.conv_in.conv.bias', 'encoder.conv_in.conv.weight', 'encoder.conv_norm_out.bias', 'encoder.conv_norm_out.weight', 'encoder.conv_out.conv.bias', 'encoder.conv_out.conv.weight', 'encoder.down_blocks.0.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.0.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.0.resnets.0.conv1.conv.bias', 'encoder.down_blocks.0.resnets.0.conv1.conv.weight', 'encoder.down_blocks.0.resnets.0.conv2.conv.bias', 'encoder.down_blocks.0.resnets.0.conv2.conv.weight', 'encoder.down_blocks.0.resnets.0.norm1.bias', 'encoder.down_blocks.0.resnets.0.norm1.weight', 'encoder.down_blocks.0.resnets.0.norm2.bias', 'encoder.down_blocks.0.resnets.0.norm2.weight', 'encoder.down_blocks.0.resnets.1.conv1.conv.bias', 'encoder.down_blocks.0.resnets.1.conv1.conv.weight', 'encoder.down_blocks.0.resnets.1.conv2.conv.bias', 'encoder.down_blocks.0.resnets.1.conv2.conv.weight', 'encoder.down_blocks.0.resnets.1.norm1.bias', 'encoder.down_blocks.0.resnets.1.norm1.weight', 'encoder.down_blocks.0.resnets.1.norm2.bias', 'encoder.down_blocks.0.resnets.1.norm2.weight', 'encoder.down_blocks.1.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.1.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.1.resnets.0.conv1.conv.bias', 'encoder.down_blocks.1.resnets.0.conv1.conv.weight', 'encoder.down_blocks.1.resnets.0.conv2.conv.bias', 'encoder.down_blocks.1.resnets.0.conv2.conv.weight', 'encoder.down_blocks.1.resnets.0.conv_shortcut.conv.bias', 'encoder.down_blocks.1.resnets.0.conv_shortcut.conv.weight', 'encoder.down_blocks.1.resnets.0.norm1.bias', 'encoder.down_blocks.1.resnets.0.norm1.weight', 'encoder.down_blocks.1.resnets.0.norm2.bias', 'encoder.down_blocks.1.resnets.0.norm2.weight', 'encoder.down_blocks.1.resnets.1.conv1.conv.bias', 'encoder.down_blocks.1.resnets.1.conv1.conv.weight', 'encoder.down_blocks.1.resnets.1.conv2.conv.bias', 'encoder.down_blocks.1.resnets.1.conv2.conv.weight', 'encoder.down_blocks.1.resnets.1.norm1.bias', 'encoder.down_blocks.1.resnets.1.norm1.weight', 'encoder.down_blocks.1.resnets.1.norm2.bias', 'encoder.down_blocks.1.resnets.1.norm2.weight', 'encoder.down_blocks.2.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.2.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.2.resnets.0.conv1.conv.bias', 'encoder.down_blocks.2.resnets.0.conv1.conv.weight', 'encoder.down_blocks.2.resnets.0.conv2.conv.bias', 'encoder.down_blocks.2.resnets.0.conv2.conv.weight', 'encoder.down_blocks.2.resnets.0.conv_shortcut.conv.bias', 'encoder.down_blocks.2.resnets.0.conv_shortcut.conv.weight', 'encoder.down_blocks.2.resnets.0.norm1.bias', 'encoder.down_blocks.2.resnets.0.norm1.weight', 'encoder.down_blocks.2.resnets.0.norm2.bias', 'encoder.down_blocks.2.resnets.0.norm2.weight', 'encoder.down_blocks.2.resnets.1.conv1.conv.bias', 'encoder.down_blocks.2.resnets.1.conv1.conv.weight', 'encoder.down_blocks.2.resnets.1.conv2.conv.bias', 'encoder.down_blocks.2.resnets.1.conv2.conv.weight', 'encoder.down_blocks.2.resnets.1.norm1.bias', 'encoder.down_blocks.2.resnets.1.norm1.weight', 'encoder.down_blocks.2.resnets.1.norm2.bias', 'encoder.down_blocks.2.resnets.1.norm2.weight', 'encoder.down_blocks.3.resnets.0.conv1.conv.bias', 'encoder.down_blocks.3.resnets.0.conv1.conv.weight', 'encoder.down_blocks.3.resnets.0.conv2.conv.bias', 'encoder.down_blocks.3.resnets.0.conv2.conv.weight', 'encoder.down_blocks.3.resnets.0.norm1.bias', 'encoder.down_blocks.3.resnets.0.norm1.weight', 'encoder.down_blocks.3.resnets.0.norm2.bias', 'encoder.down_blocks.3.resnets.0.norm2.weight', 'encoder.down_blocks.3.resnets.1.conv1.conv.bias', 'encoder.down_blocks.3.resnets.1.conv1.conv.weight', 'encoder.down_blocks.3.resnets.1.conv2.conv.bias', 'encoder.down_blocks.3.resnets.1.conv2.conv.weight', 'encoder.down_blocks.3.resnets.1.norm1.bias', 'encoder.down_blocks.3.resnets.1.norm1.weight', 'encoder.down_blocks.3.resnets.1.norm2.bias', 'encoder.down_blocks.3.resnets.1.norm2.weight', 'encoder.mid_block.attentions.0.group_norm.bias', 'encoder.mid_block.attentions.0.group_norm.weight', 'encoder.mid_block.attentions.0.to_k.bias', 'encoder.mid_block.attentions.0.to_k.weight', 'encoder.mid_block.attentions.0.to_out.0.bias', 'encoder.mid_block.attentions.0.to_out.0.weight', 'encoder.mid_block.attentions.0.to_q.bias', 'encoder.mid_block.attentions.0.to_q.weight', 'encoder.mid_block.attentions.0.to_v.bias', 'encoder.mid_block.attentions.0.to_v.weight', 'encoder.mid_block.resnets.0.conv1.conv.bias', 'encoder.mid_block.resnets.0.conv1.conv.weight', 'encoder.mid_block.resnets.0.conv2.conv.bias', 'encoder.mid_block.resnets.0.conv2.conv.weight', 'encoder.mid_block.resnets.0.norm1.bias', 'encoder.mid_block.resnets.0.norm1.weight', 'encoder.mid_block.resnets.0.norm2.bias', 'encoder.mid_block.resnets.0.norm2.weight', 'encoder.mid_block.resnets.1.conv1.conv.bias', 'encoder.mid_block.resnets.1.conv1.conv.weight', 'encoder.mid_block.resnets.1.conv2.conv.bias', 'encoder.mid_block.resnets.1.conv2.conv.weight', 'encoder.mid_block.resnets.1.norm1.bias', 'encoder.mid_block.resnets.1.norm1.weight', 'encoder.mid_block.resnets.1.norm2.bias', 'encoder.mid_block.resnets.1.norm2.weight', 'quant_conv.bias', 'quant_conv.weight'][0;0m
|
| 34 |
+
[04-20 02:27:04] Loaded vae: AutoencoderKLHunyuanVideo (sgl-diffusion version). model size: 0.27 GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
|
| 35 |
+
[04-20 02:27:04] Loading transformer from /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773/transformer. avail mem: 72.27 GB
|
| 36 |
+
[04-20 02:27:04] Loading HunyuanVideoTransformer3DModel from 6 safetensors file(s) , param_dtype: torch.bfloat16
|
| 37 |
+
[04-20 02:27:04] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
|
| 38 |
+
[04-20 02:27:04] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
|
| 39 |
+
[04-20 02:27:05] [RunAI Streamer] Overall time to stream 23.9 GiB of all files to cpu: 1.65s, 14.5 GiB/s
|
| 40 |
+
[04-20 02:27:09] Loaded model with 12.82B parameters
|
| 41 |
+
[04-20 02:27:09] Loaded transformer: HunyuanVideoTransformer3DModel (sgl-diffusion version). model size: 23.88 GB, consumed GPU mem: 24.19 GB, avail GPU mem: 48.08 GB
|
| 42 |
+
|
| 43 |
+
[04-20 02:27:09] Loaded scheduler: FlowMatchEulerDiscreteScheduler (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 48.08 GB
|
| 44 |
+
|
| 45 |
+
[04-20 02:27:09] Creating pipeline stages...
|
| 46 |
+
[04-20 02:27:09] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
|
| 47 |
+
[04-20 02:27:09] Pipeline instantiated
|
| 48 |
+
[04-20 02:27:09] Worker 0: Initialized device, model, and distributed environment.
|
| 49 |
+
[04-20 02:27:09] Worker 0: Scheduler loop started.
|
| 50 |
+
[04-20 02:27:09] Engine initialized successfully
|
| 51 |
+
[04-20 02:27:09] Loading random dataset...
|
| 52 |
+
[04-20 02:27:09] Running warmup batch...
|
| 53 |
+
[04-20 02:27:09] Adjusting number of frames from 17 to 17 based on model
|
| 54 |
+
[04-20 02:27:09] Processing prompt 1/1: <redacted, len=49>
|
| 55 |
+
[04-20 02:27:09] Sampling params:
|
| 56 |
+
width: 960
|
| 57 |
+
height: 544
|
| 58 |
+
num_frames: 17
|
| 59 |
+
fps: 24
|
| 60 |
+
prompt: <redacted, len=49>
|
| 61 |
+
neg_prompt: <redacted, len=392>
|
| 62 |
+
seed: 0
|
| 63 |
+
infer_steps: 8
|
| 64 |
+
num_outputs_per_prompt: 1
|
| 65 |
+
guidance_scale: 1.0
|
| 66 |
+
embedded_guidance_scale: 6
|
| 67 |
+
n_tokens: None
|
| 68 |
+
flow_shift: 7
|
| 69 |
+
image_path: None
|
| 70 |
+
save_output: True
|
| 71 |
+
output_file_path: outputs/Random_prompt_0_for_benchmarking_diffusion_models_20260420-022709_b99ddeb1.mp4
|
| 72 |
+
|
| 73 |
+
[04-20 02:27:09] Running pipeline stages: ['InputValidationStage', 'prompt_encoding_stage_primary', 'TimestepPreparationStage', 'LatentPreparationStage', 'DenoisingStage', 'DecodingStage']
|
| 74 |
+
[04-20 02:27:09] [InputValidationStage] started...
|
| 75 |
+
[04-20 02:27:09] [InputValidationStage] finished in 0.0001 seconds
|
| 76 |
+
[04-20 02:27:09] [TextEncodingStage] started...
|
| 77 |
+
[04-20 02:27:10] [TextEncodingStage] finished in 1.0550 seconds
|
| 78 |
+
[04-20 02:27:10] [TimestepPreparationStage] started...
|
| 79 |
+
[04-20 02:27:10] [TimestepPreparationStage] finished in 0.0004 seconds
|
| 80 |
+
[04-20 02:27:10] [LatentPreparationStage] started...
|
| 81 |
+
[04-20 02:27:10] [LatentPreparationStage] finished in 0.0014 seconds
|
| 82 |
+
[04-20 02:27:10] [DenoisingStage] started...
|
| 83 |
+
|
| 84 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 85 |
12%|█▎ | 1/8 [00:02<00:14, 2.06s/it]
|
| 86 |
25%|██▌ | 2/8 [00:02<00:05, 1.10it/s]
|
| 87 |
38%|███▊ | 3/8 [00:02<00:03, 1.47it/s]
|
| 88 |
50%|█████ | 4/8 [00:03<00:02, 1.72it/s]
|
| 89 |
62%|██████▎ | 5/8 [00:03<00:01, 1.93it/s]
|
| 90 |
75%|███████▌ | 6/8 [00:03<00:00, 2.05it/s]
|
| 91 |
88%|████████▊ | 7/8 [00:04<00:00, 2.16it/s]
|
| 92 |
+
[04-20 02:27:15] [DenoisingStage] average time per step: 0.5834 seconds
|
| 93 |
+
[04-20 02:27:15] [DenoisingStage] finished in 4.6707 seconds
|
| 94 |
+
[04-20 02:27:15] [DecodingStage] started...
|
| 95 |
+
[04-20 02:27:21] [DecodingStage] finished in 5.4083 seconds
|
| 96 |
+
[04-20 02:27:21] Output saved to [1;36moutputs/Random_prompt_0_for_benchmarking_diffusion_models_20260420-022709_b99ddeb1.mp4[0;0m
|
| 97 |
+
[04-20 02:27:21] Pixel data generated successfully in [92m11.42[0;0m seconds
|
| 98 |
+
[04-20 02:27:21] Completed batch processing. Generated 1 outputs in [92m11.42[0;0m seconds
|
| 99 |
+
[04-20 02:27:21] Running benchmark with 3 prompts...
|
| 100 |
+
[04-20 02:27:21] Adjusting number of frames from 17 to 17 based on model
|
| 101 |
+
[04-20 02:27:21] Processing prompt 1/1: <redacted, len=49>
|
| 102 |
+
[04-20 02:27:21] Sampling params:
|
| 103 |
+
width: 960
|
| 104 |
+
height: 544
|
| 105 |
+
num_frames: 17
|
| 106 |
+
fps: 24
|
| 107 |
+
prompt: <redacted, len=49>
|
| 108 |
+
neg_prompt: <redacted, len=392>
|
| 109 |
+
seed: 0
|
| 110 |
+
infer_steps: 8
|
| 111 |
+
num_outputs_per_prompt: 1
|
| 112 |
+
guidance_scale: 1.0
|
| 113 |
+
embedded_guidance_scale: 6
|
| 114 |
+
n_tokens: None
|
| 115 |
+
flow_shift: 7
|
| 116 |
+
image_path: None
|
| 117 |
+
save_output: True
|
| 118 |
+
output_file_path: outputs/Random_prompt_0_for_benchmarking_diffusion_models_20260420-022721_b99ddeb1.mp4
|
| 119 |
+
|
| 120 |
+
[04-20 02:27:21] Running pipeline stages: ['InputValidationStage', 'prompt_encoding_stage_primary', 'TimestepPreparationStage', 'LatentPreparationStage', 'DenoisingStage', 'DecodingStage']
|
| 121 |
+
[04-20 02:27:21] [InputValidationStage] started...
|
| 122 |
+
[04-20 02:27:21] [InputValidationStage] finished in 0.0001 seconds
|
| 123 |
+
[04-20 02:27:21] [TextEncodingStage] started...
|
| 124 |
+
[04-20 02:27:21] [TextEncodingStage] finished in 0.3417 seconds
|
| 125 |
+
[04-20 02:27:21] [TimestepPreparationStage] started...
|
| 126 |
+
[04-20 02:27:21] [TimestepPreparationStage] finished in 0.0004 seconds
|
| 127 |
+
[04-20 02:27:21] [LatentPreparationStage] started...
|
| 128 |
+
[04-20 02:27:21] [LatentPreparationStage] finished in 0.0001 seconds
|
| 129 |
+
[04-20 02:27:21] [DenoisingStage] started...
|
| 130 |
+
|
| 131 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 132 |
12%|█▎ | 1/8 [00:00<00:00, 9.62it/s]
|
| 133 |
25%|██▌ | 2/8 [00:00<00:01, 3.56it/s]
|
| 134 |
38%|███▊ | 3/8 [00:00<00:01, 2.90it/s]
|
| 135 |
50%|█████ | 4/8 [00:01<00:01, 2.68it/s]
|
| 136 |
62%|██████▎ | 5/8 [00:01<00:01, 2.58it/s]
|
| 137 |
75%|███████▌ | 6/8 [00:02<00:00, 2.49it/s]
|
| 138 |
88%|████████▊ | 7/8 [00:02<00:00, 2.47it/s]
|
| 139 |
+
[04-20 02:27:24] [DenoisingStage] average time per step: 0.3779 seconds
|
| 140 |
+
[04-20 02:27:24] [DenoisingStage] finished in 3.0250 seconds
|
| 141 |
+
[04-20 02:27:24] [DecodingStage] started...
|
| 142 |
+
[04-20 02:27:29] [DecodingStage] finished in 5.0738 seconds
|
| 143 |
+
[04-20 02:27:29] Output saved to [1;36moutputs/Random_prompt_0_for_benchmarking_diffusion_models_20260420-022721_b99ddeb1.mp4[0;0m
|
| 144 |
+
[04-20 02:27:29] Pixel data generated successfully in [92m8.70[0;0m seconds
|
| 145 |
+
[04-20 02:27:29] Completed batch processing. Generated 1 outputs in [92m8.70[0;0m seconds
|
| 146 |
+
[04-20 02:27:29] Adjusting number of frames from 17 to 17 based on model
|
| 147 |
+
[04-20 02:27:29] Processing prompt 1/1: <redacted, len=49>
|
| 148 |
+
[04-20 02:27:29] Sampling params:
|
| 149 |
+
width: 960
|
| 150 |
+
height: 544
|
| 151 |
+
num_frames: 17
|
| 152 |
+
fps: 24
|
| 153 |
+
prompt: <redacted, len=49>
|
| 154 |
+
neg_prompt: <redacted, len=392>
|
| 155 |
+
seed: 0
|
| 156 |
+
infer_steps: 8
|
| 157 |
+
num_outputs_per_prompt: 1
|
| 158 |
+
guidance_scale: 1.0
|
| 159 |
+
embedded_guidance_scale: 6
|
| 160 |
+
n_tokens: None
|
| 161 |
+
flow_shift: 7
|
| 162 |
+
image_path: None
|
| 163 |
+
save_output: True
|
| 164 |
+
output_file_path: outputs/Random_prompt_1_for_benchmarking_diffusion_models_20260420-022729_8d91326e.mp4
|
| 165 |
+
|
| 166 |
+
[04-20 02:27:29] Running pipeline stages: ['InputValidationStage', 'prompt_encoding_stage_primary', 'TimestepPreparationStage', 'LatentPreparationStage', 'DenoisingStage', 'DecodingStage']
|
| 167 |
+
[04-20 02:27:29] [InputValidationStage] started...
|
| 168 |
+
[04-20 02:27:29] [InputValidationStage] finished in 0.0001 seconds
|
| 169 |
+
[04-20 02:27:29] [TextEncodingStage] started...
|
| 170 |
+
[04-20 02:27:30] [TextEncodingStage] finished in 0.3422 seconds
|
| 171 |
+
[04-20 02:27:30] [TimestepPreparationStage] started...
|
| 172 |
+
[04-20 02:27:30] [TimestepPreparationStage] finished in 0.0004 seconds
|
| 173 |
+
[04-20 02:27:30] [LatentPreparationStage] started...
|
| 174 |
+
[04-20 02:27:30] [LatentPreparationStage] finished in 0.0001 seconds
|
| 175 |
+
[04-20 02:27:30] [DenoisingStage] started...
|
| 176 |
+
|
| 177 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 178 |
12%|█▎ | 1/8 [00:00<00:00, 9.66it/s]
|
| 179 |
25%|██▌ | 2/8 [00:00<00:01, 3.54it/s]
|
| 180 |
38%|███▊ | 3/8 [00:00<00:01, 2.89it/s]
|
| 181 |
50%|█████ | 4/8 [00:01<00:01, 2.67it/s]
|
| 182 |
62%|██████▎ | 5/8 [00:01<00:01, 2.57it/s]
|
| 183 |
75%|███████▌ | 6/8 [00:02<00:00, 2.47it/s]
|
| 184 |
88%|████████▊ | 7/8 [00:02<00:00, 2.46it/s]
|
| 185 |
+
[04-20 02:27:33] [DenoisingStage] average time per step: 0.3794 seconds
|
| 186 |
+
[04-20 02:27:33] [DenoisingStage] finished in 3.0371 seconds
|
| 187 |
+
[04-20 02:27:33] [DecodingStage] started...
|
| 188 |
+
[04-20 02:27:38] [DecodingStage] finished in 5.0608 seconds
|
| 189 |
+
[04-20 02:27:38] Output saved to [1;36moutputs/Random_prompt_1_for_benchmarking_diffusion_models_20260420-022729_8d91326e.mp4[0;0m
|
| 190 |
+
[04-20 02:27:38] Pixel data generated successfully in [92m8.70[0;0m seconds
|
| 191 |
+
[04-20 02:27:38] Completed batch processing. Generated 1 outputs in [92m8.70[0;0m seconds
|
| 192 |
+
[04-20 02:27:38] Adjusting number of frames from 17 to 17 based on model
|
| 193 |
+
[04-20 02:27:38] Processing prompt 1/1: <redacted, len=49>
|
| 194 |
+
[04-20 02:27:38] Sampling params:
|
| 195 |
+
width: 960
|
| 196 |
+
height: 544
|
| 197 |
+
num_frames: 17
|
| 198 |
+
fps: 24
|
| 199 |
+
prompt: <redacted, len=49>
|
| 200 |
+
neg_prompt: <redacted, len=392>
|
| 201 |
+
seed: 0
|
| 202 |
+
infer_steps: 8
|
| 203 |
+
num_outputs_per_prompt: 1
|
| 204 |
+
guidance_scale: 1.0
|
| 205 |
+
embedded_guidance_scale: 6
|
| 206 |
+
n_tokens: None
|
| 207 |
+
flow_shift: 7
|
| 208 |
+
image_path: None
|
| 209 |
+
save_output: True
|
| 210 |
+
output_file_path: outputs/Random_prompt_2_for_benchmarking_diffusion_models_20260420-022738_9a3de0d5.mp4
|
| 211 |
+
|
| 212 |
+
[04-20 02:27:38] Running pipeline stages: ['InputValidationStage', 'prompt_encoding_stage_primary', 'TimestepPreparationStage', 'LatentPreparationStage', 'DenoisingStage', 'DecodingStage']
|
| 213 |
+
[04-20 02:27:38] [InputValidationStage] started...
|
| 214 |
+
[04-20 02:27:38] [InputValidationStage] finished in 0.0000 seconds
|
| 215 |
+
[04-20 02:27:38] [TextEncodingStage] started...
|
| 216 |
+
[04-20 02:27:39] [TextEncodingStage] finished in 0.3417 seconds
|
| 217 |
+
[04-20 02:27:39] [TimestepPreparationStage] started...
|
| 218 |
+
[04-20 02:27:39] [TimestepPreparationStage] finished in 0.0004 seconds
|
| 219 |
+
[04-20 02:27:39] [LatentPreparationStage] started...
|
| 220 |
+
[04-20 02:27:39] [LatentPreparationStage] finished in 0.0001 seconds
|
| 221 |
+
[04-20 02:27:39] [DenoisingStage] started...
|
| 222 |
+
|
| 223 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 224 |
12%|█▎ | 1/8 [00:00<00:00, 9.68it/s]
|
| 225 |
25%|██▌ | 2/8 [00:00<00:01, 3.55it/s]
|
| 226 |
38%|███▊ | 3/8 [00:00<00:01, 2.88it/s]
|
| 227 |
50%|█████ | 4/8 [00:01<00:01, 2.66it/s]
|
| 228 |
62%|██████▎ | 5/8 [00:01<00:01, 2.57it/s]
|
| 229 |
75%|███████▌ | 6/8 [00:02<00:00, 2.48it/s]
|
| 230 |
88%|████████▊ | 7/8 [00:02<00:00, 2.46it/s]
|
| 231 |
+
[04-20 02:27:42] [DenoisingStage] average time per step: 0.3796 seconds
|
| 232 |
+
[04-20 02:27:42] [DenoisingStage] finished in 3.0391 seconds
|
| 233 |
+
[04-20 02:27:42] [DecodingStage] started...
|
| 234 |
+
[04-20 02:27:47] [DecodingStage] finished in 5.0717 seconds
|
| 235 |
+
[04-20 02:27:47] Output saved to [1;36moutputs/Random_prompt_2_for_benchmarking_diffusion_models_20260420-022738_9a3de0d5.mp4[0;0m
|
| 236 |
+
[04-20 02:27:47] Pixel data generated successfully in [92m8.71[0;0m seconds
|
| 237 |
+
[04-20 02:27:47] Completed batch processing. Generated 1 outputs in [92m8.71[0;0m seconds
|
| 238 |
+
|
| 239 |
+
==================================== Offline Throughput Benchmark Result =====================================
|
| 240 |
+
Model: /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773
|
| 241 |
+
Dataset: random
|
| 242 |
+
Resolution: 960x544x17
|
| 243 |
+
Num Inference Steps: 8
|
| 244 |
+
---------------------------------------------------------------------------
|
| 245 |
+
Total Requests: 3
|
| 246 |
+
Successful Requests: 3
|
| 247 |
+
Failed Requests: 0
|
| 248 |
+
Total Duration (seconds): 26.11
|
| 249 |
+
---------------------------------------------------------------------------
|
| 250 |
+
Frames Generated: 3
|
| 251 |
+
Megapixels Generated: 26.63
|
| 252 |
+
---------------------------------------------------------------------------
|
| 253 |
+
Frame Throughput (frames/sec): 0.11
|
| 254 |
+
MP Throughput (MP/sec): 1.02
|
| 255 |
+
Requests Per Second: 0.11
|
| 256 |
+
Latency Per Request (sec): 8.70
|
| 257 |
+
Peak Memory (MB): 0.00
|
| 258 |
+
==============================================================================================================
|
| 259 |
+
[04-20 02:27:47] Results saved to /data/bbuf/hunyuanvideo_fp8_20260420/benchmark/bf16_offline_throughput.jsonl
|
| 260 |
+
[93m[04-20 02:27:47] Generator was garbage collected without being shut down. Attempting to shut down the local server and client.[0;0m
|
| 261 |
+
[04-20 02:27:54] Worker 0: Shutdown complete.
|
validation/h100_20260420/logs/bench_fp8_transformer_path.log
ADDED
|
@@ -0,0 +1,231 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 1 |
12%|█▎ | 1/8 [00:01<00:12, 1.86s/it]
|
| 2 |
38%|███▊ | 3/8 [00:02<00:03, 1.55it/s]
|
| 3 |
50%|█████ | 4/8 [00:02<00:02, 1.84it/s]
|
| 4 |
62%|██████▎ | 5/8 [00:02<00:01, 2.09it/s]
|
| 5 |
75%|███████▌ | 6/8 [00:03<00:00, 2.28it/s]
|
| 6 |
88%|████████▊ | 7/8 [00:03<00:00, 2.42it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 7 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 8 |
25%|██▌ | 2/8 [00:00<00:01, 4.59it/s]
|
| 9 |
38%|███▊ | 3/8 [00:00<00:01, 3.67it/s]
|
| 10 |
50%|█████ | 4/8 [00:01<00:01, 3.28it/s]
|
| 11 |
62%|██████▎ | 5/8 [00:01<00:00, 3.12it/s]
|
| 12 |
75%|███████▌ | 6/8 [00:01<00:00, 3.04it/s]
|
| 13 |
88%|████████▊ | 7/8 [00:02<00:00, 2.96it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 14 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 15 |
25%|██▌ | 2/8 [00:00<00:01, 4.55it/s]
|
| 16 |
38%|███▊ | 3/8 [00:00<00:01, 3.65it/s]
|
| 17 |
50%|█████ | 4/8 [00:01<00:01, 3.26it/s]
|
| 18 |
62%|██████▎ | 5/8 [00:01<00:00, 3.11it/s]
|
| 19 |
75%|███████▌ | 6/8 [00:01<00:00, 3.02it/s]
|
| 20 |
88%|████████▊ | 7/8 [00:02<00:00, 2.94it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 21 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 22 |
25%|██▌ | 2/8 [00:00<00:01, 4.59it/s]
|
| 23 |
38%|███▊ | 3/8 [00:00<00:01, 3.66it/s]
|
| 24 |
50%|█████ | 4/8 [00:01<00:01, 3.27it/s]
|
| 25 |
62%|██████▎ | 5/8 [00:01<00:00, 3.11it/s]
|
| 26 |
75%|███████▌ | 6/8 [00:01<00:00, 3.02it/s]
|
| 27 |
88%|████████▊ | 7/8 [00:02<00:00, 2.94it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
No platform detected. Using base SRTPlatform with defaults.
|
| 2 |
+
[04-20 02:49:09] Port 30005 was unavailable, using port 30042 instead
|
| 3 |
+
[04-20 02:49:09] server_args: {"model_path": "/root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773", "model_id": null, "backend": "sglang", "attention_backend": null, "attention_backend_config": {}, "cache_dit_config": null, "nccl_port": null, "trust_remote_code": false, "revision": null, "num_gpus": 1, "tp_size": 1, "sp_degree": 1, "ulysses_degree": 1, "ring_degree": 1, "dp_size": 1, "dp_degree": 1, "enable_cfg_parallel": false, "hsdp_replicate_dim": 1, "hsdp_shard_dim": 1, "dist_timeout": 3600, "pipeline_class_name": null, "lora_path": null, "lora_nickname": "default", "lora_scale": 1.0, "lora_weight_name": null, "component_paths": {"transformer": "/data/bbuf/hunyuanvideo_fp8_20260420/sglang_transformer"}, "transformer_weights_path": null, "lora_target_modules": null, "dit_cpu_offload": false, "dit_layerwise_offload": false, "dit_offload_prefetch_size": 0.0, "text_encoder_cpu_offload": true, "image_encoder_cpu_offload": true, "vae_cpu_offload": true, "use_fsdp_inference": false, "pin_cpu_memory": true, "ltx2_two_stage_device_mode": null, "comfyui_mode": false, "enable_torch_compile": false, "warmup": false, "warmup_resolutions": null, "warmup_steps": 1, "disable_autocast": false, "master_port": 30042, "host": "127.0.0.1", "port": 30000, "webui": false, "webui_port": 12312, "scheduler_port": 5570, "strict_ports": false, "output_path": "outputs/", "input_save_path": "inputs/uploads", "prompt_file_path": null, "model_paths": {}, "model_loaded": {"transformer": true, "vae": true, "video_vae": true, "audio_vae": true, "video_dit": true, "audio_dit": true, "dual_tower_bridge": true}, "boundary_ratio": null, "base_gpu_id": 0, "disagg_role": "monolithic", "disagg_timeout": 600, "disagg_dispatch_policy": "round_robin", "disagg_mode": false, "disagg_server_addr": null, "encoder_urls": null, "denoiser_urls": null, "decoder_urls": null, "encoder_tp": null, "denoiser_tp": null, "denoiser_sp": null, "denoiser_ulysses": null, "denoiser_ring": null, "decoder_tp": null, "disagg_transfer_pool_size": 268435456, "disagg_p2p_hostname": "127.0.0.1", "disagg_ib_device": null, "pool_work_endpoint": null, "pool_result_endpoint": null, "log_level": "info", "uvicorn_access_log_exclude_prefixes": []}
|
| 4 |
+
[04-20 02:49:09] Starting offline throughput benchmark...
|
| 5 |
+
[04-20 02:49:09] Initializing engine...
|
| 6 |
+
[04-20 02:49:09] Local mode: True
|
| 7 |
+
[04-20 02:49:09] Starting server...
|
| 8 |
+
No platform detected. Using base SRTPlatform with defaults.
|
| 9 |
+
[04-20 02:49:17] Scheduler bind at endpoint: tcp://127.0.0.1:5570
|
| 10 |
+
[04-20 02:49:17] Initializing distributed environment with world_size=1, device=cuda:0, timeout=3600
|
| 11 |
+
[04-20 02:49:17] Setting distributed timeout to 3600 seconds
|
| 12 |
+
[04-20 02:49:18] No pipeline_class_name specified, using model_index.json
|
| 13 |
+
[04-20 02:49:18] Diffusers version: 0.32.0.dev0
|
| 14 |
+
[04-20 02:49:18] Using pipeline from model_index.json: HunyuanVideoPipeline
|
| 15 |
+
[04-20 02:49:18] Loading pipeline modules...
|
| 16 |
+
[04-20 02:49:18] Model already exists locally and is complete
|
| 17 |
+
[04-20 02:49:18] Model path: /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773
|
| 18 |
+
[04-20 02:49:18] Diffusers version: 0.32.0.dev0
|
| 19 |
+
[04-20 02:49:18] Loading pipeline modules from config: {'_class_name': 'HunyuanVideoPipeline', '_diffusers_version': '0.32.0.dev0', 'scheduler': ['diffusers', 'FlowMatchEulerDiscreteScheduler'], 'text_encoder': ['transformers', 'LlamaModel'], 'text_encoder_2': ['transformers', 'CLIPTextModel'], 'tokenizer': ['transformers', 'LlamaTokenizerFast'], 'tokenizer_2': ['transformers', 'CLIPTokenizer'], 'transformer': ['diffusers', 'HunyuanVideoTransformer3DModel'], 'vae': ['diffusers', 'AutoencoderKLHunyuanVideo']}
|
| 20 |
+
[04-20 02:49:18] Loading required components: ['text_encoder', 'text_encoder_2', 'tokenizer', 'tokenizer_2', 'vae', 'transformer', 'scheduler']
|
| 21 |
+
|
| 22 |
+
[04-20 02:49:18] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
|
| 23 |
+
[04-20 02:49:20] [RunAI Streamer] Overall time to stream 14.0 GiB of all files to cpu: 1.94s, 7.2 GiB/s
|
| 24 |
+
[04-20 02:49:37] Loaded text_encoder: FSDPLlamaModel (sgl-diffusion version). model size: 13.98 GB, consumed GPU mem: 0.81 GB, avail GPU mem: 73.04 GB
|
| 25 |
+
|
| 26 |
+
[04-20 02:49:37] Using Torch SDPA backend
|
| 27 |
+
[04-20 02:49:37] [RunAI Streamer] Overall time to stream 234.7 MiB of all files to cpu: 0.03s, 6.9 GiB/s
|
| 28 |
+
[04-20 02:49:38] Loaded text_encoder_2: FSDPCLIPTextModel (sgl-diffusion version). model size: 0.23 GB, consumed GPU mem: 0.76 GB, avail GPU mem: 72.27 GB
|
| 29 |
+
|
| 30 |
+
[04-20 02:49:39] Loaded tokenizer: LlamaTokenizer (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
|
| 31 |
+
|
| 32 |
+
[04-20 02:49:39] Loaded tokenizer_2: CLIPTokenizer (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
|
| 33 |
+
[04-20 02:49:39] Loading vae from /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773/vae. avail mem: 72.27 GB
|
| 34 |
+
[93m[04-20 02:49:39] VAE unexpected keys: ['encoder.conv_in.conv.bias', 'encoder.conv_in.conv.weight', 'encoder.conv_norm_out.bias', 'encoder.conv_norm_out.weight', 'encoder.conv_out.conv.bias', 'encoder.conv_out.conv.weight', 'encoder.down_blocks.0.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.0.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.0.resnets.0.conv1.conv.bias', 'encoder.down_blocks.0.resnets.0.conv1.conv.weight', 'encoder.down_blocks.0.resnets.0.conv2.conv.bias', 'encoder.down_blocks.0.resnets.0.conv2.conv.weight', 'encoder.down_blocks.0.resnets.0.norm1.bias', 'encoder.down_blocks.0.resnets.0.norm1.weight', 'encoder.down_blocks.0.resnets.0.norm2.bias', 'encoder.down_blocks.0.resnets.0.norm2.weight', 'encoder.down_blocks.0.resnets.1.conv1.conv.bias', 'encoder.down_blocks.0.resnets.1.conv1.conv.weight', 'encoder.down_blocks.0.resnets.1.conv2.conv.bias', 'encoder.down_blocks.0.resnets.1.conv2.conv.weight', 'encoder.down_blocks.0.resnets.1.norm1.bias', 'encoder.down_blocks.0.resnets.1.norm1.weight', 'encoder.down_blocks.0.resnets.1.norm2.bias', 'encoder.down_blocks.0.resnets.1.norm2.weight', 'encoder.down_blocks.1.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.1.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.1.resnets.0.conv1.conv.bias', 'encoder.down_blocks.1.resnets.0.conv1.conv.weight', 'encoder.down_blocks.1.resnets.0.conv2.conv.bias', 'encoder.down_blocks.1.resnets.0.conv2.conv.weight', 'encoder.down_blocks.1.resnets.0.conv_shortcut.conv.bias', 'encoder.down_blocks.1.resnets.0.conv_shortcut.conv.weight', 'encoder.down_blocks.1.resnets.0.norm1.bias', 'encoder.down_blocks.1.resnets.0.norm1.weight', 'encoder.down_blocks.1.resnets.0.norm2.bias', 'encoder.down_blocks.1.resnets.0.norm2.weight', 'encoder.down_blocks.1.resnets.1.conv1.conv.bias', 'encoder.down_blocks.1.resnets.1.conv1.conv.weight', 'encoder.down_blocks.1.resnets.1.conv2.conv.bias', 'encoder.down_blocks.1.resnets.1.conv2.conv.weight', 'encoder.down_blocks.1.resnets.1.norm1.bias', 'encoder.down_blocks.1.resnets.1.norm1.weight', 'encoder.down_blocks.1.resnets.1.norm2.bias', 'encoder.down_blocks.1.resnets.1.norm2.weight', 'encoder.down_blocks.2.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.2.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.2.resnets.0.conv1.conv.bias', 'encoder.down_blocks.2.resnets.0.conv1.conv.weight', 'encoder.down_blocks.2.resnets.0.conv2.conv.bias', 'encoder.down_blocks.2.resnets.0.conv2.conv.weight', 'encoder.down_blocks.2.resnets.0.conv_shortcut.conv.bias', 'encoder.down_blocks.2.resnets.0.conv_shortcut.conv.weight', 'encoder.down_blocks.2.resnets.0.norm1.bias', 'encoder.down_blocks.2.resnets.0.norm1.weight', 'encoder.down_blocks.2.resnets.0.norm2.bias', 'encoder.down_blocks.2.resnets.0.norm2.weight', 'encoder.down_blocks.2.resnets.1.conv1.conv.bias', 'encoder.down_blocks.2.resnets.1.conv1.conv.weight', 'encoder.down_blocks.2.resnets.1.conv2.conv.bias', 'encoder.down_blocks.2.resnets.1.conv2.conv.weight', 'encoder.down_blocks.2.resnets.1.norm1.bias', 'encoder.down_blocks.2.resnets.1.norm1.weight', 'encoder.down_blocks.2.resnets.1.norm2.bias', 'encoder.down_blocks.2.resnets.1.norm2.weight', 'encoder.down_blocks.3.resnets.0.conv1.conv.bias', 'encoder.down_blocks.3.resnets.0.conv1.conv.weight', 'encoder.down_blocks.3.resnets.0.conv2.conv.bias', 'encoder.down_blocks.3.resnets.0.conv2.conv.weight', 'encoder.down_blocks.3.resnets.0.norm1.bias', 'encoder.down_blocks.3.resnets.0.norm1.weight', 'encoder.down_blocks.3.resnets.0.norm2.bias', 'encoder.down_blocks.3.resnets.0.norm2.weight', 'encoder.down_blocks.3.resnets.1.conv1.conv.bias', 'encoder.down_blocks.3.resnets.1.conv1.conv.weight', 'encoder.down_blocks.3.resnets.1.conv2.conv.bias', 'encoder.down_blocks.3.resnets.1.conv2.conv.weight', 'encoder.down_blocks.3.resnets.1.norm1.bias', 'encoder.down_blocks.3.resnets.1.norm1.weight', 'encoder.down_blocks.3.resnets.1.norm2.bias', 'encoder.down_blocks.3.resnets.1.norm2.weight', 'encoder.mid_block.attentions.0.group_norm.bias', 'encoder.mid_block.attentions.0.group_norm.weight', 'encoder.mid_block.attentions.0.to_k.bias', 'encoder.mid_block.attentions.0.to_k.weight', 'encoder.mid_block.attentions.0.to_out.0.bias', 'encoder.mid_block.attentions.0.to_out.0.weight', 'encoder.mid_block.attentions.0.to_q.bias', 'encoder.mid_block.attentions.0.to_q.weight', 'encoder.mid_block.attentions.0.to_v.bias', 'encoder.mid_block.attentions.0.to_v.weight', 'encoder.mid_block.resnets.0.conv1.conv.bias', 'encoder.mid_block.resnets.0.conv1.conv.weight', 'encoder.mid_block.resnets.0.conv2.conv.bias', 'encoder.mid_block.resnets.0.conv2.conv.weight', 'encoder.mid_block.resnets.0.norm1.bias', 'encoder.mid_block.resnets.0.norm1.weight', 'encoder.mid_block.resnets.0.norm2.bias', 'encoder.mid_block.resnets.0.norm2.weight', 'encoder.mid_block.resnets.1.conv1.conv.bias', 'encoder.mid_block.resnets.1.conv1.conv.weight', 'encoder.mid_block.resnets.1.conv2.conv.bias', 'encoder.mid_block.resnets.1.conv2.conv.weight', 'encoder.mid_block.resnets.1.norm1.bias', 'encoder.mid_block.resnets.1.norm1.weight', 'encoder.mid_block.resnets.1.norm2.bias', 'encoder.mid_block.resnets.1.norm2.weight', 'quant_conv.bias', 'quant_conv.weight'][0;0m
|
| 35 |
+
[04-20 02:49:39] Loaded vae: AutoencoderKLHunyuanVideo (sgl-diffusion version). model size: 0.27 GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
|
| 36 |
+
[04-20 02:49:39] Loading transformer from /data/bbuf/hunyuanvideo_fp8_20260420/sglang_transformer. avail mem: 72.27 GB
|
| 37 |
+
[93m[04-20 02:49:39] Detected ModelOpt FP8 checkpoint. The format is experimental and subject to change.[0;0m
|
| 38 |
+
[04-20 02:49:39] Loading HunyuanVideoTransformer3DModel from 3 safetensors file(s) , param_dtype: None
|
| 39 |
+
[04-20 02:49:39] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
|
| 40 |
+
[04-20 02:49:39] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
|
| 41 |
+
[04-20 02:49:40] [RunAI Streamer] Overall time to stream 15.4 GiB of all files to cpu: 1.3s, 11.9 GiB/s
|
| 42 |
+
[04-20 02:49:43] Loaded model with 12.82B parameters
|
| 43 |
+
[04-20 02:49:43] Loaded transformer: HunyuanVideoTransformer3DModel (sgl-diffusion version). model size: 15.45 GB, consumed GPU mem: 15.57 GB, avail GPU mem: 56.71 GB
|
| 44 |
+
|
| 45 |
+
[04-20 02:49:43] Loaded scheduler: FlowMatchEulerDiscreteScheduler (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 56.71 GB
|
| 46 |
+
|
| 47 |
+
[04-20 02:49:43] Creating pipeline stages...
|
| 48 |
+
[04-20 02:49:43] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
|
| 49 |
+
[04-20 02:49:43] Pipeline instantiated
|
| 50 |
+
[04-20 02:49:43] Worker 0: Initialized device, model, and distributed environment.
|
| 51 |
+
[04-20 02:49:43] Worker 0: Scheduler loop started.
|
| 52 |
+
[04-20 02:49:43] Engine initialized successfully
|
| 53 |
+
[04-20 02:49:43] Loading random dataset...
|
| 54 |
+
[04-20 02:49:43] Running warmup batch...
|
| 55 |
+
[04-20 02:49:43] Adjusting number of frames from 17 to 17 based on model
|
| 56 |
+
[04-20 02:49:43] Processing prompt 1/1: <redacted, len=49>
|
| 57 |
+
[04-20 02:49:43] Sampling params:
|
| 58 |
+
width: 960
|
| 59 |
+
height: 544
|
| 60 |
+
num_frames: 17
|
| 61 |
+
fps: 24
|
| 62 |
+
prompt: <redacted, len=49>
|
| 63 |
+
neg_prompt: <redacted, len=392>
|
| 64 |
+
seed: 0
|
| 65 |
+
infer_steps: 8
|
| 66 |
+
num_outputs_per_prompt: 1
|
| 67 |
+
guidance_scale: 1.0
|
| 68 |
+
embedded_guidance_scale: 6
|
| 69 |
+
n_tokens: None
|
| 70 |
+
flow_shift: 7
|
| 71 |
+
image_path: None
|
| 72 |
+
save_output: True
|
| 73 |
+
output_file_path: outputs/Random_prompt_0_for_benchmarking_diffusion_models_20260420-024943_b99ddeb1.mp4
|
| 74 |
+
|
| 75 |
+
[04-20 02:49:43] Running pipeline stages: ['InputValidationStage', 'prompt_encoding_stage_primary', 'TimestepPreparationStage', 'LatentPreparationStage', 'DenoisingStage', 'DecodingStage']
|
| 76 |
+
[04-20 02:49:43] [InputValidationStage] started...
|
| 77 |
+
[04-20 02:49:43] [InputValidationStage] finished in 0.0000 seconds
|
| 78 |
+
[04-20 02:49:43] [TextEncodingStage] started...
|
| 79 |
+
[04-20 02:49:44] [TextEncodingStage] finished in 1.0464 seconds
|
| 80 |
+
[04-20 02:49:44] [TimestepPreparationStage] started...
|
| 81 |
+
[04-20 02:49:44] [TimestepPreparationStage] finished in 0.0004 seconds
|
| 82 |
+
[04-20 02:49:44] [LatentPreparationStage] started...
|
| 83 |
+
[04-20 02:49:44] [LatentPreparationStage] finished in 0.0008 seconds
|
| 84 |
+
[04-20 02:49:44] [DenoisingStage] started...
|
| 85 |
+
|
| 86 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 87 |
12%|█▎ | 1/8 [00:01<00:12, 1.86s/it]
|
| 88 |
38%|███▊ | 3/8 [00:02<00:03, 1.55it/s]
|
| 89 |
50%|█████ | 4/8 [00:02<00:02, 1.84it/s]
|
| 90 |
62%|██████▎ | 5/8 [00:02<00:01, 2.09it/s]
|
| 91 |
75%|███████▌ | 6/8 [00:03<00:00, 2.28it/s]
|
| 92 |
88%|████████▊ | 7/8 [00:03<00:00, 2.42it/s]
|
| 93 |
+
[04-20 02:49:48] [DenoisingStage] average time per step: 0.5073 seconds
|
| 94 |
+
[04-20 02:49:48] [DenoisingStage] finished in 4.0600 seconds
|
| 95 |
+
[04-20 02:49:48] [DecodingStage] started...
|
| 96 |
+
[04-20 02:49:54] [DecodingStage] finished in 5.3690 seconds
|
| 97 |
+
[04-20 02:49:54] Output saved to [1;36moutputs/Random_prompt_0_for_benchmarking_diffusion_models_20260420-024943_b99ddeb1.mp4[0;0m
|
| 98 |
+
[04-20 02:49:54] Pixel data generated successfully in [92m10.74[0;0m seconds
|
| 99 |
+
[04-20 02:49:54] Completed batch processing. Generated 1 outputs in [92m10.74[0;0m seconds
|
| 100 |
+
[04-20 02:49:54] Running benchmark with 3 prompts...
|
| 101 |
+
[04-20 02:49:54] Adjusting number of frames from 17 to 17 based on model
|
| 102 |
+
[04-20 02:49:54] Processing prompt 1/1: <redacted, len=49>
|
| 103 |
+
[04-20 02:49:54] Sampling params:
|
| 104 |
+
width: 960
|
| 105 |
+
height: 544
|
| 106 |
+
num_frames: 17
|
| 107 |
+
fps: 24
|
| 108 |
+
prompt: <redacted, len=49>
|
| 109 |
+
neg_prompt: <redacted, len=392>
|
| 110 |
+
seed: 0
|
| 111 |
+
infer_steps: 8
|
| 112 |
+
num_outputs_per_prompt: 1
|
| 113 |
+
guidance_scale: 1.0
|
| 114 |
+
embedded_guidance_scale: 6
|
| 115 |
+
n_tokens: None
|
| 116 |
+
flow_shift: 7
|
| 117 |
+
image_path: None
|
| 118 |
+
save_output: True
|
| 119 |
+
output_file_path: outputs/Random_prompt_0_for_benchmarking_diffusion_models_20260420-024954_b99ddeb1.mp4
|
| 120 |
+
|
| 121 |
+
[04-20 02:49:54] Running pipeline stages: ['InputValidationStage', 'prompt_encoding_stage_primary', 'TimestepPreparationStage', 'LatentPreparationStage', 'DenoisingStage', 'DecodingStage']
|
| 122 |
+
[04-20 02:49:54] [InputValidationStage] started...
|
| 123 |
+
[04-20 02:49:54] [InputValidationStage] finished in 0.0001 seconds
|
| 124 |
+
[04-20 02:49:54] [TextEncodingStage] started...
|
| 125 |
+
[04-20 02:49:54] [TextEncodingStage] finished in 0.3712 seconds
|
| 126 |
+
[04-20 02:49:54] [TimestepPreparationStage] started...
|
| 127 |
+
[04-20 02:49:54] [TimestepPreparationStage] finished in 0.0004 seconds
|
| 128 |
+
[04-20 02:49:54] [LatentPreparationStage] started...
|
| 129 |
+
[04-20 02:49:54] [LatentPreparationStage] finished in 0.0001 seconds
|
| 130 |
+
[04-20 02:49:54] [DenoisingStage] started...
|
| 131 |
+
|
| 132 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 133 |
25%|██▌ | 2/8 [00:00<00:01, 4.59it/s]
|
| 134 |
38%|███▊ | 3/8 [00:00<00:01, 3.67it/s]
|
| 135 |
50%|█████ | 4/8 [00:01<00:01, 3.28it/s]
|
| 136 |
62%|██████▎ | 5/8 [00:01<00:00, 3.12it/s]
|
| 137 |
75%|███████▌ | 6/8 [00:01<00:00, 3.04it/s]
|
| 138 |
88%|████████▊ | 7/8 [00:02<00:00, 2.96it/s]
|
| 139 |
+
[04-20 02:49:57] [DenoisingStage] average time per step: 0.3184 seconds
|
| 140 |
+
[04-20 02:49:57] [DenoisingStage] finished in 2.5504 seconds
|
| 141 |
+
[04-20 02:49:57] [DecodingStage] started...
|
| 142 |
+
[04-20 02:50:02] [DecodingStage] finished in 5.2985 seconds
|
| 143 |
+
[04-20 02:50:02] Output saved to [1;36moutputs/Random_prompt_0_for_benchmarking_diffusion_models_20260420-024954_b99ddeb1.mp4[0;0m
|
| 144 |
+
[04-20 02:50:02] Pixel data generated successfully in [92m8.54[0;0m seconds
|
| 145 |
+
[04-20 02:50:02] Completed batch processing. Generated 1 outputs in [92m8.54[0;0m seconds
|
| 146 |
+
[04-20 02:50:02] Adjusting number of frames from 17 to 17 based on model
|
| 147 |
+
[04-20 02:50:02] Processing prompt 1/1: <redacted, len=49>
|
| 148 |
+
[04-20 02:50:02] Sampling params:
|
| 149 |
+
width: 960
|
| 150 |
+
height: 544
|
| 151 |
+
num_frames: 17
|
| 152 |
+
fps: 24
|
| 153 |
+
prompt: <redacted, len=49>
|
| 154 |
+
neg_prompt: <redacted, len=392>
|
| 155 |
+
seed: 0
|
| 156 |
+
infer_steps: 8
|
| 157 |
+
num_outputs_per_prompt: 1
|
| 158 |
+
guidance_scale: 1.0
|
| 159 |
+
embedded_guidance_scale: 6
|
| 160 |
+
n_tokens: None
|
| 161 |
+
flow_shift: 7
|
| 162 |
+
image_path: None
|
| 163 |
+
save_output: True
|
| 164 |
+
output_file_path: outputs/Random_prompt_1_for_benchmarking_diffusion_models_20260420-025002_8d91326e.mp4
|
| 165 |
+
|
| 166 |
+
[04-20 02:50:02] Running pipeline stages: ['InputValidationStage', 'prompt_encoding_stage_primary', 'TimestepPreparationStage', 'LatentPreparationStage', 'DenoisingStage', 'DecodingStage']
|
| 167 |
+
[04-20 02:50:02] [InputValidationStage] started...
|
| 168 |
+
[04-20 02:50:02] [InputValidationStage] finished in 0.0001 seconds
|
| 169 |
+
[04-20 02:50:02] [TextEncodingStage] started...
|
| 170 |
+
[04-20 02:50:03] [TextEncodingStage] finished in 0.3517 seconds
|
| 171 |
+
[04-20 02:50:03] [TimestepPreparationStage] started...
|
| 172 |
+
[04-20 02:50:03] [TimestepPreparationStage] finished in 0.0008 seconds
|
| 173 |
+
[04-20 02:50:03] [LatentPreparationStage] started...
|
| 174 |
+
[04-20 02:50:03] [LatentPreparationStage] finished in 0.0002 seconds
|
| 175 |
+
[04-20 02:50:03] [DenoisingStage] started...
|
| 176 |
+
|
| 177 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 178 |
25%|██▌ | 2/8 [00:00<00:01, 4.55it/s]
|
| 179 |
38%|███▊ | 3/8 [00:00<00:01, 3.65it/s]
|
| 180 |
50%|█████ | 4/8 [00:01<00:01, 3.26it/s]
|
| 181 |
62%|██████▎ | 5/8 [00:01<00:00, 3.11it/s]
|
| 182 |
75%|███████▌ | 6/8 [00:01<00:00, 3.02it/s]
|
| 183 |
88%|████████▊ | 7/8 [00:02<00:00, 2.94it/s]
|
| 184 |
+
[04-20 02:50:05] [DenoisingStage] average time per step: 0.3199 seconds
|
| 185 |
+
[04-20 02:50:05] [DenoisingStage] finished in 2.5623 seconds
|
| 186 |
+
[04-20 02:50:05] [DecodingStage] started...
|
| 187 |
+
[04-20 02:50:10] [DecodingStage] finished in 5.0055 seconds
|
| 188 |
+
[04-20 02:50:11] Output saved to [1;36moutputs/Random_prompt_1_for_benchmarking_diffusion_models_20260420-025002_8d91326e.mp4[0;0m
|
| 189 |
+
[04-20 02:50:11] Pixel data generated successfully in [92m8.18[0;0m seconds
|
| 190 |
+
[04-20 02:50:11] Completed batch processing. Generated 1 outputs in [92m8.18[0;0m seconds
|
| 191 |
+
[04-20 02:50:11] Adjusting number of frames from 17 to 17 based on model
|
| 192 |
+
[04-20 02:50:11] Processing prompt 1/1: <redacted, len=49>
|
| 193 |
+
[04-20 02:50:11] Sampling params:
|
| 194 |
+
width: 960
|
| 195 |
+
height: 544
|
| 196 |
+
num_frames: 17
|
| 197 |
+
fps: 24
|
| 198 |
+
prompt: <redacted, len=49>
|
| 199 |
+
neg_prompt: <redacted, len=392>
|
| 200 |
+
seed: 0
|
| 201 |
+
infer_steps: 8
|
| 202 |
+
num_outputs_per_prompt: 1
|
| 203 |
+
guidance_scale: 1.0
|
| 204 |
+
embedded_guidance_scale: 6
|
| 205 |
+
n_tokens: None
|
| 206 |
+
flow_shift: 7
|
| 207 |
+
image_path: None
|
| 208 |
+
save_output: True
|
| 209 |
+
output_file_path: outputs/Random_prompt_2_for_benchmarking_diffusion_models_20260420-025011_9a3de0d5.mp4
|
| 210 |
+
|
| 211 |
+
[04-20 02:50:11] Running pipeline stages: ['InputValidationStage', 'prompt_encoding_stage_primary', 'TimestepPreparationStage', 'LatentPreparationStage', 'DenoisingStage', 'DecodingStage']
|
| 212 |
+
[04-20 02:50:11] [InputValidationStage] started...
|
| 213 |
+
[04-20 02:50:11] [InputValidationStage] finished in 0.0001 seconds
|
| 214 |
+
[04-20 02:50:11] [TextEncodingStage] started...
|
| 215 |
+
[04-20 02:50:11] [TextEncodingStage] finished in 0.3426 seconds
|
| 216 |
+
[04-20 02:50:11] [TimestepPreparationStage] started...
|
| 217 |
+
[04-20 02:50:11] [TimestepPreparationStage] finished in 0.0004 seconds
|
| 218 |
+
[04-20 02:50:11] [LatentPreparationStage] started...
|
| 219 |
+
[04-20 02:50:11] [LatentPreparationStage] finished in 0.0001 seconds
|
| 220 |
+
[04-20 02:50:11] [DenoisingStage] started...
|
| 221 |
+
|
| 222 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 223 |
25%|██▌ | 2/8 [00:00<00:01, 4.59it/s]
|
| 224 |
38%|███▊ | 3/8 [00:00<00:01, 3.66it/s]
|
| 225 |
50%|█████ | 4/8 [00:01<00:01, 3.27it/s]
|
| 226 |
62%|██████▎ | 5/8 [00:01<00:00, 3.11it/s]
|
| 227 |
75%|███████▌ | 6/8 [00:01<00:00, 3.02it/s]
|
| 228 |
88%|████████▊ | 7/8 [00:02<00:00, 2.94it/s]
|
| 229 |
+
[04-20 02:50:13] [DenoisingStage] average time per step: 0.3202 seconds
|
| 230 |
+
[04-20 02:50:13] [DenoisingStage] finished in 2.5644 seconds
|
| 231 |
+
[04-20 02:50:13] [DecodingStage] started...
|
| 232 |
+
[04-20 02:50:19] [DecodingStage] finished in 5.0319 seconds
|
| 233 |
+
[04-20 02:50:19] Output saved to [1;36moutputs/Random_prompt_2_for_benchmarking_diffusion_models_20260420-025011_9a3de0d5.mp4[0;0m
|
| 234 |
+
[04-20 02:50:19] Pixel data generated successfully in [92m8.18[0;0m seconds
|
| 235 |
+
[04-20 02:50:19] Completed batch processing. Generated 1 outputs in [92m8.18[0;0m seconds
|
| 236 |
+
|
| 237 |
+
==================================== Offline Throughput Benchmark Result =====================================
|
| 238 |
+
Model: /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773
|
| 239 |
+
Dataset: random
|
| 240 |
+
Resolution: 960x544x17
|
| 241 |
+
Num Inference Steps: 8
|
| 242 |
+
---------------------------------------------------------------------------
|
| 243 |
+
Total Requests: 3
|
| 244 |
+
Successful Requests: 3
|
| 245 |
+
Failed Requests: 0
|
| 246 |
+
Total Duration (seconds): 24.90
|
| 247 |
+
---------------------------------------------------------------------------
|
| 248 |
+
Frames Generated: 3
|
| 249 |
+
Megapixels Generated: 26.63
|
| 250 |
+
---------------------------------------------------------------------------
|
| 251 |
+
Frame Throughput (frames/sec): 0.12
|
| 252 |
+
MP Throughput (MP/sec): 1.07
|
| 253 |
+
Requests Per Second: 0.12
|
| 254 |
+
Latency Per Request (sec): 8.30
|
| 255 |
+
Peak Memory (MB): 0.00
|
| 256 |
+
==============================================================================================================
|
| 257 |
+
[04-20 02:50:19] Results saved to /data/bbuf/hunyuanvideo_fp8_20260420/benchmark/fp8_offline_throughput_transformer_path.jsonl
|
| 258 |
+
[93m[04-20 02:50:19] Generator was garbage collected without being shut down. Attempting to shut down the local server and client.[0;0m
|
| 259 |
+
[04-20 02:50:26] Worker 0: Shutdown complete.
|
validation/h100_20260420/logs/convert_sglang_fp8.log
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
No platform detected. Using base SRTPlatform with defaults.
|
| 2 |
+
{
|
| 3 |
+
"added_scale_tensors": 880,
|
| 4 |
+
"bf16_fallback_weights": 112,
|
| 5 |
+
"output_shards": 3,
|
| 6 |
+
"preserved_ignored_weights": 160,
|
| 7 |
+
"quantized_weights": 440
|
| 8 |
+
}
|
validation/h100_20260420/logs/modelopt_quantize_fp8.log
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
validation/h100_20260420/logs/profile_bf16.log
ADDED
|
@@ -0,0 +1,90 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 1 |
12%|█▎ | 1/8 [00:01<00:13, 1.86s/it]
|
| 2 |
25%|██▌ | 2/8 [00:01<00:05, 1.20it/s]
|
| 3 |
38%|███▊ | 3/8 [00:02<00:03, 1.55it/s]
|
| 4 |
50%|█████ | 4/8 [00:02<00:02, 1.80it/s]
|
| 5 |
62%|██████▎ | 5/8 [00:03<00:01, 1.98it/s]
|
| 6 |
75%|███████▌ | 6/8 [00:03<00:00, 2.08it/s]
|
| 7 |
88%|████████▊ | 7/8 [00:04<00:00, 2.19it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
No platform detected. Using base SRTPlatform with defaults.
|
| 2 |
+
[04-20 02:36:50] server_args: {"model_path": "/root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773", "model_id": null, "backend": "sglang", "attention_backend": null, "attention_backend_config": {}, "cache_dit_config": null, "nccl_port": null, "trust_remote_code": false, "revision": null, "num_gpus": 1, "tp_size": 1, "sp_degree": 1, "ulysses_degree": 1, "ring_degree": 1, "dp_size": 1, "dp_degree": 1, "enable_cfg_parallel": false, "hsdp_replicate_dim": 1, "hsdp_shard_dim": 1, "dist_timeout": 3600, "pipeline_class_name": null, "lora_path": null, "lora_nickname": "default", "lora_scale": 1.0, "lora_weight_name": null, "component_paths": {}, "transformer_weights_path": null, "lora_target_modules": null, "dit_cpu_offload": false, "dit_layerwise_offload": false, "dit_offload_prefetch_size": 0.0, "text_encoder_cpu_offload": true, "image_encoder_cpu_offload": true, "vae_cpu_offload": true, "use_fsdp_inference": false, "pin_cpu_memory": true, "ltx2_two_stage_device_mode": null, "comfyui_mode": false, "enable_torch_compile": false, "warmup": false, "warmup_resolutions": null, "warmup_steps": 1, "disable_autocast": false, "master_port": 30005, "host": "127.0.0.1", "port": 30000, "webui": false, "webui_port": 12312, "scheduler_port": 5637, "strict_ports": false, "output_path": "outputs/", "input_save_path": "inputs/uploads", "prompt_file_path": null, "model_paths": {}, "model_loaded": {"transformer": true, "vae": true, "video_vae": true, "audio_vae": true, "video_dit": true, "audio_dit": true, "dual_tower_bridge": true}, "boundary_ratio": null, "base_gpu_id": 0, "disagg_role": "monolithic", "disagg_timeout": 600, "disagg_dispatch_policy": "round_robin", "disagg_mode": false, "disagg_server_addr": null, "encoder_urls": null, "denoiser_urls": null, "decoder_urls": null, "encoder_tp": null, "denoiser_tp": null, "denoiser_sp": null, "denoiser_ulysses": null, "denoiser_ring": null, "decoder_tp": null, "disagg_transfer_pool_size": 268435456, "disagg_p2p_hostname": "127.0.0.1", "disagg_ib_device": null, "pool_work_endpoint": null, "pool_result_endpoint": null, "log_level": "info", "uvicorn_access_log_exclude_prefixes": []}
|
| 3 |
+
[04-20 02:36:50] Local mode: True
|
| 4 |
+
[04-20 02:36:50] Starting server...
|
| 5 |
+
No platform detected. Using base SRTPlatform with defaults.
|
| 6 |
+
[04-20 02:36:57] Scheduler bind at endpoint: tcp://127.0.0.1:5637
|
| 7 |
+
[04-20 02:36:57] Initializing distributed environment with world_size=1, device=cuda:0, timeout=3600
|
| 8 |
+
[04-20 02:36:57] Setting distributed timeout to 3600 seconds
|
| 9 |
+
[04-20 02:36:58] No pipeline_class_name specified, using model_index.json
|
| 10 |
+
[04-20 02:36:58] Diffusers version: 0.32.0.dev0
|
| 11 |
+
[04-20 02:36:58] Using pipeline from model_index.json: HunyuanVideoPipeline
|
| 12 |
+
[04-20 02:36:58] Loading pipeline modules...
|
| 13 |
+
[04-20 02:36:58] Model already exists locally and is complete
|
| 14 |
+
[04-20 02:36:58] Model path: /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773
|
| 15 |
+
[04-20 02:36:58] Diffusers version: 0.32.0.dev0
|
| 16 |
+
[04-20 02:36:58] Loading pipeline modules from config: {'_class_name': 'HunyuanVideoPipeline', '_diffusers_version': '0.32.0.dev0', 'scheduler': ['diffusers', 'FlowMatchEulerDiscreteScheduler'], 'text_encoder': ['transformers', 'LlamaModel'], 'text_encoder_2': ['transformers', 'CLIPTextModel'], 'tokenizer': ['transformers', 'LlamaTokenizerFast'], 'tokenizer_2': ['transformers', 'CLIPTokenizer'], 'transformer': ['diffusers', 'HunyuanVideoTransformer3DModel'], 'vae': ['diffusers', 'AutoencoderKLHunyuanVideo']}
|
| 17 |
+
[04-20 02:36:58] Loading required components: ['text_encoder', 'text_encoder_2', 'tokenizer', 'tokenizer_2', 'vae', 'transformer', 'scheduler']
|
| 18 |
+
|
| 19 |
+
[04-20 02:36:59] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
|
| 20 |
+
[04-20 02:37:01] [RunAI Streamer] Overall time to stream 14.0 GiB of all files to cpu: 2.24s, 6.3 GiB/s
|
| 21 |
+
[04-20 02:37:19] Loaded text_encoder: FSDPLlamaModel (sgl-diffusion version). model size: 13.98 GB, consumed GPU mem: 0.81 GB, avail GPU mem: 73.04 GB
|
| 22 |
+
|
| 23 |
+
[04-20 02:37:20] Using Torch SDPA backend
|
| 24 |
+
[04-20 02:37:20] [RunAI Streamer] Overall time to stream 234.7 MiB of all files to cpu: 0.03s, 8.2 GiB/s
|
| 25 |
+
[04-20 02:37:20] Loaded text_encoder_2: FSDPCLIPTextModel (sgl-diffusion version). model size: 0.23 GB, consumed GPU mem: 0.76 GB, avail GPU mem: 72.27 GB
|
| 26 |
+
|
| 27 |
+
[04-20 02:37:21] Loaded tokenizer: LlamaTokenizer (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
|
| 28 |
+
|
| 29 |
+
[04-20 02:37:21] Loaded tokenizer_2: CLIPTokenizer (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
|
| 30 |
+
[04-20 02:37:21] Loading vae from /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773/vae. avail mem: 72.27 GB
|
| 31 |
+
[93m[04-20 02:37:21] VAE unexpected keys: ['encoder.conv_in.conv.bias', 'encoder.conv_in.conv.weight', 'encoder.conv_norm_out.bias', 'encoder.conv_norm_out.weight', 'encoder.conv_out.conv.bias', 'encoder.conv_out.conv.weight', 'encoder.down_blocks.0.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.0.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.0.resnets.0.conv1.conv.bias', 'encoder.down_blocks.0.resnets.0.conv1.conv.weight', 'encoder.down_blocks.0.resnets.0.conv2.conv.bias', 'encoder.down_blocks.0.resnets.0.conv2.conv.weight', 'encoder.down_blocks.0.resnets.0.norm1.bias', 'encoder.down_blocks.0.resnets.0.norm1.weight', 'encoder.down_blocks.0.resnets.0.norm2.bias', 'encoder.down_blocks.0.resnets.0.norm2.weight', 'encoder.down_blocks.0.resnets.1.conv1.conv.bias', 'encoder.down_blocks.0.resnets.1.conv1.conv.weight', 'encoder.down_blocks.0.resnets.1.conv2.conv.bias', 'encoder.down_blocks.0.resnets.1.conv2.conv.weight', 'encoder.down_blocks.0.resnets.1.norm1.bias', 'encoder.down_blocks.0.resnets.1.norm1.weight', 'encoder.down_blocks.0.resnets.1.norm2.bias', 'encoder.down_blocks.0.resnets.1.norm2.weight', 'encoder.down_blocks.1.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.1.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.1.resnets.0.conv1.conv.bias', 'encoder.down_blocks.1.resnets.0.conv1.conv.weight', 'encoder.down_blocks.1.resnets.0.conv2.conv.bias', 'encoder.down_blocks.1.resnets.0.conv2.conv.weight', 'encoder.down_blocks.1.resnets.0.conv_shortcut.conv.bias', 'encoder.down_blocks.1.resnets.0.conv_shortcut.conv.weight', 'encoder.down_blocks.1.resnets.0.norm1.bias', 'encoder.down_blocks.1.resnets.0.norm1.weight', 'encoder.down_blocks.1.resnets.0.norm2.bias', 'encoder.down_blocks.1.resnets.0.norm2.weight', 'encoder.down_blocks.1.resnets.1.conv1.conv.bias', 'encoder.down_blocks.1.resnets.1.conv1.conv.weight', 'encoder.down_blocks.1.resnets.1.conv2.conv.bias', 'encoder.down_blocks.1.resnets.1.conv2.conv.weight', 'encoder.down_blocks.1.resnets.1.norm1.bias', 'encoder.down_blocks.1.resnets.1.norm1.weight', 'encoder.down_blocks.1.resnets.1.norm2.bias', 'encoder.down_blocks.1.resnets.1.norm2.weight', 'encoder.down_blocks.2.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.2.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.2.resnets.0.conv1.conv.bias', 'encoder.down_blocks.2.resnets.0.conv1.conv.weight', 'encoder.down_blocks.2.resnets.0.conv2.conv.bias', 'encoder.down_blocks.2.resnets.0.conv2.conv.weight', 'encoder.down_blocks.2.resnets.0.conv_shortcut.conv.bias', 'encoder.down_blocks.2.resnets.0.conv_shortcut.conv.weight', 'encoder.down_blocks.2.resnets.0.norm1.bias', 'encoder.down_blocks.2.resnets.0.norm1.weight', 'encoder.down_blocks.2.resnets.0.norm2.bias', 'encoder.down_blocks.2.resnets.0.norm2.weight', 'encoder.down_blocks.2.resnets.1.conv1.conv.bias', 'encoder.down_blocks.2.resnets.1.conv1.conv.weight', 'encoder.down_blocks.2.resnets.1.conv2.conv.bias', 'encoder.down_blocks.2.resnets.1.conv2.conv.weight', 'encoder.down_blocks.2.resnets.1.norm1.bias', 'encoder.down_blocks.2.resnets.1.norm1.weight', 'encoder.down_blocks.2.resnets.1.norm2.bias', 'encoder.down_blocks.2.resnets.1.norm2.weight', 'encoder.down_blocks.3.resnets.0.conv1.conv.bias', 'encoder.down_blocks.3.resnets.0.conv1.conv.weight', 'encoder.down_blocks.3.resnets.0.conv2.conv.bias', 'encoder.down_blocks.3.resnets.0.conv2.conv.weight', 'encoder.down_blocks.3.resnets.0.norm1.bias', 'encoder.down_blocks.3.resnets.0.norm1.weight', 'encoder.down_blocks.3.resnets.0.norm2.bias', 'encoder.down_blocks.3.resnets.0.norm2.weight', 'encoder.down_blocks.3.resnets.1.conv1.conv.bias', 'encoder.down_blocks.3.resnets.1.conv1.conv.weight', 'encoder.down_blocks.3.resnets.1.conv2.conv.bias', 'encoder.down_blocks.3.resnets.1.conv2.conv.weight', 'encoder.down_blocks.3.resnets.1.norm1.bias', 'encoder.down_blocks.3.resnets.1.norm1.weight', 'encoder.down_blocks.3.resnets.1.norm2.bias', 'encoder.down_blocks.3.resnets.1.norm2.weight', 'encoder.mid_block.attentions.0.group_norm.bias', 'encoder.mid_block.attentions.0.group_norm.weight', 'encoder.mid_block.attentions.0.to_k.bias', 'encoder.mid_block.attentions.0.to_k.weight', 'encoder.mid_block.attentions.0.to_out.0.bias', 'encoder.mid_block.attentions.0.to_out.0.weight', 'encoder.mid_block.attentions.0.to_q.bias', 'encoder.mid_block.attentions.0.to_q.weight', 'encoder.mid_block.attentions.0.to_v.bias', 'encoder.mid_block.attentions.0.to_v.weight', 'encoder.mid_block.resnets.0.conv1.conv.bias', 'encoder.mid_block.resnets.0.conv1.conv.weight', 'encoder.mid_block.resnets.0.conv2.conv.bias', 'encoder.mid_block.resnets.0.conv2.conv.weight', 'encoder.mid_block.resnets.0.norm1.bias', 'encoder.mid_block.resnets.0.norm1.weight', 'encoder.mid_block.resnets.0.norm2.bias', 'encoder.mid_block.resnets.0.norm2.weight', 'encoder.mid_block.resnets.1.conv1.conv.bias', 'encoder.mid_block.resnets.1.conv1.conv.weight', 'encoder.mid_block.resnets.1.conv2.conv.bias', 'encoder.mid_block.resnets.1.conv2.conv.weight', 'encoder.mid_block.resnets.1.norm1.bias', 'encoder.mid_block.resnets.1.norm1.weight', 'encoder.mid_block.resnets.1.norm2.bias', 'encoder.mid_block.resnets.1.norm2.weight', 'quant_conv.bias', 'quant_conv.weight'][0;0m
|
| 32 |
+
[04-20 02:37:21] Loaded vae: AutoencoderKLHunyuanVideo (sgl-diffusion version). model size: 0.27 GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
|
| 33 |
+
[04-20 02:37:21] Loading transformer from /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773/transformer. avail mem: 72.27 GB
|
| 34 |
+
[04-20 02:37:21] Loading HunyuanVideoTransformer3DModel from 6 safetensors file(s) , param_dtype: torch.bfloat16
|
| 35 |
+
[04-20 02:37:21] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
|
| 36 |
+
[04-20 02:37:21] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
|
| 37 |
+
[04-20 02:37:23] [RunAI Streamer] Overall time to stream 23.9 GiB of all files to cpu: 1.69s, 14.1 GiB/s
|
| 38 |
+
[04-20 02:37:28] Loaded model with 12.82B parameters
|
| 39 |
+
[04-20 02:37:28] Loaded transformer: HunyuanVideoTransformer3DModel (sgl-diffusion version). model size: 23.88 GB, consumed GPU mem: 24.19 GB, avail GPU mem: 48.08 GB
|
| 40 |
+
|
| 41 |
+
[04-20 02:37:28] Loaded scheduler: FlowMatchEulerDiscreteScheduler (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 48.08 GB
|
| 42 |
+
|
| 43 |
+
[04-20 02:37:28] Creating pipeline stages...
|
| 44 |
+
[04-20 02:37:28] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
|
| 45 |
+
[04-20 02:37:28] Pipeline instantiated
|
| 46 |
+
[04-20 02:37:28] Worker 0: Initialized device, model, and distributed environment.
|
| 47 |
+
[04-20 02:37:28] Worker 0: Scheduler loop started.
|
| 48 |
+
[04-20 02:37:28] Adjusting number of frames from 17 to 17 based on model
|
| 49 |
+
[04-20 02:37:28] Processing prompt 1/1: <redacted, len=119>
|
| 50 |
+
[04-20 02:37:28] Sampling params:
|
| 51 |
+
width: 960
|
| 52 |
+
height: 544
|
| 53 |
+
num_frames: 17
|
| 54 |
+
fps: 24
|
| 55 |
+
prompt: <redacted, len=119>
|
| 56 |
+
neg_prompt: <redacted, len=392>
|
| 57 |
+
seed: 0
|
| 58 |
+
infer_steps: 8
|
| 59 |
+
num_outputs_per_prompt: 1
|
| 60 |
+
guidance_scale: 1.0
|
| 61 |
+
embedded_guidance_scale: 6
|
| 62 |
+
n_tokens: None
|
| 63 |
+
flow_shift: 7
|
| 64 |
+
image_path: None
|
| 65 |
+
save_output: False
|
| 66 |
+
output_file_path: outputs/A_cinematic_shot_of_a_red_sports_car_driving_through_rain_at_night_reflections_on_wet_streets_smoo_20260420-023728_58260e89.mp4
|
| 67 |
+
|
| 68 |
+
[04-20 02:37:28] Running pipeline stages: ['InputValidationStage', 'prompt_encoding_stage_primary', 'TimestepPreparationStage', 'LatentPreparationStage', 'DenoisingStage', 'DecodingStage']
|
| 69 |
+
[04-20 02:37:28] Profiling request: 5b26da87-550f-4d35-8cb8-bf9aa9c535d0 for 5 steps...
|
| 70 |
+
[04-20 02:37:28] Starting Profiler...
|
| 71 |
+
[04-20 02:37:28] [InputValidationStage] started...
|
| 72 |
+
[04-20 02:37:28] [InputValidationStage] finished in 0.0004 seconds
|
| 73 |
+
[04-20 02:37:28] [TextEncodingStage] started...
|
| 74 |
+
[04-20 02:37:29] [TextEncodingStage] finished in 0.9494 seconds
|
| 75 |
+
[04-20 02:37:29] [TimestepPreparationStage] started...
|
| 76 |
+
[04-20 02:37:29] [TimestepPreparationStage] finished in 0.0003 seconds
|
| 77 |
+
[04-20 02:37:29] [LatentPreparationStage] started...
|
| 78 |
+
[04-20 02:37:29] [LatentPreparationStage] finished in 0.0008 seconds
|
| 79 |
+
[04-20 02:37:29] [DenoisingStage] started...
|
| 80 |
+
|
| 81 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 82 |
12%|█▎ | 1/8 [00:01<00:13, 1.86s/it]
|
| 83 |
25%|██▌ | 2/8 [00:01<00:05, 1.20it/s]
|
| 84 |
38%|███▊ | 3/8 [00:02<00:03, 1.55it/s]
|
| 85 |
50%|█████ | 4/8 [00:02<00:02, 1.80it/s]
|
| 86 |
62%|██████▎ | 5/8 [00:03<00:01, 1.98it/s]
|
| 87 |
75%|███████▌ | 6/8 [00:03<00:00, 2.08it/s]
|
| 88 |
88%|████████▊ | 7/8 [00:04<00:00, 2.19it/s]
|
| 89 |
+
[04-20 02:37:45] Saved profiler traces to: [1;36m/data/bbuf/hunyuanvideo_fp8_20260420/profiler/bf16/5b26da87-550f-4d35-8cb8-bf9aa9c535d0-5_steps-global-rank0.trace.json.gz[0;0m
|
| 90 |
+
|
| 91 |
+
[04-20 02:37:45] [DenoisingStage] average time per step: 1.9787 seconds
|
| 92 |
+
[04-20 02:37:45] [DenoisingStage] finished in 15.8313 seconds
|
| 93 |
+
[04-20 02:37:45] [DecodingStage] started...
|
| 94 |
+
[04-20 02:37:50] [DecodingStage] finished in 5.4015 seconds
|
| 95 |
+
[04-20 02:37:51] Pixel data generated successfully in [92m22.59[0;0m seconds
|
| 96 |
+
[04-20 02:37:51] Completed batch processing. Generated 1 outputs in [92m22.59[0;0m seconds
|
| 97 |
+
[93m[04-20 02:37:51] Generator was garbage collected without being shut down. Attempting to shut down the local server and client.[0;0m
|
| 98 |
+
[04-20 02:37:58] Worker 0: Shutdown complete.
|
validation/h100_20260420/logs/profile_fp8.log
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 0 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 1 |
12%|█▎ | 1/8 [00:01<00:12, 1.80s/it]
|
| 2 |
25%|██▌ | 2/8 [00:01<00:04, 1.22it/s]
|
| 3 |
38%|███▊ | 3/8 [00:02<00:03, 1.66it/s]
|
| 4 |
50%|█████ | 4/8 [00:02<00:02, 1.98it/s]
|
| 5 |
62%|██████▎ | 5/8 [00:02<00:01, 2.23it/s]
|
| 6 |
75%|███████▌ | 6/8 [00:03<00:00, 2.40it/s]
|
| 7 |
88%|████████▊ | 7/8 [00:03<00:00, 2.52it/s]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
No platform detected. Using base SRTPlatform with defaults.
|
| 2 |
+
[04-20 02:38:13] server_args: {"model_path": "/data/bbuf/hunyuanvideo_fp8_20260420/sglang_fp8_pipeline", "model_id": null, "backend": "sglang", "attention_backend": null, "attention_backend_config": {}, "cache_dit_config": null, "nccl_port": null, "trust_remote_code": false, "revision": null, "num_gpus": 1, "tp_size": 1, "sp_degree": 1, "ulysses_degree": 1, "ring_degree": 1, "dp_size": 1, "dp_degree": 1, "enable_cfg_parallel": false, "hsdp_replicate_dim": 1, "hsdp_shard_dim": 1, "dist_timeout": 3600, "pipeline_class_name": null, "lora_path": null, "lora_nickname": "default", "lora_scale": 1.0, "lora_weight_name": null, "component_paths": {}, "transformer_weights_path": null, "lora_target_modules": null, "dit_cpu_offload": false, "dit_layerwise_offload": false, "dit_offload_prefetch_size": 0.0, "text_encoder_cpu_offload": true, "image_encoder_cpu_offload": true, "vae_cpu_offload": true, "use_fsdp_inference": false, "pin_cpu_memory": true, "ltx2_two_stage_device_mode": null, "comfyui_mode": false, "enable_torch_compile": false, "warmup": false, "warmup_resolutions": null, "warmup_steps": 1, "disable_autocast": false, "master_port": 30005, "host": "127.0.0.1", "port": 30000, "webui": false, "webui_port": 12312, "scheduler_port": 5618, "strict_ports": false, "output_path": "outputs/", "input_save_path": "inputs/uploads", "prompt_file_path": null, "model_paths": {}, "model_loaded": {"transformer": true, "vae": true, "video_vae": true, "audio_vae": true, "video_dit": true, "audio_dit": true, "dual_tower_bridge": true}, "boundary_ratio": null, "base_gpu_id": 0, "disagg_role": "monolithic", "disagg_timeout": 600, "disagg_dispatch_policy": "round_robin", "disagg_mode": false, "disagg_server_addr": null, "encoder_urls": null, "denoiser_urls": null, "decoder_urls": null, "encoder_tp": null, "denoiser_tp": null, "denoiser_sp": null, "denoiser_ulysses": null, "denoiser_ring": null, "decoder_tp": null, "disagg_transfer_pool_size": 268435456, "disagg_p2p_hostname": "127.0.0.1", "disagg_ib_device": null, "pool_work_endpoint": null, "pool_result_endpoint": null, "log_level": "info", "uvicorn_access_log_exclude_prefixes": []}
|
| 3 |
+
[04-20 02:38:13] Local mode: True
|
| 4 |
+
[04-20 02:38:13] Starting server...
|
| 5 |
+
No platform detected. Using base SRTPlatform with defaults.
|
| 6 |
+
[04-20 02:38:20] Scheduler bind at endpoint: tcp://127.0.0.1:5618
|
| 7 |
+
[04-20 02:38:20] Initializing distributed environment with world_size=1, device=cuda:0, timeout=3600
|
| 8 |
+
[04-20 02:38:20] Setting distributed timeout to 3600 seconds
|
| 9 |
+
[04-20 02:38:21] No pipeline_class_name specified, using model_index.json
|
| 10 |
+
[04-20 02:38:21] Diffusers version: 0.32.0.dev0
|
| 11 |
+
[04-20 02:38:21] Diffusers version: 0.32.0.dev0
|
| 12 |
+
[04-20 02:38:21] Using pipeline from model_index.json: HunyuanVideoPipeline
|
| 13 |
+
[04-20 02:38:21] Loading pipeline modules...
|
| 14 |
+
[04-20 02:38:21] Model already exists locally and is complete
|
| 15 |
+
[04-20 02:38:21] Model path: /data/bbuf/hunyuanvideo_fp8_20260420/sglang_fp8_pipeline
|
| 16 |
+
[04-20 02:38:21] Diffusers version: 0.32.0.dev0
|
| 17 |
+
[04-20 02:38:21] Loading pipeline modules from config: {'_class_name': 'HunyuanVideoPipeline', '_diffusers_version': '0.32.0.dev0', 'scheduler': ['diffusers', 'FlowMatchEulerDiscreteScheduler'], 'text_encoder': ['transformers', 'LlamaModel'], 'text_encoder_2': ['transformers', 'CLIPTextModel'], 'tokenizer': ['transformers', 'LlamaTokenizerFast'], 'tokenizer_2': ['transformers', 'CLIPTokenizer'], 'transformer': ['diffusers', 'HunyuanVideoTransformer3DModel'], 'vae': ['diffusers', 'AutoencoderKLHunyuanVideo']}
|
| 18 |
+
[04-20 02:38:21] Loading required components: ['text_encoder', 'text_encoder_2', 'tokenizer', 'tokenizer_2', 'vae', 'transformer', 'scheduler']
|
| 19 |
+
|
| 20 |
+
[04-20 02:38:21] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
|
| 21 |
+
[04-20 02:38:23] [RunAI Streamer] Overall time to stream 14.0 GiB of all files to cpu: 2.05s, 6.8 GiB/s
|
| 22 |
+
[04-20 02:38:38] Loaded text_encoder: FSDPLlamaModel (sgl-diffusion version). model size: 13.98 GB, consumed GPU mem: 0.81 GB, avail GPU mem: 73.04 GB
|
| 23 |
+
|
| 24 |
+
[04-20 02:38:39] Using Torch SDPA backend
|
| 25 |
+
[04-20 02:38:39] [RunAI Streamer] Overall time to stream 234.7 MiB of all files to cpu: 0.03s, 7.7 GiB/s
|
| 26 |
+
[04-20 02:38:39] Loaded text_encoder_2: FSDPCLIPTextModel (sgl-diffusion version). model size: 0.23 GB, consumed GPU mem: 0.76 GB, avail GPU mem: 72.27 GB
|
| 27 |
+
|
| 28 |
+
[04-20 02:38:40] Loaded tokenizer: LlamaTokenizer (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
|
| 29 |
+
|
| 30 |
+
[04-20 02:38:40] Loaded tokenizer_2: CLIPTokenizer (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
|
| 31 |
+
[04-20 02:38:40] Loading vae from /data/bbuf/hunyuanvideo_fp8_20260420/sglang_fp8_pipeline/vae. avail mem: 72.27 GB
|
| 32 |
+
[93m[04-20 02:38:40] VAE unexpected keys: ['encoder.conv_in.conv.bias', 'encoder.conv_in.conv.weight', 'encoder.conv_norm_out.bias', 'encoder.conv_norm_out.weight', 'encoder.conv_out.conv.bias', 'encoder.conv_out.conv.weight', 'encoder.down_blocks.0.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.0.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.0.resnets.0.conv1.conv.bias', 'encoder.down_blocks.0.resnets.0.conv1.conv.weight', 'encoder.down_blocks.0.resnets.0.conv2.conv.bias', 'encoder.down_blocks.0.resnets.0.conv2.conv.weight', 'encoder.down_blocks.0.resnets.0.norm1.bias', 'encoder.down_blocks.0.resnets.0.norm1.weight', 'encoder.down_blocks.0.resnets.0.norm2.bias', 'encoder.down_blocks.0.resnets.0.norm2.weight', 'encoder.down_blocks.0.resnets.1.conv1.conv.bias', 'encoder.down_blocks.0.resnets.1.conv1.conv.weight', 'encoder.down_blocks.0.resnets.1.conv2.conv.bias', 'encoder.down_blocks.0.resnets.1.conv2.conv.weight', 'encoder.down_blocks.0.resnets.1.norm1.bias', 'encoder.down_blocks.0.resnets.1.norm1.weight', 'encoder.down_blocks.0.resnets.1.norm2.bias', 'encoder.down_blocks.0.resnets.1.norm2.weight', 'encoder.down_blocks.1.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.1.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.1.resnets.0.conv1.conv.bias', 'encoder.down_blocks.1.resnets.0.conv1.conv.weight', 'encoder.down_blocks.1.resnets.0.conv2.conv.bias', 'encoder.down_blocks.1.resnets.0.conv2.conv.weight', 'encoder.down_blocks.1.resnets.0.conv_shortcut.conv.bias', 'encoder.down_blocks.1.resnets.0.conv_shortcut.conv.weight', 'encoder.down_blocks.1.resnets.0.norm1.bias', 'encoder.down_blocks.1.resnets.0.norm1.weight', 'encoder.down_blocks.1.resnets.0.norm2.bias', 'encoder.down_blocks.1.resnets.0.norm2.weight', 'encoder.down_blocks.1.resnets.1.conv1.conv.bias', 'encoder.down_blocks.1.resnets.1.conv1.conv.weight', 'encoder.down_blocks.1.resnets.1.conv2.conv.bias', 'encoder.down_blocks.1.resnets.1.conv2.conv.weight', 'encoder.down_blocks.1.resnets.1.norm1.bias', 'encoder.down_blocks.1.resnets.1.norm1.weight', 'encoder.down_blocks.1.resnets.1.norm2.bias', 'encoder.down_blocks.1.resnets.1.norm2.weight', 'encoder.down_blocks.2.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.2.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.2.resnets.0.conv1.conv.bias', 'encoder.down_blocks.2.resnets.0.conv1.conv.weight', 'encoder.down_blocks.2.resnets.0.conv2.conv.bias', 'encoder.down_blocks.2.resnets.0.conv2.conv.weight', 'encoder.down_blocks.2.resnets.0.conv_shortcut.conv.bias', 'encoder.down_blocks.2.resnets.0.conv_shortcut.conv.weight', 'encoder.down_blocks.2.resnets.0.norm1.bias', 'encoder.down_blocks.2.resnets.0.norm1.weight', 'encoder.down_blocks.2.resnets.0.norm2.bias', 'encoder.down_blocks.2.resnets.0.norm2.weight', 'encoder.down_blocks.2.resnets.1.conv1.conv.bias', 'encoder.down_blocks.2.resnets.1.conv1.conv.weight', 'encoder.down_blocks.2.resnets.1.conv2.conv.bias', 'encoder.down_blocks.2.resnets.1.conv2.conv.weight', 'encoder.down_blocks.2.resnets.1.norm1.bias', 'encoder.down_blocks.2.resnets.1.norm1.weight', 'encoder.down_blocks.2.resnets.1.norm2.bias', 'encoder.down_blocks.2.resnets.1.norm2.weight', 'encoder.down_blocks.3.resnets.0.conv1.conv.bias', 'encoder.down_blocks.3.resnets.0.conv1.conv.weight', 'encoder.down_blocks.3.resnets.0.conv2.conv.bias', 'encoder.down_blocks.3.resnets.0.conv2.conv.weight', 'encoder.down_blocks.3.resnets.0.norm1.bias', 'encoder.down_blocks.3.resnets.0.norm1.weight', 'encoder.down_blocks.3.resnets.0.norm2.bias', 'encoder.down_blocks.3.resnets.0.norm2.weight', 'encoder.down_blocks.3.resnets.1.conv1.conv.bias', 'encoder.down_blocks.3.resnets.1.conv1.conv.weight', 'encoder.down_blocks.3.resnets.1.conv2.conv.bias', 'encoder.down_blocks.3.resnets.1.conv2.conv.weight', 'encoder.down_blocks.3.resnets.1.norm1.bias', 'encoder.down_blocks.3.resnets.1.norm1.weight', 'encoder.down_blocks.3.resnets.1.norm2.bias', 'encoder.down_blocks.3.resnets.1.norm2.weight', 'encoder.mid_block.attentions.0.group_norm.bias', 'encoder.mid_block.attentions.0.group_norm.weight', 'encoder.mid_block.attentions.0.to_k.bias', 'encoder.mid_block.attentions.0.to_k.weight', 'encoder.mid_block.attentions.0.to_out.0.bias', 'encoder.mid_block.attentions.0.to_out.0.weight', 'encoder.mid_block.attentions.0.to_q.bias', 'encoder.mid_block.attentions.0.to_q.weight', 'encoder.mid_block.attentions.0.to_v.bias', 'encoder.mid_block.attentions.0.to_v.weight', 'encoder.mid_block.resnets.0.conv1.conv.bias', 'encoder.mid_block.resnets.0.conv1.conv.weight', 'encoder.mid_block.resnets.0.conv2.conv.bias', 'encoder.mid_block.resnets.0.conv2.conv.weight', 'encoder.mid_block.resnets.0.norm1.bias', 'encoder.mid_block.resnets.0.norm1.weight', 'encoder.mid_block.resnets.0.norm2.bias', 'encoder.mid_block.resnets.0.norm2.weight', 'encoder.mid_block.resnets.1.conv1.conv.bias', 'encoder.mid_block.resnets.1.conv1.conv.weight', 'encoder.mid_block.resnets.1.conv2.conv.bias', 'encoder.mid_block.resnets.1.conv2.conv.weight', 'encoder.mid_block.resnets.1.norm1.bias', 'encoder.mid_block.resnets.1.norm1.weight', 'encoder.mid_block.resnets.1.norm2.bias', 'encoder.mid_block.resnets.1.norm2.weight', 'quant_conv.bias', 'quant_conv.weight'][0;0m
|
| 33 |
+
[04-20 02:38:40] Loaded vae: AutoencoderKLHunyuanVideo (sgl-diffusion version). model size: 0.27 GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
|
| 34 |
+
[04-20 02:38:40] Loading transformer from /data/bbuf/hunyuanvideo_fp8_20260420/sglang_fp8_pipeline/transformer. avail mem: 72.27 GB
|
| 35 |
+
[93m[04-20 02:38:40] Detected ModelOpt FP8 checkpoint. The format is experimental and subject to change.[0;0m
|
| 36 |
+
[04-20 02:38:40] Loading HunyuanVideoTransformer3DModel from 3 safetensors file(s) , param_dtype: None
|
| 37 |
+
[04-20 02:38:40] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
|
| 38 |
+
[04-20 02:38:40] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
|
| 39 |
+
[04-20 02:38:42] [RunAI Streamer] Overall time to stream 15.4 GiB of all files to cpu: 1.24s, 12.4 GiB/s
|
| 40 |
+
[04-20 02:38:45] Loaded model with 12.82B parameters
|
| 41 |
+
[04-20 02:38:45] Loaded transformer: HunyuanVideoTransformer3DModel (sgl-diffusion version). model size: 15.45 GB, consumed GPU mem: 15.57 GB, avail GPU mem: 56.71 GB
|
| 42 |
+
|
| 43 |
+
[04-20 02:38:45] Loaded scheduler: FlowMatchEulerDiscreteScheduler (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 56.71 GB
|
| 44 |
+
|
| 45 |
+
[04-20 02:38:45] Creating pipeline stages...
|
| 46 |
+
[04-20 02:38:45] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
|
| 47 |
+
[04-20 02:38:45] Pipeline instantiated
|
| 48 |
+
[04-20 02:38:45] Worker 0: Initialized device, model, and distributed environment.
|
| 49 |
+
[04-20 02:38:45] Worker 0: Scheduler loop started.
|
| 50 |
+
[04-20 02:38:45] Adjusting number of frames from 17 to 17 based on model
|
| 51 |
+
[04-20 02:38:45] Processing prompt 1/1: <redacted, len=119>
|
| 52 |
+
[04-20 02:38:45] Sampling params:
|
| 53 |
+
width: 960
|
| 54 |
+
height: 544
|
| 55 |
+
num_frames: 17
|
| 56 |
+
fps: 24
|
| 57 |
+
prompt: <redacted, len=119>
|
| 58 |
+
neg_prompt: <redacted, len=392>
|
| 59 |
+
seed: 0
|
| 60 |
+
infer_steps: 8
|
| 61 |
+
num_outputs_per_prompt: 1
|
| 62 |
+
guidance_scale: 1.0
|
| 63 |
+
embedded_guidance_scale: 6
|
| 64 |
+
n_tokens: None
|
| 65 |
+
flow_shift: 7
|
| 66 |
+
image_path: None
|
| 67 |
+
save_output: False
|
| 68 |
+
output_file_path: outputs/A_cinematic_shot_of_a_red_sports_car_driving_through_rain_at_night_reflections_on_wet_streets_smoo_20260420-023845_50e60ea3.mp4
|
| 69 |
+
|
| 70 |
+
[04-20 02:38:45] Running pipeline stages: ['InputValidationStage', 'prompt_encoding_stage_primary', 'TimestepPreparationStage', 'LatentPreparationStage', 'DenoisingStage', 'DecodingStage']
|
| 71 |
+
[04-20 02:38:45] Profiling request: 63e4807c-f120-4fdc-b5b0-c1ab96522a7e for 5 steps...
|
| 72 |
+
[04-20 02:38:45] Starting Profiler...
|
| 73 |
+
[04-20 02:38:45] [InputValidationStage] started...
|
| 74 |
+
[04-20 02:38:45] [InputValidationStage] finished in 0.0004 seconds
|
| 75 |
+
[04-20 02:38:45] [TextEncodingStage] started...
|
| 76 |
+
[04-20 02:38:46] [TextEncodingStage] finished in 0.9378 seconds
|
| 77 |
+
[04-20 02:38:46] [TimestepPreparationStage] started...
|
| 78 |
+
[04-20 02:38:46] [TimestepPreparationStage] finished in 0.0003 seconds
|
| 79 |
+
[04-20 02:38:46] [LatentPreparationStage] started...
|
| 80 |
+
[04-20 02:38:46] [LatentPreparationStage] finished in 0.0008 seconds
|
| 81 |
+
[04-20 02:38:46] [DenoisingStage] started...
|
| 82 |
+
|
| 83 |
0%| | 0/8 [00:00<?, ?it/s]
|
| 84 |
12%|█▎ | 1/8 [00:01<00:12, 1.80s/it]
|
| 85 |
25%|██▌ | 2/8 [00:01<00:04, 1.22it/s]
|
| 86 |
38%|███▊ | 3/8 [00:02<00:03, 1.66it/s]
|
| 87 |
50%|█████ | 4/8 [00:02<00:02, 1.98it/s]
|
| 88 |
62%|██████▎ | 5/8 [00:02<00:01, 2.23it/s]
|
| 89 |
75%|███████▌ | 6/8 [00:03<00:00, 2.40it/s]
|
| 90 |
88%|████████▊ | 7/8 [00:03<00:00, 2.52it/s]
|
| 91 |
+
[04-20 02:39:04] Saved profiler traces to: [1;36m/data/bbuf/hunyuanvideo_fp8_20260420/profiler/fp8/63e4807c-f120-4fdc-b5b0-c1ab96522a7e-5_steps-global-rank0.trace.json.gz[0;0m
|
| 92 |
+
|
| 93 |
+
[04-20 02:39:04] [DenoisingStage] average time per step: 2.2673 seconds
|
| 94 |
+
[04-20 02:39:04] [DenoisingStage] finished in 18.1401 seconds
|
| 95 |
+
[04-20 02:39:04] [DecodingStage] started...
|
| 96 |
+
[04-20 02:39:09] [DecodingStage] finished in 5.3886 seconds
|
| 97 |
+
[04-20 02:39:10] Pixel data generated successfully in [92m24.93[0;0m seconds
|
| 98 |
+
[04-20 02:39:10] Completed batch processing. Generated 1 outputs in [92m24.93[0;0m seconds
|
| 99 |
+
[93m[04-20 02:39:10] Generator was garbage collected without being shut down. Attempting to shut down the local server and client.[0;0m
|
| 100 |
+
[04-20 02:39:17] Worker 0: Shutdown complete.
|
validation/h100_20260420/profiler/5b26da87-550f-4d35-8cb8-bf9aa9c535d0-5_steps-global-rank0.trace.json.gz
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a5e2d63975d0b13aa8233fcd7d2660ed2ec590ab66dbdee2ee37a75ae70d50ce
|
| 3 |
+
size 13790754
|
validation/h100_20260420/profiler/63e4807c-f120-4fdc-b5b0-c1ab96522a7e-5_steps-global-rank0.trace.json.gz
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c248e9cc2f382fc2f7a28d19939a5ebfb8b9fbb3a8353b46f50a6dd103526ea6
|
| 3 |
+
size 17625456
|
validation/h100_20260420/profiler/kernel_summary.md
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# BF16 trace
|
| 2 |
+
trace: /data/bbuf/hunyuanvideo_fp8_20260420/profiler/bf16/5b26da87-550f-4d35-8cb8-bf9aa9c535d0-5_steps-global-rank0.trace.json.gz
|
| 3 |
+
total kernel time: 2481.464 ms
|
| 4 |
+
|
| 5 |
+
| rank | kernel | count | time_ms | share |
|
| 6 |
+
|---:|---|---:|---:|---:|
|
| 7 |
+
| 1 | `void cutlass::device_kernel<flash::enable_sm90_or_later<flash::FlashAttnFwdSm90<flash::CollectiveMainloopFwdSm90<2, cute::tuple<cute::C<2>, ` | 360 | 691.122 | 27.85% |
|
| 8 |
+
| 2 | `nvjet_tst_192x208_64x4_2x1_v_bz_coopB_bias_TNT` | 240 | 437.843 | 17.64% |
|
| 9 |
+
| 3 | `nvjet_tst_256x136_64x4_1x2_h_bz_coopA_bias_TNT` | 240 | 299.272 | 12.06% |
|
| 10 |
+
| 4 | `nvjet_tst_128x232_64x4_2x1_v_bz_coopA_bias_TNT` | 246 | 152.702 | 6.15% |
|
| 11 |
+
| 5 | `void at::native::(anonymous namespace)::CatArrayBatchedCopy<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int, 4, 64, 64>(at::` | 720 | 151.870 | 6.12% |
|
| 12 |
+
| 6 | `nvjet_tst_256x144_64x4_2x1_v_bz_coopA_bias_TNT` | 120 | 126.300 | 5.09% |
|
| 13 |
+
| 7 | `void at::native::(anonymous namespace)::CatArrayBatchedCopy_alignedK_contig<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int,` | 600 | 111.988 | 4.51% |
|
| 14 |
+
| 8 | `nvjet_tst_192x192_64x4_2x1_v_bz_coopB_bias_TNN` | 120 | 97.632 | 3.93% |
|
| 15 |
+
| 9 | `void at::native::elementwise_kernel<128, 4, at::native::gpu_kernel_impl_nocast<at::native::GeluCUDAKernelImpl(at::TensorIteratorBase&, at::n` | 240 | 94.940 | 3.83% |
|
| 16 |
+
| 10 | `void at::native::(anonymous namespace)::CatArrayBatchedCopy_alignedK_contig<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int,` | 246 | 73.201 | 2.95% |
|
| 17 |
+
| 11 | `void at::native::elementwise_kernel<128, 4, at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&):` | 972 | 62.019 | 2.50% |
|
| 18 |
+
| 12 | `_rotary_embedding_kernel` | 720 | 42.580 | 1.72% |
|
| 19 |
+
| 13 | `_rms_norm_tiled_onepass` | 960 | 29.808 | 1.20% |
|
| 20 |
+
| 14 | `fuse_scale_shift_kernel_blc_opt` | 480 | 22.100 | 0.89% |
|
| 21 |
+
| 15 | `void at::native::vectorized_elementwise_kernel<8, at::native::GeluCUDAKernelImpl(at::TensorIteratorBase&, at::native::GeluType)::{lambda()#1` | 240 | 21.438 | 0.86% |
|
| 22 |
+
| 16 | `kernel_cutlass_kernel_sglangjit_kerneldiffusioncutedslscale_residual_norm_scale_shiftScaleResidualNormScaleShift_object_at__tensorptrbf16_gm` | 480 | 18.923 | 0.76% |
|
| 23 |
+
| 17 | `kernel_cutlass_kernel_sglangjit_kerneldiffusioncutedslscale_residual_norm_scale_shiftScaleResidualNormScaleShift_object_at__tensorptrbf16_gm` | 240 | 10.332 | 0.42% |
|
| 24 |
+
| 18 | `nvjet_tst_256x8_64x6_4x1_v_bz_bias_TNT` | 240 | 9.724 | 0.39% |
|
| 25 |
+
| 19 | `nvjet_tst_128x32_64x10_4x1_v_bz_bias_TNT` | 264 | 7.963 | 0.32% |
|
| 26 |
+
| 20 | `nvjet_tst_128x8_64x12_4x1_v_bz_bias_TNT` | 240 | 6.732 | 0.27% |
|
| 27 |
+
|
| 28 |
+
# FP8 trace
|
| 29 |
+
trace: /data/bbuf/hunyuanvideo_fp8_20260420/profiler/fp8/63e4807c-f120-4fdc-b5b0-c1ab96522a7e-5_steps-global-rank0.trace.json.gz
|
| 30 |
+
total kernel time: 2088.351 ms
|
| 31 |
+
|
| 32 |
+
| rank | kernel | count | time_ms | share |
|
| 33 |
+
|---:|---|---:|---:|---:|
|
| 34 |
+
| 1 | `void cutlass::device_kernel<flash::enable_sm90_or_later<flash::FlashAttnFwdSm90<flash::CollectiveMainloopFwdSm90<2, cute::tuple<cute::C<2>, ` | 360 | 692.486 | 33.16% |
|
| 35 |
+
| 2 | `_ZN7cutlass13device_kernelINS_4gemm6kernel13GemmUniversalIN4cute5tupleIJiiiiEEENS1_10collective13CollectiveMmaINS1_34MainloopSm90TmaGmmaWarp` | 960 | 650.610 | 31.15% |
|
| 36 |
+
| 3 | `void at::native::(anonymous namespace)::CatArrayBatchedCopy<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int, 4, 64, 64>(at::` | 720 | 152.032 | 7.28% |
|
| 37 |
+
| 4 | `void at::native::(anonymous namespace)::CatArrayBatchedCopy_alignedK_contig<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int,` | 600 | 112.280 | 5.38% |
|
| 38 |
+
| 5 | `void at::native::elementwise_kernel<128, 4, at::native::gpu_kernel_impl_nocast<at::native::GeluCUDAKernelImpl(at::TensorIteratorBase&, at::n` | 240 | 95.617 | 4.58% |
|
| 39 |
+
| 6 | `void at::native::(anonymous namespace)::CatArrayBatchedCopy_alignedK_contig<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int,` | 246 | 73.946 | 3.54% |
|
| 40 |
+
| 7 | `_static_quant_fp8` | 1440 | 71.500 | 3.42% |
|
| 41 |
+
| 8 | `void at::native::elementwise_kernel<128, 4, at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&):` | 972 | 62.236 | 2.98% |
|
| 42 |
+
| 9 | `_rotary_embedding_kernel` | 720 | 42.578 | 2.04% |
|
| 43 |
+
| 10 | `_rms_norm_tiled_onepass` | 960 | 29.710 | 1.42% |
|
| 44 |
+
| 11 | `fuse_scale_shift_kernel_blc_opt` | 480 | 21.832 | 1.05% |
|
| 45 |
+
| 12 | `void at::native::vectorized_elementwise_kernel<8, at::native::GeluCUDAKernelImpl(at::TensorIteratorBase&, at::native::GeluType)::{lambda()#1` | 240 | 21.108 | 1.01% |
|
| 46 |
+
| 13 | `kernel_cutlass_kernel_sglangjit_kerneldiffusioncutedslscale_residual_norm_scale_shiftScaleResidualNormScaleShift_object_at__tensorptrbf16_gm` | 480 | 18.938 | 0.91% |
|
| 47 |
+
| 14 | `kernel_cutlass_kernel_sglangjit_kerneldiffusioncutedslscale_residual_norm_scale_shiftScaleResidualNormScaleShift_object_at__tensorptrbf16_gm` | 240 | 10.268 | 0.49% |
|
| 48 |
+
| 15 | `nvjet_tst_256x8_64x6_4x1_v_bz_bias_TNT` | 240 | 9.710 | 0.46% |
|
| 49 |
+
| 16 | `_ZN7cutlass13device_kernelINS_4gemm6kernel13GemmUniversalIN4cute5tupleIJiiiiEEENS1_10collective13CollectiveMmaINS1_34MainloopSm90TmaGmmaWarp` | 480 | 8.525 | 0.41% |
|
| 50 |
+
| 17 | `nvjet_tst_128x8_64x12_4x1_v_bz_bias_TNT` | 240 | 6.747 | 0.32% |
|
| 51 |
+
| 18 | `void at::native::vectorized_elementwise_kernel<8, at::native::(anonymous namespace)::silu_kernel(at::TensorIteratorBase&)::{lambda()#1}::ope` | 540 | 1.212 | 0.06% |
|
| 52 |
+
| 19 | `void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const:` | 90 | 0.932 | 0.04% |
|
| 53 |
+
| 20 | `void at::native::elementwise_kernel<128, 4, at::native::gpu_kernel_impl<at::native::BinaryFunctor<float, float, float, at::native::binary_in` | 6 | 0.849 | 0.04% |
|
validation/h100_20260420/result_summary.md
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# HunyuanVideo ModelOpt FP8 Validation Summary
|
| 2 |
+
|
| 3 |
+
HF repo: https://huggingface.co/BBuf/HunyuanVideo-ModelOpt-FP8-SGLang
|
| 4 |
+
|
| 5 |
+
## Environment
|
| 6 |
+
|
| 7 |
+
- Host/GPU: H100, rank0 (`CUDA_VISIBLE_DEVICES=0`)
|
| 8 |
+
- SGLang base: latest main plus Qwen Image FP8 PR changes and HunyuanVideo FP8 changes
|
| 9 |
+
- ModelOpt: latest main for quantization
|
| 10 |
+
- Base model: `hunyuanvideo-community/HunyuanVideo`
|
| 11 |
+
- Resolution: 960x544, 17 frames
|
| 12 |
+
- Steps: 8
|
| 13 |
+
- Seed: 0
|
| 14 |
+
- Offload: `--dit-cpu-offload false --dit-layerwise-offload false`
|
| 15 |
+
|
| 16 |
+
## Output Artifacts
|
| 17 |
+
|
| 18 |
+
- BF16 video: `artifacts/hunyuanvideo_bf16_544x960_17f_8steps.mp4`
|
| 19 |
+
- FP8 video: `artifacts/hunyuanvideo_fp8_544x960_17f_8steps.mp4`
|
| 20 |
+
- Contact sheet: `artifacts/hunyuanvideo_bf16_fp8_contact_sheet.png`
|
| 21 |
+
- BF16 perf JSON: `artifacts/hunyuanvideo_bf16_544x960_17f_8steps_perf.json`
|
| 22 |
+
- FP8 perf JSON: `artifacts/hunyuanvideo_fp8_544x960_17f_8steps_perf.json`
|
| 23 |
+
- Kernel summary: `profiler/kernel_summary.md`
|
| 24 |
+
|
| 25 |
+
## Benchmark
|
| 26 |
+
|
| 27 |
+
Offline throughput, random dataset, 3 prompts, batch size 1. FP8 uses `--transformer-path /data/bbuf/hunyuanvideo_fp8_20260420/sglang_transformer`.
|
| 28 |
+
|
| 29 |
+
| checkpoint | total duration | latency/request | requests/s | MP/s |
|
| 30 |
+
| --- | ---: | ---: | ---: | ---: |
|
| 31 |
+
| BF16 | 26.11s | 8.70s | 0.115 | 1.020 |
|
| 32 |
+
| FP8 | 24.90s | 8.30s | 0.120 | 1.070 |
|
| 33 |
+
|
| 34 |
+
Speedup:
|
| 35 |
+
|
| 36 |
+
- End-to-end latency: 1.049x
|
| 37 |
+
- End-to-end throughput: 1.049x
|
| 38 |
+
- Steady denoise step from logs: about 1.19x (`~0.379s/step` BF16 to `~0.319s/step` FP8)
|
| 39 |
+
- Transformer load memory: 23.88 GB BF16 to 15.45 GB FP8
|
| 40 |
+
|
| 41 |
+
## Commands
|
| 42 |
+
|
| 43 |
+
BF16 benchmark:
|
| 44 |
+
|
| 45 |
+
```bash
|
| 46 |
+
python -m sglang.multimodal_gen.benchmarks.bench_offline_throughput --backend=sglang --model-path /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773 --dataset random --num-prompts 3 --batch-size 1 --width 960 --height 544 --num-frames 17 --num-inference-steps 8 --guidance-scale 1.0 --seed 0 --dit-cpu-offload false --dit-layerwise-offload false
|
| 47 |
+
```
|
| 48 |
+
|
| 49 |
+
FP8 benchmark:
|
| 50 |
+
|
| 51 |
+
```bash
|
| 52 |
+
python -m sglang.multimodal_gen.benchmarks.bench_offline_throughput --backend=sglang --model-path /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773 --transformer-path /data/bbuf/hunyuanvideo_fp8_20260420/sglang_transformer --dataset random --num-prompts 3 --batch-size 1 --width 960 --height 544 --num-frames 17 --num-inference-steps 8 --guidance-scale 1.0 --seed 0 --dit-cpu-offload false --dit-layerwise-offload false
|
| 53 |
+
```
|