BBuf commited on
Commit
7e4ad16
·
verified ·
1 Parent(s): 6959738

Add H100 validation artifacts

Browse files
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ validation/h100_20260420/artifacts/hunyuanvideo_bf16_fp8_contact_sheet.png filter=lfs diff=lfs merge=lfs -text
validation/h100_20260420/artifacts/hunyuanvideo_bf16_544x960_17f_8steps.mp4 ADDED
Binary file (39.3 kB). View file
 
validation/h100_20260420/artifacts/hunyuanvideo_bf16_544x960_17f_8steps_perf.json ADDED
@@ -0,0 +1,87 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "timestamp": "2026-04-20T02:24:39.346613+00:00",
3
+ "request_id": "f14e3fae-01d7-4d44-bc16-ac805cdc4aa9",
4
+ "commit_hash": "N/A",
5
+ "tag": "cli_generate",
6
+ "total_duration_ms": 10885.856637032703,
7
+ "steps": [
8
+ {
9
+ "name": "InputValidationStage",
10
+ "duration_ms": 0.039525795727968216
11
+ },
12
+ {
13
+ "name": "TextEncodingStage",
14
+ "duration_ms": 952.3325660265982
15
+ },
16
+ {
17
+ "name": "TimestepPreparationStage",
18
+ "duration_ms": 0.37180911749601364
19
+ },
20
+ {
21
+ "name": "LatentPreparationStage",
22
+ "duration_ms": 0.8033178746700287
23
+ },
24
+ {
25
+ "name": "DenoisingStage",
26
+ "duration_ms": 4592.238221084699
27
+ },
28
+ {
29
+ "name": "DecodingStage",
30
+ "duration_ms": 5328.874800121412
31
+ }
32
+ ],
33
+ "denoise_steps_ms": [
34
+ {
35
+ "step": 0,
36
+ "duration_ms": 1984.2948359437287
37
+ },
38
+ {
39
+ "step": 1,
40
+ "duration_ms": 104.41974294371903
41
+ },
42
+ {
43
+ "step": 2,
44
+ "duration_ms": 412.1756220702082
45
+ },
46
+ {
47
+ "step": 3,
48
+ "duration_ms": 426.5072650741786
49
+ },
50
+ {
51
+ "step": 4,
52
+ "duration_ms": 406.0544071253389
53
+ },
54
+ {
55
+ "step": 5,
56
+ "duration_ms": 423.0810720473528
57
+ },
58
+ {
59
+ "step": 6,
60
+ "duration_ms": 414.66441797092557
61
+ },
62
+ {
63
+ "step": 7,
64
+ "duration_ms": 418.7105300370604
65
+ }
66
+ ],
67
+ "memory_checkpoints": {
68
+ "before_forward": {
69
+ "allocated_mb": 24456.18,
70
+ "reserved_mb": 24796.0,
71
+ "peak_allocated_mb": 24456.18,
72
+ "peak_reserved_mb": 24796.0
73
+ },
74
+ "after_forward": {
75
+ "allocated_mb": 24542.47,
76
+ "reserved_mb": 45960.0,
77
+ "peak_allocated_mb": 44298.8,
78
+ "peak_reserved_mb": 45960.0
79
+ }
80
+ },
81
+ "meta": {
82
+ "prompt": [
83
+ "A cinematic shot of a red sports car driving through rain at night, reflections on wet streets, smooth camera movement."
84
+ ],
85
+ "model": "/root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773"
86
+ }
87
+ }
validation/h100_20260420/artifacts/hunyuanvideo_bf16_fp8_contact_sheet.png ADDED

Git LFS Details

  • SHA256: a63683eee948bbc18e1735e01f4d3f8826eee71ecbc0c3bb0ad8862407cd09ee
  • Pointer size: 131 Bytes
  • Size of remote file: 154 kB
validation/h100_20260420/artifacts/hunyuanvideo_fp8_544x960_17f_8steps.mp4 ADDED
Binary file (37.2 kB). View file
 
validation/h100_20260420/artifacts/hunyuanvideo_fp8_544x960_17f_8steps_perf.json ADDED
@@ -0,0 +1,87 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "timestamp": "2026-04-20T02:25:58.322820+00:00",
3
+ "request_id": "221d84f7-7f54-4343-9d8a-13ca3fa105dd",
4
+ "commit_hash": "N/A",
5
+ "tag": "cli_generate",
6
+ "total_duration_ms": 10391.746508190408,
7
+ "steps": [
8
+ {
9
+ "name": "InputValidationStage",
10
+ "duration_ms": 0.036803074181079865
11
+ },
12
+ {
13
+ "name": "TextEncodingStage",
14
+ "duration_ms": 973.2257057912648
15
+ },
16
+ {
17
+ "name": "TimestepPreparationStage",
18
+ "duration_ms": 0.33611198887228966
19
+ },
20
+ {
21
+ "name": "LatentPreparationStage",
22
+ "duration_ms": 0.7671341300010681
23
+ },
24
+ {
25
+ "name": "DenoisingStage",
26
+ "duration_ms": 4008.4011738654226
27
+ },
28
+ {
29
+ "name": "DecodingStage",
30
+ "duration_ms": 5397.510214010254
31
+ }
32
+ ],
33
+ "denoise_steps_ms": [
34
+ {
35
+ "step": 0,
36
+ "duration_ms": 1814.3004791345447
37
+ },
38
+ {
39
+ "step": 1,
40
+ "duration_ms": 89.51880293898284
41
+ },
42
+ {
43
+ "step": 2,
44
+ "duration_ms": 345.99901596084237
45
+ },
46
+ {
47
+ "step": 3,
48
+ "duration_ms": 357.83073981292546
49
+ },
50
+ {
51
+ "step": 4,
52
+ "duration_ms": 347.1649489365518
53
+ },
54
+ {
55
+ "step": 5,
56
+ "duration_ms": 348.6302839592099
57
+ },
58
+ {
59
+ "step": 6,
60
+ "duration_ms": 355.295421089977
61
+ },
62
+ {
63
+ "step": 7,
64
+ "duration_ms": 347.20492409542203
65
+ }
66
+ ],
67
+ "memory_checkpoints": {
68
+ "before_forward": {
69
+ "allocated_mb": 15906.27,
70
+ "reserved_mb": 15964.0,
71
+ "peak_allocated_mb": 15906.27,
72
+ "peak_reserved_mb": 15964.0
73
+ },
74
+ "after_forward": {
75
+ "allocated_mb": 15992.56,
76
+ "reserved_mb": 51272.0,
77
+ "peak_allocated_mb": 35748.73,
78
+ "peak_reserved_mb": 51272.0
79
+ }
80
+ },
81
+ "meta": {
82
+ "prompt": [
83
+ "A cinematic shot of a red sports car driving through rain at night, reflections on wet streets, smooth camera movement."
84
+ ],
85
+ "model": "/root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773"
86
+ }
87
+ }
validation/h100_20260420/benchmark/bf16_offline_throughput.jsonl ADDED
@@ -0,0 +1 @@
 
 
1
+ {"metadata": {"timestamp": "2026-04-20T02:27:47", "model_path": "/root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773", "task_type": "unknown", "backend": "engine"}, "configuration": {"num_inference_steps": 8, "guidance_scale": 1.0, "seed": 0, "batch_size": 1, "num_prompts": 3, "resolution": "960x544x17", "dataset": "random"}, "results": {"num_requests": 3, "successful_requests": 3, "failed_requests": 0, "total_duration_seconds": 26.11314248898998, "total_frames_generated": 3, "total_pixels_generated": 26634240, "images_per_second": 0.11488467928610595, "frames_per_second": 0.11488467928610595, "megapixels_per_second": 1.0199553734763915, "requests_per_second": 0.11488467928610595, "latency_per_request_seconds": 8.704380829663327, "peak_memory_mb": 0.0}}
validation/h100_20260420/benchmark/fp8_offline_throughput.jsonl ADDED
@@ -0,0 +1 @@
 
 
1
+ {"metadata": {"timestamp": "2026-04-20T02:31:53", "model_path": "/data/bbuf/hunyuanvideo_fp8_20260420/sglang_fp8_pipeline", "task_type": "unknown", "backend": "engine"}, "configuration": {"num_inference_steps": 8, "guidance_scale": 1.0, "seed": 0, "batch_size": 1, "num_prompts": 3, "resolution": "960x544x17", "dataset": "random"}, "results": {"num_requests": 3, "successful_requests": 3, "failed_requests": 0, "total_duration_seconds": 24.55099329398945, "total_frames_generated": 3, "total_pixels_generated": 26634240, "images_per_second": 0.12219464866761448, "frames_per_second": 0.12219464866761448, "megapixels_per_second": 1.0848538664429748, "requests_per_second": 0.12219464866761448, "latency_per_request_seconds": 8.183664431329817, "peak_memory_mb": 0.0}}
validation/h100_20260420/benchmark/fp8_offline_throughput_transformer_path.jsonl ADDED
@@ -0,0 +1 @@
 
 
1
+ {"metadata": {"timestamp": "2026-04-20T02:50:19", "model_path": "/root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773", "task_type": "unknown", "backend": "engine"}, "configuration": {"num_inference_steps": 8, "guidance_scale": 1.0, "seed": 0, "batch_size": 1, "num_prompts": 3, "resolution": "960x544x17", "dataset": "random"}, "results": {"num_requests": 3, "successful_requests": 3, "failed_requests": 0, "total_duration_seconds": 24.901405425043777, "total_frames_generated": 3, "total_pixels_generated": 26634240, "images_per_second": 0.12047512776057402, "frames_per_second": 0.12047512776057402, "megapixels_per_second": 1.069587822268597, "requests_per_second": 0.12047512776057402, "latency_per_request_seconds": 8.300468475014592, "peak_memory_mb": 0.0}}
validation/h100_20260420/logs/bench_bf16.log ADDED
@@ -0,0 +1,229 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
0
  0%| | 0/8 [00:00<?, ?it/s]
1
  12%|█▎ | 1/8 [00:02<00:14, 2.06s/it]
2
  25%|██▌ | 2/8 [00:02<00:05, 1.10it/s]
3
  38%|███▊ | 3/8 [00:02<00:03, 1.47it/s]
4
  50%|█████ | 4/8 [00:03<00:02, 1.72it/s]
5
  62%|██████▎ | 5/8 [00:03<00:01, 1.93it/s]
6
  75%|███████▌ | 6/8 [00:03<00:00, 2.05it/s]
7
  88%|████████▊ | 7/8 [00:04<00:00, 2.16it/s]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
8
  0%| | 0/8 [00:00<?, ?it/s]
9
  12%|█▎ | 1/8 [00:00<00:00, 9.62it/s]
10
  25%|██▌ | 2/8 [00:00<00:01, 3.56it/s]
11
  38%|███▊ | 3/8 [00:00<00:01, 2.90it/s]
12
  50%|█████ | 4/8 [00:01<00:01, 2.68it/s]
13
  62%|██████▎ | 5/8 [00:01<00:01, 2.58it/s]
14
  75%|███████▌ | 6/8 [00:02<00:00, 2.49it/s]
15
  88%|████████▊ | 7/8 [00:02<00:00, 2.47it/s]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
16
  0%| | 0/8 [00:00<?, ?it/s]
17
  12%|█▎ | 1/8 [00:00<00:00, 9.66it/s]
18
  25%|██▌ | 2/8 [00:00<00:01, 3.54it/s]
19
  38%|███▊ | 3/8 [00:00<00:01, 2.89it/s]
20
  50%|█████ | 4/8 [00:01<00:01, 2.67it/s]
21
  62%|██████▎ | 5/8 [00:01<00:01, 2.57it/s]
22
  75%|███████▌ | 6/8 [00:02<00:00, 2.47it/s]
23
  88%|████████▊ | 7/8 [00:02<00:00, 2.46it/s]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
24
  0%| | 0/8 [00:00<?, ?it/s]
25
  12%|█▎ | 1/8 [00:00<00:00, 9.68it/s]
26
  25%|██▌ | 2/8 [00:00<00:01, 3.55it/s]
27
  38%|███▊ | 3/8 [00:00<00:01, 2.88it/s]
28
  50%|█████ | 4/8 [00:01<00:01, 2.66it/s]
29
  62%|██████▎ | 5/8 [00:01<00:01, 2.57it/s]
30
  75%|███████▌ | 6/8 [00:02<00:00, 2.48it/s]
31
  88%|████████▊ | 7/8 [00:02<00:00, 2.46it/s]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ No platform detected. Using base SRTPlatform with defaults.
2
+ [04-20 02:26:33] server_args: {"model_path": "/root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773", "model_id": null, "backend": "sglang", "attention_backend": null, "attention_backend_config": {}, "cache_dit_config": null, "nccl_port": null, "trust_remote_code": false, "revision": null, "num_gpus": 1, "tp_size": 1, "sp_degree": 1, "ulysses_degree": 1, "ring_degree": 1, "dp_size": 1, "dp_degree": 1, "enable_cfg_parallel": false, "hsdp_replicate_dim": 1, "hsdp_shard_dim": 1, "dist_timeout": 3600, "pipeline_class_name": null, "lora_path": null, "lora_nickname": "default", "lora_scale": 1.0, "lora_weight_name": null, "component_paths": {}, "transformer_weights_path": null, "lora_target_modules": null, "dit_cpu_offload": false, "dit_layerwise_offload": false, "dit_offload_prefetch_size": 0.0, "text_encoder_cpu_offload": true, "image_encoder_cpu_offload": true, "vae_cpu_offload": true, "use_fsdp_inference": false, "pin_cpu_memory": true, "ltx2_two_stage_device_mode": null, "comfyui_mode": false, "enable_torch_compile": false, "warmup": false, "warmup_resolutions": null, "warmup_steps": 1, "disable_autocast": false, "master_port": 30005, "host": "127.0.0.1", "port": 30000, "webui": false, "webui_port": 12312, "scheduler_port": 5644, "strict_ports": false, "output_path": "outputs/", "input_save_path": "inputs/uploads", "prompt_file_path": null, "model_paths": {}, "model_loaded": {"transformer": true, "vae": true, "video_vae": true, "audio_vae": true, "video_dit": true, "audio_dit": true, "dual_tower_bridge": true}, "boundary_ratio": null, "base_gpu_id": 0, "disagg_role": "monolithic", "disagg_timeout": 600, "disagg_dispatch_policy": "round_robin", "disagg_mode": false, "disagg_server_addr": null, "encoder_urls": null, "denoiser_urls": null, "decoder_urls": null, "encoder_tp": null, "denoiser_tp": null, "denoiser_sp": null, "denoiser_ulysses": null, "denoiser_ring": null, "decoder_tp": null, "disagg_transfer_pool_size": 268435456, "disagg_p2p_hostname": "127.0.0.1", "disagg_ib_device": null, "pool_work_endpoint": null, "pool_result_endpoint": null, "log_level": "info", "uvicorn_access_log_exclude_prefixes": []}
3
+ [04-20 02:26:33] Starting offline throughput benchmark...
4
+ [04-20 02:26:33] Initializing engine...
5
+ [04-20 02:26:33] Local mode: True
6
+ [04-20 02:26:33] Starting server...
7
+ No platform detected. Using base SRTPlatform with defaults.
8
+ [04-20 02:26:40] Scheduler bind at endpoint: tcp://127.0.0.1:5644
9
+ [04-20 02:26:40] Initializing distributed environment with world_size=1, device=cuda:0, timeout=3600
10
+ [04-20 02:26:40] Setting distributed timeout to 3600 seconds
11
+ [04-20 02:26:41] No pipeline_class_name specified, using model_index.json
12
+ [04-20 02:26:41] Diffusers version: 0.32.0.dev0
13
+ [04-20 02:26:41] Using pipeline from model_index.json: HunyuanVideoPipeline
14
+ [04-20 02:26:41] Loading pipeline modules...
15
+ [04-20 02:26:41] Model already exists locally and is complete
16
+ [04-20 02:26:41] Model path: /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773
17
+ [04-20 02:26:41] Diffusers version: 0.32.0.dev0
18
+ [04-20 02:26:41] Loading pipeline modules from config: {'_class_name': 'HunyuanVideoPipeline', '_diffusers_version': '0.32.0.dev0', 'scheduler': ['diffusers', 'FlowMatchEulerDiscreteScheduler'], 'text_encoder': ['transformers', 'LlamaModel'], 'text_encoder_2': ['transformers', 'CLIPTextModel'], 'tokenizer': ['transformers', 'LlamaTokenizerFast'], 'tokenizer_2': ['transformers', 'CLIPTokenizer'], 'transformer': ['diffusers', 'HunyuanVideoTransformer3DModel'], 'vae': ['diffusers', 'AutoencoderKLHunyuanVideo']}
19
+ [04-20 02:26:41] Loading required components: ['text_encoder', 'text_encoder_2', 'tokenizer', 'tokenizer_2', 'vae', 'transformer', 'scheduler']
20
+
21
+ [04-20 02:26:41] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
22
+ [04-20 02:26:43] [RunAI Streamer] Overall time to stream 14.0 GiB of all files to cpu: 2.07s, 6.8 GiB/s
23
+ [04-20 02:27:01] Loaded text_encoder: FSDPLlamaModel (sgl-diffusion version). model size: 13.98 GB, consumed GPU mem: 0.81 GB, avail GPU mem: 73.04 GB
24
+
25
+ [04-20 02:27:02] Using Torch SDPA backend
26
+ [04-20 02:27:02] [RunAI Streamer] Overall time to stream 234.7 MiB of all files to cpu: 0.03s, 7.7 GiB/s
27
+ [04-20 02:27:03] Loaded text_encoder_2: FSDPCLIPTextModel (sgl-diffusion version). model size: 0.23 GB, consumed GPU mem: 0.76 GB, avail GPU mem: 72.27 GB
28
+
29
+ [04-20 02:27:04] Loaded tokenizer: LlamaTokenizer (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
30
+
31
+ [04-20 02:27:04] Loaded tokenizer_2: CLIPTokenizer (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
32
+ [04-20 02:27:04] Loading vae from /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773/vae. avail mem: 72.27 GB
33
+ [04-20 02:27:04] VAE unexpected keys: ['encoder.conv_in.conv.bias', 'encoder.conv_in.conv.weight', 'encoder.conv_norm_out.bias', 'encoder.conv_norm_out.weight', 'encoder.conv_out.conv.bias', 'encoder.conv_out.conv.weight', 'encoder.down_blocks.0.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.0.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.0.resnets.0.conv1.conv.bias', 'encoder.down_blocks.0.resnets.0.conv1.conv.weight', 'encoder.down_blocks.0.resnets.0.conv2.conv.bias', 'encoder.down_blocks.0.resnets.0.conv2.conv.weight', 'encoder.down_blocks.0.resnets.0.norm1.bias', 'encoder.down_blocks.0.resnets.0.norm1.weight', 'encoder.down_blocks.0.resnets.0.norm2.bias', 'encoder.down_blocks.0.resnets.0.norm2.weight', 'encoder.down_blocks.0.resnets.1.conv1.conv.bias', 'encoder.down_blocks.0.resnets.1.conv1.conv.weight', 'encoder.down_blocks.0.resnets.1.conv2.conv.bias', 'encoder.down_blocks.0.resnets.1.conv2.conv.weight', 'encoder.down_blocks.0.resnets.1.norm1.bias', 'encoder.down_blocks.0.resnets.1.norm1.weight', 'encoder.down_blocks.0.resnets.1.norm2.bias', 'encoder.down_blocks.0.resnets.1.norm2.weight', 'encoder.down_blocks.1.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.1.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.1.resnets.0.conv1.conv.bias', 'encoder.down_blocks.1.resnets.0.conv1.conv.weight', 'encoder.down_blocks.1.resnets.0.conv2.conv.bias', 'encoder.down_blocks.1.resnets.0.conv2.conv.weight', 'encoder.down_blocks.1.resnets.0.conv_shortcut.conv.bias', 'encoder.down_blocks.1.resnets.0.conv_shortcut.conv.weight', 'encoder.down_blocks.1.resnets.0.norm1.bias', 'encoder.down_blocks.1.resnets.0.norm1.weight', 'encoder.down_blocks.1.resnets.0.norm2.bias', 'encoder.down_blocks.1.resnets.0.norm2.weight', 'encoder.down_blocks.1.resnets.1.conv1.conv.bias', 'encoder.down_blocks.1.resnets.1.conv1.conv.weight', 'encoder.down_blocks.1.resnets.1.conv2.conv.bias', 'encoder.down_blocks.1.resnets.1.conv2.conv.weight', 'encoder.down_blocks.1.resnets.1.norm1.bias', 'encoder.down_blocks.1.resnets.1.norm1.weight', 'encoder.down_blocks.1.resnets.1.norm2.bias', 'encoder.down_blocks.1.resnets.1.norm2.weight', 'encoder.down_blocks.2.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.2.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.2.resnets.0.conv1.conv.bias', 'encoder.down_blocks.2.resnets.0.conv1.conv.weight', 'encoder.down_blocks.2.resnets.0.conv2.conv.bias', 'encoder.down_blocks.2.resnets.0.conv2.conv.weight', 'encoder.down_blocks.2.resnets.0.conv_shortcut.conv.bias', 'encoder.down_blocks.2.resnets.0.conv_shortcut.conv.weight', 'encoder.down_blocks.2.resnets.0.norm1.bias', 'encoder.down_blocks.2.resnets.0.norm1.weight', 'encoder.down_blocks.2.resnets.0.norm2.bias', 'encoder.down_blocks.2.resnets.0.norm2.weight', 'encoder.down_blocks.2.resnets.1.conv1.conv.bias', 'encoder.down_blocks.2.resnets.1.conv1.conv.weight', 'encoder.down_blocks.2.resnets.1.conv2.conv.bias', 'encoder.down_blocks.2.resnets.1.conv2.conv.weight', 'encoder.down_blocks.2.resnets.1.norm1.bias', 'encoder.down_blocks.2.resnets.1.norm1.weight', 'encoder.down_blocks.2.resnets.1.norm2.bias', 'encoder.down_blocks.2.resnets.1.norm2.weight', 'encoder.down_blocks.3.resnets.0.conv1.conv.bias', 'encoder.down_blocks.3.resnets.0.conv1.conv.weight', 'encoder.down_blocks.3.resnets.0.conv2.conv.bias', 'encoder.down_blocks.3.resnets.0.conv2.conv.weight', 'encoder.down_blocks.3.resnets.0.norm1.bias', 'encoder.down_blocks.3.resnets.0.norm1.weight', 'encoder.down_blocks.3.resnets.0.norm2.bias', 'encoder.down_blocks.3.resnets.0.norm2.weight', 'encoder.down_blocks.3.resnets.1.conv1.conv.bias', 'encoder.down_blocks.3.resnets.1.conv1.conv.weight', 'encoder.down_blocks.3.resnets.1.conv2.conv.bias', 'encoder.down_blocks.3.resnets.1.conv2.conv.weight', 'encoder.down_blocks.3.resnets.1.norm1.bias', 'encoder.down_blocks.3.resnets.1.norm1.weight', 'encoder.down_blocks.3.resnets.1.norm2.bias', 'encoder.down_blocks.3.resnets.1.norm2.weight', 'encoder.mid_block.attentions.0.group_norm.bias', 'encoder.mid_block.attentions.0.group_norm.weight', 'encoder.mid_block.attentions.0.to_k.bias', 'encoder.mid_block.attentions.0.to_k.weight', 'encoder.mid_block.attentions.0.to_out.0.bias', 'encoder.mid_block.attentions.0.to_out.0.weight', 'encoder.mid_block.attentions.0.to_q.bias', 'encoder.mid_block.attentions.0.to_q.weight', 'encoder.mid_block.attentions.0.to_v.bias', 'encoder.mid_block.attentions.0.to_v.weight', 'encoder.mid_block.resnets.0.conv1.conv.bias', 'encoder.mid_block.resnets.0.conv1.conv.weight', 'encoder.mid_block.resnets.0.conv2.conv.bias', 'encoder.mid_block.resnets.0.conv2.conv.weight', 'encoder.mid_block.resnets.0.norm1.bias', 'encoder.mid_block.resnets.0.norm1.weight', 'encoder.mid_block.resnets.0.norm2.bias', 'encoder.mid_block.resnets.0.norm2.weight', 'encoder.mid_block.resnets.1.conv1.conv.bias', 'encoder.mid_block.resnets.1.conv1.conv.weight', 'encoder.mid_block.resnets.1.conv2.conv.bias', 'encoder.mid_block.resnets.1.conv2.conv.weight', 'encoder.mid_block.resnets.1.norm1.bias', 'encoder.mid_block.resnets.1.norm1.weight', 'encoder.mid_block.resnets.1.norm2.bias', 'encoder.mid_block.resnets.1.norm2.weight', 'quant_conv.bias', 'quant_conv.weight']
34
+ [04-20 02:27:04] Loaded vae: AutoencoderKLHunyuanVideo (sgl-diffusion version). model size: 0.27 GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
35
+ [04-20 02:27:04] Loading transformer from /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773/transformer. avail mem: 72.27 GB
36
+ [04-20 02:27:04] Loading HunyuanVideoTransformer3DModel from 6 safetensors file(s) , param_dtype: torch.bfloat16
37
+ [04-20 02:27:04] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
38
+ [04-20 02:27:04] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
39
+ [04-20 02:27:05] [RunAI Streamer] Overall time to stream 23.9 GiB of all files to cpu: 1.65s, 14.5 GiB/s
40
+ [04-20 02:27:09] Loaded model with 12.82B parameters
41
+ [04-20 02:27:09] Loaded transformer: HunyuanVideoTransformer3DModel (sgl-diffusion version). model size: 23.88 GB, consumed GPU mem: 24.19 GB, avail GPU mem: 48.08 GB
42
+
43
+ [04-20 02:27:09] Loaded scheduler: FlowMatchEulerDiscreteScheduler (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 48.08 GB
44
+
45
+ [04-20 02:27:09] Creating pipeline stages...
46
+ [04-20 02:27:09] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
47
+ [04-20 02:27:09] Pipeline instantiated
48
+ [04-20 02:27:09] Worker 0: Initialized device, model, and distributed environment.
49
+ [04-20 02:27:09] Worker 0: Scheduler loop started.
50
+ [04-20 02:27:09] Engine initialized successfully
51
+ [04-20 02:27:09] Loading random dataset...
52
+ [04-20 02:27:09] Running warmup batch...
53
+ [04-20 02:27:09] Adjusting number of frames from 17 to 17 based on model
54
+ [04-20 02:27:09] Processing prompt 1/1: <redacted, len=49>
55
+ [04-20 02:27:09] Sampling params:
56
+ width: 960
57
+ height: 544
58
+ num_frames: 17
59
+ fps: 24
60
+ prompt: <redacted, len=49>
61
+ neg_prompt: <redacted, len=392>
62
+ seed: 0
63
+ infer_steps: 8
64
+ num_outputs_per_prompt: 1
65
+ guidance_scale: 1.0
66
+ embedded_guidance_scale: 6
67
+ n_tokens: None
68
+ flow_shift: 7
69
+ image_path: None
70
+ save_output: True
71
+ output_file_path: outputs/Random_prompt_0_for_benchmarking_diffusion_models_20260420-022709_b99ddeb1.mp4
72
+
73
+ [04-20 02:27:09] Running pipeline stages: ['InputValidationStage', 'prompt_encoding_stage_primary', 'TimestepPreparationStage', 'LatentPreparationStage', 'DenoisingStage', 'DecodingStage']
74
+ [04-20 02:27:09] [InputValidationStage] started...
75
+ [04-20 02:27:09] [InputValidationStage] finished in 0.0001 seconds
76
+ [04-20 02:27:09] [TextEncodingStage] started...
77
+ [04-20 02:27:10] [TextEncodingStage] finished in 1.0550 seconds
78
+ [04-20 02:27:10] [TimestepPreparationStage] started...
79
+ [04-20 02:27:10] [TimestepPreparationStage] finished in 0.0004 seconds
80
+ [04-20 02:27:10] [LatentPreparationStage] started...
81
+ [04-20 02:27:10] [LatentPreparationStage] finished in 0.0014 seconds
82
+ [04-20 02:27:10] [DenoisingStage] started...
83
+
84
  0%| | 0/8 [00:00<?, ?it/s]
85
  12%|█▎ | 1/8 [00:02<00:14, 2.06s/it]
86
  25%|██▌ | 2/8 [00:02<00:05, 1.10it/s]
87
  38%|███▊ | 3/8 [00:02<00:03, 1.47it/s]
88
  50%|█████ | 4/8 [00:03<00:02, 1.72it/s]
89
  62%|██████▎ | 5/8 [00:03<00:01, 1.93it/s]
90
  75%|███████▌ | 6/8 [00:03<00:00, 2.05it/s]
91
  88%|████████▊ | 7/8 [00:04<00:00, 2.16it/s]
92
+ [04-20 02:27:15] [DenoisingStage] average time per step: 0.5834 seconds
93
+ [04-20 02:27:15] [DenoisingStage] finished in 4.6707 seconds
94
+ [04-20 02:27:15] [DecodingStage] started...
95
+ [04-20 02:27:21] [DecodingStage] finished in 5.4083 seconds
96
+ [04-20 02:27:21] Output saved to outputs/Random_prompt_0_for_benchmarking_diffusion_models_20260420-022709_b99ddeb1.mp4
97
+ [04-20 02:27:21] Pixel data generated successfully in 11.42 seconds
98
+ [04-20 02:27:21] Completed batch processing. Generated 1 outputs in 11.42 seconds
99
+ [04-20 02:27:21] Running benchmark with 3 prompts...
100
+ [04-20 02:27:21] Adjusting number of frames from 17 to 17 based on model
101
+ [04-20 02:27:21] Processing prompt 1/1: <redacted, len=49>
102
+ [04-20 02:27:21] Sampling params:
103
+ width: 960
104
+ height: 544
105
+ num_frames: 17
106
+ fps: 24
107
+ prompt: <redacted, len=49>
108
+ neg_prompt: <redacted, len=392>
109
+ seed: 0
110
+ infer_steps: 8
111
+ num_outputs_per_prompt: 1
112
+ guidance_scale: 1.0
113
+ embedded_guidance_scale: 6
114
+ n_tokens: None
115
+ flow_shift: 7
116
+ image_path: None
117
+ save_output: True
118
+ output_file_path: outputs/Random_prompt_0_for_benchmarking_diffusion_models_20260420-022721_b99ddeb1.mp4
119
+
120
+ [04-20 02:27:21] Running pipeline stages: ['InputValidationStage', 'prompt_encoding_stage_primary', 'TimestepPreparationStage', 'LatentPreparationStage', 'DenoisingStage', 'DecodingStage']
121
+ [04-20 02:27:21] [InputValidationStage] started...
122
+ [04-20 02:27:21] [InputValidationStage] finished in 0.0001 seconds
123
+ [04-20 02:27:21] [TextEncodingStage] started...
124
+ [04-20 02:27:21] [TextEncodingStage] finished in 0.3417 seconds
125
+ [04-20 02:27:21] [TimestepPreparationStage] started...
126
+ [04-20 02:27:21] [TimestepPreparationStage] finished in 0.0004 seconds
127
+ [04-20 02:27:21] [LatentPreparationStage] started...
128
+ [04-20 02:27:21] [LatentPreparationStage] finished in 0.0001 seconds
129
+ [04-20 02:27:21] [DenoisingStage] started...
130
+
131
  0%| | 0/8 [00:00<?, ?it/s]
132
  12%|█▎ | 1/8 [00:00<00:00, 9.62it/s]
133
  25%|██▌ | 2/8 [00:00<00:01, 3.56it/s]
134
  38%|███▊ | 3/8 [00:00<00:01, 2.90it/s]
135
  50%|█████ | 4/8 [00:01<00:01, 2.68it/s]
136
  62%|██████▎ | 5/8 [00:01<00:01, 2.58it/s]
137
  75%|███████▌ | 6/8 [00:02<00:00, 2.49it/s]
138
  88%|████████▊ | 7/8 [00:02<00:00, 2.47it/s]
139
+ [04-20 02:27:24] [DenoisingStage] average time per step: 0.3779 seconds
140
+ [04-20 02:27:24] [DenoisingStage] finished in 3.0250 seconds
141
+ [04-20 02:27:24] [DecodingStage] started...
142
+ [04-20 02:27:29] [DecodingStage] finished in 5.0738 seconds
143
+ [04-20 02:27:29] Output saved to outputs/Random_prompt_0_for_benchmarking_diffusion_models_20260420-022721_b99ddeb1.mp4
144
+ [04-20 02:27:29] Pixel data generated successfully in 8.70 seconds
145
+ [04-20 02:27:29] Completed batch processing. Generated 1 outputs in 8.70 seconds
146
+ [04-20 02:27:29] Adjusting number of frames from 17 to 17 based on model
147
+ [04-20 02:27:29] Processing prompt 1/1: <redacted, len=49>
148
+ [04-20 02:27:29] Sampling params:
149
+ width: 960
150
+ height: 544
151
+ num_frames: 17
152
+ fps: 24
153
+ prompt: <redacted, len=49>
154
+ neg_prompt: <redacted, len=392>
155
+ seed: 0
156
+ infer_steps: 8
157
+ num_outputs_per_prompt: 1
158
+ guidance_scale: 1.0
159
+ embedded_guidance_scale: 6
160
+ n_tokens: None
161
+ flow_shift: 7
162
+ image_path: None
163
+ save_output: True
164
+ output_file_path: outputs/Random_prompt_1_for_benchmarking_diffusion_models_20260420-022729_8d91326e.mp4
165
+
166
+ [04-20 02:27:29] Running pipeline stages: ['InputValidationStage', 'prompt_encoding_stage_primary', 'TimestepPreparationStage', 'LatentPreparationStage', 'DenoisingStage', 'DecodingStage']
167
+ [04-20 02:27:29] [InputValidationStage] started...
168
+ [04-20 02:27:29] [InputValidationStage] finished in 0.0001 seconds
169
+ [04-20 02:27:29] [TextEncodingStage] started...
170
+ [04-20 02:27:30] [TextEncodingStage] finished in 0.3422 seconds
171
+ [04-20 02:27:30] [TimestepPreparationStage] started...
172
+ [04-20 02:27:30] [TimestepPreparationStage] finished in 0.0004 seconds
173
+ [04-20 02:27:30] [LatentPreparationStage] started...
174
+ [04-20 02:27:30] [LatentPreparationStage] finished in 0.0001 seconds
175
+ [04-20 02:27:30] [DenoisingStage] started...
176
+
177
  0%| | 0/8 [00:00<?, ?it/s]
178
  12%|█▎ | 1/8 [00:00<00:00, 9.66it/s]
179
  25%|██▌ | 2/8 [00:00<00:01, 3.54it/s]
180
  38%|███▊ | 3/8 [00:00<00:01, 2.89it/s]
181
  50%|█████ | 4/8 [00:01<00:01, 2.67it/s]
182
  62%|██████▎ | 5/8 [00:01<00:01, 2.57it/s]
183
  75%|███████▌ | 6/8 [00:02<00:00, 2.47it/s]
184
  88%|████████▊ | 7/8 [00:02<00:00, 2.46it/s]
185
+ [04-20 02:27:33] [DenoisingStage] average time per step: 0.3794 seconds
186
+ [04-20 02:27:33] [DenoisingStage] finished in 3.0371 seconds
187
+ [04-20 02:27:33] [DecodingStage] started...
188
+ [04-20 02:27:38] [DecodingStage] finished in 5.0608 seconds
189
+ [04-20 02:27:38] Output saved to outputs/Random_prompt_1_for_benchmarking_diffusion_models_20260420-022729_8d91326e.mp4
190
+ [04-20 02:27:38] Pixel data generated successfully in 8.70 seconds
191
+ [04-20 02:27:38] Completed batch processing. Generated 1 outputs in 8.70 seconds
192
+ [04-20 02:27:38] Adjusting number of frames from 17 to 17 based on model
193
+ [04-20 02:27:38] Processing prompt 1/1: <redacted, len=49>
194
+ [04-20 02:27:38] Sampling params:
195
+ width: 960
196
+ height: 544
197
+ num_frames: 17
198
+ fps: 24
199
+ prompt: <redacted, len=49>
200
+ neg_prompt: <redacted, len=392>
201
+ seed: 0
202
+ infer_steps: 8
203
+ num_outputs_per_prompt: 1
204
+ guidance_scale: 1.0
205
+ embedded_guidance_scale: 6
206
+ n_tokens: None
207
+ flow_shift: 7
208
+ image_path: None
209
+ save_output: True
210
+ output_file_path: outputs/Random_prompt_2_for_benchmarking_diffusion_models_20260420-022738_9a3de0d5.mp4
211
+
212
+ [04-20 02:27:38] Running pipeline stages: ['InputValidationStage', 'prompt_encoding_stage_primary', 'TimestepPreparationStage', 'LatentPreparationStage', 'DenoisingStage', 'DecodingStage']
213
+ [04-20 02:27:38] [InputValidationStage] started...
214
+ [04-20 02:27:38] [InputValidationStage] finished in 0.0000 seconds
215
+ [04-20 02:27:38] [TextEncodingStage] started...
216
+ [04-20 02:27:39] [TextEncodingStage] finished in 0.3417 seconds
217
+ [04-20 02:27:39] [TimestepPreparationStage] started...
218
+ [04-20 02:27:39] [TimestepPreparationStage] finished in 0.0004 seconds
219
+ [04-20 02:27:39] [LatentPreparationStage] started...
220
+ [04-20 02:27:39] [LatentPreparationStage] finished in 0.0001 seconds
221
+ [04-20 02:27:39] [DenoisingStage] started...
222
+
223
  0%| | 0/8 [00:00<?, ?it/s]
224
  12%|█▎ | 1/8 [00:00<00:00, 9.68it/s]
225
  25%|██▌ | 2/8 [00:00<00:01, 3.55it/s]
226
  38%|███▊ | 3/8 [00:00<00:01, 2.88it/s]
227
  50%|█████ | 4/8 [00:01<00:01, 2.66it/s]
228
  62%|██████▎ | 5/8 [00:01<00:01, 2.57it/s]
229
  75%|███████▌ | 6/8 [00:02<00:00, 2.48it/s]
230
  88%|████████▊ | 7/8 [00:02<00:00, 2.46it/s]
231
+ [04-20 02:27:42] [DenoisingStage] average time per step: 0.3796 seconds
232
+ [04-20 02:27:42] [DenoisingStage] finished in 3.0391 seconds
233
+ [04-20 02:27:42] [DecodingStage] started...
234
+ [04-20 02:27:47] [DecodingStage] finished in 5.0717 seconds
235
+ [04-20 02:27:47] Output saved to outputs/Random_prompt_2_for_benchmarking_diffusion_models_20260420-022738_9a3de0d5.mp4
236
+ [04-20 02:27:47] Pixel data generated successfully in 8.71 seconds
237
+ [04-20 02:27:47] Completed batch processing. Generated 1 outputs in 8.71 seconds
238
+
239
+ ==================================== Offline Throughput Benchmark Result =====================================
240
+ Model: /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773
241
+ Dataset: random
242
+ Resolution: 960x544x17
243
+ Num Inference Steps: 8
244
+ ---------------------------------------------------------------------------
245
+ Total Requests: 3
246
+ Successful Requests: 3
247
+ Failed Requests: 0
248
+ Total Duration (seconds): 26.11
249
+ ---------------------------------------------------------------------------
250
+ Frames Generated: 3
251
+ Megapixels Generated: 26.63
252
+ ---------------------------------------------------------------------------
253
+ Frame Throughput (frames/sec): 0.11
254
+ MP Throughput (MP/sec): 1.02
255
+ Requests Per Second: 0.11
256
+ Latency Per Request (sec): 8.70
257
+ Peak Memory (MB): 0.00
258
+ ==============================================================================================================
259
+ [04-20 02:27:47] Results saved to /data/bbuf/hunyuanvideo_fp8_20260420/benchmark/bf16_offline_throughput.jsonl
260
+ [04-20 02:27:47] Generator was garbage collected without being shut down. Attempting to shut down the local server and client.
261
+ [04-20 02:27:54] Worker 0: Shutdown complete.
validation/h100_20260420/logs/bench_fp8_transformer_path.log ADDED
@@ -0,0 +1,231 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
0
  0%| | 0/8 [00:00<?, ?it/s]
1
  12%|█▎ | 1/8 [00:01<00:12, 1.86s/it]
2
  38%|███▊ | 3/8 [00:02<00:03, 1.55it/s]
3
  50%|█████ | 4/8 [00:02<00:02, 1.84it/s]
4
  62%|██████▎ | 5/8 [00:02<00:01, 2.09it/s]
5
  75%|███████▌ | 6/8 [00:03<00:00, 2.28it/s]
6
  88%|████████▊ | 7/8 [00:03<00:00, 2.42it/s]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
7
  0%| | 0/8 [00:00<?, ?it/s]
8
  25%|██▌ | 2/8 [00:00<00:01, 4.59it/s]
9
  38%|███▊ | 3/8 [00:00<00:01, 3.67it/s]
10
  50%|█████ | 4/8 [00:01<00:01, 3.28it/s]
11
  62%|██████▎ | 5/8 [00:01<00:00, 3.12it/s]
12
  75%|███████▌ | 6/8 [00:01<00:00, 3.04it/s]
13
  88%|████████▊ | 7/8 [00:02<00:00, 2.96it/s]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
14
  0%| | 0/8 [00:00<?, ?it/s]
15
  25%|██▌ | 2/8 [00:00<00:01, 4.55it/s]
16
  38%|███▊ | 3/8 [00:00<00:01, 3.65it/s]
17
  50%|█████ | 4/8 [00:01<00:01, 3.26it/s]
18
  62%|██████▎ | 5/8 [00:01<00:00, 3.11it/s]
19
  75%|███████▌ | 6/8 [00:01<00:00, 3.02it/s]
20
  88%|████████▊ | 7/8 [00:02<00:00, 2.94it/s]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
21
  0%| | 0/8 [00:00<?, ?it/s]
22
  25%|██▌ | 2/8 [00:00<00:01, 4.59it/s]
23
  38%|███▊ | 3/8 [00:00<00:01, 3.66it/s]
24
  50%|█████ | 4/8 [00:01<00:01, 3.27it/s]
25
  62%|██████▎ | 5/8 [00:01<00:00, 3.11it/s]
26
  75%|███████▌ | 6/8 [00:01<00:00, 3.02it/s]
27
  88%|████████▊ | 7/8 [00:02<00:00, 2.94it/s]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ No platform detected. Using base SRTPlatform with defaults.
2
+ [04-20 02:49:09] Port 30005 was unavailable, using port 30042 instead
3
+ [04-20 02:49:09] server_args: {"model_path": "/root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773", "model_id": null, "backend": "sglang", "attention_backend": null, "attention_backend_config": {}, "cache_dit_config": null, "nccl_port": null, "trust_remote_code": false, "revision": null, "num_gpus": 1, "tp_size": 1, "sp_degree": 1, "ulysses_degree": 1, "ring_degree": 1, "dp_size": 1, "dp_degree": 1, "enable_cfg_parallel": false, "hsdp_replicate_dim": 1, "hsdp_shard_dim": 1, "dist_timeout": 3600, "pipeline_class_name": null, "lora_path": null, "lora_nickname": "default", "lora_scale": 1.0, "lora_weight_name": null, "component_paths": {"transformer": "/data/bbuf/hunyuanvideo_fp8_20260420/sglang_transformer"}, "transformer_weights_path": null, "lora_target_modules": null, "dit_cpu_offload": false, "dit_layerwise_offload": false, "dit_offload_prefetch_size": 0.0, "text_encoder_cpu_offload": true, "image_encoder_cpu_offload": true, "vae_cpu_offload": true, "use_fsdp_inference": false, "pin_cpu_memory": true, "ltx2_two_stage_device_mode": null, "comfyui_mode": false, "enable_torch_compile": false, "warmup": false, "warmup_resolutions": null, "warmup_steps": 1, "disable_autocast": false, "master_port": 30042, "host": "127.0.0.1", "port": 30000, "webui": false, "webui_port": 12312, "scheduler_port": 5570, "strict_ports": false, "output_path": "outputs/", "input_save_path": "inputs/uploads", "prompt_file_path": null, "model_paths": {}, "model_loaded": {"transformer": true, "vae": true, "video_vae": true, "audio_vae": true, "video_dit": true, "audio_dit": true, "dual_tower_bridge": true}, "boundary_ratio": null, "base_gpu_id": 0, "disagg_role": "monolithic", "disagg_timeout": 600, "disagg_dispatch_policy": "round_robin", "disagg_mode": false, "disagg_server_addr": null, "encoder_urls": null, "denoiser_urls": null, "decoder_urls": null, "encoder_tp": null, "denoiser_tp": null, "denoiser_sp": null, "denoiser_ulysses": null, "denoiser_ring": null, "decoder_tp": null, "disagg_transfer_pool_size": 268435456, "disagg_p2p_hostname": "127.0.0.1", "disagg_ib_device": null, "pool_work_endpoint": null, "pool_result_endpoint": null, "log_level": "info", "uvicorn_access_log_exclude_prefixes": []}
4
+ [04-20 02:49:09] Starting offline throughput benchmark...
5
+ [04-20 02:49:09] Initializing engine...
6
+ [04-20 02:49:09] Local mode: True
7
+ [04-20 02:49:09] Starting server...
8
+ No platform detected. Using base SRTPlatform with defaults.
9
+ [04-20 02:49:17] Scheduler bind at endpoint: tcp://127.0.0.1:5570
10
+ [04-20 02:49:17] Initializing distributed environment with world_size=1, device=cuda:0, timeout=3600
11
+ [04-20 02:49:17] Setting distributed timeout to 3600 seconds
12
+ [04-20 02:49:18] No pipeline_class_name specified, using model_index.json
13
+ [04-20 02:49:18] Diffusers version: 0.32.0.dev0
14
+ [04-20 02:49:18] Using pipeline from model_index.json: HunyuanVideoPipeline
15
+ [04-20 02:49:18] Loading pipeline modules...
16
+ [04-20 02:49:18] Model already exists locally and is complete
17
+ [04-20 02:49:18] Model path: /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773
18
+ [04-20 02:49:18] Diffusers version: 0.32.0.dev0
19
+ [04-20 02:49:18] Loading pipeline modules from config: {'_class_name': 'HunyuanVideoPipeline', '_diffusers_version': '0.32.0.dev0', 'scheduler': ['diffusers', 'FlowMatchEulerDiscreteScheduler'], 'text_encoder': ['transformers', 'LlamaModel'], 'text_encoder_2': ['transformers', 'CLIPTextModel'], 'tokenizer': ['transformers', 'LlamaTokenizerFast'], 'tokenizer_2': ['transformers', 'CLIPTokenizer'], 'transformer': ['diffusers', 'HunyuanVideoTransformer3DModel'], 'vae': ['diffusers', 'AutoencoderKLHunyuanVideo']}
20
+ [04-20 02:49:18] Loading required components: ['text_encoder', 'text_encoder_2', 'tokenizer', 'tokenizer_2', 'vae', 'transformer', 'scheduler']
21
+
22
+ [04-20 02:49:18] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
23
+ [04-20 02:49:20] [RunAI Streamer] Overall time to stream 14.0 GiB of all files to cpu: 1.94s, 7.2 GiB/s
24
+ [04-20 02:49:37] Loaded text_encoder: FSDPLlamaModel (sgl-diffusion version). model size: 13.98 GB, consumed GPU mem: 0.81 GB, avail GPU mem: 73.04 GB
25
+
26
+ [04-20 02:49:37] Using Torch SDPA backend
27
+ [04-20 02:49:37] [RunAI Streamer] Overall time to stream 234.7 MiB of all files to cpu: 0.03s, 6.9 GiB/s
28
+ [04-20 02:49:38] Loaded text_encoder_2: FSDPCLIPTextModel (sgl-diffusion version). model size: 0.23 GB, consumed GPU mem: 0.76 GB, avail GPU mem: 72.27 GB
29
+
30
+ [04-20 02:49:39] Loaded tokenizer: LlamaTokenizer (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
31
+
32
+ [04-20 02:49:39] Loaded tokenizer_2: CLIPTokenizer (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
33
+ [04-20 02:49:39] Loading vae from /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773/vae. avail mem: 72.27 GB
34
+ [04-20 02:49:39] VAE unexpected keys: ['encoder.conv_in.conv.bias', 'encoder.conv_in.conv.weight', 'encoder.conv_norm_out.bias', 'encoder.conv_norm_out.weight', 'encoder.conv_out.conv.bias', 'encoder.conv_out.conv.weight', 'encoder.down_blocks.0.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.0.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.0.resnets.0.conv1.conv.bias', 'encoder.down_blocks.0.resnets.0.conv1.conv.weight', 'encoder.down_blocks.0.resnets.0.conv2.conv.bias', 'encoder.down_blocks.0.resnets.0.conv2.conv.weight', 'encoder.down_blocks.0.resnets.0.norm1.bias', 'encoder.down_blocks.0.resnets.0.norm1.weight', 'encoder.down_blocks.0.resnets.0.norm2.bias', 'encoder.down_blocks.0.resnets.0.norm2.weight', 'encoder.down_blocks.0.resnets.1.conv1.conv.bias', 'encoder.down_blocks.0.resnets.1.conv1.conv.weight', 'encoder.down_blocks.0.resnets.1.conv2.conv.bias', 'encoder.down_blocks.0.resnets.1.conv2.conv.weight', 'encoder.down_blocks.0.resnets.1.norm1.bias', 'encoder.down_blocks.0.resnets.1.norm1.weight', 'encoder.down_blocks.0.resnets.1.norm2.bias', 'encoder.down_blocks.0.resnets.1.norm2.weight', 'encoder.down_blocks.1.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.1.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.1.resnets.0.conv1.conv.bias', 'encoder.down_blocks.1.resnets.0.conv1.conv.weight', 'encoder.down_blocks.1.resnets.0.conv2.conv.bias', 'encoder.down_blocks.1.resnets.0.conv2.conv.weight', 'encoder.down_blocks.1.resnets.0.conv_shortcut.conv.bias', 'encoder.down_blocks.1.resnets.0.conv_shortcut.conv.weight', 'encoder.down_blocks.1.resnets.0.norm1.bias', 'encoder.down_blocks.1.resnets.0.norm1.weight', 'encoder.down_blocks.1.resnets.0.norm2.bias', 'encoder.down_blocks.1.resnets.0.norm2.weight', 'encoder.down_blocks.1.resnets.1.conv1.conv.bias', 'encoder.down_blocks.1.resnets.1.conv1.conv.weight', 'encoder.down_blocks.1.resnets.1.conv2.conv.bias', 'encoder.down_blocks.1.resnets.1.conv2.conv.weight', 'encoder.down_blocks.1.resnets.1.norm1.bias', 'encoder.down_blocks.1.resnets.1.norm1.weight', 'encoder.down_blocks.1.resnets.1.norm2.bias', 'encoder.down_blocks.1.resnets.1.norm2.weight', 'encoder.down_blocks.2.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.2.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.2.resnets.0.conv1.conv.bias', 'encoder.down_blocks.2.resnets.0.conv1.conv.weight', 'encoder.down_blocks.2.resnets.0.conv2.conv.bias', 'encoder.down_blocks.2.resnets.0.conv2.conv.weight', 'encoder.down_blocks.2.resnets.0.conv_shortcut.conv.bias', 'encoder.down_blocks.2.resnets.0.conv_shortcut.conv.weight', 'encoder.down_blocks.2.resnets.0.norm1.bias', 'encoder.down_blocks.2.resnets.0.norm1.weight', 'encoder.down_blocks.2.resnets.0.norm2.bias', 'encoder.down_blocks.2.resnets.0.norm2.weight', 'encoder.down_blocks.2.resnets.1.conv1.conv.bias', 'encoder.down_blocks.2.resnets.1.conv1.conv.weight', 'encoder.down_blocks.2.resnets.1.conv2.conv.bias', 'encoder.down_blocks.2.resnets.1.conv2.conv.weight', 'encoder.down_blocks.2.resnets.1.norm1.bias', 'encoder.down_blocks.2.resnets.1.norm1.weight', 'encoder.down_blocks.2.resnets.1.norm2.bias', 'encoder.down_blocks.2.resnets.1.norm2.weight', 'encoder.down_blocks.3.resnets.0.conv1.conv.bias', 'encoder.down_blocks.3.resnets.0.conv1.conv.weight', 'encoder.down_blocks.3.resnets.0.conv2.conv.bias', 'encoder.down_blocks.3.resnets.0.conv2.conv.weight', 'encoder.down_blocks.3.resnets.0.norm1.bias', 'encoder.down_blocks.3.resnets.0.norm1.weight', 'encoder.down_blocks.3.resnets.0.norm2.bias', 'encoder.down_blocks.3.resnets.0.norm2.weight', 'encoder.down_blocks.3.resnets.1.conv1.conv.bias', 'encoder.down_blocks.3.resnets.1.conv1.conv.weight', 'encoder.down_blocks.3.resnets.1.conv2.conv.bias', 'encoder.down_blocks.3.resnets.1.conv2.conv.weight', 'encoder.down_blocks.3.resnets.1.norm1.bias', 'encoder.down_blocks.3.resnets.1.norm1.weight', 'encoder.down_blocks.3.resnets.1.norm2.bias', 'encoder.down_blocks.3.resnets.1.norm2.weight', 'encoder.mid_block.attentions.0.group_norm.bias', 'encoder.mid_block.attentions.0.group_norm.weight', 'encoder.mid_block.attentions.0.to_k.bias', 'encoder.mid_block.attentions.0.to_k.weight', 'encoder.mid_block.attentions.0.to_out.0.bias', 'encoder.mid_block.attentions.0.to_out.0.weight', 'encoder.mid_block.attentions.0.to_q.bias', 'encoder.mid_block.attentions.0.to_q.weight', 'encoder.mid_block.attentions.0.to_v.bias', 'encoder.mid_block.attentions.0.to_v.weight', 'encoder.mid_block.resnets.0.conv1.conv.bias', 'encoder.mid_block.resnets.0.conv1.conv.weight', 'encoder.mid_block.resnets.0.conv2.conv.bias', 'encoder.mid_block.resnets.0.conv2.conv.weight', 'encoder.mid_block.resnets.0.norm1.bias', 'encoder.mid_block.resnets.0.norm1.weight', 'encoder.mid_block.resnets.0.norm2.bias', 'encoder.mid_block.resnets.0.norm2.weight', 'encoder.mid_block.resnets.1.conv1.conv.bias', 'encoder.mid_block.resnets.1.conv1.conv.weight', 'encoder.mid_block.resnets.1.conv2.conv.bias', 'encoder.mid_block.resnets.1.conv2.conv.weight', 'encoder.mid_block.resnets.1.norm1.bias', 'encoder.mid_block.resnets.1.norm1.weight', 'encoder.mid_block.resnets.1.norm2.bias', 'encoder.mid_block.resnets.1.norm2.weight', 'quant_conv.bias', 'quant_conv.weight']
35
+ [04-20 02:49:39] Loaded vae: AutoencoderKLHunyuanVideo (sgl-diffusion version). model size: 0.27 GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
36
+ [04-20 02:49:39] Loading transformer from /data/bbuf/hunyuanvideo_fp8_20260420/sglang_transformer. avail mem: 72.27 GB
37
+ [04-20 02:49:39] Detected ModelOpt FP8 checkpoint. The format is experimental and subject to change.
38
+ [04-20 02:49:39] Loading HunyuanVideoTransformer3DModel from 3 safetensors file(s) , param_dtype: None
39
+ [04-20 02:49:39] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
40
+ [04-20 02:49:39] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
41
+ [04-20 02:49:40] [RunAI Streamer] Overall time to stream 15.4 GiB of all files to cpu: 1.3s, 11.9 GiB/s
42
+ [04-20 02:49:43] Loaded model with 12.82B parameters
43
+ [04-20 02:49:43] Loaded transformer: HunyuanVideoTransformer3DModel (sgl-diffusion version). model size: 15.45 GB, consumed GPU mem: 15.57 GB, avail GPU mem: 56.71 GB
44
+
45
+ [04-20 02:49:43] Loaded scheduler: FlowMatchEulerDiscreteScheduler (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 56.71 GB
46
+
47
+ [04-20 02:49:43] Creating pipeline stages...
48
+ [04-20 02:49:43] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
49
+ [04-20 02:49:43] Pipeline instantiated
50
+ [04-20 02:49:43] Worker 0: Initialized device, model, and distributed environment.
51
+ [04-20 02:49:43] Worker 0: Scheduler loop started.
52
+ [04-20 02:49:43] Engine initialized successfully
53
+ [04-20 02:49:43] Loading random dataset...
54
+ [04-20 02:49:43] Running warmup batch...
55
+ [04-20 02:49:43] Adjusting number of frames from 17 to 17 based on model
56
+ [04-20 02:49:43] Processing prompt 1/1: <redacted, len=49>
57
+ [04-20 02:49:43] Sampling params:
58
+ width: 960
59
+ height: 544
60
+ num_frames: 17
61
+ fps: 24
62
+ prompt: <redacted, len=49>
63
+ neg_prompt: <redacted, len=392>
64
+ seed: 0
65
+ infer_steps: 8
66
+ num_outputs_per_prompt: 1
67
+ guidance_scale: 1.0
68
+ embedded_guidance_scale: 6
69
+ n_tokens: None
70
+ flow_shift: 7
71
+ image_path: None
72
+ save_output: True
73
+ output_file_path: outputs/Random_prompt_0_for_benchmarking_diffusion_models_20260420-024943_b99ddeb1.mp4
74
+
75
+ [04-20 02:49:43] Running pipeline stages: ['InputValidationStage', 'prompt_encoding_stage_primary', 'TimestepPreparationStage', 'LatentPreparationStage', 'DenoisingStage', 'DecodingStage']
76
+ [04-20 02:49:43] [InputValidationStage] started...
77
+ [04-20 02:49:43] [InputValidationStage] finished in 0.0000 seconds
78
+ [04-20 02:49:43] [TextEncodingStage] started...
79
+ [04-20 02:49:44] [TextEncodingStage] finished in 1.0464 seconds
80
+ [04-20 02:49:44] [TimestepPreparationStage] started...
81
+ [04-20 02:49:44] [TimestepPreparationStage] finished in 0.0004 seconds
82
+ [04-20 02:49:44] [LatentPreparationStage] started...
83
+ [04-20 02:49:44] [LatentPreparationStage] finished in 0.0008 seconds
84
+ [04-20 02:49:44] [DenoisingStage] started...
85
+
86
  0%| | 0/8 [00:00<?, ?it/s]
87
  12%|█▎ | 1/8 [00:01<00:12, 1.86s/it]
88
  38%|███▊ | 3/8 [00:02<00:03, 1.55it/s]
89
  50%|█████ | 4/8 [00:02<00:02, 1.84it/s]
90
  62%|██████▎ | 5/8 [00:02<00:01, 2.09it/s]
91
  75%|███████▌ | 6/8 [00:03<00:00, 2.28it/s]
92
  88%|████████▊ | 7/8 [00:03<00:00, 2.42it/s]
93
+ [04-20 02:49:48] [DenoisingStage] average time per step: 0.5073 seconds
94
+ [04-20 02:49:48] [DenoisingStage] finished in 4.0600 seconds
95
+ [04-20 02:49:48] [DecodingStage] started...
96
+ [04-20 02:49:54] [DecodingStage] finished in 5.3690 seconds
97
+ [04-20 02:49:54] Output saved to outputs/Random_prompt_0_for_benchmarking_diffusion_models_20260420-024943_b99ddeb1.mp4
98
+ [04-20 02:49:54] Pixel data generated successfully in 10.74 seconds
99
+ [04-20 02:49:54] Completed batch processing. Generated 1 outputs in 10.74 seconds
100
+ [04-20 02:49:54] Running benchmark with 3 prompts...
101
+ [04-20 02:49:54] Adjusting number of frames from 17 to 17 based on model
102
+ [04-20 02:49:54] Processing prompt 1/1: <redacted, len=49>
103
+ [04-20 02:49:54] Sampling params:
104
+ width: 960
105
+ height: 544
106
+ num_frames: 17
107
+ fps: 24
108
+ prompt: <redacted, len=49>
109
+ neg_prompt: <redacted, len=392>
110
+ seed: 0
111
+ infer_steps: 8
112
+ num_outputs_per_prompt: 1
113
+ guidance_scale: 1.0
114
+ embedded_guidance_scale: 6
115
+ n_tokens: None
116
+ flow_shift: 7
117
+ image_path: None
118
+ save_output: True
119
+ output_file_path: outputs/Random_prompt_0_for_benchmarking_diffusion_models_20260420-024954_b99ddeb1.mp4
120
+
121
+ [04-20 02:49:54] Running pipeline stages: ['InputValidationStage', 'prompt_encoding_stage_primary', 'TimestepPreparationStage', 'LatentPreparationStage', 'DenoisingStage', 'DecodingStage']
122
+ [04-20 02:49:54] [InputValidationStage] started...
123
+ [04-20 02:49:54] [InputValidationStage] finished in 0.0001 seconds
124
+ [04-20 02:49:54] [TextEncodingStage] started...
125
+ [04-20 02:49:54] [TextEncodingStage] finished in 0.3712 seconds
126
+ [04-20 02:49:54] [TimestepPreparationStage] started...
127
+ [04-20 02:49:54] [TimestepPreparationStage] finished in 0.0004 seconds
128
+ [04-20 02:49:54] [LatentPreparationStage] started...
129
+ [04-20 02:49:54] [LatentPreparationStage] finished in 0.0001 seconds
130
+ [04-20 02:49:54] [DenoisingStage] started...
131
+
132
  0%| | 0/8 [00:00<?, ?it/s]
133
  25%|██▌ | 2/8 [00:00<00:01, 4.59it/s]
134
  38%|███▊ | 3/8 [00:00<00:01, 3.67it/s]
135
  50%|█████ | 4/8 [00:01<00:01, 3.28it/s]
136
  62%|██████▎ | 5/8 [00:01<00:00, 3.12it/s]
137
  75%|███████▌ | 6/8 [00:01<00:00, 3.04it/s]
138
  88%|████████▊ | 7/8 [00:02<00:00, 2.96it/s]
139
+ [04-20 02:49:57] [DenoisingStage] average time per step: 0.3184 seconds
140
+ [04-20 02:49:57] [DenoisingStage] finished in 2.5504 seconds
141
+ [04-20 02:49:57] [DecodingStage] started...
142
+ [04-20 02:50:02] [DecodingStage] finished in 5.2985 seconds
143
+ [04-20 02:50:02] Output saved to outputs/Random_prompt_0_for_benchmarking_diffusion_models_20260420-024954_b99ddeb1.mp4
144
+ [04-20 02:50:02] Pixel data generated successfully in 8.54 seconds
145
+ [04-20 02:50:02] Completed batch processing. Generated 1 outputs in 8.54 seconds
146
+ [04-20 02:50:02] Adjusting number of frames from 17 to 17 based on model
147
+ [04-20 02:50:02] Processing prompt 1/1: <redacted, len=49>
148
+ [04-20 02:50:02] Sampling params:
149
+ width: 960
150
+ height: 544
151
+ num_frames: 17
152
+ fps: 24
153
+ prompt: <redacted, len=49>
154
+ neg_prompt: <redacted, len=392>
155
+ seed: 0
156
+ infer_steps: 8
157
+ num_outputs_per_prompt: 1
158
+ guidance_scale: 1.0
159
+ embedded_guidance_scale: 6
160
+ n_tokens: None
161
+ flow_shift: 7
162
+ image_path: None
163
+ save_output: True
164
+ output_file_path: outputs/Random_prompt_1_for_benchmarking_diffusion_models_20260420-025002_8d91326e.mp4
165
+
166
+ [04-20 02:50:02] Running pipeline stages: ['InputValidationStage', 'prompt_encoding_stage_primary', 'TimestepPreparationStage', 'LatentPreparationStage', 'DenoisingStage', 'DecodingStage']
167
+ [04-20 02:50:02] [InputValidationStage] started...
168
+ [04-20 02:50:02] [InputValidationStage] finished in 0.0001 seconds
169
+ [04-20 02:50:02] [TextEncodingStage] started...
170
+ [04-20 02:50:03] [TextEncodingStage] finished in 0.3517 seconds
171
+ [04-20 02:50:03] [TimestepPreparationStage] started...
172
+ [04-20 02:50:03] [TimestepPreparationStage] finished in 0.0008 seconds
173
+ [04-20 02:50:03] [LatentPreparationStage] started...
174
+ [04-20 02:50:03] [LatentPreparationStage] finished in 0.0002 seconds
175
+ [04-20 02:50:03] [DenoisingStage] started...
176
+
177
  0%| | 0/8 [00:00<?, ?it/s]
178
  25%|██▌ | 2/8 [00:00<00:01, 4.55it/s]
179
  38%|███▊ | 3/8 [00:00<00:01, 3.65it/s]
180
  50%|█████ | 4/8 [00:01<00:01, 3.26it/s]
181
  62%|██████▎ | 5/8 [00:01<00:00, 3.11it/s]
182
  75%|███████▌ | 6/8 [00:01<00:00, 3.02it/s]
183
  88%|████████▊ | 7/8 [00:02<00:00, 2.94it/s]
184
+ [04-20 02:50:05] [DenoisingStage] average time per step: 0.3199 seconds
185
+ [04-20 02:50:05] [DenoisingStage] finished in 2.5623 seconds
186
+ [04-20 02:50:05] [DecodingStage] started...
187
+ [04-20 02:50:10] [DecodingStage] finished in 5.0055 seconds
188
+ [04-20 02:50:11] Output saved to outputs/Random_prompt_1_for_benchmarking_diffusion_models_20260420-025002_8d91326e.mp4
189
+ [04-20 02:50:11] Pixel data generated successfully in 8.18 seconds
190
+ [04-20 02:50:11] Completed batch processing. Generated 1 outputs in 8.18 seconds
191
+ [04-20 02:50:11] Adjusting number of frames from 17 to 17 based on model
192
+ [04-20 02:50:11] Processing prompt 1/1: <redacted, len=49>
193
+ [04-20 02:50:11] Sampling params:
194
+ width: 960
195
+ height: 544
196
+ num_frames: 17
197
+ fps: 24
198
+ prompt: <redacted, len=49>
199
+ neg_prompt: <redacted, len=392>
200
+ seed: 0
201
+ infer_steps: 8
202
+ num_outputs_per_prompt: 1
203
+ guidance_scale: 1.0
204
+ embedded_guidance_scale: 6
205
+ n_tokens: None
206
+ flow_shift: 7
207
+ image_path: None
208
+ save_output: True
209
+ output_file_path: outputs/Random_prompt_2_for_benchmarking_diffusion_models_20260420-025011_9a3de0d5.mp4
210
+
211
+ [04-20 02:50:11] Running pipeline stages: ['InputValidationStage', 'prompt_encoding_stage_primary', 'TimestepPreparationStage', 'LatentPreparationStage', 'DenoisingStage', 'DecodingStage']
212
+ [04-20 02:50:11] [InputValidationStage] started...
213
+ [04-20 02:50:11] [InputValidationStage] finished in 0.0001 seconds
214
+ [04-20 02:50:11] [TextEncodingStage] started...
215
+ [04-20 02:50:11] [TextEncodingStage] finished in 0.3426 seconds
216
+ [04-20 02:50:11] [TimestepPreparationStage] started...
217
+ [04-20 02:50:11] [TimestepPreparationStage] finished in 0.0004 seconds
218
+ [04-20 02:50:11] [LatentPreparationStage] started...
219
+ [04-20 02:50:11] [LatentPreparationStage] finished in 0.0001 seconds
220
+ [04-20 02:50:11] [DenoisingStage] started...
221
+
222
  0%| | 0/8 [00:00<?, ?it/s]
223
  25%|██▌ | 2/8 [00:00<00:01, 4.59it/s]
224
  38%|███▊ | 3/8 [00:00<00:01, 3.66it/s]
225
  50%|█████ | 4/8 [00:01<00:01, 3.27it/s]
226
  62%|██████▎ | 5/8 [00:01<00:00, 3.11it/s]
227
  75%|███████▌ | 6/8 [00:01<00:00, 3.02it/s]
228
  88%|████████▊ | 7/8 [00:02<00:00, 2.94it/s]
229
+ [04-20 02:50:13] [DenoisingStage] average time per step: 0.3202 seconds
230
+ [04-20 02:50:13] [DenoisingStage] finished in 2.5644 seconds
231
+ [04-20 02:50:13] [DecodingStage] started...
232
+ [04-20 02:50:19] [DecodingStage] finished in 5.0319 seconds
233
+ [04-20 02:50:19] Output saved to outputs/Random_prompt_2_for_benchmarking_diffusion_models_20260420-025011_9a3de0d5.mp4
234
+ [04-20 02:50:19] Pixel data generated successfully in 8.18 seconds
235
+ [04-20 02:50:19] Completed batch processing. Generated 1 outputs in 8.18 seconds
236
+
237
+ ==================================== Offline Throughput Benchmark Result =====================================
238
+ Model: /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773
239
+ Dataset: random
240
+ Resolution: 960x544x17
241
+ Num Inference Steps: 8
242
+ ---------------------------------------------------------------------------
243
+ Total Requests: 3
244
+ Successful Requests: 3
245
+ Failed Requests: 0
246
+ Total Duration (seconds): 24.90
247
+ ---------------------------------------------------------------------------
248
+ Frames Generated: 3
249
+ Megapixels Generated: 26.63
250
+ ---------------------------------------------------------------------------
251
+ Frame Throughput (frames/sec): 0.12
252
+ MP Throughput (MP/sec): 1.07
253
+ Requests Per Second: 0.12
254
+ Latency Per Request (sec): 8.30
255
+ Peak Memory (MB): 0.00
256
+ ==============================================================================================================
257
+ [04-20 02:50:19] Results saved to /data/bbuf/hunyuanvideo_fp8_20260420/benchmark/fp8_offline_throughput_transformer_path.jsonl
258
+ [04-20 02:50:19] Generator was garbage collected without being shut down. Attempting to shut down the local server and client.
259
+ [04-20 02:50:26] Worker 0: Shutdown complete.
validation/h100_20260420/logs/convert_sglang_fp8.log ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ No platform detected. Using base SRTPlatform with defaults.
2
+ {
3
+ "added_scale_tensors": 880,
4
+ "bf16_fallback_weights": 112,
5
+ "output_shards": 3,
6
+ "preserved_ignored_weights": 160,
7
+ "quantized_weights": 440
8
+ }
validation/h100_20260420/logs/modelopt_quantize_fp8.log ADDED
The diff for this file is too large to render. See raw diff
 
validation/h100_20260420/logs/profile_bf16.log ADDED
@@ -0,0 +1,90 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
0
  0%| | 0/8 [00:00<?, ?it/s]
1
  12%|█▎ | 1/8 [00:01<00:13, 1.86s/it]
2
  25%|██▌ | 2/8 [00:01<00:05, 1.20it/s]
3
  38%|███▊ | 3/8 [00:02<00:03, 1.55it/s]
4
  50%|█████ | 4/8 [00:02<00:02, 1.80it/s]
5
  62%|██████▎ | 5/8 [00:03<00:01, 1.98it/s]
6
  75%|███████▌ | 6/8 [00:03<00:00, 2.08it/s]
7
  88%|████████▊ | 7/8 [00:04<00:00, 2.19it/s]
 
 
 
 
 
 
 
 
 
 
 
1
+ No platform detected. Using base SRTPlatform with defaults.
2
+ [04-20 02:36:50] server_args: {"model_path": "/root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773", "model_id": null, "backend": "sglang", "attention_backend": null, "attention_backend_config": {}, "cache_dit_config": null, "nccl_port": null, "trust_remote_code": false, "revision": null, "num_gpus": 1, "tp_size": 1, "sp_degree": 1, "ulysses_degree": 1, "ring_degree": 1, "dp_size": 1, "dp_degree": 1, "enable_cfg_parallel": false, "hsdp_replicate_dim": 1, "hsdp_shard_dim": 1, "dist_timeout": 3600, "pipeline_class_name": null, "lora_path": null, "lora_nickname": "default", "lora_scale": 1.0, "lora_weight_name": null, "component_paths": {}, "transformer_weights_path": null, "lora_target_modules": null, "dit_cpu_offload": false, "dit_layerwise_offload": false, "dit_offload_prefetch_size": 0.0, "text_encoder_cpu_offload": true, "image_encoder_cpu_offload": true, "vae_cpu_offload": true, "use_fsdp_inference": false, "pin_cpu_memory": true, "ltx2_two_stage_device_mode": null, "comfyui_mode": false, "enable_torch_compile": false, "warmup": false, "warmup_resolutions": null, "warmup_steps": 1, "disable_autocast": false, "master_port": 30005, "host": "127.0.0.1", "port": 30000, "webui": false, "webui_port": 12312, "scheduler_port": 5637, "strict_ports": false, "output_path": "outputs/", "input_save_path": "inputs/uploads", "prompt_file_path": null, "model_paths": {}, "model_loaded": {"transformer": true, "vae": true, "video_vae": true, "audio_vae": true, "video_dit": true, "audio_dit": true, "dual_tower_bridge": true}, "boundary_ratio": null, "base_gpu_id": 0, "disagg_role": "monolithic", "disagg_timeout": 600, "disagg_dispatch_policy": "round_robin", "disagg_mode": false, "disagg_server_addr": null, "encoder_urls": null, "denoiser_urls": null, "decoder_urls": null, "encoder_tp": null, "denoiser_tp": null, "denoiser_sp": null, "denoiser_ulysses": null, "denoiser_ring": null, "decoder_tp": null, "disagg_transfer_pool_size": 268435456, "disagg_p2p_hostname": "127.0.0.1", "disagg_ib_device": null, "pool_work_endpoint": null, "pool_result_endpoint": null, "log_level": "info", "uvicorn_access_log_exclude_prefixes": []}
3
+ [04-20 02:36:50] Local mode: True
4
+ [04-20 02:36:50] Starting server...
5
+ No platform detected. Using base SRTPlatform with defaults.
6
+ [04-20 02:36:57] Scheduler bind at endpoint: tcp://127.0.0.1:5637
7
+ [04-20 02:36:57] Initializing distributed environment with world_size=1, device=cuda:0, timeout=3600
8
+ [04-20 02:36:57] Setting distributed timeout to 3600 seconds
9
+ [04-20 02:36:58] No pipeline_class_name specified, using model_index.json
10
+ [04-20 02:36:58] Diffusers version: 0.32.0.dev0
11
+ [04-20 02:36:58] Using pipeline from model_index.json: HunyuanVideoPipeline
12
+ [04-20 02:36:58] Loading pipeline modules...
13
+ [04-20 02:36:58] Model already exists locally and is complete
14
+ [04-20 02:36:58] Model path: /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773
15
+ [04-20 02:36:58] Diffusers version: 0.32.0.dev0
16
+ [04-20 02:36:58] Loading pipeline modules from config: {'_class_name': 'HunyuanVideoPipeline', '_diffusers_version': '0.32.0.dev0', 'scheduler': ['diffusers', 'FlowMatchEulerDiscreteScheduler'], 'text_encoder': ['transformers', 'LlamaModel'], 'text_encoder_2': ['transformers', 'CLIPTextModel'], 'tokenizer': ['transformers', 'LlamaTokenizerFast'], 'tokenizer_2': ['transformers', 'CLIPTokenizer'], 'transformer': ['diffusers', 'HunyuanVideoTransformer3DModel'], 'vae': ['diffusers', 'AutoencoderKLHunyuanVideo']}
17
+ [04-20 02:36:58] Loading required components: ['text_encoder', 'text_encoder_2', 'tokenizer', 'tokenizer_2', 'vae', 'transformer', 'scheduler']
18
+
19
+ [04-20 02:36:59] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
20
+ [04-20 02:37:01] [RunAI Streamer] Overall time to stream 14.0 GiB of all files to cpu: 2.24s, 6.3 GiB/s
21
+ [04-20 02:37:19] Loaded text_encoder: FSDPLlamaModel (sgl-diffusion version). model size: 13.98 GB, consumed GPU mem: 0.81 GB, avail GPU mem: 73.04 GB
22
+
23
+ [04-20 02:37:20] Using Torch SDPA backend
24
+ [04-20 02:37:20] [RunAI Streamer] Overall time to stream 234.7 MiB of all files to cpu: 0.03s, 8.2 GiB/s
25
+ [04-20 02:37:20] Loaded text_encoder_2: FSDPCLIPTextModel (sgl-diffusion version). model size: 0.23 GB, consumed GPU mem: 0.76 GB, avail GPU mem: 72.27 GB
26
+
27
+ [04-20 02:37:21] Loaded tokenizer: LlamaTokenizer (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
28
+
29
+ [04-20 02:37:21] Loaded tokenizer_2: CLIPTokenizer (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
30
+ [04-20 02:37:21] Loading vae from /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773/vae. avail mem: 72.27 GB
31
+ [04-20 02:37:21] VAE unexpected keys: ['encoder.conv_in.conv.bias', 'encoder.conv_in.conv.weight', 'encoder.conv_norm_out.bias', 'encoder.conv_norm_out.weight', 'encoder.conv_out.conv.bias', 'encoder.conv_out.conv.weight', 'encoder.down_blocks.0.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.0.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.0.resnets.0.conv1.conv.bias', 'encoder.down_blocks.0.resnets.0.conv1.conv.weight', 'encoder.down_blocks.0.resnets.0.conv2.conv.bias', 'encoder.down_blocks.0.resnets.0.conv2.conv.weight', 'encoder.down_blocks.0.resnets.0.norm1.bias', 'encoder.down_blocks.0.resnets.0.norm1.weight', 'encoder.down_blocks.0.resnets.0.norm2.bias', 'encoder.down_blocks.0.resnets.0.norm2.weight', 'encoder.down_blocks.0.resnets.1.conv1.conv.bias', 'encoder.down_blocks.0.resnets.1.conv1.conv.weight', 'encoder.down_blocks.0.resnets.1.conv2.conv.bias', 'encoder.down_blocks.0.resnets.1.conv2.conv.weight', 'encoder.down_blocks.0.resnets.1.norm1.bias', 'encoder.down_blocks.0.resnets.1.norm1.weight', 'encoder.down_blocks.0.resnets.1.norm2.bias', 'encoder.down_blocks.0.resnets.1.norm2.weight', 'encoder.down_blocks.1.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.1.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.1.resnets.0.conv1.conv.bias', 'encoder.down_blocks.1.resnets.0.conv1.conv.weight', 'encoder.down_blocks.1.resnets.0.conv2.conv.bias', 'encoder.down_blocks.1.resnets.0.conv2.conv.weight', 'encoder.down_blocks.1.resnets.0.conv_shortcut.conv.bias', 'encoder.down_blocks.1.resnets.0.conv_shortcut.conv.weight', 'encoder.down_blocks.1.resnets.0.norm1.bias', 'encoder.down_blocks.1.resnets.0.norm1.weight', 'encoder.down_blocks.1.resnets.0.norm2.bias', 'encoder.down_blocks.1.resnets.0.norm2.weight', 'encoder.down_blocks.1.resnets.1.conv1.conv.bias', 'encoder.down_blocks.1.resnets.1.conv1.conv.weight', 'encoder.down_blocks.1.resnets.1.conv2.conv.bias', 'encoder.down_blocks.1.resnets.1.conv2.conv.weight', 'encoder.down_blocks.1.resnets.1.norm1.bias', 'encoder.down_blocks.1.resnets.1.norm1.weight', 'encoder.down_blocks.1.resnets.1.norm2.bias', 'encoder.down_blocks.1.resnets.1.norm2.weight', 'encoder.down_blocks.2.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.2.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.2.resnets.0.conv1.conv.bias', 'encoder.down_blocks.2.resnets.0.conv1.conv.weight', 'encoder.down_blocks.2.resnets.0.conv2.conv.bias', 'encoder.down_blocks.2.resnets.0.conv2.conv.weight', 'encoder.down_blocks.2.resnets.0.conv_shortcut.conv.bias', 'encoder.down_blocks.2.resnets.0.conv_shortcut.conv.weight', 'encoder.down_blocks.2.resnets.0.norm1.bias', 'encoder.down_blocks.2.resnets.0.norm1.weight', 'encoder.down_blocks.2.resnets.0.norm2.bias', 'encoder.down_blocks.2.resnets.0.norm2.weight', 'encoder.down_blocks.2.resnets.1.conv1.conv.bias', 'encoder.down_blocks.2.resnets.1.conv1.conv.weight', 'encoder.down_blocks.2.resnets.1.conv2.conv.bias', 'encoder.down_blocks.2.resnets.1.conv2.conv.weight', 'encoder.down_blocks.2.resnets.1.norm1.bias', 'encoder.down_blocks.2.resnets.1.norm1.weight', 'encoder.down_blocks.2.resnets.1.norm2.bias', 'encoder.down_blocks.2.resnets.1.norm2.weight', 'encoder.down_blocks.3.resnets.0.conv1.conv.bias', 'encoder.down_blocks.3.resnets.0.conv1.conv.weight', 'encoder.down_blocks.3.resnets.0.conv2.conv.bias', 'encoder.down_blocks.3.resnets.0.conv2.conv.weight', 'encoder.down_blocks.3.resnets.0.norm1.bias', 'encoder.down_blocks.3.resnets.0.norm1.weight', 'encoder.down_blocks.3.resnets.0.norm2.bias', 'encoder.down_blocks.3.resnets.0.norm2.weight', 'encoder.down_blocks.3.resnets.1.conv1.conv.bias', 'encoder.down_blocks.3.resnets.1.conv1.conv.weight', 'encoder.down_blocks.3.resnets.1.conv2.conv.bias', 'encoder.down_blocks.3.resnets.1.conv2.conv.weight', 'encoder.down_blocks.3.resnets.1.norm1.bias', 'encoder.down_blocks.3.resnets.1.norm1.weight', 'encoder.down_blocks.3.resnets.1.norm2.bias', 'encoder.down_blocks.3.resnets.1.norm2.weight', 'encoder.mid_block.attentions.0.group_norm.bias', 'encoder.mid_block.attentions.0.group_norm.weight', 'encoder.mid_block.attentions.0.to_k.bias', 'encoder.mid_block.attentions.0.to_k.weight', 'encoder.mid_block.attentions.0.to_out.0.bias', 'encoder.mid_block.attentions.0.to_out.0.weight', 'encoder.mid_block.attentions.0.to_q.bias', 'encoder.mid_block.attentions.0.to_q.weight', 'encoder.mid_block.attentions.0.to_v.bias', 'encoder.mid_block.attentions.0.to_v.weight', 'encoder.mid_block.resnets.0.conv1.conv.bias', 'encoder.mid_block.resnets.0.conv1.conv.weight', 'encoder.mid_block.resnets.0.conv2.conv.bias', 'encoder.mid_block.resnets.0.conv2.conv.weight', 'encoder.mid_block.resnets.0.norm1.bias', 'encoder.mid_block.resnets.0.norm1.weight', 'encoder.mid_block.resnets.0.norm2.bias', 'encoder.mid_block.resnets.0.norm2.weight', 'encoder.mid_block.resnets.1.conv1.conv.bias', 'encoder.mid_block.resnets.1.conv1.conv.weight', 'encoder.mid_block.resnets.1.conv2.conv.bias', 'encoder.mid_block.resnets.1.conv2.conv.weight', 'encoder.mid_block.resnets.1.norm1.bias', 'encoder.mid_block.resnets.1.norm1.weight', 'encoder.mid_block.resnets.1.norm2.bias', 'encoder.mid_block.resnets.1.norm2.weight', 'quant_conv.bias', 'quant_conv.weight']
32
+ [04-20 02:37:21] Loaded vae: AutoencoderKLHunyuanVideo (sgl-diffusion version). model size: 0.27 GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
33
+ [04-20 02:37:21] Loading transformer from /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773/transformer. avail mem: 72.27 GB
34
+ [04-20 02:37:21] Loading HunyuanVideoTransformer3DModel from 6 safetensors file(s) , param_dtype: torch.bfloat16
35
+ [04-20 02:37:21] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
36
+ [04-20 02:37:21] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
37
+ [04-20 02:37:23] [RunAI Streamer] Overall time to stream 23.9 GiB of all files to cpu: 1.69s, 14.1 GiB/s
38
+ [04-20 02:37:28] Loaded model with 12.82B parameters
39
+ [04-20 02:37:28] Loaded transformer: HunyuanVideoTransformer3DModel (sgl-diffusion version). model size: 23.88 GB, consumed GPU mem: 24.19 GB, avail GPU mem: 48.08 GB
40
+
41
+ [04-20 02:37:28] Loaded scheduler: FlowMatchEulerDiscreteScheduler (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 48.08 GB
42
+
43
+ [04-20 02:37:28] Creating pipeline stages...
44
+ [04-20 02:37:28] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
45
+ [04-20 02:37:28] Pipeline instantiated
46
+ [04-20 02:37:28] Worker 0: Initialized device, model, and distributed environment.
47
+ [04-20 02:37:28] Worker 0: Scheduler loop started.
48
+ [04-20 02:37:28] Adjusting number of frames from 17 to 17 based on model
49
+ [04-20 02:37:28] Processing prompt 1/1: <redacted, len=119>
50
+ [04-20 02:37:28] Sampling params:
51
+ width: 960
52
+ height: 544
53
+ num_frames: 17
54
+ fps: 24
55
+ prompt: <redacted, len=119>
56
+ neg_prompt: <redacted, len=392>
57
+ seed: 0
58
+ infer_steps: 8
59
+ num_outputs_per_prompt: 1
60
+ guidance_scale: 1.0
61
+ embedded_guidance_scale: 6
62
+ n_tokens: None
63
+ flow_shift: 7
64
+ image_path: None
65
+ save_output: False
66
+ output_file_path: outputs/A_cinematic_shot_of_a_red_sports_car_driving_through_rain_at_night_reflections_on_wet_streets_smoo_20260420-023728_58260e89.mp4
67
+
68
+ [04-20 02:37:28] Running pipeline stages: ['InputValidationStage', 'prompt_encoding_stage_primary', 'TimestepPreparationStage', 'LatentPreparationStage', 'DenoisingStage', 'DecodingStage']
69
+ [04-20 02:37:28] Profiling request: 5b26da87-550f-4d35-8cb8-bf9aa9c535d0 for 5 steps...
70
+ [04-20 02:37:28] Starting Profiler...
71
+ [04-20 02:37:28] [InputValidationStage] started...
72
+ [04-20 02:37:28] [InputValidationStage] finished in 0.0004 seconds
73
+ [04-20 02:37:28] [TextEncodingStage] started...
74
+ [04-20 02:37:29] [TextEncodingStage] finished in 0.9494 seconds
75
+ [04-20 02:37:29] [TimestepPreparationStage] started...
76
+ [04-20 02:37:29] [TimestepPreparationStage] finished in 0.0003 seconds
77
+ [04-20 02:37:29] [LatentPreparationStage] started...
78
+ [04-20 02:37:29] [LatentPreparationStage] finished in 0.0008 seconds
79
+ [04-20 02:37:29] [DenoisingStage] started...
80
+
81
  0%| | 0/8 [00:00<?, ?it/s]
82
  12%|█▎ | 1/8 [00:01<00:13, 1.86s/it]
83
  25%|██▌ | 2/8 [00:01<00:05, 1.20it/s]
84
  38%|███▊ | 3/8 [00:02<00:03, 1.55it/s]
85
  50%|█████ | 4/8 [00:02<00:02, 1.80it/s]
86
  62%|██████▎ | 5/8 [00:03<00:01, 1.98it/s]
87
  75%|███████▌ | 6/8 [00:03<00:00, 2.08it/s]
88
  88%|████████▊ | 7/8 [00:04<00:00, 2.19it/s]
89
+ [04-20 02:37:45] Saved profiler traces to: /data/bbuf/hunyuanvideo_fp8_20260420/profiler/bf16/5b26da87-550f-4d35-8cb8-bf9aa9c535d0-5_steps-global-rank0.trace.json.gz
90
+
91
+ [04-20 02:37:45] [DenoisingStage] average time per step: 1.9787 seconds
92
+ [04-20 02:37:45] [DenoisingStage] finished in 15.8313 seconds
93
+ [04-20 02:37:45] [DecodingStage] started...
94
+ [04-20 02:37:50] [DecodingStage] finished in 5.4015 seconds
95
+ [04-20 02:37:51] Pixel data generated successfully in 22.59 seconds
96
+ [04-20 02:37:51] Completed batch processing. Generated 1 outputs in 22.59 seconds
97
+ [04-20 02:37:51] Generator was garbage collected without being shut down. Attempting to shut down the local server and client.
98
+ [04-20 02:37:58] Worker 0: Shutdown complete.
validation/h100_20260420/logs/profile_fp8.log ADDED
@@ -0,0 +1,92 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
0
  0%| | 0/8 [00:00<?, ?it/s]
1
  12%|█▎ | 1/8 [00:01<00:12, 1.80s/it]
2
  25%|██▌ | 2/8 [00:01<00:04, 1.22it/s]
3
  38%|███▊ | 3/8 [00:02<00:03, 1.66it/s]
4
  50%|█████ | 4/8 [00:02<00:02, 1.98it/s]
5
  62%|██████▎ | 5/8 [00:02<00:01, 2.23it/s]
6
  75%|███████▌ | 6/8 [00:03<00:00, 2.40it/s]
7
  88%|████████▊ | 7/8 [00:03<00:00, 2.52it/s]
 
 
 
 
 
 
 
 
 
 
 
1
+ No platform detected. Using base SRTPlatform with defaults.
2
+ [04-20 02:38:13] server_args: {"model_path": "/data/bbuf/hunyuanvideo_fp8_20260420/sglang_fp8_pipeline", "model_id": null, "backend": "sglang", "attention_backend": null, "attention_backend_config": {}, "cache_dit_config": null, "nccl_port": null, "trust_remote_code": false, "revision": null, "num_gpus": 1, "tp_size": 1, "sp_degree": 1, "ulysses_degree": 1, "ring_degree": 1, "dp_size": 1, "dp_degree": 1, "enable_cfg_parallel": false, "hsdp_replicate_dim": 1, "hsdp_shard_dim": 1, "dist_timeout": 3600, "pipeline_class_name": null, "lora_path": null, "lora_nickname": "default", "lora_scale": 1.0, "lora_weight_name": null, "component_paths": {}, "transformer_weights_path": null, "lora_target_modules": null, "dit_cpu_offload": false, "dit_layerwise_offload": false, "dit_offload_prefetch_size": 0.0, "text_encoder_cpu_offload": true, "image_encoder_cpu_offload": true, "vae_cpu_offload": true, "use_fsdp_inference": false, "pin_cpu_memory": true, "ltx2_two_stage_device_mode": null, "comfyui_mode": false, "enable_torch_compile": false, "warmup": false, "warmup_resolutions": null, "warmup_steps": 1, "disable_autocast": false, "master_port": 30005, "host": "127.0.0.1", "port": 30000, "webui": false, "webui_port": 12312, "scheduler_port": 5618, "strict_ports": false, "output_path": "outputs/", "input_save_path": "inputs/uploads", "prompt_file_path": null, "model_paths": {}, "model_loaded": {"transformer": true, "vae": true, "video_vae": true, "audio_vae": true, "video_dit": true, "audio_dit": true, "dual_tower_bridge": true}, "boundary_ratio": null, "base_gpu_id": 0, "disagg_role": "monolithic", "disagg_timeout": 600, "disagg_dispatch_policy": "round_robin", "disagg_mode": false, "disagg_server_addr": null, "encoder_urls": null, "denoiser_urls": null, "decoder_urls": null, "encoder_tp": null, "denoiser_tp": null, "denoiser_sp": null, "denoiser_ulysses": null, "denoiser_ring": null, "decoder_tp": null, "disagg_transfer_pool_size": 268435456, "disagg_p2p_hostname": "127.0.0.1", "disagg_ib_device": null, "pool_work_endpoint": null, "pool_result_endpoint": null, "log_level": "info", "uvicorn_access_log_exclude_prefixes": []}
3
+ [04-20 02:38:13] Local mode: True
4
+ [04-20 02:38:13] Starting server...
5
+ No platform detected. Using base SRTPlatform with defaults.
6
+ [04-20 02:38:20] Scheduler bind at endpoint: tcp://127.0.0.1:5618
7
+ [04-20 02:38:20] Initializing distributed environment with world_size=1, device=cuda:0, timeout=3600
8
+ [04-20 02:38:20] Setting distributed timeout to 3600 seconds
9
+ [04-20 02:38:21] No pipeline_class_name specified, using model_index.json
10
+ [04-20 02:38:21] Diffusers version: 0.32.0.dev0
11
+ [04-20 02:38:21] Diffusers version: 0.32.0.dev0
12
+ [04-20 02:38:21] Using pipeline from model_index.json: HunyuanVideoPipeline
13
+ [04-20 02:38:21] Loading pipeline modules...
14
+ [04-20 02:38:21] Model already exists locally and is complete
15
+ [04-20 02:38:21] Model path: /data/bbuf/hunyuanvideo_fp8_20260420/sglang_fp8_pipeline
16
+ [04-20 02:38:21] Diffusers version: 0.32.0.dev0
17
+ [04-20 02:38:21] Loading pipeline modules from config: {'_class_name': 'HunyuanVideoPipeline', '_diffusers_version': '0.32.0.dev0', 'scheduler': ['diffusers', 'FlowMatchEulerDiscreteScheduler'], 'text_encoder': ['transformers', 'LlamaModel'], 'text_encoder_2': ['transformers', 'CLIPTextModel'], 'tokenizer': ['transformers', 'LlamaTokenizerFast'], 'tokenizer_2': ['transformers', 'CLIPTokenizer'], 'transformer': ['diffusers', 'HunyuanVideoTransformer3DModel'], 'vae': ['diffusers', 'AutoencoderKLHunyuanVideo']}
18
+ [04-20 02:38:21] Loading required components: ['text_encoder', 'text_encoder_2', 'tokenizer', 'tokenizer_2', 'vae', 'transformer', 'scheduler']
19
+
20
+ [04-20 02:38:21] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
21
+ [04-20 02:38:23] [RunAI Streamer] Overall time to stream 14.0 GiB of all files to cpu: 2.05s, 6.8 GiB/s
22
+ [04-20 02:38:38] Loaded text_encoder: FSDPLlamaModel (sgl-diffusion version). model size: 13.98 GB, consumed GPU mem: 0.81 GB, avail GPU mem: 73.04 GB
23
+
24
+ [04-20 02:38:39] Using Torch SDPA backend
25
+ [04-20 02:38:39] [RunAI Streamer] Overall time to stream 234.7 MiB of all files to cpu: 0.03s, 7.7 GiB/s
26
+ [04-20 02:38:39] Loaded text_encoder_2: FSDPCLIPTextModel (sgl-diffusion version). model size: 0.23 GB, consumed GPU mem: 0.76 GB, avail GPU mem: 72.27 GB
27
+
28
+ [04-20 02:38:40] Loaded tokenizer: LlamaTokenizer (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
29
+
30
+ [04-20 02:38:40] Loaded tokenizer_2: CLIPTokenizer (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
31
+ [04-20 02:38:40] Loading vae from /data/bbuf/hunyuanvideo_fp8_20260420/sglang_fp8_pipeline/vae. avail mem: 72.27 GB
32
+ [04-20 02:38:40] VAE unexpected keys: ['encoder.conv_in.conv.bias', 'encoder.conv_in.conv.weight', 'encoder.conv_norm_out.bias', 'encoder.conv_norm_out.weight', 'encoder.conv_out.conv.bias', 'encoder.conv_out.conv.weight', 'encoder.down_blocks.0.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.0.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.0.resnets.0.conv1.conv.bias', 'encoder.down_blocks.0.resnets.0.conv1.conv.weight', 'encoder.down_blocks.0.resnets.0.conv2.conv.bias', 'encoder.down_blocks.0.resnets.0.conv2.conv.weight', 'encoder.down_blocks.0.resnets.0.norm1.bias', 'encoder.down_blocks.0.resnets.0.norm1.weight', 'encoder.down_blocks.0.resnets.0.norm2.bias', 'encoder.down_blocks.0.resnets.0.norm2.weight', 'encoder.down_blocks.0.resnets.1.conv1.conv.bias', 'encoder.down_blocks.0.resnets.1.conv1.conv.weight', 'encoder.down_blocks.0.resnets.1.conv2.conv.bias', 'encoder.down_blocks.0.resnets.1.conv2.conv.weight', 'encoder.down_blocks.0.resnets.1.norm1.bias', 'encoder.down_blocks.0.resnets.1.norm1.weight', 'encoder.down_blocks.0.resnets.1.norm2.bias', 'encoder.down_blocks.0.resnets.1.norm2.weight', 'encoder.down_blocks.1.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.1.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.1.resnets.0.conv1.conv.bias', 'encoder.down_blocks.1.resnets.0.conv1.conv.weight', 'encoder.down_blocks.1.resnets.0.conv2.conv.bias', 'encoder.down_blocks.1.resnets.0.conv2.conv.weight', 'encoder.down_blocks.1.resnets.0.conv_shortcut.conv.bias', 'encoder.down_blocks.1.resnets.0.conv_shortcut.conv.weight', 'encoder.down_blocks.1.resnets.0.norm1.bias', 'encoder.down_blocks.1.resnets.0.norm1.weight', 'encoder.down_blocks.1.resnets.0.norm2.bias', 'encoder.down_blocks.1.resnets.0.norm2.weight', 'encoder.down_blocks.1.resnets.1.conv1.conv.bias', 'encoder.down_blocks.1.resnets.1.conv1.conv.weight', 'encoder.down_blocks.1.resnets.1.conv2.conv.bias', 'encoder.down_blocks.1.resnets.1.conv2.conv.weight', 'encoder.down_blocks.1.resnets.1.norm1.bias', 'encoder.down_blocks.1.resnets.1.norm1.weight', 'encoder.down_blocks.1.resnets.1.norm2.bias', 'encoder.down_blocks.1.resnets.1.norm2.weight', 'encoder.down_blocks.2.downsamplers.0.conv.conv.bias', 'encoder.down_blocks.2.downsamplers.0.conv.conv.weight', 'encoder.down_blocks.2.resnets.0.conv1.conv.bias', 'encoder.down_blocks.2.resnets.0.conv1.conv.weight', 'encoder.down_blocks.2.resnets.0.conv2.conv.bias', 'encoder.down_blocks.2.resnets.0.conv2.conv.weight', 'encoder.down_blocks.2.resnets.0.conv_shortcut.conv.bias', 'encoder.down_blocks.2.resnets.0.conv_shortcut.conv.weight', 'encoder.down_blocks.2.resnets.0.norm1.bias', 'encoder.down_blocks.2.resnets.0.norm1.weight', 'encoder.down_blocks.2.resnets.0.norm2.bias', 'encoder.down_blocks.2.resnets.0.norm2.weight', 'encoder.down_blocks.2.resnets.1.conv1.conv.bias', 'encoder.down_blocks.2.resnets.1.conv1.conv.weight', 'encoder.down_blocks.2.resnets.1.conv2.conv.bias', 'encoder.down_blocks.2.resnets.1.conv2.conv.weight', 'encoder.down_blocks.2.resnets.1.norm1.bias', 'encoder.down_blocks.2.resnets.1.norm1.weight', 'encoder.down_blocks.2.resnets.1.norm2.bias', 'encoder.down_blocks.2.resnets.1.norm2.weight', 'encoder.down_blocks.3.resnets.0.conv1.conv.bias', 'encoder.down_blocks.3.resnets.0.conv1.conv.weight', 'encoder.down_blocks.3.resnets.0.conv2.conv.bias', 'encoder.down_blocks.3.resnets.0.conv2.conv.weight', 'encoder.down_blocks.3.resnets.0.norm1.bias', 'encoder.down_blocks.3.resnets.0.norm1.weight', 'encoder.down_blocks.3.resnets.0.norm2.bias', 'encoder.down_blocks.3.resnets.0.norm2.weight', 'encoder.down_blocks.3.resnets.1.conv1.conv.bias', 'encoder.down_blocks.3.resnets.1.conv1.conv.weight', 'encoder.down_blocks.3.resnets.1.conv2.conv.bias', 'encoder.down_blocks.3.resnets.1.conv2.conv.weight', 'encoder.down_blocks.3.resnets.1.norm1.bias', 'encoder.down_blocks.3.resnets.1.norm1.weight', 'encoder.down_blocks.3.resnets.1.norm2.bias', 'encoder.down_blocks.3.resnets.1.norm2.weight', 'encoder.mid_block.attentions.0.group_norm.bias', 'encoder.mid_block.attentions.0.group_norm.weight', 'encoder.mid_block.attentions.0.to_k.bias', 'encoder.mid_block.attentions.0.to_k.weight', 'encoder.mid_block.attentions.0.to_out.0.bias', 'encoder.mid_block.attentions.0.to_out.0.weight', 'encoder.mid_block.attentions.0.to_q.bias', 'encoder.mid_block.attentions.0.to_q.weight', 'encoder.mid_block.attentions.0.to_v.bias', 'encoder.mid_block.attentions.0.to_v.weight', 'encoder.mid_block.resnets.0.conv1.conv.bias', 'encoder.mid_block.resnets.0.conv1.conv.weight', 'encoder.mid_block.resnets.0.conv2.conv.bias', 'encoder.mid_block.resnets.0.conv2.conv.weight', 'encoder.mid_block.resnets.0.norm1.bias', 'encoder.mid_block.resnets.0.norm1.weight', 'encoder.mid_block.resnets.0.norm2.bias', 'encoder.mid_block.resnets.0.norm2.weight', 'encoder.mid_block.resnets.1.conv1.conv.bias', 'encoder.mid_block.resnets.1.conv1.conv.weight', 'encoder.mid_block.resnets.1.conv2.conv.bias', 'encoder.mid_block.resnets.1.conv2.conv.weight', 'encoder.mid_block.resnets.1.norm1.bias', 'encoder.mid_block.resnets.1.norm1.weight', 'encoder.mid_block.resnets.1.norm2.bias', 'encoder.mid_block.resnets.1.norm2.weight', 'quant_conv.bias', 'quant_conv.weight']
33
+ [04-20 02:38:40] Loaded vae: AutoencoderKLHunyuanVideo (sgl-diffusion version). model size: 0.27 GB, consumed GPU mem: 0.00 GB, avail GPU mem: 72.27 GB
34
+ [04-20 02:38:40] Loading transformer from /data/bbuf/hunyuanvideo_fp8_20260420/sglang_fp8_pipeline/transformer. avail mem: 72.27 GB
35
+ [04-20 02:38:40] Detected ModelOpt FP8 checkpoint. The format is experimental and subject to change.
36
+ [04-20 02:38:40] Loading HunyuanVideoTransformer3DModel from 3 safetensors file(s) , param_dtype: None
37
+ [04-20 02:38:40] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
38
+ [04-20 02:38:40] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
39
+ [04-20 02:38:42] [RunAI Streamer] Overall time to stream 15.4 GiB of all files to cpu: 1.24s, 12.4 GiB/s
40
+ [04-20 02:38:45] Loaded model with 12.82B parameters
41
+ [04-20 02:38:45] Loaded transformer: HunyuanVideoTransformer3DModel (sgl-diffusion version). model size: 15.45 GB, consumed GPU mem: 15.57 GB, avail GPU mem: 56.71 GB
42
+
43
+ [04-20 02:38:45] Loaded scheduler: FlowMatchEulerDiscreteScheduler (sgl-diffusion version). model size: NA GB, consumed GPU mem: 0.00 GB, avail GPU mem: 56.71 GB
44
+
45
+ [04-20 02:38:45] Creating pipeline stages...
46
+ [04-20 02:38:45] Using FlashAttention (FA3 for hopper, FA4 for blackwell) backend
47
+ [04-20 02:38:45] Pipeline instantiated
48
+ [04-20 02:38:45] Worker 0: Initialized device, model, and distributed environment.
49
+ [04-20 02:38:45] Worker 0: Scheduler loop started.
50
+ [04-20 02:38:45] Adjusting number of frames from 17 to 17 based on model
51
+ [04-20 02:38:45] Processing prompt 1/1: <redacted, len=119>
52
+ [04-20 02:38:45] Sampling params:
53
+ width: 960
54
+ height: 544
55
+ num_frames: 17
56
+ fps: 24
57
+ prompt: <redacted, len=119>
58
+ neg_prompt: <redacted, len=392>
59
+ seed: 0
60
+ infer_steps: 8
61
+ num_outputs_per_prompt: 1
62
+ guidance_scale: 1.0
63
+ embedded_guidance_scale: 6
64
+ n_tokens: None
65
+ flow_shift: 7
66
+ image_path: None
67
+ save_output: False
68
+ output_file_path: outputs/A_cinematic_shot_of_a_red_sports_car_driving_through_rain_at_night_reflections_on_wet_streets_smoo_20260420-023845_50e60ea3.mp4
69
+
70
+ [04-20 02:38:45] Running pipeline stages: ['InputValidationStage', 'prompt_encoding_stage_primary', 'TimestepPreparationStage', 'LatentPreparationStage', 'DenoisingStage', 'DecodingStage']
71
+ [04-20 02:38:45] Profiling request: 63e4807c-f120-4fdc-b5b0-c1ab96522a7e for 5 steps...
72
+ [04-20 02:38:45] Starting Profiler...
73
+ [04-20 02:38:45] [InputValidationStage] started...
74
+ [04-20 02:38:45] [InputValidationStage] finished in 0.0004 seconds
75
+ [04-20 02:38:45] [TextEncodingStage] started...
76
+ [04-20 02:38:46] [TextEncodingStage] finished in 0.9378 seconds
77
+ [04-20 02:38:46] [TimestepPreparationStage] started...
78
+ [04-20 02:38:46] [TimestepPreparationStage] finished in 0.0003 seconds
79
+ [04-20 02:38:46] [LatentPreparationStage] started...
80
+ [04-20 02:38:46] [LatentPreparationStage] finished in 0.0008 seconds
81
+ [04-20 02:38:46] [DenoisingStage] started...
82
+
83
  0%| | 0/8 [00:00<?, ?it/s]
84
  12%|█▎ | 1/8 [00:01<00:12, 1.80s/it]
85
  25%|██▌ | 2/8 [00:01<00:04, 1.22it/s]
86
  38%|███▊ | 3/8 [00:02<00:03, 1.66it/s]
87
  50%|█████ | 4/8 [00:02<00:02, 1.98it/s]
88
  62%|██████▎ | 5/8 [00:02<00:01, 2.23it/s]
89
  75%|███████▌ | 6/8 [00:03<00:00, 2.40it/s]
90
  88%|████████▊ | 7/8 [00:03<00:00, 2.52it/s]
91
+ [04-20 02:39:04] Saved profiler traces to: /data/bbuf/hunyuanvideo_fp8_20260420/profiler/fp8/63e4807c-f120-4fdc-b5b0-c1ab96522a7e-5_steps-global-rank0.trace.json.gz
92
+
93
+ [04-20 02:39:04] [DenoisingStage] average time per step: 2.2673 seconds
94
+ [04-20 02:39:04] [DenoisingStage] finished in 18.1401 seconds
95
+ [04-20 02:39:04] [DecodingStage] started...
96
+ [04-20 02:39:09] [DecodingStage] finished in 5.3886 seconds
97
+ [04-20 02:39:10] Pixel data generated successfully in 24.93 seconds
98
+ [04-20 02:39:10] Completed batch processing. Generated 1 outputs in 24.93 seconds
99
+ [04-20 02:39:10] Generator was garbage collected without being shut down. Attempting to shut down the local server and client.
100
+ [04-20 02:39:17] Worker 0: Shutdown complete.
validation/h100_20260420/profiler/5b26da87-550f-4d35-8cb8-bf9aa9c535d0-5_steps-global-rank0.trace.json.gz ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a5e2d63975d0b13aa8233fcd7d2660ed2ec590ab66dbdee2ee37a75ae70d50ce
3
+ size 13790754
validation/h100_20260420/profiler/63e4807c-f120-4fdc-b5b0-c1ab96522a7e-5_steps-global-rank0.trace.json.gz ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c248e9cc2f382fc2f7a28d19939a5ebfb8b9fbb3a8353b46f50a6dd103526ea6
3
+ size 17625456
validation/h100_20260420/profiler/kernel_summary.md ADDED
@@ -0,0 +1,53 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # BF16 trace
2
+ trace: /data/bbuf/hunyuanvideo_fp8_20260420/profiler/bf16/5b26da87-550f-4d35-8cb8-bf9aa9c535d0-5_steps-global-rank0.trace.json.gz
3
+ total kernel time: 2481.464 ms
4
+
5
+ | rank | kernel | count | time_ms | share |
6
+ |---:|---|---:|---:|---:|
7
+ | 1 | `void cutlass::device_kernel<flash::enable_sm90_or_later<flash::FlashAttnFwdSm90<flash::CollectiveMainloopFwdSm90<2, cute::tuple<cute::C<2>, ` | 360 | 691.122 | 27.85% |
8
+ | 2 | `nvjet_tst_192x208_64x4_2x1_v_bz_coopB_bias_TNT` | 240 | 437.843 | 17.64% |
9
+ | 3 | `nvjet_tst_256x136_64x4_1x2_h_bz_coopA_bias_TNT` | 240 | 299.272 | 12.06% |
10
+ | 4 | `nvjet_tst_128x232_64x4_2x1_v_bz_coopA_bias_TNT` | 246 | 152.702 | 6.15% |
11
+ | 5 | `void at::native::(anonymous namespace)::CatArrayBatchedCopy<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int, 4, 64, 64>(at::` | 720 | 151.870 | 6.12% |
12
+ | 6 | `nvjet_tst_256x144_64x4_2x1_v_bz_coopA_bias_TNT` | 120 | 126.300 | 5.09% |
13
+ | 7 | `void at::native::(anonymous namespace)::CatArrayBatchedCopy_alignedK_contig<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int,` | 600 | 111.988 | 4.51% |
14
+ | 8 | `nvjet_tst_192x192_64x4_2x1_v_bz_coopB_bias_TNN` | 120 | 97.632 | 3.93% |
15
+ | 9 | `void at::native::elementwise_kernel<128, 4, at::native::gpu_kernel_impl_nocast<at::native::GeluCUDAKernelImpl(at::TensorIteratorBase&, at::n` | 240 | 94.940 | 3.83% |
16
+ | 10 | `void at::native::(anonymous namespace)::CatArrayBatchedCopy_alignedK_contig<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int,` | 246 | 73.201 | 2.95% |
17
+ | 11 | `void at::native::elementwise_kernel<128, 4, at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&):` | 972 | 62.019 | 2.50% |
18
+ | 12 | `_rotary_embedding_kernel` | 720 | 42.580 | 1.72% |
19
+ | 13 | `_rms_norm_tiled_onepass` | 960 | 29.808 | 1.20% |
20
+ | 14 | `fuse_scale_shift_kernel_blc_opt` | 480 | 22.100 | 0.89% |
21
+ | 15 | `void at::native::vectorized_elementwise_kernel<8, at::native::GeluCUDAKernelImpl(at::TensorIteratorBase&, at::native::GeluType)::{lambda()#1` | 240 | 21.438 | 0.86% |
22
+ | 16 | `kernel_cutlass_kernel_sglangjit_kerneldiffusioncutedslscale_residual_norm_scale_shiftScaleResidualNormScaleShift_object_at__tensorptrbf16_gm` | 480 | 18.923 | 0.76% |
23
+ | 17 | `kernel_cutlass_kernel_sglangjit_kerneldiffusioncutedslscale_residual_norm_scale_shiftScaleResidualNormScaleShift_object_at__tensorptrbf16_gm` | 240 | 10.332 | 0.42% |
24
+ | 18 | `nvjet_tst_256x8_64x6_4x1_v_bz_bias_TNT` | 240 | 9.724 | 0.39% |
25
+ | 19 | `nvjet_tst_128x32_64x10_4x1_v_bz_bias_TNT` | 264 | 7.963 | 0.32% |
26
+ | 20 | `nvjet_tst_128x8_64x12_4x1_v_bz_bias_TNT` | 240 | 6.732 | 0.27% |
27
+
28
+ # FP8 trace
29
+ trace: /data/bbuf/hunyuanvideo_fp8_20260420/profiler/fp8/63e4807c-f120-4fdc-b5b0-c1ab96522a7e-5_steps-global-rank0.trace.json.gz
30
+ total kernel time: 2088.351 ms
31
+
32
+ | rank | kernel | count | time_ms | share |
33
+ |---:|---|---:|---:|---:|
34
+ | 1 | `void cutlass::device_kernel<flash::enable_sm90_or_later<flash::FlashAttnFwdSm90<flash::CollectiveMainloopFwdSm90<2, cute::tuple<cute::C<2>, ` | 360 | 692.486 | 33.16% |
35
+ | 2 | `_ZN7cutlass13device_kernelINS_4gemm6kernel13GemmUniversalIN4cute5tupleIJiiiiEEENS1_10collective13CollectiveMmaINS1_34MainloopSm90TmaGmmaWarp` | 960 | 650.610 | 31.15% |
36
+ | 3 | `void at::native::(anonymous namespace)::CatArrayBatchedCopy<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int, 4, 64, 64>(at::` | 720 | 152.032 | 7.28% |
37
+ | 4 | `void at::native::(anonymous namespace)::CatArrayBatchedCopy_alignedK_contig<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int,` | 600 | 112.280 | 5.38% |
38
+ | 5 | `void at::native::elementwise_kernel<128, 4, at::native::gpu_kernel_impl_nocast<at::native::GeluCUDAKernelImpl(at::TensorIteratorBase&, at::n` | 240 | 95.617 | 4.58% |
39
+ | 6 | `void at::native::(anonymous namespace)::CatArrayBatchedCopy_alignedK_contig<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int,` | 246 | 73.946 | 3.54% |
40
+ | 7 | `_static_quant_fp8` | 1440 | 71.500 | 3.42% |
41
+ | 8 | `void at::native::elementwise_kernel<128, 4, at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&):` | 972 | 62.236 | 2.98% |
42
+ | 9 | `_rotary_embedding_kernel` | 720 | 42.578 | 2.04% |
43
+ | 10 | `_rms_norm_tiled_onepass` | 960 | 29.710 | 1.42% |
44
+ | 11 | `fuse_scale_shift_kernel_blc_opt` | 480 | 21.832 | 1.05% |
45
+ | 12 | `void at::native::vectorized_elementwise_kernel<8, at::native::GeluCUDAKernelImpl(at::TensorIteratorBase&, at::native::GeluType)::{lambda()#1` | 240 | 21.108 | 1.01% |
46
+ | 13 | `kernel_cutlass_kernel_sglangjit_kerneldiffusioncutedslscale_residual_norm_scale_shiftScaleResidualNormScaleShift_object_at__tensorptrbf16_gm` | 480 | 18.938 | 0.91% |
47
+ | 14 | `kernel_cutlass_kernel_sglangjit_kerneldiffusioncutedslscale_residual_norm_scale_shiftScaleResidualNormScaleShift_object_at__tensorptrbf16_gm` | 240 | 10.268 | 0.49% |
48
+ | 15 | `nvjet_tst_256x8_64x6_4x1_v_bz_bias_TNT` | 240 | 9.710 | 0.46% |
49
+ | 16 | `_ZN7cutlass13device_kernelINS_4gemm6kernel13GemmUniversalIN4cute5tupleIJiiiiEEENS1_10collective13CollectiveMmaINS1_34MainloopSm90TmaGmmaWarp` | 480 | 8.525 | 0.41% |
50
+ | 17 | `nvjet_tst_128x8_64x12_4x1_v_bz_bias_TNT` | 240 | 6.747 | 0.32% |
51
+ | 18 | `void at::native::vectorized_elementwise_kernel<8, at::native::(anonymous namespace)::silu_kernel(at::TensorIteratorBase&)::{lambda()#1}::ope` | 540 | 1.212 | 0.06% |
52
+ | 19 | `void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const:` | 90 | 0.932 | 0.04% |
53
+ | 20 | `void at::native::elementwise_kernel<128, 4, at::native::gpu_kernel_impl<at::native::BinaryFunctor<float, float, float, at::native::binary_in` | 6 | 0.849 | 0.04% |
validation/h100_20260420/result_summary.md ADDED
@@ -0,0 +1,53 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # HunyuanVideo ModelOpt FP8 Validation Summary
2
+
3
+ HF repo: https://huggingface.co/BBuf/HunyuanVideo-ModelOpt-FP8-SGLang
4
+
5
+ ## Environment
6
+
7
+ - Host/GPU: H100, rank0 (`CUDA_VISIBLE_DEVICES=0`)
8
+ - SGLang base: latest main plus Qwen Image FP8 PR changes and HunyuanVideo FP8 changes
9
+ - ModelOpt: latest main for quantization
10
+ - Base model: `hunyuanvideo-community/HunyuanVideo`
11
+ - Resolution: 960x544, 17 frames
12
+ - Steps: 8
13
+ - Seed: 0
14
+ - Offload: `--dit-cpu-offload false --dit-layerwise-offload false`
15
+
16
+ ## Output Artifacts
17
+
18
+ - BF16 video: `artifacts/hunyuanvideo_bf16_544x960_17f_8steps.mp4`
19
+ - FP8 video: `artifacts/hunyuanvideo_fp8_544x960_17f_8steps.mp4`
20
+ - Contact sheet: `artifacts/hunyuanvideo_bf16_fp8_contact_sheet.png`
21
+ - BF16 perf JSON: `artifacts/hunyuanvideo_bf16_544x960_17f_8steps_perf.json`
22
+ - FP8 perf JSON: `artifacts/hunyuanvideo_fp8_544x960_17f_8steps_perf.json`
23
+ - Kernel summary: `profiler/kernel_summary.md`
24
+
25
+ ## Benchmark
26
+
27
+ Offline throughput, random dataset, 3 prompts, batch size 1. FP8 uses `--transformer-path /data/bbuf/hunyuanvideo_fp8_20260420/sglang_transformer`.
28
+
29
+ | checkpoint | total duration | latency/request | requests/s | MP/s |
30
+ | --- | ---: | ---: | ---: | ---: |
31
+ | BF16 | 26.11s | 8.70s | 0.115 | 1.020 |
32
+ | FP8 | 24.90s | 8.30s | 0.120 | 1.070 |
33
+
34
+ Speedup:
35
+
36
+ - End-to-end latency: 1.049x
37
+ - End-to-end throughput: 1.049x
38
+ - Steady denoise step from logs: about 1.19x (`~0.379s/step` BF16 to `~0.319s/step` FP8)
39
+ - Transformer load memory: 23.88 GB BF16 to 15.45 GB FP8
40
+
41
+ ## Commands
42
+
43
+ BF16 benchmark:
44
+
45
+ ```bash
46
+ python -m sglang.multimodal_gen.benchmarks.bench_offline_throughput --backend=sglang --model-path /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773 --dataset random --num-prompts 3 --batch-size 1 --width 960 --height 544 --num-frames 17 --num-inference-steps 8 --guidance-scale 1.0 --seed 0 --dit-cpu-offload false --dit-layerwise-offload false
47
+ ```
48
+
49
+ FP8 benchmark:
50
+
51
+ ```bash
52
+ python -m sglang.multimodal_gen.benchmarks.bench_offline_throughput --backend=sglang --model-path /root/.cache/huggingface/hub/models--hunyuanvideo-community--HunyuanVideo/snapshots/e8c2aaa66fe3742a32c11a6766aecbf07c56e773 --transformer-path /data/bbuf/hunyuanvideo_fp8_20260420/sglang_transformer --dataset random --num-prompts 3 --batch-size 1 --width 960 --height 544 --num-frames 17 --num-inference-steps 8 --guidance-scale 1.0 --seed 0 --dit-cpu-offload false --dit-layerwise-offload false
53
+ ```