Failed to get device capability: SM 12.x requires CUDA >= 12.9. Failed to get device capability: SM 12.x requires CUDA >= 12.9. [08-31 15:03:40] Applying performance_mode=speed [08-31 15:03:40] server_args: {"model_path": "/data/hf_models/MiniMax-H3", "model_subfolder": null, "model_variant": "Ref2VA", "model_id": null, "backend": "sglang", "attention_backend": null, "attention_backend_config": {}, "component_attention_backends": {}, "cache_dit_config": null, "nccl_port": null, "trust_remote_code": false, "revision": null, "num_gpus": 2, "performance_mode": "speed", "base_gpu_id": 0, "gpu_ids": null, "tp_size": 2, "sp_degree": 1, "ulysses_degree": 1, "ring_degree": 1, "dp_size": 1, "dp_degree": 1, "enable_cfg_parallel": false, "cfg_parallel_degree": 1, "encoder_parallel": "auto", "hsdp_replicate_dim": 1, "hsdp_shard_dim": 2, "dist_timeout": 3600, "pipeline_class_name": null, "lora_path": null, "lora_nickname": "default", "lora_scale": 1.0, "lora_merge_mode": "auto", "lora_weight_name": null, "component_paths": {}, "transformer_weights_path": null, "component_transformer_weights_paths": {}, "quantization": null, "quantization_ignored_layers": null, "lora_target_modules": null, "dit_cpu_offload": false, "dit_layerwise_offload": false, "layerwise_offload_components": null, "dit_offload_prefetch_size": 0.0, "dit_layerwise_resident_layers": 0.0, "offload_during_compile": true, "text_encoder_cpu_offload": false, "image_encoder_cpu_offload": false, "vae_cpu_offload": false, "use_fsdp_inference": false, "pin_cpu_memory": true, "ltx2_two_stage_device_mode": null, "comfyui_mode": false, "enable_torch_compile": false, "regional_compile": false, "enable_breakable_cuda_graph": false, "bcg_text_buckets": null, "enable_layerwise_nvtx_marker": false, "warmup_mode": "server", "warmup": true, "server_warmup": true, "warmup_resolutions": null, "warmup_steps": 1, "disable_autocast": false, "master_port": 35010, "host": "0.0.0.0", "port": 34020, "webui": false, "webui_port": 12312, "scheduler_port": 36010, "batching_mode": "dynamic", "batching_max_size": 1, "batching_delay_ms": 0.0, "batching_config": null, "enable_batching_metrics": false, "strict_ports": false, "output_path": "/data/wxy/sskj-h3/throughput/sglang-base/results/ref2va-feishu-base-tp2x4-768p-15s-20steps-20260831-150318/server_1_port34020/outputs", "input_save_path": "inputs/uploads", "prompt_file_path": null, "model_paths": {}, "model_loaded": {"transformer": true, "vae": true, "video_vae": true, "audio_vae": true, "video_dit": true, "audio_dit": true, "dual_tower_bridge": true}, "boundary_ratio": null, "disagg_role": "monolithic", "disagg_timeout": 3600, "disagg_downstream_wait_timeout": 1800, "disagg_dispatch_policy": "round_robin", "disagg_mode": false, "disagg_instance_id": 0, "disagg_max_slots_per_instance": 8, "disagg_transfer_redundancy": 1.25, "disagg_role_device": "auto", "disagg_transfer_backend": "auto", "disagg_transfer_pool_size": 268435456, "disagg_transfer_pin_memory": "auto", "disagg_p2p_hostname": "127.0.0.1", "disagg_ib_device": null, "disagg_server_addr": null, "encoder_urls": null, "denoiser_urls": null, "decoder_urls": null, "encoder_tp": null, "denoiser_tp": null, "denoiser_sp": null, "denoiser_ulysses": null, "denoiser_ring": null, "decoder_sp": null, "decoder_tp": null, "pool_work_endpoint": null, "pool_result_endpoint": null, "pool_control_endpoint": null, "pool_control_advertised_endpoint": null, "log_level": "info", "log_requests": false, "log_requests_level": 2, "log_requests_format": "text", "log_requests_target": null, "uvicorn_access_log_exclude_prefixes": [], "enable_trace": false, "otlp_traces_endpoint": "localhost:4317", "srt_encoder_url": null, "srt_encoder_connect_timeout": 3.05, "srt_encoder_timeout": 100, "pe_server_url": null} [08-31 15:03:40] Starting server... Failed to get device capability: SM 12.x requires CUDA >= 12.9. Failed to get device capability: SM 12.x requires CUDA >= 12.9. Failed to get device capability: SM 12.x requires CUDA >= 12.9. Failed to get device capability: SM 12.x requires CUDA >= 12.9. [08-31 15:03:59] Scheduler bind at endpoint: tcp://0.0.0.0:36010 [08-31 15:04:00] torch.compile cache: TORCHINDUCTOR_CACHE_DIR=/root/.cache/sgl_diffusion/torch_compile_cache/inductor TRITON_CACHE_DIR=/root/.cache/sgl_diffusion/torch_compile_cache/triton [08-31 15:04:00] Initializing distributed environment with world_size=2, device=cuda:0, timeout=3600 [08-31 15:04:00] Setting distributed timeout to 3600 seconds [08-31 15:04:01] Found nccl from library libnccl.so.2 [08-31 15:04:01] sglang-diffusion is using nccl==2.28.9 [08-31 15:04:04] reading GPU P2P access cache from /root/.cache/sglang/gpu_p2p_access_cache_for_2,3.json [08-31 15:04:04] reading GPU P2P access cache from /root/.cache/sglang/gpu_p2p_access_cache_for_2,3.json [08-31 15:04:04] Found nccl from library libnccl.so.2 [08-31 15:04:04] sglang-diffusion is using nccl==2.28.9 [08-31 15:04:04] No pipeline_class_name specified, using model_index.json Loading required modules: 0%| | 0/6 [00:00