diff --git a/src/h3_blackwell_runtime/t2v.py b/src/h3_blackwell_runtime/t2v.py index e1e5913..833fef8 100644 --- a/src/h3_blackwell_runtime/t2v.py +++ b/src/h3_blackwell_runtime/t2v.py @@ -28,3 +28,12 @@ def empty_av_latents(width: int, height: int, frames: int, *, device: torch.devi video = torch.zeros((1, 24, video_steps, height // 16, width // 16), device=device) audio = torch.zeros((1, 32, 2, audio_steps), device=device) return video, audio, frames + + +def random_av_latents(width: int, height: int, frames: int, seed: int, *, device: torch.device | str = "cuda") -> tuple[torch.Tensor, torch.Tensor, int]: + """Generate Comfy-equivalent CPU-seeded joint AV noise latents.""" + video, audio, frames = empty_av_latents(width, height, frames, device="cpu") + generator = torch.manual_seed(seed) + video = torch.randn(video.shape, dtype=torch.float32, generator=generator, device="cpu").to(video.dtype) + audio = torch.randn(audio.shape, dtype=torch.float32, generator=generator, device="cpu").to(audio.dtype) + return video.to(device), audio.to(device), frames diff --git a/tools/direct_t2v_preview.py b/tools/direct_t2v_preview.py index 612c738..958b4e9 100644 --- a/tools/direct_t2v_preview.py +++ b/tools/direct_t2v_preview.py @@ -14,7 +14,7 @@ from h3_blackwell_runtime.denoiser import H3PackedDenoiser from h3_blackwell_runtime.packing import H3PromptPacker from h3_blackwell_runtime.qwen3vl_text import Qwen3VLPromptConditioner from h3_blackwell_runtime.sampler import sample_video_res_multistep -from h3_blackwell_runtime.t2v import empty_av_latents +from h3_blackwell_runtime.t2v import random_av_latents from h3_blackwell_runtime.token_refiner import H3TokenRefiner from h3_blackwell_runtime.vae_decoder import MiniMaxH3VideoVAE @@ -30,14 +30,12 @@ parser.add_argument("--seed", type=int, default=440204) parser.add_argument("--attention", choices=("sage2", "sdpa", "sage3"), default="sage2") args = parser.parse_args() -torch.manual_seed(args.seed) checkpoint = H3Checkpoint("/models/minimax_h3_fl2va_pruned_nvfp4.safetensors") conditioner = Qwen3VLPromptConditioner( "/text-encoders/qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors", "/opt/h3-blackwell-runtime/src/h3_blackwell_runtime/qwen25_tokenizer", ) -video, audio, frames = empty_av_latents(args.width, args.height, args.frames) -video.normal_() +video, audio, frames = random_av_latents(args.width, args.height, args.frames, args.seed) model = H3PackedDenoiser.from_checkpoint(checkpoint, attention_backend=args.attention).eval() text = H3TokenRefiner(checkpoint, attention_backend=args.attention)(conditioner(args.prompt)) latent = sample_video_res_multistep(model, H3PromptPacker(checkpoint), text, video, audio, steps=args.steps)