diff --git a/README.md b/README.md index 66b9fb3e..144f36bf 100644 --- a/README.md +++ b/README.md @@ -1523,6 +1523,8 @@ Example code for LingBot-Video is available at: [/examples/lingbot_video/](/exam |[Robbyant/lingbot-video-moe-30b-a3b: T2V](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_t2v.py)|[code](/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_t2v.py)|[code](/examples/lingbot_video/model_training/full/lingbot-video-moe-30b-a3b_t2v.sh)|[code](/examples/lingbot_video/model_training/validate_full/lingbot-video-moe-30b-a3b_t2v.py)|[code](/examples/lingbot_video/model_training/lora/lingbot-video-moe-30b-a3b_t2v.sh)|[code](/examples/lingbot_video/model_training/validate_lora/lingbot-video-moe-30b-a3b_t2v.py)| |[Robbyant/lingbot-video-moe-30b-a3b: TI2V](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_ti2v.py)|[code](/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_ti2v.py)|[code](/examples/lingbot_video/model_training/full/lingbot-video-moe-30b-a3b_ti2v.sh)|[code](/examples/lingbot_video/model_training/validate_full/lingbot-video-moe-30b-a3b_ti2v.py)|[code](/examples/lingbot_video/model_training/lora/lingbot-video-moe-30b-a3b_ti2v.sh)|[code](/examples/lingbot_video/model_training/validate_lora/lingbot-video-moe-30b-a3b_ti2v.py)| |[Robbyant/lingbot-video-moe-30b-a3b: T2I](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_t2i.py)|[code](/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_t2i.py)|-|-|-|-| +|[Robbyant/lingbot-video-moe-30b-a3b: T2V + Refinement](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_t2v_refiner.py)|[code](/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_t2v_refiner.py)|-|-|-|-| +|[Robbyant/lingbot-video-moe-30b-a3b: TI2V + Refinement](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_ti2v_refiner.py)|[code](/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_ti2v_refiner.py)|-|-|-|-| diff --git a/README_zh.md b/README_zh.md index 91f7e6e9..7f65e32c 100644 --- a/README_zh.md +++ b/README_zh.md @@ -1523,6 +1523,8 @@ LingBot-Video 的示例代码位于:[/examples/lingbot_video/](/examples/lingb |[Robbyant/lingbot-video-moe-30b-a3b: T2V](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_t2v.py)|[code](/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_t2v.py)|[code](/examples/lingbot_video/model_training/full/lingbot-video-moe-30b-a3b_t2v.sh)|[code](/examples/lingbot_video/model_training/validate_full/lingbot-video-moe-30b-a3b_t2v.py)|[code](/examples/lingbot_video/model_training/lora/lingbot-video-moe-30b-a3b_t2v.sh)|[code](/examples/lingbot_video/model_training/validate_lora/lingbot-video-moe-30b-a3b_t2v.py)| |[Robbyant/lingbot-video-moe-30b-a3b: TI2V](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_ti2v.py)|[code](/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_ti2v.py)|[code](/examples/lingbot_video/model_training/full/lingbot-video-moe-30b-a3b_ti2v.sh)|[code](/examples/lingbot_video/model_training/validate_full/lingbot-video-moe-30b-a3b_ti2v.py)|[code](/examples/lingbot_video/model_training/lora/lingbot-video-moe-30b-a3b_ti2v.sh)|[code](/examples/lingbot_video/model_training/validate_lora/lingbot-video-moe-30b-a3b_ti2v.py)| |[Robbyant/lingbot-video-moe-30b-a3b: T2I](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_t2i.py)|[code](/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_t2i.py)|-|-|-|-| +|[Robbyant/lingbot-video-moe-30b-a3b: T2V + Refinement](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_t2v_refiner.py)|[code](/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_t2v_refiner.py)|-|-|-|-| +|[Robbyant/lingbot-video-moe-30b-a3b: TI2V + Refinement](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_ti2v_refiner.py)|[code](/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_ti2v_refiner.py)|-|-|-|-| diff --git a/diffsynth/diffusion/flow_match.py b/diffsynth/diffusion/flow_match.py index 2547d199..2567934f 100644 --- a/diffsynth/diffusion/flow_match.py +++ b/diffsynth/diffusion/flow_match.py @@ -5,7 +5,7 @@ class FlowMatchScheduler(): - def __init__(self, template: Literal["FLUX.1", "Wan", "Qwen-Image", "FLUX.2", "Z-Image", "LTX-2", "Qwen-Image-Lightning", "ERNIE-Image", "ACE-Step", "Ideogram4", "Krea-2", "Boogu", "MiniMax-H3"] = "FLUX.1"): + def __init__(self, template: Literal["FLUX.1", "Wan", "Qwen-Image", "FLUX.2", "Z-Image", "LTX-2", "Qwen-Image-Lightning", "ERNIE-Image", "ACE-Step", "Ideogram4", "Krea-2", "Boogu", "MiniMax-H3", "LingBot-Video"] = "FLUX.1"): self.set_timesteps_fn = { "FLUX.1": FlowMatchScheduler.set_timesteps_flux, "Wan": FlowMatchScheduler.set_timesteps_wan, @@ -21,6 +21,7 @@ def __init__(self, template: Literal["FLUX.1", "Wan", "Qwen-Image", "FLUX.2", "Z "Krea-2": FlowMatchScheduler.set_timesteps_krea2, "Boogu": FlowMatchScheduler.set_timesteps_boogu, "MiniMax-H3": FlowMatchScheduler.set_timesteps_minimax_h3, + "LingBot-Video": FlowMatchScheduler.set_timesteps_lingbot_video, }.get(template, FlowMatchScheduler.set_timesteps_flux) self.num_train_timesteps = 1000 @@ -80,6 +81,28 @@ def set_timesteps_qwen_image(num_inference_steps=100, denoising_strength=1.0, ex timesteps = sigmas * num_train_timesteps return sigmas, timesteps + @staticmethod + def set_timesteps_lingbot_video(num_inference_steps=100, denoising_strength=1.0, shift=None, t_thresh=None, sigma_tail_steps=0): + sigma_min = 0.0 + sigma_max = 1.0 + shift = 5 if shift is None else shift + num_train_timesteps = 1000 + sigma_start = sigma_min + (sigma_max - sigma_min) * denoising_strength + sigmas = torch.linspace(sigma_start, sigma_min, num_inference_steps + 1)[:-1] + sigmas = shift * sigmas / (1 + (shift - 1) * sigmas) + if t_thresh is not None: + # Refinement schedule: keep the sub-threshold part of the shifted grid, pin the first + # sigma exactly at t_thresh, then append extra low-noise steps that end at sigma_min. + sigmas = sigmas[sigmas <= t_thresh + 1e-6] + if sigmas.numel() == 0 or abs(float(sigmas[0]) - t_thresh) > 1e-6: + sigmas = torch.cat([torch.tensor([t_thresh], dtype=sigmas.dtype), sigmas]) + if sigma_tail_steps > 0: + tail_start = float(sigmas[-1]) + tail = torch.linspace(tail_start, min(sigma_min, tail_start), sigma_tail_steps + 2)[1:-1] + sigmas = torch.cat([sigmas, tail.to(dtype=sigmas.dtype)]) + timesteps = sigmas * num_train_timesteps + return sigmas, timesteps + @staticmethod def set_timesteps_qwen_image_lightning(num_inference_steps=100, denoising_strength=1.0, exponential_shift_mu=None, dynamic_shift_len=None): sigma_min = 0.0 diff --git a/diffsynth/pipelines/lingbot_video.py b/diffsynth/pipelines/lingbot_video.py index 28973198..21f67e9a 100644 --- a/diffsynth/pipelines/lingbot_video.py +++ b/diffsynth/pipelines/lingbot_video.py @@ -25,7 +25,7 @@ def __init__(self, device=get_device_type(), torch_dtype=torch.bfloat16): height_division_factor=16, width_division_factor=16, time_division_factor=4, time_division_remainder=1, ) - self.scheduler = FlowMatchScheduler(template="Wan") + self.scheduler = FlowMatchScheduler(template="LingBot-Video") self.text_encoder: Krea2TextEncoder = None self.dit: LingBotVideoDiT = None self.vae: QwenImageVAE = None @@ -102,11 +102,17 @@ def __call__( # Scheduler num_inference_steps: int = 40, sigma_shift: float = 3.0, + # High-Resolution Refinement + t_thresh: float = None, + sigma_tail_steps: int = 2, # progress_bar progress_bar_cmd=tqdm, ): # Scheduler - self.scheduler.set_timesteps(num_inference_steps, denoising_strength=denoising_strength, shift=sigma_shift) + self.scheduler.set_timesteps( + num_inference_steps, denoising_strength=denoising_strength, shift=sigma_shift, + t_thresh=t_thresh, sigma_tail_steps=sigma_tail_steps, + ) # Inputs inputs_posi = {"prompt": prompt} @@ -116,7 +122,7 @@ def __call__( "input_video": input_video, "denoising_strength": denoising_strength, "seed": seed, "rand_device": rand_device, "height": height, "width": width, "num_frames": num_frames, - "cfg_scale": cfg_scale, + "cfg_scale": cfg_scale, "t_thresh": t_thresh, } for unit in self.units: inputs_shared, inputs_posi, inputs_nega = self.unit_runner(unit, self, inputs_shared, inputs_posi, inputs_nega) @@ -197,12 +203,12 @@ class LingBotVideoUnit_ImageEmbedder(PipelineUnit): def __init__(self): super().__init__( - input_params=("input_image", "latents", "height", "width"), + input_params=("input_image", "latents", "height", "width", "t_thresh"), output_params=("latents", "first_frame_latents", "vlm_image"), onload_model_names=("vae",), ) - def process(self, pipe: LingBotVideoPipeline, input_image, latents, height, width): + def process(self, pipe: LingBotVideoPipeline, input_image, latents, height, width, t_thresh=None): if input_image is None: return {} pipe.load_models_to_device(self.onload_model_names) @@ -211,7 +217,8 @@ def process(self, pipe: LingBotVideoPipeline, input_image, latents, height, widt pixel = self.preprocess_cond_image(input_image, height, width) pixel = pixel.to(dtype=pipe.torch_dtype, device=pipe.device) first_frame_latents = pipe.vae.encode_video(pixel * 2.0 - 1.0).to(dtype=pipe.torch_dtype, device=pipe.device) - vlm_image = self.vlm_image(pipe, pixel) + # The refiner conditions on text only and re-pins the frame-0 latent every step instead. + vlm_image = None if t_thresh is not None else self.vlm_image(pipe, pixel) # Pin the clean condition latent into the first temporal slot before sampling. cond_t = first_frame_latents.shape[2] latents[:, :, :cond_t] = first_frame_latents diff --git a/docs/en/Model_Details/LingBot-Video.md b/docs/en/Model_Details/LingBot-Video.md index 0e6768ae..20ce2fd9 100644 --- a/docs/en/Model_Details/LingBot-Video.md +++ b/docs/en/Model_Details/LingBot-Video.md @@ -78,6 +78,8 @@ save_video(video, "video.mp4", fps=15, quality=10) |[Robbyant/lingbot-video-moe-30b-a3b: T2V](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_t2v.py)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_t2v.py)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_training/full/lingbot-video-moe-30b-a3b_t2v.sh)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_training/validate_full/lingbot-video-moe-30b-a3b_t2v.py)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_training/lora/lingbot-video-moe-30b-a3b_t2v.sh)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_training/validate_lora/lingbot-video-moe-30b-a3b_t2v.py)| |[Robbyant/lingbot-video-moe-30b-a3b: TI2V](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_ti2v.py)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_ti2v.py)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_training/full/lingbot-video-moe-30b-a3b_ti2v.sh)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_training/validate_full/lingbot-video-moe-30b-a3b_ti2v.py)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_training/lora/lingbot-video-moe-30b-a3b_ti2v.sh)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_training/validate_lora/lingbot-video-moe-30b-a3b_ti2v.py)| |[Robbyant/lingbot-video-moe-30b-a3b: T2I](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_t2i.py)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_t2i.py)|-|-|-|-| +|[Robbyant/lingbot-video-moe-30b-a3b: T2V + Refinement](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_t2v_refiner.py)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_t2v_refiner.py)|-|-|-|-| +|[Robbyant/lingbot-video-moe-30b-a3b: TI2V + Refinement](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_ti2v_refiner.py)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_ti2v_refiner.py)|-|-|-|-| ## Model Inference @@ -96,12 +98,33 @@ The input parameters for `LingBotVideoPipeline` inference include: * `cfg_scale`: Classifier-free guidance scale, default `3.0`. * `num_inference_steps`: Number of inference steps, default `40`. * `sigma_shift`: Flow-matching timestep shift, default `3.0`. +* `t_thresh`: Refinement start sigma, default `None` (plain generation). When set, the schedule is truncated so that sampling starts at `sigma=t_thresh` and `input_video` is noised to exactly that level. Only meaningful together with `input_video`; TI2V additionally re-pins the clean first-frame latent after every step. The official refiner setting is `0.85`. +* `sigma_tail_steps`: Number of extra low-noise steps appended to the tail of the refinement schedule, default `2`. Only effective when `t_thresh` is set. * `seed`: Random seed. Default is `None`, meaning completely random. * `rand_device`: Device for generating the initial noise, default `"cpu"`. * `progress_bar_cmd`: Progress bar, default `tqdm`. Can be disabled by setting to `lambda x: x`. If VRAM is insufficient, please enable [VRAM Management](../Pipeline_Usage/VRAM_management.md). We provide recommended low-VRAM configurations for each task in the example code, see the table in the "Model Overview" section above. +### Two-stage refinement + +The MoE refiner performs a short second pass at a higher resolution: the official setup generates at 480×832 with 40 steps, then refines at 1088×1920 with 8 steps. Load the pipeline with the `refiner/` shards instead of `transformer/`, feed the base clip back in through `input_video` at the higher resolution, and set `t_thresh`: + +```python +input_video = VideoData("video_base.mp4", height=1088, width=1920) +video = pipe( + prompt=caption, + negative_prompt=pipe.default_negative_prompt, + input_video=input_video, + height=1088, width=1920, num_frames=81, + num_inference_steps=8, cfg_scale=3.0, + t_thresh=0.85, sigma_tail_steps=2, + seed=0, +) +``` + +The upscaled clip is VAE-encoded and noised back to `sigma=t_thresh`, so the pass keeps the structure of the base clip and regenerates detail at the target resolution. Pass the same caption as the base pass and keep the same aspect ratio. The refinement resolution dominates the cost — at 1088×1920 the sequence is ~5× longer than at 480×832 — so run this pass with VRAM management enabled. + ### Prompt rewriting LingBot-Video is trained on **structured-JSON captions**, not free-form prose. Feeding a flat sentence is out-of-distribution and visibly degrades quality. The pipeline accepts a caption as a `dict` (the format used at training time) or a plain string, and normalises the `dict` internally. diff --git a/docs/zh/Model_Details/LingBot-Video.md b/docs/zh/Model_Details/LingBot-Video.md index 74c111f3..49b79055 100644 --- a/docs/zh/Model_Details/LingBot-Video.md +++ b/docs/zh/Model_Details/LingBot-Video.md @@ -78,6 +78,8 @@ save_video(video, "video.mp4", fps=15, quality=10) |[Robbyant/lingbot-video-moe-30b-a3b: T2V](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_t2v.py)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_t2v.py)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_training/full/lingbot-video-moe-30b-a3b_t2v.sh)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_training/validate_full/lingbot-video-moe-30b-a3b_t2v.py)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_training/lora/lingbot-video-moe-30b-a3b_t2v.sh)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_training/validate_lora/lingbot-video-moe-30b-a3b_t2v.py)| |[Robbyant/lingbot-video-moe-30b-a3b: TI2V](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_ti2v.py)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_ti2v.py)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_training/full/lingbot-video-moe-30b-a3b_ti2v.sh)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_training/validate_full/lingbot-video-moe-30b-a3b_ti2v.py)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_training/lora/lingbot-video-moe-30b-a3b_ti2v.sh)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_training/validate_lora/lingbot-video-moe-30b-a3b_ti2v.py)| |[Robbyant/lingbot-video-moe-30b-a3b: T2I](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_t2i.py)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_t2i.py)|-|-|-|-| +|[Robbyant/lingbot-video-moe-30b-a3b: T2V + Refinement](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_t2v_refiner.py)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_t2v_refiner.py)|-|-|-|-| +|[Robbyant/lingbot-video-moe-30b-a3b: TI2V + Refinement](https://modelscope.cn/models/Robbyant/lingbot-video-moe-30b-a3b)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_ti2v_refiner.py)|[code](https://github.com/modelscope/DiffSynth-Studio/blob/main/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_ti2v_refiner.py)|-|-|-|-| ## 模型推理 @@ -96,12 +98,33 @@ save_video(video, "video.mp4", fps=15, quality=10) * `cfg_scale`: 无分类器指导强度,默认 `3.0`。 * `num_inference_steps`: 推理步数,默认 `40`。 * `sigma_shift`: Flow-matching 时间步 shift,默认 `3.0`。 +* `t_thresh`: 精修起始 sigma,默认 `None`(普通生成)。设置后调度会被截断,使采样从 `sigma=t_thresh` 开始,并把 `input_video` 加噪到该噪声水平。仅在提供 `input_video` 时有意义;TI2V 还会在每个采样步之后重新写入干净的首帧 latent。官方精修配置为 `0.85`。 +* `sigma_tail_steps`: 精修调度尾部追加的额外低噪声步数,默认 `2`。仅在设置了 `t_thresh` 时生效。 * `seed`: 随机种子,默认 `None`(完全随机)。 * `rand_device`: 生成初始噪声的设备,默认 `"cpu"`。 * `progress_bar_cmd`: 进度条,默认 `tqdm`,可设为 `lambda x: x` 关闭。 显存不足时请参考[显存管理](../Pipeline_Usage/VRAM_management.md)启用显存管理功能。我们在示例代码中提供了每个任务的推荐低显存配置,见上方"模型总览"中的表格。 +### 两阶段精修 + +MoE 的 refiner 会在更高分辨率上执行一次短程精修:官方配置先以 480×832、40 步生成,再以 1088×1920、8 步精修。加载时把分片通配符从 `transformer/` 换成 `refiner/`,将基础阶段的视频以目标分辨率通过 `input_video` 传回,并设置 `t_thresh`: + +```python +input_video = VideoData("video_base.mp4", height=1088, width=1920) +video = pipe( + prompt=caption, + negative_prompt=pipe.default_negative_prompt, + input_video=input_video, + height=1088, width=1920, num_frames=81, + num_inference_steps=8, cfg_scale=3.0, + t_thresh=0.85, sigma_tail_steps=2, + seed=0, +) +``` + +放大后的视频会被 VAE 编码并重新加噪到 `sigma=t_thresh`,因此这一阶段保留基础阶段的结构,并在目标分辨率上重新生成细节。请使用与基础阶段相同的 caption,并保持相同的宽高比。精修分辨率决定了主要开销——1088×1920 的序列长度约为 480×832 的 5 倍——建议开启显存管理运行该阶段。 + ### 提示词改写 LingBot-Video 训练时使用的是**结构化 JSON caption**,直接喂平铺句子属于分布外输入,会明显降低生成质量。Pipeline 接受 `dict` 形式的 caption(与训练一致的格式)或纯字符串,`dict` 会被内部归一化。 diff --git a/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_t2v_refiner.py b/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_t2v_refiner.py new file mode 100644 index 00000000..8df03eb5 --- /dev/null +++ b/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_t2v_refiner.py @@ -0,0 +1,49 @@ +import torch +import json +from diffsynth.utils.data import save_video, VideoData +from diffsynth.pipelines.lingbot_video import LingBotVideoPipeline, ModelConfig +from modelscope import dataset_snapshot_download + +vram_config = { + "offload_dtype": torch.bfloat16, + "offload_device": "cpu", + "onload_dtype": torch.bfloat16, + "onload_device": "cpu", + "preparing_dtype": torch.bfloat16, + "preparing_device": "cuda", + "computation_dtype": torch.bfloat16, + "computation_device": "cuda", +} + +dataset_snapshot_download( + dataset_id="DiffSynth-Studio/diffsynth_example_dataset", + local_dir="data/diffsynth_example_dataset", + allow_file_pattern="lingbot_video/lingbot-video-dense-1.3b_t2v/*", +) +base = "data/diffsynth_example_dataset/lingbot_video/lingbot-video-dense-1.3b_t2v" +with open(f"{base}/t2v_example_1.json", "r", encoding="utf-8") as f: + caption = json.load(f) + +pipe = LingBotVideoPipeline.from_pretrained( + torch_dtype=torch.bfloat16, + device="cuda", + model_configs=[ + ModelConfig(model_id="Robbyant/lingbot-video-moe-30b-a3b", origin_file_pattern="refiner/diffusion_pytorch_model*.safetensors", **vram_config), + ModelConfig(model_id="Qwen/Qwen3-VL-4B-Instruct", origin_file_pattern="*.safetensors", **vram_config), + ModelConfig(model_id="Robbyant/lingbot-video-moe-30b-a3b", origin_file_pattern="vae/diffusion_pytorch_model.safetensors", **vram_config), + ], + processor_config=ModelConfig(model_id="Qwen/Qwen3-VL-4B-Instruct", origin_file_pattern=""), + vram_limit=torch.cuda.mem_get_info("cuda")[1] / (1024 ** 3) - 5, +) + +input_video = VideoData(f"{base}/video_1.mp4", height=1088, width=1920) +video = pipe( + prompt=caption, + negative_prompt=pipe.default_negative_prompt, + input_video=input_video, + height=1088, width=1920, num_frames=81, + num_inference_steps=8, cfg_scale=3.0, + t_thresh=0.85, sigma_tail_steps=2, + seed=0, +) +save_video(video, "video_lingbot-video-moe-30b-a3b_t2v_refined.mp4", fps=15, quality=10) diff --git a/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_ti2v_refiner.py b/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_ti2v_refiner.py new file mode 100644 index 00000000..dc1859bc --- /dev/null +++ b/examples/lingbot_video/model_inference/lingbot-video-moe-30b-a3b_ti2v_refiner.py @@ -0,0 +1,53 @@ +import os +import json +import torch +from PIL import Image +from diffsynth.utils.data import save_video, VideoData +from diffsynth.pipelines.lingbot_video import LingBotVideoPipeline, ModelConfig +from modelscope import dataset_snapshot_download + +vram_config = { + "offload_dtype": torch.bfloat16, + "offload_device": "cpu", + "onload_dtype": torch.bfloat16, + "onload_device": "cpu", + "preparing_dtype": torch.bfloat16, + "preparing_device": "cuda", + "computation_dtype": torch.bfloat16, + "computation_device": "cuda", +} + +dataset_snapshot_download( + dataset_id="DiffSynth-Studio/diffsynth_example_dataset", + local_dir="data/diffsynth_example_dataset", + allow_file_pattern="lingbot_video/lingbot-video-dense-1.3b_ti2v/*", +) +base = "data/diffsynth_example_dataset/lingbot_video/lingbot-video-dense-1.3b_ti2v" +with open(os.path.join(base, "ti2v_example.json"), "r", encoding="utf-8") as f: + caption = json.load(f) +input_image = Image.open(os.path.join(base, "ti2v_first_frame.png")).convert("RGB") + +pipe = LingBotVideoPipeline.from_pretrained( + torch_dtype=torch.bfloat16, + device="cuda", + model_configs=[ + ModelConfig(model_id="Robbyant/lingbot-video-moe-30b-a3b", origin_file_pattern="refiner/diffusion_pytorch_model*.safetensors", **vram_config), + ModelConfig(model_id="Qwen/Qwen3-VL-4B-Instruct", origin_file_pattern="*.safetensors", **vram_config), + ModelConfig(model_id="Robbyant/lingbot-video-moe-30b-a3b", origin_file_pattern="vae/diffusion_pytorch_model.safetensors", **vram_config), + ], + processor_config=ModelConfig(model_id="Qwen/Qwen3-VL-4B-Instruct", origin_file_pattern=""), + vram_limit=torch.cuda.mem_get_info("cuda")[1] / (1024 ** 3) - 5, +) + +input_video = VideoData(os.path.join(base, "video_1.mp4"), height=1088, width=1920) +video = pipe( + prompt=caption, + negative_prompt=pipe.default_negative_prompt, + input_image=input_image, + input_video=input_video, + height=1088, width=1920, num_frames=81, + num_inference_steps=8, cfg_scale=3.0, + t_thresh=0.85, sigma_tail_steps=2, + seed=0, +) +save_video(video, "video_lingbot-video-moe-30b-a3b_ti2v_refined.mp4", fps=15, quality=10) diff --git a/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_t2v_refiner.py b/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_t2v_refiner.py new file mode 100644 index 00000000..450bedfe --- /dev/null +++ b/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_t2v_refiner.py @@ -0,0 +1,49 @@ +import torch +import json +from diffsynth.utils.data import save_video, VideoData +from diffsynth.pipelines.lingbot_video import LingBotVideoPipeline, ModelConfig +from modelscope import dataset_snapshot_download + +vram_config = { + "offload_dtype": "disk", + "offload_device": "disk", + "onload_dtype": torch.float8_e4m3fn, + "onload_device": "cpu", + "preparing_dtype": torch.float8_e4m3fn, + "preparing_device": "cuda", + "computation_dtype": torch.bfloat16, + "computation_device": "cuda", +} + +dataset_snapshot_download( + dataset_id="DiffSynth-Studio/diffsynth_example_dataset", + local_dir="data/diffsynth_example_dataset", + allow_file_pattern="lingbot_video/lingbot-video-dense-1.3b_t2v/*", +) +base = "data/diffsynth_example_dataset/lingbot_video/lingbot-video-dense-1.3b_t2v" +with open(f"{base}/t2v_example_1.json", "r", encoding="utf-8") as f: + caption = json.load(f) + +pipe = LingBotVideoPipeline.from_pretrained( + torch_dtype=torch.bfloat16, + device="cuda", + model_configs=[ + ModelConfig(model_id="Robbyant/lingbot-video-moe-30b-a3b", origin_file_pattern="refiner/diffusion_pytorch_model*.safetensors", **vram_config), + ModelConfig(model_id="Qwen/Qwen3-VL-4B-Instruct", origin_file_pattern="*.safetensors", **vram_config), + ModelConfig(model_id="Robbyant/lingbot-video-moe-30b-a3b", origin_file_pattern="vae/diffusion_pytorch_model.safetensors", **vram_config), + ], + processor_config=ModelConfig(model_id="Qwen/Qwen3-VL-4B-Instruct", origin_file_pattern=""), + vram_limit=torch.cuda.mem_get_info("cuda")[1] / (1024 ** 3) - 5, +) + +input_video = VideoData(f"{base}/video_1.mp4", height=1088, width=1920) +video = pipe( + prompt=caption, + negative_prompt=pipe.default_negative_prompt, + input_video=input_video, + height=1088, width=1920, num_frames=81, + num_inference_steps=8, cfg_scale=3.0, + t_thresh=0.85, sigma_tail_steps=2, + seed=0, +) +save_video(video, "video_lingbot-video-moe-30b-a3b_t2v_refined.mp4", fps=15, quality=10) diff --git a/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_ti2v_refiner.py b/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_ti2v_refiner.py new file mode 100644 index 00000000..0323922f --- /dev/null +++ b/examples/lingbot_video/model_inference_low_vram/lingbot-video-moe-30b-a3b_ti2v_refiner.py @@ -0,0 +1,53 @@ +import os +import json +import torch +from PIL import Image +from diffsynth.utils.data import save_video, VideoData +from diffsynth.pipelines.lingbot_video import LingBotVideoPipeline, ModelConfig +from modelscope import dataset_snapshot_download + +vram_config = { + "offload_dtype": "disk", + "offload_device": "disk", + "onload_dtype": torch.float8_e4m3fn, + "onload_device": "cpu", + "preparing_dtype": torch.float8_e4m3fn, + "preparing_device": "cuda", + "computation_dtype": torch.bfloat16, + "computation_device": "cuda", +} + +dataset_snapshot_download( + dataset_id="DiffSynth-Studio/diffsynth_example_dataset", + local_dir="data/diffsynth_example_dataset", + allow_file_pattern="lingbot_video/lingbot-video-dense-1.3b_ti2v/*", +) +base = "data/diffsynth_example_dataset/lingbot_video/lingbot-video-dense-1.3b_ti2v" +with open(os.path.join(base, "ti2v_example.json"), "r", encoding="utf-8") as f: + caption = json.load(f) +input_image = Image.open(os.path.join(base, "ti2v_first_frame.png")).convert("RGB") + +pipe = LingBotVideoPipeline.from_pretrained( + torch_dtype=torch.bfloat16, + device="cuda", + model_configs=[ + ModelConfig(model_id="Robbyant/lingbot-video-moe-30b-a3b", origin_file_pattern="refiner/diffusion_pytorch_model*.safetensors", **vram_config), + ModelConfig(model_id="Qwen/Qwen3-VL-4B-Instruct", origin_file_pattern="*.safetensors", **vram_config), + ModelConfig(model_id="Robbyant/lingbot-video-moe-30b-a3b", origin_file_pattern="vae/diffusion_pytorch_model.safetensors", **vram_config), + ], + processor_config=ModelConfig(model_id="Qwen/Qwen3-VL-4B-Instruct", origin_file_pattern=""), + vram_limit=torch.cuda.mem_get_info("cuda")[1] / (1024 ** 3) - 5, +) + +input_video = VideoData(os.path.join(base, "video_1.mp4"), height=1088, width=1920) +video = pipe( + prompt=caption, + negative_prompt=pipe.default_negative_prompt, + input_image=input_image, + input_video=input_video, + height=1088, width=1920, num_frames=81, + num_inference_steps=8, cfg_scale=3.0, + t_thresh=0.85, sigma_tail_steps=2, + seed=0, +) +save_video(video, "video_lingbot-video-moe-30b-a3b_ti2v_refined.mp4", fps=15, quality=10)