From 005ce0e70b2c17b91f6a331a29f38c16ccb1b539 Mon Sep 17 00:00:00 2001 From: Bowen Xue Date: Tue, 6 Oct 2026 23:16:52 +0000 Subject: [PATCH] Use a faster int8 model on GeForce and RTX 30-series cards, with a one-click upgrade (0.3.0) GeForce RTX 40/50 and RTX 30-series cards run the prepared ConvRot int8 model: activations get the cache's group-256 rotation and one scale per row, and a Triton GEMM multiplies them with the int8 weight rows. Workstation and datacenter cards keep FP8. Setup checks the int8 kernels on the GPU. On an RTX 3090 a 5 s 1344x768 Light request went from 434 s to 188 s. Existing installations keep their model until the user upgrades from the launcher: the new model downloads beside the current one, the switch runs between videos after the launcher stops its own ComfyUI and setup checks the int8 kernels, and then the retired variant is removed file by file with a receipt. Leftovers are offered again and removed only after confirmation. Release 0.3.0: release notes, README News and the planning docs. --- README.md | 3 +- README.zh-CN.md | 3 +- docs/execution-planning.md | 6 +- docs/zh-CN/execution-planning.md | 6 +- freevideo_engine/backends/cuda.py | 7 +- freevideo_engine/bootstrap.py | 225 +++++++++-- freevideo_engine/comfy_launcher_runtime.py | 15 +- freevideo_engine/comfy_setup.py | 2 + freevideo_engine/diagnostic_summary.py | 4 +- freevideo_engine/doctor.py | 3 +- freevideo_engine/fp8_ops.py | 3 +- freevideo_engine/generate.py | 8 +- freevideo_engine/int8_ops.py | 189 +++++++++ freevideo_engine/kernel_capabilities.py | 13 +- freevideo_engine/launcher/Main.qml | 130 ++++++ freevideo_engine/launcher_copy.py | 1 + freevideo_engine/launcher_session.py | 174 +++++++- freevideo_engine/lora_cache.py | 3 + freevideo_engine/lora_online_cache.py | 7 +- freevideo_engine/model_upgrade.py | 276 +++++++++++++ freevideo_engine/modern_launcher.py | 12 + freevideo_engine/policy.py | 41 +- freevideo_engine/prepared_model.py | 17 + freevideo_engine/probe.py | 20 + freevideo_engine/provision.py | 40 +- freevideo_engine/release_notes.json | 22 +- freevideo_engine/residual.py | 40 +- freevideo_engine/runtime.py | 2 +- freevideo_engine/system.py | 6 +- freevideo_engine/tuning.py | 2 +- freevideo_engine/two_pass.py | 1 + freevideo_engine/variant_cleanup.py | 439 +++++++++++++++++++++ freevideo_engine/worker.py | 4 +- web/updates.js | 12 +- 34 files changed, 1635 insertions(+), 101 deletions(-) create mode 100644 freevideo_engine/int8_ops.py create mode 100644 freevideo_engine/model_upgrade.py create mode 100644 freevideo_engine/variant_cleanup.py diff --git a/README.md b/README.md index dce33b4..04501c3 100644 --- a/README.md +++ b/README.md @@ -19,6 +19,7 @@ https://github.com/user-attachments/assets/ecda7d0d-7fbe-4e0c-8c29-8f3315bafc15 ## News +- **2026-10-07** · **[v0.3.0](https://github.com/FlashML-org/FreeVideo/releases/tag/v0.3.0): a faster int8 model and prompt enhancement.** GeForce RTX 40/50 and RTX 30-series cards switch to an int8 model that is faster and closer to the original quality; an RTX 3090 makes a 5-second video about 2.3x faster. Existing installations upgrade in one click, and an optional local prompt enhancer rewrites prompts for MiniMax H3. - **2026-10-06** · **[Gallery](https://freevideo-community.pages.dev/#gallery) is live.** Watch 20-second clips made with FreeVideo, and the four quality levels side by side. - **2026-10-06** · **[v0.2.3](https://github.com/FlashML-org/FreeVideo/releases/tag/v0.2.3): videos carry their workflow.** Drop a FreeVideo video onto the ComfyUI canvas to restore its prompt, seed and settings. Earlier videos can get theirs too. - **2026-10-05** · **[v0.2.0](https://github.com/FlashML-org/FreeVideo/releases/tag/v0.2.0): four quality levels.** Choose Light, Medium, High or Max for each video; higher levels give higher quality but take longer. Results can be exported as sharing images or videos with the generation time and GPU. @@ -32,7 +33,7 @@ FreeVideo is a local inference engine for MiniMax H3 on consumer GPUs, built on It coordinates VRAM, system memory and disk, adapting weight placement, compute precision and attention kernels to the available hardware. FreeVideo runs as a ComfyUI plugin, with a Windows launcher for setup and command-line support on Linux. Its core features include: -- **Hardware-adaptive execution**: Chooses the FP8 compute path for each GPU architecture, either native FP8 or FP8 storage with BF16 compute, and automatically probes the available attention kernels. +- **Hardware-adaptive execution**: Chooses the weight format for each GPU, int8 on GeForce and RTX 30-series cards and FP8 on workstation and datacenter cards, and automatically probes the available attention kernels. - **Low-memory inference**: Weight streaming, asynchronous prefetching and chunked computation keep peak memory low, enabling inference with as little as 8 GB of VRAM and 16 GB of RAM. - **Multimodal inputs**: Text prompts, first and last frames, and image, video and audio references. - **Community LoRAs**: Use MiniMax H3 LoRAs in your workflow. See [examples](docs/LoRA.md). diff --git a/README.zh-CN.md b/README.zh-CN.md index f7a68cb..5b7c285 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -19,6 +19,7 @@ https://github.com/user-attachments/assets/ecda7d0d-7fbe-4e0c-8c29-8f3315bafc15 ## 更新动态 +- **2026-10-07** · **[v0.3.0](https://github.com/FlashML-org/FreeVideo/releases/tag/v0.3.0):更快的 int8 模型和提示词增强。** GeForce RTX 40/50 系和 RTX 30 系显卡改用更快、画质更接近原版的 int8 模型,RTX 3090 生成 5 秒视频快约 2.3 倍;已安装的用户可一键升级。新增可选的本地提示词增强,把提示词改写成适合 MiniMax H3 的写法。 - **2026-10-06** · **[作品展示](https://freevideo-community.pages.dev/#gallery)上线。** 观看 FreeVideo 生成的 20 秒视频,以及四档质量的并排对比。 - **2026-10-06** · **[v0.2.3](https://github.com/FlashML-org/FreeVideo/releases/tag/v0.2.3):视频自带工作流。** 把 FreeVideo 生成的视频拖到 ComfyUI 画布上,即可还原提示词、种子和全部参数,之前的视频也能补上。 - **2026-10-05** · **[v0.2.0](https://github.com/FlashML-org/FreeVideo/releases/tag/v0.2.0):四档质量。** 每次生成可选择轻量、标准、精细或极致,档位越高,生成质量越高,但耗时更长;生成结果可导出为带生成耗时和显卡信息的分享图片或视频。 @@ -32,7 +33,7 @@ FreeVideo 是面向消费级显卡的 MiniMax H3 本地推理引擎,基于 [Op 它统一调度显存、内存与磁盘,并根据硬件条件调整权重放置、计算精度和注意力内核。FreeVideo 以 ComfyUI 插件形式提供,配备 Windows 启动器,也支持 Linux 命令行。主要特性包括: -- **硬件自适应**:针对不同显卡架构选择 FP8 计算路径(原生 FP8,或 FP8 存储配合 BF16 计算),并自动探测可用的注意力内核,无需手动配置。 +- **硬件自适应**:针对不同显卡选择权重格式(GeForce 和 RTX 30 系用 int8,专业卡和数据中心卡用 FP8),并自动探测可用的注意力内核,无需手动配置。 - **低显存推理**:通过权重流式加载、异步预取与分块计算降低峰值显存,最低只需 8GB 显存和 16GB 内存。 - **多模态输入**:支持文本、首帧、尾帧,以及图像、视频、音频参考输入。 - **社区 LoRA**:支持在工作流中使用 MiniMax H3 社区 LoRA。[查看效果对比](docs/LoRA.zh-CN.md)。 diff --git a/docs/execution-planning.md b/docs/execution-planning.md index 737fb3d..dde633d 100644 --- a/docs/execution-planning.md +++ b/docs/execution-planning.md @@ -1,12 +1,12 @@ # FreeVideo Adaptive Execution Planner -The FreeVideo Adaptive Execution Planner computes an execution plan for every request from live device and host measurements. The plan fixes the residency of the 50 transformer blocks across VRAM, pinned host memory and disk, the transfer schedule, the FP8 GEMM path, the attention backend and head grouping, activation staging and the VAE decoder placement, within the measured VRAM and host-memory budgets. +The FreeVideo Adaptive Execution Planner computes an execution plan for every request from live device and host measurements. The plan fixes the residency of the 50 transformer blocks across VRAM, pinned host memory and disk, the transfer schedule, the int8 or FP8 GEMM path, the attention backend and head grouping, activation staging and the VAE decoder placement, within the measured VRAM and host-memory budgets. ## Inputs | Input | Source | Determines | | --- | --- | --- | -| Architecture and compute capability | CUDA device properties | FP8 GEMM path and kernel set | +| Architecture and compute capability | CUDA device properties | Weight format, GEMM path and kernel set | | Free VRAM | `torch.cuda.mem_get_info` | VRAM budget | | Available host memory | OS memory counters, cgroup v1/v2 limit, Windows commit headroom | Host-memory budget | | Attention backends | On-device kernel probes | Backend selection | @@ -23,7 +23,7 @@ VRAM budget: free VRAM minus a reserve of 2.5% of free VRAM, clamped to 0.5–1 | Transfer schedule | Two transfer slots (prefetch) at a VRAM budget of 14 GiB or more, with lower thresholds on Windows Blackwell and Ampere; one slot otherwise. | | Attention | The first backend that passes its probe, in the order SageAttention 2, PyTorch flash attention, cuDNN, FlashAttention 2, FlashAttention 4. Head group: the widest of 16, 8 and 4 heads whose measured activation footprint at the request's token count fits the VRAM budget. | | Activation staging | Below a 10 GiB VRAM budget, the residual stream and attention outputs are staged in host buffers when the on-device activation path does not fit. | -| FP8 GEMM | Per-tensor scales on Blackwell (SM120), per-channel scales on Ada (SM89) and Hopper (SM90), FP8 weights with BF16 compute on Ampere (SM80, SM86). | +| GEMM | int8 on GeForce Ada and Blackwell and on every Ampere card: activations get the cache's ConvRot rotation and one int8 scale per row. Other cards use FP8: per-tensor scales on Blackwell (SM120), per-channel scales on Ada (SM89) and Hopper (SM90). | | Chunking | Feed-forward chunk 2048, projection chunk 1024, window batch 4 (1 on Ampere). | | Text encoder | Runs in a separate process that exits before the transformer is loaded. | | VAE decoder | Below a 20 GiB VRAM budget, ⌊(VRAM budget − 3.25 GiB) / 268.6 MB⌋ of its 36 blocks stay resident and the rest are streamed; temporal clips are decoded in sequence with the original tiles and blending. | diff --git a/docs/zh-CN/execution-planning.md b/docs/zh-CN/execution-planning.md index 6643a6d..cd1cf4b 100644 --- a/docs/zh-CN/execution-planning.md +++ b/docs/zh-CN/execution-planning.md @@ -1,12 +1,12 @@ # FreeVideo Adaptive Execution Planner -FreeVideo Adaptive Execution Planner(自适应执行规划器)根据设备与主机的实时测量结果,为每个请求计算执行计划。执行计划在实测的显存与内存预算内确定:50 个 Transformer 块在显存、锁页内存与磁盘之间的驻留位置,传输调度,FP8 GEMM 路径,注意力后端与头分组,激活暂存,以及 VAE 解码器的放置。 +FreeVideo Adaptive Execution Planner(自适应执行规划器)根据设备与主机的实时测量结果,为每个请求计算执行计划。执行计划在实测的显存与内存预算内确定:50 个 Transformer 块在显存、锁页内存与磁盘之间的驻留位置,传输调度,int8 或 FP8 GEMM 路径,注意力后端与头分组,激活暂存,以及 VAE 解码器的放置。 ## 输入 | 输入 | 来源 | 决定 | | --- | --- | --- | -| 架构与计算能力 | CUDA 设备属性 | FP8 GEMM 路径与内核集 | +| 架构与计算能力 | CUDA 设备属性 | 权重格式、GEMM 路径与内核集 | | 空闲显存 | `torch.cuda.mem_get_info` | 显存预算 | | 可用内存 | 系统内存计数、cgroup v1/v2 上限、Windows 提交余量 | 内存预算 | | 注意力后端 | 设备上的内核探测 | 后端选择 | @@ -23,7 +23,7 @@ FreeVideo Adaptive Execution Planner(自适应执行规划器)根据设备 | 传输调度 | 显存预算不低于 14 GiB 时使用两个传输槽(预取),Windows Blackwell 与 Ampere 的门槛更低;否则使用一个传输槽。 | | 注意力 | 按 SageAttention 2、PyTorch flash attention、cuDNN、FlashAttention 2、FlashAttention 4 的顺序选择第一个通过探测的后端。头分组:在 16、8、4 头中选择该请求 token 数下实测激活占用不超过显存预算的最宽分组。 | | 激活暂存 | 显存预算低于 10 GiB 且设备上的激活路径放不下时,把残差流和注意力输出暂存在主机缓冲区。 | -| FP8 GEMM | Blackwell(SM120)逐张量缩放,Ada(SM89)与 Hopper(SM90)逐通道缩放,Ampere(SM80、SM86)以 FP8 存储权重、BF16 计算。 | +| GEMM | GeForce Ada、Blackwell 与所有 Ampere 显卡使用 int8:激活按缓存的 ConvRot 旋转后逐行 int8 量化。其他显卡使用 FP8:Blackwell(SM120)逐张量缩放,Ada(SM89)与 Hopper(SM90)逐通道缩放。 | | 分块 | 前馈分块 2048,投影分块 1024,窗口批次 4(Ampere 为 1)。 | | 文本编码器 | 在独立进程中运行,Transformer 加载前退出。 | | VAE 解码器 | 显存预算低于 20 GiB 时,36 个块中 ⌊(显存预算 − 3.25 GiB) / 268.6 MB⌋ 个常驻,其余流式加载;按时间片段依次解码,分块与融合方式与原解码器相同。 | diff --git a/freevideo_engine/backends/cuda.py b/freevideo_engine/backends/cuda.py index a3ff72d..841530f 100644 --- a/freevideo_engine/backends/cuda.py +++ b/freevideo_engine/backends/cuda.py @@ -6,7 +6,7 @@ class CUDABackend(DeviceBackend): device = 'cuda' capabilities = BackendCapabilities( name='cuda', memory_model='dedicated', - linear_policies=('native-fp8', 'bf16-weight-only'), + linear_policies=('native-fp8', 'bf16-weight-only', 'int8'), attention_candidates=('cudnn', 'torch-flash', 'sage2', 'fa2', 'fa4'), pinned_host_weights=True, streamed_weights=True) @@ -85,6 +85,11 @@ def prepare_linears(self, model, manifest, linear_compute, fp8_gemm): if linear_compute == 'native-fp8': from ..fp8_gemm import install actual_fp8_gemm = install(model, manifest['scale_granularity'], requested=fp8_gemm) + elif precision == 'int8': + if linear_compute != 'int8': + raise ValueError('Int8 weights on CUDA run with the int8 linear policy') + from ..int8_ops import install_cached_linears as install_int8_linears + install_int8_linears(model, manifest['linears'], manifest.get('rotation')) elif precision != 'bf16': raise ValueError('Unknown prepared weight precision') return actual_fp8_gemm diff --git a/freevideo_engine/bootstrap.py b/freevideo_engine/bootstrap.py index eb37233..4fa40ab 100644 --- a/freevideo_engine/bootstrap.py +++ b/freevideo_engine/bootstrap.py @@ -233,6 +233,43 @@ def model_target(row, model_dir, encoder_dir, prepared_dir=None): return (model_dir if row['repo'].startswith('OpenVDN/') else encoder_dir) / row['file'] +def resolve_prepared_format(requested, saved, hardware): + """The prepared model format of this setup run: 'int8_convrot' or 'fp8'. + + An explicit choice wins. Otherwise an installation keeps the format it + recorded, and one from before the choice existed keeps its FP8 model, so an + update never starts a model download by itself. A fresh installation takes + the faster format for its GPU. + """ + choices = {'int8': 'int8_convrot', 'int8_convrot': 'int8_convrot', 'fp8': 'fp8'} + if requested in choices: + return choices[requested] + if requested not in (None, 'auto'): + raise ValueError('Unknown prepared model format: ' + str(requested)) + recorded = saved.get('prepared_format') + if recorded in ('int8_convrot', 'fp8'): + return recorded + if saved.get('cache'): + return 'fp8' + from .prepared_model import preferred_format + return preferred_format(hardware) or 'fp8' + + +def require_int8_kernels(plan, kernel_report): + """Stop a setup that installs the int8 model unless its GPU kernels passed on this machine. + + machine.json is written only after every step, so the installation keeps + the model it had. + """ + if plan.get('prepared_format') != 'int8_convrot' or plan.get('inventory', {}).get('hardware', {}).get('system') == 'Darwin': + return + from .kernel_capabilities import readiness + rows = json.loads(Path(kernel_report).read_text(encoding='utf-8')).get('kernel_probes', []) + if not readiness(rows).get('int8_ready'): + raise RuntimeError('The int8 model needs int8 GPU kernels, and they did not pass on this GPU. The ' + 'installation keeps its current model. Details: ' + str(kernel_report)) + + def plan(args, *, local_progress=None): root = args.root.expanduser().resolve() saved = json.loads((root / 'machine.json').read_text(encoding='utf-8')) if (root / 'machine.json').is_file() else {} @@ -253,6 +290,7 @@ def plan(args, *, local_progress=None): snapshot, resources, errors = preflight(args, fixture, ram_gib=ram_gib, vram_gib=vram_gib) system, windows_target, layout = 'Darwin', False, 'unified' cache_format = dict(capability=None, scale_granularity='int8_convrot') + prepared_format = 'int8_convrot' else: if not args.hardware_json and (platform.system() not in ('Linux', 'Windows') or platform.machine().lower() not in ('x86_64', 'amd64')): raise RuntimeError('One-click setup supports Linux x86_64 and native Windows x64.') @@ -287,6 +325,21 @@ def plan(args, *, local_progress=None): errors.append(str(error)) system = hardware.system cache_format = dict(capability=hardware.capability) + prepared_format = resolve_prepared_format(getattr(args, 'prepared_format', 'auto'), saved, hardware) + if (prepared_format == 'int8_convrot' and getattr(args, 'prepared_format', 'auto') in (None, 'auto') + and getattr(args, 'reuse_models', None) and not getattr(args, 'cache', None)): + # A fresh installation pointed at an existing FP8 model reuses it + # instead of downloading the int8 one; switching stays a choice. + from . import install_tuning + if (install_tuning.discover_prepared(args.reuse_models, hardware.capability, scale_granularity='int8_convrot') is None + and install_tuning.discover_prepared(args.reuse_models, hardware.capability) is not None): + prepared_format = 'fp8' + if getattr(args, 'cache', None) and getattr(args, 'prepared_format', 'auto') in (None, 'auto'): + # An explicitly named cache keeps its own format. + prepared_format = ('int8_convrot' if cache_compatible(args.cache, hardware.capability, scale_granularity='int8_convrot') + else 'fp8') + if prepared_format == 'int8_convrot': + cache_format['scale_granularity'] = 'int8_convrot' model_dir = Path(args.models or saved.get('model_root') or root / 'models' / 'vdn').expanduser().resolve() encoder_dir = Path(args.encoder_models or saved.get('encoder_model_root') or prior.get('encoder_dir') or root / 'models' / 'encoder').expanduser().resolve() reuse_cache = args.cache.expanduser().resolve() if args.cache else None @@ -436,6 +489,7 @@ def plan(args, *, local_progress=None): 'model_dir': str(model_dir), 'encoder_dir': str(encoder_dir), 'reuse_cache': str(reuse_cache) if reuse_cache else None, 'prepared_model': prepared, 'sampling_caches': sampling_caches, + 'prepared_format': prepared_format, 'model_scale_granularity': cache_format.get('scale_granularity'), 'model_source': 'prepared' if prepared else 'existing' if reuse_cache else 'source', 'model_source_reason': ('Pinned slim model, matching the existing GPU precision policy' if prepared else 'Reuse existing compatible cache' if reuse_cache else @@ -506,7 +560,8 @@ def display(value, ui=None, *, verbose=False): rows.append((label, '~%.1f GiB additional needed · %.1f GiB free' % (disk['needed_bytes']/GiB, disk['free_bytes']/GiB))) prepared = value.get('prepared_model') rows.append(('Storage', 'Download slim %s model · no local conversion' % prepared_precision(prepared) if prepared else - 'Reuse prepared FP8 cache' if value['reuse_cache'] else 'Prepare compact FP8 model')) + ('Reuse prepared %s cache' % ('ConvRot int8' if value.get('prepared_format') == 'int8_convrot' else 'FP8')) + if value['reuse_cache'] else 'Prepare compact FP8 model')) if prepared: rows.append(('Prepared model', prepared['repo'] + ' · ' + prepared['scale_granularity'] + (' · private, authorized HF token required' if prepared.get('private') else ''))) @@ -618,7 +673,7 @@ def confirmed(args, value, ask=input, ui=None): if not args.yes or not args.accept_model_license: raise ValueError('--approved-plan requires explicit plan/license acceptance.') reviewed = json.loads(Path(reviewed_path).read_text(encoding='utf-8')) - keys = ('root', 'engine_version', 'environment_layout', 'model_dir', 'encoder_dir', 'vram_gib', 'ram_gib', 'reuse_cache', 'storage', 'storage_cleanup', 'model_downloader', 'prepared_model', 'model_source', 'disk_mode', 'frontend', 'sampling_caches') + keys = ('root', 'engine_version', 'environment_layout', 'model_dir', 'encoder_dir', 'vram_gib', 'ram_gib', 'reuse_cache', 'storage', 'storage_cleanup', 'model_downloader', 'prepared_model', 'prepared_format', 'model_source', 'disk_mode', 'frontend', 'sampling_caches') changed = any(reviewed.get(key) != value.get(key) for key in keys) changed |= reviewed.get('device_backend') != value.get('device_backend') if value.get('device_backend') == 'mps': @@ -708,7 +763,7 @@ def setup_task(label): 'vdn-source': ('Download model code', 'Prepare model source files while runtime packages download'), 'library-check': ('Check text encoder', 'Verify that the text encoding library loads'), 'models': ('Download model weights', 'Download and verify the video model and text encoder'), - 'prepare': ('Optimize model storage', 'Prepare compact FP8 weights for your GPU'), + 'prepare': ('Optimize model storage', 'Prepare the model weights for your GPU'), 'kernels': ('Test GPU acceleration', 'Run small checks on your GPU'), 'storage': ('Finish model storage', 'Verify prepared weights and apply your storage choice'), 'dependency-check': ('Check installed packages', 'Verify dependency compatibility'), @@ -779,16 +834,13 @@ def initialize(self, value, ui): if previous.is_file(): shutil.copyfile(previous, self.run_dir / 'machine.before.json') saved = json.loads(previous.read_text(encoding='utf-8')) - # Persist a confirmed choice for interrupted first installs and upgrades. - # The former complete configuration is retained in machine.before.json. - save(previous, dict(saved, root=str(self.root), ready=False, setup_run=str(self.run_dir), - storage=value.get('storage', 'compact'), - disk_mode=value.get('disk_mode', 'normal'), - pending_environment_layout=self.layout, model_root=value['model_dir'], - encoder_model_root=value.get('encoder_dir', saved.get('encoder_model_root')), - wheel_cache=value.get('wheel_cache', saved.get('wheel_cache')), - vram_gib=value.get('vram_gib', saved.get('vram_gib')), - ram_gib=value.get('ram_gib', saved.get('ram_gib')))) + self.saved = saved + # Moving a working installation to int8 checks the int8 kernels first + # (execute), so a GPU that fails them keeps its configuration as it was. + self.int8_check_first = (value.get('prepared_format') == 'int8_convrot' and saved.get('ready') is True + and self.system != 'Darwin' and self.pythons['engine'].is_file()) + if not self.int8_check_first: + self.mark_unfinished() self.env = dict(os.environ, FREEVIDEO_HOME=str(self.root), FREEVIDEO_VDN_ROOT=str(self.root / 'vendor' / 'vdn'), FREEVIDEO_MODEL_ROOT=value['model_dir'], @@ -821,6 +873,21 @@ def initialize(self, value, ui): self.monitor_thread = None self.started = time.monotonic() + def mark_unfinished(self): + """Persist a confirmed choice for interrupted first installs and upgrades. + + The former complete configuration is retained in machine.before.json. + """ + value, saved = self.plan, self.saved + save(self.root / 'machine.json', dict(saved, root=str(self.root), ready=False, setup_run=str(self.run_dir), + storage=value.get('storage', 'compact'), + disk_mode=value.get('disk_mode', 'normal'), + pending_environment_layout=self.layout, model_root=value['model_dir'], + encoder_model_root=value.get('encoder_dir', saved.get('encoder_model_root')), + wheel_cache=value.get('wheel_cache', saved.get('wheel_cache')), + vram_gib=value.get('vram_gib', saved.get('vram_gib')), + ram_gib=value.get('ram_gib', saved.get('ram_gib')))) + def device_environment(self, env): from .triton_compat import environment return environment(self.root, dict(env, @@ -999,7 +1066,8 @@ def update_progress(*, final=False): with log.open('w', encoding='utf-8') as stream: with PackageOutput(row['command'], stream, env or self.env) as output: child = processes.popen(row['command'], env=output.env, cwd=cwd, stdout=output.stdout, stderr=subprocess.STDOUT, - start_new_session=True, pass_fds=(self.runtime_fd,), supervise=True) + start_new_session=True, supervise=True, + pass_fds=() if self.runtime_fd is None else (self.runtime_fd,)) output.spawned() try: while child.poll() is None: @@ -1307,6 +1375,30 @@ def install_toolkit(self, cuda): save(receipt, spec) return toolkit + def machine_configuration(self, python, encoder_python, comfy, prepared): + """The complete machine.json written once every setup step has passed.""" + return {'schema_version': 1, 'root': str(self.root), 'source': str(SOURCE), + 'storage': self.plan.get('storage', 'compact'), + 'disk_mode': self.plan.get('disk_mode', 'normal'), + 'environment_layout': self.layout, + 'system': self.system, 'engine_version': __version__, + 'git': self.env.get('FREEVIDEO_GIT') or shutil.which('git', path=self.env['PATH']), + 'platform_validation': 'Local dependency and small GPU probes passed; full consumer-GPU validation comes from user test reports.', + 'kernel_capabilities': str(self.root / 'kernel-capabilities.json'), + 'python': str(python), 'comfy_python': str(encoder_python), 'comfy_root': str(comfy), + 'vdn_root': str(self.root / 'vendor' / 'vdn'), 'model_root': self.plan['model_dir'], + 'encoder_model_root': self.plan['encoder_dir'], 'wheel_cache': self.plan['wheel_cache'], + 'base': str(Path(self.plan['model_dir']) / 'h3-base'), + 'checkpoint': str(Path(self.plan['model_dir']) / 'stage-dmd-step-250'), + 'cache': json.loads(prepared.read_text(encoding='utf-8'))['cache'], + 'encoder': self.spec['models']['encoder_file'].split('/')[-1], + 'model_paths': str(self.root / 'encoder-paths.yaml'), + **self.device_configuration(), + 'vram_gib': self.plan.get('vram_gib'), 'ram_gib': self.plan.get('ram_gib'), + 'model_revision': self.spec['models']['vdn_revision'], + 'prepared_format': self.plan.get('prepared_format', 'fp8'), + 'setup_run': str(self.run_dir), 'ready': True} + def device_configuration(self): return {'gpu_uuid': self.plan['inventory']['selected_gpu']['uuid']} @@ -1356,6 +1448,14 @@ def prepare_tools(self): return uv def execute(self): + if self.int8_check_first: + # The installed environment runs this source's probes (PYTHONPATH); + # the receipt is reused by the check after installation. + self.ui.phase('Verify installation on your GPU', 0, 8) + report = self.run_dir / 'kernel-capabilities.before.json' + self.check_kernels(self.pythons['engine'], report) + require_int8_kernels(self.plan, report) + self.mark_unfinished() self.ui.phase('Prepare download tools', 0, 8) uv = self.prepare_tools() self.env['FREEVIDEO_UV'] = str(uv) @@ -1383,30 +1483,12 @@ def scheduling(name, status, completed, count): self.command(name + '-freeze', [uv, 'pip', 'freeze', '--python', executable]) kernel_report = self.run_dir / 'kernel-capabilities.json' self.check_kernels(python, kernel_report) + require_int8_kernels(self.plan, kernel_report) self.command('storage', [python, '-m', 'freevideo_engine.provision', '--plan', self.run_dir / 'plan.json', '--cleanup', '--out', self.run_dir / 'storage.json']) self.record_kernel_validation(results, kernel_report) self.ui.phase('Finish setup', total - 1, total) - configuration = {'schema_version': 1, 'root': str(self.root), 'source': str(SOURCE), - 'storage': self.plan.get('storage', 'compact'), - 'disk_mode': self.plan.get('disk_mode', 'normal'), - 'environment_layout': self.layout, - 'system': self.system, 'engine_version': __version__, - 'git': self.env.get('FREEVIDEO_GIT') or shutil.which('git', path=self.env['PATH']), - 'platform_validation': 'Local dependency and small GPU probes passed; full consumer-GPU validation comes from user test reports.', - 'kernel_capabilities': str(self.root / 'kernel-capabilities.json'), - 'python': str(python), 'comfy_python': str(encoder_python), 'comfy_root': str(comfy), - 'vdn_root': str(self.root / 'vendor' / 'vdn'), 'model_root': self.plan['model_dir'], - 'encoder_model_root': self.plan['encoder_dir'], 'wheel_cache': self.plan['wheel_cache'], - 'base': str(Path(self.plan['model_dir']) / 'h3-base'), - 'checkpoint': str(Path(self.plan['model_dir']) / 'stage-dmd-step-250'), - 'cache': json.loads(prepared.read_text(encoding='utf-8'))['cache'], - 'encoder': self.spec['models']['encoder_file'].split('/')[-1], - 'model_paths': str(self.root / 'encoder-paths.yaml'), - **self.device_configuration(), - 'vram_gib': self.plan.get('vram_gib'), 'ram_gib': self.plan.get('ram_gib'), - 'model_revision': self.spec['models']['vdn_revision'], - 'setup_run': str(self.run_dir), 'ready': True} + configuration = self.machine_configuration(python, encoder_python, comfy, prepared) # Commit readiness only after all steps pass. Failed/rerun setup files # remain available, and no model/output cleanup runs automatically. self.monitor_stop.set() @@ -1418,6 +1500,71 @@ def scheduling(name, status, completed, count): return configuration +class Prefetcher: + """Download the prepared model an installation is switching to, beside the one in use. + + provision --prefetch holds the setup lease (no concurrent setup run) and + nothing else: a running request keeps the engine lease, so generation + continues on the installed model. machine.json is unchanged and nothing + is removed. Progress uses the installer's step reporting. + """ + command = Installer.command + + def __init__(self, value, ui=None): + self.plan = value + self.root = Path(value['root']) + self.system = value['inventory'].get('hardware', {}).get('system', platform.system()) + self.layout = value.get('environment_layout', 'unified') + self.pythons = role_pythons(self.root, self.layout, self.system) + self.runtime_fd = None + self.run_dir = self.root / 'setup-runs' / (time.strftime('%Y%m%dT%H%M%SZ', time.gmtime()) + '-prefetch-' + str(os.getpid())) + self.run_dir.mkdir(parents=True, exist_ok=False) + save(self.run_dir / 'plan.json', value) + env = dict(os.environ, FREEVIDEO_HOME=str(self.root), FREEVIDEO_MODEL_ROOT=value['model_dir'], + HF_HOME=str(self.root / 'downloads' / 'huggingface'), HF_HUB_DISABLE_TELEMETRY='1', + HF_HUB_OFFLINE='0', TRANSFORMERS_OFFLINE='0', PYTHONUNBUFFERED='1', PYTHONUTF8='1', + PYTHONIOENCODING='utf-8', PYTHONPATH=str(SOURCE), + FREEVIDEO_NETWORK_PLAN=str(self.run_dir / 'plan.json'), + FREEVIDEO_NETWORK_EVENTS=str(self.run_dir / 'network.jsonl')) + env.pop(LOCK_ENV, None) + self.env = network.proxy_environment(env) + self.state = {'status': 'running', 'steps': [], 'plan': value, 'prefetch': True} + self.state_lock = threading.Lock() + self.cancel = threading.Event() + self.ui = ui or TerminalUI('Model download', plain=True) + + def run(self): + out = self.run_dir / 'prefetch.json' + self.ui.phase('Download the faster model', 1, 2) + self.command('models', [self.pythons['engine'], '-m', 'freevideo_engine.provision', + '--plan', self.run_dir / 'plan.json', '--prefetch', '--out', out]) + result = json.loads(out.read_text(encoding='utf-8')) + self.state.update(status='complete' if result.get('complete') else 'incomplete', result=result) + save(self.run_dir / 'status.json', self.state) + self.ui.phase('Model downloaded' if result.get('complete') else 'Model download incomplete', 2, 2) + return result + + +def run_prefetch(value, ui): + try: + prefetcher = Prefetcher(value, ui) + ui.start(prefetcher.run_dir) + result = prefetcher.run() + except BaseException as error: + message = str(error) if not isinstance(error, KeyboardInterrupt) else 'Stopped' + ui.event('failure', error=message) + print('\nModel download stopped: %s\nThe installed model is unchanged and still in use; downloaded files ' + 'are kept for the next attempt.' % message, file=sys.stderr) + return 130 if isinstance(error, KeyboardInterrupt) else 1 + finally: + ui.close() + if not result.get('complete'): + print('\nSome model files are still missing; run the download again.', file=sys.stderr) + return 1 + print('\nThe model is downloaded and verified. Run setup with the same --prepared-format to switch to it.') + return 0 + + def main(argv=None): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument('--root', type=Path, default=Path(os.environ.get('FREEVIDEO_HOME', DEFAULT_ROOT))) @@ -1428,6 +1575,12 @@ def main(argv=None): parser.add_argument('--sampling-caches', action=argparse.BooleanOptionalAction, default=None, help='Install every quality level in advance (default for new installations; existing ones keep ' 'their choice). The default 8 + 3 refinement tables are always installed.') + parser.add_argument('--prepared-format', choices=('auto', 'int8', 'fp8'), default='auto', + help='auto: an installation keeps its recorded format and new installations use int8 on GeForce ' + 'and Ampere cards, FP8 elsewhere. int8/fp8 switch explicitly; files of the other format are kept') + parser.add_argument('--prefetch-only', action='store_true', + help='With --prepared-format: download and verify that model beside the installed one, which stays ' + 'in use. No other setup step runs and machine.json is unchanged') parser.add_argument('--model-source', choices=('prepared', 'source'), default='prepared', help='Default: download a pinned slim model matching the GPU format. source: explicitly download original weights and convert locally') parser.add_argument('--encoder-models', type=Path, help='Directory containing text_encoders/') @@ -1474,6 +1627,8 @@ def main(argv=None): parser.error('--auto-resources cannot be combined with --vram-gib or --ram-gib') if args.copy_existing_models and not (args.reuse_models or args.reuse_models_manifest): parser.error('--copy-existing-models requires --reuse-models or --reuse-models-manifest') + if args.prefetch_only and args.prepared_format == 'auto': + parser.error('--prefetch-only needs --prepared-format int8 or fp8') ui = TerminalUI(platform.system() + ' setup', plain=args.plain, no_color=args.no_color, verbose=args.verbose, show_location=args.verbose) try: @@ -1503,6 +1658,8 @@ def curl_progress(message): if not accepted: print('Cancelled. No engine or model installation was started.') return 0 + if args.prefetch_only: + return run_prefetch(value, ui) try: installer_class = Installer if value.get('device_backend') == 'mps': diff --git a/freevideo_engine/comfy_launcher_runtime.py b/freevideo_engine/comfy_launcher_runtime.py index a99487a..3dc5b1f 100644 --- a/freevideo_engine/comfy_launcher_runtime.py +++ b/freevideo_engine/comfy_launcher_runtime.py @@ -508,7 +508,7 @@ def _inspect(self, values): raise ValueError('Model folders must be a list of directory paths') extra = extra + ([values['models']] if values.get('models') else []) self.setup.inspect(dict(root=str(engine), extra_libraries=extra, copy=False, - sampling_caches=bool(values.get('sampling_caches')), + sampling_caches=bool(values.get('sampling_caches')), prepared_format=values.get('prepared_format'), frontend=dict(root=descriptor['root'], separate=descriptor['separate'], download=fresh))) row = self._wait_setup() self.state = dict(self.state, plan=row['plan']) @@ -530,6 +530,7 @@ def _install(self, accepted): validate_target(selected['root']) if not selected['ready']: self.stage('engine', label='Prepare FreeVideo') + self._stop_server_for_setup(selected['url']) current = self.setup.status() self.setup.install(dict(plan_id=current.get('plan_id'), accept_licenses=True)) self._wait_setup() @@ -555,6 +556,18 @@ def _install(self, accepted): self.state = dict(self.state, selection=dict(selected), deployed=deployed) self._connect() + def _stop_server_for_setup(self, url): + """Setup reinstalls engine packages that our ComfyUI's resident worker + keeps loaded after a video (Windows refuses to replace SageAttention's + _fused.pyd), and a model switch must not leave the old model in that + worker. Stop our own idle server; connecting after setup starts it + again. A running job is never stopped.""" + if not self.owns_server(): + return + if queue_busy(url): + raise RuntimeError('ComfyUI is running a job. Try again after it finishes; the job is left running.') + self.stop_owned_server() + def ensure_shortcut(self): from .desktop_shortcut import create_after_install self.state = dict(self.state, shortcut=create_after_install( diff --git a/freevideo_engine/comfy_setup.py b/freevideo_engine/comfy_setup.py index f907cd4..cdc3d3f 100644 --- a/freevideo_engine/comfy_setup.py +++ b/freevideo_engine/comfy_setup.py @@ -250,6 +250,8 @@ def inspect(self, value): arguments.append('--frontend-download') if isinstance(value.get('sampling_caches'), bool): arguments.append('--sampling-caches' if value['sampling_caches'] else '--no-sampling-caches') + if value.get('prepared_format') in ('int8', 'fp8'): + arguments += ['--prepared-format', value['prepared_format']] if value.get('copy'): arguments.append('--copy-existing-models') self.selection = dict(root=str(root), arguments=arguments) diff --git a/freevideo_engine/diagnostic_summary.py b/freevideo_engine/diagnostic_summary.py index a4330d2..659b360 100644 --- a/freevideo_engine/diagnostic_summary.py +++ b/freevideo_engine/diagnostic_summary.py @@ -73,12 +73,12 @@ def summary_report(report): 'inference_kernels','window_varlen','varlen_smooth_k','reuse_block_outputs','preload_host','pin_host_weights','cache_refined_text'): if type(config.get(k)) is bool: result['config'][k] = config[k] - for k, allowed in (('task', ('t2va','fl2va','ref2va')), ('linear_compute',('native-fp8','bf16-weight-only')), + for k, allowed in (('task', ('t2va','fl2va','ref2va')), ('linear_compute',('native-fp8','bf16-weight-only','int8')), ('adaln_mode', ('portable-model-asset','optional-model-asset','local-precompute','original-projections'))): if config.get(k) in allowed: result['config'][k] = config[k] for k, allowed in (('fp8_gemm', ('torch','scaled-mm-epilogue','triton')), - ('fp8_scale_granularity', ('per_tensor','rowwise')), ('precision', ('fp8','bf16','fp16'))): + ('fp8_scale_granularity', ('per_tensor','rowwise')), ('precision', ('fp8','int8','bf16','fp16'))): if config.get(k) in allowed: result['config'][k] = config[k] attention = config.get('attention', '') diff --git a/freevideo_engine/doctor.py b/freevideo_engine/doctor.py index 99559f7..2093987 100644 --- a/freevideo_engine/doctor.py +++ b/freevideo_engine/doctor.py @@ -25,7 +25,8 @@ def doctor(*, probe=False, hardware=None, backends=None): result['kernel_probes'] = [] with runtime_lock() as descriptor: env = dict(os.environ, **{LOCK_ENV: str(descriptor)}) - for backend in [*sorted(backends), 'linear']: + linear = ['linear'] + (['linear-int8'] if hardware.system in ('Windows', 'Linux') else []) + for backend in [*sorted(backends), *linear]: command = [sys.executable, '-m', 'freevideo_engine.probe', backend] try: child = processes.run(command, capture_output=True, text=True, timeout=180, diff --git a/freevideo_engine/fp8_ops.py b/freevideo_engine/fp8_ops.py index a6a6db5..50c52b1 100644 --- a/freevideo_engine/fp8_ops.py +++ b/freevideo_engine/fp8_ops.py @@ -63,7 +63,8 @@ def project_with_scale(module, value, scale=None): def sliced_projection(module, value, channels, quantized=None): from .weight_only import WeightOnlyLinear - if isinstance(module, WeightOnlyLinear): + from .int8_ops import Int8Linear + if isinstance(module, (WeightOnlyLinear, Int8Linear)): return apply_lora(module, value, module.project(value, channels), channels) if not isinstance(module, official.Fp8Linear): result = torch.nn.functional.linear(value, module.weight[channels], diff --git a/freevideo_engine/generate.py b/freevideo_engine/generate.py index 744b2c7..0b9a5e6 100644 --- a/freevideo_engine/generate.py +++ b/freevideo_engine/generate.py @@ -154,6 +154,10 @@ def automatic_profile(args, canvas, *, stage, evidence, descriptor=None, environ planning_canvas['task'] = args.task options.update(available_backends=available_backends(hardware, probe_missing=False), canvas=planning_canvas, demonstrated_ram_bytes=demonstrated_local_ram(hardware, args, canvas)) + cache = getattr(args, 'cache', None) + if cache is not None and (Path(cache) / 'manifest.json').is_file(): + manifest = json.loads((Path(cache) / 'manifest.json').read_text(encoding='utf-8')) + options['precision'] = 'int8' if manifest.get('precision') == 'int8' else 'fp8' if getattr(args, '_lora_max_block_bytes', 0): options['lora_max_block_bytes'] = args._lora_max_block_bytes if getattr(args, '_lora_root_bytes', 0): @@ -348,8 +352,8 @@ def _run(args): if args.out.suffix.lower() != '.mp4': raise ValueError('--out must name an MP4 file') manifest = json.loads((args.cache / 'manifest.json').read_text(encoding='utf-8')) - if manifest.get('precision') != 'fp8': - raise ValueError('Select the prepared official FP8 cache') + if manifest.get('precision') not in ('fp8', 'int8', 'bf16'): # BF16: the unquantized reference + raise ValueError('Select the prepared official FP8 or int8 cache') destination = args.out.resolve() if destination.exists(): raise FileExistsError(destination) diff --git a/freevideo_engine/int8_ops.py b/freevideo_engine/int8_ops.py new file mode 100644 index 0000000..85c0639 --- /dev/null +++ b/freevideo_engine/int8_ops.py @@ -0,0 +1,189 @@ +"""ConvRot int8 projections for CUDA (W8A8, int32 accumulation). + +Weights are the prepared int8 rows of a ConvRot cache: each row was rotated by +the group-256 regular Hadamard transform H256 = H16 (x) H16 before per-row +quantization, the convention of ComfyUI's int8 H3 checkpoints. Activations get +the same rotation and one scale per row here; the int8 product is exact in +int32 and the epilogue applies both scales. Because H256 is symmetric and +orthogonal, x W^T = (x H)(W H)^T. +""" +import torch +import triton +import triton.language as tl + +from .triton_compat import activate +activate() + +GROUP = 256 +_H16 = {} + + +def hadamard16(device): + """H4 (x) H4 / 4 with H4 = [[1,1,1,-1],[1,1,-1,1],[1,-1,1,1],[-1,1,1,1]] (FP32).""" + key = str(device) + if key not in _H16: + h4 = torch.tensor([[1, 1, 1, -1], [1, 1, -1, 1], [1, -1, 1, 1], [-1, 1, 1, 1]], dtype=torch.float32) + _H16[key] = (torch.kron(h4, h4) / 4).to(device).contiguous() + return _H16[key] + + +@triton.jit +def _rotate(x, h16, BM: tl.constexpr): + # Channel c = 16 a + b of a 256-wide group: y = H16 X H16 over (a, b). + z = tl.dot(tl.reshape(x, (BM * 16, 16)), h16, input_precision='ieee') + z = tl.permute(tl.reshape(z, (BM, 16, 16)), (0, 2, 1)) + z = tl.dot(tl.reshape(z, (BM * 16, 16)), h16, input_precision='ieee') + return tl.reshape(tl.permute(tl.reshape(z, (BM, 16, 16)), (0, 2, 1)), (BM, 256)) + + +@triton.jit +def _rotate_quantize(X, Q, S, H, M, K: tl.constexpr, SX, BM: tl.constexpr): + rows = tl.program_id(0).to(tl.int64) * BM + tl.arange(0, BM) + live = rows < M + lane = tl.arange(0, 256) + pair = tl.arange(0, 16) + h16 = tl.load(H + pair[:, None] * 16 + pair[None, :]) + largest = tl.zeros((BM,), tl.float32) + for group in range(K // 256): + x = tl.load(X + rows[:, None] * SX + group * 256 + lane[None, :], live[:, None], other=0.).to(tl.float32) + largest = tl.maximum(largest, tl.max(tl.abs(_rotate(x, h16, BM)), 1)) + scale = tl.maximum(largest, 1e-12) / 127. + tl.store(S + rows, scale, live) + for group in range(K // 256): + x = tl.load(X + rows[:, None] * SX + group * 256 + lane[None, :], live[:, None], other=0.).to(tl.float32) + value = tl.extra.cuda.libdevice.rint(tl.fdiv(_rotate(x, h16, BM), scale[:, None], ieee_rounding=True)) + value = tl.minimum(tl.maximum(value, -127.), 127.) + tl.store(Q + rows[:, None] * K + group * 256 + lane[None, :], value.to(tl.int8), live[:, None]) + + +def rotate_quantize(value): + """[..., K] BF16/FP16 -> (int8 [M, K], FP32 [M] row scales) after the ConvRot rotation.""" + rows = value.reshape(-1, value.shape[-1]) + if rows.stride(-1) != 1: + rows = rows.contiguous() + m, k = rows.shape + if k % GROUP: + raise ValueError('ConvRot activations need a multiple of 256 channels') + quantized = torch.empty((m, k), device=rows.device, dtype=torch.int8) + scales = torch.empty(m, device=rows.device, dtype=torch.float32) + block = 16 + _rotate_quantize[(triton.cdiv(m, block),)](rows, quantized, scales, hadamard16(rows.device), m, k, rows.stride(0), + BM=block, num_warps=4) + return quantized, scales + + +@triton.autotune(configs=[ + triton.Config({'BM': 128, 'BN': 128, 'BK': 128, 'GROUP_M': 8}, num_warps=8, num_stages=3), + triton.Config({'BM': 128, 'BN': 256, 'BK': 128, 'GROUP_M': 8}, num_warps=8, num_stages=3), + triton.Config({'BM': 256, 'BN': 128, 'BK': 128, 'GROUP_M': 8}, num_warps=8, num_stages=3), + triton.Config({'BM': 128, 'BN': 128, 'BK': 64, 'GROUP_M': 8}, num_warps=4, num_stages=4), + triton.Config({'BM': 64, 'BN': 128, 'BK': 128, 'GROUP_M': 8}, num_warps=4, num_stages=4), +], key=['ROWS', 'N', 'K'], cache_results=True) # tuned once per machine, kept in the Triton cache +@triton.jit +def _matmul(A, W, SA, SW, B, O, M, N, K: tl.constexpr, SAM, SWN, SOM, ROWS, + HAS_BIAS: tl.constexpr, BM: tl.constexpr, BN: tl.constexpr, BK: tl.constexpr, GROUP_M: tl.constexpr): + pid = tl.program_id(0) + blocks_m, blocks_n = tl.cdiv(M, BM), tl.cdiv(N, BN) + per_group = GROUP_M * blocks_n + first_m = (pid // per_group) * GROUP_M + size_m = tl.minimum(blocks_m - first_m, GROUP_M) + pid_m = first_m + (pid % per_group) % size_m + pid_n = (pid % per_group) // size_m + rows = pid_m.to(tl.int64) * BM + tl.arange(0, BM) + cols = pid_n.to(tl.int64) * BN + tl.arange(0, BN) + inner = tl.arange(0, BK) + a_ptrs = A + rows[:, None] * SAM + inner[None, :] + w_ptrs = W + cols[None, :] * SWN + inner[:, None] + acc = tl.zeros((BM, BN), dtype=tl.int32) + for _ in range(K // BK): + a = tl.load(a_ptrs, rows[:, None] < M, other=0) + w = tl.load(w_ptrs, cols[None, :] < N, other=0) + acc = tl.dot(a, w, acc, out_dtype=tl.int32) + a_ptrs += BK + w_ptrs += BK + sa = tl.load(SA + rows, rows < M, other=0.) + sw = tl.load(SW + cols, cols < N, other=0.) + out = acc.to(tl.float32) * sa[:, None] * sw[None, :] + if HAS_BIAS: + out += tl.load(B + cols, cols < N, other=0.).to(tl.float32)[None, :] + tl.store(O + rows[:, None] * SOM + cols[None, :], out.to(O.dtype.element_ty), + (rows[:, None] < M) & (cols[None, :] < N)) + + +def matmul(quantized, weight, weight_scale, bias=None, out_dtype=torch.bfloat16): + """(int8 [M, K], FP32 [M]) x int8 [N, K] rows with FP32 [N] scales -> out_dtype [M, N].""" + values, scales = quantized + m, k = values.shape + n, weight_k = weight.shape + if k != weight_k or k % 128: + raise ValueError('Int8 projection widths differ or are not multiples of 128') + if weight.stride(1) != 1 or values.stride(1) != 1: + raise ValueError('Int8 operands need unit column stride') + weight_scale = weight_scale.reshape(-1) + output = torch.empty((m, n), device=values.device, dtype=out_dtype) + # Tune once per power-of-two row band, not for every chunk and remainder row count. + rows = max(256, 1 << (m - 1).bit_length()) + _matmul[lambda meta: (triton.cdiv(m, meta['BM']) * triton.cdiv(n, meta['BN']),)]( + values, weight, scales, weight_scale, bias if bias is not None else weight_scale, output, + m, n, k, values.stride(0), weight.stride(0), output.stride(0), rows, bias is not None) + return output + + +class Int8Linear(torch.nn.Module): + """A prepared ConvRot int8 Linear: per-row int8 activations times int8 rows. + + The most recent quantized input is shared by every Int8Linear, so Q/K/V and + their head-chunk slices quantize the same activation once. The entry keeps + its source alive: a freed input's address could otherwise be reused by a + later activation of the same shape. + """ + _last = {} + + @property + def bias(self): + return self.original.bias + + def quantized(self, value): + rows = value.reshape(-1, value.shape[-1]) + key = (rows.data_ptr(), rows._version, tuple(rows.shape), tuple(rows.stride()), str(rows.device)) + last = Int8Linear._last + if last.get('key') != key: + last.clear() + last.update(key=key, source=rows, value=rotate_quantize(rows)) + return last['value'] + + def project(self, value, channels): + scales = self.weight_scale.reshape(-1)[channels] + output = matmul(self.quantized(value), self.weight_int8[channels], scales, + None if self.bias is None else self.bias[channels], out_dtype=value.dtype) + return output.reshape(*value.shape[:-1], output.shape[-1]) + + def forward(self, value): + output = matmul(self.quantized(value), self.weight_int8, self.weight_scale.reshape(-1), self.bias, + out_dtype=value.dtype) + return output.reshape(*value.shape[:-1], self.out_features) + + +def install_cached_linears(model, linears, rotation): + """Replace prepared int8 Linears with meta int8 placeholders for layer loading.""" + if rotation is None or rotation.get('kind') != 'convrot' or rotation.get('group') != GROUP: + raise ValueError('CUDA int8 projections need ConvRot group-256 weights') + for name, spec in linears.items(): + if spec.get('storage') != 'int8' or spec.get('convrot_group') != GROUP: + raise ValueError('Int8 cache entry is not a ConvRot int8 matrix: ' + name) + original = model.get_submodule(name) + if not isinstance(original, torch.nn.Linear): + raise ValueError('Int8 cache target is not a Linear: ' + name) + if list(original.weight.shape) != spec['weight_shape'] or original.in_features % GROUP: + raise ValueError('Int8 cache shape differs from the official model: ' + name) + replacement = Int8Linear.__new__(Int8Linear) + torch.nn.Module.__init__(replacement) + replacement.register_buffer('weight_int8', torch.empty(spec['weight_shape'], dtype=torch.int8, device='meta')) + replacement.register_buffer('weight_scale', torch.empty(spec['scale_shape'], dtype=torch.float32, device='meta')) + replacement.original = torch.nn.Module() + replacement.original.register_parameter('bias', original.bias) + replacement.in_features = original.in_features + replacement.out_features = original.out_features + replacement.input_dtype = getattr(torch, spec.get('input_dtype', 'bfloat16')) + prefix, _, child = name.rpartition('.') + setattr(model.get_submodule(prefix) if prefix else model, child, replacement) diff --git a/freevideo_engine/kernel_capabilities.py b/freevideo_engine/kernel_capabilities.py index c99d5e5..f9a0eb1 100644 --- a/freevideo_engine/kernel_capabilities.py +++ b/freevideo_engine/kernel_capabilities.py @@ -10,7 +10,7 @@ PACKAGES = ('torch', 'triton', 'triton-windows', 'diffusers', 'sageattention', 'flash-attn', 'flash-attn-4') COMPUTE_FILES = ('attention.py', 'backends/__init__.py', 'backends/base.py', 'backends/cuda.py', - 'backends/cuda_attention.py', 'probe.py', 'fp8_gemm.py', 'fp8_ops.py', 'weight_only.py', + 'backends/cuda_attention.py', 'probe.py', 'fp8_gemm.py', 'fp8_ops.py', 'int8_ops.py', 'weight_only.py', 'kernel_capabilities.py', 'doctor.py', 'fa4_guard.py', 'triton_compat.py', 'runtime.py', 'head_chunk.py', 'blocks.py', 'packing.py', 'dependencies.json') @@ -47,10 +47,14 @@ def receipt_path(): return data_root() / 'kernel-capabilities.json' +LINEAR_PROBES = ('linear', 'linear-int8') + + def readiness(rows): passed = {row['backend'] for row in rows if row.get('status') == 'complete'} - attention = sorted(passed - {'linear'}) + attention = sorted(passed - set(LINEAR_PROBES)) return dict(ready='linear' in passed and bool(attention), usable_attention_backends=attention, + int8_ready='linear-int8' in passed, failed_optional_backends=[row['backend'] for row in rows if row.get('status') != 'complete' and row['backend'] != 'linear']) @@ -70,8 +74,9 @@ def available_backends(hardware, *, discovered=None, probe_missing=True): if receipt['identity'] != expected or set(receipt['installed_attention_packages']) != discovered: receipt = None elif (not isinstance(receipt.get('kernel_probes'), list) - or {row.get('backend') for row in receipt['kernel_probes'] if isinstance(row, dict)} != discovered | {'linear'} - or len(receipt['kernel_probes']) != len(discovered) + 1 + or not discovered | {'linear'} <= {row.get('backend') for row in receipt['kernel_probes'] if isinstance(row, dict)} + <= discovered | set(LINEAR_PROBES) + or len(receipt['kernel_probes']) != len({row.get('backend') for row in receipt['kernel_probes'] if isinstance(row, dict)}) or any(not isinstance(row, dict) or row.get('status') not in ('complete', 'error') for row in receipt['kernel_probes'])): receipt = None diff --git a/freevideo_engine/launcher/Main.qml b/freevideo_engine/launcher/Main.qml index 017084a..3d90876 100644 --- a/freevideo_engine/launcher/Main.qml +++ b/freevideo_engine/launcher/Main.qml @@ -17,6 +17,7 @@ ApplicationWindow { property bool modelInfoOpen: false property bool closePending: false property bool cleanupConfirm: false + property bool releaseConfirm: false property bool accepted: false property bool manualUpdate: false property bool releaseNotesOpen: false @@ -47,6 +48,69 @@ ApplicationWindow { // An engine update needs no download; offer it as the way to launch. readonly property bool updateFirst: s.page === "launcher" && s.update.engine && !s.update.phase && !s.busy && s.status !== "open" && s.status !== "restart-required" && !s.needs_consent function percent(row) { return row && row.total ? Math.floor(100 * row.done / row.total) + "%" : "" } + readonly property var upgrade: s.model_upgrade || ({}) + readonly property bool upgradeShown: !upgrade.dismissed && ["available", "low-disk", "downloading", "downloaded", "switching", "releasing", "releasable", "complete", "failed"].indexOf(upgrade.status) >= 0 + function upgradeHeadline() { + var u = upgrade + if (u.status === "available") return t("A faster model is available for your GPU", "你的显卡有更快的模型可用") + if (u.status === "low-disk") return t("Not enough disk space for the faster model", "磁盘空间不够,暂时无法升级到更快的模型") + if (u.status === "downloading") return t("Downloading the faster model", "正在下载更快的模型") + if (u.status === "downloaded" && u.waiting === "comfy") return t("Switching after ComfyUI is closed", "关闭 ComfyUI 后切换") + if (u.status === "downloaded" || (u.status === "switching" && !s.busy)) return t("Switching after the current video", "当前视频完成后切换") + if (u.status === "switching") return t("Switching to the faster model", "正在切换到更快的模型") + if (u.status === "releasing") return u.waiting ? t("Removing the old model after the current video", "当前视频完成后删除旧模型") : t("Removing the old model", "正在删除旧模型") + if (u.status === "releasable") return u.in_use === "int8_convrot" ? t("The old model is still on disk", "旧模型还占着磁盘空间") + : t("Some model files are not in use", "有没在使用的模型文件") + if (u.status === "complete" && u.in_use) return u.in_use === "int8_convrot" ? t("The old model was removed", "旧模型已删除") + : t("Unused model files were removed", "没在使用的模型文件已删除") + if (u.status === "complete") return t("Upgraded to the faster model", "已升级到更快的模型") + if (u.status === "failed") return u.step === "release" ? t("The old model was not removed", "旧模型没有删除") : t("The upgrade stopped", "升级未完成") + return "" + } + function upgradeBlocker(code) { + if (code === "prepared-folder-redirected") return t("The model folder is linked to another location, and FreeVideo never deletes through a link.", "模型文件夹被链接到了别的位置,FreeVideo 不会通过链接删除文件。") + if (code === "unexpected-active-variant") return t("FreeVideo is not using the int8 model, so the old model stays.", "当前使用的不是 int8 模型,旧模型保留。") + if (code === "active-variant-incomplete") return t("Some int8 files are missing, so the old model stays.", "int8 模型文件不完整,旧模型保留。") + return t("The model in use is not a standard FreeVideo model, so nothing was deleted.", "当前使用的模型不是标准安装的模型,为安全起见没有删除任何文件。") + } + function upgradeExplanation() { + var u = upgrade + var lora = u.lora_variants && u.lora_variants.length + var speed = u.architecture === "ampere" ? t("about twice as fast on this GPU", "在这张显卡上预计快约 2 倍") + : t("about 10–30% faster on this GPU", "在这张显卡上预计快约 10–30%") + if (u.status === "available") + return t("The int8 model is ", "int8 模型") + speed + t(" and closer to the original quality. It downloads ", ",画质也更接近原版。需要下载约 ") + bytes(u.download_bytes) + + (lora ? t("; your old model stays for the fused LoRAs that int8 cannot merge yet.", ";你有需要合并的 LoRA,int8 暂不支持,旧模型会保留,不释放空间。") + : t(", then removes the old model and frees about ", ",完成后删除旧模型,释放约 ") + bytes(u.release_bytes) + t(".", "。")) + + t(" You can keep generating while it downloads.", "下载期间可以继续生成。") + if (u.status === "low-disk") + return t("It needs about ", "需要约 ") + bytes(u.required_bytes) + t(" free; ", " 可用空间,目前剩 ") + bytes(u.free_bytes) + + t(" is free. Clean the download cache in Settings → Storage or free some space, then reopen FreeVideo.", "。可以在 设置 → 存储 清理下载缓存,或者腾出空间后重新打开 FreeVideo。") + if (u.status === "downloading") + return u.progress && u.progress.unit === "bytes" && u.progress.total ? bytes(u.progress.done) + " / " + bytes(u.progress.total) + t(" · you can keep generating", " · 可以继续生成") + : t("Checking the download…", "正在检查下载内容…") + if (u.status === "downloaded" && u.waiting === "comfy") + return t("This ComfyUI was not opened by the launcher, so FreeVideo cannot restart it. Close it and the switch starts on its own; until then the current model keeps working.", + "这个 ComfyUI 不是启动器打开的,FreeVideo 没法重启它。关闭它后会自动开始切换,在此之前当前模型照常可用。") + if (u.status === "downloaded" || u.status === "switching") + return t("ComfyUI restarts once and generation pauses for about 2 minutes while FreeVideo switches; settings and videos are kept.", "切换约需 2 分钟,ComfyUI 会重启一次,期间暂停生成,设置和视频都会保留。") + if (u.status === "releasable" && u.in_use === "int8_convrot") + return t("FreeVideo now uses the int8 model. Removing the old model frees about ", "现在用的是 int8 模型,删除不再使用的旧模型可以释放约 ") + bytes(u.release_bytes) + + t("; only its own files are removed.", ",只删除旧模型自己的文件。") + if (u.status === "releasable") + return t("These model files are not in use, for example an int8 download this GPU could not run. Removing them frees about ", "这些模型文件没有在使用,比如这张显卡没能运行的 int8 下载。删除可以释放约 ") + bytes(u.release_bytes) + + t("; the model in use is not touched.", ",正在使用的模型不受影响。") + if (u.status === "releasing") return t("Only the old model's own files are removed.", "只删除旧模型自己的文件。") + if (u.status === "complete") + return (u.released_bytes ? t("Freed ", "已释放 ") + bytes(u.released_bytes) + t(".", "。") : "") + + (lora ? t(" The old model stays for your fused LoRAs.", "旧模型为需要合并的 LoRA 保留。") : "") + if (u.status === "failed") + return (u.blockers && u.blockers.length ? upgradeBlocker(u.blockers[0]) + t(" ", "") : u.error ? u.error.split("\n")[0] + "\n" : "") + + (u.step === "release" ? t("The int8 model is in use; the old files stay on disk for now.", "int8 模型已在使用,旧模型文件暂时保留在磁盘上。") + : u.step === "switch" && !u.kept ? t("The previous model's files are all kept. Retry on the setup page restores it; you can upgrade again afterwards.", "原来的模型文件都还在。在安装页面点「重试」即可恢复,之后可以再次升级。") + : t("Your previous model is unchanged and still works.", "原来的模型没有变化,可以继续使用。")) + return "" + } function updateHeadline() { var phase = s.update.phase if (phase === "downloading") return t("Downloading update ", "正在下载更新 ") + percent(s.update.progress) @@ -586,6 +650,38 @@ ApplicationWindow { FText { visible: !s.update.phase && !!text; text: releaseSummary(availableRelease); Layout.fillWidth: true; color: theme.muted; font.pixelSize: theme.micro } FButton { text: t("What's new", "更新内容"); flat: true; visible: !s.update.phase; implicitHeight: theme.heightSm; onClicked: releaseNotesOpen = true } } + FCard { + objectName: "modelUpgradeCard" + visible: upgradeShown + reveal: true; Layout.fillWidth: true; padding: 18; spacing: 10 + color: upgrade.status === "failed" || upgrade.status === "low-disk" ? theme.surface : theme.accentSubtle + border.color: upgrade.status === "failed" || upgrade.status === "low-disk" ? theme.border : theme.accentDim + RowLayout { + Layout.fillWidth: true; spacing: 14 + FIcon { kind: upgrade.status === "releasable" ? "folder" : "download"; ink: upgrade.status === "failed" ? theme.muted : theme.accent; Layout.preferredWidth: 22; Layout.preferredHeight: 22 } + ColumnLayout { + Layout.fillWidth: true; spacing: 2 + FText { objectName: "modelUpgradeHeadline"; text: upgradeHeadline(); font.weight: Font.DemiBold; Layout.fillWidth: true } + FText { objectName: "modelUpgradeText"; visible: text !== ""; text: upgradeExplanation(); color: theme.muted; font.pixelSize: theme.micro; Layout.fillWidth: true; wrapMode: Text.WordWrap } + } + FButton { objectName: "modelUpgradeLater"; visible: ["available", "low-disk", "releasable", "complete", "failed"].indexOf(upgrade.status) >= 0; flat: true + text: upgrade.status === "complete" || upgrade.status === "low-disk" ? t("Close", "关闭") : t("Later", "稍后"); onClicked: backend.dismissModelUpgrade() } + FButton { objectName: "modelUpgradeStop"; visible: upgrade.status === "downloading"; flat: true; text: t("Stop download", "停止下载"); onClicked: backend.cancelModelUpgrade() } + FButton { + objectName: "modelUpgradeButton"; primary: true + visible: upgrade.status === "available" || upgrade.status === "releasable" + || (upgrade.status === "failed" && (upgrade.step === "download" || upgrade.step === "release" || (upgrade.step === "switch" && upgrade.kept))) + enabled: !s.busy && !upgrade.busy + text: upgrade.status === "failed" ? t("Try again", "重试") : upgrade.status === "releasable" ? t("Remove…", "删除…") : t("Download & upgrade", "下载并升级") + onClicked: upgrade.status === "releasable" ? releaseConfirm = true : backend.upgradeModel() + } + } + FMeter { + // Waiting for the user to close ComfyUI is not progress. + visible: ["downloading", "downloaded", "switching", "releasing"].indexOf(upgrade.status) >= 0 && upgrade.waiting !== "comfy"; Layout.fillWidth: true; active: visible + fraction: upgrade.status === "downloading" && upgrade.progress && upgrade.progress.fraction !== null && upgrade.progress.fraction !== undefined ? upgrade.progress.fraction : -1 + } + } FCard { Layout.fillWidth: true; padding: 26; spacing: 18 RowLayout { @@ -828,6 +924,29 @@ ApplicationWindow { } } FText { visible: !!s.cleanup.error; text: s.cleanup.error || ""; color: theme.danger; font.pixelSize: theme.micro; Layout.fillWidth: true } + RowLayout { + objectName: "modelStorageRow" + visible: ["available", "low-disk", "downloading", "downloaded", "switching", "releasing", "releasable", "complete", "failed"].indexOf(upgrade.status) >= 0 + Layout.fillWidth: true; spacing: 12 + ColumnLayout { + Layout.fillWidth: true; spacing: 2 + FText { text: t("Video model", "视频模型"); Layout.fillWidth: true } + FText { + objectName: "modelStorageStatus"; Layout.fillWidth: true; font.pixelSize: theme.micro; color: theme.muted; wrapMode: Text.WordWrap + text: upgrade.status === "available" ? t("A faster int8 model is available: about ", "有更快的 int8 模型:下载约 ") + bytes(upgrade.download_bytes) + + (upgrade.release_bytes ? t(" to download, about ", ",完成后删除旧模型,释放约 ") + bytes(upgrade.release_bytes) + t(" freed afterwards.", "。") : t(" to download.", "。")) + : upgrade.status === "releasable" ? (upgrade.in_use === "int8_convrot" ? t("The old model is no longer used; about ", "旧模型已不再使用,删除可释放约 ") : t("Model files not in use; about ", "有没在使用的模型文件,删除可释放约 ")) + bytes(upgrade.release_bytes) + t(" can be freed.", "。") + : upgradeHeadline() + } + } + FButton { + objectName: "modelStorageButton"; visible: upgrade.status === "available" || upgrade.status === "releasable" + text: upgrade.status === "releasable" ? t("Remove…", "删除…") : t("Upgrade", "升级") + implicitHeight: theme.heightSm; font.pixelSize: theme.micro + 1 + enabled: !s.busy && !upgrade.busy + onClicked: upgrade.status === "releasable" ? releaseConfirm = true : backend.upgradeModel() + } + } } FGroup { Layout.fillWidth: true; title: t("Compatibility", "兼容性") @@ -974,6 +1093,17 @@ ApplicationWindow { onAccepted: { cleanupConfirm = false; backend.cleanupDownloads(true) } onRejected: cleanupConfirm = false } + FDialog { + objectName: "releaseConfirmation"; visible: releaseConfirm + title: upgrade.in_use === "int8_convrot" ? t("Remove the old model?", "删除旧模型?") : t("Remove unused model files?", "删除没在使用的模型文件?") + text: (upgrade.in_use === "int8_convrot" ? t("The FP8 model files FreeVideo no longer uses are removed, about ", "将删除 FreeVideo 不再使用的 FP8 模型文件,释放约 ") + : t("The model files FreeVideo does not use are removed, about ", "将删除 FreeVideo 没在使用的模型文件,释放约 ")) + bytes(upgrade.release_bytes || 0) + + (upgrade.in_use === "int8_convrot" ? t(". The int8 model, your videos and settings are not touched.", "。int8 模型、视频和设置都不受影响。") + : t(". The model in use, your videos and settings are not touched.", "。正在使用的模型、视频和设置都不受影响。")) + acceptText: t("Remove", "删除"); rejectText: t("Cancel", "取消"); destructive: true + onAccepted: { releaseConfirm = false; backend.upgradeModel() } + onRejected: releaseConfirm = false + } FDialog { visible: closePending title: t("Pause and close?", "暂停并退出?") diff --git a/freevideo_engine/launcher_copy.py b/freevideo_engine/launcher_copy.py index 9c58f6c..1fe3c89 100644 --- a/freevideo_engine/launcher_copy.py +++ b/freevideo_engine/launcher_copy.py @@ -66,6 +66,7 @@ ('Verify that the text encoding library loads', '检查文本编码器'), ('Download and verify the video model and text encoder', '下载并校验模型'), ('Prepare compact FP8 weights for your GPU', '准备适配显卡的模型'), + ('Prepare the model weights for your GPU', '准备适配显卡的模型'), ('Run small checks on your GPU', '检查 GPU 加速是否可用'), ('Verify prepared weights and apply your storage choice', '校验模型文件'), ('Verify dependency compatibility', '检查运行组件'), diff --git a/freevideo_engine/launcher_session.py b/freevideo_engine/launcher_session.py index f34b447..7037494 100644 --- a/freevideo_engine/launcher_session.py +++ b/freevideo_engine/launcher_session.py @@ -16,12 +16,14 @@ from .model_guidance import package_instructions, video_instructions, runtime_packages_supported from .setup_progress import progress_text from .terminal_ui import clean, duration +from .model_upgrade import TARGET # Background launcher release checks while the window stays open. CHECK_SECONDS = 30 * 60 # "Later" hides one reminder for this long; a newer release reminds at once. SNOOZE_SECONDS = 4 * 3600 QUEUE_POLL_SECONDS = 3 +RECEIPT_POLL_SECONDS = 5 class Session: @@ -32,7 +34,8 @@ class Session: update_waiting = update_restarting = False engine_updating = engine_autoinstall = reload_expected = resume_pages = False browser_wait = queue_state = queue_thread = bridge = None - clients_polled = queue_polled = resume_until = resume_polled = bridge_polled = 0. + foreign_state = foreign_thread = None + clients_polled = queue_polled = foreign_polled = receipt_polled = resume_until = resume_polled = bridge_polled = 0. bridge_pending = False bridge_written = (0., None) update_checked = 0. @@ -43,12 +46,15 @@ class Session: def __init__(self, source=None, *, controller=None, store=None, updater=None, smoke=False): from .offline_packages import Importer from .installation_cleanup import Cleaner + from .model_upgrade import ModelUpgrade self.cleaner = Cleaner() self.importer = Importer() self.imported_batch = None self.import_retry = None self.source = Path(source or materialize_source()) self.controller = controller or Controller(self.source) + self.upgrade = ModelUpgrade(self.source) + self.model_switch = None self.store = store or Store(launcher_root()) saved = self.store.read() home = Path(os.environ.get('USERPROFILE') or os.environ.get('HOME') or launcher_root().parent) @@ -221,6 +227,11 @@ def activate_offline(self, root): def action(self, name, accepted=False): if self.cleaner.busy: return + # The model download holds the setup lease; launching and opening + # FreeVideo continue, anything that would start setup waits for it. + if (self.upgrade.active or self.model_switch) and (name in ('retry', 'setup') or + (name == 'primary' and self.page != 'launcher')): + return if self.importer.busy: if name == 'stop': self.importer.cancelled.set() @@ -325,11 +336,127 @@ def retry_kind(self): return '' def cleanup_downloads(self, remove=False): - if (self.controller.busy or self.importer.busy or self.closing + if (self.controller.busy or self.importer.busy or self.closing or self.upgrade.active or self.model_switch or self.update_intent or self.engine_updating or self.updater and self.updater.busy): return self.cleaner.start(self.engine_root(), remove=remove) + def upgrade_model(self): + """The user's one consent: download int8 beside the current model, switch, then release the old one. + + After a failure only the step that failed runs again: a stopped download, + or a switch that failed before setup started, is offered afresh (files + already fetched count); a failed release retries the release. A switch + that failed during setup is retried from the setup page, which restores + the previous model. An old model left after the switch is released here + too, when the user asks. + """ + state = self.upgrade.state + if (state.get('status') not in ('available', 'failed', 'releasable') or self.upgrade.busy or self.model_switch + or self.closing or self.controller.busy or self.importer.busy or self.cleaner.busy or self.engine_updating): + return + root = self.engine_root() + # Started from Settings after "Later": show the card again. + self.upgrade.state = state = dict(state, dismissed=False) + if state.get('status') == 'releasable' or state.get('status') == 'failed' and state.get('step') == 'release': + # After the switch the int8 model must be in use; files offered on + # their own name the variant they keep. + self.upgrade.release(root, state.get('in_use') or TARGET) + return + if state.get('status') == 'failed': + # A switch that failed after setup started leaves the installation + # unfinished; Retry on the setup page restores it first. + if state.get('step') not in ('download', 'switch') or state.get('step') == 'switch' and not state.get('kept'): + return + from .model_upgrade import offer + state = offer(root) + self.upgrade.state = state + if state.get('status') != 'available': + return + from .desktop_runtime import materialize_source + self.upgrade.source = materialize_source(self.source, root / 'launcher' / 'source') + self.upgrade.download(root, state) + + def cancel_model_upgrade(self): + self.upgrade.cancel() + + def dismiss_model_upgrade(self): + # Hides the card for this session; Settings → Storage keeps the entry. + if not self.upgrade.busy and not self.model_switch: + self.upgrade.state = dict(self.upgrade.state, dismissed=True) + + def _tick_model_upgrade(self, row, busy): + upgrade = self.upgrade + state = upgrade.state + if self.closing or upgrade.busy: + return + if (state.get('status') == 'idle' and self.page == 'launcher' and not busy + and row.get('status') in ('open', 'ready')): + try: + upgrade.inspect(self.engine_root()) + except ValueError: + pass + return + if (state.get('status') == 'unavailable' and state.get('reason') == 'int8-kernels' + and self.page == 'launcher' and not busy): + # A failed int8 probe is final until the GPU checks run again (a + # setup or repair, for example after a driver update). + now = time.monotonic() + if now - self.receipt_polled >= RECEIPT_POLL_SECONDS: + self.receipt_polled = now + from .model_upgrade import kernel_receipt + try: + changed = kernel_receipt(self.engine_root()) != state.get('receipt') + except ValueError: + changed = False + if changed: + upgrade.inspect(self.engine_root()) + return + if state.get('status') == 'downloaded' and self.model_switch is None: + # Switch only between videos, as an engine update does. + if busy or self.engine_updating or self._queue_busy() is not False: + return + # The switch restarts our ComfyUI so its resident worker lets go of + # the old model and the GPU packages setup reinstalls. A ComfyUI + # this launcher did not start cannot be restarted: wait until the + # user closes it. + foreign = self._foreign_server() + if foreign is not False: + if foreign: + upgrade.waiting_for('comfy') + return + self.model_switch = 'inspect' + upgrade.switching() + self.controller.run('inspect', dict(self.form, token=self.token, repair=True, prepared_format='int8')) + self.page = 'progress' + self.model_groups = [] + self.started = time.monotonic() + return + if self.model_switch == 'inspect' and not busy and row.get('action') == 'inspect' and row.get('status') != 'running': + plan = row.get('plan') or {} + if (row.get('status') == 'review' and not row.get('errors') and plan.get('prepared_format') == 'int8_convrot' + and plan.get('model_download_bytes', 1 << 62) <= 64 * (1 << 20)): + # Everything was downloaded and verified beside the old model; + # the user agreed to this switch when starting the upgrade. + self.model_switch = 'install' + self.controller.run('install', True) + else: + # Only a plan was made; machine.json is unchanged. + self.model_switch = None + upgrade.switch_failed('\n'.join(row.get('errors') or []) or row.get('error') or + self.t('The switch needs more files than were downloaded; check again.', + '切换还需要未下载的文件,请重新检查。'), kept=True) + return + if self.model_switch == 'install' and not busy and row.get('action') == 'install' and row.get('status') != 'running': + self.model_switch = None + if row.get('status') in ('open', 'restart-required') or row.get('deployed'): + upgrade.release(self.engine_root()) + else: + # Setup marks the installation unfinished when it starts; then + # Retry on the setup page restores the previous model. + upgrade.switch_failed(row.get('error') or self.t('The switch stopped.', '切换未完成。'), + kept=self._installation_ready()) + def _browser_url(self, address): """Return the URL used by the browser, with the launcher locale hint.""" if not address or not self.language.startswith('zh'): @@ -518,6 +645,30 @@ def poll(): self.queue_thread.start() return self.queue_state + def _foreign_server(self): + """Whether a ComfyUI this launcher did not start answers at the address; None before known.""" + owns = getattr(self.controller, 'owns_server', None) + if callable(owns) and owns() is True: + return False + now = time.monotonic() + if (self.foreign_thread is None or not self.foreign_thread.is_alive()) and ( + self.foreign_state is None or now - self.foreign_polled >= QUEUE_POLL_SECONDS): + from .comfy_launcher_runtime import server_info + url = (self.controller.selection or self.selected or {}).get('url') or self.form['url'] + self.foreign_polled = now + def poll(): + # Off the UI thread, like the queue poll. + self.foreign_state = server_info(url).get('status') != 'offline' + self.foreign_thread = threading.Thread(target=poll, name='freevideo-server', daemon=True) + self.foreign_thread.start() + return self.foreign_state + + def _installation_ready(self): + try: + return json.loads((self.engine_root() / 'machine.json').read_text(encoding='utf-8')).get('ready') is True + except (OSError, ValueError, AttributeError): + return False + def _restart_when_idle(self): row = self.updater.state if row.get('status') != 'ready' or self.update_restarting: @@ -544,7 +695,7 @@ def _restart_when_idle(self): self.closing = True def _update_engine_when_idle(self): - if self.controller.busy or self.importer.busy: + if self.controller.busy or self.importer.busy or self.upgrade.active or self.model_switch: return busy = self._queue_busy() self.update_waiting = bool(busy) @@ -638,6 +789,15 @@ def update_phase(self): return 'review' return '' + def model_phase(self): + """For open pages: switching to the int8 model restarts our ComfyUI once.""" + if self.model_switch: + return 'switching' + state = self.upgrade.state + if state.get('status') == 'downloaded' and not state.get('waiting'): + return 'waiting' + return '' + def source_stamp(self, source): """(version, built_at) of an engine source; a source checkout has neither.""" if not source: @@ -686,7 +846,7 @@ def _write_bridge_status(self, force=False): view = self.update_view() from .release_notes import public_details candidate = view.get('candidate') or {} - value = dict(version=__version__, phase=view['phase'], status=view.get('status'), + value = dict(version=__version__, phase=view['phase'], model=self.model_phase(), status=view.get('status'), progress=view.get('progress'), manual=view['manual'], channel=view['channel'], track=view['track'], candidate=public_details(candidate) if candidate.get('version') else None, engine=dict(view['current_release'], pending=view['engine'], installed=view['installed']), @@ -712,6 +872,7 @@ def close(self): self.closing = True if self.updater: self.updater.cancelled.set() + self.upgrade.close() self.controller.close() from .launcher_bridge import remove remove(self.bridge) @@ -787,6 +948,7 @@ def tick(self): elif not self.reload_expected or time.monotonic() >= self.browser_wait: self.reload_expected = False self.open_browser() + self._tick_model_upgrade(row, busy) sources = self.controller.terminal_sources() if sources: selected = next((p for _, p in sources if p == self.log_source), sources[-1][1]) @@ -863,7 +1025,9 @@ def snapshot(self): page=self.page, status=row.get('status', 'idle'), busy=self.controller.busy or self.importer.busy or self.cleaner.busy, cleanup=dict(self.cleaner.state, busy=self.cleaner.busy), can_cleanup=not (self.controller.busy or self.importer.busy or self.cleaner.busy or self.closing - or self.update_intent or self.engine_updating or self.updater and self.updater.busy), + or self.update_intent or self.engine_updating or self.updater and self.updater.busy + or self.upgrade.active or self.model_switch), + model_upgrade=dict(self.upgrade.state, busy=self.upgrade.busy, switching=bool(self.model_switch)), offline=dict(progress_view(self.importer.state, zh), runtime=bool(self.form['offline_runtime']), runtime_supported=runtime_packages_supported(), models=len(self.form['offline_models']), guide=package_instructions(self.form['new_comfy'], zh)), diff --git a/freevideo_engine/lora_cache.py b/freevideo_engine/lora_cache.py index 7f9523a..c71fef0 100644 --- a/freevideo_engine/lora_cache.py +++ b/freevideo_engine/lora_cache.py @@ -176,6 +176,9 @@ def prepare(cache, adapters, output_root=None): verify_files({'loras': adapters}) manifest_path = cache / 'manifest.json' manifest = json.loads(manifest_path.read_text(encoding='utf-8')) + if manifest.get('precision') == 'int8': + raise ValueError('This LoRA also changes non-linear or modulation weights, which cannot be merged ' + 'into the int8 model yet. LoRAs that only adapt linear layers work.') if manifest.get('precision') != 'fp8': raise ValueError('LoRA variants require the prepared FP8 model') from .adaln_assets import SLIM_FORMATS, restore_projections diff --git a/freevideo_engine/lora_online_cache.py b/freevideo_engine/lora_online_cache.py index 296b3b0..3e5b390 100644 --- a/freevideo_engine/lora_online_cache.py +++ b/freevideo_engine/lora_online_cache.py @@ -40,7 +40,10 @@ def prepare(cache, adapters, output_root=None): manifest = json.loads((cache / 'manifest.json').read_text(encoding='utf-8')) if manifest.get('online_lora'): raise ValueError('Select the original model when changing the LoRA combination') - if manifest.get('precision') != 'fp8' or not _supported(adapters, manifest.get('linears', {})): + precision = manifest.get('precision') + # Low-rank branches run beside the base projection, so they need no change + # to FP8 or ConvRot int8 weights. + if precision not in ('fp8', 'int8') or not _supported(adapters, manifest.get('linears', {})): # Full-difference, bias and modulation patches retain the existing # exact preparation path. Never silently omit an unsupported patch. path, report = prepare_fused(cache, adapters, output_root) @@ -48,7 +51,7 @@ def prepare(cache, adapters, output_root=None): identity = dict(source_manifest_sha256=digest(cache / 'manifest.json'), implementation_sha256=digest(__file__), index_sha256=digest(Path(__file__).with_name('lora_cache.py')), adapters=[{k: v for k, v in row.items() if k != 'path'} for row in adapters], - arithmetic='FP8 base plus independent BF16 low-rank branches') + arithmetic=('FP8' if precision == 'fp8' else 'ConvRot int8') + ' base plus independent BF16 low-rank branches') key = hashlib.sha256(json.dumps(identity, sort_keys=True).encode()).hexdigest() output = Path(output_root or cache.parent) / ('vdn-online-lora-' + key[:24]) marker = output / 'manifest.json' diff --git a/freevideo_engine/model_upgrade.py b/freevideo_engine/model_upgrade.py new file mode 100644 index 0000000..bfbff79 --- /dev/null +++ b/freevideo_engine/model_upgrade.py @@ -0,0 +1,276 @@ +"""Move an existing installation to the faster prepared model, then release the old one. + +One consent covers the whole move, and each step starts only after the +previous one succeeded: + +1. download: setup --prepared-format int8 --prefetch-only, planned and approved + first, so the download cannot grow beyond what the user was shown. Only the + setup lease is held, so the installed FP8 model keeps generating. +2. switch: the launcher's normal setup run with --prepared-format int8, which + the session starts once the queue is idle. int8 kernels must pass on the + GPU, and machine.json changes only when every step succeeded. +3. release: variant_cleanup removes the retired FP8 files one by one, only once + the int8 model is in use and complete, retrying while a request holds the + engine lease. Fused LoRA variants keep the old model whole. + +Model files no longer in use - the old model left after the switch (the +launcher closed before the release, or a file was in use), or an int8 download +the GPU then failed to run - are offered for removal, and removed only on +request. +""" +import json +from pathlib import Path +import shutil +import threading +import time + +GiB = 1 << 30 +MARGIN = 2 * GiB +TARGET = 'int8_convrot' +RETRY_SECONDS = 30 + + +def offer(root): + """Whether this installation should move to int8 and what that costs, or which + model files it no longer uses. Reads only.""" + from .hardware import Hardware + from .kernel_capabilities import readiness + from .prepared_model import catalog as load_catalog, files, preferred_format, select + from . import variant_cleanup + root = Path(root) + try: + machine = json.loads((root / 'machine.json').read_text(encoding='utf-8')) + stamp = kernel_receipt(root) + receipt = json.loads((root / 'kernel-capabilities.json').read_text(encoding='utf-8')) + hardware = Hardware.from_dict(receipt['hardware']) + except (OSError, ValueError, KeyError, TypeError): + return dict(status='unavailable', reason='installation-unknown') + if not machine.get('ready') or machine.get('device_backend') == 'mps': + return dict(status='unavailable', reason='not-applicable') + catalog = load_catalog() + if machine.get('prepared_format') == TARGET: + return unused(root, catalog, TARGET) or dict(status='unavailable', reason='current') + current, _ = variant_cleanup.active_variant(root, catalog) + if current is None or current == TARGET: + # A reused or custom cache outside the prepared folders: never offered. + return dict(status='unavailable', reason='custom-model') + # A failed int8 probe on this GPU offers nothing. A receipt without one + # comes from an engine before int8 (an engine update does not run setup): + # the move is offered, and setup checks the int8 kernels before it changes + # anything. + try: + rows = receipt.get('kernel_probes') or [] + checked = any(row.get('backend') == 'linear-int8' for row in rows) + int8_ready = bool(readiness(rows).get('int8_ready')) + except (KeyError, TypeError, AttributeError): + checked = int8_ready = False + reason = ('fp8-is-faster' if preferred_format(hardware) != TARGET + else 'int8-kernels' if checked and not int8_ready else None) + if reason: + # For example an int8 download this GPU then failed to run. + return unused(root, catalog, current) or dict(status='unavailable', reason=reason, receipt=stamp) + selection = select(hardware.capability, root, scale_granularity=TARGET) + directory = Path(selection['directory']) + rows = files(selection) + present = 0 + for row in rows: + path = directory / row['file'] + try: + if path.is_file() and path.stat().st_size == row['bytes']: + present += row['bytes'] + except OSError: + pass + download = sum(row['bytes'] for row in rows) - present + fused = variant_cleanup.fused_lora_variants(root, catalog, current) + release = 0 if fused else variant_cleanup.retire(root, catalog, current)['bytes'] + free = shutil.disk_usage(root).free + return dict(status='available' if free >= download + MARGIN else 'low-disk', + architecture=hardware.architecture, gpu=hardware.gpu_name, current=current, + download_bytes=download, release_bytes=release, free_bytes=free, + required_bytes=download + MARGIN, lora_variants=fused, int8_checked=checked) + + +def kernel_receipt(root): + """Identifies the GPU check receipt as written; it changes when the checks run again.""" + try: + info = (Path(root) / 'kernel-capabilities.json').stat() + return [info.st_mtime_ns, info.st_size] + except OSError: + return None + + +def unused(root, catalog, in_use): + """Prepared variants on disk beside the one in use, removed only when the user asks; or None. + + The same scan and blockers as the release: the variant in use must be + complete, and fused LoRA variants keep theirs whole. + """ + from . import variant_cleanup + plan = variant_cleanup.scan(root, catalog, expect_active=in_use) + if plan['blockers'] or not plan['bytes']: + return None + return dict(status='releasable', in_use=in_use, release_bytes=plan['bytes'], + variants=[retired['variant'] for retired in plan['retired']], + lora_variants=[path for kept in plan['kept_variants'] for path in kept['lora_variants']]) + + +def default_runner(source, logs): + from .comfy_launcher_runtime import LauncherRunner + from .comfy_setup import Events + events = Events() + return LauncherRunner(source, events, logs), events + + +class ModelUpgrade: + def __init__(self, source, runner_factory=default_runner): + self.source = Path(source) + self.runner_factory = runner_factory + self.state = dict(status='idle') + self.thread = None + self.runner = None + self.closing = threading.Event() + + @property + def busy(self): + return self.thread is not None and self.thread.is_alive() + + @property + def active(self): + """Downloading or releasing: holding the setup lease or about to remove files.""" + return self.busy and self.state.get('status') in ('downloading', 'releasing') + + def _start(self, work): + if self.busy: + return False + self.thread = threading.Thread(target=work, name='freevideo-model-upgrade') + self.thread.start() + return True + + def inspect(self, root): + def work(): + try: + self.state = offer(root) + except Exception as error: + self.state = dict(status='unavailable', reason='error', error=str(error)) + return self._start(work) + + @staticmethod + def progress(snapshot, phase): + """The whole model download, never just the file being fetched at the moment.""" + view = snapshot.get('progress') or {} + if phase != 'download': + return {key: view.get(key) for key in ('label', 'done', 'total', 'unit', 'fraction')} + groups = snapshot.get('model_groups') or [] + total = sum(group.get('download_bytes') or 0 for group in groups) + if total <= 0: + # Until the download reports its files, show no numbers rather than one file's. + return dict(label=view.get('label'), done=None, total=None, unit=None, fraction=None) + done = min(total, sum(group.get('downloaded_bytes') or 0 for group in groups)) + return dict(label=view.get('label'), done=done, total=total, unit='bytes', fraction=done / total) + + def _run(self, runner, events, action, root, arguments, phase): + """Run one setup command and follow its progress until it ends.""" + runner.start(action, root, arguments) + result = None + while result is None: + for kind, value in events.drain(): + if kind == 'done': + result = value + self.state = dict(self.state, phase=phase, progress=self.progress(events.snapshot(), phase)) + if result is None: + time.sleep(.25) + return result + + def download(self, root, offered): + """Plan, approve exactly what was offered, then prefetch beside the model in use.""" + from .desktop_runtime import preflight_json + from .monitoring import save + root = Path(root) + def work(): + runner, events = self.runner_factory(self.source, root / 'launcher' / 'runs') + self.runner = runner + base = dict(offered, status='downloading') + self.state = dict(base, phase='plan') + try: + planned = self._run(runner, events, 'plan', root, + ['setup', '--prepared-format', 'int8', '--plan', '--json'], 'plan') + if planned['status'] == 'cancelled': + self.state = dict(offered) + return + value = preflight_json(planned.get('output', '')) + if value.get('errors'): + raise ValueError('\n'.join(value['errors'])) + if value.get('prepared_format') != TARGET or not value.get('prepared_model'): + raise ValueError('The setup plan does not move this installation to int8.') + # Nothing beyond what the user agreed to: the approval receipt + # makes the run refuse a larger download or a changed model. + if value['model_download_bytes'] > offered['download_bytes'] + 64 * (1 << 20): + raise ValueError('The download is larger than shown; check again.') + receipt = root / 'launcher' / 'model-upgrade-plan.json' + save(receipt, value) + fetched = self._run(runner, events, 'setup', root, + ['setup', '--prepared-format', 'int8', '--prefetch-only', '--yes', + '--accept-model-license', '--approved-plan', str(receipt)], 'download') + if fetched['status'] == 'cancelled': + self.state = dict(offered, status='available') + return + if fetched['status'] != 'complete': + raise RuntimeError(fetched.get('error') or 'The model download stopped; downloaded files are kept.') + self.state = dict(self.state, status='downloaded', phase='switch') + except Exception as error: + self.state = dict(offered, status='failed', step='download', error=str(error)) + finally: + self.runner = None + return self._start(work) + + def cancel(self): + runner = self.runner + if runner is not None: + runner.cancel() + + def waiting_for(self, reason): + """Downloaded, and the switch waits for this (a ComfyUI the launcher cannot restart).""" + if self.state.get('status') == 'downloaded' and self.state.get('waiting') != reason: + self.state = dict(self.state, waiting=reason) + + def switching(self): + if self.state.get('status') == 'downloaded': + self.state = dict(self.state, status='switching', waiting=None) + + def switch_failed(self, error, kept): + """kept: machine.json still describes the previous model, which keeps working.""" + self.state = dict(self.state, status='failed', step='switch', error=error, kept=kept, waiting=None) + + def release(self, root, expect=TARGET): + """Remove the variants not in use once `expect` is in use and complete, waiting out running requests.""" + from . import variant_cleanup + root = Path(root) + def work(): + self.state = dict(self.state, status='releasing', waiting=False) + while not self.closing.is_set(): + try: + plan = variant_cleanup.scan(root, expect_active=expect) + if plan['blockers']: + # Nothing was removed; the page explains each reason. + self.state = dict(self.state, status='failed', step='release', blockers=plan['blockers'], + error='', waiting=False) + return + receipt = root / 'logs' / ('retired-models-' + time.strftime('%Y%m%dT%H%M%SZ', time.gmtime()) + '.json') + result = variant_cleanup.clean(plan, receipt=receipt) + self.state = dict(self.state, status='complete', released_bytes=result['released_bytes'], + kept_bytes=result['kept_bytes'], skipped=len(result['skipped']), + lora_variants=[path for kept in plan['kept_variants'] for path in kept['lora_variants']], + receipt=str(receipt), waiting=False) + return + except BlockingIOError: + # A request or another setup holds a lease; try again after it. + self.state = dict(self.state, waiting=True) + self.closing.wait(RETRY_SECONDS) + except Exception as error: + self.state = dict(self.state, status='failed', step='release', error=str(error), waiting=False) + return + return self._start(work) + + def close(self): + self.closing.set() + self.cancel() diff --git a/freevideo_engine/modern_launcher.py b/freevideo_engine/modern_launcher.py index 7eb0bca..79b2220 100644 --- a/freevideo_engine/modern_launcher.py +++ b/freevideo_engine/modern_launcher.py @@ -124,6 +124,18 @@ def action(self, name, accepted=False): def cleanupDownloads(self, remove=False): self.invoke(session.cleanup_downloads, remove) + @Slot() + def upgradeModel(self): + self.invoke(session.upgrade_model) + + @Slot() + def cancelModelUpgrade(self): + self.invoke(session.cancel_model_upgrade) + + @Slot() + def dismissModelUpgrade(self): + self.invoke(session.dismiss_model_upgrade) + @Slot(str) def browse(self, key): if session.controller.busy: diff --git a/freevideo_engine/policy.py b/freevideo_engine/policy.py index 8fab397..9016a3c 100644 --- a/freevideo_engine/policy.py +++ b/freevideo_engine/policy.py @@ -271,9 +271,16 @@ def interpolate(group): MIN_GPU_RESERVE_GIB = .2 -def windows_gpu_output_workspace(video_tokens, *, prefetch=False): +FF_STASH_WIDTH = 14336 # BF16 SwiGLU output channels kept for the per-tensor FP8 scale + + +def windows_gpu_output_workspace(video_tokens, *, prefetch=False, ff_stash=True): """Starting workspace for head-8 GPU outputs with the full per-tensor FF stash. + Int8 projections scale each activation row on its own, so every FF row tile + finishes immediately and nothing is stashed; ff_stash=False removes that + term (1.94 GiB at 72576 tokens, linear in the token count). + Two complete Windows 5060 Ti requests constrain this estimate: 896-square / 243 frames used 11.14 GiB reserved with zero resident blocks; 1344x768 / 243 used 13.75 GiB with eight blocks and FF recomputation. Subtracting @@ -283,7 +290,8 @@ def windows_gpu_output_workspace(video_tokens, *, prefetch=False): The OS growth reserve has already been subtracted from the budget. """ tokens = 72576 if video_tokens is None else video_tokens - return int((6.5 + 6. * tokens / 72576) * GiB) + (BLOCK_BYTES if prefetch else 0) + workspace = int((6.5 + 6. * tokens / 72576) * GiB) + (BLOCK_BYTES if prefetch else 0) + return workspace if ff_stash else workspace - tokens * FF_STASH_WIDTH * 2 class ResourceBudgetError(ValueError): @@ -453,10 +461,14 @@ def adaln_extra_bytes(canvas=None): def choose(hardware: Hardware, *, vram_gib=None, ram_gib=None, attention='auto', gpu_reserve_gib=None, ram_reserve_gib=None, available_backends=None, canvas=None, demonstrated_ram_bytes=None, allow_capacity_trial=False, stage='generation', - lora_max_block_bytes=0, lora_root_bytes=0): + lora_max_block_bytes=0, lora_root_bytes=0, precision='fp8'): + """precision is the prepared cache's: int8 caches run int8 projections.""" from .geometry import geometry if stage not in ('encoding', 'generation'): raise ValueError('Resource planning stage must be encoding or generation') + if precision not in ('fp8', 'int8'): + raise ValueError('Resource planning needs the FP8 or int8 prepared model') + int8 = precision == 'int8' if type(lora_max_block_bytes) is not int or lora_max_block_bytes < 0: raise ValueError('LoRA block bytes must be a nonnegative integer') if type(lora_root_bytes) is not int or lora_root_bytes < 0: @@ -586,6 +598,14 @@ def choose(hardware: Hardware, *, vram_gib=None, ram_gib=None, attention='auto', legs *= 2 if len(legs) != 2 or any(leg not in available for leg in legs): raise ValueError('Requested attention backend is unavailable or failed its kernel probe: ' + attention) + # Ampere's own plan exists for its bounded BF16 compute, which widens FP8 + # weights per projection: one attention window, and residency taken from + # the activation table alone. Int8 projections run the same kernels on + # every architecture, so they take the common plan. On a 12 GiB card + # emulating Ampere the Ampere plan held four blocks and ran out of memory + # in attention quantization (9.61 GiB allocated against 9.85 allowed), + # where the common plan holds one and completes. + ampere = hardware.architecture == 'ampere' and not int8 # Small cards trade extra transfers for lower activation storage. The # 12 GiB native-FP8 preset comes from the matched 12/32 capacity experiments. small = gpu_budget < 10 * GiB @@ -593,15 +613,14 @@ def choose(hardware: Hardware, *, vram_gib=None, ram_gib=None, attention='auto', # 10 GiB boundary. Keep its affordable head-4 path in that narrow band # until the larger path's actual workspace fits; entering that path only # to fall back to CPU residuals would waste the newly available memory. - if (desktop and hardware.architecture != 'ampere' and gpu_budget < 11 * GiB - and gpu_budget < windows_gpu_output_workspace(effective_tokens, prefetch=False)): + if (desktop and not ampere and gpu_budget < 11 * GiB + and gpu_budget < windows_gpu_output_workspace(effective_tokens, prefetch=False, ff_stash=not int8)): small = True # No gap between the small and twelve bands. A 0.25 GiB sliver used to fall # through both and take an unmeasured branch with three resident blocks, # which made residency drop as the budget grew: three blocks at 10.1 GiB # against one at 10.3 GiB. twelve = not small and gpu_budget < 14 * GiB - ampere = hardware.architecture == 'ampere' prefetch = prefetch_for_budget(gpu_budget, ram_budget=ram_budget, twelve=twelve, ampere=ampere, system=hardware.system, architecture=hardware.architecture) @@ -647,7 +666,7 @@ def choose(hardware: Hardware, *, vram_gib=None, ram_gib=None, attention='auto', # 360-595 s/step; eight heads needed 25.07 of its 30.40 GiB budget. # Without this, 26 to 30 s at 1344x768 staged and 31 to 38 s did not. if not small and not ampere: - while head > 8 and (windows_gpu_output_workspace(effective_tokens, prefetch=False) + while head > 8 and (windows_gpu_output_workspace(effective_tokens, prefetch=False, ff_stash=not int8) + activation_bytes(head, effective_tokens) - activation_bytes(8, effective_tokens)) > gpu_budget: head //= 2 @@ -859,7 +878,7 @@ def choose(hardware: Hardware, *, vram_gib=None, ram_gib=None, attention='auto', # FF stash. In addition, crediting a shorter geometry against its old # 345-frame anchor added five blocks to a request that already OOMed. # Budget the current execution path before host placement is planned. - windows_workspace = windows_gpu_output_workspace(effective_tokens, prefetch=prefetch) + windows_workspace = windows_gpu_output_workspace(effective_tokens, prefetch=prefetch, ff_stash=not int8) windows_workspace += max(0, activation_bytes(head, effective_tokens) - activation_bytes(8, effective_tokens)) if (prefetch and gpu_budget < 14 * GiB and hardware.architecture == 'blackwell-rtx'): @@ -1013,7 +1032,7 @@ def choose(hardware: Hardware, *, vram_gib=None, ram_gib=None, attention='auto', # matched the recomputing run to the byte across three probe # steps (6.63, 7.37, 8.11 GiB) because those activations are # freed within each block, while it cost 5% of sampling time. - fp8_ff_recompute=(capacity_trial and hardware.capability[0] >= 10), cache_refined_text=True, + fp8_ff_recompute=(capacity_trial and hardware.capability[0] >= 10 and not int8), cache_refined_text=True, resident_blocks=resident, pin_host_gb=round(pin, 3), stream_weights=streamed_weights, # Larger looped batches are optimization candidates; full @@ -1034,7 +1053,7 @@ def choose(hardware: Hardware, *, vram_gib=None, ram_gib=None, attention='auto', # unchanged at 15.83 and 18.09 GiB. Dropping to two at a # 12 GiB cap cost 0.2%, so one rule covers every band. window_batch=1 if ampere else 4, - linear_compute='bf16-weight-only' if ampere else 'native-fp8') + linear_compute='int8' if int8 else 'bf16-weight-only' if ampere else 'native-fp8') notes = ['Request resolution, frame count, steps and attention backend are preserved.', 'Budgets use current free VRAM and available physical/commit RAM, with growth headroom; OS usage is already excluded.', 'Automatic execution uses matching GPU kernel probes; read-only plans may use package discovery. Small probes do not prove full-video capacity.', @@ -1068,7 +1087,7 @@ def choose(hardware: Hardware, *, vram_gib=None, ram_gib=None, attention='auto', else: notes.append('Live RAM cannot retain all offloaded mappings: keep at most %.3f GB pinned, reopen only remaining layers per transfer; no arithmetic or attention change.' % pin) notes.append('Host weight cache keeps %.2f GiB for this path\'s working memory, inside the separate OS growth reserve; the runtime rechecks available RAM.' % (host_headroom / GiB)) - if hardware.architecture == 'ampere': + if ampere: notes.append('Ampere keeps FP8 weight storage and uses bounded BF16 compute; no native FP8 tensor-core claim.') if desktop: notes.append('Windows keeps %.2f GiB beyond current GPU usage and %.2f GiB beyond current available RAM/commit; pagefile capacity is not added to RAM.' % (reserve_gpu, reserve_ram)) diff --git a/freevideo_engine/prepared_model.py b/freevideo_engine/prepared_model.py index 3b39f36..117feaf 100644 --- a/freevideo_engine/prepared_model.py +++ b/freevideo_engine/prepared_model.py @@ -39,6 +39,23 @@ def catalog(): return value +def preferred_format(hardware): + """The prepared format a fresh installation should use on this GPU. + + int8 where it is the faster arithmetic: GeForce Ada and Blackwell cards run + FP8 with FP32 accumulation at half rate but int8 at full rate, and Ampere + has no FP8 tensor cores at all. Workstation and datacenter Ada/Blackwell + cards run FP8 at full rate and keep it (int8 measured 5% slower on an RTX + PRO 6000). Returns None to keep the GPU's FP8 granularity. + """ + architecture = hardware.architecture + if architecture == 'ampere': + return 'int8_convrot' + if architecture in ('ada', 'blackwell-rtx') and 'geforce' in (hardware.gpu_name or '').lower(): + return 'int8_convrot' + return None + + def select(capability, root, *, scale_granularity=None): value = catalog() scale = scale_granularity or ('per_tensor' if capability[0] >= 10 else 'rowwise') diff --git a/freevideo_engine/probe.py b/freevideo_engine/probe.py index a6b406a..7f265f1 100644 --- a/freevideo_engine/probe.py +++ b/freevideo_engine/probe.py @@ -43,6 +43,26 @@ def main(): reference_x = x.float() if torch.cuda.get_device_capability() < (8, 9) else xq.float() * xs reference = reference_x @ (fp8.float() * scale.reshape(-1, 1)).t() tolerance = .01 + elif backend == 'linear-int8': + from .int8_ops import GROUP, hadamard16, matmul, rotate_quantize + # The prepared int8 path on the user's GPU: ConvRot rotation and + # per-row quantization, then the int8 GEMM against the same rows + # dequantized in FP32. A tail row block is included; no model loads. + x = torch.randn(701, 2 * GROUP, device='cuda', dtype=torch.bfloat16) + weight = torch.randint(-127, 128, (1024, 2 * GROUP), device='cuda', dtype=torch.int8) + weight_scale = torch.rand(1024, device='cuda') / 64 + 1e-3 + values, scales = rotate_quantize(x) + h16 = hadamard16(x.device) + rotated = (h16 @ x.float().reshape(len(x), -1, 16, 16) @ h16).reshape(len(x), -1) + dequantized = values.float() * scales[:, None] + rotation = ((dequantized - rotated).square().mean() / rotated.square().mean()).sqrt().item() + if not math.isfinite(rotation) or rotation > .03: + raise RuntimeError(f'Int8 rotation/quantization relative RMSE {rotation:.6f} exceeded 0.03') + value = matmul((values, scales), weight, weight_scale, out_dtype=torch.float32) + reference = dequantized @ (weight.float() * weight_scale[:, None]).t() + shapes = {'activation': list(x.shape), 'weight': list(weight.shape), 'rotation_relative_rmse': rotation} + policy = 'int8-convrot' + tolerance = 1e-4 else: if backend == 'torch-flash' and not torch.backends.cuda.is_flash_attention_available(): raise RuntimeError('This PyTorch wheel was built without CUDA Flash Attention. ' diff --git a/freevideo_engine/provision.py b/freevideo_engine/provision.py index 0583dc3..bc83f04 100644 --- a/freevideo_engine/provision.py +++ b/freevideo_engine/provision.py @@ -436,16 +436,50 @@ def prepare(plan, out): save(root / 'prepared-cache.json', record) +def prefetch(plan, out): + """Fetch the prepared format an installation is about to switch to, beside the one in use. + + Generation keeps running on the installed model: only the setup lease is + held (no concurrent setup run), machine.json and prepared-cache.json are + not touched, and nothing is removed. The setup run that switches afterwards + finds every file present and verified. + """ + from .prepared_model import files as prepared_files + root = Path(plan['root']).expanduser().resolve() + prepared = plan.get('prepared_model') + if not prepared or plan.get('reuse_cache'): + raise ValueError('Prefetch needs a prepared model this installation does not use yet') + machine = json.loads((root / 'machine.json').read_text(encoding='utf-8')) + if not machine.get('ready') or not machine.get('cache'): + raise ValueError('Prefetch runs beside a working installation; complete setup first') + if Path(machine['cache']).resolve() == Path(prepared['cache']).resolve(): + raise ValueError('This prepared model is already in use') + with runtime_lock(root / 'setup.lock', inherit=False): + models(plan) + directory = Path(prepared['directory']) + rows = prepared_files(prepared) + present = [row for row in rows if (directory / row['file']).is_file() + and (directory / row['file']).stat().st_size == row['bytes']] + result = dict(schema_version=1, scale_granularity=prepared['scale_granularity'], revision=prepared['revision'], + cache=prepared['cache'], files=len(rows), present=len(present), + bytes=sum(row['bytes'] for row in rows), complete=len(present) == len(rows), finished=time.time()) + save(out, result) + print(json.dumps(dict(event='prefetch_complete', **result)), flush=True) + return result + + def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument('--plan', required=True, type=Path) operation = parser.add_mutually_exclusive_group() operation.add_argument('--prepare', action='store_true') operation.add_argument('--cleanup', action='store_true') + operation.add_argument('--prefetch', action='store_true', + help='Download and verify another prepared format while the installed one stays in use') parser.add_argument('--out', type=Path) args = parser.parse_args() - if (args.prepare or args.cleanup) and args.out is None: - parser.error('--out is required for preparation and cleanup reports') + if (args.prepare or args.cleanup or args.prefetch) and args.out is None: + parser.error('--out is required for preparation, cleanup and prefetch reports') plan = json.loads(args.plan.read_text(encoding='utf-8')) if args.cleanup: record = json.loads((Path(plan['root']) / 'prepared-cache.json').read_text(encoding='utf-8')) @@ -454,6 +488,8 @@ def main(): print(json.dumps({'event': 'storage_compacted', **result}), flush=True) elif args.prepare: prepare(plan, args.out) + elif args.prefetch: + prefetch(plan, args.out) else: models(plan) diff --git a/freevideo_engine/release_notes.json b/freevideo_engine/release_notes.json index db7e975..9a4bdd9 100644 --- a/freevideo_engine/release_notes.json +++ b/freevideo_engine/release_notes.json @@ -1,22 +1,22 @@ { "schema": 1, - "product_version": "0.2.5", + "product_version": "0.3.0", "en": { - "summary": "Optional prompt enhancement, retries after failed installations and updates, and download cache cleanup.", + "summary": "A faster int8 model for GeForce and RTX 30-series cards with a one-click upgrade, and optional prompt enhancement.", "changes": [ - "Optional FreeToken prompt enhancement in the creation panel: rewrite your prompt and reference images for MiniMax H3 with a local model, keep the original and the enhanced version, and choose which one to generate. It is off by default and asks before downloading the model (8.3 GiB); your prompt and images stay on this computer.", - "After freeing disk space, check the installation plan again without reopening FreeVideo. Existing models and download progress are kept.", - "Clear installation download caches from Settings, with a space estimate before removal. Installed environments, models and videos are kept.", - "Update buttons stay visible with long release notes. Failed update downloads no longer leave temporary files behind." + "GeForce RTX 40/50-series and RTX 30-series cards use a faster int8 model, closer to the original quality.", + "Optional prompt enhancement in the creation panel: a local model rewrites your prompt for MiniMax H3, and you choose the original or the enhanced version. It is off by default and asks before downloading its model (8.3 GiB).", + "Faster generation on 8 GB cards.", + "Failed installations and updates can be retried in place, and download caches can be cleaned from Settings." ] }, "zh": { - "summary": "新增可选的提示词增强;安装和更新失败可直接重试,设置中可清理下载缓存。", + "summary": "GeForce 和 RTX 30 系显卡改用更快的 int8 模型,老用户一键升级;新增可选的提示词增强。", "changes": [ - "创作面板新增可选的 FreeToken 提示词增强:用本地模型结合参考图,把提示词改写成适合 MiniMax H3 的写法;原文和改写版都会保留,生成时可任选其一。默认关闭,首次使用会先询问再下载模型(8.3 GiB),提示词和图片只在本机处理。", - "释放磁盘空间后可原地重新检查安装计划,无需重开软件;保留已有模型和下载进度。", - "设置中可查看下载缓存大小并清理,保留已安装的运行环境、模型和视频。", - "更新说明较长时,操作按钮仍固定可见;更新下载失败后自动回收本次临时文件。" + "GeForce RTX 40/50 系和 RTX 30 系显卡换用更快的 int8 模型,画质也更接近原版。", + "创作面板新增可选的提示词增强:用本地模型把提示词改写成适合 MiniMax H3 的写法,原文和改写版可任选其一生成。默认关闭,首次使用会先询问再下载模型(8.3 GiB)。", + "8 GB 显卡生成更快。", + "安装或更新失败可直接重试,设置中可以清理下载缓存。" ] } } diff --git a/freevideo_engine/residual.py b/freevideo_engine/residual.py index abdeb24..94d069c 100644 --- a/freevideo_engine/residual.py +++ b/freevideo_engine/residual.py @@ -17,6 +17,7 @@ def __init__(self, tensor): self.tensor = tensor self.shape, self.dtype, self.device = tuple(tensor.shape), tensor.dtype, tensor.device self.host = None + self.pinned = False self.stored = False self.closed = False self.copies = {'to_host': 0, 'to_device': 0} @@ -38,20 +39,40 @@ def store(self, tensor): raise RuntimeError('Consume the device residual before staging it') tick = time.perf_counter() if self.host is None: - # Pageable storage avoids consuming the platform's limited locked - # pages in addition to attention readouts and the weight cache. - self.host = torch.empty(self.shape, dtype=self.dtype, device='cpu') - self.host.copy_(tensor, non_blocking=False) + self.host = self._allocate() + self.host.copy_(tensor, non_blocking=self.pinned) self.stored = True self.copies['to_host'] += 1 self.copy_wall_seconds += time.perf_counter()-tick + def _allocate(self): + # Locked pages make both copies asynchronous DMA on the compute + # stream, so the CPU keeps queueing kernels instead of draining the + # GPU twice per block. Stream order is all the synchronization this + # needs: the restore and every later kernel run after the store on the + # same stream, and the host allocator holds a freed block until the + # copies recorded on it finish. The planner already keeps this buffer + # out of the weight cache (residual_host_headroom), so locking it + # takes no pages that pinned weights were promised. On a 12 GiB + # RTX 4070 capped at 8 GiB, 1344x768x243, second-pass steps measured + # 108 -> 96 s with native FP8 and 99 -> 87 s with int8 projections. + # Pageable storage remains the fallback when pages cannot be locked. + if self.device.type == 'cuda': + try: + host = torch.empty(self.shape, dtype=self.dtype, device='cpu', pin_memory=True) + except RuntimeError: + pass + else: + self.pinned = True + return host + return torch.empty(self.shape, dtype=self.dtype, device='cpu') + def restore(self): if self.closed or not self.stored: raise RuntimeError('No complete host residual is available') tick = time.perf_counter() # copy=True also makes CPU contract checks use independent storage. - tensor = self.host.to(self.device, copy=True, non_blocking=False) + tensor = self.host.to(self.device, copy=True, non_blocking=self.pinned) self.copies['to_device'] += 1 self.copy_wall_seconds += time.perf_counter()-tick return tensor @@ -63,9 +84,12 @@ def replace(self, tensor): self.tensor = tensor def stats(self): - return dict(host_buffer_bytes=self.host.numel()*self.host.element_size() if self.host is not None else 0, - pinned_host_bytes=0, copies=dict(self.copies), blocking_copy_wall_seconds=self.copy_wall_seconds, - timing_scope='Blocking copy calls can include waiting for prior compute; not pure PCIe transfer time.') + host_bytes = self.host.numel()*self.host.element_size() if self.host is not None else 0 + return dict(host_buffer_bytes=host_bytes, pinned_host_bytes=host_bytes if self.pinned else 0, + copies=dict(self.copies), blocking_copy_wall_seconds=self.copy_wall_seconds, + timing_scope=('Locked-page copies are queued asynchronously; this is the time spent issuing them.' + if self.pinned else + 'Blocking copy calls can include waiting for prior compute; not pure PCIe transfer time.')) def close(self): self.tensor = self.host = None diff --git a/freevideo_engine/runtime.py b/freevideo_engine/runtime.py index bf3ce8d..90a59b5 100644 --- a/freevideo_engine/runtime.py +++ b/freevideo_engine/runtime.py @@ -79,7 +79,7 @@ def __init__(self, cache, *, base=None, checkpoint=None, self.weight_cache_headroom_bytes += residual_host_headroom(self.canvas) base = base_path() if base is None else Path(base) checkpoint = checkpoint_path() if checkpoint is None else Path(checkpoint) - if linear_compute not in ('native-fp8', 'bf16-weight-only'): + if linear_compute not in ('native-fp8', 'bf16-weight-only', 'int8'): raise ValueError('Unknown linear compute policy') if min(query_chunk, ff_chunk, head_chunk, projection_chunk) < 0: raise ValueError('Chunk sizes cannot be negative; zero disables chunking') diff --git a/freevideo_engine/system.py b/freevideo_engine/system.py index ac59bdc..6fb115f 100644 --- a/freevideo_engine/system.py +++ b/freevideo_engine/system.py @@ -96,7 +96,11 @@ def system_memory(): def residual_host_headroom(canvas=None): - """Room for one pageable packed residual, separate from the weight cache.""" + """Room for one packed residual, separate from the weight cache. + + The staging buffer is locked on CUDA when the platform allows, and this + room is what keeps those pages out of the pinned weight budget. + """ rows = (sum(canvas.get(key, 0) for key in ('video_tokens', 'reference_video_tokens', 'reference_audio_tokens')) if canvas is not None else 72576) diff --git a/freevideo_engine/tuning.py b/freevideo_engine/tuning.py index e102d66..4b67dab 100644 --- a/freevideo_engine/tuning.py +++ b/freevideo_engine/tuning.py @@ -22,7 +22,7 @@ SCHEMA = 1 COMPUTE_FILES = ('activation_staging.py', 'residual.py', 'adaln.py', 'adaln_assets.py', 'attention.py', 'blocks.py', 'keyframes.py', 'media_request.py', 'media_conditioning.py', 'media_encoding.py', 'reference_sampler.py', 'lora_cache.py', - 'conditioning.py', 'decode.py', 'decode_prefetch.py', 'resident_models.py', 'decode_stream.py', 'streamed_weights.py', 'encode_worker.py', 'idle_encoder.py', 'media_encoding.py', 'fp8.py', 'fp8_ops.py', 'fp8_gemm.py', + 'conditioning.py', 'decode.py', 'decode_prefetch.py', 'resident_models.py', 'decode_stream.py', 'streamed_weights.py', 'encode_worker.py', 'idle_encoder.py', 'media_encoding.py', 'fp8.py', 'fp8_ops.py', 'fp8_gemm.py', 'int8_ops.py', 'encoder_precision.py', 'encoder_lowmem.py', 'kernel_capabilities.py', 'gpu_budget.py', 'worker.py', 'resident_worker.py', 'failure_cleanup.py', 'geometry.py', 'head_chunk.py', 'offload.py', 'packing.py', 'runtime.py', diff --git a/freevideo_engine/two_pass.py b/freevideo_engine/two_pass.py index f7f4914..2b9b472 100644 --- a/freevideo_engine/two_pass.py +++ b/freevideo_engine/two_pass.py @@ -208,6 +208,7 @@ def first_pass_policy(profile, canvas, sampling_plan): gpu_reserve_gib=gpu_reserve / GiB, ram_reserve_gib=ram_reserve / GiB, lora_max_block_bytes=policy.get('lora_max_block_bytes', 0), lora_root_bytes=policy.get('lora_root_bytes', 0), + precision='int8' if profile['engine'].get('linear_compute') == 'int8' else 'fp8', canvas=first, allow_capacity_trial=policy.get('capacity_trial', False)).legacy_profile() selected['engine']['steps'] = sampling_plan['base_steps'] selected['policy']['engine']['steps'] = sampling_plan['base_steps'] diff --git a/freevideo_engine/variant_cleanup.py b/freevideo_engine/variant_cleanup.py new file mode 100644 index 0000000..f84a980 --- /dev/null +++ b/freevideo_engine/variant_cleanup.py @@ -0,0 +1,439 @@ +"""Remove a retired prepared model variant once its replacement is in use. + +After an installation moves to another prepared variant (FP8 to int8), the old +variant's files are no longer read. Removal is limited to what this installer +placed there, one file at a time: + +* each pinned catalog file of a variant that is not in use, at its catalogued + size, regular and unshared. LoRA variants hardlink their unchanged groups, so + a shared file stays with the LoRA that still uses it; +* that file's download sidecar, which must name the same SHA-256, and its + interrupted partial downloads; +* that file's native staging folder below .freevideo-downloads; +* AdaLN tables the engine generated in the variant's cache folder, only when + the folder's name is the digest of its identity marker and it holds nothing + but table files. + +A variant with fused LoRA variants beside its cache is kept whole: int8 cannot +merge those adapters yet, so the old model is still the only way to use them. + +Every path must lie inside the installation and be reached without symbolic +links or junctions. Everything else in the folder (LoRA variants, user files) +is reported and kept, and a directory is removed only once it is empty. The +scan records a fingerprint of every file; removal requires it unchanged, under +the setup, engine and host-setup leases, with the replacement complete and in +use. Uncertain ownership is never permission to delete. +""" +import json +import os +from pathlib import Path +import re +import stat +import time + +from .locking import runtime_lock +from .storage import fingerprint + +SIDECAR = '.partial.source.json' +SIDECAR_LIMIT = 4096 +STAGES = '.freevideo-downloads' +SOURCE = re.compile(r'[a-z0-9][a-z0-9-]{0,31}') +FUSED_LORA = 'vdn-fp8-lora-' +PARTIAL = re.compile(r'\.partial(-\d{1,20})?') +ADALN_DIRECTORY = re.compile(r'adaln-tables-v2-[0-9a-f]{24}') +ADALN_FILE = re.compile(r'identity\.json|producer\.json|\d{2}\.(safetensors|json)') +ADALN_IDENTITY_LIMIT = 65536 +REPARSE_POINT = 0x400 + + +def redirected(info): + return stat.S_ISLNK(info.st_mode) or bool(getattr(info, 'st_file_attributes', 0) & REPARSE_POINT) + + +def inside(base, path): + """True when path is base or below it and nothing between them is a link or junction.""" + try: + path.relative_to(base) + current = path + while True: + if redirected(current.lstat()): + return False + if current == base: + return True + current = current.parent + except (OSError, ValueError): + return False + + +def unshared_file(path): + try: + info = path.lstat() + except OSError: + return None + if redirected(info) or not stat.S_ISREG(info.st_mode) or info.st_nlink != 1: + return None + return info + + +def variant_folder(root, catalog, name): + variant = catalog['variants'][name] + return root / 'prepared' / ('edge-' + variant.get('revision', catalog['revision'])[:16]) + + +def active_variant(root, catalog): + """The catalog variant whose cache machine.json names, or None when that is unknown.""" + try: + record = json.loads((root / 'machine.json').read_text(encoding='utf-8')) + cache = Path(record['cache']) + except (OSError, ValueError, KeyError, TypeError): + return None, None + for name, variant in catalog['variants'].items(): + expected = variant_folder(root, catalog, name) / variant['cache_prefix'] + try: + if cache.absolute() == expected or cache.resolve() == expected.resolve(): + return name, expected + except OSError: + continue + return None, cache + + +def complete(root, catalog, name): + """Every catalog file of the variant is present, regular, local and of its catalogued size.""" + folder = variant_folder(root, catalog, name) + prepared = root / 'prepared' + for row in catalog['variants'][name]['files']: + path = folder / row['file'] + try: + info = path.lstat() + except OSError: + return False + if redirected(info) or not stat.S_ISREG(info.st_mode) or info.st_size != row['bytes'] or not inside(prepared, path): + return False + return True + + +def _sidecar_matches(path, row): + info = unshared_file(path) + if info is None or info.st_size > SIDECAR_LIMIT: + return False + try: + value = json.loads(path.read_text(encoding='utf-8')) + except (OSError, ValueError): + return False + return isinstance(value, dict) and value.get('expected') == row['sha256'] + + +def _stage(folder, prepared, stage): + """Files and directories of one native staging folder, or None when anything is redirected or shared.""" + if not inside(prepared, stage) or not stage.is_dir(): + return None + files, directories = [], [stage] + for current, names, entries in os.walk(stage, followlinks=False): + for name in names: + path = Path(current) / name + try: + if redirected(path.lstat()): + return None + except OSError: + return None + directories.append(path) + for name in entries: + path = Path(current) / name + if unshared_file(path) is None: + return None + files.append(path) + return files, directories + + +def _generated_tables(prepared, table): + """Files of one engine-generated AdaLN table folder, or None unless it is bound to its identity.""" + from .adaln_assets import FORMAT, directory + if not ADALN_DIRECTORY.fullmatch(table.name) or not inside(prepared, table) or not table.is_dir(): + return None + entries = list(table.iterdir()) + if any(not ADALN_FILE.fullmatch(entry.name) or unshared_file(entry) is None for entry in entries): + return None + marker = table / 'identity.json' + info = unshared_file(marker) + if info is None or info.st_size > ADALN_IDENTITY_LIMIT: + return None + try: + identity = json.loads(marker.read_text(encoding='utf-8')) + if not isinstance(identity, dict) or identity.get('format') != FORMAT or directory(identity) != table.name: + return None + except (OSError, ValueError, TypeError): + return None + return entries + + +def _plan_file(path, kind, entries): + entries.append(dict(path=str(path), kind=kind, bytes=path.lstat().st_size, stamp=fingerprint(path))) + + +def retire(root, catalog, name): + """Owned files of one retired variant, and what stays in its folder.""" + folder = variant_folder(root, catalog, name) + prepared = root / 'prepared' + entries, kept, planned, stage_directories = [], [], set(), [] + if not folder.is_dir() or not inside(prepared, folder): + return dict(variant=name, folder=str(folder), files=entries, kept=kept, bytes=0, kept_bytes=0, + stage_directories=stage_directories, present=folder.exists() or folder.is_symlink()) + stages = folder / STAGES + for row in catalog['variants'][name]['files']: + path = folder / row['file'] + if path.exists() or path.is_symlink(): + info = unshared_file(path) + if info is not None and info.st_size == row['bytes'] and inside(prepared, path): + _plan_file(path, 'catalog', entries) + planned.add(path) + sidecar = path.with_name(path.name + SIDECAR) + if sidecar.exists() and inside(prepared, sidecar) and _sidecar_matches(sidecar, row): + _plan_file(sidecar, 'sidecar', entries) + planned.add(sidecar) + if path.parent.is_dir() and inside(prepared, path.parent): + for candidate in path.parent.iterdir(): + suffix = candidate.name[len(path.name):] + if (candidate.name.startswith(path.name) and PARTIAL.fullmatch(suffix) + and unshared_file(candidate) is not None and inside(prepared, candidate)): + _plan_file(candidate, 'partial', entries) + planned.add(candidate) + if stages.is_dir() and inside(prepared, stages): + for stage in stages.iterdir(): + identity, _, source = stage.name.partition('-') + if identity != row['sha256'][:16] or not SOURCE.fullmatch(source): + continue + found = _stage(folder, prepared, stage) + if found is None: + continue + for path_in_stage in found[0]: + if path_in_stage not in planned: + _plan_file(path_in_stage, 'stage', entries) + planned.add(path_in_stage) + for directory in found[1]: + if str(directory) not in stage_directories: + stage_directories.append(str(directory)) + cache = folder / catalog['variants'][name]['cache_prefix'] + if cache.is_dir() and inside(prepared, cache): + for table in sorted(cache.iterdir()): + found = _generated_tables(prepared, table) + for path in found or (): + if path not in planned: + _plan_file(path, 'adaln', entries) + planned.add(path) + kept_bytes = 0 + for current, names, files in os.walk(folder, followlinks=False): + for name_ in list(names): + path = Path(current) / name_ + try: + if redirected(path.lstat()): + names.remove(name_) + kept.append(dict(path=str(path), bytes=0, reason='link or junction')) + except OSError: + names.remove(name_) + for name_ in files: + path = Path(current) / name_ + if path in planned: + continue + try: + size = path.lstat().st_size + except OSError: + size = 0 + kept_bytes += size + kept.append(dict(path=str(path), bytes=size, reason='not placed by this installer for this variant')) + return dict(variant=name, folder=str(folder), files=entries, kept=kept, present=True, + stage_directories=stage_directories, + bytes=sum(row['bytes'] for row in entries), kept_bytes=kept_bytes) + + +def fused_lora_variants(root, catalog, name): + """Fused LoRA variants built from this variant (they live beside its cache).""" + beside = (variant_folder(root, catalog, name) / catalog['variants'][name]['cache_prefix']).parent + try: + return sorted(str(path) for path in beside.iterdir() if path.name.startswith(FUSED_LORA)) + except OSError: + return [] + + +def scan(root, catalog=None, expect_active=None): + """What removing the variants that are not in use would release; nothing is changed. + + expect_active names the variant the caller has just switched to; a scan + that finds any other variant in use plans nothing. + """ + from .prepared_model import catalog as load_catalog + root = Path(root).absolute() + catalog = catalog or load_catalog() + active, cache = active_variant(root, catalog) + plan = dict(root=str(root), active=active, cache=str(cache) if cache else None, expect_active=expect_active, + retired=[], kept_variants=[], bytes=0, kept_bytes=0, blockers=[]) + try: + if redirected((root / 'prepared').lstat()): + # A models folder moved elsewhere behind a link is the user's own + # arrangement; deleting through it is not ours to decide. + plan['blockers'].append('prepared-folder-redirected') + return plan + except OSError: + pass + if active is None: + plan['blockers'].append('active-variant-unknown') + return plan + if expect_active is not None and active != expect_active: + plan['blockers'].append('unexpected-active-variant') + return plan + if not complete(root, catalog, active): + plan['blockers'].append('active-variant-incomplete') + return plan + active_folder = variant_folder(root, catalog, active) + for name in catalog['variants']: + if name == active: + continue + folder = variant_folder(root, catalog, name) + if folder == active_folder: + continue # A shared revision folder also holds the files in use. + if not folder.exists() and not folder.is_symlink(): + continue + fused = fused_lora_variants(root, catalog, name) + if fused: + plan['kept_variants'].append(dict(variant=name, folder=str(folder), reason='fused-lora-variants', + lora_variants=fused)) + continue + retired = retire(root, catalog, name) + plan['retired'].append(retired) + plan['bytes'] += retired['bytes'] + plan['kept_bytes'] += retired['kept_bytes'] + return plan + + +def allowed(root, catalog, name): + """The paths removal may touch for one retired variant: catalog files and their own metadata names.""" + folder = variant_folder(root, catalog, name) + rows = {} + for row in catalog['variants'][name]['files']: + rows[str(folder / row['file'])] = ('catalog', row) + rows[str(folder / (row['file'] + SIDECAR))] = ('sidecar', row) + return folder, rows + + +def permitted(path, kind, folder, rows, cache_prefix=None): + """Recheck at removal time that a planned path still has exactly the name it was planned under.""" + if not path.is_absolute() or '..' in path.parts or Path(os.path.normpath(path)) != path: + return False + key = str(path) + if kind in ('catalog', 'sidecar'): + return key in rows and rows[key][0] == kind + if kind == 'partial': + for name, (row_kind, row) in rows.items(): + if row_kind == 'catalog' and key.startswith(name) and PARTIAL.fullmatch(key[len(name):]): + return Path(name).parent == path.parent + return False + if kind == 'adaln': + if cache_prefix is None: + return False + try: + relative = path.relative_to(folder / cache_prefix) + except ValueError: + return False + return (len(relative.parts) == 2 and ADALN_DIRECTORY.fullmatch(relative.parts[0]) is not None + and ADALN_FILE.fullmatch(relative.parts[1]) is not None) + if kind == 'stage': + try: + relative = path.relative_to(folder / STAGES) + except ValueError: + return False + identity, _, source = relative.parts[0].partition('-') if relative.parts else ('', '', '') + return (len(relative.parts) >= 2 and SOURCE.fullmatch(source) is not None + and any(row_kind == 'catalog' and row['sha256'][:16] == identity for row_kind, row in rows.values())) + return False + + +def stage_directory_permitted(directory, folder, rows): + """A staging folder of a catalog file of this variant, or a directory inside one.""" + if not directory.is_absolute() or '..' in directory.parts or Path(os.path.normpath(directory)) != directory: + return False + try: + relative = directory.relative_to(folder / STAGES) + except ValueError: + return False + if not relative.parts: + return False + identity, _, source = relative.parts[0].partition('-') + return (SOURCE.fullmatch(source) is not None + and any(kind == 'catalog' and row['sha256'][:16] == identity for kind, row in rows.values())) + + +def clean(plan, catalog=None, receipt=None): + """Remove exactly the planned files that are still unchanged; report everything else.""" + from .prepared_model import catalog as load_catalog + catalog = catalog or load_catalog() + root = Path(plan['root']) + removed, skipped, released = [], [], 0 + if plan.get('blockers'): + raise ValueError('Retired model cleanup is blocked: ' + ', '.join(plan['blockers'])) + with runtime_lock(root / 'setup.lock', inherit=False), \ + runtime_lock(root / 'engine.lock', inherit=False), \ + runtime_lock(root / 'launcher' / 'host-setup.lock', inherit=False): + active, _ = active_variant(root, catalog) + if (active is None or active != plan['active'] or not complete(root, catalog, active) + or (plan.get('expect_active') is not None and active != plan['expect_active'])): + raise ValueError('The model in use changed; check storage again.') + prepared = root / 'prepared' + for retired in plan['retired']: + name = retired['variant'] + if name == active or name not in catalog['variants']: + raise ValueError('Retired model plan names the model in use; nothing was removed.') + folder, rows = allowed(root, catalog, name) + if str(folder) != retired['folder'] or folder == variant_folder(root, catalog, active): + raise ValueError('Retired model folder changed; check storage again.') + directories = set() + for entry in retired['files']: + path = Path(entry['path']) + reason = None + if not permitted(path, entry['kind'], folder, rows, catalog['variants'][name]['cache_prefix']): + reason = 'not a planned name' + elif not inside(prepared, path) or unshared_file(path) is None: + reason = 'link, junction, shared or missing' + else: + try: + if fingerprint(path) != entry['stamp']: + reason = 'changed since the scan' + except OSError: + reason = 'missing' + if reason: + skipped.append(dict(path=str(path), reason=reason)) + continue + try: + path.unlink() + except OSError as error: + skipped.append(dict(path=str(path), reason=type(error).__name__)) + continue + removed.append(dict(path=str(path), bytes=entry['bytes'], kind=entry['kind'])) + released += entry['bytes'] + current = path.parent + while current != folder and folder in current.parents: + directories.add(current) + current = current.parent + # Staging folders hold the SDK's own empty work directories; remove + # them only when empty, deepest first, each named under its stage. + for value in sorted(retired.get('stage_directories', []), key=lambda p: len(Path(p).parts), reverse=True): + directory = Path(value) + if not stage_directory_permitted(directory, folder, rows): + continue + try: + if inside(prepared, directory) and directory.is_dir() and not any(directory.iterdir()): + directory.rmdir() + except OSError: + pass + # Only directories that held removed files, deepest first, and only when empty. + for directory in sorted(directories, key=lambda p: len(p.parts), reverse=True) + [folder / STAGES, folder]: + try: + if inside(prepared, directory) and directory.is_dir() and not any(directory.iterdir()): + directory.rmdir() + except OSError: + pass + result = dict(released_bytes=released, removed_files=len(removed), skipped=skipped, + kept_bytes=plan.get('kept_bytes', 0), active=plan['active']) + if receipt is not None: + receipt = Path(receipt) + receipt.parent.mkdir(parents=True, exist_ok=True) + receipt.write_text(json.dumps(dict(result, removed=removed, finished=time.time()), indent=1), encoding='utf-8') + return result diff --git a/freevideo_engine/worker.py b/freevideo_engine/worker.py index e9669f3..9f119fe 100644 --- a/freevideo_engine/worker.py +++ b/freevideo_engine/worker.py @@ -112,8 +112,8 @@ def release(target): resident.gpu_budget = min(resident.gpu_budget, limit) save(request['metrics'], metrics) manifest = json.loads((Path(request['cache']) / 'manifest.json').read_text(encoding='utf-8')) - if manifest.get('precision') != 'fp8': - raise ValueError('The Engine requires an official FP8 cache') + if manifest.get('precision') not in ('fp8', 'int8', 'bf16'): + raise ValueError('The Engine requires an official FP8 or int8 cache') options, decoder = request['engine_options'], request['decoder_options'] resume = request.get('resume_decode') is not None if not resume and options.get('task', 't2va') != 't2va': diff --git a/web/updates.js b/web/updates.js index 10dcbce..5eb6306 100644 --- a/web/updates.js +++ b/web/updates.js @@ -40,7 +40,8 @@ export async function checkUpdates() { } value = next; offline = false; const phase = next.launcher?.phase || ''; - if (working.has(phase)) updating = true; + // Switching to a faster model also restarts ComfyUI once. + if (working.has(phase) || next.launcher?.model) updating = true; else if (updating && phase !== 'restarting') updating = false; publish(); // The server returns immediately while its initial network check runs. @@ -127,10 +128,15 @@ export function createUpdateNotice(cn) { ? t('After your task finishes, open the new FreeVideo.app to update.', '当前任务完成后,打开新版 FreeVideo.app 更新。') : t('After your task finishes, open the new FreeVideo.exe to update.', '当前任务完成后,打开新版 FreeVideo.exe 更新。'); const progress = launcher?.progress, percent = progress?.total ? ' ' + Math.floor(100 * progress.done / progress.total) + '%' : ''; - const busy = updating && (working.has(phase) || offline); + // The last model state seen stays while ComfyUI is away. + const model = launcher?.model || ''; + const busy = updating && (working.has(phase) || !!model || offline); let text = '', actions = false; if (busy) { - text = offline || phase === 'restarting' || phase === 'engine' + text = model === 'switching' || (model && offline) + ? t('Switching to the faster model… ComfyUI restarts once and this page reconnects by itself.', '正在切换到更快的模型…ComfyUI 会重启一次,页面会自动恢复。') + : model === 'waiting' ? t('The faster model is ready. FreeVideo switches after the current video; ComfyUI restarts once.', '更快的模型已下载好,当前视频生成完后自动切换,ComfyUI 会重启一次。') + : offline || phase === 'restarting' || phase === 'engine' ? t('Restarting FreeVideo to finish the update… this page refreshes by itself.', '正在重启 FreeVideo 完成更新…页面会自动刷新。') : phase === 'downloading' ? t('Downloading the update', '正在下载更新') + percent : phase === 'waiting' ? t('The update installs as soon as the current video finishes.', '当前视频生成完成后会自动更新。')