diff --git a/THIRD_PARTY_NOTICES.md b/THIRD_PARTY_NOTICES.md
index 9335523..1cd37ba 100644
--- a/THIRD_PARTY_NOTICES.md
+++ b/THIRD_PARTY_NOTICES.md
@@ -11,7 +11,11 @@ FreeVideo code uses [Apache-2.0](LICENSE). Its dependencies and models retain th
| [VDN-H3](https://github.com/OpenVDN/vdn-minimax-h3), [Diffusers](https://github.com/huggingface/diffusers), [SageAttention](https://github.com/thu-ml/SageAttention) | Apache-2.0 |
| [VDN-H3 and H3 model weights](https://huggingface.co/OpenVDN/vdn-minimax-h3-edge/blob/main/LICENSE) | MiniMax H3 Community License |
| [MiniMax H3 latent upscaler](https://huggingface.co/LBH-123-AI/Minimax_h3_latent_Upscaler) | [MIT](freevideo_engine/licenses/latent-upscaler-MIT.txt) |
+| [FreeToken](https://github.com/FlashML-org/FreeToken) prompt VLM operators, [Qwen3-VL-4B-Instruct](https://huggingface.co/Qwen/Qwen3-VL-4B-Instruct) optional weights | [Apache-2.0](freevideo_engine/licenses/FreeToken-Apache-2.0.txt) |
Original launcher license texts are included in Windows packages. Other runtime dependencies are listed in [constraints](constraints), with their license information in the installed packages.
`decode_stream.py`, `lora_cache.py` and `reference_sampler.py` include adaptations of upstream code; their source headers retain the attribution.
+
+`prompt_vlm/core.py` adapts FreeToken's merged MLP/QKV and bounded vision operators from revision
+`555efd89447232a555e05d187256b76a51c1aaa0`. Its header describes the changes and integration boundary.
diff --git a/__init__.py b/__init__.py
index a91861c..9615c5c 100644
--- a/__init__.py
+++ b/__init__.py
@@ -9,6 +9,8 @@ async def comfy_entrypoint():
register()
from .freevideo_engine.comfy_setup import register as setup_routes
setup_routes()
+ from .freevideo_engine.prompt_vlm.service import register as prompt_routes
+ prompt_routes()
from .freevideo_engine.comfy_launcher_api import register as launcher_routes
launcher_routes()
from .freevideo_engine.comfy_updates import register as update_routes
diff --git a/freevideo_engine/comfy_bridge.py b/freevideo_engine/comfy_bridge.py
index d75dcb3..0e76536 100644
--- a/freevideo_engine/comfy_bridge.py
+++ b/freevideo_engine/comfy_bridge.py
@@ -466,7 +466,8 @@ def engine_environment(root, source, environ=None):
def generate(prompt, width, height, seconds, seed, output_directory, *,
source=None, environ=None, metadata=None, progress=None, interrupted=None,
release_models=None, export_inputs=None, two_pass=True, encoder_prewarm=None,
- force_regenerate=False, base_steps=8, refine_steps=3, comfy_metadata=None):
+ force_regenerate=False, base_steps=8, refine_steps=3, comfy_metadata=None,
+ prompt_rewrite_report=None):
if type(two_pass) is not bool:
raise ValueError('Two-pass generation must be a boolean')
if type(force_regenerate) is not bool:
@@ -488,6 +489,9 @@ def generate(prompt, width, height, seconds, seed, output_directory, *,
run = Path(output_directory).resolve() / 'FreeVideo' / time.strftime('%Y-%m-%d', time.gmtime()) / uuid.uuid4().hex
run.mkdir(parents=True, exist_ok=False)
output = run / 'video.mp4'
+ if prompt_rewrite_report:
+ from .prompt_vlm.receipt import attach
+ attach(root, prompt_rewrite_report, run)
(run / 'prompt.txt').write_text(prompt, encoding='utf-8')
if metadata is not None:
save(run / 'workflow.json', metadata)
diff --git a/freevideo_engine/comfy_nodes.py b/freevideo_engine/comfy_nodes.py
index 5acc3ff..7645fe1 100644
--- a/freevideo_engine/comfy_nodes.py
+++ b/freevideo_engine/comfy_nodes.py
@@ -114,6 +114,12 @@ def release():
memory.unload_all_models()
memory.soft_empty_cache()
metadata = cls.hidden.extra_pnginfo or {}
+ workflow_nodes = (metadata.get('workflow') or {}).get('nodes', [])
+ properties = next((row.get('properties', {}) for row in workflow_nodes
+ if str(row.get('id')) == str(node_id)), {})
+ versions = properties.get('freevideo_prompt_versions', {})
+ rewrite_report = dict(report_id=properties.get('freevideo_prompt_report'),
+ selected=versions.get('selected') if versions.get('enabled') else 'original')
try:
from comfy.cli_args import args as comfy_args
embed = not comfy_args.disable_metadata
@@ -125,6 +131,7 @@ def release():
try:
output = comfy_bridge.generate(text, width, height, seconds, seed, output_root,
metadata=metadata.get('workflow', metadata), progress=progress, two_pass=two_pass,
+ prompt_rewrite_report=rewrite_report,
encoder_prewarm=prewarm, force_regenerate=force_regenerate,
base_steps=base_steps, refine_steps=refine_steps, comfy_metadata=graph,
interrupted=memory.throw_exception_if_processing_interrupted, release_models=release,
diff --git a/freevideo_engine/diagnostics.py b/freevideo_engine/diagnostics.py
index 449442d..4ae2566 100644
--- a/freevideo_engine/diagnostics.py
+++ b/freevideo_engine/diagnostics.py
@@ -25,6 +25,7 @@
TOTAL_LIMIT = 12 * 1024 * 1024
FILE_COUNT = 256
NAMES = {'report.json', 'report.csv', 'inventory.json', 'case.json', 'status.json',
+ 'prompt-rewrite.json',
'plan.json', 'machine.before.json', 'comparison.json', 'profile.json',
'result.json', 'gpu.csv', 'ram.jsonl', 'network.jsonl', 'optimization.json', 'storage.json', 'kernel-capabilities.json'}
SENSITIVE = re.compile(r'(?:^|[_-])(?:token|password|secret|credential|api[_-]?key)(?:$|[_-])', re.I)
@@ -254,6 +255,14 @@ def add_path(name, path):
kernels = root / 'kernel-capabilities.json'
if kernels.is_file():
add_path('installation/kernel-capabilities.json', kernels)
+ prompt_report = root / 'prompt-vlm-report.json'
+ if prompt_report.is_file():
+ add_path('installation/prompt-vlm-report.json', prompt_report)
+ prompt_runs = root / 'prompt-vlm-runs'
+ if prompt_runs.is_dir() and not is_link(prompt_runs):
+ for path in sorted(prompt_runs.glob('*/prompt-rewrite.json'), key=lambda p: p.stat().st_mtime, reverse=True):
+ if not is_link(path) and not is_link(path.parent):
+ add_path('prompt-vlm/' + path.parent.name + '.json', path)
if tuning_state.is_file():
add_path('installation/tuning.json', tuning_state)
download_ownership = root / '.freevideo/downloaded-models.json'
diff --git a/freevideo_engine/licenses/FreeToken-Apache-2.0.txt b/freevideo_engine/licenses/FreeToken-Apache-2.0.txt
new file mode 100644
index 0000000..9eeb151
--- /dev/null
+++ b/freevideo_engine/licenses/FreeToken-Apache-2.0.txt
@@ -0,0 +1,201 @@
+ Apache License
+ Version 2.0, January 2004
+ http://www.apache.org/licenses/
+
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
+
+ 1. Definitions.
+
+ "License" shall mean the terms and conditions for use, reproduction,
+ and distribution as defined by Sections 1 through 9 of this document.
+
+ "Licensor" shall mean the copyright owner or entity authorized by
+ the copyright owner that is granting the License.
+
+ "Legal Entity" shall mean the union of the acting entity and all
+ other entities that control, are controlled by, or are under common
+ control with that entity. For the purposes of this definition,
+ "control" means (i) the power, direct or indirect, to cause the
+ direction or management of such entity, whether by contract or
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
+ outstanding shares, or (iii) beneficial ownership of such entity.
+
+ "You" (or "Your") shall mean an individual or Legal Entity
+ exercising permissions granted by this License.
+
+ "Source" form shall mean the preferred form for making modifications,
+ including but not limited to software source code, documentation
+ source, and configuration files.
+
+ "Object" form shall mean any form resulting from mechanical
+ transformation or translation of a Source form, including but
+ not limited to compiled object code, generated documentation,
+ and conversions to other media types.
+
+ "Work" shall mean the work of authorship, whether in Source or
+ Object form, made available under the License, as indicated by a
+ copyright notice that is included in or attached to the work
+ (an example is provided in the Appendix below).
+
+ "Derivative Works" shall mean any work, whether in Source or Object
+ form, that is based on (or derived from) the Work and for which the
+ editorial revisions, annotations, elaborations, or other modifications
+ represent, as a whole, an original work of authorship. For the purposes
+ of this License, Derivative Works shall not include works that remain
+ separable from, or merely link (or bind by name) to the interfaces of,
+ the Work and Derivative Works thereof.
+
+ "Contribution" shall mean any work of authorship, including
+ the original version of the Work and any modifications or additions
+ to that Work or Derivative Works thereof, that is intentionally
+ submitted to Licensor for inclusion in the Work by the copyright owner
+ or by an individual or Legal Entity authorized to submit on behalf of
+ the copyright owner. For the purposes of this definition, "submitted"
+ means any form of electronic, verbal, or written communication sent
+ to the Licensor or its representatives, including but not limited to
+ communication on electronic mailing lists, source code control systems,
+ and issue tracking systems that are managed by, or on behalf of, the
+ Licensor for the purpose of discussing and improving the Work, but
+ excluding communication that is conspicuously marked or otherwise
+ designated in writing by the copyright owner as "Not a Contribution."
+
+ "Contributor" shall mean Licensor and any individual or Legal Entity
+ on behalf of whom a Contribution has been received by Licensor and
+ subsequently incorporated within the Work.
+
+ 2. Grant of Copyright License. Subject to the terms and conditions of
+ this License, each Contributor hereby grants to You a perpetual,
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
+ copyright license to reproduce, prepare Derivative Works of,
+ publicly display, publicly perform, sublicense, and distribute the
+ Work and such Derivative Works in Source or Object form.
+
+ 3. Grant of Patent License. Subject to the terms and conditions of
+ this License, each Contributor hereby grants to You a perpetual,
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
+ (except as stated in this section) patent license to make, have made,
+ use, offer to sell, sell, import, and otherwise transfer the Work,
+ where such license applies only to those patent claims licensable
+ by such Contributor that are necessarily infringed by their
+ Contribution(s) alone or by combination of their Contribution(s)
+ with the Work to which such Contribution(s) was submitted. If You
+ institute patent litigation against any entity (including a
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
+ or a Contribution incorporated within the Work constitutes direct
+ or contributory patent infringement, then any patent licenses
+ granted to You under this License for that Work shall terminate
+ as of the date such litigation is filed.
+
+ 4. Redistribution. You may reproduce and distribute copies of the
+ Work or Derivative Works thereof in any medium, with or without
+ modifications, and in Source or Object form, provided that You
+ meet the following conditions:
+
+ (a) You must give any other recipients of the Work or
+ Derivative Works a copy of this License; and
+
+ (b) You must cause any modified files to carry prominent notices
+ stating that You changed the files; and
+
+ (c) You must retain, in the Source form of any Derivative Works
+ that You distribute, all copyright, patent, trademark, and
+ attribution notices from the Source form of the Work,
+ excluding those notices that do not pertain to any part of
+ the Derivative Works; and
+
+ (d) If the Work includes a "NOTICE" text file as part of its
+ distribution, then any Derivative Works that You distribute must
+ include a readable copy of the attribution notices contained
+ within such NOTICE file, excluding those notices that do not
+ pertain to any part of the Derivative Works, in at least one
+ of the following places: within a NOTICE text file distributed
+ as part of the Derivative Works; within the Source form or
+ documentation, if provided along with the Derivative Works; or,
+ within a display generated by the Derivative Works, if and
+ wherever such third-party notices normally appear. The contents
+ of the NOTICE file are for informational purposes only and
+ do not modify the License. You may add Your own attribution
+ notices within Derivative Works that You distribute, alongside
+ or as an addendum to the NOTICE text from the Work, provided
+ that such additional attribution notices cannot be construed
+ as modifying the License.
+
+ You may add Your own copyright statement to Your modifications and
+ may provide additional or different license terms and conditions
+ for use, reproduction, or distribution of Your modifications, or
+ for any such Derivative Works as a whole, provided Your use,
+ reproduction, and distribution of the Work otherwise complies with
+ the conditions stated in this License.
+
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
+ any Contribution intentionally submitted for inclusion in the Work
+ by You to the Licensor shall be under the terms and conditions of
+ this License, without any additional terms or conditions.
+ Notwithstanding the above, nothing herein shall supersede or modify
+ the terms of any separate license agreement you may have executed
+ with Licensor regarding such Contributions.
+
+ 6. Trademarks. This License does not grant permission to use the trade
+ names, trademarks, service marks, or product names of the Licensor,
+ except as required for reasonable and customary use in describing the
+ origin of the Work and reproducing the content of the NOTICE file.
+
+ 7. Disclaimer of Warranty. Unless required by applicable law or
+ agreed to in writing, Licensor provides the Work (and each
+ Contributor provides its Contributions) on an "AS IS" BASIS,
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
+ implied, including, without limitation, any warranties or conditions
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
+ PARTICULAR PURPOSE. You are solely responsible for determining the
+ appropriateness of using or redistributing the Work and assume any
+ risks associated with Your exercise of permissions under this License.
+
+ 8. Limitation of Liability. In no event and under no legal theory,
+ whether in tort (including negligence), contract, or otherwise,
+ unless required by applicable law (such as deliberate and grossly
+ negligent acts) or agreed to in writing, shall any Contributor be
+ liable to You for damages, including any direct, indirect, special,
+ incidental, or consequential damages of any character arising as a
+ result of this License or out of the use or inability to use the
+ Work (including but not limited to damages for loss of goodwill,
+ work stoppage, computer failure or malfunction, or any and all
+ other commercial damages or losses), even if such Contributor
+ has been advised of the possibility of such damages.
+
+ 9. Accepting Warranty or Additional Liability. While redistributing
+ the Work or Derivative Works thereof, You may choose to offer,
+ and charge a fee for, acceptance of support, warranty, indemnity,
+ or other liability obligations and/or rights consistent with this
+ License. However, in accepting such obligations, You may act only
+ on Your own behalf and on Your sole responsibility, not on behalf
+ of any other Contributor, and only if You agree to indemnify,
+ defend, and hold each Contributor harmless for any liability
+ incurred by, or claims asserted against, such Contributor by reason
+ of your accepting any such warranty or additional liability.
+
+ END OF TERMS AND CONDITIONS
+
+ APPENDIX: How to apply the Apache License to your work.
+
+ To apply the Apache License to your work, attach the following
+ boilerplate notice, with the fields enclosed by brackets "[]"
+ replaced with your own identifying information. (Don't include
+ the brackets!) The text should be enclosed in the appropriate
+ comment syntax for the file format. We also recommend that a
+ file or class name and description of purpose be included on the
+ same "printed page" as the copyright notice for easier
+ identification within third-party archives.
+
+ Copyright 2026 FreeToken Authors
+
+ Licensed under the Apache License, Version 2.0 (the "License");
+ you may not use this file except in compliance with the License.
+ You may obtain a copy of the License at
+
+ http://www.apache.org/licenses/LICENSE-2.0
+
+ Unless required by applicable law or agreed to in writing, software
+ distributed under the License is distributed on an "AS IS" BASIS,
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ See the License for the specific language governing permissions and
+ limitations under the License.
diff --git a/freevideo_engine/prompt_vlm/__init__.py b/freevideo_engine/prompt_vlm/__init__.py
new file mode 100644
index 0000000..5fc44cc
--- /dev/null
+++ b/freevideo_engine/prompt_vlm/__init__.py
@@ -0,0 +1 @@
+"""Optional local prompt enhancement. Importing this package never loads a model."""
diff --git a/freevideo_engine/prompt_vlm/assets.py b/freevideo_engine/prompt_vlm/assets.py
new file mode 100644
index 0000000..1395016
--- /dev/null
+++ b/freevideo_engine/prompt_vlm/assets.py
@@ -0,0 +1,74 @@
+"""Explicit, resumable installation through FreeVideo's download routing."""
+import json
+from pathlib import Path
+import shutil
+
+
+def catalog():
+ return json.loads(Path(__file__).with_name('model.json').read_text(encoding='utf-8'))
+
+
+def directory(root):
+ spec = catalog()
+ return Path(root) / 'models' / 'prompt-vlm' / spec['revision']
+
+
+def ready(root):
+ spec, folder = catalog(), directory(root)
+ try:
+ receipt = json.loads((folder / 'verified.json').read_text(encoding='utf-8'))
+ return receipt.get('revision') == spec['revision'] and all(
+ (folder / row['file']).stat().st_size == row['bytes'] and
+ receipt['files'].get(row['file']) == (folder / row['file']).stat().st_mtime_ns
+ for row in spec['files'])
+ except (OSError, ValueError, KeyError, TypeError):
+ return False
+
+
+def install(root, progress, check):
+ from .. import network, provision
+ from ..adaln_assets import _download_plan
+ from ..monitoring import save
+ if ready(root):
+ return
+ folder, spec = directory(root), catalog()
+ folder.mkdir(parents=True, exist_ok=True)
+ ledger_path = folder / 'download-verified.json'
+ try:
+ ledger = json.loads(ledger_path.read_text(encoding='utf-8'))
+ except (OSError, ValueError):
+ ledger = {}
+ missing = [row for row in spec['files'] if not provision.verified(folder / row['file'], row, ledger)]
+ total = sum(row['bytes'] for row in missing)
+ retained = 0
+ for row in missing:
+ path = folder / row['file']
+ partial = path.with_suffix(path.suffix + '.partial')
+ if partial.is_file():
+ size = partial.stat().st_size
+ if size < row['bytes']:
+ retained += size
+ if shutil.disk_usage(folder).free < total - retained + 256 * 2**20:
+ raise ValueError('disk_space')
+ plan = _download_plan(root)
+ # This catalog has no verified ModelScope mirror yet. Preserve configured
+ # routing, but never guess a mirror revision or send credentials to it.
+ plan['sources'] = dict(plan.get('sources', {}), models=[
+ row for row in plan.get('sources', {}).get('models', []) if row['id'] in ('official', 'hf-mirror')])
+ if not plan['sources']['models']:
+ plan['sources']['models'] = [{'id': 'official'}, {'id': 'hf-mirror'}]
+ plan.update(quiet=True, resource_check=check)
+ done = 0
+ for row in missing:
+ check()
+ path = folder / row['file']
+ remote = dict(repo=spec['repo'], revision=spec['revision'], file=row['file'])
+ def moved(current, size, speed, **kwargs):
+ progress(dict(phase='download', done=done + current, total=total))
+ network.download(network.model_urls(plan, remote), path, row['sha256'], moved,
+ network=plan, size=row['bytes'], keep_partial=True, stall_seconds=30)
+ ledger[str(path)] = provision.file_identity(path, row)
+ save(ledger_path, ledger)
+ done += row['bytes']
+ save(folder / 'verified.json', dict(revision=spec['revision'], files={
+ row['file']: (folder / row['file']).stat().st_mtime_ns for row in spec['files']}))
diff --git a/freevideo_engine/prompt_vlm/core.py b/freevideo_engine/prompt_vlm/core.py
new file mode 100644
index 0000000..81faa48
--- /dev/null
+++ b/freevideo_engine/prompt_vlm/core.py
@@ -0,0 +1,114 @@
+"""Portable single-request subset adapted from FreeToken (Apache-2.0).
+
+Source: FlashML-org/FreeToken, 555efd89447232a555e05d187256b76a51c1aaa0,
+models/blocks.py GatedMLP, models/qwen3/attention.py merged QKV,
+models/qwen3_vl/vision.py VisionAttention/VisionMLP.
+Changes: nn.Module weights, single-device linears, Torch SiLU/SDPA, and the
+Transformers vision-call signature. No server, scheduler or global context.
+Tokenizer, checkpoint loading, language attention/KV cache and decoding remain
+Transformers implementations; this is not the complete FreeToken engine.
+"""
+import types
+
+import torch
+from torch import nn
+from torch.nn import functional as F
+
+
+class GatedMLP(nn.Module):
+ def __init__(self, source):
+ super().__init__()
+ self.gate_up_proj = nn.Linear(source.gate_proj.in_features,
+ 2 * source.gate_proj.out_features, bias=False,
+ device='meta', dtype=source.gate_proj.weight.dtype)
+ self.gate_up_proj.weight = nn.Parameter(torch.cat(
+ (source.gate_proj.weight, source.up_proj.weight)), requires_grad=False)
+ self.down_proj = source.down_proj
+
+ def forward(self, x):
+ gate_up = self.gate_up_proj(x)
+ del x
+ gate, up = gate_up.chunk(2, dim=-1)
+ y = F.silu(gate) * up
+ del gate, up, gate_up
+ return self.down_proj(y)
+
+
+def text_attention(self, hidden_states, position_embeddings, attention_mask,
+ past_key_values=None, **kwargs):
+ # FreeToken's merged QKV, retaining Transformers' rotary, cache and SDPA
+ # contract. No new kernels, JIT compiler, persistent cache or CUDA graphs.
+ from transformers.models.qwen3_vl.modeling_qwen3_vl import (
+ apply_rotary_pos_emb, ALL_ATTENTION_FUNCTIONS, eager_attention_forward)
+ shape = hidden_states.shape[:-1]
+ head_shape = (*shape, -1, self.head_dim)
+ q, k, v = self.qkv_proj(hidden_states).split(self.qkv_sizes, dim=-1)
+ q = self.q_norm(q.reshape(head_shape)).transpose(1, 2)
+ k = self.k_norm(k.reshape(head_shape)).transpose(1, 2)
+ v = v.reshape(head_shape).transpose(1, 2)
+ q, k = apply_rotary_pos_emb(q, k, *position_embeddings)
+ if past_key_values is not None:
+ k, v = past_key_values.update(k, v, self.layer_idx)
+ attention = ALL_ATTENTION_FUNCTIONS.get_interface(self.config._attn_implementation,
+ eager_attention_forward)
+ out, weights = attention(self, q, k, v, attention_mask,
+ dropout=self.attention_dropout if self.training else 0.,
+ scaling=self.scaling, **kwargs)
+ return self.o_proj(out.reshape(*shape, -1).contiguous()), weights
+
+
+def merge_qkv(attention):
+ projections = [attention.q_proj, attention.k_proj, attention.v_proj]
+ attention.qkv_sizes = [p.out_features for p in projections]
+ merged = nn.Linear(projections[0].in_features, sum(attention.qkv_sizes),
+ bias=projections[0].bias is not None, device='meta',
+ dtype=projections[0].weight.dtype)
+ merged.weight = nn.Parameter(torch.cat([p.weight for p in projections]), requires_grad=False)
+ if merged.bias is not None:
+ merged.bias = nn.Parameter(torch.cat([p.bias for p in projections]), requires_grad=False)
+ attention.qkv_proj = merged
+ del attention.q_proj, attention.k_proj, attention.v_proj
+ attention.forward = types.MethodType(text_attention, attention)
+
+
+def vision_attention(self, hidden_states, cu_seqlens, position_embeddings=None, **kwargs):
+ # Per-image bidirectional attention: unrelated references cannot attend
+ # across image boundaries. Match FreeToken's fp32 rotary arithmetic.
+ size = hidden_states.shape[0]
+ q, k, v = self.qkv(hidden_states).view(size, 3, self.num_heads, self.head_dim).unbind(1)
+ cos, sin = (x.unsqueeze(-2).float() for x in position_embeddings)
+ def rotate(x):
+ x = x.float()
+ a, b = x.chunk(2, dim=-1)
+ return x * cos + torch.cat((-b, a), dim=-1) * sin
+ q, k = rotate(q).to(q.dtype), rotate(k).to(k.dtype)
+ q, k, v = (x.transpose(0, 1).unsqueeze(0) for x in (q, k, v))
+ lengths = torch.diff(cu_seqlens).tolist()
+ outs = [F.scaled_dot_product_attention(qs, ks, vs)
+ for qs, ks, vs in zip(torch.split(q, lengths, dim=2),
+ torch.split(k, lengths, dim=2), torch.split(v, lengths, dim=2))]
+ output = outs[0] if len(outs) == 1 else torch.cat(outs, dim=2)
+ return self.proj(output[0].transpose(0, 1).reshape(size, -1))
+
+
+def vision_mlp(self, x):
+ if x.shape[0] <= 4096:
+ return self.linear_fc2(F.gelu(self.linear_fc1(x), approximate='tanh'))
+ out = torch.empty_like(x)
+ for lo in range(0, x.shape[0], 4096):
+ rows = x[lo:lo + 4096]
+ out[lo:lo + 4096] = self.linear_fc2(F.gelu(self.linear_fc1(rows), approximate='tanh'))
+ return out
+
+
+def prepare(model):
+ """Transform on CPU before accelerate installs placement/offload hooks."""
+ if model.config.model_type != 'qwen3_vl' or model.config.text_config.hidden_act != 'silu':
+ raise ValueError('Unsupported prompt model architecture')
+ for layer in model.model.language_model.layers:
+ layer.mlp = GatedMLP(layer.mlp)
+ merge_qkv(layer.self_attn)
+ for block in model.model.visual.blocks:
+ block.attn.forward = types.MethodType(vision_attention, block.attn)
+ block.mlp.forward = types.MethodType(vision_mlp, block.mlp)
+ return model
diff --git a/freevideo_engine/prompt_vlm/model.json b/freevideo_engine/prompt_vlm/model.json
new file mode 100644
index 0000000..8e59980
--- /dev/null
+++ b/freevideo_engine/prompt_vlm/model.json
@@ -0,0 +1,67 @@
+{
+ "repo": "Qwen/Qwen3-VL-4B-Instruct",
+ "revision": "ebb281ec70b05090aa6165b016eac8ec08e71b17",
+ "license": "Apache-2.0",
+ "files": [
+ {
+ "file": "chat_template.json",
+ "bytes": 5502,
+ "sha256": "6f8a6a55027e3da5160105556cda5dd69f6423f1c32645f6730d32de7773d0c4"
+ },
+ {
+ "file": "config.json",
+ "bytes": 1505,
+ "sha256": "edac7703329133edfc53e46ac0081835144c99d7eebf28b71c732694d435224d"
+ },
+ {
+ "file": "generation_config.json",
+ "bytes": 269,
+ "sha256": "8469742d1fce0de951c8909b26a2c0c0d8490837ce476efb114da9e0cefc4d44"
+ },
+ {
+ "file": "merges.txt",
+ "bytes": 1671839,
+ "sha256": "599bab54075088774b1733fde865d5bd747cbcc7a547c5bc12610e874e26f5e3"
+ },
+ {
+ "file": "model-00001-of-00002.safetensors",
+ "bytes": 4967229296,
+ "sha256": "30a01a0556622645a3cce87b655bbbbbc1f170c196099f1b666c93202c3339a9"
+ },
+ {
+ "file": "model-00002-of-00002.safetensors",
+ "bytes": 3908490048,
+ "sha256": "046296a2a387efb43b0c997d5833c789604d168834f6e0d3064bf7bb13d002a6"
+ },
+ {
+ "file": "model.safetensors.index.json",
+ "bytes": 64742,
+ "sha256": "58a7841d7bff2548dd91577d216274a83cf1b500bc6a534b809d6c1b1707cf2b"
+ },
+ {
+ "file": "preprocessor_config.json",
+ "bytes": 390,
+ "sha256": "27225450ac9c6529872ee1924fcb0962ff5634834f817040f444118116f4e516"
+ },
+ {
+ "file": "tokenizer.json",
+ "bytes": 7032403,
+ "sha256": "a5d85b6dcc535e6b93115a9ef287e6132fdbf30270da6218194ba742261173c7"
+ },
+ {
+ "file": "tokenizer_config.json",
+ "bytes": 10868,
+ "sha256": "c2da771801886ad9ae98181793ffd3dfb7f1af30f6f7c6a4e15d7dbba52e2399"
+ },
+ {
+ "file": "video_preprocessor_config.json",
+ "bytes": 385,
+ "sha256": "7768af27c1fafa9cc9011c1dc20067e03f8915e03b63504550e11d5066986d13"
+ },
+ {
+ "file": "vocab.json",
+ "bytes": 2776833,
+ "sha256": "ca10d7e9fb3ed18575dd1e277a2579c16d108e32f27439684afa0e10b1440910"
+ }
+ ]
+}
diff --git a/freevideo_engine/prompt_vlm/receipt.py b/freevideo_engine/prompt_vlm/receipt.py
new file mode 100644
index 0000000..2219182
--- /dev/null
+++ b/freevideo_engine/prompt_vlm/receipt.py
@@ -0,0 +1,26 @@
+"""Attach the selected rewrite receipt, never the installation's latest job."""
+import json
+from pathlib import Path
+import re
+
+
+def attach(root, identity, run):
+ from ..diagnostics import is_link, read_bounded
+ from ..monitoring import save
+ selected = identity.get('selected') if isinstance(identity, dict) else None
+ identity = identity.get('report_id') if isinstance(identity, dict) else identity
+ if not isinstance(identity, str) or not re.fullmatch('[a-f0-9]{32}', identity):
+ return
+ parent = Path(root) / 'prompt-vlm-runs'
+ path = parent / identity / 'prompt-rewrite.json'
+ if is_link(parent) or is_link(path.parent) or is_link(path):
+ return
+ try:
+ raw, truncated = read_bounded(path, 128 * 1024)
+ row = json.loads(raw)
+ if truncated or row.get('schema') != 'freevideo.prompt-vlm' or row.get('report_id') != identity:
+ return
+ row['video_input_version'] = selected if selected in ('original', 'rewritten') else 'unknown'
+ save(Path(run) / 'prompt-rewrite.json', row)
+ except (OSError, ValueError, AttributeError):
+ pass # Missing historical diagnostics must never block video generation.
diff --git a/freevideo_engine/prompt_vlm/rules.py b/freevideo_engine/prompt_vlm/rules.py
new file mode 100644
index 0000000..d29f594
--- /dev/null
+++ b/freevideo_engine/prompt_vlm/rules.py
@@ -0,0 +1,106 @@
+"""MiniMax H3 rewrite contract, derived from the official writing guides.
+
+https://github.com/MiniMax-AI/MiniMax-H3/tree/main/skills/h3-prompt-writing
+The public guide describes a prompt format, not a released Context-IR model.
+"""
+import math
+import re
+
+BASE_FIELDS = ('integrated_multimodal_description', 'overall_soundscape', 'non_diegetic_music')
+REF_FIELDS = ('subject_definitions', 'summary', 'retention_analysis', 'detailed_description',
+ 'overall_soundscape', 'non_diegetic_music')
+
+
+def request(value):
+ if not isinstance(value, dict):
+ raise ValueError('invalid_request')
+ prompt, seconds, media = value.get('text'), value.get('seconds'), value.get('media', [])
+ if not isinstance(prompt, str) or not prompt.strip() or len(prompt) > 12000:
+ raise ValueError('invalid_prompt')
+ if type(seconds) not in (float, int) or not math.isfinite(seconds) or not 1 <= seconds <= 60:
+ raise ValueError('invalid_duration')
+ if not isinstance(media, list) or len(media) > 8:
+ raise ValueError('too_many_images')
+ roles = []
+ for row in media:
+ if (not isinstance(row, dict) or set(row) != {'file', 'role'} or
+ not isinstance(row['file'], str) or row['role'] not in ('first', 'last', 'reference')):
+ raise ValueError('unsupported_media')
+ roles.append(row['role'])
+ if 'reference' in roles and any(r in roles for r in ('first', 'last')):
+ raise ValueError('mixed_media')
+ if roles.count('first') > 1 or roles.count('last') > 1:
+ raise ValueError('mixed_media')
+ mode = ('ref2va' if 'reference' in roles else 'fl2va' if 'first' in roles and 'last' in roles
+ else 'i2va' if 'first' in roles else 'l2va' if 'last' in roles else 't2va')
+ # Anchor numbering must match the encoder even when the UI added last first.
+ if mode != 'ref2va':
+ media = sorted(media, key=lambda r: r['role'] != 'first')
+ return dict(text=prompt, seconds=float(seconds), media=media, mode=mode)
+
+
+def instruction(value):
+ rule = """Rewrite the user scene as an English MiniMax H3 video prompt. Follow the requested
+actions and camera movement exactly. Image references describe appearance, not whether an
+object moves. Keep dialogue, lyrics and visible writing verbatim in their original language.
+Add concrete motion, framing, lighting and sounds without changing the intended scene.
+Do not invent dialogue, captions or music. Return only the sections listed below as plain
+headings followed by colons. No preface, commentary, markdown fences or JSON.
+Use [Shot 1] for the opening; if the user wants cuts, add [Shot 2] At MM:SS.mmm, etc.
+Keep unbroken shots unbroken. Literal dialogue uses (S1) [Language] speech.
+"""
+ if value['mode'] == 'ref2va':
+ rule += """subject_definitions: Name each visible subject literally , , etc.,
+with its appearance and source . Preserve the supplied picture numbering.
+summary: [reference generation] State what happens in the requested video.
+retention_analysis: Mark each subject fully_preserved, partially_preserved, attribute_transfer,
+or weak_reference. This describes retained visual attributes, not motion restrictions.
+detailed_description: [Shot 1] Describe the target video, using identifiers.
+"""
+ else:
+ rule += 'integrated_multimodal_description: [Shot 1] Describe the target video.\n'
+ for i, row in enumerate(value['media'], 1):
+ timestamp = '0.00' if row['role'] == 'first' else f'{value["seconds"]:.2f}'
+ rule += (f'Anchor to the {row["role"]} frame at {timestamp} seconds; '
+ 'keep its pictured layout at that anchor.\n')
+ rule += """overall_soundscape: Environmental sounds, or N/A.
+non_diegetic_music: Music only if requested, otherwise N/A.
+Use only supplied media labels; do not invent audio or video references.
+"""
+ fields = REF_FIELDS if value['mode'] == 'ref2va' else BASE_FIELDS
+ return (rule + f'Cover {value["seconds"]:.3f} seconds. All sections are required, in this order: '
+ + ', '.join(fields) + f'. Begin with "{fields[0]}:".')
+
+
+def validate_output(text, value):
+ if not isinstance(text, str) or not 100 <= len(text.strip()) <= 20000:
+ raise ValueError('invalid_output')
+ text = text.strip()
+ if text.startswith('```') and text.endswith('```'):
+ text = '\n'.join(text.splitlines()[1:-1]).strip()
+ text = re.sub(r'(?m)^#{1,3} +([a-z_]+):', r'\1:', text)
+ fields = REF_FIELDS if value['mode'] == 'ref2va' else BASE_FIELDS
+ found = re.findall(r'(?m)^([a-z_]+):', text)
+ if found != list(fields) or not text.startswith(fields[0] + ':'):
+ raise ValueError('invalid_output')
+ if not re.search(r'\[Shot \d+\]', text):
+ name = 'detailed_description' if value['mode'] == 'ref2va' else fields[0]
+ text = text.replace(name + ':', name + ': [Shot 1]', 1)
+ if '[Shot 1]' not in text:
+ raise ValueError('invalid_output')
+ for marker in fields:
+ section = text.split(marker + ':', 1)[1].split('\n\n', 1)[0].strip()
+ if not section:
+ raise ValueError('invalid_output')
+ for minutes, seconds in re.findall(r'\bAt (\d{2}):(\d{2}\.\d{3})', text):
+ if int(minutes) * 60 + float(seconds) >= value['seconds']:
+ raise ValueError('invalid_output')
+ for index in re.findall(r'', text):
+ if not 1 <= int(index) <= len(value['media']):
+ raise ValueError('invalid_output')
+ if re.search(r'<(?:Audio|Video) \d+>', text):
+ raise ValueError('invalid_output')
+ for i in range(1, len(value['media']) + 1):
+ if f'' not in text:
+ raise ValueError('invalid_output')
+ return text
diff --git a/freevideo_engine/prompt_vlm/service.py b/freevideo_engine/prompt_vlm/service.py
new file mode 100644
index 0000000..e95c1e7
--- /dev/null
+++ b/freevideo_engine/prompt_vlm/service.py
@@ -0,0 +1,325 @@
+"""Optional rewrite jobs; prompts stay in memory and never enter job reports."""
+import atexit
+import json
+from pathlib import Path
+import secrets
+import subprocess
+import threading
+import time
+import uuid
+
+from . import assets, rules
+
+
+class Service:
+ def __init__(self, installation, input_directory):
+ self.installation = installation
+ self.input_directory = input_directory
+ self.mutex = threading.Lock()
+ self.stop = threading.Event()
+ self.state = {}
+ self.key = None
+ self.thread = None
+ self.process = None
+ self.result_expires = 0.
+ self.report = {}
+ atexit.register(self.close)
+
+ def info(self):
+ root, _ = self.installation()
+ spec = assets.catalog()
+ return dict(ready=assets.ready(root), model=spec['repo'], license=spec['license'],
+ bytes=sum(row['bytes'] for row in spec['files']), busy=self.busy())
+
+ def busy(self):
+ return self.thread is not None and self.thread.is_alive()
+
+ def check(self):
+ if self.stop.is_set():
+ raise InterruptedError('cancelled')
+
+ def progress(self, row):
+ with self.mutex:
+ self.state.update(row)
+
+ def start(self, action, value):
+ if action not in ('install', 'rewrite'):
+ raise ValueError('invalid_request')
+ if self.busy():
+ raise ValueError('busy')
+ root, machine = self.installation()
+ if action == 'install':
+ if value.get('accept_download') is not True:
+ raise ValueError('consent_required')
+ else:
+ if not assets.ready(root):
+ raise ValueError('model_missing')
+ value = rules.request(value)
+ from ..comfy_assets import resolve_assets
+ resolved = resolve_assets(json.dumps(value['media']), self.input_directory())
+ for i, row in enumerate(value['media']):
+ path = (resolved['references'][i]['path'] if row['role'] == 'reference'
+ else resolved[row['role']])
+ from ..comfy_assets import media_kind
+ if media_kind(path) != 'image':
+ raise ValueError('unsupported_media')
+ row['path'] = path
+ self.key = secrets.token_urlsafe(24)
+ self.stop.clear()
+ self.result_expires = 0.
+ self.state = dict(phase='starting')
+ self.report = {}
+ self.thread = threading.Thread(target=self.work, args=(root, machine, action, value),
+ name='freevideo-prompt-vlm', daemon=True)
+ self.thread.start()
+ return dict(job=self.key)
+
+ def status(self, key):
+ if not isinstance(key, str) or not self.key or not secrets.compare_digest(key, self.key):
+ raise ValueError('unknown_job')
+ with self.mutex:
+ if self.result_expires and time.monotonic() > self.result_expires:
+ self.state.pop('text', None)
+ self.state.update(phase='failed', error='expired')
+ row = dict(self.state)
+ # The browser may immediately queue video work or the next rewrite.
+ # Do not announce completion while we still hold the runtime lease.
+ if self.busy() and row.get('phase') in ('complete', 'failed', 'cancelled'):
+ row['phase'] = 'finishing'
+ row.pop('text', None)
+ return row
+
+ def cancel(self, key):
+ self.status(key)
+ self.stop.set()
+ return dict(phase='cancelling')
+
+ def close(self):
+ self.stop.set()
+ process = self.process
+ if process is not None and process.poll() is None:
+ from .. import processes
+ processes.stop(process, grace=5)
+
+ def work(self, root, machine, action, value):
+ from ..locking import runtime_lock
+ from ..monitoring import save
+ started = time.monotonic()
+ try:
+ if action == 'rewrite':
+ from ..resident_process import OWNER
+ if OWNER.active:
+ raise BlockingIOError('busy')
+ OWNER.stop_prewarm()
+ from ..encoder_prewarm import IDLE
+ IDLE.stop()
+ with runtime_lock(Path(root) / 'engine.lock', inherit=False) as descriptor:
+ self.check()
+ if action == 'install':
+ assets.install(root, self.progress, self.check)
+ self.check()
+ self.progress(dict(phase='complete'))
+ else:
+ self.infer(root, machine, descriptor, value)
+ except InterruptedError:
+ self.progress(dict(phase='cancelled'))
+ except BlockingIOError:
+ self.progress(dict(phase='failed', error='busy'))
+ except Exception as error:
+ known = {'disk_space', 'ram_space', 'rewrite_failed', 'worker_exit', 'timeout', 'worker_cleanup_failed'}
+ self.progress(dict(phase='failed', error=str(error) if str(error) in known
+ else 'download_failed' if action == 'install' else 'rewrite_failed',
+ exception=type(error).__name__))
+ finally:
+ if self.stop.is_set():
+ self.progress(dict(phase='cancelled'))
+ self.result_expires = time.monotonic() + 600
+ # Allowlisted metadata only, never original/rewritten text or input paths.
+ report = {key: item for key, item in self.state.items()
+ if key in ('phase', 'error', 'exception', 'error_details', 'timings', 'diagnostics',
+ 'resources', 'cleanup', 'worker_returncode')}
+ spec = assets.catalog()
+ report.update(schema='freevideo.prompt-vlm', schema_version=2, action=action,
+ created_epoch=time.time(), report_id=uuid.uuid4().hex,
+ elapsed_seconds=time.monotonic() - started, model=spec['repo'],
+ model_revision=spec['revision'])
+ self.report = report
+ try:
+ directory = Path(root) / 'prompt-vlm-runs' / report['report_id']
+ save(directory / 'prompt-rewrite.json', report)
+ save(Path(root) / 'prompt-vlm-report.json', report)
+ if action == 'rewrite':
+ from ..comfy_progress import REPORTS
+ self.progress(dict(rewrite_report_id=report['report_id'],
+ report_id=REPORTS.register(directory / 'video.mp4')))
+ except OSError:
+ pass
+
+ def infer(self, root, machine, descriptor, value):
+ from .. import processes
+ from ..comfy_environment import isolated_environment
+ from ..comfy_bridge import source_root
+ from .. import resident_process
+ from ..ram import ProcessMemory
+ import psutil
+ env = isolated_environment(root, source_root())
+ owner = resident_process.OWNER
+ if owner.process is not None and owner.process.poll() is None and owner.endpoint:
+ env[resident_process.ENV] = str(owner.endpoint)
+ if env.get(resident_process.ENV) and Path(env[resident_process.ENV]).is_file():
+ resident_process.release_idle_cache(env, descriptor)
+ command = processes.module_command('freevideo_engine.prompt_vlm.worker', assets.directory(root))
+ command[0] = machine['comfy_python']
+ child = processes.popen(command, env=env, stdin=subprocess.PIPE, stdout=subprocess.PIPE,
+ stderr=subprocess.PIPE, text=True, encoding='utf-8', errors='replace',
+ pass_fds=(descriptor,), supervise=True, start_new_session=True)
+ self.process = child
+ memory = ProcessMemory()
+ memory_errors = 0
+ next_sample = 0.
+ observed = {}
+ result = {}
+ result_received_at = []
+ native = dict(stderr_characters=0, out_of_memory=False, cuda_error=False, fatal_python_error=False)
+ def read_errors():
+ # Native libraries can bypass the structured worker exception. Keep
+ # symptom flags, never stderr text (which can contain prompt pieces).
+ tail = ''
+ for chunk in iter(lambda: child.stderr.read(1024), ''):
+ native['stderr_characters'] += len(chunk)
+ content = (tail + chunk).lower()
+ for key, phrase in [('out_of_memory', 'out of memory'), ('cuda_error', 'cuda error'),
+ ('fatal_python_error', 'fatal python error')]:
+ native[key] |= phrase in content
+ tail = content[-32:]
+ def read():
+ for line in child.stdout:
+ if len(line) > 128 * 1024:
+ continue
+ try:
+ row = json.loads(line)
+ except ValueError:
+ continue
+ if row.get('phase') in ('complete', 'failed'):
+ result.update(row)
+ result_received_at[:] = [time.monotonic()]
+ elif row.get('phase') in ('loading', 'rewriting'):
+ self.progress(row)
+ reader = threading.Thread(target=read, daemon=True)
+ error_reader = threading.Thread(target=read_errors, daemon=True)
+ reader.start()
+ error_reader.start()
+ try:
+ child.stdin.write(json.dumps(value, ensure_ascii=True))
+ child.stdin.close()
+ deadline = time.monotonic() + 600
+ while child.poll() is None:
+ self.check()
+ if time.monotonic() > deadline:
+ raise TimeoutError('timeout')
+ if time.monotonic() >= next_sample:
+ try:
+ memory.sample(child.pid)
+ except (OSError, RuntimeError):
+ memory_errors += 1
+ # Track identity, not only PIDs, to verify descendants after
+ # cancellation. Observation failure is diagnostic, not fatal.
+ try:
+ for process in psutil.Process(child.pid).children(recursive=True):
+ observed[process.pid] = process
+ except psutil.Error:
+ memory_errors += 1
+ next_sample = time.monotonic() + .5
+ self.stop.wait(.1)
+ reader.join(timeout=5)
+ self.check()
+ if reader.is_alive() or not result or child.returncode and result.get('phase') == 'complete':
+ raise RuntimeError('worker_exit')
+ if result.get('phase') == 'complete':
+ rules.validate_output(result.get('text'), value)
+ self.progress(dict(result, worker_returncode=child.returncode))
+ finally:
+ releasing = result_received_at[0] if result_received_at else time.monotonic()
+ if child.poll() is None:
+ # Killing only the Linux supervisor bypasses its descendant
+ # cleanup. Stop the owned group/Windows Job and wait instead.
+ processes.stop(child, grace=5)
+ child.wait()
+ reader.join(timeout=5)
+ error_reader.join(timeout=5)
+ child.stdout.close()
+ child.stderr.close()
+ surviving = []
+ for process in observed.values():
+ try:
+ if process.is_running() and process.status() != 'zombie':
+ surviving.append(process)
+ except psutil.NoSuchProcess:
+ pass
+ if surviving:
+ for process in surviving:
+ try:
+ process.kill()
+ except psutil.NoSuchProcess:
+ pass
+ _, surviving = psutil.wait_procs(surviving, timeout=5)
+ resource = {key: item for key, item in memory.result().items()
+ if item is None or type(item) in (int, float, bool) or key == 'ram_guard_metric'}
+ resource.update(sample_errors=memory_errors, poll_interval_seconds=.5,
+ scope='Sampled worker tree, not an enforced RAM limit', native_exit=native)
+ self.progress(dict(resources=resource, worker_returncode=child.returncode,
+ cleanup=dict(supervisor_exited=child.poll() is not None,
+ observed_descendants=len(observed), remaining_descendants=len(surviving),
+ seconds=time.monotonic() - releasing,
+ model_retained=bool(surviving), kv_retained=bool(surviving),
+ release_basis='Process exit; sampled descendant identities',
+ verified=not surviving)))
+ self.process = None
+ if surviving:
+ raise RuntimeError('worker_cleanup_failed')
+
+
+def register():
+ from aiohttp import web
+ from server import PromptServer
+ import folder_paths
+ from ..comfy_bridge import installation
+ server = PromptServer.instance
+ if server is None or getattr(server, '_freevideo_prompt_vlm', None) is not None:
+ return
+ service = Service(installation, folder_paths.get_input_directory)
+ server._freevideo_prompt_vlm = service
+ token = secrets.token_urlsafe(24)
+
+ @server.routes.get('/freevideo/prompt-vlm')
+ async def info(request):
+ try:
+ return web.json_response(dict(service.info(), token=token))
+ except (OSError, ValueError):
+ return web.json_response(dict(error='setup_required'), status=400)
+
+ @server.routes.post('/freevideo/prompt-vlm/{action}')
+ async def action(request):
+ if not secrets.compare_digest(request.headers.get('X-FreeVideo-Prompt', ''), token):
+ return web.json_response(dict(error='reload'), status=403)
+ try:
+ if request.content_length and request.content_length > 128 * 1024:
+ raise ValueError('invalid_request')
+ value = await request.json()
+ if not isinstance(value, dict):
+ raise ValueError('invalid_request')
+ name = request.match_info['action']
+ if name == 'status':
+ result = service.status(value.get('job'))
+ elif name == 'cancel':
+ result = service.cancel(value.get('job'))
+ else:
+ running, pending = server.prompt_queue.get_current_queue()
+ if running or pending:
+ raise ValueError('busy')
+ result = service.start(name, value)
+ return web.json_response(result)
+ except (ValueError, OSError) as error:
+ codes = {'invalid_request', 'invalid_prompt', 'invalid_duration', 'too_many_images',
+ 'unsupported_media', 'mixed_media', 'busy', 'model_missing', 'consent_required', 'unknown_job'}
+ return web.json_response(dict(error=str(error) if str(error) in codes else 'invalid_media'), status=400)
diff --git a/freevideo_engine/prompt_vlm/worker.py b/freevideo_engine/prompt_vlm/worker.py
new file mode 100644
index 0000000..862033d
--- /dev/null
+++ b/freevideo_engine/prompt_vlm/worker.py
@@ -0,0 +1,188 @@
+"""One rewrite per process: model, CUDA context and private commit die together."""
+import json
+import os
+from pathlib import Path
+import sys
+import time
+import traceback
+
+
+def available_memory():
+ import psutil
+ available = psutil.virtual_memory().available
+ if os.name == 'nt':
+ from ..win32 import memory_status
+ available = min(available, memory_status()['commit_available_bytes'])
+ return available
+
+
+def emit(**value):
+ print(json.dumps(value, ensure_ascii=True), flush=True)
+
+
+def gpu_snapshot(stats):
+ torch = sys.modules.get('torch')
+ if torch is not None and torch.cuda.is_initialized():
+ try:
+ stats.update(gpu_peak_allocated_bytes=torch.cuda.max_memory_allocated(),
+ gpu_peak_reserved_bytes=torch.cuda.max_memory_reserved(),
+ gpu_free_bytes=torch.cuda.mem_get_info()[0],
+ gpu_allocator_ooms=torch.cuda.memory_stats().get('num_ooms'))
+ except (RuntimeError, OSError):
+ stats['gpu_sample_unavailable'] = True
+
+
+def failure(error):
+ """Keep stack locations and typed causes, never arbitrary exception text.
+
+ Tokenizers may quote only a fragment of either draft: replacing the full
+ prompt in an exception is not sufficient to make that exception private.
+ """
+ allowed = {'ram_space', 'image_size', 'context_length', 'invalid_output'}
+ message = str(error)
+ code = message if message in allowed else 'out_of_memory' if 'out of memory' in message.lower() else 'rewrite_failed'
+ frames = [dict(module=Path(f.filename).name, function=f.name, line=f.lineno)
+ for f in traceback.extract_tb(error.__traceback__)]
+ return dict(error=code, exception=type(error).__name__, error_details=frames)
+
+
+def run(folder, value, *, stats=None):
+ stats = {} if stats is None else stats
+ started = time.monotonic()
+ stats.update(stage='imports', pid=os.getpid(), image_count=len(value['media']),
+ mode=value['mode'], dtype='bfloat16', attention='sdpa', kv_cache='dynamic',
+ backend='transformers', accelerations=[],
+ persistent_model=False)
+ import hashlib
+ stats['code_sha256'] = {name: hashlib.sha256(Path(__file__).with_name(name).read_bytes()).hexdigest()
+ for name in ('core.py', 'worker.py', 'rules.py')}
+ emit(phase='loading', diagnostics=dict(stats))
+ os.environ.update(HF_HUB_OFFLINE='1', TRANSFORMERS_OFFLINE='1', HF_HUB_DISABLE_TELEMETRY='1')
+ import psutil
+ import torch
+ import transformers
+ from PIL import Image, ImageOps
+ from transformers import AutoProcessor, Qwen3VLForConditionalGeneration, StoppingCriteria, StoppingCriteriaList
+ from .core import prepare
+ from .rules import instruction, validate_output
+ stats.update(import_seconds=time.monotonic() - started, torch_version=torch.__version__,
+ transformers_version=transformers.__version__, cuda_version=torch.version.cuda,
+ platform=sys.platform, python_version=sys.version.split()[0])
+ def checkpoint(stage, phase='loading'):
+ stats['stage'] = stage
+ stats['worker_seconds'] = time.monotonic() - started
+ gpu_snapshot(stats)
+ emit(phase=phase, diagnostics=dict(stats))
+ stats['available_memory_before_load_bytes'] = available_memory()
+ if stats['available_memory_before_load_bytes'] < 11 * 2**30:
+ raise ValueError('ram_space')
+ torch.set_num_threads(min(8, os.cpu_count() or 1))
+ checkpoint('preprocess')
+ preparing = time.monotonic()
+ processor = AutoProcessor.from_pretrained(folder, local_files_only=True, trust_remote_code=False)
+ messages = [{'role': 'system', 'content': instruction(value)}]
+ content = []
+ for index, row in enumerate(value['media'], 1):
+ with Image.open(row['path']) as source:
+ if source.width * source.height > 50_000_000:
+ raise ValueError('image_size')
+ image = ImageOps.exif_transpose(source).convert('RGB')
+ image.thumbnail((768, 768))
+ content.extend([{'type': 'text', 'text': f' ({row["role"]})'},
+ {'type': 'image', 'image': image}])
+ content.append({'type': 'text', 'text': f'User scene to rewrite (preserve every requested action):\n{value["text"]}'})
+ messages.append({'role': 'user', 'content': content})
+ inputs = processor.apply_chat_template(messages, tokenize=True, add_generation_prompt=True,
+ return_dict=True, return_tensors='pt', processor_kwargs={'images_kwargs': {'max_pixels': 512 * 32 * 32}})
+ if inputs.input_ids.shape[-1] > 6500:
+ raise ValueError('context_length')
+ stats.update(input_tokens=int(inputs.input_ids.shape[-1]), preprocess_seconds=time.monotonic() - preparing)
+ checkpoint('load_weights')
+ loading = time.monotonic()
+ model = Qwen3VLForConditionalGeneration.from_pretrained(folder, dtype=torch.bfloat16,
+ device_map='cpu', local_files_only=True, trust_remote_code=False, attn_implementation='sdpa').eval()
+ stats['load_seconds'] = time.monotonic() - loading
+ preparing = time.monotonic()
+ prepare(model)
+ from .core import GatedMLP
+ accelerations = []
+ if isinstance(model.model.language_model.layers[0].mlp, GatedMLP):
+ accelerations += ['merged_gate_up', 'per_image_sdpa', 'chunked_vision_mlp']
+ if hasattr(model.model.language_model.layers[0].self_attn, 'qkv_proj'):
+ accelerations.append('merged_qkv')
+ stats.update(accelerations=accelerations,
+ backend='freetoken-portable-ops+transformers' if accelerations else 'transformers')
+ stats['prepare_ops_seconds'] = time.monotonic() - preparing
+ checkpoint('placement')
+ placing = time.monotonic()
+ mapping = {'': 'cpu'}
+ if torch.cuda.is_available():
+ from accelerate import dispatch_model, infer_auto_device_map
+ from .. import gpu_budget
+ free, _ = torch.cuda.mem_get_info()
+ stats.update(gpu_name=torch.cuda.get_device_name(), gpu_free_before_load_bytes=free,
+ gpu_compute_capability=list(torch.cuda.get_device_capability()))
+ admission = gpu_budget.configure(torch, free, reserve_bytes=256 * 2**20)
+ free = min(free, admission.get('effective_allocator_limit_bytes') or free)
+ # Keep activation/KV space and the desktop outside the weight budget.
+ weight_budget = min(9 * 2**30, max(0, free - 2 * 2**30))
+ stats.update(gpu_allocator_budget_bytes=free, gpu_weight_budget_bytes=weight_budget)
+ if weight_budget > 2 * 2**30:
+ weights = sum(p.numel() * p.element_size() for p in model.parameters())
+ mapping = infer_auto_device_map(model, max_memory={0: weight_budget,
+ 'cpu': min(psutil.virtual_memory().total - 2 * 2**30, weights + available_memory() - 2 * 2**30)},
+ no_split_module_classes=['Qwen3VLTextDecoderLayer', 'Qwen3VLVisionBlock'], dtype=torch.bfloat16)
+ if 'disk' in mapping.values():
+ raise ValueError('ram_space')
+ model = dispatch_model(model, mapping)
+ inputs = inputs.to('cuda:0' if any(v == 0 for v in mapping.values()) else 'cpu')
+ stats.update(placement_seconds=time.monotonic() - placing,
+ cpu_offload=any(v == 'cpu' for v in mapping.values()) and any(v == 0 for v in mapping.values()),
+ execution_device=str(inputs.input_ids.device), device_map=mapping)
+ loaded = time.monotonic()
+ checkpoint('prefill', 'rewriting')
+ class Progress(StoppingCriteria):
+ shown = 0.
+ first = None
+ def __call__(self, input_ids, scores, **kwargs):
+ now = time.monotonic()
+ tokens = int(input_ids.shape[-1] - inputs.input_ids.shape[-1])
+ if self.first is None:
+ if inputs.input_ids.is_cuda:
+ torch.cuda.synchronize()
+ now = time.monotonic()
+ self.first = now
+ stats['time_to_first_output_seconds'] = now - loaded
+ stats.update(output_tokens=tokens, decode_seconds=now - self.first,
+ decode_tokens_per_second=(tokens - 1) / (now - self.first) if now > self.first else None)
+ if now - self.shown > .5:
+ checkpoint('decode', 'rewriting')
+ self.shown = now
+ return False
+ with torch.inference_mode():
+ output = model.generate(**inputs, max_new_tokens=1800, do_sample=False, use_cache=True,
+ stopping_criteria=StoppingCriteriaList([Progress()]))
+ ids = output[0, inputs.input_ids.shape[-1]:]
+ stats.update(rewrite_seconds=time.monotonic() - loaded, output_tokens=len(ids))
+ checkpoint('validate', 'rewriting')
+ if len(ids) >= 1800:
+ raise ValueError('context_length')
+ rewritten = validate_output(processor.decode(ids, skip_special_tokens=True), value)
+ stats['stage'] = 'complete'
+ emit(phase='complete', text=rewritten, diagnostics=dict(stats))
+
+
+def main():
+ stats = {}
+ try:
+ value = json.loads(sys.stdin.buffer.read(512 * 1024))
+ run(Path(sys.argv[1]), value, stats=stats)
+ except Exception as error:
+ gpu_snapshot(stats)
+ emit(phase='failed', diagnostics=stats, **failure(error))
+ return 1
+ return 0
+
+
+if __name__ == '__main__':
+ raise SystemExit(main())
diff --git a/freevideo_engine/release_notes.json b/freevideo_engine/release_notes.json
index 80f0511..db7e975 100644
--- a/freevideo_engine/release_notes.json
+++ b/freevideo_engine/release_notes.json
@@ -2,16 +2,18 @@
"schema": 1,
"product_version": "0.2.5",
"en": {
- "summary": "Retry failed installations and updates, and free disk space from Settings.",
+ "summary": "Optional prompt enhancement, retries after failed installations and updates, and download cache cleanup.",
"changes": [
+ "Optional FreeToken prompt enhancement in the creation panel: rewrite your prompt and reference images for MiniMax H3 with a local model, keep the original and the enhanced version, and choose which one to generate. It is off by default and asks before downloading the model (8.3 GiB); your prompt and images stay on this computer.",
"After freeing disk space, check the installation plan again without reopening FreeVideo. Existing models and download progress are kept.",
"Clear installation download caches from Settings, with a space estimate before removal. Installed environments, models and videos are kept.",
"Update buttons stay visible with long release notes. Failed update downloads no longer leave temporary files behind."
]
},
"zh": {
- "summary": "安装和更新失败后可直接重试,设置中新增下载缓存清理。",
+ "summary": "新增可选的提示词增强;安装和更新失败可直接重试,设置中可清理下载缓存。",
"changes": [
+ "创作面板新增可选的 FreeToken 提示词增强:用本地模型结合参考图,把提示词改写成适合 MiniMax H3 的写法;原文和改写版都会保留,生成时可任选其一。默认关闭,首次使用会先询问再下载模型(8.3 GiB),提示词和图片只在本机处理。",
"释放磁盘空间后可原地重新检查安装计划,无需重开软件;保留已有模型和下载进度。",
"设置中可查看下载缓存大小并清理,保留已安装的运行环境、模型和视频。",
"更新说明较长时,操作按钮仍固定可见;更新下载失败后自动回收本次临时文件。"
diff --git a/freevideo_engine/support_report.py b/freevideo_engine/support_report.py
index d33e2dd..f48696e 100644
--- a/freevideo_engine/support_report.py
+++ b/freevideo_engine/support_report.py
@@ -368,6 +368,10 @@ def count(value):
'Cached encoder receipts do not describe this request’s peak.'),
bridge={k:bridge[k] for k in ('status', 'error', 'bridge_seconds', 'bridge_wall_seconds', 'sampling_cache_install', 'encoder_prewarm') if k in bridge},
collection_notes=notes, log_tails={})
+ rewrite = read_json('prompt-rewrite', output.parent / 'prompt-rewrite.json')
+ if rewrite.get('schema') == 'freevideo.prompt-vlm':
+ payload['prompt_rewrite'] = rewrite
+ payload['prompt_rewrite']['timing_scope'] = 'Before video generation; excluded from video request_seconds'
if isinstance(bridge.get('result_cache'), dict):
payload['bridge']['result_cache'] = {key: bridge['result_cache'][key]
for key in ('enabled', 'hit', 'forced', 'stored')
diff --git a/pyproject.toml b/pyproject.toml
index a425e88..4157caf 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -31,4 +31,5 @@ freevideo = "freevideo_engine.cli:main"
include = ["freevideo_engine*"]
[tool.setuptools.package-data]
+"freevideo_engine.prompt_vlm" = ["model.json"]
freevideo_engine = ["build-version.txt", "build-identity.json", "release_notes.json", "dependencies.json", "model_files.json", "model_shares.json", "prepared_models.json", "macos_reference_catalog.json", "bootstrap_versions.json", "uv_downloads.json", "torch_downloads.json", "test_prompts.json", "windows_bootstrap.ps1", "assets/*.png", "assets/*.ico", "launcher/*.qml", "launcher/licenses/*.txt", "licenses/*.txt"]
diff --git a/web/error_panel.js b/web/error_panel.js
index 7738625..67fcbc8 100644
--- a/web/error_panel.js
+++ b/web/error_panel.js
@@ -52,8 +52,13 @@ function reportRedactor(detail, context) {
if (Array.isArray(input) && input.length === 2 && Number.isInteger(input[1])) continue;
remember(input);
}
- for (const node of list(context.graph?._nodes))
+ for (const node of list(context.graph?._nodes)) {
+ remember(node.properties?.freevideo_prompt_versions?.original);
+ remember(node.properties?.freevideo_prompt_versions?.rewritten);
for (const widget of list(node.widgets)) remember(widget.value);
+ }
+ remember(context.node?.properties?.freevideo_prompt_versions?.original);
+ remember(context.node?.properties?.freevideo_prompt_versions?.rewritten);
for (const widget of list(context.node?.widgets)) remember(widget.value);
for (const node of Object.values(record(detail?.response?.node_errors ?? detail?.node_errors)))
for (const reason of list(node?.errors)) remember(reason?.extra_info?.received_value);
diff --git a/web/prompt_enhance.css b/web/prompt_enhance.css
new file mode 100644
index 0000000..30e08f2
--- /dev/null
+++ b/web/prompt_enhance.css
@@ -0,0 +1,32 @@
+.fv-enhance{color:var(--fv-muted,#aaaeb9);font-size:12px;border:1px solid var(--fv-border,#ffffff15);border-radius:12px;background:var(--fv-card,#1b2431);overflow:hidden;transition:border-color .2s}
+.fv-enhance:focus-within{border-color:color-mix(in srgb,var(--fv-accent,#6db8fa) 60%,transparent)}
+.fv-enhance-bar{display:flex;align-items:center;gap:6px;min-height:44px;padding:6px 10px;border-bottom:1px solid #ffffff08;white-space:nowrap}
+.fv-enhance-toggle{display:inline-flex;align-items:center;gap:7px;cursor:pointer;font-size:11px;margin-right:auto;min-width:0}
+.fv-enhance-toggle span{overflow:hidden;text-overflow:ellipsis}
+.fv-studio .fv-enhance-toggle input{appearance:none;position:relative;flex:none;width:26px;height:16px;min-height:16px;padding:0;border:0;border-radius:12px;margin:0;background:#ffffff24;cursor:pointer;transition:background .2s}
+.fv-enhance-toggle input::before{content:'';position:absolute;top:3px;left:3px;width:10px;height:10px;border-radius:50%;background:white;transition:transform .2s}
+.fv-enhance-toggle input:checked{background:var(--fv-accent,#6db8fa)}
+.fv-enhance-toggle input:checked::before{transform:translateX(10px)}
+.fv-enhance-tabs{display:flex;flex:none;gap:2px;padding:2px;background:#0002;border-radius:7px}
+.fv-studio .fv-enhance-tabs button{min-height:25px;background:transparent;border:0;color:inherit;padding:3px 7px;font-size:11px;font-weight:400;border-radius:5px;transition:color .18s,background .18s;cursor:pointer}
+.fv-studio .fv-enhance-tabs button[aria-pressed=true]{color:#eef4ff;background:#ffffff15}
+.fv-studio .fv-enhance-bar>.fv-quiet{padding:3px 6px;min-height:28px;font-size:11px}
+.fv-studio .fv-enhance-bar>.fv-enhance-redo{width:28px;flex:none;padding:0}
+.fv-studio .fv-enhance textarea{border:0;border-radius:0;background:transparent;box-shadow:none}
+.fv-enhance [hidden]{display:none!important}
+.fv-enhance-status{display:none;padding:9px 12px 10px;line-height:1.6;border-top:1px solid #ffffff08;font-size:11px}
+.fv-enhance-status:is([data-state=progress],[data-state=error],[data-state=note]){display:block}
+.fv-enhance-status[data-state=error]{color:var(--fv-danger,#ff8a80);background:color-mix(in srgb,var(--fv-danger,#ff8a80) 7%,transparent)}
+.fv-enhance-meter{position:relative;height:3px;margin-top:7px;border-radius:2px;background:#ffffff12;overflow:hidden}
+.fv-enhance-meter>i{position:absolute;inset:0 auto 0 0;width:0;border-radius:inherit;background:var(--fv-accent,#6db8fa);transition:width .4s ease-out}
+.fv-enhance-meter[data-indeterminate=true]>i{width:32%;animation:fv-enhance-sweep 1.4s ease-in-out infinite}
+@keyframes fv-enhance-sweep{from{transform:translateX(-100%)}to{transform:translateX(320%)}}
+.fv-enhance-install{background:#202a38;border-top:1px solid #ffffff12;padding:12px;line-height:1.7;font-size:11px;animation:fv-enhance-in .2s ease-out}
+.fv-enhance-install p{margin:0 0 10px}
+.fv-enhance-install-actions{display:flex;align-items:center;gap:5px;flex-wrap:wrap}
+.fv-studio .fv-enhance-install-actions :is(button,a){min-height:29px;font-size:11px;padding:5px 9px}
+.fv-studio .fv-enhance-install-actions a{margin-left:auto;background:none;border:0;color:var(--fv-muted,#99aabb);padding-inline:4px;text-decoration:underline;text-underline-offset:3px}
+.fv-studio a.fv-prompt-help{font-size:11px;font-weight:400;min-height:0;padding:0;border:0;background:none;color:var(--fv-muted)}
+.fv-enhance button:focus-visible,.fv-enhance input:focus-visible{outline:2px solid var(--fv-accent,#6db8fa);outline-offset:2px}
+@keyframes fv-enhance-in{from{opacity:0;transform:translateY(-4px)}to{opacity:1;transform:none}}
+@media(prefers-reduced-motion:reduce){.fv-enhance *{transition:none;animation:none}}
diff --git a/web/prompt_enhance.js b/web/prompt_enhance.js
new file mode 100644
index 0000000..7eb5f8c
--- /dev/null
+++ b/web/prompt_enhance.js
@@ -0,0 +1,201 @@
+// Original and enhanced prompts are separate workflow drafts, never diagnostics.
+export class PromptVersions {
+ constructor(saved, text) {
+ this.value = saved && typeof saved.original === 'string' && typeof saved.rewritten === 'string'
+ ? {...saved} : {enabled: false, original: text, rewritten: '', selected: 'original', context: ''};
+ if (!['original', 'rewritten'].includes(this.value.selected)) this.value.selected = 'original';
+ if (this.text() !== text) this.edit(text);
+ }
+ text() { return this.value[this.value.selected]; }
+ edit(text) {
+ this.value.edits = (this.value.edits || 0) + 1;
+ this.value[this.value.selected] = text;
+ if (this.value.selected === 'original') this.value.context = '';
+ }
+ key(context) { return JSON.stringify({text: this.value.original, ...context}); }
+ accept(key, context, text) {
+ if (this.key(context) !== key || !this.value.enabled || this.value.useOriginal) return false;
+ this.value.rewritten = text; this.value.selected = 'rewritten'; this.value.context = key;
+ return true;
+ }
+ fresh(context) { return Boolean(this.value.rewritten && this.value.context === this.key(context)); }
+ select(name) { if (name === 'original' || this.value.rewritten) { this.value.selected = name; this.value.useOriginal = name === 'original'; } }
+}
+
+export function createPromptEnhancer({node, input, editor, api, context, setText, t, onReport = () => {}}) {
+ const versions = new PromptVersions(node.properties?.freevideo_prompt_versions, input.value);
+ const el = (tag, text, cls) => { const e = document.createElement(tag); if (text) e.textContent = text; if (cls) e.className = cls; return e; };
+ const element = el('div', null, 'fv-enhance');
+ const bar = el('div', null, 'fv-enhance-bar'), label = el('label', null, 'fv-enhance-toggle');
+ const toggle = el('input'); toggle.type = 'checkbox'; toggle.setAttribute('role', 'switch'); toggle.disabled = input.disabled;
+ toggle.setAttribute('aria-label', t('Prompt enhancement', '提示词增强'));
+ label.append(toggle, el('span', t('FreeToken enhancement', 'FreeToken 提示词增强')));
+ label.title = t('Based on FreeToken. Rewrite locally with reference images, following MiniMax H3 guidelines.', '基于 FreeToken,结合参考图,在本机按 MiniMax H3 规则改写。');
+ const tabs = el('div', null, 'fv-enhance-tabs'); tabs.setAttribute('role', 'group'); tabs.setAttribute('aria-label', t('Prompt version', '提示词版本'));
+ const buttons = {};
+ for (const [name, text] of [['original', t('Original', '原文')], ['rewritten', t('Enhanced', '改写')]]) {
+ const b = el('button', text); b.type = 'button'; buttons[name] = b; tabs.append(b);
+ b.onclick = () => { if (running) cancelJob().catch(() => {}); versions.select(name); apply(); };
+ }
+ const redo = el('button', t('Rewrite', '改写'), 'fv-quiet'); redo.type = 'button';
+ redo.classList.add('fv-enhance-redo');
+ redo.innerHTML = '';
+ const cancel = el('button', t('Cancel', '取消'), 'fv-quiet'); cancel.type = 'button';
+ const status = el('div', null, 'fv-enhance-status'); status.setAttribute('role', 'status');
+ const statusText = el('span', null, 'fv-enhance-status-text');
+ const meter = el('div', null, 'fv-enhance-meter'), fill = el('i');
+ meter.setAttribute('role', 'progressbar'); meter.setAttribute('aria-valuemin', '0'); meter.setAttribute('aria-valuemax', '100');
+ meter.append(fill); status.append(statusText, meter);
+ // One place sets what the status line says and how it looks.
+ function show(text, state = '', fraction = null) {
+ statusText.textContent = text || '';
+ status.dataset.state = text ? state : '';
+ meter.hidden = state !== 'progress';
+ meter.dataset.indeterminate = String(fraction === null);
+ if (fraction === null) { meter.removeAttribute('aria-valuenow'); fill.style.width = ''; }
+ else { const percent = Math.round(100 * Math.max(0, Math.min(1, fraction))); meter.setAttribute('aria-valuenow', String(percent)); fill.style.width = percent + '%'; }
+ }
+ const install = el('div', null, 'fv-enhance-install'); install.hidden = true;
+ const detail = el('p');
+ const agree = el('button', t('Download and enable', '下载并启用'), 'fv-primary'); agree.type = 'button';
+ const decline = el('button', t('Not now', '暂不启用'), 'fv-quiet'); decline.type = 'button';
+ const license = el('a', 'Apache 2.0'); license.href = 'https://huggingface.co/Qwen/Qwen3-VL-4B-Instruct'; license.target = '_blank'; license.rel = 'noopener noreferrer';
+ const actions = el('div', null, 'fv-enhance-install-actions'); actions.append(agree, decline, license);
+ install.append(detail, actions);
+ bar.append(label, tabs, redo, cancel); element.append(bar, editor || input, status, install);
+ let disposed = false, token = '', job = '', running = null, consent = null, cancelling = false;
+ const errorText = error => ({
+ busy: t('Finish the current queue before enhancing this prompt.', '请等待当前队列完成后再改写。'),
+ setup_required: t('Complete FreeVideo installation first.', '请先完成 FreeVideo 安装。'),
+ model_missing: t('Download the optional model to enable enhancement.', '下载可选模型后即可启用。'),
+ download_failed: t('Download paused. Retry to resume.', '下载已暂停,重试即可继续。'),
+ disk_space: t('Not enough disk space for the optional model.', '可选模型的磁盘空间不足,请清理后重试。'),
+ ram_space: t('This model needs about 11 GiB of available RAM to load. You can keep using the original.', '此模型加载需约 11 GiB 可用内存,可先使用原文生成。'),
+ unsupported_media: t('Use images added in Media for enhancement; connected nodes, video and audio are not supported yet.', '增强暂支持在素材中添加的图片;连接节点、视频和音频暂不支持。'),
+ too_many_images: t('Use up to 8 reference images for enhancement.', '增强暂支持最多 8 张参考图。'),
+ invalid_prompt: t('Enter a prompt of up to 12,000 characters.', '请输入不超过 12,000 字符的提示词。'),
+ invalid_output: t('The rewrite was incomplete. Retry or use the original.', '改写结果不完整,请重试或使用原文。'),
+ context_length: t('The scene is too long for this model. Shorten it or use the original.', '内容超出改写模型容量,请精简或使用原文。'),
+ out_of_memory: t('Not enough memory to rewrite. Close other GPU apps or use the original.', '改写时内存不足,请关闭其他显卡程序或使用原文。'),
+ timeout: t('Rewriting took too long. Retry or use the original.', '改写超时,请重试或使用原文。'),
+ stale: t('The prompt or references changed. Rewrite again for the new scene.', '原文或素材已改变,请按新内容重新改写。'),
+ cancelled: t('Cancelled. Your original prompt is retained.', '已取消,原文已保留。'),
+ worker_cleanup_failed: t('The rewrite process could not close. Restart FreeVideo before generating.', '改写进程未能退出,请重启 FreeVideo 后再生成。'),
+ }[error?.message] || t('Could not enhance the prompt. Retry or switch enhancement off to use the original.', '暂时无法改写,请重试,或关闭增强使用原文。'));
+ function save() {
+ node.properties ||= {};
+ node.properties.freevideo_prompt_versions = {...versions.value};
+ node.graph?.change();
+ }
+ function render() {
+ toggle.checked = Boolean(versions.value.enabled);
+ tabs.hidden = !versions.value.rewritten;
+ for (const [name, b] of Object.entries(buttons)) {
+ b.setAttribute('aria-pressed', String(versions.value.selected === name));
+ b.disabled = input.disabled || (name === 'rewritten' && !versions.value.enabled);
+ }
+ redo.hidden = !versions.value.enabled || Boolean(running); redo.disabled = input.disabled;
+ redo.title = versions.value.rewritten ? t('Rewrite again', '重新改写') : t('Rewrite', '改写');
+ redo.setAttribute('aria-label', redo.title);
+ cancel.hidden = !running || install.hidden === false; cancel.disabled = cancelling;
+ element.dataset.busy = String(Boolean(running));
+ element.dataset.enabled = String(Boolean(versions.value.enabled));
+ }
+ function apply() { input.value = versions.text(); setText(input.value); save(); render(); }
+ function changed() { versions.edit(input.value); save(); render(); }
+ input.addEventListener('input', changed);
+ async function post(action, body) {
+ const reply = await api.fetchApi('/freevideo/prompt-vlm/' + action, {method: 'POST',
+ headers: {'Content-Type': 'application/json', 'X-FreeVideo-Prompt': token}, body: JSON.stringify(body)});
+ const data = await reply.json();
+ if (!reply.ok) throw new Error(data.error || 'request_failed');
+ return data;
+ }
+ async function cancelJob() {
+ cancelling = true; consent?.(false); consent = null; render();
+ if (job) await post('cancel', {job});
+ }
+ cancel.onclick = () => cancelJob().catch(error => { show(errorText(error), 'error'); });
+ async function poll(action, body) {
+ if (disposed || cancelling) throw new Error('cancelled');
+ const started = await post(action, body); job = started.job;
+ if (disposed || cancelling) await post('cancel', {job});
+ try {
+ for (;;) {
+ const row = await post('status', {job});
+ if (row.rewrite_report_id) {
+ node.properties.freevideo_prompt_report = row.rewrite_report_id;
+ node.graph?.change();
+ onReport(row);
+ }
+ if (row.phase === 'complete') return row;
+ if (row.phase === 'failed') throw new Error(row.error || 'rewrite_failed');
+ if (row.phase === 'cancelled') throw new Error('cancelled');
+ if (disposed) throw new Error('cancelled');
+ if (row.phase === 'download') {
+ const fraction = (row.done || 0) / (row.total || 1), gib = value => (value / 2 ** 30).toFixed(1);
+ show(t(`Downloading the model · ${gib(row.done || 0)} / ${gib(row.total || 0)} GiB`, `正在下载模型 · ${gib(row.done || 0)} / ${gib(row.total || 0)} GiB`), 'progress', fraction);
+ } else show(row.phase === 'rewriting' ? t('Enhancing the prompt…', '正在改写提示词…') : t('Preparing enhancement…', '正在准备提示词增强…'), 'progress');
+ await new Promise(resolve => setTimeout(resolve, 400));
+ }
+ } catch (error) {
+ // Losing a status reply must not leave an invisible model running.
+ await post('cancel', {job}).catch(() => {});
+ throw error;
+ } finally { job = ''; }
+ }
+ async function enhance() {
+ if (running) return running;
+ const execute = async () => {
+ cancelling = false;
+ versions.value.useOriginal = false;
+ const scene = context(), key = versions.key(scene), original = versions.value.original, edits = versions.value.edits;
+ const reply = await api.fetchApi('/freevideo/prompt-vlm');
+ const info = await reply.json();
+ if (!reply.ok) throw new Error(info.error);
+ token = info.token;
+ if (!info.ready) {
+ const model = String(info.model || 'Qwen/Qwen3-VL-4B-Instruct').split('/').pop().replace(/-Instruct$/, '');
+ detail.textContent = t(`Local model ${model} · ${(info.bytes / 2 ** 30).toFixed(1)} GiB download · needs 11 GiB free RAM. Your prompt and images stay on this computer.`,
+ `本地模型 ${model} · 下载 ${(info.bytes / 2 ** 30).toFixed(1)} GiB · 需 11 GiB 可用内存。提示词和图片只在本机处理。`);
+ install.hidden = false; render();
+ const accepted = await new Promise(resolve => { consent = resolve; agree.onclick = () => resolve(true); decline.onclick = () => resolve(false); });
+ consent = null; install.hidden = true; render();
+ if (!accepted || disposed || cancelling) {
+ versions.value.enabled = false;
+ versions.select('original'); apply();
+ throw new Error('cancelled');
+ }
+ await poll('install', {accept_download: true});
+ }
+ const result = await poll('rewrite', {text: original, ...scene});
+ if (disposed || cancelling) throw new Error('cancelled');
+ if (edits !== versions.value.edits || !versions.accept(key, context(), result.text)) throw new Error('stale');
+ apply(); show('');
+ };
+ running = Promise.resolve().then(execute).catch(error => {
+ if (!disposed) show(errorText(error), error?.message === 'cancelled' ? '' : 'error');
+ throw error;
+ }).finally(() => { running = null; install.hidden = true; render(); });
+ render();
+ return running;
+ }
+ toggle.onchange = () => {
+ versions.value.enabled = toggle.checked;
+ if (!toggle.checked) { cancelJob().catch(() => {}); versions.select('original'); apply(); }
+ else { versions.value.useOriginal = false; save(); render(); enhance().catch(() => {}); }
+ };
+ redo.onclick = () => enhance().catch(() => {});
+ show(''); save(); render();
+ return {element, async beforeGenerate() {
+ if (running) {
+ try { await running; }
+ catch (error) { if (versions.value.enabled && !versions.value.useOriginal) return false; }
+ }
+ if (!versions.value.enabled || versions.value.useOriginal) return;
+ try { if (!versions.fresh(context())) await enhance(); }
+ catch (error) { show(errorText(error), error?.message === 'cancelled' ? '' : 'error'); return false; }
+ }, sync() { if (input.value !== versions.text()) changed(); }, dispose() {
+ disposed = true; cancelJob().catch(() => {}); input.removeEventListener('input', changed);
+ }};
+}
diff --git a/web/studio.js b/web/studio.js
index 56c1011..0969752 100644
--- a/web/studio.js
+++ b/web/studio.js
@@ -13,6 +13,7 @@ import { createStudioQueue, randomSeed } from './studio_queue.js';
import { attachReferencePicker, referenceItems, syncReferencePrompt } from './prompt_references.js';
import { createSamplingEffort } from './sampling_effort.js';
import { resultActions } from './result_actions.js';
+import { createPromptEnhancer } from './prompt_enhance.js';
const languageOverride = typeof location !== 'undefined'
? new URLSearchParams(location.search).get('freevideo_lang') : null;
@@ -21,6 +22,7 @@ const cn = languageOverride === 'zh' || (languageOverride !== 'en'
const t = (en, zh) => cn ? zh : en;
const css = document.createElement('link'); css.rel = 'stylesheet'; css.href = new URL('./studio.css', import.meta.url).href; document.head.append(css);
const effortCss = document.createElement('link'); effortCss.rel = 'stylesheet'; effortCss.href = new URL('./sampling_effort.css', import.meta.url).href; document.head.append(effortCss);
+const enhanceCss = document.createElement('link'); enhanceCss.rel = 'stylesheet'; enhanceCss.href = new URL('./prompt_enhance.css', import.meta.url).href; document.head.append(enhanceCss);
const el = (tag, label, cls) => { const e = document.createElement(tag); if (label != null) e.textContent = label; if (cls) e.className = cls; return e; };
const button = (label, action, cls = '') => { const e = el('button', label, cls); e.type = 'button'; e.onclick = action; return e; };
const widget = (node, name) => node?.widgets?.find(w => w.name === name);
@@ -157,10 +159,24 @@ export function openStudio(node) {
mention.title = t('Reference media (@)', '引用素材(@)'); mention.setAttribute('aria-label', mention.title);
mention.disabled = prompt.disabled; mention.onpointerdown = e => e.preventDefault(); promptEditor.append(mention);
cleanup.push(() => references.dispose());
- const promptChanged = e => { if (String(e.detail) === String(node.id)) { if (prompt.value !== value(node, 'text')) prompt.value = value(node, 'text') || ''; references.refresh(); } };
+ const enhancer = createPromptEnhancer({node, input: prompt, editor: promptEditor, api, t,
+ onReport: row => { node.freevideoReportId = row.report_id; progress.updateReport(row); },
+ setText: text => set(node, 'text', text), context: () => {
+ const items = referenceItems(node);
+ if (linked(node, 'conditioning') || items.some(row => !row.file || row.kind !== 'image')) throw new Error('unsupported_media');
+ return {seconds: Number(value(node, 'seconds')), media: items.map(row => ({file: row.file, role: row.role || 'reference'}))};
+ }});
+ cleanup.push(() => enhancer.dispose());
+ const promptChanged = e => { if (String(e.detail) === String(node.id)) { if (prompt.value !== value(node, 'text')) prompt.value = value(node, 'text') || ''; references.refresh(); enhancer.sync(); } };
window.addEventListener('freevideo-reference-prompt', promptChanged);
cleanup.push(() => window.removeEventListener('freevideo-reference-prompt', promptChanged));
- section(t('Describe your scene', '描述画面')).append(promptEditor, promptGuide());
+ const promptSection = section(t('Describe your scene', '描述画面'));
+ const promptHelp = el('a', t('Guide ↗', '指南 ↗'), 'fv-prompt-help');
+ promptHelp.href = 'https://github.com/MiniMax-AI/MiniMax-H3/blob/main/skills/h3-prompt-writing/SKILL.md';
+ promptHelp.target = '_blank'; promptHelp.rel = 'noopener noreferrer';
+ promptHelp.title = t('MiniMax H3 prompt guide', 'MiniMax H3 提示词指南');
+ promptSection.querySelector('.fv-label').append(promptHelp);
+ promptSection.append(enhancer.element);
const canvas = section(t('Frame & duration', '画幅与时长'));
const shapes = el('div', null, 'fv-shapes'); canvas.append(shapes);
const dimensionsLinked = linked(node, 'width') || linked(node, 'height');
@@ -596,6 +612,8 @@ export function openStudio(node) {
try {
lastQueueError = null; failure.clear(); node.freevideoFailure = ''; node.freevideoFailureReport = null;
capturing = true; updateRunButton(); status.dataset.error = 'false'; runOptions.open = false;
+ if (await enhancer.beforeGenerate() === false) return;
+ if (disposed) return;
// Only serialization briefly locks the editor. Network submission
// and all GPU execution leave the next draft fully editable.
for (const section of controls.querySelectorAll('.fv-section')) section.inert = true;