{
  "catalog_version": "2.0.0",
  "title": "Tunx AI 视频制作提示词工程目录",
  "as_of": "2026-08-14",
  "maintainer": "cooperatebot",
  "canonical_url": "https://tunx.ai/guides/video-prompt-engineering/",
  "scope": "覆盖截至核验日的主流基础视频模型、重要创作平台、开源权重与数字人工作流。‘主流’指有官方产品或文档、实际创作入口且仍在维护的代表性系统；不穷举换壳站、区域转售商与尚未公开可用的预告模型。",
  "verification_policy": [
    "能力、版本、参数和语法优先采用模型厂商或平台官方文档。",
    "平台与模型分层：聚合平台的 UI 控件不能被误写成底层模型通用能力。",
    "版本会变化；生产前重新检查官方文档、区域可用性、计费、内容政策与许可条款。",
    "供应商营销指标只作产品声明，不作为跨模型质量排名。",
    "Sora 2 视频 API 已被官方标为 Legacy/Deprecated，仅保留迁移与历史方法论。",
    "开源条目必须同时核验推理代码、模型权重与模型许可证；GitHub 可见不自动等于可自由商用。",
    "每个模型保存 last_verified_at 与官方来源；超过 90 天未核验的版本进入 freshness review。"
  ],
  "freshness": {
    "last_verified_at": "2026-08-14",
    "review_after_days": 90,
    "states": ["current", "legacy", "preview", "license_review", "availability_review"]
  },
  "openness_taxonomy": [
    {"id": "open_source", "meaning": "代码和权重均公开，且许可证经核验；仍需遵守许可证与可接受使用条款。"},
    {"id": "open_weight", "meaning": "权重可下载，但代码、数据或许可证可能并非 OSI 意义的开源。"},
    {"id": "restricted_open_weight", "meaning": "权重和代码可用，但商业规模、用途、地域或再分发存在额外限制。"},
    {"id": "open_source_pipeline", "meaning": "推理/控制工具链开源，底层模型需单独记录并核验许可证。"},
    {"id": "hosted_only", "meaning": "只能通过厂商或平台 UI/API 使用。"}
  ],
  "research_watchlist": [
    {"name": "Tora3", "state": "paper_or_project_only", "reason": "截至核验日，官方仓库公告论文/项目页，但未核验到 Tora3 权重与推理代码同时发布；不列为可用适配器。", "source": "https://github.com/alibaba/Tora"},
    {"name": "SANA-Video 2.0", "state": "release_not_verified", "reason": "截至核验日未从官方仓库核验到代码与权重完整发布；不根据社区帖子建立生产适配器。"},
    {"name": "Seedance 2.5", "state": "availability_review", "reason": "官方入口的预告、区域与稳定可用性仍不一致，生产默认保留 Seedance 2.0。"}
  ],
  "architecture": [
    {
      "step": 1,
      "name": "Creative Brief",
      "output": "目标、受众、渠道、时长、画幅、叙事、CTA 与合规边界"
    },
    {
      "step": 2,
      "name": "Continuity Bible",
      "output": "角色、服装、产品、场景地理、色彩、声音与不可漂移项"
    },
    {
      "step": 3,
      "name": "ShotSpec",
      "output": "供应商中立的视觉、动作、摄影、时间、音频、参考与约束"
    },
    {
      "step": 4,
      "name": "Model Adapter",
      "output": "按模型、模式与平台控件编译成实际提示词和参数"
    },
    {
      "step": 5,
      "name": "Render",
      "output": "保存原始提示、有效提示、模型版本、设置、引用和输出 ID"
    },
    {
      "step": 6,
      "name": "Evaluate & Iterate",
      "output": "先硬拒，再分维度评分；一次只改变一个主要变量"
    }
  ],
  "modes": [
    {
      "id": "text_to_video",
      "name_zh": "文生视频",
      "input_contract": "纯文本，可选结构化设置",
      "prompt_focus": "完整定义主体、动作、场景、摄影、美学、时间和声音",
      "base_grammar": "主体与属性 → 动作与物理 → 场景 → 构图与摄影机 → 光色与风格 → 时间节拍 → 音频",
      "failure_risk": "静态信息过多、动作归属不清、多个复杂事件争夺注意力"
    },
    {
      "id": "image_to_video",
      "name_zh": "图生视频",
      "input_contract": "首图或源图 + 文本",
      "prompt_focus": "主要写从图像开始发生的主体动作、环境动作、运镜与时序变化",
      "base_grammar": "主体动作 → 环境响应 → 运镜 → 速度/方向/物理 → 可选的时序性变化",
      "failure_risk": "复述图片造成冲突；试图用文本重新定义已锁定的构图和身份"
    },
    {
      "id": "first_last_frame",
      "name_zh": "首尾帧 / 关键帧",
      "input_contract": "开始帧 + 结束帧 + 文本",
      "prompt_focus": "描述两帧之间如何运动与过渡，不重复端点",
      "base_grammar": "起始状态 → 连续动作链 → 相机路径 → 过渡节奏 → 稳定抵达结束状态",
      "failure_risk": "端点差异过大、路径不可能、尾帧没有稳定停留时间"
    },
    {
      "id": "reference_to_video",
      "name_zh": "多参考生视频",
      "input_contract": "角色/产品/场景/构图/动作/声音等多资产 + 文本",
      "prompt_focus": "逐一声明每个参考的职责、只借用什么、必须保留什么",
      "base_grammar": "@参考及角色映射 → 镜头目标 → 动作/摄影/节拍 → 保留与排除边界",
      "failure_risk": "参考职责重叠、只写‘参考这些素材’而未绑定用途"
    },
    {
      "id": "video_to_video",
      "name_zh": "视频重绘 / 风格迁移",
      "input_contract": "源视频 + 目标状态文本，可选参考",
      "prompt_focus": "源视频已给出时序与镜头；描述输出应该变成什么",
      "base_grammar": "最终主体/环境/材质/光色/风格 → 明确保留运动、表演、构图和机位",
      "failure_risk": "写操作步骤而非目标状态；忘记指定保留项导致全局漂移"
    },
    {
      "id": "edit_video",
      "name_zh": "局部编辑",
      "input_contract": "源视频 + 选择区域/帧 + 编辑意图",
      "prompt_focus": "明确改变对象、目标结果、时间范围，并写‘其余保持不变’",
      "base_grammar": "在何时何处把 X 改为 Y → 保持身份/动作/相机/背景/声音不变",
      "failure_risk": "编辑范围不明确造成全片重绘"
    },
    {
      "id": "extend_video",
      "name_zh": "视频延长 / 续写",
      "input_contract": "已有片段 + 后续事件",
      "prompt_focus": "从末帧状态继续，延续方向、速度、光线、声音与因果",
      "base_grammar": "承接当前状态 → 下一动作 → 相机延续 → 新信息 → 结尾状态",
      "failure_risk": "重新介绍角色与场景、动作方向突变、光线或音乐断裂"
    },
    {
      "id": "performance_transfer",
      "name_zh": "动作 / 表演迁移",
      "input_contract": "驱动视频或姿态 + 目标角色/场景",
      "prompt_focus": "把运动交给控制输入，文本负责目标角色、材质、环境和保留项",
      "base_grammar": "目标外观与场景 → 沿用驱动动作/节奏/镜头 → 身份与材质稳定",
      "failure_risk": "文本动作与驱动视频冲突"
    },
    {
      "id": "avatar_presenter",
      "name_zh": "数字人口播",
      "input_contract": "已授权数字人 + 脚本/录音 + 声音、背景与版式设置",
      "prompt_focus": "脚本结构、发音、停顿、语气、手势意图、镜头和屏幕信息",
      "base_grammar": "目标/受众 → 分场脚本 → 逐句语气和停顿 → B-roll/版式 → 字幕/品牌/CTA",
      "failure_risk": "把视觉生成 prompt 当成演讲稿；句子过长；缺少发音与停顿标注"
    },
    {
      "id": "interactive_avatar",
      "name_zh": "实时交互数字人",
      "input_contract": "数字人 + persona/系统指令 + 知识与会话策略",
      "prompt_focus": "角色边界、对话目标、工具、知识、回退、隐私和升级人工规则",
      "base_grammar": "身份与目标 → 知识边界 → 对话行为 → 安全护栏 → 失败回退/人工升级",
      "failure_risk": "把实时 persona 与异步视频脚本混用；缺少知识边界与敏感操作确认"
    }
  ],
  "prompt_element_groups": [
    {
      "id": "brief",
      "name_zh": "制作任务",
      "elements": ["目标", "受众", "发布渠道", "核心信息", "CTA", "时长", "画幅", "交付分辨率", "单镜头/多镜头", "合规边界"]
    },
    {
      "id": "continuity",
      "name_zh": "连续性圣经",
      "elements": ["角色身份", "面部/体态", "服装", "声音", "产品与道具", "场景地理", "色彩系统", "风格边界", "不可漂移项"]
    },
    {
      "id": "visual",
      "name_zh": "画面世界",
      "elements": ["主体", "属性", "材质", "服装", "道具", "环境", "时代", "时间/天气", "光源", "光质", "色板", "媒介/风格", "纹理/后期质感"]
    },
    {
      "id": "composition_camera",
      "name_zh": "构图与摄影",
      "elements": ["景别", "机位角度", "POV", "机位高度/距离", "画面占比", "前中后景", "焦距", "光圈/景深", "焦点对象", "拉焦", "稳定方式", "运镜路径"]
    },
    {
      "id": "motion_physics",
      "name_zh": "动作与物理",
      "elements": ["主体动作", "物体动作", "环境运动", "动作归属", "空间位置", "方向", "速度", "幅度", "轨迹", "重量", "惯性", "摩擦", "碰撞/反弹", "布料/毛发/流体", "因果链"]
    },
    {
      "id": "time_editing",
      "name_zh": "时间与剪辑",
      "elements": ["总时长", "时间码", "节拍", "动作顺序", "停顿", "节奏", "慢动作", "延时", "循环接缝", "剪切点", "转场", "尾帧停留"]
    },
    {
      "id": "audio",
      "name_zh": "声音",
      "elements": ["说话人标签", "逐字对白", "语言", "口音", "音色", "情绪", "语速", "停顿", "旁白", "拟音", "环境声", "音乐", "静默", "画音同步点", "无对白/无音乐"]
    },
    {
      "id": "references",
      "name_zh": "参考资产",
      "elements": ["身份", "服装", "产品", "道具", "场景", "构图", "风格", "动作", "姿态", "首帧", "尾帧", "声音", "音乐", "权利/授权", "只借用项", "必须保留项"]
    },
    {
      "id": "constraints",
      "name_zh": "约束与负向",
      "elements": ["必须出现", "必须保持", "必须避免", "安全限制", "法律限制", "品牌限制", "独立 negative 字段", "其余保持不变", "文字/Logo 转后期"]
    },
    {
      "id": "settings",
      "name_zh": "模型设置",
      "elements": ["供应商", "平台", "模型 ID", "版本/快照", "模式", "时长", "画幅", "分辨率", "FPS", "种子", "运动强度", "参考强度", "原生音频", "提示增强器", "原始/有效提示"]
    },
    {
      "id": "governance",
      "name_zh": "治理与评测",
      "elements": ["org_id", "project_id", "shot_id", "提示版本", "作者/审核人", "参考哈希", "渲染任务/输出 ID", "成本/延迟", "硬拒原因", "分维度评分", "变更说明"]
    }
  ],
  "camera_lexicon": [
    {"term": "pan / 摇摄", "meaning": "机位不移动，镜头水平旋转", "do_not_confuse": "truck 横移"},
    {"term": "tilt / 俯仰", "meaning": "机位不移动，镜头垂直旋转", "do_not_confuse": "pedestal 升降"},
    {"term": "truck / 横移", "meaning": "摄影机整体向左或右平移", "do_not_confuse": "pan 摇摄"},
    {"term": "pedestal / 升降", "meaning": "摄影机整体垂直上升或下降", "do_not_confuse": "tilt 俯仰"},
    {"term": "dolly in/out / 推拉", "meaning": "摄影机在空间里靠近或远离，透视关系会变化", "do_not_confuse": "zoom 只改变焦距"},
    {"term": "zoom / 变焦", "meaning": "机位不动，改变焦距与视角", "do_not_confuse": "dolly 推拉"},
    {"term": "orbit / arc / 环绕", "meaning": "摄影机沿弧线绕主体运动", "do_not_confuse": "角色原地旋转"},
    {"term": "tracking / 跟拍", "meaning": "相机与移动主体保持相对关系", "do_not_confuse": "handheld 只是稳定风格"},
    {"term": "crane / jib / 摇臂", "meaning": "沿三维弧线升降并改变视角", "do_not_confuse": "简单 pedestal"},
    {"term": "rack focus / 拉焦", "meaning": "焦点从一个深度平面转移到另一个", "do_not_confuse": "zoom"},
    {"term": "dolly zoom / 希区柯克变焦", "meaning": "推拉与反向变焦同步，主体尺寸近似不变而背景透视改变", "do_not_confuse": "普通 zoom"},
    {"term": "static / locked-off / 固定机位", "meaning": "相机完全静止，仅主体和环境运动", "do_not_confuse": "静态画面"}
  ],
  "platforms": [
    {
      "name": "Google Gemini API / AI Studio / Vertex AI",
      "kind": "第一方模型平台",
      "models": ["Gemini Omni Flash", "Veo 3.1"],
      "prompt_rule": "默认先按 Gemini Omni 的多轮多模态交互工作；需要首尾帧、扩展或既有 Veo 管线时切换 Veo 3.1。"
    },
    {
      "name": "Runway",
      "kind": "多模型创作平台",
      "models": ["Gen-4.5", "Aleph 2.0", "合作伙伴模型"],
      "prompt_rule": "先确认模型与模式；I2V 重点写动作，Aleph 编辑写目标修改和保留边界。"
    },
    {
      "name": "Adobe Firefly",
      "kind": "多模型创作与后期平台",
      "models": ["Firefly Video", "合作伙伴模型"],
      "prompt_rule": "模型选择器改变可用设置；构图、机位、风格、种子等 UI 控件优先于重复堆进文本。"
    },
    {
      "name": "Dreamina / 即梦 / CapCut",
      "kind": "字节系创作平台",
      "models": ["Seedance 2.0 系列", "第三方模型入口"],
      "prompt_rule": "多模态素材用 @AssetName 或 UI 绑定角色；Seedance 2.5 在官方入口仍存在预告/区域差异，不作为稳定默认。"
    },
    {
      "name": "Kling AI / 可灵",
      "kind": "第一方创作平台",
      "models": ["Kling 3.0", "Kling 3.0 Omni"],
      "prompt_rule": "多镜头用 Shot 1/2…或平台分镜；为元素绑定角色、动作、说话人与声音。"
    },
    {
      "name": "Luma Dream Machine",
      "kind": "第一方创作平台",
      "models": ["Ray3.14", "Ray3.2 Modify"],
      "prompt_rule": "生成与编辑语法不同：Ray3.14 写镜头，Ray3.2 修改写最终目标状态。"
    },
    {
      "name": "Hailuo / 海螺",
      "kind": "第一方创作与 API 平台",
      "models": ["Hailuo 2.3", "Hailuo 2.3 Fast"],
      "prompt_rule": "相机命令用官方方括号词汇，单段组合不宜超过三个同时运镜。"
    },
    {
      "name": "Vidu",
      "kind": "第一方创作与 API 平台",
      "models": ["Vidu Q3 family", "Vidu S1 interactive"],
      "prompt_rule": "Q3 按 pro/mix/drama/ad/turbo 选任务；S1 属实时交互，不与离线镜头提示混用。"
    },
    {
      "name": "Pika",
      "kind": "效果化短视频平台",
      "models": ["Pika 2.5", "Pikaframes", "Pikaformance"],
      "prompt_rule": "优先简洁动作/效果指令与 UI 参数；旧 Discord dash 参数不是当前通用语法。"
    },
    {
      "name": "Alibaba Cloud Model Studio / 百炼",
      "kind": "第一方 API 平台",
      "models": ["Wan 2.7 API", "Wan 2.6 family"],
      "prompt_rule": "按 T2V、I2V、多镜头、参考生视频分别使用官方公式，并保存 prompt_extend 后的有效提示。"
    },
    {
      "name": "Tencent Cloud MPS / 混元",
      "kind": "聚合云平台与开源模型",
      "models": ["HunyuanVideo-1.5", "YT-Video", "Kling/Hailuo/Vidu connectors"],
      "prompt_rule": "先区分自有模型与聚合连接器；连接器仍服从其底层模型语法。"
    },
    {
      "name": "Higgsfield",
      "kind": "电影化控制与多模型平台",
      "models": ["Cinema Studio 3.5 model picker"],
      "prompt_rule": "Hero Frame First；镜头、焦距、光圈、灯光和类型交给 Director Panel，文本聚焦内容与动作。"
    },
    {
      "name": "fal.ai / Replicate / 模型聚合 API",
      "kind": "模型托管与分发平台",
      "models": ["多家闭源 API 与开源权重"],
      "prompt_rule": "平台只是传输层；模型版本、输入 schema、增强器和负向字段必须按具体 endpoint 固定。"
    },
    {
      "name": "ComfyUI",
      "kind": "开源工作流编排",
      "models": ["LTX-2.5", "Wan2.2", "HunyuanVideo-1.5", "SkyReels V3", "LongCat-Video", "MAGI-1.1 等"],
      "prompt_rule": "节点、模型、采样器、LoRA、seed、分辨率与工作流 JSON 都属于提示工程可复现记录。"
    },
    {
      "name": "Midjourney Video",
      "kind": "图生视频创作平台",
      "models": ["Midjourney Video"],
      "prompt_rule": "以起始图像为主体和构图来源，文本主要写运动；用 --motion low/high、--raw、--loop、--end 等当前参数控制。"
    },
    {
      "name": "PixVerse",
      "kind": "第一方创作与 API 平台",
      "models": ["PixVerse v6"],
      "prompt_rule": "先选 T2V/I2V/Transition/Effect/Lip-sync，再把 aspect、duration、quality、camera、seed 与 negative 放结构化字段。"
    },
    {
      "name": "Canva AI / Magic Media",
      "kind": "设计、剪辑与模型聚合平台",
      "models": ["Magic Video", "Veo 3 connector", "Canva AI workflows"],
      "prompt_rule": "区分素材自动剪辑和生成式视频；记录实际底层模型，生成后利用可编辑图层、版式和品牌系统完成交付。"
    },
    {
      "name": "Krea / Freepik AI Suite",
      "kind": "多模型创作与聚合平台",
      "models": ["多家闭源 API", "主流开放权重模型"],
      "prompt_rule": "模型选择器就是适配器边界；每次保存底层模型、版本、模式、输入槽位和平台增强后的有效提示。"
    },
    {
      "name": "InVideo AI",
      "kind": "脚本到成片的 agentic 平台",
      "models": ["Agent One/Two", "Autopilot"],
      "prompt_rule": "先给项目 brief、受众、渠道、时长、素材与品牌限制，再逐轮审核脚本、分镜、旁白、素材和 CTA。"
    },
    {
      "name": "Captions / Mirage",
      "kind": "AI 演员与内容创作平台",
      "models": ["mirage-video-1-latest"],
      "prompt_rule": "静态人物图与音频是核心输入；提示写表演和画面目标，身份与声音授权必须作为生成前置门禁。"
    },
    {
      "name": "HeyGen / Synthesia / Tavus / D-ID",
      "kind": "数字人、口播与实时交互平台",
      "models": ["Avatar V", "Synthesia Assistant", "Tavus Replicas/CVI", "D-ID V4/V3/V2"],
      "prompt_rule": "核心不是摄影形容词，而是脚本、语气、停顿、发音、授权身份、背景/B-roll 与 persona 护栏。"
    }
  ],
  "model_adapters": [
    {
      "id": "google-gemini-omni-flash",
      "vendor": "Google",
      "model": "Gemini Omni Flash",
      "model_id": "gemini-omni-flash-preview",
      "status": "public_preview",
      "layer": "foundation_model",
      "modes": ["text_to_video", "image_to_video", "reference_to_video", "edit_video"],
      "best_for": "默认多模态创作、带音频视频输出、多轮对话式修改、故事板与任意素材参考",
      "prompt_grammar": "先给创作目标和素材职责，再给按时间排列的画面/动作/镜头/声音；后续回合只描述需要修改的差异。",
      "priorities": ["把每个图像/视频/音频输入的用途说清", "多镜头先给 storyboard 和连续性", "对白标说话人、内容、语言与语气"],
      "controls": ["多模态输入", "视频+音频输出", "多轮修改", "故事板"],
      "negative_strategy": "用明确正向目标和局部修改范围，勿假设跨模型 negative 语法。",
      "gotchas": ["Preview 版本可能变化", "需要首尾帧/扩展等专用控制时 Veo 3.1 可能更适合"],
      "official_sources": ["https://deepmind.google/models/gemini-omni/prompt-guide/", "https://ai.google.dev/gemini-api/docs/omni", "https://ai.google.dev/gemini-api/docs/video"]
    },
    {
      "id": "google-veo-3-1",
      "vendor": "Google",
      "model": "Veo 3.1",
      "status": "active",
      "layer": "foundation_model",
      "modes": ["text_to_video", "image_to_video", "first_last_frame", "reference_to_video", "extend_video"],
      "best_for": "原生音频、首尾帧、素材 ingredients、场景扩展和既有 Veo 生产管线",
      "prompt_grammar": "主体+动作+场景+摄影+光色+逐拍时间+对白/环境声/音乐；参考素材逐一绑定角色。",
      "priorities": ["短片内只保留一个主要动作弧", "对白使用明确说话人标签", "关键帧之间写过程而非端点"],
      "controls": ["ingredients/reference", "first/last frame", "scene extension", "native audio"],
      "negative_strategy": "优先正向描述所需画面；使用 API/UI 明确支持的控制字段。",
      "gotchas": ["文本、参考和参数三者冲突时会降低遵循度", "不要在素材已锁定外观时重复改写身份"],
      "official_sources": ["https://deepmind.google/models/veo/", "https://cloud.google.com/blog/products/ai-machine-learning/ultimate-prompting-guide-for-veo-3-1/", "https://ai.google.dev/gemini-api/docs/video"]
    },
    {
      "id": "runway-gen-4-5",
      "vendor": "Runway",
      "model": "Gen-4.5",
      "status": "active",
      "layer": "foundation_model_on_platform",
      "modes": ["text_to_video", "image_to_video"],
      "best_for": "高质量 T2V/I2V、精细相机编排、短镜头生成",
      "prompt_grammar": "T2V 写视觉与运动；I2V 几乎只写主体动作、环境运动、运镜、速度和时间。可用自然语言按顺序写复杂摄影机路径。",
      "priorities": ["清晰、直接、无歧义", "把主体动作与相机动作分开", "I2V 不复述源图"],
      "controls": ["2–10 秒", "T2V", "I2V", "camera choreography"],
      "negative_strategy": "不要沿用旧模型未经确认的 negative 行为；以正向可见目标为主。",
      "gotchas": ["结构顺序不如清晰意图重要", "一次塞入过多互斥动作会让模型自行取舍"],
      "official_sources": ["https://help.runwayml.com/hc/en-us/articles/46974685288467-Creating-with-Gen-4-5", "https://help.runwayml.com/hc/en-us/articles/48324313115155-Image-to-Video-Prompting-Guide", "https://help.runwayml.com/hc/en-us/articles/47313504791059-Camera-Terms-Prompts-Examples"]
    },
    {
      "id": "runway-aleph-2",
      "vendor": "Runway",
      "model": "Aleph 2.0 / Edit Studio",
      "status": "active",
      "layer": "video_editing_model",
      "modes": ["edit_video", "video_to_video"],
      "best_for": "最长约 30 秒 1080p 的局部/全局视频编辑、多镜头一致修改",
      "prompt_grammar": "把 X 改成目标 Y；限定对象、区域和时间；以‘保持其余人物、运动、相机、光线和声音不变’收尾。",
      "priorities": ["一个请求一个主要编辑目标", "说最终结果", "明确保留边界"],
      "controls": ["localized edits", "frame/image-level edit", "multi-shot preservation"],
      "negative_strategy": "用 preserve/keep unchanged 边界约束，避免长串负向形容词。",
      "gotchas": ["编辑模型与生成模型语法不同", "未指定时间范围可能扩大编辑影响"],
      "official_sources": ["https://runwayml.com/news/introducing-aleph-2-and-edit-studio"]
    },
    {
      "id": "adobe-firefly-video",
      "vendor": "Adobe",
      "model": "Firefly Video",
      "status": "active",
      "layer": "foundation_model_on_platform",
      "modes": ["text_to_video", "image_to_video", "reference_to_video"],
      "best_for": "与 Adobe 后期工作流衔接、构图/相机/风格 UI 控制、透明背景等平台能力",
      "prompt_grammar": "Shot type + Character + Action + Location + Aesthetic；UI 已设定的机位、镜头、风格不要在文本里重复冲突。",
      "priorities": ["明确人物和动作", "使用 shot type 与 aesthetic", "把可确定设置交给 UI"],
      "controls": ["composition reference", "camera motion reference", "shot/angle/motion", "style preset", "seed", "transparent background（适用模型）"],
      "negative_strategy": "使用所选模型实际提供的字段；Firefly 平台内合作伙伴模型的行为不等于 Adobe 自有模型。",
      "gotchas": ["模型选择器会改变可用控件", "Adobe 对自有 Firefly 模型的训练/商业安全表述不自动覆盖合作伙伴模型，也不构成法律保证"],
      "official_sources": ["https://helpx.adobe.com/firefly/web/work-with-audio-and-video/work-with-video/writing-effective-text-prompts-for-video-generation.html", "https://helpx.adobe.com/firefly/web/work-with-audio-and-video/work-with-video/generate-videos-using-text-prompts.html", "https://helpx.adobe.com/firefly/web/create-mood-boards/firefly-boards/partner-models-to-generate-videos.html"]
    },
    {
      "id": "kling-3",
      "vendor": "Kuaishou",
      "model": "Kling 3.0 / Kling 3.0 Omni",
      "status": "active",
      "layer": "foundation_model_on_platform",
      "modes": ["text_to_video", "image_to_video", "first_last_frame", "reference_to_video", "edit_video"],
      "best_for": "3–15 秒多镜头、元素/角色引用、多语种对白与原生音频",
      "prompt_grammar": "单镜头写主体+动作+场景+摄影+声音；多镜头按 Shot 1 / Shot 2…明确每镜时间、人物、动作、机位、对白和转场。",
      "priorities": ["元素 ID 与人物绑定", "对白标注说话人", "跨镜头重复连续性锚点"],
      "controls": ["multi-shot", "element reference", "start/end frame", "native audio", "multilingual dialogue"],
      "negative_strategy": "若 UI 提供独立 negative 字段再使用；正向提示仍写可见目标。",
      "gotchas": ["多人物场景必须明确动作/台词归属", "Omni 与普通 3.0 模式的输入能力不同"],
      "official_sources": ["https://kling.ai/quickstart/klingai-video-3-model-user-guide"]
    },
    {
      "id": "bytedance-seedance-2",
      "vendor": "ByteDance Seed",
      "model": "Seedance 2.0",
      "status": "active",
      "layer": "foundation_model",
      "modes": ["text_to_video", "image_to_video", "reference_to_video", "edit_video", "extend_video"],
      "best_for": "最多模态素材驱动、多镜头音画生成、角色/动作/镜头/音频参考与局部修改",
      "prompt_grammar": "先用 @Image/@Video/@Audio 映射每项素材职责，再给总目标和按节拍镜头；写‘角色来自…、场景来自…、动作参考…、声音参考…’。",
      "priorities": ["明确参考角色而不是堆素材", "动作/摄影/音频分槽", "需要安静时显式写无对白/无背景音乐"],
      "controls": ["text/image/audio/video input", "multi-shot", "native audiovisual output", "reference/edit/extend"],
      "negative_strategy": "使用清楚的无对白/无 BGM 等声音边界；视觉约束优先写保留目标。",
      "gotchas": ["访问入口与区域可能不同", "Seedance 2.5 在 Dreamina 官方页面仍有预告/可用性不一致，生产默认锁定 2.0"],
      "official_sources": ["https://seed.bytedance.com/en/blog/seedance-2-0-official-launch", "https://seed.bytedance.com/en/seedance2_0", "https://dreamina.capcut.com/tools/seedance-2-0"]
    },
    {
      "id": "alibaba-wan-2-7",
      "vendor": "Alibaba Cloud",
      "model": "Wan 2.7 API family",
      "status": "active_region_dependent",
      "layer": "foundation_model_api",
      "modes": ["text_to_video", "image_to_video", "first_last_frame", "reference_to_video", "extend_video"],
      "best_for": "最长 15 秒、多镜头、原生音频、图像/音频/视频参考与 API 生产",
      "prompt_grammar": "T2V 基础=主体+场景+运动；进阶加美学和风格。I2V=运动+运镜。多镜头=总述+镜头号+时间码+镜头内容；参考用 Image n / Video n 明确绑定。",
      "priorities": ["I2V 不重写图像", "多镜头写时间码", "音频分别写人声/音效/BGM"],
      "controls": ["2–15 秒", "720p/1080p", "native audio", "multi-shot", "prompt_extend"],
      "negative_strategy": "支持时写独立负向字段；声音边界可明确 No dialogue / No background music。",
      "gotchas": ["Wan 2.7/2.6 的地区与 endpoint 不同", "prompt_extend 开启后必须保存实际有效提示"],
      "official_sources": ["https://www.alibabacloud.com/help/en/model-studio/use-video-generation/", "https://www.alibabacloud.com/help/en/model-studio/text-to-video-prompt"]
    },
    {
      "id": "alibaba-wan-2-2-open",
      "vendor": "Alibaba Wan Team",
      "model": "Wan2.2 Open",
      "status": "active_open_weight",
      "layer": "open_weight_model",
      "adapter_family": "wan-open",
      "openness": "open_source",
      "last_verified_at": "2026-08-14",
      "release_date": "2025-07-28",
      "license": {"name": "Apache-2.0", "production_gate": "standard_review", "url": "https://github.com/Wan-Video/Wan2.2/blob/main/LICENSE.txt"},
      "runtime": {"weights": true, "inference_code": true, "families": ["T2V-A14B", "I2V-A14B", "TI2V-5B", "S2V-14B", "Animate-14B"], "workflow_record_required": true},
      "capabilities": {"resolution": ["480p", "720p"], "fps": [24], "speech_to_video": true, "pose_driving": true, "character_replacement": true},
      "modes": ["text_to_video", "image_to_video", "first_last_frame", "avatar_presenter", "performance_transfer"],
      "best_for": "本地/私有化部署、ComfyUI、音频驱动人像、动作迁移/角色替换与可复现实验",
      "prompt_grammar": "采用详细视觉描述并结合仓库推荐的提示扩展；固定工作流、模型文件、采样参数和 seed。",
      "priorities": ["保存完整 workflow JSON", "记录 checkpoint/LoRA 哈希", "本地提示扩展也要保存前后文本"],
      "controls": ["open weights", "ComfyUI/inference scripts", "T2V/I2V/TI2V", "S2V audio/pose driving", "Animate character animation/replacement"],
      "negative_strategy": "按具体工作流的 negative encoder 输入，不假设与托管 API 一致。",
      "gotchas": ["开源 Wan2.2 与托管 Wan2.7 不是同一版本", "S2V/Animate 与普通 T2V 使用不同输入 contract", "硬件、量化和节点版本会显著影响结果"],
      "official_sources": ["https://github.com/Wan-Video/Wan2.2"]
    },
    {
      "id": "luma-ray-3-14",
      "vendor": "Luma AI",
      "model": "Ray3.14",
      "status": "active",
      "layer": "foundation_model_on_platform",
      "modes": ["text_to_video", "image_to_video", "first_last_frame", "video_to_video"],
      "best_for": "原生 1080p 短镜头、关键帧、Modify/V2V 与稳定动作",
      "prompt_grammar": "T2V 写完整镜头；I2V 写运动；关键帧写路径。生成与修改必须分开使用适配器。",
      "priorities": ["一个镜头一个动作弧", "关键帧端点差异可实现", "明确相机运动"],
      "controls": ["T2V 5/10 秒", "I2V 5 秒", "V2V/Modify 最长约 18 秒", "24fps", "1080p", "keyframes"],
      "negative_strategy": "以正向目标状态和保留项为主。",
      "gotchas": ["Ray3.14 当前不提供 character reference", "Modify 与纯生成的提示逻辑不同"],
      "official_sources": ["https://lumalabs.ai/learning-hub/ray314-user-guide", "https://lumalabs.ai/learning-center/articles/luma-video-models-field-guide"]
    },
    {
      "id": "luma-ray-3-2-modify",
      "vendor": "Luma AI",
      "model": "Ray3.2 Modify",
      "status": "active",
      "layer": "video_editing_model",
      "modes": ["video_to_video", "edit_video"],
      "best_for": "源视频驱动的目标状态修改与风格/角色/场景转换",
      "prompt_grammar": "直接描述最终视频看起来是什么；不要写‘把/替换/转换’的过程。结尾加保留提示：保持原始运动、节奏、构图和摄影机。",
      "priorities": ["目标状态用肯定句", "源视频负责运动", "保留项放结尾"],
      "controls": ["source-video timing", "target-state prompting", "preservation cues"],
      "negative_strategy": "不要列长负向清单；写清最终状态和 preserved attributes。",
      "gotchas": ["重复描述源动作可能与控制输入冲突", "写编辑步骤不如写结果有效"],
      "official_sources": ["https://lumalabs.ai/learning-center/articles/ray-3-2-prompting-outputs-and-controls", "https://lumalabs.ai/learning-hub/how-to-use-modify-video"]
    },
    {
      "id": "minimax-hailuo-2-3",
      "vendor": "MiniMax",
      "model": "Hailuo 2.3 / 2.3 Fast",
      "status": "active",
      "layer": "foundation_model_api",
      "modes": ["text_to_video", "image_to_video", "first_last_frame", "reference_to_video"],
      "best_for": "短镜头、主体参考与显式相机命令",
      "prompt_grammar": "可用自然语言动作描述，并插入官方 [Push in]、[Pan left]、[Tracking shot] 等方括号相机命令；顺序命令按文本顺序执行。",
      "priorities": ["单次最多约三个同时相机命令", "顺序运镜按发生顺序写", "开启优化器时保存有效提示"],
      "controls": ["15 camera commands", "6/10 秒（依模型/分辨率）", "prompt_optimizer", "subject reference"],
      "negative_strategy": "按 API 字段使用；精准控制时可关闭 prompt_optimizer 做对照。",
      "gotchas": ["prompt_optimizer 默认可能改写输入", "相机命令过多会互相竞争"],
      "official_sources": ["https://platform.minimax.io/docs/api-reference/api-overview", "https://platform.minimax.io/docs/guides/video-generation", "https://platform.minimax.io/docs/api-reference/video-generation-t2v"]
    },
    {
      "id": "vidu-q3",
      "vendor": "ShengShu Technology",
      "model": "Vidu Q3 family",
      "status": "active",
      "layer": "foundation_model_api",
      "modes": ["text_to_video", "image_to_video", "reference_to_video"],
      "best_for": "音画直出、智能分镜、广告/剧情等专用变体与参考生视频",
      "prompt_grammar": "按变体先确定任务；再写主体/动作/场景/镜头/声音，多镜头按节拍分段并保持角色锚点。",
      "priorities": ["pro/mix/drama/ad/turbo 选对变体", "音频和画面同步写", "参考角色职责清楚"],
      "controls": ["up to 16s", "540p/720p/1080p", "audio-video output", "smart cuts", "reference-to-video"],
      "negative_strategy": "以具体模型 endpoint 的 schema 为准。",
      "gotchas": ["Q3 变体参数不同", "Vidu S1 是实时交互模型，不是普通离线镜头替代品"],
      "official_sources": ["https://platform.vidu.com/docs/model-map", "https://platform.vidu.com/docs/text-to-video", "https://platform.vidu.com/docs/image-to-video", "https://www.genspi.com/en/about/"]
    },
    {
      "id": "pika-2-5",
      "vendor": "Pika",
      "model": "Pika 2.5 ecosystem",
      "status": "active",
      "layer": "creator_platform_model",
      "modes": ["text_to_video", "image_to_video", "first_last_frame", "edit_video", "performance_transfer"],
      "best_for": "社交短视频、关键帧过渡、快速物理效果、替换/添加/扭转与音频驱动表演",
      "prompt_grammar": "简短写可见动作和效果；Pikaframes 给全局目标和每段关键帧之间的过渡；Pikaffects 用直接动词。",
      "priorities": ["效果化任务保持单一", "关键帧写过渡", "参数由当前 UI 选择"],
      "controls": ["T2V/I2V 5/10 秒", "Pikaframes 5–25 秒", "Pikascenes", "Pikadditions", "Pikaswaps", "Pikatwists", "Pikaffects", "Pikaformance"],
      "negative_strategy": "使用当前产品 UI 提供的能力，不复制旧版 Discord 命令。",
      "gotchas": ["旧 /create 和 -motion/-gs/-camera 文档属于遗留语法", "不同 Pika 工具不是同一个 prompt contract"],
      "official_sources": ["https://pika.art/pricing", "https://pika.art/faq"]
    },
    {
      "id": "xai-grok-imagine-video-1-5",
      "vendor": "xAI",
      "model": "Grok Imagine Video 1.5",
      "model_id": "grok-imagine-video-1.5",
      "status": "active",
      "layer": "foundation_model_api",
      "modes": ["text_to_video", "image_to_video"],
      "best_for": "快速文本/图像视频、运动、物理、音频与对白",
      "prompt_grammar": "直接写主体、动作、场景、相机和情绪；I2V 让源图负责外观，提示词描述接下来如何运动及声音。",
      "priorities": ["一个清晰动作", "相机语言直接", "声音与说话人明确"],
      "controls": ["image input", "duration", "resolution", "audio/speech"],
      "negative_strategy": "按 Imagine API 当前参数使用，文本以肯定目标为主。",
      "gotchas": ["产品 UI 与 API 暴露的模式/限制可能不同", "精确时长和分辨率应放参数"],
      "official_sources": ["https://x.ai/news/grok-imagine-video-1-5", "https://docs.x.ai/developers/model-capabilities/video/generation", "https://docs.x.ai/developers/models/grok-imagine-video"]
    },
    {
      "id": "lightricks-ltx-2-5",
      "vendor": "Lightricks",
      "model": "LTX-2.5",
      "status": "active_open_weight",
      "layer": "open_weight_model",
      "adapter_family": "ltx",
      "openness": "restricted_open_weight",
      "last_verified_at": "2026-08-14",
      "release_date": "2026-08-11",
      "license": {"name": "LTX-2.x Community License", "production_gate": "manual_review", "commercial_threshold": "年营收达到 1000 万美元的实体，商业用途需另行取得付费许可；以许可证原文为准。", "url": "https://github.com/Lightricks/LTX-2/blob/main/LICENSE.md"},
      "runtime": {"weights": true, "inference_code": true, "parameter_scale": "22B distilled transformer", "download_size_note": "官方 quickstart 所列核心文件约 66 GiB", "workflow_record_required": true},
      "capabilities": {"joint_audio_video": true, "pipelines": ["distilled", "keyframe", "retake", "extend"], "fine_tuning": ["LoRA"]},
      "modes": ["text_to_video", "image_to_video", "first_last_frame", "reference_to_video", "extend_video", "edit_video"],
      "best_for": "开放权重音画联合生成、关键帧、retake、extend 与 LoRA 训练",
      "prompt_grammar": "把主动作放第一句，以连续自然语言按时间写人物/物体运动、摄影、灯光和同步音频；参数走 pipeline 字段。",
      "priorities": ["画面和声音按同一时间轴", "固定 2.5 checkpoint 家族", "保存原始/增强提示和完整推理配置"],
      "controls": ["joint audio-video", "keyframe conditioning", "retake", "extend", "LoRA trainer", "distilled pipeline"],
      "negative_strategy": "使用对应 pipeline 的 negative 字段；不要复用 2.3 的权重、LoRA 或 VAE 文件。",
      "gotchas": ["LTX-2.3 文件和 LoRA 与 2.5 不互换", "许可证含商业规模门槛，生产接入必须单独审查"],
      "official_sources": ["https://github.com/Lightricks/LTX-2", "https://github.com/Lightricks/LTX-2/blob/main/LICENSE.md", "https://huggingface.co/Lightricks/LTX-2.5"]
    },
    {
      "id": "lightricks-ltx-2-3",
      "vendor": "Lightricks",
      "model": "LTX-2.3",
      "status": "legacy_open_weight",
      "layer": "foundation_model",
      "adapter_family": "ltx",
      "openness": "restricted_open_weight",
      "last_verified_at": "2026-08-14",
      "replaced_by": "lightricks-ltx-2-5",
      "license": {"name": "Version-specific LTX license", "production_gate": "manual_review", "url": "https://github.com/Lightricks/LTX-2/blob/main/MODELS-LTX-2.3.md"},
      "runtime": {"weights": true, "inference_code": true, "legacy": true, "workflow_record_required": true},
      "modes": ["text_to_video", "image_to_video", "extend_video", "edit_video", "reference_to_video"],
      "best_for": "音画联合生成、较长 API 镜头、开源 ComfyUI、retake/extend/audio-to-video",
      "prompt_grammar": "少于约 200 词的连续段落：先主动作，再按时间写动作/手势、外观、背景、摄影、灯光色彩和变化，音频嵌入相应时刻。",
      "priorities": ["按时间顺序", "主动作开头", "画面与音频同一段编排"],
      "controls": ["joint audio-video", "API up to 4K/20s", "T2V/I2V", "audio-to-video", "retake", "extend", "ComfyUI"],
      "negative_strategy": "按 API 或工作流节点区分；不要把旧 LTX-2 设置复制到 2.3。",
      "gotchas": ["官方当前仓库已把 2.3 标为 Legacy", "2.3 权重、VAE 与 LoRA 不可和 2.5 混用", "开源工作流必须记录 checkpoint、节点和采样参数"],
      "official_sources": ["https://docs.ltx.video/", "https://docs.ltx.video/pricing", "https://github.com/Lightricks/LTX-2", "https://docs.ltx.video/open-source-model/integration-tools/comfy-ui"]
    },
    {
      "id": "tencent-hunyuan-video-1-5",
      "vendor": "Tencent Hunyuan",
      "model": "HunyuanVideo-1.5",
      "status": "active_open_weight",
      "layer": "open_weight_model",
      "adapter_family": "hunyuan",
      "openness": "restricted_open_weight",
      "last_verified_at": "2026-08-14",
      "license": {"name": "Tencent Hunyuan Community License", "production_gate": "manual_review", "url": "https://github.com/Tencent-Hunyuan/HunyuanVideo-1.5/blob/main/LICENSE"},
      "runtime": {"weights": true, "inference_code": true, "training_code": true, "parameter_scale": "8.3B", "variants": ["T2V", "I2V", "480p I2V step-distilled"]},
      "capabilities": {"text_to_video": true, "image_to_video": true, "step_distillation": ["4", "8", "12"], "fine_tuning": ["LoRA"]},
      "modes": ["text_to_video", "image_to_video"],
      "best_for": "8.3B 级开源本地部署、中英文提示、研究与定制工作流",
      "prompt_grammar": "Subject + Motion + Scene + 可选 Shot/Camera/Lighting/Style/Atmosphere；详细、客观、按时间，方向和动作归属明确。",
      "priorities": ["用物理可观察语言", "明确左右/前后与谁在动作", "固定推理配置"],
      "controls": ["open weights", "T2V/I2V", "Chinese/English"],
      "negative_strategy": "按官方/ComfyUI 工作流的独立负向编码设置。",
      "gotchas": ["腾讯云聚合的 Kling/Hailuo/Vidu 不是混元模型", "云 API 的长度和字段与本地开源推理不同"],
      "official_sources": ["https://github.com/Tencent-Hunyuan/HunyuanVideo-1.5", "https://github.com/Tencent-Hunyuan/HunyuanVideo-1.5/blob/main/assets/HunyuanVideo_1_5_Prompt_Handbook_EN.md", "https://cloud.tencent.com/document/product/1823/130081"]
    },
    {
      "id": "baidu-musesteamer-2-1",
      "vendor": "Baidu",
      "model": "MuseSteamer 2.1 / 2.0 audio variants",
      "status": "active_api",
      "layer": "foundation_model_api",
      "modes": ["image_to_video", "reference_to_video"],
      "best_for": "中文复杂运镜、人物对白/音频与图生视频",
      "prompt_grammar": "从源图开始写人物动作、镜头和声音；多人对白显式标位置与说话人，例如‘左边人物说话：…’。",
      "priorities": ["中文角色定位", "对白逐字和说话人", "运镜与人物动作分开"],
      "controls": ["I2V", "audio/dialogue variants", "camera instructions"],
      "negative_strategy": "遵循千帆具体 endpoint 字段。",
      "gotchas": ["2.1 与 2.0 音频/效果能力需按接口区分", "平台配额与区域可用性需实时核验"],
      "official_sources": ["https://cloud.baidu.com/doc/qianfan-docs/s/rmejr4y27", "https://cloud.baidu.com/doc/qianfan-api/s/Xme6ul5g4"]
    },
    {
      "id": "moonvalley-marey",
      "vendor": "Moonvalley",
      "model": "Marey / Realism v1.5",
      "status": "active",
      "layer": "professional_video_model",
      "modes": ["text_to_video", "image_to_video", "first_last_frame", "reference_to_video", "performance_transfer", "extend_video"],
      "best_for": "专业镜头控制、动作迁移、轨迹、关键帧、姿态和强调数据许可来源的制作",
      "prompt_grammar": "从核心动作到细节写完整目标；动作迁移时文本主要定义主体、场景、造型与外观，并由控制视频承担运动。",
      "priorities": ["主动作开头", "50+ 词动作迁移描述作为官方建议起点", "结构控制交给轨迹/姿态/关键帧"],
      "controls": ["camera control", "motion transfer", "trajectory", "keyframes", "references", "pose", "extension"],
      "negative_strategy": "用控制输入和保留项减少文本否定。",
      "gotchas": ["‘licensed-data training’是厂商声明，不等于对任何输入/输出的法律许可", "动作控制与文本冲突时要删掉冲突文本"],
      "official_sources": ["https://www.moonvalley.com/marey", "https://www.moonvalley.com/beyondtheframe/marey-launches-on-fal-ai", "https://help.moonvalley.com/en/articles/11898189-motion-transfer"]
    },
    {
      "id": "higgsfield-cinema-studio-3-5",
      "vendor": "Higgsfield",
      "model": "Cinema Studio 3.5 workflow",
      "status": "active_platform",
      "layer": "multi_model_creator_platform",
      "modes": ["text_to_video", "image_to_video", "reference_to_video"],
      "best_for": "电影化 Hero Frame、导演面板、镜头/焦距/光圈/灯光和多模型对比",
      "prompt_grammar": "先制作 Hero Frame；文本写主体、场景和动作，类型、灯光、颜色、镜头、焦距、光圈、运动由 Director Panel 固定。",
      "priorities": ["Hero Frame First", "控件与文本不重复冲突", "每次对比只换模型或一个设置"],
      "controls": ["genre", "lighting", "color", "camera", "lens", "focal length", "aperture", "movement", "model picker"],
      "negative_strategy": "按底层模型适配，不把平台字段误当通用 negative。",
      "gotchas": ["Higgsfield 是多模型平台，不是单一基础模型", "平台排行榜属于营销材料，不作为本库质量结论"],
      "official_sources": ["https://higgsfield.ai/cinematic-video-generator", "https://higgsfield.ai/academy/courses/cinema-studio-complete-tour/video-generation", "https://higgsfield.ai/blog/ai-video-camera-control"]
    },
    {
      "id": "heygen-avatar-v",
      "vendor": "HeyGen",
      "model": "Avatar V / v3 Video API",
      "status": "active",
      "layer": "avatar_video_platform",
      "modes": ["avatar_presenter"],
      "best_for": "授权数字人、脚本或录音口播、声音参数、动作意图、批量个性化与翻译",
      "prompt_grammar": "先写逐场脚本；句子短、口语化，标注停顿、重音、发音和语气。motion_prompt 只写与台词匹配的自然表演意图，背景/画幅/字幕用结构化字段。",
      "priorities": ["脚本优先", "声音 ID/语言/速度独立设置", "身份和声音必须有授权"],
      "controls": ["avatar_id", "script/audio", "voice settings", "motion_prompt", "background", "caption", "aspect ratio"],
      "negative_strategy": "不适用传统视觉 negative；用脚本、动作提示和版式字段精确约束。",
      "gotchas": ["Avatar III 使用遗留 API，v3 支持 Avatar IV/V", "不要在脚本中混入不可朗读的摄影说明"],
      "official_sources": ["https://developers.heygen.com/reference/create-video", "https://developers.heygen.com/"]
    },
    {
      "id": "synthesia-assistant",
      "vendor": "Synthesia",
      "model": "Synthesia Assistant",
      "status": "beta_active",
      "layer": "avatar_video_platform",
      "modes": ["avatar_presenter"],
      "best_for": "企业培训、讲解、文档/URL 到多场景视频、品牌模板与协作迭代",
      "prompt_grammar": "主题 + 受众 + 目标 + 必须使用/忽略的来源范围 + 时长 + delivery style；生成后在 Storyboard 按脚本、视觉与节奏逐场修改。",
      "priorities": ["主题/受众/目标写全", "说明如何使用上传资料", "Cinematic 与 Presentation 二选一"],
      "controls": ["files/URLs", "30s–5min", "delivery style", "template", "brand kit", "Storyboard chat"],
      "negative_strategy": "用来源边界和逐场修改，不使用视觉模型式负向提示。",
      "gotchas": ["Assistant 仍为 Beta", "每个 scene 的 spoken script 与制作说明分开"],
      "official_sources": ["https://docs.synthesia.io/docs/assistant", "https://help.synthesia.io/en/articles/13214011-how-do-i-create-a-script-in-synthesia"]
    },
    {
      "id": "tavus-replica",
      "vendor": "Tavus",
      "model": "Replicas / Video API / CVI",
      "status": "active",
      "layer": "avatar_video_and_realtime_platform",
      "modes": ["avatar_presenter", "interactive_avatar"],
      "best_for": "脚本/录音异步数字人视频与 persona 驱动的实时视频会话",
      "prompt_grammar": "异步 Video API 主要传 script 或 audio_url；实时 CVI 另写 persona：目标、语气、知识、限制、回退和人工升级。两类提示不可混用。",
      "priorities": ["区分异步视频与实时对话", "脚本控制句长和自然停顿", "persona 明确知识和行为边界"],
      "controls": ["replica_id", "script/audio_url", "background URL/video", "transparent background", "persona", "guardrails"],
      "negative_strategy": "用 persona/guardrails 与结构化参数约束。",
      "gotchas": ["克隆训练要求授权/同意材料", "过长脚本可能出现重复手势或不自然表现"],
      "official_sources": ["https://docs.tavus.io/sections/video/quickstart", "https://docs.tavus.io/api-reference/video-request/create-video", "https://docs.tavus.io/api-reference/overview", "https://docs.tavus.io/sections/onboarding-guide/prompting-guide"]
    },
    {
      "id": "did-avatar",
      "vendor": "D-ID",
      "model": "V4/V3/V2 Avatars",
      "status": "active",
      "layer": "avatar_video_platform",
      "modes": ["avatar_presenter", "interactive_avatar"],
      "best_for": "照片/演员 + 文本或录音生成 talking head、视频翻译与实时数字人",
      "prompt_grammar": "视觉由 source/actor 决定；提示工程集中在可朗读脚本、情绪/语音设置、停顿、背景和互动 agent 指令。",
      "priorities": ["脚本和音频可二选一", "逐字检查发音和停顿", "实时 agent 与异步 talk 分开"],
      "controls": ["source image/actor", "text/audio script", "voice", "avatar sentiment", "realtime agents"],
      "negative_strategy": "使用结构化配置与 agent 规则，不套生成视频负向词。",
      "gotchas": ["名人/身份与内容有审核规则", "图片和声音使用权必须留档"],
      "official_sources": ["https://www.d-id.com/api/", "https://docs.d-id.com/reference/createtalk", "https://docs.d-id.com/docs/quickstart"]
    },
    {
      "id": "midjourney-video",
      "vendor": "Midjourney",
      "model": "Midjourney Video",
      "status": "active_hosted",
      "layer": "creator_platform_model",
      "adapter_family": "midjourney-video",
      "openness": "hosted_only",
      "last_verified_at": "2026-08-14",
      "modes": ["image_to_video", "first_last_frame", "extend_video"],
      "best_for": "从 Midjourney 图像或外部起始帧生成风格一致的短运动片段、循环和首尾帧过渡",
      "prompt_grammar": "起始图像负责画面；motion prompt 只写主体、环境和摄影机接下来如何运动，参数放在文本尾部。",
      "priorities": ["运动描述短而明确", "主体运动和相机运动分开", "需要端点时使用 --end 而不是在 prose 里重画尾帧"],
      "controls": ["5-second initial clip", "--motion low/high", "--raw", "--loop", "--end", "--bs"],
      "negative_strategy": "使用正向运动目标和平台参数；不要假设图像模型的 --no 在视频中具有同等行为。",
      "gotchas": ["视频以起始图像为必需输入", "参数与可用入口会变化，提交前检查当前文档"],
      "official_sources": ["https://docs.midjourney.com/hc/en-us/articles/37460773864589-Video"]
    },
    {
      "id": "pixverse-v6",
      "vendor": "PixVerse",
      "model": "PixVerse v6",
      "status": "active_api",
      "layer": "foundation_model_api",
      "adapter_family": "pixverse",
      "openness": "hosted_only",
      "last_verified_at": "2026-08-14",
      "modes": ["text_to_video", "image_to_video", "first_last_frame", "edit_video", "avatar_presenter"],
      "best_for": "T2V/I2V、转场、效果、口型同步和可结构化控制的 API 生产",
      "prompt_grammar": "文本写可见主体、动作、场景与摄影机；aspect、duration、quality、seed、camera 与 negative 分别放 API 字段。",
      "priorities": ["先选正确 endpoint", "相机控制不与 prose 冲突", "保存 seed 与 negative 字段"],
      "controls": ["aspect_ratio", "duration", "quality", "seed", "camera_movement", "negative_prompt", "transition", "lip-sync", "sound"],
      "negative_strategy": "使用 API 的独立 negative_prompt 字段。",
      "gotchas": ["effect/transition/lip-sync 不是普通 T2V 的同一输入 contract", "能力以 v6 capability matrix 为准"],
      "official_sources": ["https://docs.platform.pixverse.ai/text-to-video-generation-13016634e0", "https://docs.platform.pixverse.ai/capability-matrix-2144288m0"]
    },
    {
      "id": "canva-magic-video",
      "vendor": "Canva",
      "model": "Canva AI / Magic Video / Veo connector",
      "status": "active_platform",
      "layer": "multi_model_creator_platform",
      "adapter_family": "model-picker-platform",
      "openness": "hosted_only",
      "last_verified_at": "2026-08-14",
      "modes": ["text_to_video", "image_to_video", "edit_video", "avatar_presenter"],
      "best_for": "生成、素材自动剪辑、可编辑图层、品牌版式与团队交付的一体化工作流",
      "prompt_grammar": "先声明是生成新镜头还是剪辑已有素材；brief 写受众、渠道、时长、叙事、品牌与 CTA，底层模型另行记录。",
      "priorities": ["任务类型先于视觉形容词", "记录底层模型", "生成后用可编辑图层和品牌系统完成交付"],
      "controls": ["Magic Media", "Magic Video", "Veo 3 integration", "editable layers", "brand kit"],
      "negative_strategy": "由实际选择的底层模型决定；平台版式限制用编辑控件实现。",
      "gotchas": ["Canva 是平台而不是单一视频基础模型", "素材剪辑 brief 与生成式 prompt 不可混用"],
      "official_sources": ["https://www.canva.com/newsroom/news/canva-create-2026-ai/", "https://www.canva.com/newsroom/news/canva-video/", "https://www.canva.com/newsroom/news/veo3-canva-ai-video/"]
    },
    {
      "id": "krea-video-model-picker",
      "vendor": "Krea",
      "model": "Krea AI Video model picker",
      "status": "active_platform",
      "layer": "multi_model_creator_platform",
      "adapter_family": "model-picker-platform",
      "openness": "hosted_only",
      "last_verified_at": "2026-08-14",
      "modes": ["text_to_video", "image_to_video", "first_last_frame", "reference_to_video", "video_to_video"],
      "best_for": "从同一界面对比多家视频模型并进行增强、关键帧和工作流迭代",
      "prompt_grammar": "先选模型和模式，再按底层模型适配；平台级增强与底层 prompt 必须分别保存。",
      "priorities": ["模型 ID 必须入库", "相同 ShotSpec 跨模型对比", "不要把平台能力归因于底层模型"],
      "controls": ["multi-model picker", "keyframes", "enhance", "video workflows"],
      "negative_strategy": "按底层模型 endpoint 处理。",
      "gotchas": ["模型列表和参数会动态变化", "同一文本跨模型不能视为公平对比，需经过各自适配器"],
      "official_sources": ["https://www.krea.ai/features/ai-video-generator"]
    },
    {
      "id": "freepik-ai-video-suite",
      "vendor": "Freepik",
      "model": "Freepik AI Video model suite",
      "status": "active_platform",
      "layer": "multi_model_creator_platform",
      "adapter_family": "model-picker-platform",
      "openness": "hosted_only",
      "last_verified_at": "2026-08-14",
      "modes": ["text_to_video", "image_to_video", "first_last_frame", "reference_to_video", "video_to_video", "avatar_presenter"],
      "best_for": "多模型生成、素材库与设计资产衔接的创作工作流",
      "prompt_grammar": "以当前模型卡为准建立 input contract；记录模型、模式、比例、时长、参考槽位和增强后的有效提示。",
      "priorities": ["按模型卡适配", "资产来源与许可单独留档", "输出回到设计工作流完成文字和品牌元素"],
      "controls": ["multi-model selection", "T2V/I2V", "references", "platform assets"],
      "negative_strategy": "仅在所选模型提供独立字段时使用。",
      "gotchas": ["平台模型阵容变化快", "素材库许可和生成模型许可是两件事"],
      "official_sources": ["https://www.freepik.com/ai/docs/video-ai-models"]
    },
    {
      "id": "invideo-agentic-video",
      "vendor": "InVideo",
      "model": "Agent One/Two / Autopilot",
      "status": "active_platform",
      "layer": "agentic_video_platform",
      "adapter_family": "script-platform",
      "openness": "hosted_only",
      "last_verified_at": "2026-08-14",
      "modes": ["text_to_video", "avatar_presenter"],
      "best_for": "从 brief 或已有脚本生成旁白、分镜、素材、字幕、音乐与成片",
      "prompt_grammar": "项目 brief = 目标 + 受众 + 渠道 + 时长 + 结构 + 事实来源 + 品牌 + 素材策略 + 旁白风格 + CTA；逐轮只改一个制作层。",
      "priorities": ["先锁脚本事实与结构", "再审分镜和素材", "最后调节奏、字幕、音乐与 CTA"],
      "controls": ["script-to-video", "conversational agents", "Autopilot", "voiceover", "stock/generated media"],
      "negative_strategy": "以来源范围、禁用素材类型和逐层修改约束，不使用镜头模型式长 negative。",
      "gotchas": ["项目级 agent brief 不是单镜头 prompt", "AI 生成事实与素材许可均需人工审核"],
      "official_sources": ["https://help.invideo.io/en/articles/9382180-how-can-i-create-a-video-using-my-script", "https://invideo.io/blog/agentic-ai-video-workflow/"]
    },
    {
      "id": "captions-mirage-video-1",
      "vendor": "Captions",
      "model": "mirage-video-1-latest",
      "model_id": "mirage-video-1-latest",
      "status": "active_api",
      "layer": "avatar_video_platform",
      "adapter_family": "mirage-avatar",
      "openness": "hosted_only",
      "last_verified_at": "2026-08-14",
      "modes": ["avatar_presenter"],
      "best_for": "静态人物图 + 音频驱动的真人感表演视频",
      "prompt_grammar": "图像锁定人物与造型，音频锁定台词和节奏；文本只写合适的表情、眼神、手势、景别、背景和必须保持项。",
      "priorities": ["人物与声音均需授权", "音频先完成", "表演提示服从语音节奏"],
      "controls": ["still image", "audio", "human performance generation"],
      "negative_strategy": "用身份、构图和背景保留边界。",
      "gotchas": ["不是纯文本视频模型", "真人身份或声音缺少可验证同意记录时不得进入生产"],
      "official_sources": ["https://captions.ai/help/docs/api/video-generation"]
    },
    {
      "id": "skywork-skyreels-v3",
      "vendor": "SkyworkAI",
      "model": "SkyReels V3",
      "status": "active_open_weight",
      "layer": "open_weight_model",
      "adapter_family": "skyreels-v3",
      "openness": "restricted_open_weight",
      "last_verified_at": "2026-08-14",
      "release_date": "2026-01-29",
      "license": {"name": "Skywork Community License", "production_gate": "manual_review", "commercial_use": "许可证声明支持商业用途，但要求遵守社区许可与部署合规条件。", "url": "https://github.com/SkyworkAI/SkyReels-V3/blob/main/LICENSE.txt"},
      "runtime": {"weights": true, "inference_code": true, "variants": ["R2V-14B-720P", "Video-Extension-14B-720P", "Talking-Avatar-19B-720P"]},
      "capabilities": {"reference_images": {"min": 1, "max": 4}, "extension_seconds": {"min": 5, "max": 30}, "avatar_audio_seconds_max": 200, "resolution": ["480P", "540P", "720P"], "fps": [24]},
      "modes": ["reference_to_video", "video_to_video", "extend_video", "avatar_presenter"],
      "best_for": "多主体参考、视频续写/智能切镜和单图+音频长数字人",
      "prompt_grammar": "R2V 逐一绑定 1–4 张参考图的角色/物体/背景职责；续写从末帧状态继续；数字人由图像和音频负责身份/台词，文本写表演与机位。",
      "priorities": ["参考职责互斥且明确", "续写标注 single-shot 或 shot-switching", "保存 task_type、模型变体、分辨率与 seed"],
      "controls": ["1–4 reference images", "single-shot extension 5–30s", "shot-switching extension", "talking avatar <=200s", "low_vram"],
      "negative_strategy": "由具体推理脚本/工作流决定；参考冲突优先通过资产职责修复。",
      "gotchas": ["三个模型变体不能混用输入参数", "社区许可证不是 Apache/MIT，生产前需法务核验"],
      "official_sources": ["https://github.com/SkyworkAI/SkyReels-V3", "https://github.com/SkyworkAI/SkyReels-V3/blob/main/LICENSE.txt"]
    },
    {
      "id": "sandai-magi-1-1",
      "vendor": "SandAI",
      "model": "MAGI-1.1 24B",
      "status": "active_open_source",
      "layer": "open_weight_model",
      "adapter_family": "magi",
      "openness": "open_source",
      "last_verified_at": "2026-08-14",
      "release_date": "2026-06-17",
      "license": {"name": "Apache-2.0", "production_gate": "standard_review", "url": "https://github.com/SandAI-org/MAGI-1/blob/main/LICENSE"},
      "runtime": {"weights": true, "inference_code": true, "variants": ["24B", "24B distilled", "24B quantized", "4.5B distilled"]},
      "capabilities": {"streaming_generation": true, "chunk_wise_prompting": true, "long_horizon": true},
      "modes": ["image_to_video", "extend_video"],
      "best_for": "自回归长视频、流式生成、按 chunk 改变场景和细粒度时序控制",
      "prompt_grammar": "先给全局主体/世界/风格锚点，再为每个连续 chunk 写时间范围、动作、摄影和从上一段继承的状态。",
      "priorities": ["chunk 间重复连续性锚点", "每段只引入一个主要变化", "记录蒸馏/量化 checkpoint"],
      "controls": ["chunk-wise prompts", "streaming generation", "distilled/quantized variants", "ComfyUI"],
      "negative_strategy": "跟随具体 workflow 的 negative 编码；长序列漂移用锚点与小步续写修复。",
      "gotchas": ["核心强项是 I2V/长时序而非通用短 T2V 替代", "不同 24B/4.5B 变体不可直接比较"],
      "official_sources": ["https://github.com/SandAI-org/MAGI-1"]
    },
    {
      "id": "meituan-longcat-video",
      "vendor": "Meituan LongCat",
      "model": "LongCat-Video / Avatar-1.5",
      "status": "active_open_source",
      "layer": "open_weight_model",
      "adapter_family": "longcat",
      "openness": "open_source",
      "last_verified_at": "2026-08-14",
      "release_date": "2026-05-21",
      "license": {"name": "MIT", "production_gate": "standard_review", "url": "https://github.com/meituan-longcat/LongCat-Video/blob/main/LICENSE"},
      "runtime": {"weights": true, "inference_code": true, "parameter_scale": "13.6B", "entrypoints": ["T2V", "I2V", "video continuation", "long video", "interactive video", "avatar"]},
      "capabilities": {"resolution": ["720p"], "fps": [30], "minutes_long_continuation": true, "audio_driven_avatar": true},
      "modes": ["text_to_video", "image_to_video", "extend_video", "avatar_presenter"],
      "best_for": "统一 T2V/I2V/续写、分钟级长视频与音频驱动人物",
      "prompt_grammar": "短镜头写完整视觉动作；长视频先固定世界/人物/相机规则，再为每次 continuation 写承接状态、下一事件和不漂移项。",
      "priorities": ["长片按续写段版本化", "每段记录末帧状态", "avatar 按人物绑定各自音频"],
      "controls": ["T2V", "I2V", "video continuation", "long video", "interactive video", "Avatar-1.5"],
      "negative_strategy": "按仓库脚本字段；长视频优先用连续性锚点。",
      "gotchas": ["长时序输出仍需逐段质量门禁", "Avatar 与基础视频任务使用不同入口"],
      "official_sources": ["https://github.com/meituan-longcat/LongCat-Video"]
    },
    {
      "id": "hpcaitech-open-sora-2",
      "vendor": "HPC-AI Tech",
      "model": "Open-Sora 2.0 11B",
      "status": "active_open_source",
      "layer": "open_weight_model",
      "adapter_family": "open-sora",
      "openness": "open_source",
      "last_verified_at": "2026-08-14",
      "release_date": "2025-03-12",
      "license": {"name": "Apache-2.0", "production_gate": "standard_review", "url": "https://github.com/hpcaitech/Open-Sora/blob/main/LICENSE"},
      "runtime": {"weights": true, "inference_code": true, "training_code": true, "parameter_scale": "11B"},
      "capabilities": {"resolution": ["256px", "768px"], "aspect_ratios": ["16:9", "9:16", "1:1", "2.39:1"], "frame_rule": "4k+1, <129", "motion_score": true},
      "modes": ["text_to_video", "image_to_video"],
      "best_for": "完全开放的训练/推理研究、T2V/I2V、motion score 与可复现实验",
      "prompt_grammar": "使用完整视觉描述；I2V 提供 ref 并写接下来发生的运动。aspect_ratio、num_frames、motion_score、seed 与 refine_prompt 都是独立参数。",
      "priorities": ["参数不混入 prose", "固定 seed 和 num_frames", "保存是否使用 prompt refine"],
      "controls": ["T2V", "I2V", "motion score", "prompt refine", "seed", "multi-GPU"],
      "negative_strategy": "按配置文件的实际字段；不要用不存在的托管 API 语法。",
      "gotchas": ["2.0 更偏 I2V，T2V 可走 T2I2V pipeline", "768px 官方示例对显存与多卡要求较高"],
      "official_sources": ["https://github.com/hpcaitech/Open-Sora"]
    },
    {
      "id": "stepfun-step-video-t2v-ti2v",
      "vendor": "StepFun",
      "model": "Step-Video-T2V / TI2V",
      "status": "stable_open_source",
      "layer": "open_weight_model",
      "adapter_family": "step-video",
      "openness": "open_source",
      "last_verified_at": "2026-08-14",
      "release_date": "2025-03-17",
      "license": {"name": "MIT", "production_gate": "standard_review", "url": "https://github.com/stepfun-ai/Step-Video-T2V/blob/main/LICENSE"},
      "runtime": {"weights": true, "inference_code": true, "parameter_scale": "30B", "variants": ["T2V", "T2V-Turbo", "TI2V"]},
      "capabilities": {"frames_max": 204, "text_to_video": true, "image_to_video": true},
      "modes": ["text_to_video", "image_to_video"],
      "best_for": "30B 级开放权重 T2V、Turbo 与后续 TI2V 研究/部署",
      "prompt_grammar": "长描述按主体、动作、场景、摄影、光线和时间顺序组织；I2V 不复述源图，重点写运动。",
      "priorities": ["T2V/Turbo/TI2V checkpoint 分开", "固定帧数和推理配置", "大模型部署先估显存"],
      "controls": ["T2V", "T2V-Turbo", "TI2V", "up to 204 frames"],
      "negative_strategy": "跟随官方推理配置。",
      "gotchas": ["模型较大，部署成本高", "仓库更新节奏较早，使用前检查依赖兼容性"],
      "official_sources": ["https://github.com/stepfun-ai/Step-Video-T2V"]
    },
    {
      "id": "zai-cogvideox-1-5",
      "vendor": "Z.ai / THUDM",
      "model": "CogVideoX1.5-5B / I2V",
      "status": "stable_open_weight",
      "layer": "open_weight_model",
      "adapter_family": "cogvideo",
      "openness": "open_weight",
      "last_verified_at": "2026-08-14",
      "release_date": "2024-11-08",
      "license": {"name": "Model-specific; repository code Apache-2.0", "production_gate": "manual_review", "url": "https://github.com/zai-org/CogVideo"},
      "runtime": {"weights": true, "inference_code": true, "parameter_scale": "5B", "integrations": ["SAT", "Diffusers", "CogKit"]},
      "capabilities": {"duration_seconds": {"max": 10}, "t2v_resolution": "1360x768", "i2v_resolution": "min side 768, max side <=1360", "frame_rules": ["T2V 16N+1", "I2V 8N+1"]},
      "modes": ["text_to_video", "image_to_video", "video_to_video"],
      "best_for": "成熟 5B 级 T2V/I2V、Diffusers、LoRA 与 DDIM inversion 视频编辑研究",
      "prompt_grammar": "详细写主体、动作、场景、摄影和风格；I2V 只补充运动与未由图像确定的信息。",
      "priorities": ["遵守帧数规则", "T2V/I2V 分 checkpoint", "模型许可证逐个核验"],
      "controls": ["T2V", "I2V", "LoRA", "DDIM inverse", "Diffusers", "CogKit"],
      "negative_strategy": "使用对应 pipeline 的 negative_prompt。",
      "gotchas": ["仓库 Apache 许可不自动代表每个模型权重同为 Apache", "属于稳定/基线级而非 2026 最新发布"],
      "official_sources": ["https://github.com/zai-org/CogVideo"]
    },
    {
      "id": "genmo-mochi-1",
      "vendor": "Genmo",
      "model": "Mochi 1 Preview 10B",
      "status": "stable_open_source",
      "layer": "open_weight_model",
      "adapter_family": "mochi",
      "openness": "open_source",
      "last_verified_at": "2026-08-14",
      "release_date": "2024-11-26",
      "license": {"name": "Apache-2.0", "production_gate": "standard_review", "url": "https://github.com/genmoai/mochi/blob/main/LICENSE"},
      "runtime": {"weights": true, "inference_code": true, "parameter_scale": "10B", "fine_tuning": ["LoRA"]},
      "capabilities": {"text_to_video": true, "single_gpu_reference_vram": "约 60GB（官方仓库实现）"},
      "modes": ["text_to_video"],
      "best_for": "Apache-2.0 开放基线、运动质量研究、LoRA 与 ComfyUI 生态",
      "prompt_grammar": "使用自然、具体的主体、动作、环境与摄影描述，避免互斥动作和文字渲染要求。",
      "priorities": ["固定 seed/CFG/采样配置", "记录优化/量化实现", "作为稳定基线而不是追逐最新能力"],
      "controls": ["T2V", "LoRA", "multi/single GPU", "ComfyUI community integrations"],
      "negative_strategy": "按具体 pipeline 使用。",
      "gotchas": ["官方称 Preview", "原生仓库显存需求较高，社区低显存实现需记录差异"],
      "official_sources": ["https://github.com/genmoai/mochi"]
    },
    {
      "id": "nvidia-cosmos-3-generator",
      "vendor": "NVIDIA",
      "model": "Cosmos 3 Generator",
      "status": "active_open_weight_world_model",
      "layer": "open_weight_world_model",
      "adapter_family": "cosmos",
      "openness": "restricted_open_weight",
      "last_verified_at": "2026-08-14",
      "license": {"name": "OpenMDW-1.1", "production_gate": "manual_review", "url": "https://github.com/NVIDIA/cosmos/blob/main/LICENSE"},
      "runtime": {"weights": true, "inference_code": true, "integrations": ["Diffusers", "vLLM-Omni", "NIM", "SGLang"]},
      "capabilities": {"world_model": true, "modalities": ["language", "image", "video", "audio", "action"]},
      "modes": ["text_to_video", "image_to_video", "video_to_video"],
      "best_for": "机器人、自动驾驶、智能基础设施等 Physical AI 的世界生成与推理",
      "prompt_grammar": "写可观测环境状态、主体/传感器视角、动作条件、物理变化、时间跨度和需要预测的未来状态。",
      "priorities": ["先确认 Physical AI 适用性", "动作与观察视角结构化", "模型/数据/工具许可证分别核验"],
      "controls": ["omnimodal generator", "world reasoning", "action-conditioned workflows", "multiple inference backends"],
      "negative_strategy": "使用任务配置；安全关键场景必须用仿真/现实验证，不能靠 negative prompt。",
      "gotchas": ["不是默认的广告/影视创作模型", "OpenMDW-1.1 与 Apache/MIT 不同，生产前需审查"],
      "official_sources": ["https://github.com/NVIDIA/cosmos"]
    },
    {
      "id": "alibaba-vace-wan2-1",
      "vendor": "Alibaba Tongyi Lab",
      "model": "VACE / Wan2.1-VACE 14B",
      "status": "stable_open_weight",
      "layer": "open_weight_video_editing_model",
      "adapter_family": "vace",
      "openness": "open_weight",
      "last_verified_at": "2026-08-14",
      "release_date": "2025-05-14",
      "license": {"name": "Per base model (Apache-2.0 or RAIL-M)", "production_gate": "manual_review", "url": "https://github.com/ali-vilab/VACE"},
      "runtime": {"weights": true, "inference_code": true, "variants": ["Wan2.1-VACE-1.3B", "Wan2.1-VACE-14B", "VACE-LTX-Video-0.9"]},
      "capabilities": {"tasks": ["R2V", "V2V", "MV2V"], "controls": ["Move", "Swap", "Reference", "Expand", "Animate"]},
      "modes": ["reference_to_video", "video_to_video", "edit_video"],
      "best_for": "统一参考生视频、视频重绘与蒙版局部编辑",
      "prompt_grammar": "先声明 R2V/V2V/MV2V 任务与各控制输入，再写最终目标状态、时间范围、被编辑区域和必须保持项。",
      "priorities": ["mask/control video 与文本职责分开", "写最终状态", "记录底模和对应许可证"],
      "controls": ["R2V", "V2V", "masked V2V", "Move/Swap/Expand/Animate"],
      "negative_strategy": "按底模 pipeline 使用；局部编辑主要靠 mask 与 preserve boundary。",
      "gotchas": ["VACE 是控制/编辑体系而非单一最新基础模型", "各变体继承底模许可证"],
      "official_sources": ["https://github.com/ali-vilab/VACE"]
    },
    {
      "id": "framepack-pipeline",
      "vendor": "lllyasviel",
      "model": "FramePack / FramePack-F1/P1 pipeline",
      "status": "active_open_source_pipeline",
      "layer": "open_source_pipeline",
      "adapter_family": "framepack",
      "openness": "open_source_pipeline",
      "last_verified_at": "2026-08-14",
      "release_date": "2025-07-14",
      "license": {"name": "Apache-2.0 code; base model separate", "production_gate": "manual_review", "url": "https://github.com/lllyasviel/FramePack/blob/main/LICENSE"},
      "runtime": {"weights": false, "inference_code": true, "base_model_required": true, "progressive_generation": true},
      "capabilities": {"constant_context_workload": true, "long_video": true, "reference_implementation": "13B-class base model"},
      "modes": ["text_to_video", "image_to_video", "extend_video"],
      "best_for": "消费级显卡上的渐进式长视频与防漂移工作流",
      "prompt_grammar": "全局写人物/世界/风格不变量，分段写当前动作和下一状态；每次续写保留上一段末态并只引入一个变化。",
      "priorities": ["记录确切底模", "保存每段 seed/提示/末帧", "长视频逐段评测"],
      "controls": ["progressive generation", "constant context packing", "F1/P1 variants", "long video"],
      "negative_strategy": "由底模决定。",
      "gotchas": ["FramePack 是推理结构/软件，不是独立基础模型", "网站冒充较多，官方仓库声明 GitHub 是唯一官方网站"],
      "official_sources": ["https://github.com/lllyasviel/FramePack"]
    },
    {
      "id": "meigen-infinitetalk",
      "vendor": "MeiGen-AI",
      "model": "InfiniteTalk",
      "status": "active_open_source",
      "layer": "open_weight_avatar_model",
      "adapter_family": "infinitetalk",
      "openness": "open_source",
      "last_verified_at": "2026-08-14",
      "release_date": "2025-08-20",
      "license": {"name": "Apache-2.0 for repository models; dependencies separate", "production_gate": "manual_review", "url": "https://github.com/MeiGen-AI/InfiniteTalk"},
      "runtime": {"weights": true, "inference_code": true, "inputs": ["image or video", "audio"]},
      "capabilities": {"unlimited_length_claim": true, "image_to_video": true, "video_to_video": true, "video_dubbing": true},
      "modes": ["avatar_presenter", "video_to_video"],
      "best_for": "长时音频驱动口播、单图人物与原视频配音/重配",
      "prompt_grammar": "图像/视频锁定身份、动作基线和构图，音频锁定台词与节奏；文本只写表情、手势、场景和必须保持项。",
      "priorities": ["身份与声音同意记录", "音频先做清理与对齐", "长片分段质检"],
      "controls": ["audio-driven I2V", "audio-driven V2V", "long-form", "video dubbing"],
      "negative_strategy": "主要用保留边界和输入资产控制。",
      "gotchas": ["底层依赖许可证需分别核验", "‘无限长度’是架构/项目表述，生产仍应按段门禁"],
      "official_sources": ["https://github.com/MeiGen-AI/InfiniteTalk"]
    },
    {
      "id": "tencent-hunyuan-video-avatar",
      "vendor": "Tencent Hunyuan",
      "model": "HunyuanVideo-Avatar",
      "status": "stable_open_weight",
      "layer": "open_weight_avatar_model",
      "adapter_family": "hunyuan-avatar",
      "openness": "restricted_open_weight",
      "last_verified_at": "2026-08-14",
      "release_date": "2025-05-28",
      "license": {"name": "Tencent Hunyuan Community License", "production_gate": "manual_review", "url": "https://github.com/Tencent-Hunyuan/HunyuanVideo-Avatar/blob/main/LICENSE"},
      "runtime": {"weights": true, "inference_code": true, "low_vram_community_path": "10GB"},
      "capabilities": {"audio_driven": true, "emotion_reference": true, "multi_character": true},
      "modes": ["avatar_presenter"],
      "best_for": "情绪可控、多人物、分别绑定音频的高动态人物视频",
      "prompt_grammar": "为每个人物绑定图像、音频和可选情绪参考；文本写场景、镜头、动作意图与每个角色的空间/说话归属。",
      "priorities": ["逐角色资产映射", "逐角色同意记录", "情绪参考和台词含义一致"],
      "controls": ["character image injection", "audio emotion module", "face-aware per-character audio", "multi-character dialogue"],
      "negative_strategy": "用逐角色资产与 face-aware mask 控制，不靠长负向提示解决串人。",
      "gotchas": ["社区许可需生产核验", "多人音频绑定错误会造成口型/角色串联"],
      "official_sources": ["https://github.com/Tencent-Hunyuan/HunyuanVideo-Avatar"]
    },
    {
      "id": "character-ai-ovi-1-1",
      "vendor": "Character.AI",
      "model": "Ovi 1.1",
      "status": "active_open_source",
      "layer": "open_weight_audio_video_model",
      "adapter_family": "ovi",
      "openness": "open_source",
      "last_verified_at": "2026-08-14",
      "release_date": "2025-11-10",
      "license": {"name": "Apache-2.0", "production_gate": "standard_review", "url": "https://github.com/character-ai/Ovi/blob/main/LICENSE"},
      "runtime": {"weights": true, "inference_code": true, "parameter_scale": "11B total with 5B audio branch", "quantization": ["FP8", "QInt8"]},
      "capabilities": {"joint_audio_video": true, "duration_seconds": {"values": [5, 10], "max": 10}, "fps": [24], "resolution": ["960x960"], "aspect_ratios": ["9:16", "16:9", "1:1"]},
      "modes": ["text_to_video", "image_to_video"],
      "best_for": "开放权重同步音画 T2AV/I2AV、短对白、拟音和环境声",
      "prompt_grammar": "先写完整画面与动作；逐字对白使用 <S>台词<E>，提示末尾另起 Audio: 描述音效、环境与音乐。",
      "priorities": ["对白标签与说话人明确", "Audio: 放末尾", "保存 5s/10s checkpoint 与量化版本"],
      "controls": ["T2AV", "I2AV", "speech tags", "Audio: caption", "ComfyUI community integration"],
      "negative_strategy": "按推理配置使用；无对白时显式写 Audio: no speech。",
      "gotchas": ["Ovi 1.1 已把旧 <AUDCAP> 语法改为 Audio:", "视频分支基于 Wan2.2，依赖和派生组件许可证仍需分别记录"],
      "official_sources": ["https://github.com/character-ai/Ovi"]
    },
    {
      "id": "kandinskylab-kandinsky-5-video",
      "vendor": "Kandinsky Lab",
      "model": "Kandinsky 5.0 Video Pro / Lite",
      "status": "active_open_source",
      "layer": "open_weight_model",
      "adapter_family": "kandinsky-video",
      "openness": "open_source",
      "last_verified_at": "2026-08-14",
      "release_date": "2025-11-20",
      "license": {"name": "MIT", "production_gate": "standard_review", "url": "https://github.com/kandinskylab/kandinsky-5/blob/main/LICENSE"},
      "runtime": {"weights": true, "inference_code": true, "variants": ["Video Pro 19B", "Video Lite 2B", "SFT", "pretrain", "CFG-distilled", "diffusion-distilled"]},
      "capabilities": {"text_to_video": true, "image_to_video": true, "duration_seconds": {"values": [5, 10], "max": 10}, "languages": ["English", "Russian"], "camera_control_lora": true},
      "modes": ["text_to_video", "image_to_video"],
      "best_for": "英语/俄语 T2V/I2V、19B Pro 高质量、2B Lite 轻量部署和相机控制 LoRA",
      "prompt_grammar": "按主体、动作、环境、摄影机、光线、风格与时间顺序描述；机位控制优先使用官方 camera LoRA 并避免文本冲突。",
      "priorities": ["Pro/Lite 和 5s/10s checkpoint 精确锁定", "记录 attention engine", "相机 LoRA 与 prose 只保留一个主导控制"],
      "controls": ["T2V", "I2V", "5s/10s", "camera LoRAs", "Diffusers", "ComfyUI", "LoRA/full-rank training"],
      "negative_strategy": "使用具体配置的 negative prompt；相机错误优先检查 LoRA 和文本冲突。",
      "gotchas": ["Pro 与 Lite 的能力、显存和速度差异大", "厂商排行属于其评测/声明，不作为本库质量排名"],
      "official_sources": ["https://github.com/kandinskylab/kandinsky-5"]
    },
    {
      "id": "bytedance-bernini",
      "vendor": "ByteDance",
      "model": "Bernini / Bernini-R",
      "status": "active_open_source",
      "layer": "open_weight_video_editing_model",
      "adapter_family": "bernini",
      "openness": "open_source",
      "last_verified_at": "2026-08-14",
      "release_date": "2026-07-13",
      "license": {"name": "Apache-2.0", "production_gate": "standard_review", "url": "https://github.com/bytedance/Bernini/blob/main/LICENSE"},
      "runtime": {"weights": true, "inference_code": true, "training_code": true, "variants": ["Bernini 7+14B", "Bernini v2", "Bernini-R 14B", "Bernini-R 1.3B"]},
      "capabilities": {"semantic_planner": true, "generation": true, "video_editing": true, "prompt_enhancer": true},
      "modes": ["text_to_video", "image_to_video", "reference_to_video", "video_to_video", "edit_video"],
      "best_for": "复杂编辑指令的语义规划、视频生成/重绘、局部编辑和轻量 1.3B 编辑",
      "prompt_grammar": "把编辑对象、区域、时间、最终状态与必须保持项写清；完整 Bernini 让 MLLM 规划语义变化，Bernini-R 适合直接渲染。",
      "priorities": ["先选 full planner 或 renderer-only", "prompt enhancer 前后文本都保存", "guidance_mode 与上传素材槽位一致"],
      "controls": ["semantic planner", "renderer", "V2V", "reference V2V", "prompt enhancer", "1.3B/14B variants"],
      "negative_strategy": "局部编辑以 mask/control 和 preserve boundary 为主，negative 走具体推理配置。",
      "gotchas": ["--use_pe 会通过 OpenAI-compatible endpoint 改写提示，必须捕获 effective prompt", "1.3B 对复杂人物生成弱于大变体"],
      "official_sources": ["https://github.com/bytedance/Bernini"]
    },
    {
      "id": "bytedance-video-as-prompt",
      "vendor": "ByteDance",
      "model": "Video-As-Prompt (VAP)",
      "status": "active_open_source_research",
      "layer": "open_weight_control_model",
      "adapter_family": "video-as-prompt",
      "openness": "open_source",
      "last_verified_at": "2026-08-14",
      "release_date": "2025-10-24",
      "license": {"name": "Apache-2.0", "production_gate": "manual_review", "url": "https://github.com/bytedance/Video-As-Prompt/blob/main/LICENSE"},
      "runtime": {"weights": true, "inference_code": true, "training_code": true, "backbones": ["CogVideoX-I2V-5B", "Wan2.1-I2V-14B"]},
      "capabilities": {"reference_video_as_semantic_prompt": true, "reference_image_identity": true, "semantic_control": ["motion", "depth", "pose", "layout"]},
      "modes": ["reference_to_video", "performance_transfer"],
      "best_for": "用参考视频传递运动/深度/姿态等语义，同时用参考图保持目标身份",
      "prompt_grammar": "明确 reference image 负责谁/什么，reference video 只负责哪种时序语义；文本写目标场景、材质和不得从视频参考继承的内容。",
      "priorities": ["图像身份与视频语义职责解耦", "选择对应 backbone", "参考视频权利与人物同意留档"],
      "controls": ["video semantic prompt", "reference image", "CogVideoX/Wan backbones", "LoRA/full training integrations"],
      "negative_strategy": "用‘只借用’和‘不得继承’边界控制语义污染。",
      "gotchas": ["这是控制模型/研究体系，不是独立通用 T2V", "底模许可证与参考视频权利需另外核验"],
      "official_sources": ["https://github.com/bytedance/Video-As-Prompt"]
    },
    {
      "id": "javisverse-javisdit-plusplus",
      "vendor": "JavisVerse",
      "model": "JavisDiT++ v1.0",
      "status": "active_open_source_research",
      "layer": "open_weight_audio_video_model",
      "adapter_family": "javisdit",
      "openness": "open_source",
      "last_verified_at": "2026-08-14",
      "release_date": "2026-02-26",
      "license": {"name": "Apache-2.0", "production_gate": "manual_review", "url": "https://github.com/JavisVerse/JavisDiT/blob/main/LICENSE"},
      "runtime": {"weights": true, "inference_code": true, "training_code": true, "version": "JavisDiT++ v1.0"},
      "capabilities": {"joint_audio_video": true, "text_conditioning": true, "temporal_audio_video_alignment": true},
      "modes": ["text_to_video"],
      "best_for": "研究级同步音画生成、声音事件与视觉事件的细粒度时间对齐",
      "prompt_grammar": "画面按时间写主体/动作/摄影；声音另写来源、发生时刻、持续时间、空间位置和与动作的同步点。",
      "priorities": ["每个声音绑定可见来源", "音画使用相同时间轴", "保存模型快照与评测配置"],
      "controls": ["joint audio-video diffusion", "TA-RoPE temporal alignment", "AV-DPO", "training/inference"],
      "negative_strategy": "显式写无对白/无音乐等声音边界；视觉 negative 按配置。",
      "gotchas": ["属于研究级模型，生产稳定性和硬件需自行验证", "仓库主线已从 JavisDiT v0.1 升级到 JavisDiT++ v1.0"],
      "official_sources": ["https://github.com/JavisVerse/JavisDiT"]
    },
    {
      "id": "openai-sora-2-legacy",
      "vendor": "OpenAI",
      "model": "Sora 2",
      "status": "legacy_deprecated_api",
      "layer": "legacy_foundation_model",
      "modes": ["text_to_video", "image_to_video", "reference_to_video", "extend_video"],
      "best_for": "仅维护既有集成、学习历史上仍可迁移的导演式提示方法",
      "prompt_grammar": "历史官方方法：风格/基调 → 场景 → 摄影与光线 → 逐拍动作 → 对白 → 背景音；参数与 prose 分工。",
      "priorities": ["把动作按 beat 拆开", "写质量/摩擦/遮挡等物理", "对白和声音单列"],
      "controls": ["legacy video endpoints", "historical prompting cookbook"],
      "negative_strategy": "不建议为新项目设计专用适配；优先迁移到仍受支持的模型。",
      "gotchas": ["官方模型页标 Legacy", "官方 /videos endpoints 标 Deprecated", "不要新建长期生产依赖"],
      "official_sources": ["https://developers.openai.com/api/docs/models/sora-2", "https://developers.openai.com/api/reference/resources/videos", "https://developers.openai.com/cookbook/examples/sora/sora2_prompting_guide"]
    }
  ],
  "templates": [
    {
      "id": "t2v-shot",
      "mode": "text_to_video",
      "template": "[景别/构图]，[主体与稳定身份特征]在[场景]中[主动作]。[物体/环境响应与物理]。摄影机[运镜路径]，[焦距/景深/焦点]。[光线、色板、媒介风格]。[0–Xs 节拍；Xs–Ys 节拍]。[对白/拟音/环境声/音乐]。"
    },
    {
      "id": "i2v-motion",
      "mode": "image_to_video",
      "template": "从输入图开始，[主体动作]，[方向/速度/幅度]；[环境或物体的因果响应]。摄影机[运镜]，[焦点变化]。最后[结束状态与停留]。"
    },
    {
      "id": "keyframe-transition",
      "mode": "first_last_frame",
      "template": "从首帧状态开始：[动作 1]，随后[动作 2]；摄影机沿[路径]连续运动，在[时间]自然减速并稳定抵达尾帧。保持[身份/产品/材质/光色]连续。"
    },
    {
      "id": "reference-video",
      "mode": "reference_to_video",
      "template": "使用@图1仅保持角色身份，@图2仅保持产品外观，@视频1仅参考动作与镜头，@音频1仅参考节奏。生成：[完整镜头目标]。必须保持：[锚点]；不要借用：[排除项]。"
    },
    {
      "id": "v2v-target",
      "mode": "video_to_video",
      "template": "[最终主体外观、材质与场景]，[最终光线、色彩和风格]。保持源视频原有的表演、运动轨迹、节奏、构图和摄影机运动不变。"
    },
    {
      "id": "localized-edit",
      "mode": "edit_video",
      "template": "在[时间范围/区域]，将[对象]改为[明确目标状态]。保持人物身份、动作、遮挡关系、相机、背景、光线和声音的其他部分不变。"
    },
    {
      "id": "multishot-av",
      "mode": "text_to_video",
      "template": "总述：[连续性/风格/声音规则]。Shot 1 [00:00–00:04]：[画面、动作、机位、声音]。Shot 2 [00:04–00:09]：[画面、动作、机位、对白、转场]。Shot 3 [00:09–00:15]：[画面、动作、产品/CTA 收束、尾帧停留]。"
    },
    {
      "id": "avatar-presenter",
      "mode": "avatar_presenter",
      "template": "目标：[受众要理解/采取什么行动]。语气：[角色、情绪、语速]。Scene 1：[短句脚本，标停顿和重音]；视觉：[版式/B-roll]。Scene 2：[…]。发音表：[…]。字幕/品牌/CTA：[…]。"
    },
    {
      "id": "interactive-persona",
      "mode": "interactive_avatar",
      "template": "你是[身份]，目标是[会话目标]。只依据[知识范围]回答；不确定时[回退]。每次最多[长度]，先[询问/确认]再[动作]。不得[边界]；涉及[敏感操作]必须确认；出现[条件]升级人工。"
    }
  ],
  "evaluation": {
    "hard_rejects": [
      "人物或产品身份明显漂移",
      "关键镜头出现严重肢体/物体拓扑错误",
      "产品形态、包装、Logo 或法定文字错误",
      "说话人错位、对白内容错误或关键画音不同步",
      "主动作因果或物理完全失败",
      "出现未授权真人/声音、敏感身份或不合规内容",
      "组织数据或参考资产发生跨 org_id 泄漏"
    ],
    "score_scale": "0–5；硬拒项不能被平均分掩盖",
    "dimensions": [
      "prompt_adherence",
      "temporal_action",
      "spatial_binding",
      "identity_continuity",
      "camera_execution",
      "physics",
      "visual_quality",
      "audio_sync_quality",
      "editorial_utility"
    ],
    "iteration_rule": "固定模型版本、seed/设置和参考；一次只改变一个主要变量。先修动作归属与时序，再修摄影和美学。"
  },
  "failure_playbook": [
    {"symptom": "角色互换动作或说错台词", "likely_cause": "空间/人物绑定不明确", "fix": "为每个动作和台词写角色标签、位置、方向与对象；减少同拍人物数。"},
    {"symptom": "图生视频忽略动作", "likely_cause": "重复描述静态画面挤占注意力", "fix": "删掉图中已有的外观/场景，保留动作、环境响应、运镜和时序。"},
    {"symptom": "首尾帧跳变或融化", "likely_cause": "端点差异太大或缺少可实现路径", "fix": "缩小差异、拆成多段关键帧、描述连续动作链与减速抵达。"},
    {"symptom": "相机和主体一起乱动", "likely_cause": "动作槽与相机槽混写", "fix": "分句写主体动作、环境运动、摄影机运动；先固定机位做诊断。"},
    {"symptom": "提示越长结果越随机", "likely_cause": "并列目标太多或相互矛盾", "fix": "保留一个主动作和一个摄影目标；把身份/风格交给参考，把精确参数交给 UI/API。"},
    {"symptom": "V2V 改掉了原动作", "likely_cause": "写了编辑步骤或重复指挥运动", "fix": "只写最终目标状态，结尾明确保持原运动、节奏、构图和相机。"},
    {"symptom": "人物/产品跨镜头漂移", "likely_cause": "只有形容词，没有连续性资产和锚点", "fix": "建立 continuity bible；使用身份/产品参考；每镜重复少量不可漂移锚点。"},
    {"symptom": "对白有声但错人/错口型", "likely_cause": "未标说话人或同一时间多个声音事件", "fix": "每句独立标说话人、语言、语气和时间；同拍减少发言人。"},
    {"symptom": "Logo/字幕乱码", "likely_cause": "把法定文字交给生成模型", "fix": "生成干净画面，Logo、价格、免责声明和长字幕统一在后期合成。"},
    {"symptom": "复现实验失败", "likely_cause": "只存了 prompt", "fix": "同时保存原始/有效提示、模型/版本、所有设置、参考哈希、工作流与输出 ID。"}
  ],
  "research_benchmarks": [
    {"name": "VBench", "use": "视频生成质量的多维评测框架", "url": "https://github.com/Vchitect/VBench"},
    {"name": "T2V-CompBench", "use": "组合式文本到视频一致性评测", "url": "https://github.com/KaiyueSun98/T2V-CompBench"},
    {"name": "EvalCrafter", "use": "生成视频多维自动评测", "url": "https://github.com/EvalCrafter/EvalCrafter"},
    {"name": "MovieGenBench", "use": "电影级视频生成与编辑评测参考", "url": "https://ai.meta.com/research/movie-gen/"}
  ]
}
