{
  "title": "AI, Behind the Shot",
  "reviewed": "2026-10-07",
  "kind": "documented-tool-capabilities-and-editorial-workflows",
  "method": "Primary provider documentation. Editorial fits are not performance rankings. Demos are provider-selected, not our tests. Credits are provider-specific.",
  "tools": [
    {
      "id": "veo",
      "name": "Veo 3.1",
      "provider": "Google",
      "label": "Frames + sound",
      "summary": "A useful route when a shot needs a defined first/last frame, native audio, or extension of a Veo-generated clip.",
      "uses": [
        "product",
        "story"
      ],
      "specs": [
        {
          "label": "Input",
          "value": "Text, starting image; up to 3 reference images (standard/Fast)"
        },
        {
          "label": "Length",
          "value": "4, 6 or 8 seconds; references and higher resolutions require 8s"
        },
        {
          "label": "Output",
          "value": "720p, 1080p, 4K; extension is 720p"
        },
        {
          "label": "Audio",
          "value": "Native audio, always on"
        },
        {
          "label": "Controls",
          "value": "First/last frame; extension on standard/Fast, not Lite"
        }
      ],
      "pros": [
        "Frames help specify where a shot starts and ends.",
        "Audio is generated alongside the picture.",
        "Fast and Lite offer different cost/resolution tradeoffs."
      ],
      "cons": [
        "Reference, duration and resolution combinations have constraints.",
        "Extension accepts supported Veo outputs; it is not a general footage editor.",
        "Preview-model access and regional person-generation rules can change."
      ],
      "workflow": [
        "Approve a product or character image before animating.",
        "Request one camera move and explicit sound direction.",
        "Check the first, middle and last frame; extend only a usable take."
      ],
      "fit": "For our miniature-car shot, start with image-to-video and a single dolly-in. Check body geometry before trying dialogue.",
      "price": "Gemini API: standard $0.40/s at 720p/1080p; Fast $0.10/s at 720p; Lite $0.05/s at 720p. 4K and other resolutions differ.",
      "sources": [
        "veo",
        "google-price"
      ],
      "url": "https://deepmind.google/models/veo/"
    },
    {
      "id": "omni",
      "name": "Gemini Omni Flash",
      "provider": "Google",
      "label": "Conversational edits",
      "summary": "Google now recommends Omni Flash as its default video-generation route. Its distinctive workflow is refining a clip through follow-up instructions.",
      "uses": [
        "edit",
        "social"
      ],
      "specs": [
        {
          "label": "Workflow",
          "value": "Video generation and multi-turn editing through the Interactions API"
        },
        {
          "label": "Output",
          "value": "360p or 720p; 1080p and 4K are upscaled"
        },
        {
          "label": "References",
          "value": "API video references: up to 3 clips, 3s each; audio ignored"
        },
        {
          "label": "Editing limits",
          "value": "Uploaded footage ≤10s; upload editing unavailable in EEA, Switzerland and UK"
        },
        {
          "label": "Audio",
          "value": "Generated audio; uploaded audio references and voice editing unsupported in this API"
        }
      ],
      "pros": [
        "Follow-up edits preserve the conversational clip context.",
        "Can edit a generated result without starting a new prompt from scratch.",
        "Useful to explore changes in look or a specific visual element."
      ],
      "cons": [
        "The general model overview is broader than the current API supports.",
        "Upload-region and footage-length limits matter for real projects.",
        "Ask explicitly for a continuous shot; the default can include multiple shots."
      ],
      "workflow": [
        "Generate a short, simple scene with no cuts.",
        "Change one thing in a follow-up and ask to keep everything else the same.",
        "Check that the untouched product, background and timing actually remain stable."
      ],
      "fit": "Choose this when iteration on an existing generated shot matters more than a precise first/last-frame pipeline.",
      "price": "Check current Gemini API pricing for this model. Veo rates do not apply to Omni Flash.",
      "sources": [
        "google-index",
        "omni"
      ],
      "url": "https://ai.google.dev/gemini-api/docs/omni"
    },
    {
      "id": "runway",
      "name": "Gen-4.5",
      "provider": "Runway",
      "label": "Shot-by-shot direction",
      "summary": "A short-shot model within a broader filmmaking platform. Start with an image when composition is already approved, then describe the motion.",
      "uses": [
        "product",
        "social"
      ],
      "specs": [
        {
          "label": "Input",
          "value": "Text-to-video or image-to-video"
        },
        {
          "label": "Length",
          "value": "2–10 seconds"
        },
        {
          "label": "Output",
          "value": "720p, 24 or 25fps"
        },
        {
          "label": "Framing",
          "value": "Text: 16:9; image: 16:9, 9:16, 1:1, 4:3, 3:4 or 21:9"
        },
        {
          "label": "Access",
          "value": "Standard plan or higher; 12 credits per generated second"
        }
      ],
      "pros": [
        "Variable short durations suit a shot-based workflow.",
        "Image-to-video supports several useful delivery shapes.",
        "Documented motion-focused prompting is a clear starting point."
      ],
      "cons": [
        "Native model output is 720p.",
        "Other Runway models and tools have different capabilities and costs.",
        "Premium ProRes/PNG export adds credits and has plan restrictions."
      ],
      "workflow": [
        "Create an approved first frame with the intended composition.",
        "Describe camera and subject motion rather than re-describing every pixel.",
        "Choose the shortest take that covers the action, then finish in your timeline."
      ],
      "fit": "A practical starting point for controlled visual inserts and product motion. This is an editorial fit, not a quality ranking.",
      "price": "12 Runway credits/s. ProRes/PNG sequence generation adds 5 credits/s on supported premium plans. Credits are not dollars or Pika credits.",
      "sources": [
        "runway",
        "runway-prompt"
      ],
      "url": "https://runwayml.com/"
    },
    {
      "id": "kling",
      "name": "Kling 3.0 / Omni",
      "provider": "Kuaishou",
      "label": "Multi-shot + voices",
      "summary": "Video 3.0 and 3.0 Omni bring shot-based storytelling and native audio into the same family. Omni adds richer reference and storyboard controls.",
      "uses": [
        "story",
        "product"
      ],
      "specs": [
        {
          "label": "Length",
          "value": "Up to 15 seconds"
        },
        {
          "label": "Audio",
          "value": "Native audio; announcement names English, Chinese, Japanese, Korean and Spanish"
        },
        {
          "label": "References",
          "value": "Image/video elements; Omni can extract character appearance and voice from video"
        },
        {
          "label": "Storyboarding",
          "value": "Omni: per-shot duration, shot size, perspective and camera movement"
        },
        {
          "label": "Pricing / resolution",
          "value": "Not verified from a current public primary specification; check the selected host"
        }
      ],
      "pros": [
        "Per-shot direction is useful when the order of scenes matters.",
        "Character appearance and voice can be referenced together in Omni.",
        "Multi-language audio opens more options than a silent clip workflow."
      ],
      "cons": [
        "Video 3.0 and Omni are different model variants.",
        "Launch-day access is not evidence of your current plan entitlement.",
        "Text and identity consistency are provider claims to test on your own content."
      ],
      "workflow": [
        "Prepare a character reference with the correct voice and rights.",
        "Write a small sequence with one clear action in each shot.",
        "Check speaker attribution, continuity and exact text before accepting the take."
      ],
      "fit": "Consider Omni for a short character-led sequence with explicit cuts and voices. Keep the initial storyboard simple.",
      "price": "Check the current Kling plan and exact model. We do not assign an unverified dollar-per-second rate.",
      "sources": [
        "kling"
      ],
      "url": "https://klingai.com/"
    },
    {
      "id": "luma",
      "name": "Ray3.2",
      "provider": "Luma",
      "label": "Keyframes + finishing",
      "summary": "A reference-heavy workflow for shaping motion and finishing the image, including HDR and EXR output options.",
      "uses": [
        "edit",
        "finish",
        "product"
      ],
      "specs": [
        {
          "label": "Length / output",
          "value": "Up to 20 seconds at 1080p, per the release announcement"
        },
        {
          "label": "Keyframes",
          "value": "Up to 16 keyframes"
        },
        {
          "label": "Performance",
          "value": "Skeletal/gesture tracking and facial tracking for up to 8 faces"
        },
        {
          "label": "Finishing",
          "value": "Native HDR and 16-bit EXR; reframe and background replacement"
        },
        {
          "label": "Interface",
          "value": "Ray3.2 API and application workflows; select the model explicitly"
        }
      ],
      "pros": [
        "Multiple keyframes let you describe a trajectory across the shot.",
        "Performance controls are relevant to modifying existing footage.",
        "HDR/EXR support matters when the finishing pipeline needs it."
      ],
      "cons": [
        "A more complex pipeline than making a disposable social clip.",
        "Resolution, length and output format materially change the credit bill.",
        "Ray3.14 restrictions should not be assumed to describe Ray3.2."
      ],
      "workflow": [
        "Pick the finishing format before generating variations.",
        "Use keyframes to define the meaningful moments of the shot.",
        "Check motion between frames and review the result in the intended color pipeline."
      ],
      "fit": "Start here when the problem is shaping or finishing a shot with references, rather than finding a broad visual idea.",
      "price": "Ray3.2 720p SDR T2V/I2V: 100 credits/5s or 300/10s. HDR costs 2× and EXR 3×. Video editing and reframe use different tables.",
      "sources": [
        "luma",
        "luma-price"
      ],
      "url": "https://lumalabs.ai/"
    },
    {
      "id": "firefly",
      "name": "Firefly Video",
      "provider": "Adobe",
      "label": "Creative Cloud workflow",
      "summary": "Native Firefly video generation can sit close to an existing Adobe editing workflow, with frame and composition references.",
      "uses": [
        "finish",
        "product"
      ],
      "specs": [
        {
          "label": "Native duration",
          "value": "5 seconds at 24fps in the documented editor workflow"
        },
        {
          "label": "Keyframes",
          "value": "First and last images"
        },
        {
          "label": "Reference",
          "value": "Composition guided by existing video"
        },
        {
          "label": "4K",
          "value": "Available through a separate upscaling workflow; not a native-output claim here"
        },
        {
          "label": "Model choice",
          "value": "Firefly and partner models have separate features, credits and terms"
        }
      ],
      "pros": [
        "Useful when the team already edits and delivers in Adobe tools.",
        "Reference composition and frames offer concrete direction.",
        "Adobe documents native Firefly training and commercial positioning."
      ],
      "cons": [
        "Five-second native clips require a shot-and-edit approach.",
        "A partner model inside Firefly is not the native Firefly model.",
        "Commercial positioning is not a guarantee that every output clears third-party rights."
      ],
      "workflow": [
        "Bring an approved frame or composition reference.",
        "Generate a short insert using the selected native model.",
        "Edit, add exact titles and audio, then upscale only if delivery requires it."
      ],
      "fit": "Consider native Firefly for short inserts in an Adobe finishing workflow and evaluate the exact model terms for a client project.",
      "price": "Check current Firefly plan and generative credit requirements. Partner generation is billed differently.",
      "sources": [
        "adobe",
        "adobe-ref",
        "adobe-upscale",
        "adobe-training"
      ],
      "url": "https://firefly.adobe.com/"
    },
    {
      "id": "pika",
      "name": "Pika Create",
      "provider": "Pika",
      "label": "A multi-model studio",
      "summary": "The September 2026 app is a creative workspace with multiple underlying video models, character/product tools and sound tools. It is not a single video model.",
      "uses": [
        "social",
        "product"
      ],
      "specs": [
        {
          "label": "Platform",
          "value": "Video, character and product studios with a multi-model catalog"
        },
        {
          "label": "Longer sequences",
          "value": "App advertises up to 30s multi-shot video; model-dependent"
        },
        {
          "label": "Models",
          "value": "Catalog includes Seedance 2.5, MiniMax H3 and Pika 2.5 among others"
        },
        {
          "label": "Commercial license",
          "value": "Current pricing: Creator/Fancy yes; Free/Starter no"
        },
        {
          "label": "Credits",
          "value": "New app credits differ from legacy app, API and iOS credits"
        }
      ],
      "pros": [
        "One workspace for trying different models and creative tasks.",
        "Character/product studios can shorten a social-content workflow.",
        "Sound tools sit alongside picture tools."
      ],
      "cons": [
        "A Pika result can be generated by another provider’s model.",
        "An unwatermarked download does not automatically include commercial rights.",
        "Different models consume different credits; old pricing is not current-app pricing."
      ],
      "workflow": [
        "Choose the underlying model, not just the Pika app.",
        "Prepare the character or product reference, then make a short test.",
        "Check license tier and credit usage before a batch or client delivery."
      ],
      "fit": "A useful entry point for everyday social creation when you want several tools in one interface.",
      "price": "Monthly pricing: Starter $10/900 credits; Creator $35/3,150. Seedance 2.5 720p/5s example: 122 credits. Check current terms.",
      "sources": [
        "pika",
        "pika-features",
        "pika-price"
      ],
      "url": "https://pika.art/"
    },
    {
      "id": "seedance",
      "name": "Seedance 2.5",
      "provider": "ByteDance",
      "label": "Longer reference-led stories",
      "summary": "An audio-video model aimed at longer sequences and rich multimodal direction, with timestamp-level editing in the provider’s release.",
      "uses": [
        "story",
        "edit"
      ],
      "specs": [
        {
          "label": "Length",
          "value": "Up to 30 seconds per generation; multiple extension rounds"
        },
        {
          "label": "References",
          "value": "Provider model: up to 30 images, 10 videos and 10 audio clips"
        },
        {
          "label": "Editing",
          "value": "Timestamp-level audio/video edits; perspective and reference-based editing"
        },
        {
          "label": "Access",
          "value": "Release names Jimeng and Doubao; also listed in Pika’s model catalog"
        },
        {
          "label": "Host caveat",
          "value": "An app may expose fewer controls, lengths or references than the model announcement"
        }
      ],
      "pros": [
        "Longer individual takes can cover more of a small story.",
        "Rich reference packs support multiple subjects and scene directions.",
        "Timestamp-based edits are relevant when one part of a take needs changing."
      ],
      "cons": [
        "A model-level reference ceiling is not every host app’s upload limit.",
        "Longer sequences need careful continuity and audio review.",
        "Availability, region and cost depend on the access platform."
      ],
      "workflow": [
        "Build a small consistent reference pack rather than adding every image.",
        "Write timed beats with explicit cut and sound intentions.",
        "Review the entire sequence, then edit the problematic moment if supported by the host."
      ],
      "fit": "Consider it for a sequence whose references and timeline matter more than one spectacular isolated shot.",
      "price": "Host-dependent. Pika lists 122 credits for a 5s 720p generation; that does not price a 30s run or the native platform.",
      "sources": [
        "seedance",
        "pika-price"
      ],
      "url": "https://seed.bytedance.com/seedance2_5"
    },
    {
      "id": "minimax",
      "name": "MiniMax H3",
      "provider": "MiniMax / Hailuo",
      "label": "Multimodal + stereo",
      "summary": "A multimodal model that can combine visual and sound references with generation and video editing.",
      "uses": [
        "product",
        "edit",
        "story"
      ],
      "specs": [
        {
          "label": "Length / output",
          "value": "Up to 15 seconds at 2K"
        },
        {
          "label": "Audio",
          "value": "Native stereo sound"
        },
        {
          "label": "Input",
          "value": "Text, images, video and audio context"
        },
        {
          "label": "Editing",
          "value": "Video-to-video motion transfer and natural-language edits"
        },
        {
          "label": "Access",
          "value": "Hailuo and supported hosts; exact exposed controls depend on the interface"
        }
      ],
      "pros": [
        "Picture and stereo sound can be generated together.",
        "Motion-reference workflows offer a starting point for specific camera ideas.",
        "Video editing can target changes to an existing scene."
      ],
      "cons": [
        "Provider claims about brand/text fidelity still need exact-output checks.",
        "A Hailuo 2.3 spec or price does not describe H3.",
        "Published comparisons against competitors are provider-reported, not our benchmark."
      ],
      "workflow": [
        "Define what each reference contributes: subject, movement or sound.",
        "Start with a short product shot and one motion reference.",
        "Check readable branding, geometry and sound placement in the actual export."
      ],
      "fit": "Consider H3 when a product or motion-reference shot also needs integrated sound.",
      "price": "Check H3 pricing in the selected interface. We do not infer a rate from older Hailuo models or provider relative-price claims.",
      "sources": [
        "minimax"
      ],
      "url": "https://hailuoai.video/"
    }
  ],
  "budgetConfigurations": [
    {
      "id": "veo-standard",
      "name": "Veo 3.1 · 720p standard",
      "unit": "USD",
      "rate": 0.4,
      "seconds": 8,
      "source": "google-price"
    },
    {
      "id": "veo-fast",
      "name": "Veo 3.1 Fast · 720p",
      "unit": "USD",
      "rate": 0.1,
      "seconds": 8,
      "source": "google-price"
    },
    {
      "id": "veo-lite",
      "name": "Veo 3.1 Lite · 720p",
      "unit": "USD",
      "rate": 0.05,
      "seconds": 8,
      "source": "google-price"
    },
    {
      "id": "runway",
      "name": "Gen-4.5 · 720p",
      "unit": "Runway credits",
      "rate": 12,
      "seconds": 5,
      "source": "runway"
    },
    {
      "id": "pika-seedance",
      "name": "Seedance 2.5 in Pika · 720p/5s",
      "unit": "Pika credits",
      "rate": 24.4,
      "seconds": 5,
      "source": "pika-price"
    },
    {
      "id": "luma-720",
      "name": "Ray3.2 · 720p SDR/5s",
      "unit": "Luma credits",
      "rate": 20,
      "seconds": 5,
      "source": "luma-price"
    }
  ],
  "demos": [
    {
      "id": "minimax",
      "name": "H3 multimodal reference example",
      "provider": "MiniMax",
      "length": "7s example",
      "context": "Generated output from the multimodal reference example; input clips are separate on the source page.",
      "src": "https://filecdn.minimax.chat/public/h3-en-v2-video-003-1785473642166.mp4",
      "source": "minimax",
      "watch": "This is the generated output in the provider’s multimodal context example. Open the source to see the separate movement, image and audio references."
    },
    {
      "id": "luma",
      "name": "Ray3.2 release reel",
      "provider": "Luma",
      "length": "1m 25s release reel",
      "context": "A provider-edited reel demonstrating several workflows, not one unedited model generation.",
      "src": "https://static.cdn-luma.com/files/sanity/67573787-9db2-474a-a54d-526a0cc58873.mp4",
      "source": "luma",
      "watch": "Watch how the provider presents its keyframe and finishing workflow. Look for what is an input, an edit, or an output."
    },
    {
      "id": "seedance",
      "name": "Seedance 2.5 release film",
      "provider": "ByteDance",
      "length": "4m 22s provider film",
      "context": "A longer creative film showcasing the workflow. This is not a single 30-second generation.",
      "src": "https://lf3-static.bytednsdoc.com/obj/eden-cn/lapzild-tss/ljhwZthlaukjlkulzlp/user-upload/4xfa4ms8eqq9e.mp4",
      "source": "seedance",
      "watch": "Follow the shot changes, subject identity and sound across the sequence. Longer output makes continuity more important."
    }
  ]
}