bghira/minimax-h3-lora-test
Public
198
runs
Run bghira/minimax-h3-lora-test with an API
Use one of our client libraries to get started quickly. Clicking on a library will take you to the Playground tab where you can tweak different inputs, see the results, and copy the corresponding code to use in your own project.
Input schema
The fields you can use to run this model with an API. If you don't give a value for a field its default value will be used.
| Field | Type | Default value | Description |
|---|---|---|---|
| prompt |
string
|
Text prompt for MiniMax H3 video and audio generation
|
|
| acceleration_model |
None
|
Custom / base model
|
Base model or a pinned H3 Acceleration Arena preset
|
| mode |
None
|
Auto
|
Task family. Auto uses Ref2VA when reference inputs are provided, otherwise FL2VA/T2VA.
|
| cache_mode |
None
|
Fast
|
Transformer block-cache strength
|
| image |
string
|
Optional first keyframe for FL2VA generation
|
|
| last_image |
string
|
Optional last keyframe for FL2VA generation
|
|
| reference_image |
string
|
Optional Ref2VA image reference
|
|
| reference_images |
array
|
Optional ordered Ref2VA image references. Use this for multiple image uploads.
|
|
| reference_video |
string
|
Optional Ref2VA video reference. A soundtrack in the file is used as that reference's audio.
|
|
| reference_audio |
string
|
Optional Ref2VA audio reference. Must be paired with an image or video reference.
|
|
| reference_manifest |
string
|
|
Optional JSON list of Ref2VA references for URL/path inputs. Entries may be strings or objects with image, video, audio, fps, and sample_rate fields.
|
| reference_order |
string
|
image,video,audio
|
Order for the simple Ref2VA reference inputs
|
| lora |
string
|
Optional LoRA path, URL, Hugging Face repo id, comma-separated list, or JSON list
|
|
| lora_scale |
number
|
1
Min: -4 Max: 4 |
Default LoRA scale
|
| lora_scales |
string
|
Optional comma-separated or JSON list of scales matching lora
|
|
| aspect_ratio |
None
|
16:9
|
Output aspect ratio
|
| resolution |
None
|
544p
|
Output resolution tier
|
| num_frames |
integer
|
124
Min: 120 Max: 345 |
Frame count. Rounded up to MiniMax H3's 17n+5 frame grid.
|
| num_inference_steps |
integer
|
0
Max: 80 |
Actual transformer evaluations. Use 0 for the base model default of 20.
|
| seed |
integer
|
0
|
Random seed. Use 0 for a random seed.
|
| save_compile_cache |
boolean
|
False
|
Write torch compiler cache artifacts after this run if available
|
{
"type": "object",
"title": "Input",
"required": [
"prompt"
],
"properties": {
"lora": {
"type": "string",
"title": "Lora",
"x-order": 12,
"nullable": true,
"description": "Optional LoRA path, URL, Hugging Face repo id, comma-separated list, or JSON list"
},
"mode": {
"enum": [
"Auto",
"Text / keyframes (FL2VA)",
"Omni references (Ref2VA)"
],
"type": "string",
"title": "mode",
"description": "Task family. Auto uses Ref2VA when reference inputs are provided, otherwise FL2VA/T2VA.",
"default": "Auto",
"x-order": 2
},
"seed": {
"type": "integer",
"title": "Seed",
"default": 0,
"minimum": 0,
"x-order": 19,
"description": "Random seed. Use 0 for a random seed."
},
"image": {
"type": "string",
"title": "Image",
"format": "uri",
"x-order": 4,
"nullable": true,
"description": "Optional first keyframe for FL2VA generation"
},
"prompt": {
"type": "string",
"title": "Prompt",
"x-order": 0,
"description": "Text prompt for MiniMax H3 video and audio generation"
},
"cache_mode": {
"enum": [
"Off",
"Safe",
"Fast",
"Aggressive"
],
"type": "string",
"title": "cache_mode",
"description": "Transformer block-cache strength",
"default": "Fast",
"x-order": 3
},
"last_image": {
"type": "string",
"title": "Last Image",
"format": "uri",
"x-order": 5,
"nullable": true,
"description": "Optional last keyframe for FL2VA generation"
},
"lora_scale": {
"type": "number",
"title": "Lora Scale",
"default": 1,
"maximum": 4,
"minimum": -4,
"x-order": 13,
"description": "Default LoRA scale"
},
"num_frames": {
"type": "integer",
"title": "Num Frames",
"default": 124,
"maximum": 345,
"minimum": 120,
"x-order": 17,
"description": "Frame count. Rounded up to MiniMax H3's 17n+5 frame grid."
},
"resolution": {
"enum": [
"256p",
"544p",
"768p"
],
"type": "string",
"title": "resolution",
"description": "Output resolution tier",
"default": "544p",
"x-order": 16
},
"lora_scales": {
"type": "string",
"title": "Lora Scales",
"x-order": 14,
"nullable": true,
"description": "Optional comma-separated or JSON list of scales matching lora"
},
"aspect_ratio": {
"enum": [
"21:9",
"16:9",
"4:3",
"1:1",
"3:4",
"9:16",
"9:21"
],
"type": "string",
"title": "aspect_ratio",
"description": "Output aspect ratio",
"default": "16:9",
"x-order": 15
},
"reference_audio": {
"type": "string",
"title": "Reference Audio",
"format": "uri",
"x-order": 9,
"nullable": true,
"description": "Optional Ref2VA audio reference. Must be paired with an image or video reference."
},
"reference_image": {
"type": "string",
"title": "Reference Image",
"format": "uri",
"x-order": 6,
"nullable": true,
"description": "Optional Ref2VA image reference"
},
"reference_order": {
"type": "string",
"title": "Reference Order",
"default": "image,video,audio",
"x-order": 11,
"description": "Order for the simple Ref2VA reference inputs"
},
"reference_video": {
"type": "string",
"title": "Reference Video",
"format": "uri",
"x-order": 8,
"nullable": true,
"description": "Optional Ref2VA video reference. A soundtrack in the file is used as that reference's audio."
},
"reference_images": {
"type": "array",
"items": {
"type": "string",
"format": "uri"
},
"title": "Reference Images",
"x-order": 7,
"nullable": true,
"description": "Optional ordered Ref2VA image references. Use this for multiple image uploads."
},
"acceleration_model": {
"enum": [
"Custom / base model",
"Alibaba PDD Acc - FL2VA (8 NFE)",
"MiniMax-H3 - arena baseline (27 NFE)",
"FastVideo FastH3 Preview v0.2 - VSA (4 NFE)",
"FastVideo FastH3 v1 VSA-DataFree full checkpoint (4 NFE)",
"FastVideo FastH3 v1 VSA-DataFree full checkpoint (6 NFE)",
"FastVideo FastH3 v1 VSA-DataFree LoRA (4 NFE)",
"FlashGen v1.0 768p (4 NFE)",
"joyfox Turbo FL2VA pruned (4 NFE)",
"larryvrh Turbo EMA checkpoint 850 (4 NFE)",
"larryvrh Turbo v4 step 600 EMA (6 NFE)",
"LightX2V Turbo v0.1 (4 NFE)",
"LightX2V Turbo v1.0 768p (4 NFE)",
"LightX2V Turbo v1.1 768p (4 NFE)",
"LightX2V Turbo v1.0 768p (8 NFE)",
"LightX2V Turbo v1.0 native resolution (8 NFE)",
"Plaguekind Parasyte Turbo (6 NFE)",
"RAVEN Streaming Preview (4 NFE)",
"silveroxides 4-to-8-step DARE merge (6 NFE)",
"silveroxides DARE-TIES merge (6 NFE)",
"silveroxides LightX2V v0.1 resize (4 NFE)",
"Tutu AudioVideo 20-to-8 checkpoint 100 (8 NFE)",
"Tutu AudioVideo 20-to-8 checkpoint 300 (8 NFE)",
"Alibaba PDD Acc - Ref2VA (8 NFE)",
"LightX2V Ref2V Turbo v0.1 (4 NFE)"
],
"type": "string",
"title": "acceleration_model",
"description": "Base model or a pinned H3 Acceleration Arena preset",
"default": "Custom / base model",
"x-order": 1
},
"reference_manifest": {
"type": "string",
"title": "Reference Manifest",
"default": "",
"x-order": 10,
"description": "Optional JSON list of Ref2VA references for URL/path inputs. Entries may be strings or objects with image, video, audio, fps, and sample_rate fields."
},
"save_compile_cache": {
"type": "boolean",
"title": "Save Compile Cache",
"default": false,
"x-order": 20,
"description": "Write torch compiler cache artifacts after this run if available"
},
"num_inference_steps": {
"type": "integer",
"title": "Num Inference Steps",
"default": 0,
"maximum": 80,
"minimum": 0,
"x-order": 18,
"description": "Actual transformer evaluations. Use 0 for the base model default of 20."
}
}
}
Output schema
The shape of the response you’ll get when you run this model with an API.
Schema
{
"type": "string",
"title": "Output",
"format": "uri"
}