nicolascoutureau/ac
Public
52.9K
runs
Run nicolascoutureau/ac with an API
Use one of our client libraries to get started quickly. Clicking on a library will take you to the Playground tab where you can tweak different inputs, see the results, and copy the corresponding code to use in your own project.
Input schema
The fields you can use to run this model with an API. If you don't give a value for a field its default value will be used.
| Field | Type | Default value | Description |
|---|---|---|---|
| video |
string
|
Input horizontal video to convert to vertical format
|
|
| aspect_ratio |
None
|
9:16
|
Output aspect ratio
|
| speed_preset |
None
|
balanced
|
Processing speed preset. Fast uses smaller YOLO model and lower resolution analysis.
|
| detect_speaker |
boolean
|
True
|
Detect and focus on the active speaker using TalkNet ASD (audio-visual neural network). When multiple people are detected, follows whoever is talking (switching with hysteresis) and shows the two main speakers in split screen during a conversation.
|
| tracking_mode |
None
|
smooth
|
Camera tracking mode. Smooth = cinematic spring-damped movement. Fast = more responsive. Static = fixed per scene.
|
| debug_overlay |
boolean
|
False
|
Draw debug info on video (scene, strategy, speaker, ASD scores, crop).
|
| start |
number
|
Start of the range to process, in seconds (optional). Audio is trimmed too.
|
|
| end |
number
|
End of the range to process, in seconds (optional, default: end of video).
|
|
| focus |
None
|
auto
|
What to frame. auto = active speaker when detected, else the main person, motion when nobody is on screen. speaker = always follow the active speaker (no split screen). action = follow motion, snapping to the nearest person. center = fixed centre crop.
|
| allow_letterbox |
boolean
|
False
|
For scenes without people, show the whole frame over a blurred background instead of cropping on motion / saliency.
|
| min_output_height |
integer
|
0
Max: 4096 |
Minimum output height in pixels. 0 = standard size (1080x1920 for 9:16, 1080x1350 for 4:5, 1080x1080 for 1:1, 1920x1080 for 16:9).
|
| detect_interval |
number
|
0.5
Min: 0.1 Max: 5 |
Seconds between person detections used by the tracking camera.
|
| return_track |
boolean
|
False
|
Also return the crop track as JSON (per keyframe: time, box, strategy, speaker, confidence). When true the output is an object {video, track} instead of a single video URL.
|
{
"type": "object",
"title": "Input",
"required": [
"video"
],
"properties": {
"end": {
"type": "number",
"title": "End",
"minimum": 0,
"x-order": 7,
"nullable": true,
"description": "End of the range to process, in seconds (optional, default: end of video)."
},
"focus": {
"enum": [
"auto",
"speaker",
"action",
"center"
],
"type": "string",
"title": "focus",
"description": "What to frame. auto = active speaker when detected, else the main person, motion when nobody is on screen. speaker = always follow the active speaker (no split screen). action = follow motion, snapping to the nearest person. center = fixed centre crop.",
"default": "auto",
"x-order": 8
},
"start": {
"type": "number",
"title": "Start",
"minimum": 0,
"x-order": 6,
"nullable": true,
"description": "Start of the range to process, in seconds (optional). Audio is trimmed too."
},
"video": {
"type": "string",
"title": "Video",
"format": "uri",
"x-order": 0,
"description": "Input horizontal video to convert to vertical format"
},
"aspect_ratio": {
"enum": [
"9:16",
"4:5",
"1:1",
"16:9"
],
"type": "string",
"title": "aspect_ratio",
"description": "Output aspect ratio",
"default": "9:16",
"x-order": 1
},
"return_track": {
"type": "boolean",
"title": "Return Track",
"default": false,
"x-order": 12,
"description": "Also return the crop track as JSON (per keyframe: time, box, strategy, speaker, confidence). When true the output is an object {video, track} instead of a single video URL."
},
"speed_preset": {
"enum": [
"quality",
"balanced",
"fast"
],
"type": "string",
"title": "speed_preset",
"description": "Processing speed preset. Fast uses smaller YOLO model and lower resolution analysis.",
"default": "balanced",
"x-order": 2
},
"debug_overlay": {
"type": "boolean",
"title": "Debug Overlay",
"default": false,
"x-order": 5,
"description": "Draw debug info on video (scene, strategy, speaker, ASD scores, crop)."
},
"tracking_mode": {
"enum": [
"smooth",
"static",
"fast"
],
"type": "string",
"title": "tracking_mode",
"description": "Camera tracking mode. Smooth = cinematic spring-damped movement. Fast = more responsive. Static = fixed per scene.",
"default": "smooth",
"x-order": 4
},
"detect_speaker": {
"type": "boolean",
"title": "Detect Speaker",
"default": true,
"x-order": 3,
"description": "Detect and focus on the active speaker using TalkNet ASD (audio-visual neural network). When multiple people are detected, follows whoever is talking (switching with hysteresis) and shows the two main speakers in split screen during a conversation."
},
"allow_letterbox": {
"type": "boolean",
"title": "Allow Letterbox",
"default": false,
"x-order": 9,
"description": "For scenes without people, show the whole frame over a blurred background instead of cropping on motion / saliency."
},
"detect_interval": {
"type": "number",
"title": "Detect Interval",
"default": 0.5,
"maximum": 5,
"minimum": 0.1,
"x-order": 11,
"description": "Seconds between person detections used by the tracking camera."
},
"min_output_height": {
"type": "integer",
"title": "Min Output Height",
"default": 0,
"maximum": 4096,
"minimum": 0,
"x-order": 10,
"description": "Minimum output height in pixels. 0 = standard size (1080x1920 for 9:16, 1080x1350 for 4:5, 1080x1080 for 1:1, 1920x1080 for 16:9)."
}
}
}
Output schema
The shape of the response you’ll get when you run this model with an API.
Schema
{
"title": "Output"
}