From fad90016d92b912dac19aefe050b4b5e1195c223 Mon Sep 17 00:00:00 2001 From: Terence Cho Date: Wed, 30 Sep 2026 11:07:34 -0700 Subject: [PATCH] codegen: expose instant video creation and status commands --- cmd/heygen/instant_video_test.go | 80 +++++++++++++++++++++++++++ codegen/examples/folder.yaml | 6 +++ codegen/examples/model.yaml | 10 ++++ gen/folder.go | 87 ++++++++++++++++++++++++++++++ gen/lipsync.go | 6 +-- gen/model.go | 93 ++++++++++++++++++++++++++++++-- gen/registry.go | 7 +++ gen/template.go | 6 +-- gen/video-agent.go | 48 ++++++++++++----- gen/video-translate.go | 10 ++-- gen/video.go | 8 +-- gen/voice.go | 28 +++++++--- 12 files changed, 349 insertions(+), 40 deletions(-) create mode 100644 cmd/heygen/instant_video_test.go create mode 100644 codegen/examples/folder.yaml create mode 100644 gen/folder.go diff --git a/cmd/heygen/instant_video_test.go b/cmd/heygen/instant_video_test.go new file mode 100644 index 0000000..a700996 --- /dev/null +++ b/cmd/heygen/instant_video_test.go @@ -0,0 +1,80 @@ +package main + +import ( + "encoding/json" + "net/http" + "strings" + "testing" +) + +func TestInstantVideoCreateModes(t *testing.T) { + for _, mode := range []string{"text_to_video", "image_to_video", "reference_to_video"} { + t.Run(mode, func(t *testing.T) { + body := map[string]any{"model": "heygen-video-1", "mode": mode, "prompt": "A person waves", "duration": 5} + if mode == "image_to_video" { + body["image"] = map[string]string{"type": "asset_id", "asset_id": "asset-1"} + } + if mode == "reference_to_video" { + body["reference_images"] = []map[string]string{{"type": "asset_id", "asset_id": "asset-1"}} + } + encoded, err := json.Marshal(body) + if err != nil { + t.Fatal(err) + } + requests := 0 + server := setupTestServer(t, map[string]testHandler{ + "POST /v3/models/videos": { + StatusCode: http.StatusAccepted, + Body: `{"data":{"video_id":"video-1","status":"pending"}}`, + ValidateRequest: func(t *testing.T, r *http.Request) { + requests++ + if r.Header.Get("X-Api-Key") != "test-key" { + t.Error("missing API key") + } + if r.Header.Get("Idempotency-Key") != "generation-1" { + t.Error("missing idempotency key") + } + var got map[string]any + if err := json.NewDecoder(r.Body).Decode(&got); err != nil { + t.Fatal(err) + } + if got["mode"] != mode || got["model"] != "heygen-video-1" || got["prompt"] != "A person waves" { + t.Fatalf("unexpected request: %#v", got) + } + if mode == "image_to_video" && got["image"] == nil { + t.Error("first frame missing") + } + if mode == "reference_to_video" && got["reference_images"] == nil { + t.Error("references missing") + } + }, + }, + }) + defer server.Close() + result := runCommand(t, server.URL, "test-key", "model", "videos", "create", "--idempotency-key", "generation-1", "-d", string(encoded)) + if result.ExitCode != 0 || result.Stderr != "" { + t.Fatalf("exit=%d stderr=%s", result.ExitCode, result.Stderr) + } + if requests != 1 || !strings.Contains(result.Stdout, `"video_id":"video-1"`) { + t.Fatalf("requests=%d stdout=%s", requests, result.Stdout) + } + }) + } +} + +func TestInstantVideoGet(t *testing.T) { + server := setupTestServer(t, map[string]testHandler{ + "GET /v3/models/videos/video-1": { + StatusCode: http.StatusOK, + Body: `{"data":{"video_id":"video-1","status":"completed","video_url":"https://example.com/result.mp4"}}`, + }, + }) + defer server.Close() + result := runCommand(t, server.URL, "test-key", "model", "videos", "get", "video-1") + if result.ExitCode != 0 || result.Stderr != "" { + t.Fatalf("exit=%d stderr=%s", result.ExitCode, result.Stderr) + } + if !strings.Contains(result.Stdout, "https://example.com/result.mp4") { + t.Fatalf("missing result URL: %s", result.Stdout) + } +} diff --git a/codegen/examples/folder.yaml b/codegen/examples/folder.yaml new file mode 100644 index 0000000..16dae68 --- /dev/null +++ b/codegen/examples/folder.yaml @@ -0,0 +1,6 @@ +"POST /v3/folders": + - desc: "Create a folder in the workspace" + cmd: "heygen folder create --name 'Campaign'" +"GET /v3/folders/{folder_id}": + - desc: "Get a folder by its id" + cmd: "heygen folder get " diff --git a/codegen/examples/model.yaml b/codegen/examples/model.yaml index ab35cc5..780771c 100644 --- a/codegen/examples/model.yaml +++ b/codegen/examples/model.yaml @@ -17,3 +17,13 @@ "DELETE /v3/models/audio/voices/{voice_id}": - desc: "Delete a voice (rejected while its status is PENDING)" cmd: "heygen model audio voices delete " +"POST /v3/models/videos": + - desc: "Generate an Instant Video from a text prompt" + cmd: "heygen model videos create -d '{\"model\":\"heygen-video-1\",\"mode\":\"text_to_video\",\"prompt\":\"A person walking through a garden\",\"duration\":5,\"resolution\":\"768p\"}'" + - desc: "Animate a first-frame image" + cmd: "heygen model videos create -d '{\"model\":\"heygen-video-1\",\"mode\":\"image_to_video\",\"prompt\":\"The person smiles and waves\",\"image\":{\"type\":\"asset_id\",\"asset_id\":\"\"}}'" + - desc: "Generate with reference images and reuse an idempotency key for retries" + cmd: "heygen model videos create --idempotency-key -d '{\"model\":\"heygen-video-1\",\"mode\":\"reference_to_video\",\"prompt\":\"The person walks through a garden\",\"reference_images\":[{\"type\":\"asset_id\",\"asset_id\":\"\"}]}'" +"GET /v3/models/videos/{video_id}": + - desc: "Poll an Instant Video until completed, failed, or cancelled" + cmd: "heygen model videos get " diff --git a/gen/folder.go b/gen/folder.go new file mode 100644 index 0000000..93b640e --- /dev/null +++ b/gen/folder.go @@ -0,0 +1,87 @@ +// Code generated by heygen-cli/codegen. DO NOT EDIT. + +package gen + +import "github.com/heygen-com/heygen-cli/internal/command" + +var FolderCreate = &command.Spec{ + Group: "folder", + Name: "create", + Summary: "Create Folder", + Description: "Creates a folder in the caller's workspace, at the root or inside another folder, and returns it. Pass the returned folder_id as folder_id to POST /v3/videos or POST /v3/video-translations to place the output in it, or as parent_id to this endpoint to nest another folder. Folders and their contents are visible in the HeyGen web app. Sibling folders may share a name; this endpoint never looks up an existing folder by name, so store the ids you receive rather than recreating a tree on retry, and send an Idempotency-Key so a retried request returns the folder the first attempt created. API keys need the videos:write scope; a key scoped only to translations can place translations in a folder but cannot create one.", + RequestSchema: "{\n \"description\": \"Request body for POST /v3/folders.\",\n \"properties\": {\n \"name\": {\n \"description\": \"Folder name, 1-256 characters. Sibling folders may share a name.\",\n \"type\": \"string\"\n },\n \"parent_id\": {\n \"description\": \"ID of the folder to create this one in. Omit, pass null, or pass an empty string to create it at the workspace root. The parent must be a folder in the caller's workspace that is not in the trash.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"mixed\",\n \"description\": \"Kind of folder. 'mixed' is what the HeyGen web app's New folder action creates; 'video_translate' is what the Video Translate page creates. The web app lists all three kinds together in its folder views, and every kind accepts videos and translations placed with folder_id on POST /v3/videos and POST /v3/video-translations.\",\n \"enum\": [\n \"mixed\",\n \"video\",\n \"video_translate\"\n ],\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"name\"\n ],\n \"type\": \"object\"\n}", + ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"description\": \"A folder in the caller's workspace.\",\n \"properties\": {\n \"created_at\": {\n \"description\": \"Unix timestamp (seconds) when the folder was created.\",\n \"type\": \"integer\"\n },\n \"creator_username\": {\n \"description\": \"Username of the workspace member who created the folder.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"folder_id\": {\n \"description\": \"Unique folder identifier. Pass as folder_id to POST /v3/videos or POST /v3/video-translations, or as parent_id to POST /v3/folders.\",\n \"type\": \"string\"\n },\n \"is_trash\": {\n \"description\": \"Whether the folder is in the trash.\",\n \"type\": \"boolean\"\n },\n \"name\": {\n \"description\": \"Folder name.\",\n \"type\": \"string\"\n },\n \"parent_id\": {\n \"description\": \"ID of the containing folder. Absent for a folder at the workspace root.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Kind of folder; see the type field of POST /v3/folders.\",\n \"enum\": [\n \"mixed\",\n \"video\",\n \"video_translate\"\n ],\n \"type\": \"string\"\n },\n \"updated_at\": {\n \"description\": \"Unix timestamp (seconds) when the folder was last updated.\",\n \"type\": \"integer\"\n }\n },\n \"required\": [\n \"folder_id\",\n \"name\",\n \"type\",\n \"is_trash\",\n \"created_at\",\n \"updated_at\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", + Endpoint: "/v3/folders", + Method: "POST", + BodyEncoding: "json", + Examples: []string{ + "# Create a folder in the workspace\n heygen folder create --name 'Campaign'", + }, + Flags: []command.FlagSpec{ + { + Name: "idempotency-key", + Type: "string", + Default: "", + Help: "Optional client-supplied key for safely retrying mutations. Subsequent calls within 24 hours that share this key replay the original response — even if the request body differs slightly (a warning is logged). A retry that arrives while the original is still in flight gets a 409 `request_in_progress`. Keys must be 1–255 characters from `[A-Za-z0-9_:.-]`; a UUID is a safe default. Scope is per-endpoint and per-resource: the same key on a different route or path parameter is independent. Example: 550e8400-e29b-41d4-a716-446655440000", + Required: false, + Enum: nil, + Min: nil, + Max: nil, + Source: "header", + JSONName: "Idempotency-Key", + }, + { + Name: "name", + Type: "string", + Default: "", + Help: "Folder name, 1-256 characters. Sibling folders may share a name.", + Required: true, + Enum: nil, + Min: nil, + Max: nil, + Source: "body", + JSONName: "name", + }, + { + Name: "parent-id", + Type: "string", + Default: "", + Help: "ID of the folder to create this one in. Omit, pass null, or pass an empty string to create it at the workspace root. The parent must be a folder in the caller's workspace that is not in the trash.", + Required: false, + Enum: nil, + Min: nil, + Max: nil, + Source: "body", + JSONName: "parent_id", + }, + { + Name: "type", + Type: "string", + Default: "mixed", + Help: "Kind of folder. 'mixed' is what the HeyGen web app's New folder action creates; 'video_translate' is what the Video Translate page creates. The web app lists all three kinds together in its folder views, and every kind accepts videos and translations placed with folder_id on POST /v3/videos and POST /v3/video-translations.", + Required: false, + Enum: []string{"mixed", "video", "video_translate"}, + Min: nil, + Max: nil, + Source: "body", + JSONName: "type", + }, + }, +} + +var FolderGet = &command.Spec{ + Group: "folder", + Name: "get", + Summary: "Get Folder", + Description: "Returns one folder in the caller's workspace, including one that is in the trash (is_trash is true). Use it to confirm a stored folder_id still exists before placing content in it. A folder that was deleted, or that belongs to another workspace, is reported as not found. API keys need the videos:read scope.", + ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"description\": \"A folder in the caller's workspace.\",\n \"properties\": {\n \"created_at\": {\n \"description\": \"Unix timestamp (seconds) when the folder was created.\",\n \"type\": \"integer\"\n },\n \"creator_username\": {\n \"description\": \"Username of the workspace member who created the folder.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"folder_id\": {\n \"description\": \"Unique folder identifier. Pass as folder_id to POST /v3/videos or POST /v3/video-translations, or as parent_id to POST /v3/folders.\",\n \"type\": \"string\"\n },\n \"is_trash\": {\n \"description\": \"Whether the folder is in the trash.\",\n \"type\": \"boolean\"\n },\n \"name\": {\n \"description\": \"Folder name.\",\n \"type\": \"string\"\n },\n \"parent_id\": {\n \"description\": \"ID of the containing folder. Absent for a folder at the workspace root.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Kind of folder; see the type field of POST /v3/folders.\",\n \"enum\": [\n \"mixed\",\n \"video\",\n \"video_translate\"\n ],\n \"type\": \"string\"\n },\n \"updated_at\": {\n \"description\": \"Unix timestamp (seconds) when the folder was last updated.\",\n \"type\": \"integer\"\n }\n },\n \"required\": [\n \"folder_id\",\n \"name\",\n \"type\",\n \"is_trash\",\n \"created_at\",\n \"updated_at\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", + Endpoint: "/v3/folders/{folder_id}", + Method: "GET", + BodyEncoding: "", + Examples: []string{ + "# Get a folder by its id\n heygen folder get ", + }, + Args: []command.ArgSpec{ + {Name: "folder-id", Param: "folder_id", Help: ""}, + }, +} diff --git a/gen/lipsync.go b/gen/lipsync.go index cc315dd..c653736 100644 --- a/gen/lipsync.go +++ b/gen/lipsync.go @@ -9,7 +9,7 @@ var LipsyncBatchesCreate = &command.Spec{ Name: "batches create", Summary: "Create Lipsync Batch", Description: "Submit up to 100 lipsync payloads as a single batch. Each payload becomes one batch item, created and processed independently so one bad source does not fail the rest. Returns 202 with a batch_id; poll GET /v3/lipsyncs/batches/{batch_id} for progress. Pass an Idempotency-Key header to make retries safe — the same key returns the same batch.", - RequestSchema: "{\n \"properties\": {\n \"callback_url\": {\n \"description\": \"Webhook URL invoked once when every item in the batch reaches a terminal state.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"lipsyncs\": {\n \"description\": \"Lipsync payloads, identical in shape to POST /v3/lipsyncs. Each entry becomes exactly one batch item (no expansion); the item count is capped at 100.\",\n \"items\": {\n \"description\": \"Request body for POST /v3/lipsyncs.\",\n \"properties\": {\n \"audio\": {\n \"description\": \"Replacement audio — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"callback_id\": {\n \"description\": \"ID included in webhook payload\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL for completion notifications\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"disable_music_track\": {\n \"default\": false,\n \"description\": \"Remove background music\",\n \"type\": \"boolean\"\n },\n \"enable_caption\": {\n \"default\": false,\n \"deprecated\": true,\n \"description\": \"Deprecated and ignored: captions are always generated; whether to display them is a download-side choice.\",\n \"type\": \"boolean\"\n },\n \"enable_dynamic_duration\": {\n \"default\": true,\n \"description\": \"Allow dynamic duration adjustment\",\n \"type\": \"boolean\"\n },\n \"enable_speech_enhancement\": {\n \"default\": false,\n \"description\": \"Enhance speech quality\",\n \"type\": \"boolean\"\n },\n \"enable_watermark\": {\n \"default\": false,\n \"description\": \"Add watermark to output\",\n \"type\": \"boolean\"\n },\n \"end_time\": {\n \"description\": \"End time in seconds for partial lipsync\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"folder_id\": {\n \"description\": \"Project/folder ID to organize lipsync into\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"fps_mode\": {\n \"description\": \"Frame rate mode: 'vfr', 'cfr', or 'passthrough'.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"keep_the_same_format\": {\n \"description\": \"Preserve the source video's encoding specs (resolution, bitrate).\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"mode\": {\n \"default\": \"speed\",\n \"description\": \"Quality mode: 'speed' (faster) or 'precision' (higher quality, uses avatar inference)\",\n \"enum\": [\n \"speed\",\n \"precision\"\n ],\n \"type\": \"string\"\n },\n \"start_time\": {\n \"description\": \"Start time in seconds for partial lipsync\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"title\": {\n \"description\": \"Title for the lipsync job\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"video\": {\n \"description\": \"Source video — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n }\n },\n \"required\": [\n \"video\",\n \"audio\"\n ],\n \"type\": \"object\"\n },\n \"type\": \"array\"\n },\n \"title\": {\n \"description\": \"Display name for the batch, shown in the HeyGen app.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"lipsyncs\"\n ],\n \"type\": \"object\"\n}", + RequestSchema: "{\n \"properties\": {\n \"callback_url\": {\n \"description\": \"Webhook URL invoked once when every item in the batch reaches a terminal state.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"lipsyncs\": {\n \"description\": \"Lipsync payloads, identical in shape to POST /v3/lipsyncs. Each entry becomes exactly one batch item (no expansion); the item count is capped at 100.\",\n \"items\": {\n \"description\": \"Request body for POST /v3/lipsyncs.\",\n \"properties\": {\n \"audio\": {\n \"description\": \"Replacement audio — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"callback_id\": {\n \"description\": \"ID included in webhook payload\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL for completion notifications\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"disable_music_track\": {\n \"default\": false,\n \"description\": \"Remove background music\",\n \"type\": \"boolean\"\n },\n \"enable_caption\": {\n \"default\": false,\n \"deprecated\": true,\n \"description\": \"Deprecated and ignored: captions are always generated; whether to display them is a download-side choice.\",\n \"type\": \"boolean\"\n },\n \"enable_dynamic_duration\": {\n \"default\": true,\n \"description\": \"Allow dynamic duration adjustment\",\n \"type\": \"boolean\"\n },\n \"enable_speech_enhancement\": {\n \"default\": false,\n \"description\": \"Enhance speech quality\",\n \"type\": \"boolean\"\n },\n \"enable_watermark\": {\n \"default\": false,\n \"description\": \"Add watermark to output\",\n \"type\": \"boolean\"\n },\n \"end_time\": {\n \"description\": \"End time in seconds for partial lipsync\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"folder_id\": {\n \"description\": \"Destination folder ID in the caller's workspace, for example one returned by POST /v3/folders. Omit, pass null, or pass an empty string to place the result at the workspace root. The id must name a folder that is not in the trash and that the caller can write to; an unknown id, a folder in another workspace, a project of another kind such as a brand kit, or a trashed folder is rejected with 404 before anything is created.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"fps_mode\": {\n \"description\": \"Frame rate mode: 'vfr', 'cfr', or 'passthrough'.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"keep_the_same_format\": {\n \"description\": \"Preserve the source video's encoding specs (resolution, bitrate).\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"mode\": {\n \"default\": \"speed\",\n \"description\": \"Quality mode: 'speed' (faster) or 'precision' (higher quality, uses avatar inference)\",\n \"enum\": [\n \"speed\",\n \"precision\"\n ],\n \"type\": \"string\"\n },\n \"start_time\": {\n \"description\": \"Start time in seconds for partial lipsync\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"title\": {\n \"description\": \"Title for the lipsync job\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"video\": {\n \"description\": \"Source video — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n }\n },\n \"required\": [\n \"video\",\n \"audio\"\n ],\n \"type\": \"object\"\n },\n \"type\": \"array\"\n },\n \"title\": {\n \"description\": \"Display name for the batch, shown in the HeyGen app.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"lipsyncs\"\n ],\n \"type\": \"object\"\n}", Endpoint: "/v3/lipsyncs/batches", Method: "POST", BodyEncoding: "json", @@ -106,7 +106,7 @@ var LipsyncCreate = &command.Spec{ Name: "create", Summary: "Create Lipsync", Description: "Replaces the audio on an existing video and re-animates the speaker's lip movements to match the new audio. Use mode: 'speed' for fast output or 'precision' for high-quality lip-sync.", - RequestSchema: "{\n \"description\": \"Request body for POST /v3/lipsyncs.\",\n \"properties\": {\n \"audio\": {\n \"description\": \"Replacement audio — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"callback_id\": {\n \"description\": \"ID included in webhook payload\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL for completion notifications\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"disable_music_track\": {\n \"default\": false,\n \"description\": \"Remove background music\",\n \"type\": \"boolean\"\n },\n \"enable_caption\": {\n \"default\": false,\n \"deprecated\": true,\n \"description\": \"Deprecated and ignored: captions are always generated; whether to display them is a download-side choice.\",\n \"type\": \"boolean\"\n },\n \"enable_dynamic_duration\": {\n \"default\": true,\n \"description\": \"Allow dynamic duration adjustment\",\n \"type\": \"boolean\"\n },\n \"enable_speech_enhancement\": {\n \"default\": false,\n \"description\": \"Enhance speech quality\",\n \"type\": \"boolean\"\n },\n \"enable_watermark\": {\n \"default\": false,\n \"description\": \"Add watermark to output\",\n \"type\": \"boolean\"\n },\n \"end_time\": {\n \"description\": \"End time in seconds for partial lipsync\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"folder_id\": {\n \"description\": \"Project/folder ID to organize lipsync into\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"fps_mode\": {\n \"description\": \"Frame rate mode: 'vfr', 'cfr', or 'passthrough'.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"keep_the_same_format\": {\n \"description\": \"Preserve the source video's encoding specs (resolution, bitrate).\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"mode\": {\n \"default\": \"speed\",\n \"description\": \"Quality mode: 'speed' (faster) or 'precision' (higher quality, uses avatar inference)\",\n \"enum\": [\n \"speed\",\n \"precision\"\n ],\n \"type\": \"string\"\n },\n \"start_time\": {\n \"description\": \"Start time in seconds for partial lipsync\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"title\": {\n \"description\": \"Title for the lipsync job\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"video\": {\n \"description\": \"Source video — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n }\n },\n \"required\": [\n \"video\",\n \"audio\"\n ],\n \"type\": \"object\"\n}", + RequestSchema: "{\n \"description\": \"Request body for POST /v3/lipsyncs.\",\n \"properties\": {\n \"audio\": {\n \"description\": \"Replacement audio — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"callback_id\": {\n \"description\": \"ID included in webhook payload\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL for completion notifications\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"disable_music_track\": {\n \"default\": false,\n \"description\": \"Remove background music\",\n \"type\": \"boolean\"\n },\n \"enable_caption\": {\n \"default\": false,\n \"deprecated\": true,\n \"description\": \"Deprecated and ignored: captions are always generated; whether to display them is a download-side choice.\",\n \"type\": \"boolean\"\n },\n \"enable_dynamic_duration\": {\n \"default\": true,\n \"description\": \"Allow dynamic duration adjustment\",\n \"type\": \"boolean\"\n },\n \"enable_speech_enhancement\": {\n \"default\": false,\n \"description\": \"Enhance speech quality\",\n \"type\": \"boolean\"\n },\n \"enable_watermark\": {\n \"default\": false,\n \"description\": \"Add watermark to output\",\n \"type\": \"boolean\"\n },\n \"end_time\": {\n \"description\": \"End time in seconds for partial lipsync\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"folder_id\": {\n \"description\": \"Destination folder ID in the caller's workspace, for example one returned by POST /v3/folders. Omit, pass null, or pass an empty string to place the result at the workspace root. The id must name a folder that is not in the trash and that the caller can write to; an unknown id, a folder in another workspace, a project of another kind such as a brand kit, or a trashed folder is rejected with 404 before anything is created.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"fps_mode\": {\n \"description\": \"Frame rate mode: 'vfr', 'cfr', or 'passthrough'.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"keep_the_same_format\": {\n \"description\": \"Preserve the source video's encoding specs (resolution, bitrate).\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"mode\": {\n \"default\": \"speed\",\n \"description\": \"Quality mode: 'speed' (faster) or 'precision' (higher quality, uses avatar inference)\",\n \"enum\": [\n \"speed\",\n \"precision\"\n ],\n \"type\": \"string\"\n },\n \"start_time\": {\n \"description\": \"Start time in seconds for partial lipsync\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"title\": {\n \"description\": \"Title for the lipsync job\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"video\": {\n \"description\": \"Source video — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n }\n },\n \"required\": [\n \"video\",\n \"audio\"\n ],\n \"type\": \"object\"\n}", ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"description\": \"Response for POST /v3/lipsyncs.\",\n \"properties\": {\n \"lipsync_id\": {\n \"description\": \"Lipsync ID — use GET /v3/lipsyncs/{id} to poll status\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"lipsync_id\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", Endpoint: "/v3/lipsyncs", Method: "POST", @@ -228,7 +228,7 @@ var LipsyncCreate = &command.Spec{ Name: "folder-id", Type: "string", Default: "", - Help: "Project/folder ID to organize lipsync into", + Help: "Destination folder ID in the caller's workspace, for example one returned by POST /v3/folders. Omit, pass null, or pass an empty string to place the result at the workspace root. The id must name a folder that is not in the trash and that the caller can write to; an unknown id, a folder in another workspace, a project of another kind such as a brand kit, or a trashed folder is rejected with 404 before anything is created.", Required: false, Enum: nil, Min: nil, diff --git a/gen/model.go b/gen/model.go index 0dee050..b13e8e7 100644 --- a/gen/model.go +++ b/gen/model.go @@ -8,8 +8,8 @@ var ModelAudioTtsCreate = &command.Spec{ Group: "model", Name: "audio tts create", Summary: "Generate Speech", - Description: "Generates speech using the voice identified by `voice_id` and returns a URL for one completed mono PCM16 WAV file at 44.1 kHz. The request remains open until synthesis and output assembly finish. The voice must be an ACTIVE professional voice. Rate limit: 30 requests per minute per workspace member.", - RequestSchema: "{\n \"properties\": {\n \"language\": {\n \"description\": \"Language code used for speech synthesis, such as `en`.\",\n \"type\": \"string\"\n },\n \"seed\": {\n \"description\": \"Optional best-effort deterministic generation seed.\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"text\": {\n \"description\": \"Plain text to synthesize. SSML and break tags are not supported.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Identifier of the voice used for speech synthesis.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"voice_id\",\n \"text\",\n \"language\"\n ],\n \"type\": \"object\"\n}", + Description: "Generates speech using the voice identified by `voice_id` and returns a URL for one completed mono PCM16 WAV file at 44.1 kHz. The request remains open until synthesis and output assembly finish. The voice must be an ACTIVE HeyGen Voice clone, instant or professional, owned by the caller's workspace; `seed`, `speed`, `pitch_shift` and `pitch_variance` apply to professional clones only. Rate limit: 30 requests per minute per workspace member.", + RequestSchema: "{\n \"properties\": {\n \"language\": {\n \"description\": \"Language code used for speech synthesis, such as `en`.\",\n \"type\": \"string\"\n },\n \"pitch_shift\": {\n \"description\": \"Pitch shift in semitones.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"pitch_variance\": {\n \"description\": \"Pitch variation multiplier; 1.0 preserves the voice default.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"seed\": {\n \"description\": \"Optional best-effort deterministic generation seed.\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"speed\": {\n \"description\": \"Speech speed multiplier.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"text\": {\n \"description\": \"Plain text to synthesize. A professional voice clone takes \\u003cbreak time=\\\"Xs\\\"/\\u003e pauses (seconds, up to 5 per pause, e.g. \\u003cbreak time=\\\"0.5s\\\"/\\u003e); other SSML, and break tags for an instant voice clone, are not supported.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Identifier of the voice used for speech synthesis.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"voice_id\",\n \"text\",\n \"language\"\n ],\n \"type\": \"object\"\n}", ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"properties\": {\n \"audio_url\": {\n \"description\": \"URL of the generated audio file.\",\n \"type\": \"string\"\n },\n \"duration\": {\n \"description\": \"Duration of the generated audio in seconds.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"audio_url\",\n \"duration\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", Endpoint: "/v3/models/audio/tts", Method: "POST", @@ -31,6 +31,30 @@ var ModelAudioTtsCreate = &command.Spec{ Source: "body", JSONName: "language", }, + { + Name: "pitch-shift", + Type: "float64", + Default: "", + Help: "Pitch shift in semitones.", + Required: false, + Enum: nil, + Min: nil, + Max: nil, + Source: "body", + JSONName: "pitch_shift", + }, + { + Name: "pitch-variance", + Type: "float64", + Default: "", + Help: "Pitch variation multiplier; 1.0 preserves the voice default.", + Required: false, + Enum: nil, + Min: nil, + Max: nil, + Source: "body", + JSONName: "pitch_variance", + }, { Name: "seed", Type: "int", @@ -43,11 +67,23 @@ var ModelAudioTtsCreate = &command.Spec{ Source: "body", JSONName: "seed", }, + { + Name: "speed", + Type: "float64", + Default: "", + Help: "Speech speed multiplier.", + Required: false, + Enum: nil, + Min: nil, + Max: nil, + Source: "body", + JSONName: "speed", + }, { Name: "text", Type: "string", Default: "", - Help: "Plain text to synthesize. SSML and break tags are not supported.", + Help: "Plain text to synthesize. A professional voice clone takes pauses (seconds, up to 5 per pause, e.g. ); other SSML, and break tags for an instant voice clone, are not supported.", Required: true, Enum: nil, Min: nil, @@ -170,7 +206,7 @@ var ModelAudioVoicesGet = &command.Spec{ Name: "audio voices get", Summary: "Get an Audio Voice", Description: "Returns one caller-owned model-backed audio voice and its current lifecycle state. `PENDING` covers queued and running work, `ACTIVE` is ready for inference, and `FAILED` is terminal.", - ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"description\": \"Customer-visible state of one model-backed audio voice.\",\n \"properties\": {\n \"created_at\": {\n \"description\": \"Unix timestamp in seconds when the voice was created.\",\n \"type\": \"integer\"\n },\n \"failure_reason\": {\n \"description\": \"Stable failure code. Present only when `status` is `FAILED`.\",\n \"enum\": [\n \"INVALID_AUDIO\",\n \"INSUFFICIENT_AUDIO\",\n \"PREPROCESSING_FAILED\",\n \"TRAINING_FAILED\",\n \"ARTIFACT_VALIDATION_FAILED\",\n \"INTERNAL_ERROR\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"language\": {\n \"description\": \"Primary language code supplied when the voice was created.\",\n \"type\": \"string\"\n },\n \"mode\": {\n \"description\": \"Voice creation mode.\",\n \"enum\": [\n \"professional\"\n ],\n \"type\": \"string\"\n },\n \"name\": {\n \"description\": \"Display name of the voice.\",\n \"type\": \"string\"\n },\n \"status\": {\n \"description\": \"Current lifecycle state: `PENDING`, `ACTIVE`, or `FAILED`.\",\n \"enum\": [\n \"PENDING\",\n \"ACTIVE\",\n \"FAILED\"\n ],\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Stable identifier for this voice.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"voice_id\",\n \"mode\",\n \"name\",\n \"language\",\n \"status\",\n \"created_at\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", + ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"description\": \"Customer-visible state of one model-backed audio voice.\",\n \"properties\": {\n \"created_at\": {\n \"description\": \"Unix timestamp in seconds when the voice was created.\",\n \"type\": \"integer\"\n },\n \"failure_reason\": {\n \"description\": \"Stable failure code. Present only when `status` is `FAILED`.\",\n \"enum\": [\n \"INVALID_AUDIO\",\n \"INSUFFICIENT_AUDIO\",\n \"PREPROCESSING_FAILED\",\n \"TRAINING_FAILED\",\n \"DESIGN_FAILED\",\n \"ARTIFACT_VALIDATION_FAILED\",\n \"INTERNAL_ERROR\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"language\": {\n \"description\": \"Primary language code supplied when the voice was created.\",\n \"type\": \"string\"\n },\n \"mode\": {\n \"description\": \"Voice creation mode.\",\n \"enum\": [\n \"professional\"\n ],\n \"type\": \"string\"\n },\n \"name\": {\n \"description\": \"Display name of the voice.\",\n \"type\": \"string\"\n },\n \"status\": {\n \"description\": \"Current lifecycle state: `PENDING`, `ACTIVE`, or `FAILED`.\",\n \"enum\": [\n \"PENDING\",\n \"ACTIVE\",\n \"FAILED\"\n ],\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Stable identifier for this voice.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"voice_id\",\n \"mode\",\n \"name\",\n \"language\",\n \"status\",\n \"created_at\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", Endpoint: "/v3/models/audio/voices/{voice_id}", Method: "GET", BodyEncoding: "", @@ -187,7 +223,7 @@ var ModelAudioVoicesList = &command.Spec{ Name: "audio voices list", Summary: "List Audio Voices", Description: "Returns the model-backed audio voices in the caller's workspace, ordered newest first. Use `limit` and `token` to retrieve additional pages.", - ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"items\": {\n \"description\": \"Customer-visible state of one model-backed audio voice.\",\n \"properties\": {\n \"created_at\": {\n \"description\": \"Unix timestamp in seconds when the voice was created.\",\n \"type\": \"integer\"\n },\n \"failure_reason\": {\n \"description\": \"Stable failure code. Present only when `status` is `FAILED`.\",\n \"enum\": [\n \"INVALID_AUDIO\",\n \"INSUFFICIENT_AUDIO\",\n \"PREPROCESSING_FAILED\",\n \"TRAINING_FAILED\",\n \"ARTIFACT_VALIDATION_FAILED\",\n \"INTERNAL_ERROR\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"language\": {\n \"description\": \"Primary language code supplied when the voice was created.\",\n \"type\": \"string\"\n },\n \"mode\": {\n \"description\": \"Voice creation mode.\",\n \"enum\": [\n \"professional\"\n ],\n \"type\": \"string\"\n },\n \"name\": {\n \"description\": \"Display name of the voice.\",\n \"type\": \"string\"\n },\n \"status\": {\n \"description\": \"Current lifecycle state: `PENDING`, `ACTIVE`, or `FAILED`.\",\n \"enum\": [\n \"PENDING\",\n \"ACTIVE\",\n \"FAILED\"\n ],\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Stable identifier for this voice.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"voice_id\",\n \"mode\",\n \"name\",\n \"language\",\n \"status\",\n \"created_at\"\n ],\n \"type\": \"object\"\n },\n \"type\": \"array\"\n },\n \"has_more\": {\n \"description\": \"Whether more pages are available\",\n \"type\": \"boolean\"\n },\n \"next_token\": {\n \"description\": \"Opaque cursor for the next page\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", + ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"items\": {\n \"description\": \"Customer-visible state of one model-backed audio voice.\",\n \"properties\": {\n \"created_at\": {\n \"description\": \"Unix timestamp in seconds when the voice was created.\",\n \"type\": \"integer\"\n },\n \"failure_reason\": {\n \"description\": \"Stable failure code. Present only when `status` is `FAILED`.\",\n \"enum\": [\n \"INVALID_AUDIO\",\n \"INSUFFICIENT_AUDIO\",\n \"PREPROCESSING_FAILED\",\n \"TRAINING_FAILED\",\n \"DESIGN_FAILED\",\n \"ARTIFACT_VALIDATION_FAILED\",\n \"INTERNAL_ERROR\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"language\": {\n \"description\": \"Primary language code supplied when the voice was created.\",\n \"type\": \"string\"\n },\n \"mode\": {\n \"description\": \"Voice creation mode.\",\n \"enum\": [\n \"professional\"\n ],\n \"type\": \"string\"\n },\n \"name\": {\n \"description\": \"Display name of the voice.\",\n \"type\": \"string\"\n },\n \"status\": {\n \"description\": \"Current lifecycle state: `PENDING`, `ACTIVE`, or `FAILED`.\",\n \"enum\": [\n \"PENDING\",\n \"ACTIVE\",\n \"FAILED\"\n ],\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Stable identifier for this voice.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"voice_id\",\n \"mode\",\n \"name\",\n \"language\",\n \"status\",\n \"created_at\"\n ],\n \"type\": \"object\"\n },\n \"type\": \"array\"\n },\n \"has_more\": {\n \"description\": \"Whether more pages are available\",\n \"type\": \"boolean\"\n },\n \"next_token\": {\n \"description\": \"Opaque cursor for the next page\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", Endpoint: "/v3/models/audio/voices", Method: "GET", BodyEncoding: "", @@ -222,3 +258,50 @@ var ModelAudioVoicesList = &command.Spec{ }, }, } + +var ModelVideosCreate = &command.Spec{ + Group: "model", + Name: "videos create", + Summary: "Create Instant Video", + Description: "Generate an Instant Video with model `heygen-video-1` from a text prompt, a first-frame image, or reference images, videos, and audio. Select `text_to_video`, `image_to_video`, or `reference_to_video` with `mode`. Prompts accept at most 5,000 Unicode characters. Duration is 5–15 whole seconds; resolution is 480p or 768p. Image-to-video follows the image proportions and ignores aspect_ratio. Reference-to-video requires at least one image or video, with at most nine images, three videos, three audio recordings, and twelve references total. Reference videos must be within a 1:4–4:1 ratio. Assets accept HTTPS URLs, uploaded asset IDs, or inline base64. Returns 202 with a video_id; poll GET /v3/models/videos/{video_id} for completion and the download URL. Pass an Idempotency-Key header to retry safely; without a key, each submission creates a new generation.", + RequestSchema: "{\n \"description\": \"Body of POST /v3/models/videos: one schema per generation mode, selected by the required mode field.\",\n \"discriminator\": {\n \"mapping\": {\n \"image_to_video\": \"#/components/schemas/ImageToVideoRequest\",\n \"reference_to_video\": \"#/components/schemas/ReferenceToVideoRequest\",\n \"text_to_video\": \"#/components/schemas/TextToVideoRequest\"\n },\n \"propertyName\": \"mode\"\n },\n \"oneOf\": [\n {\n \"properties\": {\n \"aspect_ratio\": {\n \"default\": \"16:9\",\n \"description\": \"Output aspect ratio for text-to-video. Defaults to 16:9.\",\n \"enum\": [\n \"21:9\",\n \"16:9\",\n \"4:3\",\n \"1:1\",\n \"3:4\",\n \"9:16\"\n ],\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"Client tracking ID echoed in the terminal webhook event.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"HTTPS webhook URL for the terminal generation event.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"duration\": {\n \"default\": 5,\n \"description\": \"Requested video duration in whole seconds, from 5 through 15 inclusive.\",\n \"type\": \"integer\"\n },\n \"mode\": {\n \"description\": \"Generate from the prompt alone.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Required video generation model. Only heygen-video-1 is supported.\",\n \"type\": \"string\"\n },\n \"prompt\": {\n \"description\": \"Instructions for the video to generate. At most 5,000 Unicode characters.\",\n \"type\": \"string\"\n },\n \"prompt_enhancement\": {\n \"default\": \"turbo\",\n \"description\": \"Prompt enhancement mode: turbo, quality, or disabled. Defaults to turbo; default is an alias for turbo.\",\n \"enum\": [\n \"turbo\",\n \"quality\",\n \"default\",\n \"disabled\"\n ],\n \"type\": \"string\"\n },\n \"resolution\": {\n \"default\": \"768p\",\n \"description\": \"Output resolution: 480p or 768p. Defaults to 768p.\",\n \"enum\": [\n \"480p\",\n \"768p\"\n ],\n \"type\": \"string\"\n },\n \"seed\": {\n \"description\": \"Generation seed. A random seed is chosen when omitted.\",\n \"nullable\": true,\n \"type\": \"integer\"\n }\n },\n \"required\": [\n \"model\",\n \"prompt\",\n \"mode\"\n ],\n \"type\": \"object\"\n },\n {\n \"properties\": {\n \"aspect_ratio\": {\n \"description\": \"The output follows the image proportions, aligned to supported pixel dimensions. Any supplied aspect ratio is ignored.\",\n \"enum\": [\n \"21:9\",\n \"16:9\",\n \"4:3\",\n \"1:1\",\n \"3:4\",\n \"9:16\",\n \"adaptive\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"Client tracking ID echoed in the terminal webhook event.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"HTTPS webhook URL for the terminal generation event.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"duration\": {\n \"default\": 5,\n \"description\": \"Requested video duration in whole seconds, from 5 through 15 inclusive.\",\n \"type\": \"integer\"\n },\n \"image\": {\n \"description\": \"First-frame image.\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"mode\": {\n \"description\": \"Animate a first-frame image.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Required video generation model. Only heygen-video-1 is supported.\",\n \"type\": \"string\"\n },\n \"prompt\": {\n \"description\": \"Instructions for the video to generate. At most 5,000 Unicode characters.\",\n \"type\": \"string\"\n },\n \"prompt_enhancement\": {\n \"default\": \"turbo\",\n \"description\": \"Prompt enhancement mode: turbo, quality, or disabled. Defaults to turbo; default is an alias for turbo.\",\n \"enum\": [\n \"turbo\",\n \"quality\",\n \"default\",\n \"disabled\"\n ],\n \"type\": \"string\"\n },\n \"resolution\": {\n \"default\": \"768p\",\n \"description\": \"Output resolution: 480p or 768p. Defaults to 768p.\",\n \"enum\": [\n \"480p\",\n \"768p\"\n ],\n \"type\": \"string\"\n },\n \"seed\": {\n \"description\": \"Generation seed. A random seed is chosen when omitted.\",\n \"nullable\": true,\n \"type\": \"integer\"\n }\n },\n \"required\": [\n \"model\",\n \"prompt\",\n \"mode\",\n \"image\"\n ],\n \"type\": \"object\"\n },\n {\n \"properties\": {\n \"aspect_ratio\": {\n \"default\": \"adaptive\",\n \"description\": \"Output aspect ratio. Adaptive uses the first reference image, or the first reference video when no images are supplied. Defaults to adaptive.\",\n \"enum\": [\n \"21:9\",\n \"16:9\",\n \"4:3\",\n \"1:1\",\n \"3:4\",\n \"9:16\",\n \"adaptive\"\n ],\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"Client tracking ID echoed in the terminal webhook event.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"HTTPS webhook URL for the terminal generation event.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"duration\": {\n \"default\": 5,\n \"description\": \"Requested video duration in whole seconds, from 5 through 15 inclusive.\",\n \"type\": \"integer\"\n },\n \"mode\": {\n \"description\": \"Generate guided by reference images, videos, and audio.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Required video generation model. Only heygen-video-1 is supported.\",\n \"type\": \"string\"\n },\n \"prompt\": {\n \"description\": \"Instructions for the video to generate. At most 5,000 Unicode characters.\",\n \"type\": \"string\"\n },\n \"prompt_enhancement\": {\n \"default\": \"turbo\",\n \"description\": \"Prompt enhancement mode: turbo, quality, or disabled. Defaults to turbo; default is an alias for turbo.\",\n \"enum\": [\n \"turbo\",\n \"quality\",\n \"default\",\n \"disabled\"\n ],\n \"type\": \"string\"\n },\n \"reference_audio\": {\n \"description\": \"Reference audio recordings. At most three.\",\n \"items\": {\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": \"array\"\n },\n \"reference_images\": {\n \"description\": \"Reference images. At most nine.\",\n \"items\": {\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": \"array\"\n },\n \"reference_videos\": {\n \"description\": \"Reference videos. At most three.\",\n \"items\": {\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": \"array\"\n },\n \"resolution\": {\n \"default\": \"768p\",\n \"description\": \"Output resolution: 480p or 768p. Defaults to 768p.\",\n \"enum\": [\n \"480p\",\n \"768p\"\n ],\n \"type\": \"string\"\n },\n \"seed\": {\n \"description\": \"Generation seed. A random seed is chosen when omitted.\",\n \"nullable\": true,\n \"type\": \"integer\"\n }\n },\n \"required\": [\n \"model\",\n \"prompt\",\n \"mode\"\n ],\n \"type\": \"object\"\n }\n ]\n}", + Endpoint: "/v3/models/videos", + Method: "POST", + BodyEncoding: "json", + Examples: []string{ + "# Generate an Instant Video from a text prompt\n heygen model videos create -d '{\"model\":\"heygen-video-1\",\"mode\":\"text_to_video\",\"prompt\":\"A person walking through a garden\",\"duration\":5,\"resolution\":\"768p\"}'", + "# Animate a first-frame image\n heygen model videos create -d '{\"model\":\"heygen-video-1\",\"mode\":\"image_to_video\",\"prompt\":\"The person smiles and waves\",\"image\":{\"type\":\"asset_id\",\"asset_id\":\"\"}}'", + "# Generate with reference images and reuse an idempotency key for retries\n heygen model videos create --idempotency-key -d '{\"model\":\"heygen-video-1\",\"mode\":\"reference_to_video\",\"prompt\":\"The person walks through a garden\",\"reference_images\":[{\"type\":\"asset_id\",\"asset_id\":\"\"}]}'", + }, + Flags: []command.FlagSpec{ + { + Name: "idempotency-key", + Type: "string", + Default: "", + Help: "Optional client-supplied key for safely retrying mutations. Subsequent calls within 24 hours that share this key replay the original response — even if the request body differs slightly (a warning is logged). A retry that arrives while the original is still in flight gets a 409 `request_in_progress`. Keys must be 1–255 characters from `[A-Za-z0-9_:.-]`; a UUID is a safe default. Scope is per-endpoint and per-resource: the same key on a different route or path parameter is independent. Example: 550e8400-e29b-41d4-a716-446655440000", + Required: false, + Enum: nil, + Min: nil, + Max: nil, + Source: "header", + JSONName: "Idempotency-Key", + }, + }, +} + +var ModelVideosGet = &command.Spec{ + Group: "model", + Name: "videos get", + Summary: "Get Instant Video", + Description: "Get an Instant Video generation in the caller's workspace. Status is pending, processing, completed, failed, or cancelled. Completed results include a fresh video_url, duration, aspect_ratio, dimensions, and seed when available. Failed and cancelled results include failure_code and failure_message. Poll again to refresh an expired download URL. Missing videos and videos from another workspace return the same 404.", + ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"properties\": {\n \"aspect_ratio\": {\n \"description\": \"Output aspect ratio. Adaptive outputs report the actual width:height ratio after pixel alignment.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"created_at\": {\n \"description\": \"Request creation time as Unix seconds.\",\n \"type\": \"integer\"\n },\n \"duration\": {\n \"description\": \"Generated video duration in seconds, present when completed.\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"failure_code\": {\n \"description\": \"Failure code, present when failed or cancelled.\",\n \"enum\": [\n \"generation_failed\",\n \"generation_cancelled\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"failure_message\": {\n \"description\": \"Failure explanation, present when failed or cancelled.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"height\": {\n \"description\": \"Actual output height in pixels, present when reported by the renderer.\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"model\": {\n \"description\": \"Requested video generation model.\",\n \"enum\": [\n \"heygen-video-1\"\n ],\n \"type\": \"string\"\n },\n \"seed\": {\n \"description\": \"Generation seed, present when completed.\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"status\": {\n \"description\": \"Current generation status.\",\n \"enum\": [\n \"pending\",\n \"processing\",\n \"completed\",\n \"failed\",\n \"cancelled\"\n ],\n \"type\": \"string\"\n },\n \"timings\": {\n \"description\": \"Generation timing in seconds, when reported by the renderer.\",\n \"nullable\": true,\n \"properties\": {\n \"inference\": {\n \"description\": \"Diffusion transformer execution time in seconds. Excludes other inference stages, encoding, and queueing.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"inference\"\n ],\n \"type\": \"object\"\n },\n \"video_id\": {\n \"description\": \"Generated video identifier.\",\n \"type\": \"string\"\n },\n \"video_url\": {\n \"description\": \"Download URL, present when completed. Poll again to obtain a fresh URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"width\": {\n \"description\": \"Actual output width in pixels, present when reported by the renderer.\",\n \"nullable\": true,\n \"type\": \"integer\"\n }\n },\n \"required\": [\n \"video_id\",\n \"status\",\n \"model\",\n \"created_at\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", + Endpoint: "/v3/models/videos/{video_id}", + Method: "GET", + BodyEncoding: "", + Examples: []string{ + "# Poll an Instant Video until completed, failed, or cancelled\n heygen model videos get ", + }, + Args: []command.ArgSpec{ + {Name: "video-id", Param: "video_id", Help: ""}, + }, +} diff --git a/gen/registry.go b/gen/registry.go index efef05c..ee91eae 100644 --- a/gen/registry.go +++ b/gen/registry.go @@ -67,6 +67,10 @@ var Groups = map[string][]*command.Spec{ FillerWordRemovalCreate, FillerWordRemovalGet, }, + "folder": { + FolderCreate, + FolderGet, + }, "lipsync": { LipsyncBatchesCreate, LipsyncBatchesGet, @@ -83,6 +87,8 @@ var Groups = map[string][]*command.Spec{ ModelAudioVoicesDelete, ModelAudioVoicesGet, ModelAudioVoicesList, + ModelVideosCreate, + ModelVideosGet, }, "template": { TemplateGenerate, @@ -156,6 +162,7 @@ var GroupDescriptions = map[string]string{ "audio": "Search the background-music and sound-effects catalog", "avatar": "List and manage avatars and looks", "brand": "Brand-related resources — brand kits (colors, fonts, logos) and brand glossaries (custom term translations)", + "folder": "Create folders to organize videos and translations in the workspace", "lipsync": "Dub or replace audio on existing videos", "template": "Generate videos from reusable templates by replacing their variables", "user": "Account information and billing", diff --git a/gen/template.go b/gen/template.go index 135eb70..ef59190 100644 --- a/gen/template.go +++ b/gen/template.go @@ -9,7 +9,7 @@ var TemplateGenerate = &command.Spec{ Name: "generate", Summary: "Generate Video from Template", Description: "Generates a video from the template by replacing its variables (text, image, video, audio, character, voice). Use scene_ids to select, reorder, or repeat scenes — scenes must already exist in the template; the API cannot create new ones. Returns the created video object; poll GET /v3/videos/{video_id} or use webhooks for completion. Idempotent replays return the original creation-time snapshot (status and URLs as of the first request), not the video's current state.", - RequestSchema: "{\n \"description\": \"Request body for POST /v3/templates/{template_id}.\",\n \"properties\": {\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary controlling how custom terms are pronounced in generated speech. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"brand_voice_id\": {\n \"deprecated\": true,\n \"description\": \"Legacy field name for `brand_glossary_id`. Both are accepted and resolve to the same workspace record. Prefer `brand_glossary_id`.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"Opaque ID echoed back in webhook events for this video\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"URL called with the video result in addition to registered webhook endpoints\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"caption\": {\n \"default\": false,\n \"description\": \"Whether to burn captions into the video\",\n \"type\": \"boolean\"\n },\n \"dimension\": {\n \"description\": \"Output resolution override. Must match the template's aspect ratio.\",\n \"nullable\": true,\n \"properties\": {\n \"height\": {\n \"description\": \"Output video height in pixels (even number, 128-4096)\",\n \"type\": \"integer\"\n },\n \"width\": {\n \"description\": \"Output video width in pixels (even number, 128-4096)\",\n \"type\": \"integer\"\n }\n },\n \"required\": [\n \"width\",\n \"height\"\n ],\n \"type\": \"object\"\n },\n \"enable_sharing\": {\n \"default\": false,\n \"description\": \"Whether the generated video's share page is publicly accessible\",\n \"type\": \"boolean\"\n },\n \"folder_id\": {\n \"description\": \"Folder to place the generated video in\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"fps\": {\n \"default\": 25,\n \"description\": \"Output frame rate. One of 25, 30, or 60.\",\n \"type\": \"number\"\n },\n \"include_gif\": {\n \"default\": false,\n \"description\": \"Whether to include a GIF preview in the webhook payload\",\n \"type\": \"boolean\"\n },\n \"keep_text_vertically_centered\": {\n \"default\": false,\n \"description\": \"When true, replaced text elements are vertically re-centered based on their rendered height\",\n \"type\": \"boolean\"\n },\n \"reorder_music\": {\n \"default\": true,\n \"description\": \"When true (default), background audio tracks move with their scenes. When false, tracks stay pinned to layout positions.\",\n \"type\": \"boolean\"\n },\n \"scene_ids\": {\n \"description\": \"Scene IDs to render, in order (repeats allowed). Scenes must already exist in the template; the API can select, reorder, and repeat scenes but cannot create new ones. Omit to render all scenes in template order.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"nullable\": true,\n \"type\": \"array\"\n },\n \"subtitles\": {\n \"description\": \"Subtitle style settings. Implies captions when provided.\",\n \"nullable\": true,\n \"properties\": {\n \"alignment\": {\n \"default\": 2,\n \"description\": \"Subtitle alignment\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"disable_highlight\": {\n \"default\": false,\n \"description\": \"Override the preset's word-highlight style\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"font_size\": {\n \"description\": \"Font size override for the preset\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"position\": {\n \"description\": \"Subtitle position override\",\n \"nullable\": true,\n \"properties\": {\n \"x\": {\n \"default\": 0,\n \"description\": \"Horizontal subtitle position\",\n \"type\": \"number\"\n },\n \"y\": {\n \"default\": 0,\n \"description\": \"Vertical subtitle position\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"preset_name\": {\n \"description\": \"Subtitle preset name, e.g. 'classic', 'bold', 'bright'\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"preset_name\"\n ],\n \"type\": \"object\"\n },\n \"title\": {\n \"description\": \"Title for the generated video\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"variables\": {\n \"description\": \"Template variable replacements, keyed by the variable name defined in the template. Supply every text variable you want filled: an omitted text variable is not substituted, so its literal `{{variable_name}}` placeholder remains in the text it is bound to, whether that is a spoken script or an on-screen text element. Omitting an image, video, audio, character or voice variable is safe and keeps the value the template already holds. The defaults returned by `GET /v3/templates/{template_id}` are the template's current values, not fallbacks applied at generation time.\",\n \"properties\": {},\n \"required\": [],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"variables\"\n ],\n \"type\": \"object\"\n}", + RequestSchema: "{\n \"description\": \"Request body for POST /v3/templates/{template_id}.\",\n \"properties\": {\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary controlling how custom terms are pronounced in generated speech. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"brand_voice_id\": {\n \"deprecated\": true,\n \"description\": \"Legacy field name for `brand_glossary_id`. Both are accepted and resolve to the same workspace record. Prefer `brand_glossary_id`.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"Opaque ID echoed back in webhook events for this video\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"URL called with the video result in addition to registered webhook endpoints\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"caption\": {\n \"default\": false,\n \"description\": \"Whether to burn captions into the video\",\n \"type\": \"boolean\"\n },\n \"dimension\": {\n \"description\": \"Output resolution override. Must match the template's aspect ratio.\",\n \"nullable\": true,\n \"properties\": {\n \"height\": {\n \"description\": \"Output video height in pixels (even number, 128-4096)\",\n \"type\": \"integer\"\n },\n \"width\": {\n \"description\": \"Output video width in pixels (even number, 128-4096)\",\n \"type\": \"integer\"\n }\n },\n \"required\": [\n \"width\",\n \"height\"\n ],\n \"type\": \"object\"\n },\n \"enable_sharing\": {\n \"default\": false,\n \"description\": \"Whether the generated video's share page is publicly accessible\",\n \"type\": \"boolean\"\n },\n \"folder_id\": {\n \"description\": \"Destination folder ID in the caller's workspace, for example one returned by POST /v3/folders. Omit, pass null, or pass an empty string to place the result at the workspace root. The id must name a folder that is not in the trash and that the caller can write to; an unknown id, a folder in another workspace, a project of another kind such as a brand kit, or a trashed folder is rejected with 404 before anything is created. Supported only for single-video creation; batch video items cannot set their own destination folder.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"fps\": {\n \"default\": 25,\n \"description\": \"Output frame rate. One of 25, 30, or 60.\",\n \"type\": \"number\"\n },\n \"include_gif\": {\n \"default\": false,\n \"description\": \"Whether to include a GIF preview in the webhook payload\",\n \"type\": \"boolean\"\n },\n \"keep_text_vertically_centered\": {\n \"default\": false,\n \"description\": \"When true, replaced text elements are vertically re-centered based on their rendered height\",\n \"type\": \"boolean\"\n },\n \"reorder_music\": {\n \"default\": true,\n \"description\": \"When true (default), background audio tracks move with their scenes. When false, tracks stay pinned to layout positions.\",\n \"type\": \"boolean\"\n },\n \"scene_ids\": {\n \"description\": \"Scene IDs to render, in order (repeats allowed). Scenes must already exist in the template; the API can select, reorder, and repeat scenes but cannot create new ones. Omit to render all scenes in template order.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"nullable\": true,\n \"type\": \"array\"\n },\n \"subtitles\": {\n \"description\": \"Subtitle style settings. Implies captions when provided.\",\n \"nullable\": true,\n \"properties\": {\n \"alignment\": {\n \"default\": 2,\n \"description\": \"Subtitle alignment\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"disable_highlight\": {\n \"default\": false,\n \"description\": \"Override the preset's word-highlight style\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"font_size\": {\n \"description\": \"Font size override for the preset\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"position\": {\n \"description\": \"Subtitle position override\",\n \"nullable\": true,\n \"properties\": {\n \"x\": {\n \"default\": 0,\n \"description\": \"Horizontal subtitle position\",\n \"type\": \"number\"\n },\n \"y\": {\n \"default\": 0,\n \"description\": \"Vertical subtitle position\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"preset_name\": {\n \"description\": \"Subtitle preset name, e.g. 'classic', 'bold', 'bright'\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"preset_name\"\n ],\n \"type\": \"object\"\n },\n \"title\": {\n \"description\": \"Title for the generated video\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"variables\": {\n \"description\": \"Template variable replacements, keyed by the variable name defined in the template. Supply every text variable you want filled: an omitted text variable is not substituted, so its literal `{{variable_name}}` placeholder remains in the text it is bound to, whether that is a spoken script or an on-screen text element. Omitting an image, video, audio, character or voice variable is safe and keeps the value the template already holds. The defaults returned by `GET /v3/templates/{template_id}` are the template's current values, not fallbacks applied at generation time.\",\n \"properties\": {},\n \"required\": [],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"variables\"\n ],\n \"type\": \"object\"\n}", ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"description\": \"Video resource returned by list and detail endpoints.\\n\\nIf ``output_language`` is present the video is a translated video;\\notherwise it is a generated video.\",\n \"properties\": {\n \"captioned_video_url\": {\n \"description\": \"Presigned URL to download the video file with captions burned in\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"completed_at\": {\n \"description\": \"Unix timestamp when video generation finished\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"created_at\": {\n \"description\": \"Unix timestamp of creation\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"duration\": {\n \"description\": \"Video duration in seconds\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"failure_code\": {\n \"description\": \"Machine-readable failure reason. Only present when status is failed.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"failure_message\": {\n \"description\": \"Human-readable failure description. Only present when status is failed.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"folder_id\": {\n \"description\": \"ID of containing folder\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"gif_url\": {\n \"description\": \"URL to animated GIF preview\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"id\": {\n \"description\": \"Unique video identifier\",\n \"type\": \"string\"\n },\n \"output_language\": {\n \"description\": \"BCP-47 output language code. Present only for translated videos.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"status\": {\n \"description\": \"Current video status\",\n \"enum\": [\n \"pending\",\n \"processing\",\n \"completed\",\n \"failed\"\n ],\n \"type\": \"string\"\n },\n \"subtitle_url\": {\n \"description\": \"Presigned URL to download the SRT subtitle file\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"thumbnail_url\": {\n \"description\": \"URL to video thumbnail image\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"Video title\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"video_page_url\": {\n \"description\": \"URL to the video page in the HeyGen app\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"video_url\": {\n \"description\": \"Presigned URL to download the video file\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\",\n \"status\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", Endpoint: "/v3/templates/{template_id}", Method: "POST", @@ -112,7 +112,7 @@ var TemplateGenerate = &command.Spec{ Name: "folder-id", Type: "string", Default: "", - Help: "Folder to place the generated video in", + Help: "Destination folder ID in the caller's workspace, for example one returned by POST /v3/folders. Omit, pass null, or pass an empty string to place the result at the workspace root. The id must name a folder that is not in the trash and that the caller can write to; an unknown id, a folder in another workspace, a project of another kind such as a brand kit, or a trashed folder is rejected with 404 before anything is created. Supported only for single-video creation; batch video items cannot set their own destination folder.", Required: false, Enum: nil, Min: nil, @@ -200,7 +200,7 @@ var TemplateGet = &command.Spec{ Name: "get", Summary: "Get Template", Description: "Returns template details including its variable schema (with current default values) and scenes. Variable defaults are returned in the same shape the generate request accepts, so a response can be edited and posted back. Only draft version 4 templates (the current editor format) are supported.", - ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"description\": \"Template detail including its variable schema and scenes.\",\n \"properties\": {\n \"aspect_ratio\": {\n \"description\": \"Template aspect ratio\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"created_at\": {\n \"description\": \"Unix timestamp of creation\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"id\": {\n \"description\": \"Unique template identifier\",\n \"type\": \"string\"\n },\n \"name\": {\n \"description\": \"Template name\",\n \"type\": \"string\"\n },\n \"scene_ids\": {\n \"description\": \"Scene IDs in template order\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"scenes\": {\n \"description\": \"Scenes defined in the template\",\n \"items\": {\n \"description\": \"A scene defined in the template.\",\n \"properties\": {\n \"scene_id\": {\n \"description\": \"Scene ID, usable in the generate request's scene_ids\",\n \"type\": \"string\"\n },\n \"script\": {\n \"description\": \"Scene script text, with variable placeholders unreplaced\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"variables\": {\n \"description\": \"Variables used in this scene\",\n \"items\": {\n \"description\": \"A variable used within a scene.\",\n \"properties\": {\n \"name\": {\n \"description\": \"Variable name, matching a key in the template's variables\",\n \"type\": \"string\"\n },\n \"variable_type\": {\n \"description\": \"Variable type: text, image, video, audio, character, or voice\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"name\",\n \"variable_type\"\n ],\n \"type\": \"object\"\n },\n \"type\": \"array\"\n }\n },\n \"required\": [\n \"scene_id\"\n ],\n \"type\": \"object\"\n },\n \"type\": \"array\"\n },\n \"thumbnail_url\": {\n \"description\": \"URL to the template thumbnail image\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"updated_at\": {\n \"description\": \"Unix timestamp of last update\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"variables\": {\n \"description\": \"Variables defined in the template with their current default values, keyed by variable name\",\n \"properties\": {},\n \"required\": [],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"id\",\n \"name\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", + ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"description\": \"Template detail: its variable schema, the July scene summary and, for authoring workspaces, its composition.\",\n \"properties\": {\n \"aspect_ratio\": {\n \"description\": \"Template aspect ratio\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"composition\": {\n \"description\": \"The template's scenes and cross-scene audio tracks in the shape GET /v3/videos/{video_id}/scenes uses, with each node carrying the variable types it can be bound to (bindable_as), the variable bound to it (variable) and the text variables in its {{placeholders}} (text_variables). Present only for workspaces with access to the template authoring API, and omitted when the template's draft cannot be read.\",\n \"nullable\": true,\n \"properties\": {\n \"background_audio\": {\n \"description\": \"Music and sound-effect tracks that play across scenes, with what an audio variable can replace.\",\n \"items\": {\n \"description\": \"A music or sound-effect track that plays across scenes; an audio variable replaces it.\",\n \"properties\": {\n \"bindable_as\": {\n \"description\": \"Variable types that can be bound to this node.\",\n \"items\": {\n \"enum\": [\n \"text\",\n \"image\",\n \"video\",\n \"audio\",\n \"character\",\n \"voice\"\n ],\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"id\": {\n \"description\": \"Opaque identifier of this track within the template. Stable for the life of the template.\",\n \"type\": \"string\"\n },\n \"text_variables\": {\n \"description\": \"Text variable names appearing as {{placeholders}} in this node's text, in order of first appearance.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"type\": {\n \"default\": \"audio\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the track. Present when the template records a link and, if that link is signed, it is not at or near its deadline. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"variable\": {\n \"description\": \"Name of the image, video, audio, character or voice variable bound to this node, if any.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n \"type\": \"array\"\n },\n \"scenes\": {\n \"description\": \"Scenes in template order, each in the video-scenes shape plus bindings.\",\n \"items\": {\n \"description\": \"One scene of the template, in template order, with what a variable can bind to in it.\",\n \"properties\": {\n \"background\": {\n \"description\": \"What fills the frame behind this scene's elements.\",\n \"discriminator\": {\n \"mapping\": {\n \"color\": \"#/components/schemas/ColorBackground\",\n \"image\": \"#/components/schemas/TemplateImageBackground\",\n \"video\": \"#/components/schemas/TemplateVideoBackground\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"A solid colour filling the frame behind the scene's elements.\",\n \"properties\": {\n \"color\": {\n \"description\": \"Hex colour, e.g. '#f6f6fc'.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"color\",\n \"description\": \"Background type discriminator.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"color\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A frame-filling image; an image variable replaces it.\",\n \"properties\": {\n \"bindable_as\": {\n \"description\": \"Variable types that can be bound to this node.\",\n \"items\": {\n \"enum\": [\n \"text\",\n \"image\",\n \"video\",\n \"audio\",\n \"character\",\n \"voice\"\n ],\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"text_variables\": {\n \"description\": \"Text variable names appearing as {{placeholders}} in this node's text, in order of first appearance.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"type\": {\n \"default\": \"image\",\n \"description\": \"Background type discriminator.\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the image. Present when the video records a link for this background and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"variable\": {\n \"description\": \"Name of the image, video, audio, character or voice variable bound to this node, if any.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A frame-filling clip; a video variable replaces it.\",\n \"properties\": {\n \"bindable_as\": {\n \"description\": \"Variable types that can be bound to this node.\",\n \"items\": {\n \"enum\": [\n \"text\",\n \"image\",\n \"video\",\n \"audio\",\n \"character\",\n \"voice\"\n ],\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"playback\": {\n \"description\": \"How the clip is reconciled to the scene. Same values as a video element's playback.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"speed_multiplier\": {\n \"description\": \"Playback speed multiplier. 2.0 = twice as fast.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"text_variables\": {\n \"description\": \"Text variable names appearing as {{placeholders}} in this node's text, in order of first appearance.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"trim\": {\n \"description\": \"Seconds trimmed from each end of the clip.\",\n \"nullable\": true,\n \"properties\": {\n \"end_offset\": {\n \"description\": \"Seconds trimmed from the end of the clip.\",\n \"type\": \"number\"\n },\n \"start_offset\": {\n \"description\": \"Seconds trimmed from the start of the clip.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"start_offset\",\n \"end_offset\"\n ],\n \"type\": \"object\"\n },\n \"type\": {\n \"default\": \"video\",\n \"description\": \"Background type discriminator.\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the clip. Present when the video records a link for this background and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"variable\": {\n \"description\": \"Name of the image, video, audio, character or voice variable bound to this node, if any.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"volume\": {\n \"description\": \"Clip audio volume. 1.0 = source level, 0.0 = silent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"elements\": {\n \"description\": \"Every element this scene places, in the template's own order.\",\n \"items\": {\n \"anyOf\": [\n {\n \"description\": \"An avatar element; a character variable replaces its look.\",\n \"properties\": {\n \"avatar_id\": {\n \"description\": \"The avatar look performing this scene.\",\n \"type\": \"string\"\n },\n \"bindable_as\": {\n \"description\": \"Variable types that can be bound to this node.\",\n \"items\": {\n \"enum\": [\n \"text\",\n \"image\",\n \"video\",\n \"audio\",\n \"character\",\n \"voice\"\n ],\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"engine\": {\n \"description\": \"Generation engine for this avatar. Absent when the video does not record one and none can be determined.\",\n \"enum\": [\n \"avatar_v\",\n \"avatar_iv\",\n \"avatar_iii\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"expressiveness\": {\n \"description\": \"Expressiveness level. Absent when left at the default.\",\n \"enum\": [\n \"high\",\n \"medium\",\n \"low\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"motion_prompt\": {\n \"description\": \"Motion description authored for this avatar.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"text_variables\": {\n \"description\": \"Text variable names appearing as {{placeholders}} in this node's text, in order of first appearance.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"type\": {\n \"default\": \"avatar\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n },\n \"variable\": {\n \"description\": \"Name of the image, video, audio, character or voice variable bound to this node, if any.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\",\n \"avatar_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"An image element; an image variable replaces its picture.\",\n \"properties\": {\n \"bindable_as\": {\n \"description\": \"Variable types that can be bound to this node.\",\n \"items\": {\n \"enum\": [\n \"text\",\n \"image\",\n \"video\",\n \"audio\",\n \"character\",\n \"voice\"\n ],\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"text_variables\": {\n \"description\": \"Text variable names appearing as {{placeholders}} in this node's text, in order of first appearance.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"type\": {\n \"default\": \"image\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the image. Present when the video records a link for this element and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed; the element is still reported either way, so an image with no URL is one this read could not link to rather than an element this API does not describe. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"variable\": {\n \"description\": \"Name of the image, video, audio, character or voice variable bound to this node, if any.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A video element; a video variable replaces its clip.\",\n \"properties\": {\n \"bindable_as\": {\n \"description\": \"Variable types that can be bound to this node.\",\n \"items\": {\n \"enum\": [\n \"text\",\n \"image\",\n \"video\",\n \"audio\",\n \"character\",\n \"voice\"\n ],\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"playback\": {\n \"description\": \"How the clip is reconciled to the scene: 'freeze' holds the last frame, 'loop' repeats it, 'fit_to_scene' changes speed to match, 'full_video' plays it whole and drives the scene's length.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"speed_multiplier\": {\n \"description\": \"Playback speed multiplier. 2.0 = twice as fast.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"text_variables\": {\n \"description\": \"Text variable names appearing as {{placeholders}} in this node's text, in order of first appearance.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"trim\": {\n \"description\": \"Seconds trimmed from each end of the clip.\",\n \"nullable\": true,\n \"properties\": {\n \"end_offset\": {\n \"description\": \"Seconds trimmed from the end of the clip.\",\n \"type\": \"number\"\n },\n \"start_offset\": {\n \"description\": \"Seconds trimmed from the start of the clip.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"start_offset\",\n \"end_offset\"\n ],\n \"type\": \"object\"\n },\n \"type\": {\n \"default\": \"video\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the clip. Present when the video records a link for this element and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"variable\": {\n \"description\": \"Name of the image, video, audio, character or voice variable bound to this node, if any.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"volume\": {\n \"description\": \"Clip audio volume. 1.0 = source level, 0.0 = silent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"An on-screen text element; text variables fill its {{placeholders}}.\",\n \"properties\": {\n \"bindable_as\": {\n \"description\": \"Variable types that can be bound to this node.\",\n \"items\": {\n \"enum\": [\n \"text\",\n \"image\",\n \"video\",\n \"audio\",\n \"character\",\n \"voice\"\n ],\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"text\": {\n \"description\": \"The text as displayed. Paragraphs of styled text are joined with newlines.\",\n \"type\": \"string\"\n },\n \"text_variables\": {\n \"description\": \"Text variable names appearing as {{placeholders}} in this node's text, in order of first appearance.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"type\": {\n \"default\": \"text\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n },\n \"variable\": {\n \"description\": \"Name of the image, video, audio, character or voice variable bound to this node, if any.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\",\n \"text\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"An element this version names but does not describe.\\n\\nExists so a scene's ``elements`` array is never a filtered view presented as a complete one: a\\ncaller iterating it sees every element the scene places. Carries exactly two fields: which\\nelement this is, and what kind of thing it is.\\n\\n**Absence of content is not by itself proof that an element is undescribed.** A described\\ncategory can serialize with nothing but an id and a type when every optional field happens to be\\nunavailable, and one case is reachable today: an image with no usable link carries no ``url``,\\nleaving ``{id, type}``. Read the ``type`` to tell them apart.\\n``avatar``, ``image``, ``video`` and ``text`` are described, so a bare one of those is a described\\nelement missing an optional value rather than an unexpanded node.\\n\\nThe ``id`` matters most here. Two masked elements in one scene are otherwise identical on the\\nwire, so without it a caller can see that the scene places two things and nothing else about\\neither.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"The element's category. An **open set**: new values may be added, so treat an unrecognized one as an element to skip rather than an error. Several kinds of element can share one value, so a category says what an element is rather than naming a single underlying kind. Every value is permanent except 'other', which means 'a kind of element this API does not yet categorise' and may be replaced by a more specific category later.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\",\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A group or mask; nothing binds to the container itself, its children carry their own bindings.\",\n \"properties\": {\n \"children\": {\n \"description\": \"The elements this container holds, each described exactly as a top-level element would be.\",\n \"items\": {\n \"anyOf\": [\n {\n \"description\": \"An avatar element; a character variable replaces its look.\",\n \"properties\": {\n \"avatar_id\": {\n \"description\": \"The avatar look performing this scene.\",\n \"type\": \"string\"\n },\n \"bindable_as\": {\n \"description\": \"Variable types that can be bound to this node.\",\n \"items\": {\n \"enum\": [\n \"text\",\n \"image\",\n \"video\",\n \"audio\",\n \"character\",\n \"voice\"\n ],\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"engine\": {\n \"description\": \"Generation engine for this avatar. Absent when the video does not record one and none can be determined.\",\n \"enum\": [\n \"avatar_v\",\n \"avatar_iv\",\n \"avatar_iii\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"expressiveness\": {\n \"description\": \"Expressiveness level. Absent when left at the default.\",\n \"enum\": [\n \"high\",\n \"medium\",\n \"low\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"motion_prompt\": {\n \"description\": \"Motion description authored for this avatar.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"text_variables\": {\n \"description\": \"Text variable names appearing as {{placeholders}} in this node's text, in order of first appearance.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"type\": {\n \"default\": \"avatar\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n },\n \"variable\": {\n \"description\": \"Name of the image, video, audio, character or voice variable bound to this node, if any.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\",\n \"avatar_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"An image element; an image variable replaces its picture.\",\n \"properties\": {\n \"bindable_as\": {\n \"description\": \"Variable types that can be bound to this node.\",\n \"items\": {\n \"enum\": [\n \"text\",\n \"image\",\n \"video\",\n \"audio\",\n \"character\",\n \"voice\"\n ],\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"text_variables\": {\n \"description\": \"Text variable names appearing as {{placeholders}} in this node's text, in order of first appearance.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"type\": {\n \"default\": \"image\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the image. Present when the video records a link for this element and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed; the element is still reported either way, so an image with no URL is one this read could not link to rather than an element this API does not describe. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"variable\": {\n \"description\": \"Name of the image, video, audio, character or voice variable bound to this node, if any.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A video element; a video variable replaces its clip.\",\n \"properties\": {\n \"bindable_as\": {\n \"description\": \"Variable types that can be bound to this node.\",\n \"items\": {\n \"enum\": [\n \"text\",\n \"image\",\n \"video\",\n \"audio\",\n \"character\",\n \"voice\"\n ],\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"playback\": {\n \"description\": \"How the clip is reconciled to the scene: 'freeze' holds the last frame, 'loop' repeats it, 'fit_to_scene' changes speed to match, 'full_video' plays it whole and drives the scene's length.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"speed_multiplier\": {\n \"description\": \"Playback speed multiplier. 2.0 = twice as fast.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"text_variables\": {\n \"description\": \"Text variable names appearing as {{placeholders}} in this node's text, in order of first appearance.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"trim\": {\n \"description\": \"Seconds trimmed from each end of the clip.\",\n \"nullable\": true,\n \"properties\": {\n \"end_offset\": {\n \"description\": \"Seconds trimmed from the end of the clip.\",\n \"type\": \"number\"\n },\n \"start_offset\": {\n \"description\": \"Seconds trimmed from the start of the clip.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"start_offset\",\n \"end_offset\"\n ],\n \"type\": \"object\"\n },\n \"type\": {\n \"default\": \"video\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the clip. Present when the video records a link for this element and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"variable\": {\n \"description\": \"Name of the image, video, audio, character or voice variable bound to this node, if any.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"volume\": {\n \"description\": \"Clip audio volume. 1.0 = source level, 0.0 = silent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"An on-screen text element; text variables fill its {{placeholders}}.\",\n \"properties\": {\n \"bindable_as\": {\n \"description\": \"Variable types that can be bound to this node.\",\n \"items\": {\n \"enum\": [\n \"text\",\n \"image\",\n \"video\",\n \"audio\",\n \"character\",\n \"voice\"\n ],\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"text\": {\n \"description\": \"The text as displayed. Paragraphs of styled text are joined with newlines.\",\n \"type\": \"string\"\n },\n \"text_variables\": {\n \"description\": \"Text variable names appearing as {{placeholders}} in this node's text, in order of first appearance.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"type\": {\n \"default\": \"text\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n },\n \"variable\": {\n \"description\": \"Name of the image, video, audio, character or voice variable bound to this node, if any.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\",\n \"text\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"An element this version names but does not describe.\\n\\nExists so a scene's ``elements`` array is never a filtered view presented as a complete one: a\\ncaller iterating it sees every element the scene places. Carries exactly two fields: which\\nelement this is, and what kind of thing it is.\\n\\n**Absence of content is not by itself proof that an element is undescribed.** A described\\ncategory can serialize with nothing but an id and a type when every optional field happens to be\\nunavailable, and one case is reachable today: an image with no usable link carries no ``url``,\\nleaving ``{id, type}``. Read the ``type`` to tell them apart.\\n``avatar``, ``image``, ``video`` and ``text`` are described, so a bare one of those is a described\\nelement missing an optional value rather than an unexpanded node.\\n\\nThe ``id`` matters most here. Two masked elements in one scene are otherwise identical on the\\nwire, so without it a caller can see that the scene places two things and nothing else about\\neither.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"The element's category. An **open set**: new values may be added, so treat an unrecognized one as an element to skip rather than an error. Several kinds of element can share one value, so a category says what an element is rather than naming a single underlying kind. Every value is permanent except 'other', which means 'a kind of element this API does not yet categorise' and may be replaced by a more specific category later.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\",\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Circular reference to TemplateContainerElement\"\n }\n ]\n },\n \"type\": \"array\"\n },\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"The container's category, e.g. 'group' or 'mask'.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\",\n \"type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": \"array\"\n },\n \"id\": {\n \"description\": \"Identifier of this scene within the video.\",\n \"type\": \"string\"\n },\n \"script\": {\n \"description\": \"This scene's audio sources in order, each one synthesized speech, uploaded audio, or a silence.\",\n \"items\": {\n \"anyOf\": [\n {\n \"description\": \"Synthesized speech; a voice variable replaces its voice, text variables fill its {{placeholders}}.\",\n \"properties\": {\n \"bindable_as\": {\n \"description\": \"Variable types that can be bound to this node.\",\n \"items\": {\n \"enum\": [\n \"text\",\n \"image\",\n \"video\",\n \"audio\",\n \"character\",\n \"voice\"\n ],\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"id\": {\n \"description\": \"Opaque identifier of this script entry within the video. Not to be parsed, sorted, or assumed to encode anything. The same id under two scenes means one entry is shared between them: its text belongs to both, and concatenating both scenes' scripts would synthesize the shared words twice.\",\n \"type\": \"string\"\n },\n \"text\": {\n \"description\": \"The script, verbatim, including inline markup. Empty when the scene was left unfinished.\",\n \"type\": \"string\"\n },\n \"text_variables\": {\n \"description\": \"Text variable names appearing as {{placeholders}} in this node's text, in order of first appearance.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"trim_to_speech\": {\n \"description\": \"Whether leading and trailing silence is trimmed to the spoken window.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"type\": {\n \"default\": \"text\",\n \"description\": \"Script entry type discriminator.\",\n \"type\": \"string\"\n },\n \"variable\": {\n \"description\": \"Name of the image, video, audio, character or voice variable bound to this node, if any.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"The voice delivering this script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning applied to this script.\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Present only when the video pins an engine this API exposes; absent when the engine is left for the server to pick.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3 or eleven_v4, stability may be any value in [0.0, 1.0].\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\",\n \"eleven_v4\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"description\": \"Pitch adjustment in semitones.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"speed\": {\n \"description\": \"Playback speed multiplier.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"volume\": {\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"id\",\n \"text\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Uploaded audio; an audio variable replaces the recording.\",\n \"properties\": {\n \"audio_url\": {\n \"description\": \"URL of the audio. Present when the video records a link for this entry and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"bindable_as\": {\n \"description\": \"Variable types that can be bound to this node.\",\n \"items\": {\n \"enum\": [\n \"text\",\n \"image\",\n \"video\",\n \"audio\",\n \"character\",\n \"voice\"\n ],\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"id\": {\n \"description\": \"Identifier of this script entry within the video.\",\n \"type\": \"string\"\n },\n \"text_variables\": {\n \"description\": \"Text variable names appearing as {{placeholders}} in this node's text, in order of first appearance.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"type\": {\n \"default\": \"audio\",\n \"description\": \"Script entry type discriminator.\",\n \"type\": \"string\"\n },\n \"variable\": {\n \"description\": \"Name of the image, video, audio, character or voice variable bound to this node, if any.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A held silence, with no words spoken over the scene.\",\n \"properties\": {\n \"duration\": {\n \"description\": \"How long the silence is held, in seconds. Absent where the number does not describe what renders — on a scene whose video clip plays in full, the clip's own length governs the scene.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"id\": {\n \"description\": \"Identifier of this script entry within the video.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"silence\",\n \"description\": \"Script entry type discriminator.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": \"array\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n \"type\": \"array\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"created_at\": {\n \"description\": \"Unix timestamp of creation\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"edit_version\": {\n \"description\": \"Opaque version of the template returned by this request. Every save of the template's draft or variable set changes it, whether through PUT /v3/templates/{template_id}/variables or the HeyGen editor; renaming or deleting the template does not. Send it back as expected_edit_version on PUT /v3/templates/{template_id}/variables to detect concurrent edits. Present only for workspaces with access to the template authoring API.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"id\": {\n \"description\": \"Unique template identifier\",\n \"type\": \"string\"\n },\n \"name\": {\n \"description\": \"Template name\",\n \"type\": \"string\"\n },\n \"scene_ids\": {\n \"description\": \"Scene IDs in template order, as POST /v3/templates/{template_id} accepts them\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"scenes\": {\n \"deprecated\": true,\n \"description\": \"Per-scene summary: the scene id, its script text and the variables it uses. Deprecated in favour of composition, which describes each scene in the same shape as GET /v3/videos/{video_id}/scenes; kept until a published sunset.\",\n \"items\": {\n \"description\": \"A scene defined in the template.\",\n \"properties\": {\n \"scene_id\": {\n \"description\": \"Scene ID, usable in the generate request's scene_ids\",\n \"type\": \"string\"\n },\n \"script\": {\n \"description\": \"Scene script text, with variable placeholders unreplaced\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"variables\": {\n \"description\": \"Variables used in this scene\",\n \"items\": {\n \"description\": \"A variable used within a scene.\",\n \"properties\": {\n \"name\": {\n \"description\": \"Variable name, matching a key in the template's variables\",\n \"type\": \"string\"\n },\n \"variable_type\": {\n \"description\": \"Variable type: text, image, video, audio, character, or voice\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"name\",\n \"variable_type\"\n ],\n \"type\": \"object\"\n },\n \"type\": \"array\"\n }\n },\n \"required\": [\n \"scene_id\"\n ],\n \"type\": \"object\"\n },\n \"type\": \"array\"\n },\n \"source_video_id\": {\n \"description\": \"Video the template was created from, when known. Present only for workspaces with access to the template authoring API.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"thumbnail_url\": {\n \"description\": \"URL to the template thumbnail image\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"updated_at\": {\n \"description\": \"Unix timestamp of last update\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"variables\": {\n \"description\": \"Variables defined in the template with their current default values, keyed by variable name\",\n \"properties\": {},\n \"required\": [],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"id\",\n \"name\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", Endpoint: "/v3/templates/{template_id}", Method: "GET", BodyEncoding: "", diff --git a/gen/video-agent.go b/gen/video-agent.go index 4da1450..2f88834 100644 --- a/gen/video-agent.go +++ b/gen/video-agent.go @@ -9,7 +9,7 @@ var VideoAgentCreate = &command.Spec{ Name: "create", Summary: "Create Video Agent Session", Description: "One-shot video generation from a prompt — agent handles scripting, avatar selection, scene composition, and rendering. Supports generate (fire-and-forget) and chat (multi-turn) modes.", - RequestSchema: "{\n \"description\": \"Request body for creating a video from a prompt using Video Agent v3.\\n\\nAll configuration is flat (no nested config object). Files use the\\ntype-discriminated AssetInput union for flexible asset inputs.\\n\\nSupports two modes:\\n- ``generate`` (default): one-shot — auto-proceeds through storyboard, produces one video.\\n- ``chat``: multi-turn — may pause for user input on real decisions (e.g. pick a voice),\\n auto-proceeds on confirmations. Allows revisions and follow-up videos.\",\n \"properties\": {\n \"avatar_id\": {\n \"description\": \"Specific avatar ID to use\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary ID controlling how custom terms are pronounced in the generated video's narration (for example, saying 'HeyGen' as 'hey-jen'). Pronunciation is applied to the synthesized audio only, so caption and subtitle text still show the original script wording. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"brand_kit_id\": {\n \"description\": \"Brand kit ID to apply brand colors, fonts, and logos to the generated video. In enterprise workspaces with a locked brand policy, only the workspace default brand kit is accepted; omit this field to apply the workspace default automatically.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"Optional callback ID included in webhook payload\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL for completion/failure notifications\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"files\": {\n \"description\": \"Optional file attachments (max 20 files)\",\n \"items\": {\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"nullable\": true,\n \"type\": \"array\"\n },\n \"incognito_mode\": {\n \"default\": false,\n \"description\": \"When enabled, disables memory injection and extraction for this session\",\n \"type\": \"boolean\"\n },\n \"mode\": {\n \"default\": \"generate\",\n \"description\": \"Session mode. 'generate' produces one video (fire-and-forget). 'chat' enables multi-turn interaction — the agent may pause for decisions and allows revisions.\",\n \"enum\": [\n \"generate\",\n \"chat\"\n ],\n \"type\": \"string\"\n },\n \"orientation\": {\n \"description\": \"Video orientation. If not provided, auto-detected from content.\",\n \"enum\": [\n \"landscape\",\n \"portrait\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"prompt\": {\n \"description\": \"The message/prompt for video generation (1-10000 characters)\",\n \"type\": \"string\"\n },\n \"style_id\": {\n \"description\": \"Style ID from GET /v3/video-agents/styles. Applies a curated visual template to the generated video.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"visibility\": {\n \"default\": \"team\",\n \"description\": \"Session visibility: private (owner only), team (workspace members), or public (anyone with the link). Defaults to team. Applies to the session and its conversation; does not change existing sessions.\",\n \"enum\": [\n \"private\",\n \"team\",\n \"public\"\n ],\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Specific voice ID to use for narration\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"prompt\"\n ],\n \"type\": \"object\"\n}", + RequestSchema: "{\n \"description\": \"Request body for creating a video from a prompt using Video Agent v3.\\n\\nAll configuration is flat (no nested config object). Files use the\\ntype-discriminated AssetInput union for flexible asset inputs.\\n\\nSupports two modes:\\n- ``generate`` (default): one-shot — auto-proceeds through storyboard, produces one video.\\n- ``chat``: multi-turn — may pause for user input on real decisions (e.g. pick a voice),\\n auto-proceeds on confirmations. Allows follow-up messages, targeted edits, and additional videos.\",\n \"properties\": {\n \"avatar_id\": {\n \"description\": \"Specific avatar ID to use\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary ID controlling how custom terms are pronounced in the generated video's narration (for example, saying 'HeyGen' as 'hey-jen'). Pronunciation is applied to the synthesized audio only, so caption and subtitle text still show the original script wording. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"brand_kit_id\": {\n \"description\": \"Brand kit ID to apply brand colors, fonts, and logos to the generated video. In enterprise workspaces with a locked brand policy, only the workspace default brand kit is accepted; omit this field to apply the workspace default automatically.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"Optional callback ID included in webhook payload\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL for completion/failure notifications\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"files\": {\n \"description\": \"Optional file attachments (max 20 files)\",\n \"items\": {\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"nullable\": true,\n \"type\": \"array\"\n },\n \"incognito_mode\": {\n \"default\": false,\n \"description\": \"When enabled, disables memory injection and extraction for this session\",\n \"type\": \"boolean\"\n },\n \"members\": {\n \"description\": \"Email addresses to grant access to. Valid only with `visibility: selected`, and every address must belong to a member of the caller's workspace (max 100). May be omitted or empty, leaving the session owner-only until members are added.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"nullable\": true,\n \"type\": \"array\"\n },\n \"mode\": {\n \"default\": \"generate\",\n \"description\": \"Session mode. 'generate' produces one video (fire-and-forget). 'chat' enables multi-turn interaction — the agent may pause for decisions and accept follow-up messages.\",\n \"enum\": [\n \"generate\",\n \"chat\"\n ],\n \"type\": \"string\"\n },\n \"orientation\": {\n \"description\": \"Video orientation. If not provided, auto-detected from content.\",\n \"enum\": [\n \"landscape\",\n \"portrait\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"prompt\": {\n \"description\": \"The message/prompt for video generation (1-10000 characters)\",\n \"type\": \"string\"\n },\n \"style_id\": {\n \"description\": \"Style ID from GET /v3/video-agents/styles. Applies a curated visual template to the generated video.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"visibility\": {\n \"default\": \"space\",\n \"description\": \"Who can open the session. `private`: the owner only. `space`: everyone in the workspace can open and drive the session. `selected`: the owner plus the workspace members listed in `members`. Defaults to `space`. Applies to the session and its conversation; does not change existing sessions.\",\n \"enum\": [\n \"private\",\n \"space\",\n \"selected\"\n ],\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Specific voice ID to use for narration\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"prompt\"\n ],\n \"type\": \"object\"\n}", ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"description\": \"Response from creating a video agent session.\",\n \"properties\": {\n \"created_at\": {\n \"description\": \"Unix timestamp of session creation\",\n \"type\": \"integer\"\n },\n \"session_id\": {\n \"description\": \"Session ID — primary identifier for this video agent session\",\n \"type\": \"string\"\n },\n \"status\": {\n \"description\": \"Session status\",\n \"enum\": [\n \"generating\",\n \"thinking\",\n \"completed\",\n \"failed\"\n ],\n \"type\": \"string\"\n },\n \"video_id\": {\n \"description\": \"Video ID for polling via GET /v3/videos/{video_id}, when available.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"session_id\",\n \"status\",\n \"created_at\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", Endpoint: "/v3/video-agents", Method: "POST", @@ -92,11 +92,23 @@ var VideoAgentCreate = &command.Spec{ Source: "body", JSONName: "incognito_mode", }, + { + Name: "members", + Type: "string-slice", + Default: "", + Help: "Email addresses to grant access to. Valid only with `visibility: selected`, and every address must belong to a member of the caller's workspace (max 100). May be omitted or empty, leaving the session owner-only until members are added.", + Required: false, + Enum: nil, + Min: nil, + Max: nil, + Source: "body", + JSONName: "members", + }, { Name: "mode", Type: "string", Default: "generate", - Help: "Session mode. 'generate' produces one video (fire-and-forget). 'chat' enables multi-turn interaction — the agent may pause for decisions and allows revisions.", + Help: "Session mode. 'generate' produces one video (fire-and-forget). 'chat' enables multi-turn interaction — the agent may pause for decisions and accept follow-up messages.", Required: false, Enum: []string{"generate", "chat"}, Min: nil, @@ -143,10 +155,10 @@ var VideoAgentCreate = &command.Spec{ { Name: "visibility", Type: "string", - Default: "team", - Help: "Session visibility: private (owner only), team (workspace members), or public (anyone with the link). Defaults to team. Applies to the session and its conversation; does not change existing sessions.", + Default: "space", + Help: "Who can open the session. `private`: the owner only. `space`: everyone in the workspace can open and drive the session. `selected`: the owner plus the workspace members listed in `members`. Defaults to `space`. Applies to the session and its conversation; does not change existing sessions.", Required: false, - Enum: []string{"private", "team", "public"}, + Enum: []string{"private", "space", "selected"}, Min: nil, Max: nil, Source: "body", @@ -172,7 +184,7 @@ var VideoAgentGet = &command.Spec{ Name: "get", Summary: "Get Video Agent Session", Description: "Returns the current status, progress, video_id, and recent chat messages for a session.", - ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"description\": \"Response from getting a video agent session.\",\n \"properties\": {\n \"created_at\": {\n \"description\": \"Unix timestamp of session creation\",\n \"type\": \"integer\"\n },\n \"messages\": {\n \"description\": \"Most recent visible messages (max 40, newest-first)\",\n \"items\": {\n \"description\": \"Simplified chat message for external consumers.\",\n \"properties\": {\n \"content\": {\n \"description\": \"Message text content\",\n \"type\": \"string\"\n },\n \"created_at\": {\n \"description\": \"Unix timestamp of message creation\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"resource_ids\": {\n \"description\": \"Resource IDs referenced in this message\",\n \"items\": {\n \"type\": \"string\"\n },\n \"nullable\": true,\n \"type\": \"array\"\n },\n \"role\": {\n \"description\": \"Message author: 'user' or 'model'\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Message type: text, resource, or error\",\n \"enum\": [\n \"text\",\n \"resource\",\n \"error\"\n ],\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"role\",\n \"content\",\n \"type\"\n ],\n \"type\": \"object\"\n },\n \"type\": \"array\"\n },\n \"progress\": {\n \"default\": 0,\n \"description\": \"Progress 0-100\",\n \"type\": \"integer\"\n },\n \"session_id\": {\n \"description\": \"Session ID\",\n \"type\": \"string\"\n },\n \"status\": {\n \"description\": \"Session status. If a generate session pauses for input before creating its reserved video, the session is waiting_for_input while the reserved video is failed.\",\n \"enum\": [\n \"thinking\",\n \"waiting_for_input\",\n \"reviewing\",\n \"generating\",\n \"completed\",\n \"failed\"\n ],\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"LLM-generated session title\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"video_id\": {\n \"description\": \"Video ID once generation starts\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"session_id\",\n \"status\",\n \"created_at\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", + ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"description\": \"Response from getting a video agent session.\",\n \"properties\": {\n \"created_at\": {\n \"description\": \"Unix timestamp of session creation\",\n \"type\": \"integer\"\n },\n \"error\": {\n \"description\": \"Error details when a failed session has a specific, caller-actionable reason (e.g. a billing paywall). Absent for generic failures with no surfaced reason.\",\n \"nullable\": true,\n \"properties\": {\n \"code\": {\n \"description\": \"Machine-readable error code.\",\n \"type\": \"string\"\n },\n \"message\": {\n \"description\": \"Human-readable error description.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"code\",\n \"message\"\n ],\n \"type\": \"object\"\n },\n \"messages\": {\n \"description\": \"Most recent visible messages (max 40, newest-first)\",\n \"items\": {\n \"description\": \"Simplified chat message for external consumers.\",\n \"properties\": {\n \"content\": {\n \"description\": \"Message text content\",\n \"type\": \"string\"\n },\n \"created_at\": {\n \"description\": \"Unix timestamp of message creation\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"resource_ids\": {\n \"description\": \"Resource IDs referenced in this message\",\n \"items\": {\n \"type\": \"string\"\n },\n \"nullable\": true,\n \"type\": \"array\"\n },\n \"role\": {\n \"description\": \"Message author: 'user' or 'model'\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Message type: text, resource, or error\",\n \"enum\": [\n \"text\",\n \"resource\",\n \"error\"\n ],\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"role\",\n \"content\",\n \"type\"\n ],\n \"type\": \"object\"\n },\n \"type\": \"array\"\n },\n \"progress\": {\n \"default\": 0,\n \"description\": \"Progress 0-100\",\n \"type\": \"integer\"\n },\n \"session_id\": {\n \"description\": \"Session ID\",\n \"type\": \"string\"\n },\n \"status\": {\n \"description\": \"Session status. If a generate session pauses for input before creating its reserved video, the session is waiting_for_input while the reserved video is failed.\",\n \"enum\": [\n \"thinking\",\n \"waiting_for_input\",\n \"reviewing\",\n \"generating\",\n \"completed\",\n \"failed\"\n ],\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"LLM-generated session title\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"video_id\": {\n \"description\": \"Video ID once generation starts\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"session_id\",\n \"status\",\n \"created_at\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", Endpoint: "/v3/video-agents/{session_id}", Method: "GET", BodyEncoding: "", @@ -246,10 +258,10 @@ var VideoAgentResourcesGet = &command.Spec{ var VideoAgentSend = &command.Spec{ Group: "video-agent", Name: "send", - Summary: "Send Message or Request Revision", - Description: "Sends a follow-up message to an existing session. Use to answer agent questions, add context, or request edits to a generated video. Only valid for sessions created in chat mode.", - RequestSchema: "{\n \"description\": \"Request body for sending a follow-up message, answering the agent's question,\\nor requesting edits and revisions to a previously generated video.\",\n \"properties\": {\n \"avatar_id\": {\n \"description\": \"Override avatar for this message\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"brand_kit_id\": {\n \"description\": \"Brand kit ID to apply for this message. In enterprise workspaces with a locked brand policy, only the workspace default brand kit is accepted.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"files\": {\n \"description\": \"Optional file attachments (max 20 files)\",\n \"items\": {\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"nullable\": true,\n \"type\": \"array\"\n },\n \"message\": {\n \"description\": \"Text message to the agent\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Override voice for this message\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"message\"\n ],\n \"type\": \"object\"\n}", - ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"description\": \"Response from sending a message to a session.\",\n \"properties\": {\n \"run_id\": {\n \"description\": \"Run ID for this message processing\",\n \"type\": \"string\"\n },\n \"session_id\": {\n \"description\": \"Session ID\",\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"LLM-generated session title\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"session_id\",\n \"run_id\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", + Summary: "Send Video Agent Message", + Description: "Submits a conversational or scene-edit turn to an existing session created in either `generate` or `chat` mode. Follow-up turns preserve the session mode: `generate` continues automatically, while `chat` may pause for user input. Provide `message`, `edit_plan`, or both. Send `message` to answer agent questions, add context, or request conversational changes. For deterministic scene-scoped changes, first call GET /v3/videos/{video_id}/scenes, then include that video's ID as `scene_snapshot_video_id`, its `edit_version`, and a public `scene_id` in each `edit_plan` item. Items may reference different scene snapshots from this session, including earlier videos; none identifies the output draft. The server resolves internal Video Agent resource and storyboard scene IDs; clients must not supply them. The complete edit plan is rejected before submission if any scene is invalid or the draft changed before acceptance. Edits are asynchronous: `edit_version` is not a document lock, so two requests created from the same version may both be accepted. A successful edit turn returns the current working draft in `video_id`; further edits reuse it until rendering starts, then the next edit receives a new draft ID. Poll GET /v3/videos/{video_id} directly for status.", + RequestSchema: "{\n \"anyOf\": [\n {\n \"properties\": {\n \"message\": {\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"message\"\n ],\n \"type\": \"object\"\n },\n {\n \"properties\": {\n \"edit_plan\": {\n \"type\": \"array\"\n }\n },\n \"required\": [\n \"edit_plan\"\n ],\n \"type\": \"object\"\n }\n ],\n \"description\": \"One conversational or scene-edit turn in an existing Video Agent session.\"\n}", + ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"description\": \"Response from submitting a turn to a session.\",\n \"properties\": {\n \"run_id\": {\n \"description\": \"Run ID for this message processing\",\n \"type\": \"string\"\n },\n \"session_id\": {\n \"description\": \"Session ID\",\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"LLM-generated session title\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"video_id\": {\n \"description\": \"Current working draft video ID for an edit turn; poll GET /v3/videos/{video_id} for status\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"session_id\",\n \"run_id\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", Endpoint: "/v3/video-agents/{session_id}", Method: "POST", BodyEncoding: "json", @@ -260,6 +272,18 @@ var VideoAgentSend = &command.Spec{ {Name: "session-id", Param: "session_id", Help: ""}, }, Flags: []command.FlagSpec{ + { + Name: "idempotency-key", + Type: "string", + Default: "", + Help: "Optional client-supplied key for safely retrying mutations. Subsequent calls within 24 hours that share this key replay the original response — even if the request body differs slightly (a warning is logged). A retry that arrives while the original is still in flight gets a 409 `request_in_progress`. Keys must be 1–255 characters from `[A-Za-z0-9_:.-]`; a UUID is a safe default. Scope is per-endpoint and per-resource: the same key on a different route or path parameter is independent. Example: 550e8400-e29b-41d4-a716-446655440000", + Required: false, + Enum: nil, + Min: nil, + Max: nil, + Source: "header", + JSONName: "Idempotency-Key", + }, { Name: "avatar-id", Type: "string", @@ -288,8 +312,8 @@ var VideoAgentSend = &command.Spec{ Name: "message", Type: "string", Default: "", - Help: "Text message to the agent", - Required: true, + Help: "Text message to the agent. Required when edit_plan is omitted; may be omitted or empty when edit_plan is provided.", + Required: false, Enum: nil, Min: nil, Max: nil, diff --git a/gen/video-translate.go b/gen/video-translate.go index 0a5970a..9a63821 100644 --- a/gen/video-translate.go +++ b/gen/video-translate.go @@ -9,7 +9,7 @@ var VideoTranslateBatchesCreate = &command.Spec{ Name: "batches create", Summary: "Create Video Translation Batch", Description: "Submit up to 100 video-translation payloads (identical in shape to POST /v3/video-translations) as a single batch. A payload targeting multiple output_languages expands to one batch item per language, and each item is created and processed independently so one bad source does not fail the rest. Returns 202 with a batch_id; poll GET /v3/video-translations/batches/{batch_id} for progress. Pass an Idempotency-Key header to make retries safe — the same key returns the same batch.", - RequestSchema: "{\n \"properties\": {\n \"callback_url\": {\n \"description\": \"Webhook URL invoked once when every item in the batch reaches a terminal state.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"Display name for the batch, shown in the HeyGen app.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"video_translations\": {\n \"description\": \"Video-translation payloads, identical in shape to POST /v3/video-translations. A single entry targeting multiple output_languages expands to one batch item per language; the expanded item count is capped at 100.\",\n \"items\": {\n \"description\": \"Request body for POST /v3/video-translations.\",\n \"properties\": {\n \"audio\": {\n \"description\": \"Custom audio for dubbing — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary ID for custom term translations (e.g. translate 'Reformer' as the Pilates equipment, not 'political activist'). Alias for the legacy `brand_voice_id` field. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"brand_voice_id\": {\n \"deprecated\": true,\n \"description\": \"Brand glossary ID for custom term translations. Legacy field name for `brand_glossary_id` — both are accepted and resolve to the same workspace record. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"ID included in webhook payload\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL for completion notifications\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"disable_music_track\": {\n \"default\": false,\n \"description\": \"Remove background music\",\n \"type\": \"boolean\"\n },\n \"enable_caption\": {\n \"default\": false,\n \"deprecated\": true,\n \"description\": \"Deprecated and ignored: captions are always generated; whether to display them is a download-side choice.\",\n \"type\": \"boolean\"\n },\n \"enable_dynamic_duration\": {\n \"default\": true,\n \"description\": \"Allow dynamic duration adjustment\",\n \"type\": \"boolean\"\n },\n \"enable_speech_enhancement\": {\n \"default\": false,\n \"description\": \"Enhance speech quality\",\n \"type\": \"boolean\"\n },\n \"enable_watermark\": {\n \"default\": false,\n \"description\": \"Add watermark to output\",\n \"type\": \"boolean\"\n },\n \"end_time\": {\n \"description\": \"End time in seconds for partial translation\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"folder_id\": {\n \"description\": \"Project/folder ID to organize translation into\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"fps_mode\": {\n \"description\": \"Frame rate mode for the output video. 'vfr' = variable frame rate, 'cfr' = constant frame rate, 'passthrough' = match the source. Only takes effect when a custom 'audio' track is provided.\",\n \"enum\": [\n \"vfr\",\n \"cfr\",\n \"passthrough\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"input_language\": {\n \"description\": \"Source language code (auto-detected if omitted)\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"keep_the_same_format\": {\n \"description\": \"Preserve the source video's encoding specs (resolution, bitrate).\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"mode\": {\n \"default\": \"speed\",\n \"description\": \"Translation quality mode: 'speed' (faster) or 'precision' (higher quality, uses avatar inference)\",\n \"enum\": [\n \"speed\",\n \"precision\"\n ],\n \"type\": \"string\"\n },\n \"output_languages\": {\n \"description\": \"Target language names (e.g. 'Chinese (Cantonese, Traditional)', 'Spanish (Spain)', 'English'). Use GET /v3/video-translations/languages for valid values. Use one for single translation, multiple for batch.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"speaker_num\": {\n \"description\": \"Number of speakers (improves speaker separation)\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"srt\": {\n \"description\": \"Custom subtitle file — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}.\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"srt_role\": {\n \"description\": \"Which video the subtitle applies to: 'input' (source) or 'output' (translated).\",\n \"enum\": [\n \"input\",\n \"output\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"start_time\": {\n \"description\": \"Start time in seconds for partial translation\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stock_voice_config\": {\n \"description\": \"Use a preset stock voice for the translation instead of recreating the original speaker's voice. By default, Video Translation clones the original speaker so the result sounds like them; with this enabled, the translation is spoken by a natural preset voice optimized for clear pronunciation and accent in the target language (the result will not sound like the original speaker). Enterprise feature, available for selected accounts and languages by request — contact your HeyGen account team.\",\n \"nullable\": true,\n \"properties\": {\n \"preferred_stock_voice_ids\": {\n \"description\": \"Optional. Pin specific stock voice IDs to draw from. If omitted, the target language's default stock-voice pool is used.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"use_stock_voice\": {\n \"default\": false,\n \"description\": \"Set to true to use a preset stock voice instead of cloning the original speaker.\",\n \"type\": \"boolean\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"title\": {\n \"description\": \"Title for the translation job\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"translate_audio_only\": {\n \"default\": false,\n \"description\": \"Only translate audio, keep original video\",\n \"type\": \"boolean\"\n },\n \"video\": {\n \"description\": \"Source video — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n }\n },\n \"required\": [\n \"video\",\n \"output_languages\"\n ],\n \"type\": \"object\"\n },\n \"type\": \"array\"\n }\n },\n \"required\": [\n \"video_translations\"\n ],\n \"type\": \"object\"\n}", + RequestSchema: "{\n \"properties\": {\n \"callback_url\": {\n \"description\": \"Webhook URL invoked once when every item in the batch reaches a terminal state.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"Display name for the batch, shown in the HeyGen app.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"video_translations\": {\n \"description\": \"Video-translation payloads, identical in shape to POST /v3/video-translations. A single entry targeting multiple output_languages expands to one batch item per language; the expanded item count is capped at 100.\",\n \"items\": {\n \"description\": \"Request body for POST /v3/video-translations.\",\n \"properties\": {\n \"audio\": {\n \"description\": \"Custom audio for dubbing — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary ID for custom term translations (e.g. translate 'Reformer' as the Pilates equipment, not 'political activist'). Alias for the legacy `brand_voice_id` field. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"brand_voice_id\": {\n \"deprecated\": true,\n \"description\": \"Brand glossary ID for custom term translations. Legacy field name for `brand_glossary_id` — both are accepted and resolve to the same workspace record. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"ID included in webhook payload\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL for completion notifications\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"disable_music_track\": {\n \"default\": false,\n \"description\": \"Remove background music\",\n \"type\": \"boolean\"\n },\n \"enable_caption\": {\n \"default\": false,\n \"deprecated\": true,\n \"description\": \"Deprecated and ignored: captions are always generated; whether to display them is a download-side choice.\",\n \"type\": \"boolean\"\n },\n \"enable_dynamic_duration\": {\n \"default\": true,\n \"description\": \"Allow dynamic duration adjustment\",\n \"type\": \"boolean\"\n },\n \"enable_speech_enhancement\": {\n \"default\": false,\n \"description\": \"Enhance speech quality\",\n \"type\": \"boolean\"\n },\n \"enable_watermark\": {\n \"default\": false,\n \"description\": \"Add watermark to output\",\n \"type\": \"boolean\"\n },\n \"end_time\": {\n \"description\": \"End time in seconds for partial translation\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"folder_id\": {\n \"description\": \"Destination folder ID in the caller's workspace, for example one returned by POST /v3/folders. Omit, pass null, or pass an empty string to place the result at the workspace root. The id must name a folder that is not in the trash and that the caller can write to; an unknown id, a folder in another workspace, a project of another kind such as a brand kit, or a trashed folder is rejected with 404 before anything is created.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"fps_mode\": {\n \"description\": \"Frame rate mode for the output video. 'vfr' = variable frame rate, 'cfr' = constant frame rate, 'passthrough' = match the source. Only takes effect when a custom 'audio' track is provided.\",\n \"enum\": [\n \"vfr\",\n \"cfr\",\n \"passthrough\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"input_language\": {\n \"description\": \"Source language code (auto-detected if omitted)\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"keep_the_same_format\": {\n \"description\": \"Preserve the source video's encoding specs (resolution, bitrate).\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"mode\": {\n \"default\": \"speed\",\n \"description\": \"Translation quality mode: 'speed' (faster) or 'precision' (higher quality, uses avatar inference)\",\n \"enum\": [\n \"speed\",\n \"precision\"\n ],\n \"type\": \"string\"\n },\n \"output_languages\": {\n \"description\": \"Target language names (e.g. 'Chinese (Cantonese, Traditional)', 'Spanish (Spain)', 'English'). Use GET /v3/video-translations/languages for valid values. Use one for single translation, multiple for batch.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"speaker_num\": {\n \"description\": \"Number of speakers (improves speaker separation)\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"srt\": {\n \"description\": \"Custom subtitle file — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}.\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"srt_role\": {\n \"description\": \"Which video the subtitle applies to: 'input' (source) or 'output' (translated).\",\n \"enum\": [\n \"input\",\n \"output\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"start_time\": {\n \"description\": \"Start time in seconds for partial translation\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stock_voice_config\": {\n \"description\": \"Use a preset stock voice for the translation instead of recreating the original speaker's voice. By default, Video Translation clones the original speaker so the result sounds like them; with this enabled, the translation is spoken by a natural preset voice optimized for clear pronunciation and accent in the target language (the result will not sound like the original speaker). Enterprise feature, available for selected accounts and languages by request — contact your HeyGen account team.\",\n \"nullable\": true,\n \"properties\": {\n \"preferred_stock_voice_ids\": {\n \"description\": \"Optional. Pin specific stock voice IDs to draw from. If omitted, the target language's default stock-voice pool is used.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"use_stock_voice\": {\n \"default\": false,\n \"description\": \"Set to true to use a preset stock voice instead of cloning the original speaker.\",\n \"type\": \"boolean\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"title\": {\n \"description\": \"Title for the translation job\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"translate_audio_only\": {\n \"default\": false,\n \"description\": \"Only translate audio, keep original video\",\n \"type\": \"boolean\"\n },\n \"video\": {\n \"description\": \"Source video — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n }\n },\n \"required\": [\n \"video\",\n \"output_languages\"\n ],\n \"type\": \"object\"\n },\n \"type\": \"array\"\n }\n },\n \"required\": [\n \"video_translations\"\n ],\n \"type\": \"object\"\n}", Endpoint: "/v3/video-translations/batches", Method: "POST", BodyEncoding: "json", @@ -106,7 +106,7 @@ var VideoTranslateCreate = &command.Spec{ Name: "create", Summary: "Create Video Translation", Description: "Translates a video into one or more target languages with voice cloning and lip-sync. Returns one video_translation_id per language. Use mode: 'speed' (default) for fast turnaround or 'precision' for higher lip-sync quality.", - RequestSchema: "{\n \"description\": \"Request body for POST /v3/video-translations.\",\n \"properties\": {\n \"audio\": {\n \"description\": \"Custom audio for dubbing — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary ID for custom term translations (e.g. translate 'Reformer' as the Pilates equipment, not 'political activist'). Alias for the legacy `brand_voice_id` field. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"brand_voice_id\": {\n \"deprecated\": true,\n \"description\": \"Brand glossary ID for custom term translations. Legacy field name for `brand_glossary_id` — both are accepted and resolve to the same workspace record. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"ID included in webhook payload\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL for completion notifications\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"disable_music_track\": {\n \"default\": false,\n \"description\": \"Remove background music\",\n \"type\": \"boolean\"\n },\n \"enable_caption\": {\n \"default\": false,\n \"deprecated\": true,\n \"description\": \"Deprecated and ignored: captions are always generated; whether to display them is a download-side choice.\",\n \"type\": \"boolean\"\n },\n \"enable_dynamic_duration\": {\n \"default\": true,\n \"description\": \"Allow dynamic duration adjustment\",\n \"type\": \"boolean\"\n },\n \"enable_speech_enhancement\": {\n \"default\": false,\n \"description\": \"Enhance speech quality\",\n \"type\": \"boolean\"\n },\n \"enable_watermark\": {\n \"default\": false,\n \"description\": \"Add watermark to output\",\n \"type\": \"boolean\"\n },\n \"end_time\": {\n \"description\": \"End time in seconds for partial translation\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"folder_id\": {\n \"description\": \"Project/folder ID to organize translation into\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"fps_mode\": {\n \"description\": \"Frame rate mode for the output video. 'vfr' = variable frame rate, 'cfr' = constant frame rate, 'passthrough' = match the source. Only takes effect when a custom 'audio' track is provided.\",\n \"enum\": [\n \"vfr\",\n \"cfr\",\n \"passthrough\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"input_language\": {\n \"description\": \"Source language code (auto-detected if omitted)\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"keep_the_same_format\": {\n \"description\": \"Preserve the source video's encoding specs (resolution, bitrate).\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"mode\": {\n \"default\": \"speed\",\n \"description\": \"Translation quality mode: 'speed' (faster) or 'precision' (higher quality, uses avatar inference)\",\n \"enum\": [\n \"speed\",\n \"precision\"\n ],\n \"type\": \"string\"\n },\n \"output_languages\": {\n \"description\": \"Target language names (e.g. 'Chinese (Cantonese, Traditional)', 'Spanish (Spain)', 'English'). Use GET /v3/video-translations/languages for valid values. Use one for single translation, multiple for batch.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"speaker_num\": {\n \"description\": \"Number of speakers (improves speaker separation)\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"srt\": {\n \"description\": \"Custom subtitle file — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}.\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"srt_role\": {\n \"description\": \"Which video the subtitle applies to: 'input' (source) or 'output' (translated).\",\n \"enum\": [\n \"input\",\n \"output\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"start_time\": {\n \"description\": \"Start time in seconds for partial translation\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stock_voice_config\": {\n \"description\": \"Use a preset stock voice for the translation instead of recreating the original speaker's voice. By default, Video Translation clones the original speaker so the result sounds like them; with this enabled, the translation is spoken by a natural preset voice optimized for clear pronunciation and accent in the target language (the result will not sound like the original speaker). Enterprise feature, available for selected accounts and languages by request — contact your HeyGen account team.\",\n \"nullable\": true,\n \"properties\": {\n \"preferred_stock_voice_ids\": {\n \"description\": \"Optional. Pin specific stock voice IDs to draw from. If omitted, the target language's default stock-voice pool is used.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"use_stock_voice\": {\n \"default\": false,\n \"description\": \"Set to true to use a preset stock voice instead of cloning the original speaker.\",\n \"type\": \"boolean\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"title\": {\n \"description\": \"Title for the translation job\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"translate_audio_only\": {\n \"default\": false,\n \"description\": \"Only translate audio, keep original video\",\n \"type\": \"boolean\"\n },\n \"video\": {\n \"description\": \"Source video — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n }\n },\n \"required\": [\n \"video\",\n \"output_languages\"\n ],\n \"type\": \"object\"\n}", + RequestSchema: "{\n \"description\": \"Request body for POST /v3/video-translations.\",\n \"properties\": {\n \"audio\": {\n \"description\": \"Custom audio for dubbing — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary ID for custom term translations (e.g. translate 'Reformer' as the Pilates equipment, not 'political activist'). Alias for the legacy `brand_voice_id` field. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"brand_voice_id\": {\n \"deprecated\": true,\n \"description\": \"Brand glossary ID for custom term translations. Legacy field name for `brand_glossary_id` — both are accepted and resolve to the same workspace record. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"ID included in webhook payload\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL for completion notifications\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"disable_music_track\": {\n \"default\": false,\n \"description\": \"Remove background music\",\n \"type\": \"boolean\"\n },\n \"enable_caption\": {\n \"default\": false,\n \"deprecated\": true,\n \"description\": \"Deprecated and ignored: captions are always generated; whether to display them is a download-side choice.\",\n \"type\": \"boolean\"\n },\n \"enable_dynamic_duration\": {\n \"default\": true,\n \"description\": \"Allow dynamic duration adjustment\",\n \"type\": \"boolean\"\n },\n \"enable_speech_enhancement\": {\n \"default\": false,\n \"description\": \"Enhance speech quality\",\n \"type\": \"boolean\"\n },\n \"enable_watermark\": {\n \"default\": false,\n \"description\": \"Add watermark to output\",\n \"type\": \"boolean\"\n },\n \"end_time\": {\n \"description\": \"End time in seconds for partial translation\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"folder_id\": {\n \"description\": \"Destination folder ID in the caller's workspace, for example one returned by POST /v3/folders. Omit, pass null, or pass an empty string to place the result at the workspace root. The id must name a folder that is not in the trash and that the caller can write to; an unknown id, a folder in another workspace, a project of another kind such as a brand kit, or a trashed folder is rejected with 404 before anything is created.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"fps_mode\": {\n \"description\": \"Frame rate mode for the output video. 'vfr' = variable frame rate, 'cfr' = constant frame rate, 'passthrough' = match the source. Only takes effect when a custom 'audio' track is provided.\",\n \"enum\": [\n \"vfr\",\n \"cfr\",\n \"passthrough\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"input_language\": {\n \"description\": \"Source language code (auto-detected if omitted)\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"keep_the_same_format\": {\n \"description\": \"Preserve the source video's encoding specs (resolution, bitrate).\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"mode\": {\n \"default\": \"speed\",\n \"description\": \"Translation quality mode: 'speed' (faster) or 'precision' (higher quality, uses avatar inference)\",\n \"enum\": [\n \"speed\",\n \"precision\"\n ],\n \"type\": \"string\"\n },\n \"output_languages\": {\n \"description\": \"Target language names (e.g. 'Chinese (Cantonese, Traditional)', 'Spanish (Spain)', 'English'). Use GET /v3/video-translations/languages for valid values. Use one for single translation, multiple for batch.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"speaker_num\": {\n \"description\": \"Number of speakers (improves speaker separation)\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"srt\": {\n \"description\": \"Custom subtitle file — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}.\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"srt_role\": {\n \"description\": \"Which video the subtitle applies to: 'input' (source) or 'output' (translated).\",\n \"enum\": [\n \"input\",\n \"output\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"start_time\": {\n \"description\": \"Start time in seconds for partial translation\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stock_voice_config\": {\n \"description\": \"Use a preset stock voice for the translation instead of recreating the original speaker's voice. By default, Video Translation clones the original speaker so the result sounds like them; with this enabled, the translation is spoken by a natural preset voice optimized for clear pronunciation and accent in the target language (the result will not sound like the original speaker). Enterprise feature, available for selected accounts and languages by request — contact your HeyGen account team.\",\n \"nullable\": true,\n \"properties\": {\n \"preferred_stock_voice_ids\": {\n \"description\": \"Optional. Pin specific stock voice IDs to draw from. If omitted, the target language's default stock-voice pool is used.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"use_stock_voice\": {\n \"default\": false,\n \"description\": \"Set to true to use a preset stock voice instead of cloning the original speaker.\",\n \"type\": \"boolean\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"title\": {\n \"description\": \"Title for the translation job\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"translate_audio_only\": {\n \"default\": false,\n \"description\": \"Only translate audio, keep original video\",\n \"type\": \"boolean\"\n },\n \"video\": {\n \"description\": \"Source video — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n }\n },\n \"required\": [\n \"video\",\n \"output_languages\"\n ],\n \"type\": \"object\"\n}", ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"description\": \"Response for POST /v3/video-translations.\",\n \"properties\": {\n \"video_translation_ids\": {\n \"description\": \"Video translation IDs, one per target language\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n }\n },\n \"required\": [\n \"video_translation_ids\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", Endpoint: "/v3/video-translations", Method: "POST", @@ -254,7 +254,7 @@ var VideoTranslateCreate = &command.Spec{ Name: "folder-id", Type: "string", Default: "", - Help: "Project/folder ID to organize translation into", + Help: "Destination folder ID in the caller's workspace, for example one returned by POST /v3/folders. Omit, pass null, or pass an empty string to place the result at the workspace root. The id must name a folder that is not in the trash and that the caller can write to; an unknown id, a folder in another workspace, a project of another kind such as a brand kit, or a trashed folder is rejected with 404 before anything is created.", Required: false, Enum: nil, Min: nil, @@ -480,7 +480,7 @@ var VideoTranslateProofreadsCreate = &command.Spec{ Name: "proofreads create", Summary: "Create Proofread Session", Description: "Creates a proofread session that extracts editable subtitles from a video before final rendering.", - RequestSchema: "{\n \"description\": \"Request body for POST /v3/video-translations/proofreads.\",\n \"properties\": {\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary ID for custom term translations (e.g. translate 'Reformer' as 'Pilates equipment', not 'political activist'). Alias for the legacy `brand_voice_id` field. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"brand_voice_id\": {\n \"deprecated\": true,\n \"description\": \"Brand glossary ID for custom term translations. Legacy field name for `brand_glossary_id` — both are accepted and resolve to the same workspace record. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"disable_music_track\": {\n \"default\": false,\n \"description\": \"Remove background music\",\n \"type\": \"boolean\"\n },\n \"enable_speech_enhancement\": {\n \"default\": false,\n \"description\": \"Enhance speech quality\",\n \"type\": \"boolean\"\n },\n \"enable_video_stretching\": {\n \"default\": false,\n \"description\": \"Allow dynamic duration adjustment\",\n \"type\": \"boolean\"\n },\n \"folder_id\": {\n \"description\": \"Project/folder ID to organize proofread into\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"keep_the_same_format\": {\n \"default\": false,\n \"description\": \"Preserve the source video's encoding specs (resolution, bitrate)\",\n \"type\": \"boolean\"\n },\n \"mode\": {\n \"default\": \"speed\",\n \"description\": \"Translation quality mode: 'speed' (faster) or 'precision' (higher quality)\",\n \"enum\": [\n \"speed\",\n \"precision\"\n ],\n \"type\": \"string\"\n },\n \"output_languages\": {\n \"description\": \"Target language codes. Use one for single proofread, multiple for batch.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"speaker_num\": {\n \"description\": \"Number of speakers (improves speaker separation)\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"srt\": {\n \"description\": \"Initial SRT file — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"title\": {\n \"description\": \"Title for the proofread job\",\n \"type\": \"string\"\n },\n \"video\": {\n \"description\": \"Source video — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n }\n },\n \"required\": [\n \"video\",\n \"output_languages\",\n \"title\"\n ],\n \"type\": \"object\"\n}", + RequestSchema: "{\n \"description\": \"Request body for POST /v3/video-translations/proofreads.\",\n \"properties\": {\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary ID for custom term translations (e.g. translate 'Reformer' as 'Pilates equipment', not 'political activist'). Alias for the legacy `brand_voice_id` field. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"brand_voice_id\": {\n \"deprecated\": true,\n \"description\": \"Brand glossary ID for custom term translations. Legacy field name for `brand_glossary_id` — both are accepted and resolve to the same workspace record. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"disable_music_track\": {\n \"default\": false,\n \"description\": \"Remove background music\",\n \"type\": \"boolean\"\n },\n \"enable_speech_enhancement\": {\n \"default\": false,\n \"description\": \"Enhance speech quality\",\n \"type\": \"boolean\"\n },\n \"enable_video_stretching\": {\n \"default\": false,\n \"description\": \"Allow dynamic duration adjustment\",\n \"type\": \"boolean\"\n },\n \"folder_id\": {\n \"description\": \"Destination folder ID in the caller's workspace, for example one returned by POST /v3/folders. Omit, pass null, or pass an empty string to place the result at the workspace root. The id must name a folder that is not in the trash and that the caller can write to; an unknown id, a folder in another workspace, a project of another kind such as a brand kit, or a trashed folder is rejected with 404 before anything is created.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"keep_the_same_format\": {\n \"default\": false,\n \"description\": \"Preserve the source video's encoding specs (resolution, bitrate)\",\n \"type\": \"boolean\"\n },\n \"mode\": {\n \"default\": \"speed\",\n \"description\": \"Translation quality mode: 'speed' (faster) or 'precision' (higher quality)\",\n \"enum\": [\n \"speed\",\n \"precision\"\n ],\n \"type\": \"string\"\n },\n \"output_languages\": {\n \"description\": \"Target language codes. Use one for single proofread, multiple for batch.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"speaker_num\": {\n \"description\": \"Number of speakers (improves speaker separation)\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"srt\": {\n \"description\": \"Initial SRT file — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"title\": {\n \"description\": \"Title for the proofread job\",\n \"type\": \"string\"\n },\n \"video\": {\n \"description\": \"Source video — provide as {type: 'url', url: '...'} or {type: 'asset_id', asset_id: '...'}\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n }\n ]\n }\n },\n \"required\": [\n \"video\",\n \"output_languages\",\n \"title\"\n ],\n \"type\": \"object\"\n}", ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"description\": \"Response for POST /v3/video-translations/proofreads.\",\n \"properties\": {\n \"proofread_ids\": {\n \"description\": \"Proofread IDs, one per target language\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"status\": {\n \"description\": \"Initial status (always processing)\",\n \"enum\": [\n \"processing\",\n \"completed\",\n \"failed\"\n ],\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"proofread_ids\",\n \"status\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", Endpoint: "/v3/video-translations/proofreads", Method: "POST", @@ -566,7 +566,7 @@ var VideoTranslateProofreadsCreate = &command.Spec{ Name: "folder-id", Type: "string", Default: "", - Help: "Project/folder ID to organize proofread into", + Help: "Destination folder ID in the caller's workspace, for example one returned by POST /v3/folders. Omit, pass null, or pass an empty string to place the result at the workspace root. The id must name a folder that is not in the trash and that the caller can write to; an unknown id, a folder in another workspace, a project of another kind such as a brand kit, or a trashed folder is rejected with 404 before anything is created.", Required: false, Enum: nil, Min: nil, diff --git a/gen/video.go b/gen/video.go index 94a7b6c..577b517 100644 --- a/gen/video.go +++ b/gen/video.go @@ -9,7 +9,7 @@ var VideoBatchesCreate = &command.Spec{ Name: "batches create", Summary: "Create Video Batch", Description: "Submit up to 100 video creation payloads in one request and return a batch id immediately. Videos are created asynchronously; poll GET /v3/videos/batches/{batch_id} for per-item video ids and statuses.", - RequestSchema: "{\n \"properties\": {\n \"callback_url\": {\n \"description\": \"Webhook URL invoked once when every item in the batch reaches a terminal state.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"folder_id\": {\n \"description\": \"Destination folder ID for the batch container. The videos remain grouped inside the newly created batch; the batch itself is placed in this folder. Omit, pass null, or pass an empty string to place the batch at the workspace root.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"Display name for the batch, shown in the HeyGen app.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"videos\": {\n \"description\": \"Video creation requests (avatar / image / cinematic_avatar). Set folder_id once on the batch request, not on individual videos. Max 100 per batch.\",\n \"items\": {\n \"discriminator\": {\n \"mapping\": {\n \"avatar\": \"#/components/schemas/CreateBatchVideoFromAvatar\",\n \"cinematic_avatar\": \"#/components/schemas/CreateBatchVideoFromCinematicAvatar\",\n \"image\": \"#/components/schemas/CreateBatchVideoFromImage\",\n \"studio\": \"#/components/schemas/CreateBatchVideoFromStudio\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Create a video from a HeyGen avatar (video or photo avatar).\\n\\nProvide an avatar_id to use a previously created avatar. Supports all\\navatar types: studio_avatar, digital_twin, and photo_avatar. Optionally\\nset ``engine`` to select Avatar V for eligible avatars; when omitted, the\\nserver defaults to Avatar IV.\",\n \"properties\": {\n \"aspect_ratio\": {\n \"default\": \"16:9\",\n \"description\": \"Output video aspect ratio. Supported values: '16:9', '9:16', '4:5', '5:4', '1:1', 'auto'. Defaults to '16:9'. 'auto' preserves the source's aspect ratio (avatar source frames or uploaded image), short-edge anchored to the requested resolution and capped at the tier's long edge. Falls back to '16:9' when source dimensions can't be read.\",\n \"enum\": [\n \"16:9\",\n \"9:16\",\n \"4:5\",\n \"5:4\",\n \"1:1\",\n \"auto\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_asset_id\": {\n \"description\": \"HeyGen asset ID of an uploaded audio file. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_url\": {\n \"description\": \"Public URL of an audio file to lip-sync. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"avatar_id\": {\n \"description\": \"HeyGen avatar ID (video avatar or photo avatar look ID).\",\n \"type\": \"string\"\n },\n \"background\": {\n \"description\": \"Background settings for the video.\",\n \"nullable\": true,\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID of the background image. Used when type is 'image'. Mutually exclusive with url.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Background type. 'color' uses a solid hex color; 'image' uses an image from url or asset_id.\",\n \"enum\": [\n \"color\",\n \"image\"\n ],\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the background image. Used when type is 'image'. Mutually exclusive with asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"value\": {\n \"description\": \"Hex color code (e.g. '#ff0000'). Required when type is 'color'.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary ID controlling how custom terms are pronounced in generated speech (for example, saying 'HeyGen' as 'hey-jen'). Applies when the audio is synthesized from `script`; a caller-supplied `audio_url` or `audio_asset_id` is unaffected. Pronunciation is applied to the synthesized audio only, so caption and subtitle text still show the original script wording. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"Caller-defined identifier echoed back in the webhook payload.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL to receive a POST notification when the video is ready.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"caption\": {\n \"description\": \"Caption generation settings. A sidecar subtitle file is always returned via subtitle_url; set 'style' to additionally burn captions into the rendered video.\",\n \"nullable\": true,\n \"properties\": {\n \"file_format\": {\n \"default\": \"srt\",\n \"description\": \"Output format for the sidecar caption file.\",\n \"enum\": [\n \"srt\"\n ],\n \"type\": \"string\"\n },\n \"style\": {\n \"description\": \"Visual style for burning captions into the rendered video. Omit for sidecar-only captions.\",\n \"enum\": [\n \"default\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"engine\": {\n \"description\": \"Engine configuration for video generation. Pass {\\\"type\\\": \\\"avatar_v\\\"} to enable cross-reference-driven animation for higher quality. Check supported_api_engines on the avatar look to confirm eligibility. Defaults to Avatar IV when omitted.\",\n \"discriminator\": {\n \"mapping\": {\n \"avatar_iii\": \"#/components/schemas/AvatarIIIEngineConfig\",\n \"avatar_iv\": \"#/components/schemas/AvatarIVEngineConfig\",\n \"avatar_v\": \"#/components/schemas/AvatarVEngineConfig\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Avatar V engine configuration with cross-reference-driven animation.\",\n \"properties\": {\n \"reference_look_id\": {\n \"description\": \"Optional look to use as the animation reference. When provided, it must be a `digital_twin` look accessible to your workspace and in the same avatar group as `avatar_id` (`studio_avatar` and `photo_avatar` looks are rejected). When omitted, video avatars self-reference and photo avatars select from their group's eligible candidates, preferring digital twins (ready first, then processing / upgrading), then curated public studio looks. A photo avatar whose group has no eligible reference renders directly from its image without one; motion_prompt is rejected in that case. A non-public `digital_twin` reference, whether provided or selected automatically, must also satisfy its group's consent requirements.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_v'. Check supported_api_engines on the avatar look to confirm eligibility.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Avatar IV engine configuration (default behavior).\",\n \"properties\": {\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_iv'.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Avatar III engine configuration.\\n\\nA single engine value that resolves to the right product by the avatar's\\nlook type (mirrors how ``avatar_iv`` already serves both photo and video\\navatars):\\n\\n- video avatar looks (``digital_twin``, ``studio_avatar``) -\\u003e Digital Twin\\n (supports 4K)\\n- ``photo_avatar`` look -\\u003e Photo Avatar (no 4K output)\\n\\nNot supported for raw image input (``type: \\\"image\\\"``).\\n``motion_prompt`` and ``expressiveness`` are not supported with this engine.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_iii'. Resolves to Digital Twin for video avatar looks (digital_twin, studio_avatar) and Photo Avatar for photo_avatar looks; not supported for raw image input. Check supported_api_engines on the avatar look to confirm eligibility.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"expressiveness\": {\n \"description\": \"Avatar expressiveness level. Photo avatars only. Defaults to 'low' when omitted. Avatar IV only; rejected when engine.type is 'avatar_v'.\",\n \"enum\": [\n \"high\",\n \"medium\",\n \"low\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"fit\": {\n \"description\": \"How the subject is fitted to the output canvas. 'cover' scales to fill the frame (may crop edges). 'contain' scales to fit entirely within the frame (may show background). When omitted, the server picks the best option based on the source and canvas orientations.\",\n \"enum\": [\n \"contain\",\n \"cover\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"motion_prompt\": {\n \"description\": \"Natural-language prompt controlling avatar body motion and hand gestures. Supported for photo avatars on either engine, and for video avatars when engine.type is 'avatar_v'. Rejected for video avatars on the default Avatar IV engine, and for photo avatars on 'avatar_v' when the avatar's group has no animation reference (no digital twin or curated reference look).\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"output_format\": {\n \"default\": \"mp4\",\n \"description\": \"Output container. 'webm' returns a video with a transparent background (alpha channel); 'mp4' (default) returns a standard video. 'webm' requires an avatar that supports matting. When 'webm' is selected, any 'background' value is rejected and background removal is applied automatically — the caller does not need to set 'remove_background'.\",\n \"enum\": [\n \"mp4\",\n \"webm\"\n ],\n \"type\": \"string\"\n },\n \"remove_background\": {\n \"description\": \"Remove the avatar background. Video avatars must be trained with matting enabled.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"resolution\": {\n \"description\": \"Output video resolution. Avatar IV and Avatar V render the avatar at up to 1080p: with `4k`, the avatar is composited onto a 4K canvas rather than rendered natively. Native 4K output is available for Avatar III digital twins and studio avatars.\",\n \"enum\": [\n \"4k\",\n \"1080p\",\n \"720p\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"script\": {\n \"description\": \"Text script for the avatar to speak. Pair with voice_id, or omit voice_id when using avatar_id to use the avatar's default voice. Mutually exclusive with audio_url/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"Display title for the video in the HeyGen dashboard.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Must be 'avatar' for avatar-based video creation.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID for text-to-speech. Required when script is provided, unless avatar_id is set (the avatar's default voice is used as fallback).\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning parameters (speed, pitch, locale).\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Use the variant matching the engine backing the chosen voice (e.g. engine_type='elevenlabs' for ElevenLabs-backed voices). The request is rejected if the voice_id is not compatible with the selected engine.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3, stability must be 0, 0.5, or 1.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"default\": 0,\n \"description\": \"Pitch adjustment in semitones. -50 to +50.\",\n \"type\": \"number\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Playback speed multiplier. 0.5 (half speed) to 1.5 (1.5x speed).\",\n \"type\": \"number\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent. Useful when mixing spoken voice with background audio.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"watermark\": {\n \"description\": \"Custom watermark image to overlay on the video (PNG or JPEG). Available as a premium option for select Enterprise customers. To request access, please contact our support team.\",\n \"nullable\": true,\n \"properties\": {\n \"image\": {\n \"description\": \"Image asset to use as the watermark overlay (PNG or JPEG).\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"opacity\": {\n \"default\": 1,\n \"description\": \"Watermark opacity. 0.0 is fully transparent, 1.0 is fully opaque.\",\n \"type\": \"number\"\n },\n \"placement\": {\n \"description\": \"Watermark placement. Defaults to bottom-right with standard margins when omitted.\",\n \"nullable\": true,\n \"properties\": {\n \"offset_x\": {\n \"description\": \"Fine-tune horizontal position. Fraction of frame width; 0.05 shifts 5% rightward, -0.05 shifts 5% leftward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"offset_y\": {\n \"description\": \"Fine-tune vertical position. Fraction of frame height; 0.05 shifts 5% downward, -0.05 shifts 5% upward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"position\": {\n \"default\": \"bottom_right\",\n \"description\": \"Anchor corner for the watermark.\",\n \"enum\": [\n \"top_left\",\n \"top_right\",\n \"bottom_left\",\n \"bottom_right\"\n ],\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"scale\": {\n \"default\": 1,\n \"description\": \"Scale multiplier for the watermark image. 1.0 renders at native size.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"image\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"avatar_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Create a video by animating an arbitrary image.\\n\\nProvide an image via URL, asset ID, or inline base64. The image will be\\nanimated with lip-sync to the provided audio or generated speech.\",\n \"properties\": {\n \"aspect_ratio\": {\n \"default\": \"16:9\",\n \"description\": \"Output video aspect ratio. Supported values: '16:9', '9:16', '4:5', '5:4', '1:1', 'auto'. Defaults to '16:9'. 'auto' preserves the source's aspect ratio (avatar source frames or uploaded image), short-edge anchored to the requested resolution and capped at the tier's long edge. Falls back to '16:9' when source dimensions can't be read.\",\n \"enum\": [\n \"16:9\",\n \"9:16\",\n \"4:5\",\n \"5:4\",\n \"1:1\",\n \"auto\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_asset_id\": {\n \"description\": \"HeyGen asset ID of an uploaded audio file. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_url\": {\n \"description\": \"Public URL of an audio file to lip-sync. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"background\": {\n \"description\": \"Background settings for the video.\",\n \"nullable\": true,\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID of the background image. Used when type is 'image'. Mutually exclusive with url.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Background type. 'color' uses a solid hex color; 'image' uses an image from url or asset_id.\",\n \"enum\": [\n \"color\",\n \"image\"\n ],\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the background image. Used when type is 'image'. Mutually exclusive with asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"value\": {\n \"description\": \"Hex color code (e.g. '#ff0000'). Required when type is 'color'.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary ID controlling how custom terms are pronounced in generated speech (for example, saying 'HeyGen' as 'hey-jen'). Applies when the audio is synthesized from `script`; a caller-supplied `audio_url` or `audio_asset_id` is unaffected. Pronunciation is applied to the synthesized audio only, so caption and subtitle text still show the original script wording. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"Caller-defined identifier echoed back in the webhook payload.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL to receive a POST notification when the video is ready.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"caption\": {\n \"description\": \"Caption generation settings. A sidecar subtitle file is always returned via subtitle_url; set 'style' to additionally burn captions into the rendered video.\",\n \"nullable\": true,\n \"properties\": {\n \"file_format\": {\n \"default\": \"srt\",\n \"description\": \"Output format for the sidecar caption file.\",\n \"enum\": [\n \"srt\"\n ],\n \"type\": \"string\"\n },\n \"style\": {\n \"description\": \"Visual style for burning captions into the rendered video. Omit for sidecar-only captions.\",\n \"enum\": [\n \"default\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"expressiveness\": {\n \"description\": \"Avatar expressiveness level. Photo avatars only. Defaults to 'low' when omitted.\",\n \"enum\": [\n \"high\",\n \"medium\",\n \"low\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"fit\": {\n \"description\": \"How the subject is fitted to the output canvas. 'cover' scales to fill the frame (may crop edges). 'contain' scales to fit entirely within the frame (may show background). When omitted, the server picks the best option based on the source and canvas orientations.\",\n \"enum\": [\n \"contain\",\n \"cover\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"image\": {\n \"description\": \"Image to animate. Accepts URL, asset ID, or base64-encoded data.\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"motion_prompt\": {\n \"description\": \"Natural-language prompt controlling avatar body motion. Photo avatars only.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"output_format\": {\n \"default\": \"mp4\",\n \"description\": \"Output container. 'webm' returns a video with a transparent background (alpha channel); 'mp4' (default) returns a standard video. 'webm' requires an avatar that supports matting. When 'webm' is selected, any 'background' value is rejected and background removal is applied automatically — the caller does not need to set 'remove_background'.\",\n \"enum\": [\n \"mp4\",\n \"webm\"\n ],\n \"type\": \"string\"\n },\n \"remove_background\": {\n \"description\": \"Remove the avatar background. Video avatars must be trained with matting enabled.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"resolution\": {\n \"description\": \"Output video resolution. Avatar IV and Avatar V render the avatar at up to 1080p: with `4k`, the avatar is composited onto a 4K canvas rather than rendered natively. Native 4K output is available for Avatar III digital twins and studio avatars.\",\n \"enum\": [\n \"4k\",\n \"1080p\",\n \"720p\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"script\": {\n \"description\": \"Text script for the avatar to speak. Pair with voice_id, or omit voice_id when using avatar_id to use the avatar's default voice. Mutually exclusive with audio_url/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"Display title for the video in the HeyGen dashboard.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Must be 'image' for image-based video creation.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID for text-to-speech. Required when script is provided, unless avatar_id is set (the avatar's default voice is used as fallback).\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning parameters (speed, pitch, locale).\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Use the variant matching the engine backing the chosen voice (e.g. engine_type='elevenlabs' for ElevenLabs-backed voices). The request is rejected if the voice_id is not compatible with the selected engine.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3, stability must be 0, 0.5, or 1.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"default\": 0,\n \"description\": \"Pitch adjustment in semitones. -50 to +50.\",\n \"type\": \"number\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Playback speed multiplier. 0.5 (half speed) to 1.5 (1.5x speed).\",\n \"type\": \"number\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent. Useful when mixing spoken voice with background audio.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"watermark\": {\n \"description\": \"Custom watermark image to overlay on the video (PNG or JPEG). Available as a premium option for select Enterprise customers. To request access, please contact our support team.\",\n \"nullable\": true,\n \"properties\": {\n \"image\": {\n \"description\": \"Image asset to use as the watermark overlay (PNG or JPEG).\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"opacity\": {\n \"default\": 1,\n \"description\": \"Watermark opacity. 0.0 is fully transparent, 1.0 is fully opaque.\",\n \"type\": \"number\"\n },\n \"placement\": {\n \"description\": \"Watermark placement. Defaults to bottom-right with standard margins when omitted.\",\n \"nullable\": true,\n \"properties\": {\n \"offset_x\": {\n \"description\": \"Fine-tune horizontal position. Fraction of frame width; 0.05 shifts 5% rightward, -0.05 shifts 5% leftward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"offset_y\": {\n \"description\": \"Fine-tune vertical position. Fraction of frame height; 0.05 shifts 5% downward, -0.05 shifts 5% upward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"position\": {\n \"default\": \"bottom_right\",\n \"description\": \"Anchor corner for the watermark.\",\n \"enum\": [\n \"top_left\",\n \"top_right\",\n \"bottom_left\",\n \"bottom_right\"\n ],\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"scale\": {\n \"default\": 1,\n \"description\": \"Scale multiplier for the watermark image. 1.0 renders at native size.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"image\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"image\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Create a video from a text prompt plus avatar and asset references (Cinematic Avatar).\\n\\nCinematic Avatar generates a video from a natural-language ``prompt`` guided by\\nreference content: one to three avatar looks and optional reference assets\\n(images / videos / audio). Unlike the ``avatar`` and ``image`` modes there is\\nno script or voice — motion and speech are driven entirely by the prompt and\\nthe supplied references. Backed by the Seedance generation pipeline.\",\n \"properties\": {\n \"aspect_ratio\": {\n \"default\": \"16:9\",\n \"description\": \"Output aspect ratio. Supported for cinematic_avatar: '16:9', '9:16', '1:1'. Defaults to '16:9'.\",\n \"enum\": [\n \"16:9\",\n \"9:16\",\n \"1:1\"\n ],\n \"type\": \"string\"\n },\n \"auto_duration\": {\n \"default\": false,\n \"description\": \"Let the model choose the video length. When true, omit duration.\",\n \"type\": \"boolean\"\n },\n \"avatar_id\": {\n \"description\": \"Avatar look ID(s) used as visual references. Provide 1 to 3 look IDs.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"duration\": {\n \"description\": \"Video length in seconds (4–15). Defaults to 10. Omit when auto_duration is true.\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"enhance_prompt\": {\n \"default\": false,\n \"description\": \"Enable server-side prompt enhancement.\",\n \"type\": \"boolean\"\n },\n \"prompt\": {\n \"description\": \"Natural-language prompt describing the video to generate.\",\n \"type\": \"string\"\n },\n \"references\": {\n \"description\": \"Reference assets (images, videos, or audio) guiding the generation. Each accepts a URL, an asset_id, or inline base64. Combined limits: at most 3 videos and 9 images across avatars and references.\",\n \"items\": {\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"nullable\": true,\n \"type\": \"array\"\n },\n \"resolution\": {\n \"default\": \"720p\",\n \"description\": \"Output resolution. Supported for cinematic_avatar: '720p', '1080p'. Defaults to '720p'.\",\n \"enum\": [\n \"720p\",\n \"1080p\"\n ],\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"Display title for the video in the HeyGen dashboard.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Must be 'cinematic_avatar' for prompt-and-reference video creation.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"prompt\",\n \"avatar_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Create a single video by composing an ordered list of whole-frame scenes.\\n\\nThe server owns layout and center-crops each scene to the global output\\ncanvas. Output settings are global (one per request); a single video_id is\\nreturned and rendering is all-or-nothing. MP4 only in v1 — the output\\ncontainer is fixed and ``output_format`` is not exposed.\",\n \"properties\": {\n \"aspect_ratio\": {\n \"default\": \"16:9\",\n \"description\": \"Global output aspect ratio. Supported values: '16:9', '9:16', '4:5', '5:4', '1:1', 'auto'. Defaults to '16:9'. Each scene is center-cropped to this canvas.\",\n \"enum\": [\n \"16:9\",\n \"9:16\",\n \"4:5\",\n \"5:4\",\n \"1:1\",\n \"auto\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary ID controlling how custom terms are pronounced in generated speech (for example, saying 'HeyGen' as 'hey-jen'). Applies to every scene whose audio is synthesized from a script; scenes that supply their own audio URL or audio asset are unaffected. Pronunciation is applied to the synthesized audio only, so caption and subtitle text still show the original script wording. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"Caller-defined identifier echoed back in the webhook payload.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL to receive a POST notification when the video is ready.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"caption\": {\n \"description\": \"Caption generation settings. A sidecar subtitle file is always returned via subtitle_url; set 'style' to additionally burn captions into the rendered video.\",\n \"nullable\": true,\n \"properties\": {\n \"file_format\": {\n \"default\": \"srt\",\n \"description\": \"Output format for the sidecar caption file.\",\n \"enum\": [\n \"srt\"\n ],\n \"type\": \"string\"\n },\n \"style\": {\n \"description\": \"Visual style for burning captions into the rendered video. Omit for sidecar-only captions.\",\n \"enum\": [\n \"default\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"resolution\": {\n \"description\": \"Global output video resolution. Avatar IV and Avatar V render the avatar at up to 1080p: with `4k`, the avatar is composited onto a 4K canvas rather than rendered natively. Native 4K output is available for Avatar III digital twins and studio avatars.\",\n \"enum\": [\n \"4k\",\n \"1080p\",\n \"720p\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"scenes\": {\n \"description\": \"Ordered list of whole-frame scenes to concatenate. Each scene is one of 'avatar_video', 'image', or 'video'. Must contain 1 to 50 scenes.\",\n \"items\": {\n \"discriminator\": {\n \"mapping\": {\n \"avatar_video\": \"#/components/schemas/AvatarVideoScene\",\n \"image\": \"#/components/schemas/ImageScene\",\n \"video\": \"#/components/schemas/VideoScene\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"A whole-frame speaking scene backed by an avatar.\",\n \"properties\": {\n \"input\": {\n \"description\": \"Scene source ('type': 'avatar'): an avatar_id plus one audio source. The avatar_id accepts any avatar look — video avatars and photo avatars alike (pass a photo avatar's look id to get a talking photo). The scene duration is derived server-side from the audio.\",\n \"properties\": {\n \"audio_asset_id\": {\n \"description\": \"HeyGen asset ID of an uploaded audio file. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_url\": {\n \"description\": \"Public URL of an audio file to lip-sync. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"avatar_id\": {\n \"description\": \"HeyGen avatar ID (video avatar or photo avatar look ID).\",\n \"type\": \"string\"\n },\n \"background\": {\n \"description\": \"Optional scene background composited behind the avatar. Color-only in v1: pass {\\\"type\\\": \\\"color\\\", \\\"color\\\": \\\"#RRGGBB\\\"}. Other background types are not yet supported.\",\n \"discriminator\": {\n \"mapping\": {\n \"color\": \"#/components/schemas/StudioColorBackgroundInput\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Solid-color background for an ``avatar_video`` studio scene.\\n\\nStudio v1 supports solid-color backgrounds.\",\n \"properties\": {\n \"color\": {\n \"description\": \"Background color as a 6-digit hex string, e.g. '#1a2b3c'.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Background type discriminator. Must be 'color'.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"color\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"engine\": {\n \"description\": \"Engine configuration for video generation. Pass {\\\"type\\\": \\\"avatar_v\\\"} to enable cross-reference-driven animation for higher quality. Check supported_api_engines on the avatar look to confirm eligibility. Defaults to Avatar IV when omitted.\",\n \"discriminator\": {\n \"mapping\": {\n \"avatar_iii\": \"#/components/schemas/AvatarIIIEngineConfig\",\n \"avatar_iv\": \"#/components/schemas/AvatarIVEngineConfig\",\n \"avatar_v\": \"#/components/schemas/AvatarVEngineConfig\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Avatar V engine configuration with cross-reference-driven animation.\",\n \"properties\": {\n \"reference_look_id\": {\n \"description\": \"Optional look to use as the animation reference. When provided, it must be a `digital_twin` look accessible to your workspace and in the same avatar group as `avatar_id` (`studio_avatar` and `photo_avatar` looks are rejected). When omitted, video avatars self-reference and photo avatars select from their group's eligible candidates, preferring digital twins (ready first, then processing / upgrading), then curated public studio looks. A photo avatar whose group has no eligible reference renders directly from its image without one; motion_prompt is rejected in that case. A non-public `digital_twin` reference, whether provided or selected automatically, must also satisfy its group's consent requirements.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_v'. Check supported_api_engines on the avatar look to confirm eligibility.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Avatar IV engine configuration (default behavior).\",\n \"properties\": {\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_iv'.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Avatar III engine configuration.\\n\\nA single engine value that resolves to the right product by the avatar's\\nlook type (mirrors how ``avatar_iv`` already serves both photo and video\\navatars):\\n\\n- video avatar looks (``digital_twin``, ``studio_avatar``) -\\u003e Digital Twin\\n (supports 4K)\\n- ``photo_avatar`` look -\\u003e Photo Avatar (no 4K output)\\n\\nNot supported for raw image input (``type: \\\"image\\\"``).\\n``motion_prompt`` and ``expressiveness`` are not supported with this engine.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_iii'. Resolves to Digital Twin for video avatar looks (digital_twin, studio_avatar) and Photo Avatar for photo_avatar looks; not supported for raw image input. Check supported_api_engines on the avatar look to confirm eligibility.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"expressiveness\": {\n \"description\": \"Avatar expressiveness level. Photo avatars only. Defaults to 'low' when omitted. Avatar IV only; rejected when engine.type is 'avatar_v'.\",\n \"enum\": [\n \"high\",\n \"medium\",\n \"low\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"motion_prompt\": {\n \"description\": \"Natural-language prompt controlling avatar body motion and hand gestures. Supported for photo avatars on either engine, and for video avatars when engine.type is 'avatar_v'. Rejected for video avatars on the default Avatar IV engine, and for photo avatars on 'avatar_v' when the avatar's group has no animation reference (no digital twin or curated reference look).\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"script\": {\n \"description\": \"Text script for the avatar to speak. Pair with voice_id, or omit voice_id when using avatar_id to use the avatar's default voice. Mutually exclusive with audio_url/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Must be 'avatar' for an avatar-driven scene source.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID for text-to-speech. Required when script is provided, unless avatar_id is set (the avatar's default voice is used as fallback).\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning parameters (speed, pitch, locale).\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Use the variant matching the engine backing the chosen voice (e.g. engine_type='elevenlabs' for ElevenLabs-backed voices). The request is rejected if the voice_id is not compatible with the selected engine.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3, stability must be 0, 0.5, or 1.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"default\": 0,\n \"description\": \"Pitch adjustment in semitones. -50 to +50.\",\n \"type\": \"number\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Playback speed multiplier. 0.5 (half speed) to 1.5 (1.5x speed).\",\n \"type\": \"number\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent. Useful when mixing spoken voice with background audio.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"avatar_id\"\n ],\n \"type\": \"object\"\n },\n \"type\": {\n \"description\": \"Must be 'avatar_video' for an avatar speaking scene.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"input\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A whole-frame still-image scene: either silent (held for ``duration``) or narrated.\\n\\nExactly one mode must be chosen:\\n- silent: set ``duration`` (seconds) and no audio source.\\n- narrated: set exactly one audio source (script + voice_id, audio_url, or\\n audio_asset_id) and omit ``duration`` — the scene length follows the audio.\",\n \"properties\": {\n \"audio_asset_id\": {\n \"description\": \"Narrated mode: HeyGen asset ID of an uploaded audio file. Mutually exclusive with duration/script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_url\": {\n \"description\": \"Narrated mode: public URL of an audio file to play over the image. Mutually exclusive with duration/script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"duration\": {\n \"description\": \"Silent mode: hold the still image for this many seconds. Mutually exclusive with any audio source. Must be \\u003e 0 and \\u003c= 300.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"script\": {\n \"description\": \"Narrated mode: text to speak over the image. Pair with voice_id. Mutually exclusive with duration/audio_url/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"source\": {\n \"description\": \"Still image to display. Accepts URL, asset ID, or base64-encoded data.\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": {\n \"description\": \"Must be 'image' for a still-image scene.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID for text-to-speech. Required when script is provided.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning parameters (speed, pitch, locale).\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Use the variant matching the engine backing the chosen voice (e.g. engine_type='elevenlabs' for ElevenLabs-backed voices). The request is rejected if the voice_id is not compatible with the selected engine.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3, stability must be 0, 0.5, or 1.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"default\": 0,\n \"description\": \"Pitch adjustment in semitones. -50 to +50.\",\n \"type\": \"number\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Playback speed multiplier. 0.5 (half speed) to 1.5 (1.5x speed).\",\n \"type\": \"number\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent. Useful when mixing spoken voice with background audio.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"source\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A whole-frame scene backed by an existing video clip.\\n\\nOptional ``playback`` exposes the audio volume / mute capability; when\\nomitted the clip plays at its source volume.\\n\\nFor optional voiceover / narration, supply at most one audio source\\n(``script`` + ``voice_id``, ``audio_url``, or ``audio_asset_id``) — the *same*\\naudio inputs a narrated ``image`` scene accepts. When present, the narration\\ndrives the scene length and ``playback.mode`` controls whether the clip\\nfreezes, loops, or changes speed to fit that duration. When omitted the clip\\nplays full-length as before. The clip's own audio level is still governed by\\n``playback`` (the two compose).\",\n \"properties\": {\n \"audio_asset_id\": {\n \"description\": \"Optional voiceover: HeyGen asset ID of an uploaded audio file. Mutually exclusive with script/audio_url.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_url\": {\n \"description\": \"Optional voiceover: public URL of an audio file to play over the clip. Mutually exclusive with script/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"playback\": {\n \"description\": \"Optional playback capability: narrated-clip alignment 'mode', audio 'volume' (0.0–1.0), and 'mute'. Omit to use freeze alignment and keep the clip's source volume.\",\n \"nullable\": true,\n \"properties\": {\n \"mode\": {\n \"description\": \"How a narrated clip aligns to the voiceover-driven scene duration. 'freeze' plays once and holds the last frame; 'loop' repeats the clip; 'fit_to_scene' adjusts playback speed to exactly match the scene. Defaults to 'freeze' when omitted. Requires a video-scene voiceover.\",\n \"enum\": [\n \"freeze\",\n \"loop\",\n \"fit_to_scene\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"mute\": {\n \"default\": false,\n \"description\": \"If True, force the clip silent regardless of 'volume'.\",\n \"type\": \"boolean\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Clip audio volume. 1.0 = source level (default), 0.0 = silent.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"script\": {\n \"description\": \"Optional voiceover: text to speak over the clip. Pair with voice_id. Mutually exclusive with audio_url/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"source\": {\n \"description\": \"Video clip to include. Accepts URL, asset ID, or base64-encoded data.\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": {\n \"description\": \"Must be 'video' for a video-clip scene.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID for text-to-speech. Required when script is provided.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning parameters (speed, pitch, locale) for a script voiceover.\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Use the variant matching the engine backing the chosen voice (e.g. engine_type='elevenlabs' for ElevenLabs-backed voices). The request is rejected if the voice_id is not compatible with the selected engine.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3, stability must be 0, 0.5, or 1.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"default\": 0,\n \"description\": \"Pitch adjustment in semitones. -50 to +50.\",\n \"type\": \"number\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Playback speed multiplier. 0.5 (half speed) to 1.5 (1.5x speed).\",\n \"type\": \"number\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent. Useful when mixing spoken voice with background audio.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"source\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": \"array\"\n },\n \"title\": {\n \"description\": \"Display title for the video in the HeyGen dashboard.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Must be 'studio' for scene-composition video creation.\",\n \"type\": \"string\"\n },\n \"watermark\": {\n \"description\": \"Custom watermark image to overlay on the video (PNG or JPEG). Available as a premium option for select Enterprise customers. To request access, please contact our support team.\",\n \"nullable\": true,\n \"properties\": {\n \"image\": {\n \"description\": \"Image asset to use as the watermark overlay (PNG or JPEG).\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"opacity\": {\n \"default\": 1,\n \"description\": \"Watermark opacity. 0.0 is fully transparent, 1.0 is fully opaque.\",\n \"type\": \"number\"\n },\n \"placement\": {\n \"description\": \"Watermark placement. Defaults to bottom-right with standard margins when omitted.\",\n \"nullable\": true,\n \"properties\": {\n \"offset_x\": {\n \"description\": \"Fine-tune horizontal position. Fraction of frame width; 0.05 shifts 5% rightward, -0.05 shifts 5% leftward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"offset_y\": {\n \"description\": \"Fine-tune vertical position. Fraction of frame height; 0.05 shifts 5% downward, -0.05 shifts 5% upward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"position\": {\n \"default\": \"bottom_right\",\n \"description\": \"Anchor corner for the watermark.\",\n \"enum\": [\n \"top_left\",\n \"top_right\",\n \"bottom_left\",\n \"bottom_right\"\n ],\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"scale\": {\n \"default\": 1,\n \"description\": \"Scale multiplier for the watermark image. 1.0 renders at native size.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"image\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"scenes\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": \"array\"\n }\n },\n \"required\": [\n \"videos\"\n ],\n \"type\": \"object\"\n}", + RequestSchema: "{\n \"properties\": {\n \"callback_url\": {\n \"description\": \"Webhook URL invoked once when every item in the batch reaches a terminal state.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"folder_id\": {\n \"description\": \"Destination folder ID for the batch container. The videos remain grouped inside the newly created batch; the batch itself is placed in this folder. Omit, pass null, or pass an empty string to place the batch at the workspace root.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"Display name for the batch, shown in the HeyGen app.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"videos\": {\n \"description\": \"Video creation requests (avatar / image / cinematic_avatar). Set folder_id once on the batch request, not on individual videos. Max 100 per batch.\",\n \"items\": {\n \"discriminator\": {\n \"mapping\": {\n \"avatar\": \"#/components/schemas/CreateBatchVideoFromAvatar\",\n \"cinematic_avatar\": \"#/components/schemas/CreateBatchVideoFromCinematicAvatar\",\n \"image\": \"#/components/schemas/CreateBatchVideoFromImage\",\n \"studio\": \"#/components/schemas/CreateBatchVideoFromStudio\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Create a video from a HeyGen avatar (video or photo avatar).\\n\\nProvide an avatar_id to use a previously created avatar. Supports all\\navatar types: studio_avatar, digital_twin, and photo_avatar. Optionally\\nset ``engine`` to select Avatar V for eligible avatars; when omitted, the\\nserver defaults to Avatar IV.\",\n \"properties\": {\n \"aspect_ratio\": {\n \"default\": \"16:9\",\n \"description\": \"Output video aspect ratio. Supported values: '16:9', '9:16', '4:5', '5:4', '1:1', 'auto'. Defaults to '16:9'. 'auto' preserves the source's aspect ratio (avatar source frames or uploaded image), short-edge anchored to the requested resolution and capped at the tier's long edge. Falls back to '16:9' when source dimensions can't be read.\",\n \"enum\": [\n \"16:9\",\n \"9:16\",\n \"4:5\",\n \"5:4\",\n \"1:1\",\n \"auto\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_asset_id\": {\n \"description\": \"HeyGen asset ID of an uploaded audio file. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_url\": {\n \"description\": \"Public URL of an audio file to lip-sync. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"avatar_id\": {\n \"description\": \"HeyGen avatar ID (video avatar or photo avatar look ID).\",\n \"type\": \"string\"\n },\n \"background\": {\n \"description\": \"Background settings for the video.\",\n \"nullable\": true,\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID of the background image. Used when type is 'image'. Mutually exclusive with url.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Background type. 'color' uses a solid hex color; 'image' uses an image from url or asset_id.\",\n \"enum\": [\n \"color\",\n \"image\"\n ],\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the background image. Used when type is 'image'. Mutually exclusive with asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"value\": {\n \"description\": \"Hex color code (e.g. '#ff0000'). Required when type is 'color'.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary ID controlling how custom terms are pronounced in generated speech (for example, saying 'HeyGen' as 'hey-jen'). Applies when the audio is synthesized from `script`; a caller-supplied `audio_url` or `audio_asset_id` is unaffected. Pronunciation is applied to the synthesized audio only, so caption and subtitle text still show the original script wording. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"Caller-defined identifier echoed back in the webhook payload.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL to receive a POST notification when the video is ready.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"caption\": {\n \"description\": \"Caption generation settings. A sidecar subtitle file is always returned via subtitle_url; set 'style' to additionally burn captions into the rendered video.\",\n \"nullable\": true,\n \"properties\": {\n \"file_format\": {\n \"default\": \"srt\",\n \"description\": \"Output format for the sidecar caption file.\",\n \"enum\": [\n \"srt\"\n ],\n \"type\": \"string\"\n },\n \"style\": {\n \"description\": \"Visual style for burning captions into the rendered video. Omit for sidecar-only captions.\",\n \"enum\": [\n \"default\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"engine\": {\n \"description\": \"Engine configuration for video generation. Pass {\\\"type\\\": \\\"avatar_v\\\"} to enable cross-reference-driven animation for higher quality. Check supported_api_engines on the avatar look to confirm eligibility. Defaults to Avatar IV when omitted.\",\n \"discriminator\": {\n \"mapping\": {\n \"avatar_iii\": \"#/components/schemas/AvatarIIIEngineConfig\",\n \"avatar_iv\": \"#/components/schemas/AvatarIVEngineConfig\",\n \"avatar_v\": \"#/components/schemas/AvatarVEngineConfig\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Avatar V engine configuration with cross-reference-driven animation.\",\n \"properties\": {\n \"reference_look_id\": {\n \"description\": \"Optional look to use as the animation reference. When provided, it must be a `digital_twin` look accessible to your workspace and in the same avatar group as `avatar_id` (`studio_avatar` and `photo_avatar` looks are rejected). When omitted, video avatars self-reference and photo avatars select from their group's eligible candidates, preferring digital twins (ready first, then processing / upgrading), then curated public studio looks. A photo avatar whose group has no eligible reference renders directly from its image without one; motion_prompt is rejected in that case. A non-public `digital_twin` reference, whether provided or selected automatically, must also satisfy its group's consent requirements.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_v'. Check supported_api_engines on the avatar look to confirm eligibility.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Avatar IV engine configuration (default behavior).\",\n \"properties\": {\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_iv'.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Avatar III engine configuration.\\n\\nA single engine value that resolves to the right product by the avatar's\\nlook type (mirrors how ``avatar_iv`` already serves both photo and video\\navatars):\\n\\n- video avatar looks (``digital_twin``, ``studio_avatar``) -\\u003e Digital Twin\\n (supports 4K)\\n- ``photo_avatar`` look -\\u003e Photo Avatar (no 4K output)\\n\\nNot supported for raw image input (``type: \\\"image\\\"``).\\n``motion_prompt`` and ``expressiveness`` are not supported with this engine.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_iii'. Resolves to Digital Twin for video avatar looks (digital_twin, studio_avatar) and Photo Avatar for photo_avatar looks; not supported for raw image input. Check supported_api_engines on the avatar look to confirm eligibility.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"expressiveness\": {\n \"description\": \"Avatar expressiveness level. Photo avatars only. Defaults to 'low' when omitted. Avatar IV only; rejected when engine.type is 'avatar_v'.\",\n \"enum\": [\n \"high\",\n \"medium\",\n \"low\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"fit\": {\n \"description\": \"How the subject is fitted to the output canvas. 'cover' scales to fill the frame (may crop edges). 'contain' scales to fit entirely within the frame (may show background). When omitted, the server picks the best option based on the source and canvas orientations.\",\n \"enum\": [\n \"contain\",\n \"cover\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"motion_prompt\": {\n \"description\": \"Natural-language prompt controlling avatar body motion and hand gestures. Supported for photo avatars on either engine, and for video avatars when engine.type is 'avatar_v'. Rejected for video avatars on the default Avatar IV engine, and for photo avatars on 'avatar_v' when the avatar's group has no animation reference (no digital twin or curated reference look).\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"output_format\": {\n \"default\": \"mp4\",\n \"description\": \"Output container. 'webm' returns a video with a transparent background (alpha channel); 'mp4' (default) returns a standard video. 'webm' requires an avatar that supports matting. When 'webm' is selected, any 'background' value is rejected and background removal is applied automatically — the caller does not need to set 'remove_background'.\",\n \"enum\": [\n \"mp4\",\n \"webm\"\n ],\n \"type\": \"string\"\n },\n \"remove_background\": {\n \"description\": \"Remove the avatar background. Video avatars must be trained with matting enabled.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"resolution\": {\n \"description\": \"Output video resolution. Avatar IV and Avatar V render the avatar at up to 1080p: with `4k`, the avatar is composited onto a 4K canvas rather than rendered natively. Native 4K output is available for Avatar III digital twins and studio avatars.\",\n \"enum\": [\n \"4k\",\n \"1080p\",\n \"720p\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"script\": {\n \"description\": \"Text script for the avatar to speak. Pair with voice_id, or omit voice_id when using avatar_id to use the avatar's default voice. Mutually exclusive with audio_url/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"Display title for the video in the HeyGen dashboard.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Must be 'avatar' for avatar-based video creation.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID for text-to-speech. Required when script is provided, unless avatar_id is set (the avatar's default voice is used as fallback).\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning parameters (speed, pitch, locale).\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Use the variant matching the engine backing the chosen voice (e.g. engine_type='elevenlabs' for ElevenLabs-backed voices). The request is rejected if the voice_id is not compatible with the selected engine.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3 or eleven_v4, stability may be any value in [0.0, 1.0].\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\",\n \"eleven_v4\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"default\": 0,\n \"description\": \"Pitch adjustment in semitones. -50 to +50.\",\n \"type\": \"number\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Playback speed multiplier. 0.5 (half speed) to 1.5 (1.5x speed).\",\n \"type\": \"number\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent. Useful when mixing spoken voice with background audio.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"watermark\": {\n \"description\": \"Custom watermark image to overlay on the video (PNG or JPEG). Available as a premium option for select Enterprise customers. To request access, please contact our support team.\",\n \"nullable\": true,\n \"properties\": {\n \"image\": {\n \"description\": \"Image asset to use as the watermark overlay (PNG or JPEG).\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"opacity\": {\n \"default\": 1,\n \"description\": \"Watermark opacity. 0.0 is fully transparent, 1.0 is fully opaque.\",\n \"type\": \"number\"\n },\n \"placement\": {\n \"description\": \"Watermark placement. Defaults to bottom-right with standard margins when omitted.\",\n \"nullable\": true,\n \"properties\": {\n \"offset_x\": {\n \"description\": \"Fine-tune horizontal position. Fraction of frame width; 0.05 shifts 5% rightward, -0.05 shifts 5% leftward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"offset_y\": {\n \"description\": \"Fine-tune vertical position. Fraction of frame height; 0.05 shifts 5% downward, -0.05 shifts 5% upward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"position\": {\n \"default\": \"bottom_right\",\n \"description\": \"Anchor corner for the watermark.\",\n \"enum\": [\n \"top_left\",\n \"top_right\",\n \"bottom_left\",\n \"bottom_right\"\n ],\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"scale\": {\n \"default\": 1,\n \"description\": \"Scale multiplier for the watermark image. 1.0 renders at native size.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"image\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"avatar_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Create a video by animating an arbitrary image.\\n\\nProvide an image via URL, asset ID, or inline base64. The image will be\\nanimated with lip-sync to the provided audio or generated speech.\",\n \"properties\": {\n \"aspect_ratio\": {\n \"default\": \"16:9\",\n \"description\": \"Output video aspect ratio. Supported values: '16:9', '9:16', '4:5', '5:4', '1:1', 'auto'. Defaults to '16:9'. 'auto' preserves the source's aspect ratio (avatar source frames or uploaded image), short-edge anchored to the requested resolution and capped at the tier's long edge. Falls back to '16:9' when source dimensions can't be read.\",\n \"enum\": [\n \"16:9\",\n \"9:16\",\n \"4:5\",\n \"5:4\",\n \"1:1\",\n \"auto\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_asset_id\": {\n \"description\": \"HeyGen asset ID of an uploaded audio file. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_url\": {\n \"description\": \"Public URL of an audio file to lip-sync. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"background\": {\n \"description\": \"Background settings for the video.\",\n \"nullable\": true,\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID of the background image. Used when type is 'image'. Mutually exclusive with url.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Background type. 'color' uses a solid hex color; 'image' uses an image from url or asset_id.\",\n \"enum\": [\n \"color\",\n \"image\"\n ],\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the background image. Used when type is 'image'. Mutually exclusive with asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"value\": {\n \"description\": \"Hex color code (e.g. '#ff0000'). Required when type is 'color'.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary ID controlling how custom terms are pronounced in generated speech (for example, saying 'HeyGen' as 'hey-jen'). Applies when the audio is synthesized from `script`; a caller-supplied `audio_url` or `audio_asset_id` is unaffected. Pronunciation is applied to the synthesized audio only, so caption and subtitle text still show the original script wording. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"Caller-defined identifier echoed back in the webhook payload.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL to receive a POST notification when the video is ready.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"caption\": {\n \"description\": \"Caption generation settings. A sidecar subtitle file is always returned via subtitle_url; set 'style' to additionally burn captions into the rendered video.\",\n \"nullable\": true,\n \"properties\": {\n \"file_format\": {\n \"default\": \"srt\",\n \"description\": \"Output format for the sidecar caption file.\",\n \"enum\": [\n \"srt\"\n ],\n \"type\": \"string\"\n },\n \"style\": {\n \"description\": \"Visual style for burning captions into the rendered video. Omit for sidecar-only captions.\",\n \"enum\": [\n \"default\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"expressiveness\": {\n \"description\": \"Avatar expressiveness level. Photo avatars only. Defaults to 'low' when omitted.\",\n \"enum\": [\n \"high\",\n \"medium\",\n \"low\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"fit\": {\n \"description\": \"How the subject is fitted to the output canvas. 'cover' scales to fill the frame (may crop edges). 'contain' scales to fit entirely within the frame (may show background). When omitted, the server picks the best option based on the source and canvas orientations.\",\n \"enum\": [\n \"contain\",\n \"cover\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"image\": {\n \"description\": \"Image to animate. Accepts URL, asset ID, or base64-encoded data.\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"motion_prompt\": {\n \"description\": \"Natural-language prompt controlling avatar body motion. Photo avatars only.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"output_format\": {\n \"default\": \"mp4\",\n \"description\": \"Output container. 'webm' returns a video with a transparent background (alpha channel); 'mp4' (default) returns a standard video. 'webm' requires an avatar that supports matting. When 'webm' is selected, any 'background' value is rejected and background removal is applied automatically — the caller does not need to set 'remove_background'.\",\n \"enum\": [\n \"mp4\",\n \"webm\"\n ],\n \"type\": \"string\"\n },\n \"remove_background\": {\n \"description\": \"Remove the avatar background. Video avatars must be trained with matting enabled.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"resolution\": {\n \"description\": \"Output video resolution. Avatar IV and Avatar V render the avatar at up to 1080p: with `4k`, the avatar is composited onto a 4K canvas rather than rendered natively. Native 4K output is available for Avatar III digital twins and studio avatars.\",\n \"enum\": [\n \"4k\",\n \"1080p\",\n \"720p\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"script\": {\n \"description\": \"Text script for the avatar to speak. Pair with voice_id, or omit voice_id when using avatar_id to use the avatar's default voice. Mutually exclusive with audio_url/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"Display title for the video in the HeyGen dashboard.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Must be 'image' for image-based video creation.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID for text-to-speech. Required when script is provided, unless avatar_id is set (the avatar's default voice is used as fallback).\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning parameters (speed, pitch, locale).\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Use the variant matching the engine backing the chosen voice (e.g. engine_type='elevenlabs' for ElevenLabs-backed voices). The request is rejected if the voice_id is not compatible with the selected engine.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3 or eleven_v4, stability may be any value in [0.0, 1.0].\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\",\n \"eleven_v4\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"default\": 0,\n \"description\": \"Pitch adjustment in semitones. -50 to +50.\",\n \"type\": \"number\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Playback speed multiplier. 0.5 (half speed) to 1.5 (1.5x speed).\",\n \"type\": \"number\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent. Useful when mixing spoken voice with background audio.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"watermark\": {\n \"description\": \"Custom watermark image to overlay on the video (PNG or JPEG). Available as a premium option for select Enterprise customers. To request access, please contact our support team.\",\n \"nullable\": true,\n \"properties\": {\n \"image\": {\n \"description\": \"Image asset to use as the watermark overlay (PNG or JPEG).\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"opacity\": {\n \"default\": 1,\n \"description\": \"Watermark opacity. 0.0 is fully transparent, 1.0 is fully opaque.\",\n \"type\": \"number\"\n },\n \"placement\": {\n \"description\": \"Watermark placement. Defaults to bottom-right with standard margins when omitted.\",\n \"nullable\": true,\n \"properties\": {\n \"offset_x\": {\n \"description\": \"Fine-tune horizontal position. Fraction of frame width; 0.05 shifts 5% rightward, -0.05 shifts 5% leftward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"offset_y\": {\n \"description\": \"Fine-tune vertical position. Fraction of frame height; 0.05 shifts 5% downward, -0.05 shifts 5% upward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"position\": {\n \"default\": \"bottom_right\",\n \"description\": \"Anchor corner for the watermark.\",\n \"enum\": [\n \"top_left\",\n \"top_right\",\n \"bottom_left\",\n \"bottom_right\"\n ],\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"scale\": {\n \"default\": 1,\n \"description\": \"Scale multiplier for the watermark image. 1.0 renders at native size.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"image\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"image\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Create a video from a text prompt plus avatar and asset references (Cinematic Avatar).\\n\\nCinematic Avatar generates a video from a natural-language ``prompt`` guided by\\nreference content: one to three avatar looks and optional reference assets\\n(images / videos / audio). Unlike the ``avatar`` and ``image`` modes there is\\nno script or voice — motion and speech are driven entirely by the prompt and\\nthe supplied references. Backed by the Seedance generation pipeline.\",\n \"properties\": {\n \"aspect_ratio\": {\n \"default\": \"16:9\",\n \"description\": \"Output aspect ratio. Supported for cinematic_avatar: '16:9', '9:16', '1:1'. Defaults to '16:9'.\",\n \"enum\": [\n \"16:9\",\n \"9:16\",\n \"1:1\"\n ],\n \"type\": \"string\"\n },\n \"auto_duration\": {\n \"default\": false,\n \"description\": \"Let the model choose the video length. When true, omit duration.\",\n \"type\": \"boolean\"\n },\n \"avatar_id\": {\n \"description\": \"Avatar look ID(s) used as visual references. Provide 1 to 3 look IDs.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"duration\": {\n \"description\": \"Video length in seconds (4–15). Defaults to 10. Omit when auto_duration is true.\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"enhance_prompt\": {\n \"default\": false,\n \"description\": \"Enable server-side prompt enhancement.\",\n \"type\": \"boolean\"\n },\n \"prompt\": {\n \"description\": \"Natural-language prompt describing the video to generate.\",\n \"type\": \"string\"\n },\n \"references\": {\n \"description\": \"Reference assets (images, videos, or audio) guiding the generation. Each accepts a URL, an asset_id, or inline base64. Combined limits: at most 3 videos and 9 images across avatars and references.\",\n \"items\": {\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"nullable\": true,\n \"type\": \"array\"\n },\n \"resolution\": {\n \"default\": \"720p\",\n \"description\": \"Output resolution. Supported for cinematic_avatar: '720p', '1080p'. Defaults to '720p'.\",\n \"enum\": [\n \"720p\",\n \"1080p\"\n ],\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"Display title for the video in the HeyGen dashboard.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Must be 'cinematic_avatar' for prompt-and-reference video creation.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"prompt\",\n \"avatar_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Create a single video by composing an ordered list of whole-frame scenes.\\n\\nThe server owns layout and center-crops each scene to the global output\\ncanvas. Output settings are global (one per request); a single video_id is\\nreturned and rendering is all-or-nothing. MP4 only in v1 — the output\\ncontainer is fixed and ``output_format`` is not exposed.\",\n \"properties\": {\n \"aspect_ratio\": {\n \"default\": \"16:9\",\n \"description\": \"Global output aspect ratio. Supported values: '16:9', '9:16', '4:5', '5:4', '1:1', 'auto'. Defaults to '16:9'. Each scene is center-cropped to this canvas.\",\n \"enum\": [\n \"16:9\",\n \"9:16\",\n \"4:5\",\n \"5:4\",\n \"1:1\",\n \"auto\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary ID controlling how custom terms are pronounced in generated speech (for example, saying 'HeyGen' as 'hey-jen'). Applies to every scene whose audio is synthesized from a script; scenes that supply their own audio URL or audio asset are unaffected. Pronunciation is applied to the synthesized audio only, so caption and subtitle text still show the original script wording. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"Caller-defined identifier echoed back in the webhook payload.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL to receive a POST notification when the video is ready.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"caption\": {\n \"description\": \"Caption generation settings. A sidecar subtitle file is always returned via subtitle_url; set 'style' to additionally burn captions into the rendered video.\",\n \"nullable\": true,\n \"properties\": {\n \"file_format\": {\n \"default\": \"srt\",\n \"description\": \"Output format for the sidecar caption file.\",\n \"enum\": [\n \"srt\"\n ],\n \"type\": \"string\"\n },\n \"style\": {\n \"description\": \"Visual style for burning captions into the rendered video. Omit for sidecar-only captions.\",\n \"enum\": [\n \"default\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"resolution\": {\n \"description\": \"Global output video resolution. Avatar IV and Avatar V render the avatar at up to 1080p: with `4k`, the avatar is composited onto a 4K canvas rather than rendered natively. Native 4K output is available for Avatar III digital twins and studio avatars.\",\n \"enum\": [\n \"4k\",\n \"1080p\",\n \"720p\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"scenes\": {\n \"description\": \"Ordered list of whole-frame scenes to concatenate. Each scene is one of 'avatar_video', 'image', or 'video'. Must contain 1 to 50 scenes.\",\n \"items\": {\n \"discriminator\": {\n \"mapping\": {\n \"avatar_video\": \"#/components/schemas/AvatarVideoScene\",\n \"image\": \"#/components/schemas/ImageScene\",\n \"video\": \"#/components/schemas/VideoScene\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"A whole-frame speaking scene backed by an avatar.\",\n \"properties\": {\n \"input\": {\n \"description\": \"Scene source ('type': 'avatar'): an avatar_id plus one audio source. The avatar_id accepts any avatar look — video avatars and photo avatars alike (pass a photo avatar's look id to get a talking photo). The scene duration is derived server-side from the audio.\",\n \"properties\": {\n \"audio_asset_id\": {\n \"description\": \"HeyGen asset ID of an uploaded audio file. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_url\": {\n \"description\": \"Public URL of an audio file to lip-sync. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"avatar_id\": {\n \"description\": \"HeyGen avatar ID (video avatar or photo avatar look ID).\",\n \"type\": \"string\"\n },\n \"background\": {\n \"description\": \"Optional scene background composited behind the avatar. Color-only in v1: pass {\\\"type\\\": \\\"color\\\", \\\"color\\\": \\\"#RRGGBB\\\"}. Other background types are not yet supported.\",\n \"discriminator\": {\n \"mapping\": {\n \"color\": \"#/components/schemas/StudioColorBackgroundInput\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Solid-color background for an ``avatar_video`` studio scene.\\n\\nStudio v1 supports solid-color backgrounds.\",\n \"properties\": {\n \"color\": {\n \"description\": \"Background color as a 6-digit hex string, e.g. '#1a2b3c'.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Background type discriminator. Must be 'color'.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"color\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"engine\": {\n \"description\": \"Engine configuration for video generation. Pass {\\\"type\\\": \\\"avatar_v\\\"} to enable cross-reference-driven animation for higher quality. Check supported_api_engines on the avatar look to confirm eligibility. Defaults to Avatar IV when omitted.\",\n \"discriminator\": {\n \"mapping\": {\n \"avatar_iii\": \"#/components/schemas/AvatarIIIEngineConfig\",\n \"avatar_iv\": \"#/components/schemas/AvatarIVEngineConfig\",\n \"avatar_v\": \"#/components/schemas/AvatarVEngineConfig\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Avatar V engine configuration with cross-reference-driven animation.\",\n \"properties\": {\n \"reference_look_id\": {\n \"description\": \"Optional look to use as the animation reference. When provided, it must be a `digital_twin` look accessible to your workspace and in the same avatar group as `avatar_id` (`studio_avatar` and `photo_avatar` looks are rejected). When omitted, video avatars self-reference and photo avatars select from their group's eligible candidates, preferring digital twins (ready first, then processing / upgrading), then curated public studio looks. A photo avatar whose group has no eligible reference renders directly from its image without one; motion_prompt is rejected in that case. A non-public `digital_twin` reference, whether provided or selected automatically, must also satisfy its group's consent requirements.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_v'. Check supported_api_engines on the avatar look to confirm eligibility.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Avatar IV engine configuration (default behavior).\",\n \"properties\": {\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_iv'.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Avatar III engine configuration.\\n\\nA single engine value that resolves to the right product by the avatar's\\nlook type (mirrors how ``avatar_iv`` already serves both photo and video\\navatars):\\n\\n- video avatar looks (``digital_twin``, ``studio_avatar``) -\\u003e Digital Twin\\n (supports 4K)\\n- ``photo_avatar`` look -\\u003e Photo Avatar (no 4K output)\\n\\nNot supported for raw image input (``type: \\\"image\\\"``).\\n``motion_prompt`` and ``expressiveness`` are not supported with this engine.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_iii'. Resolves to Digital Twin for video avatar looks (digital_twin, studio_avatar) and Photo Avatar for photo_avatar looks; not supported for raw image input. Check supported_api_engines on the avatar look to confirm eligibility.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"expressiveness\": {\n \"description\": \"Avatar expressiveness level. Photo avatars only. Defaults to 'low' when omitted. Avatar IV only; rejected when engine.type is 'avatar_v'.\",\n \"enum\": [\n \"high\",\n \"medium\",\n \"low\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"motion_prompt\": {\n \"description\": \"Natural-language prompt controlling avatar body motion and hand gestures. Supported for photo avatars on either engine, and for video avatars when engine.type is 'avatar_v'. Rejected for video avatars on the default Avatar IV engine, and for photo avatars on 'avatar_v' when the avatar's group has no animation reference (no digital twin or curated reference look).\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"script\": {\n \"description\": \"Text script for the avatar to speak. Pair with voice_id, or omit voice_id when using avatar_id to use the avatar's default voice. Mutually exclusive with audio_url/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Must be 'avatar' for an avatar-driven scene source.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID for text-to-speech. Required when script is provided, unless avatar_id is set (the avatar's default voice is used as fallback).\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning parameters (speed, pitch, locale).\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Use the variant matching the engine backing the chosen voice (e.g. engine_type='elevenlabs' for ElevenLabs-backed voices). The request is rejected if the voice_id is not compatible with the selected engine.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3 or eleven_v4, stability may be any value in [0.0, 1.0].\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\",\n \"eleven_v4\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"default\": 0,\n \"description\": \"Pitch adjustment in semitones. -50 to +50.\",\n \"type\": \"number\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Playback speed multiplier. 0.5 (half speed) to 1.5 (1.5x speed).\",\n \"type\": \"number\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent. Useful when mixing spoken voice with background audio.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"avatar_id\"\n ],\n \"type\": \"object\"\n },\n \"type\": {\n \"description\": \"Must be 'avatar_video' for an avatar speaking scene.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"input\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A whole-frame still-image scene: either silent (held for ``duration``) or narrated.\\n\\nExactly one mode must be chosen:\\n- silent: set ``duration`` (seconds) and no audio source.\\n- narrated: set exactly one audio source (script + voice_id, audio_url, or\\n audio_asset_id) and omit ``duration`` — the scene length follows the audio.\",\n \"properties\": {\n \"audio_asset_id\": {\n \"description\": \"Narrated mode: HeyGen asset ID of an uploaded audio file. Mutually exclusive with duration/script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_url\": {\n \"description\": \"Narrated mode: public URL of an audio file to play over the image. Mutually exclusive with duration/script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"duration\": {\n \"description\": \"Silent mode: hold the still image for this many seconds. Mutually exclusive with any audio source. Must be \\u003e 0 and \\u003c= 300.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"script\": {\n \"description\": \"Narrated mode: text to speak over the image. Pair with voice_id. Mutually exclusive with duration/audio_url/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"source\": {\n \"description\": \"Still image to display. Accepts URL, asset ID, or base64-encoded data.\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": {\n \"description\": \"Must be 'image' for a still-image scene.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID for text-to-speech. Required when script is provided.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning parameters (speed, pitch, locale).\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Use the variant matching the engine backing the chosen voice (e.g. engine_type='elevenlabs' for ElevenLabs-backed voices). The request is rejected if the voice_id is not compatible with the selected engine.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3 or eleven_v4, stability may be any value in [0.0, 1.0].\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\",\n \"eleven_v4\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"default\": 0,\n \"description\": \"Pitch adjustment in semitones. -50 to +50.\",\n \"type\": \"number\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Playback speed multiplier. 0.5 (half speed) to 1.5 (1.5x speed).\",\n \"type\": \"number\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent. Useful when mixing spoken voice with background audio.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"source\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A whole-frame scene backed by an existing video clip.\\n\\nOptional ``playback`` exposes the audio volume / mute capability; when\\nomitted the clip plays at its source volume.\\n\\nFor optional voiceover / narration, supply at most one audio source\\n(``script`` + ``voice_id``, ``audio_url``, or ``audio_asset_id``) — the *same*\\naudio inputs a narrated ``image`` scene accepts. When present, the narration\\ndrives the scene length and ``playback.mode`` controls whether the clip\\nfreezes, loops, or changes speed to fit that duration. When omitted the clip\\nplays full-length as before. The clip's own audio level is still governed by\\n``playback`` (the two compose).\",\n \"properties\": {\n \"audio_asset_id\": {\n \"description\": \"Optional voiceover: HeyGen asset ID of an uploaded audio file. Mutually exclusive with script/audio_url.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_url\": {\n \"description\": \"Optional voiceover: public URL of an audio file to play over the clip. Mutually exclusive with script/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"playback\": {\n \"description\": \"Optional playback capability: narrated-clip alignment 'mode', audio 'volume' (0.0–1.0), and 'mute'. Omit to use freeze alignment and keep the clip's source volume.\",\n \"nullable\": true,\n \"properties\": {\n \"mode\": {\n \"description\": \"How a narrated clip aligns to the voiceover-driven scene duration. 'freeze' plays once and holds the last frame; 'loop' repeats the clip; 'fit_to_scene' adjusts playback speed to exactly match the scene. Defaults to 'freeze' when omitted. Requires a video-scene voiceover.\",\n \"enum\": [\n \"freeze\",\n \"loop\",\n \"fit_to_scene\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"mute\": {\n \"default\": false,\n \"description\": \"If True, force the clip silent regardless of 'volume'.\",\n \"type\": \"boolean\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Clip audio volume. 1.0 = source level (default), 0.0 = silent.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"script\": {\n \"description\": \"Optional voiceover: text to speak over the clip. Pair with voice_id. Mutually exclusive with audio_url/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"source\": {\n \"description\": \"Video clip to include. Accepts URL, asset ID, or base64-encoded data.\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": {\n \"description\": \"Must be 'video' for a video-clip scene.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID for text-to-speech. Required when script is provided.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning parameters (speed, pitch, locale) for a script voiceover.\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Use the variant matching the engine backing the chosen voice (e.g. engine_type='elevenlabs' for ElevenLabs-backed voices). The request is rejected if the voice_id is not compatible with the selected engine.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3 or eleven_v4, stability may be any value in [0.0, 1.0].\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\",\n \"eleven_v4\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"default\": 0,\n \"description\": \"Pitch adjustment in semitones. -50 to +50.\",\n \"type\": \"number\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Playback speed multiplier. 0.5 (half speed) to 1.5 (1.5x speed).\",\n \"type\": \"number\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent. Useful when mixing spoken voice with background audio.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"source\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": \"array\"\n },\n \"title\": {\n \"description\": \"Display title for the video in the HeyGen dashboard.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Must be 'studio' for scene-composition video creation.\",\n \"type\": \"string\"\n },\n \"watermark\": {\n \"description\": \"Custom watermark image to overlay on the video (PNG or JPEG). Available as a premium option for select Enterprise customers. To request access, please contact our support team.\",\n \"nullable\": true,\n \"properties\": {\n \"image\": {\n \"description\": \"Image asset to use as the watermark overlay (PNG or JPEG).\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"opacity\": {\n \"default\": 1,\n \"description\": \"Watermark opacity. 0.0 is fully transparent, 1.0 is fully opaque.\",\n \"type\": \"number\"\n },\n \"placement\": {\n \"description\": \"Watermark placement. Defaults to bottom-right with standard margins when omitted.\",\n \"nullable\": true,\n \"properties\": {\n \"offset_x\": {\n \"description\": \"Fine-tune horizontal position. Fraction of frame width; 0.05 shifts 5% rightward, -0.05 shifts 5% leftward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"offset_y\": {\n \"description\": \"Fine-tune vertical position. Fraction of frame height; 0.05 shifts 5% downward, -0.05 shifts 5% upward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"position\": {\n \"default\": \"bottom_right\",\n \"description\": \"Anchor corner for the watermark.\",\n \"enum\": [\n \"top_left\",\n \"top_right\",\n \"bottom_left\",\n \"bottom_right\"\n ],\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"scale\": {\n \"default\": 1,\n \"description\": \"Scale multiplier for the watermark image. 1.0 renders at native size.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"image\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"scenes\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": \"array\"\n }\n },\n \"required\": [\n \"videos\"\n ],\n \"type\": \"object\"\n}", Endpoint: "/v3/videos/batches", Method: "POST", BodyEncoding: "json", @@ -118,7 +118,7 @@ var VideoCreate = &command.Spec{ Name: "create", Summary: "Create Video", Description: "Creates a video from a HeyGen avatar or an arbitrary image. Supports scripts or pre-recorded audio for lip-sync. Supports the Avatar III, Avatar IV, and Avatar V engines; set the 'engine' field to select. Avatar IV is used by default when 'engine' is omitted.", - RequestSchema: "{\n \"description\": \"Discriminated union for POST /v3/videos request body.\",\n \"discriminator\": {\n \"mapping\": {\n \"avatar\": \"#/components/schemas/CreateVideoFromAvatar\",\n \"cinematic_avatar\": \"#/components/schemas/CreateVideoFromCinematicAvatar\",\n \"image\": \"#/components/schemas/CreateVideoFromImage\",\n \"studio\": \"#/components/schemas/CreateVideoFromStudio\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Create a video from a HeyGen avatar (video or photo avatar).\\n\\nProvide an avatar_id to use a previously created avatar. Supports all\\navatar types: studio_avatar, digital_twin, and photo_avatar. Optionally\\nset ``engine`` to select Avatar V for eligible avatars; when omitted, the\\nserver defaults to Avatar IV.\",\n \"properties\": {\n \"aspect_ratio\": {\n \"default\": \"16:9\",\n \"description\": \"Output video aspect ratio. Supported values: '16:9', '9:16', '4:5', '5:4', '1:1', 'auto'. Defaults to '16:9'. 'auto' preserves the source's aspect ratio (avatar source frames or uploaded image), short-edge anchored to the requested resolution and capped at the tier's long edge. Falls back to '16:9' when source dimensions can't be read.\",\n \"enum\": [\n \"16:9\",\n \"9:16\",\n \"4:5\",\n \"5:4\",\n \"1:1\",\n \"auto\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_asset_id\": {\n \"description\": \"HeyGen asset ID of an uploaded audio file. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_url\": {\n \"description\": \"Public URL of an audio file to lip-sync. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"avatar_id\": {\n \"description\": \"HeyGen avatar ID (video avatar or photo avatar look ID).\",\n \"type\": \"string\"\n },\n \"background\": {\n \"description\": \"Background settings for the video.\",\n \"nullable\": true,\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID of the background image. Used when type is 'image'. Mutually exclusive with url.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Background type. 'color' uses a solid hex color; 'image' uses an image from url or asset_id.\",\n \"enum\": [\n \"color\",\n \"image\"\n ],\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the background image. Used when type is 'image'. Mutually exclusive with asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"value\": {\n \"description\": \"Hex color code (e.g. '#ff0000'). Required when type is 'color'.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary ID controlling how custom terms are pronounced in generated speech (for example, saying 'HeyGen' as 'hey-jen'). Applies when the audio is synthesized from `script`; a caller-supplied `audio_url` or `audio_asset_id` is unaffected. Pronunciation is applied to the synthesized audio only, so caption and subtitle text still show the original script wording. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"Caller-defined identifier echoed back in the webhook payload.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL to receive a POST notification when the video is ready.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"caption\": {\n \"description\": \"Caption generation settings. A sidecar subtitle file is always returned via subtitle_url; set 'style' to additionally burn captions into the rendered video.\",\n \"nullable\": true,\n \"properties\": {\n \"file_format\": {\n \"default\": \"srt\",\n \"description\": \"Output format for the sidecar caption file.\",\n \"enum\": [\n \"srt\"\n ],\n \"type\": \"string\"\n },\n \"style\": {\n \"description\": \"Visual style for burning captions into the rendered video. Omit for sidecar-only captions.\",\n \"enum\": [\n \"default\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"engine\": {\n \"description\": \"Engine configuration for video generation. Pass {\\\"type\\\": \\\"avatar_v\\\"} to enable cross-reference-driven animation for higher quality. Check supported_api_engines on the avatar look to confirm eligibility. Defaults to Avatar IV when omitted.\",\n \"discriminator\": {\n \"mapping\": {\n \"avatar_iii\": \"#/components/schemas/AvatarIIIEngineConfig\",\n \"avatar_iv\": \"#/components/schemas/AvatarIVEngineConfig\",\n \"avatar_v\": \"#/components/schemas/AvatarVEngineConfig\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Avatar V engine configuration with cross-reference-driven animation.\",\n \"properties\": {\n \"reference_look_id\": {\n \"description\": \"Optional look to use as the animation reference. When provided, it must be a `digital_twin` look accessible to your workspace and in the same avatar group as `avatar_id` (`studio_avatar` and `photo_avatar` looks are rejected). When omitted, video avatars self-reference and photo avatars select from their group's eligible candidates, preferring digital twins (ready first, then processing / upgrading), then curated public studio looks. A photo avatar whose group has no eligible reference renders directly from its image without one; motion_prompt is rejected in that case. A non-public `digital_twin` reference, whether provided or selected automatically, must also satisfy its group's consent requirements.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_v'. Check supported_api_engines on the avatar look to confirm eligibility.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Avatar IV engine configuration (default behavior).\",\n \"properties\": {\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_iv'.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Avatar III engine configuration.\\n\\nA single engine value that resolves to the right product by the avatar's\\nlook type (mirrors how ``avatar_iv`` already serves both photo and video\\navatars):\\n\\n- video avatar looks (``digital_twin``, ``studio_avatar``) -\\u003e Digital Twin\\n (supports 4K)\\n- ``photo_avatar`` look -\\u003e Photo Avatar (no 4K output)\\n\\nNot supported for raw image input (``type: \\\"image\\\"``).\\n``motion_prompt`` and ``expressiveness`` are not supported with this engine.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_iii'. Resolves to Digital Twin for video avatar looks (digital_twin, studio_avatar) and Photo Avatar for photo_avatar looks; not supported for raw image input. Check supported_api_engines on the avatar look to confirm eligibility.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"expressiveness\": {\n \"description\": \"Avatar expressiveness level. Photo avatars only. Defaults to 'low' when omitted. Avatar IV only; rejected when engine.type is 'avatar_v'.\",\n \"enum\": [\n \"high\",\n \"medium\",\n \"low\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"fit\": {\n \"description\": \"How the subject is fitted to the output canvas. 'cover' scales to fill the frame (may crop edges). 'contain' scales to fit entirely within the frame (may show background). When omitted, the server picks the best option based on the source and canvas orientations.\",\n \"enum\": [\n \"contain\",\n \"cover\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"folder_id\": {\n \"description\": \"Destination folder ID in the caller's workspace. Omit, pass null, or pass an empty string to place the video at the workspace root. Supported only for single-video creation; batch video items cannot set their own destination folder.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"motion_prompt\": {\n \"description\": \"Natural-language prompt controlling avatar body motion and hand gestures. Supported for photo avatars on either engine, and for video avatars when engine.type is 'avatar_v'. Rejected for video avatars on the default Avatar IV engine, and for photo avatars on 'avatar_v' when the avatar's group has no animation reference (no digital twin or curated reference look).\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"output_format\": {\n \"default\": \"mp4\",\n \"description\": \"Output container. 'webm' returns a video with a transparent background (alpha channel); 'mp4' (default) returns a standard video. 'webm' requires an avatar that supports matting. When 'webm' is selected, any 'background' value is rejected and background removal is applied automatically — the caller does not need to set 'remove_background'.\",\n \"enum\": [\n \"mp4\",\n \"webm\"\n ],\n \"type\": \"string\"\n },\n \"remove_background\": {\n \"description\": \"Remove the avatar background. Video avatars must be trained with matting enabled.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"resolution\": {\n \"description\": \"Output video resolution. Avatar IV and Avatar V render the avatar at up to 1080p: with `4k`, the avatar is composited onto a 4K canvas rather than rendered natively. Native 4K output is available for Avatar III digital twins and studio avatars.\",\n \"enum\": [\n \"4k\",\n \"1080p\",\n \"720p\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"script\": {\n \"description\": \"Text script for the avatar to speak. Pair with voice_id, or omit voice_id when using avatar_id to use the avatar's default voice. Mutually exclusive with audio_url/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"Display title for the video in the HeyGen dashboard.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Must be 'avatar' for avatar-based video creation.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID for text-to-speech. Required when script is provided, unless avatar_id is set (the avatar's default voice is used as fallback).\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning parameters (speed, pitch, locale).\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Use the variant matching the engine backing the chosen voice (e.g. engine_type='elevenlabs' for ElevenLabs-backed voices). The request is rejected if the voice_id is not compatible with the selected engine.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3, stability must be 0, 0.5, or 1.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"default\": 0,\n \"description\": \"Pitch adjustment in semitones. -50 to +50.\",\n \"type\": \"number\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Playback speed multiplier. 0.5 (half speed) to 1.5 (1.5x speed).\",\n \"type\": \"number\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent. Useful when mixing spoken voice with background audio.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"watermark\": {\n \"description\": \"Custom watermark image to overlay on the video (PNG or JPEG). Available as a premium option for select Enterprise customers. To request access, please contact our support team.\",\n \"nullable\": true,\n \"properties\": {\n \"image\": {\n \"description\": \"Image asset to use as the watermark overlay (PNG or JPEG).\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"opacity\": {\n \"default\": 1,\n \"description\": \"Watermark opacity. 0.0 is fully transparent, 1.0 is fully opaque.\",\n \"type\": \"number\"\n },\n \"placement\": {\n \"description\": \"Watermark placement. Defaults to bottom-right with standard margins when omitted.\",\n \"nullable\": true,\n \"properties\": {\n \"offset_x\": {\n \"description\": \"Fine-tune horizontal position. Fraction of frame width; 0.05 shifts 5% rightward, -0.05 shifts 5% leftward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"offset_y\": {\n \"description\": \"Fine-tune vertical position. Fraction of frame height; 0.05 shifts 5% downward, -0.05 shifts 5% upward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"position\": {\n \"default\": \"bottom_right\",\n \"description\": \"Anchor corner for the watermark.\",\n \"enum\": [\n \"top_left\",\n \"top_right\",\n \"bottom_left\",\n \"bottom_right\"\n ],\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"scale\": {\n \"default\": 1,\n \"description\": \"Scale multiplier for the watermark image. 1.0 renders at native size.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"image\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"avatar_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Create a video by animating an arbitrary image.\\n\\nProvide an image via URL, asset ID, or inline base64. The image will be\\nanimated with lip-sync to the provided audio or generated speech.\",\n \"properties\": {\n \"aspect_ratio\": {\n \"default\": \"16:9\",\n \"description\": \"Output video aspect ratio. Supported values: '16:9', '9:16', '4:5', '5:4', '1:1', 'auto'. Defaults to '16:9'. 'auto' preserves the source's aspect ratio (avatar source frames or uploaded image), short-edge anchored to the requested resolution and capped at the tier's long edge. Falls back to '16:9' when source dimensions can't be read.\",\n \"enum\": [\n \"16:9\",\n \"9:16\",\n \"4:5\",\n \"5:4\",\n \"1:1\",\n \"auto\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_asset_id\": {\n \"description\": \"HeyGen asset ID of an uploaded audio file. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_url\": {\n \"description\": \"Public URL of an audio file to lip-sync. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"background\": {\n \"description\": \"Background settings for the video.\",\n \"nullable\": true,\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID of the background image. Used when type is 'image'. Mutually exclusive with url.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Background type. 'color' uses a solid hex color; 'image' uses an image from url or asset_id.\",\n \"enum\": [\n \"color\",\n \"image\"\n ],\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the background image. Used when type is 'image'. Mutually exclusive with asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"value\": {\n \"description\": \"Hex color code (e.g. '#ff0000'). Required when type is 'color'.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary ID controlling how custom terms are pronounced in generated speech (for example, saying 'HeyGen' as 'hey-jen'). Applies when the audio is synthesized from `script`; a caller-supplied `audio_url` or `audio_asset_id` is unaffected. Pronunciation is applied to the synthesized audio only, so caption and subtitle text still show the original script wording. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"Caller-defined identifier echoed back in the webhook payload.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL to receive a POST notification when the video is ready.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"caption\": {\n \"description\": \"Caption generation settings. A sidecar subtitle file is always returned via subtitle_url; set 'style' to additionally burn captions into the rendered video.\",\n \"nullable\": true,\n \"properties\": {\n \"file_format\": {\n \"default\": \"srt\",\n \"description\": \"Output format for the sidecar caption file.\",\n \"enum\": [\n \"srt\"\n ],\n \"type\": \"string\"\n },\n \"style\": {\n \"description\": \"Visual style for burning captions into the rendered video. Omit for sidecar-only captions.\",\n \"enum\": [\n \"default\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"expressiveness\": {\n \"description\": \"Avatar expressiveness level. Photo avatars only. Defaults to 'low' when omitted.\",\n \"enum\": [\n \"high\",\n \"medium\",\n \"low\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"fit\": {\n \"description\": \"How the subject is fitted to the output canvas. 'cover' scales to fill the frame (may crop edges). 'contain' scales to fit entirely within the frame (may show background). When omitted, the server picks the best option based on the source and canvas orientations.\",\n \"enum\": [\n \"contain\",\n \"cover\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"folder_id\": {\n \"description\": \"Destination folder ID in the caller's workspace. Omit, pass null, or pass an empty string to place the video at the workspace root. Supported only for single-video creation; batch video items cannot set their own destination folder.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"image\": {\n \"description\": \"Image to animate. Accepts URL, asset ID, or base64-encoded data.\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"motion_prompt\": {\n \"description\": \"Natural-language prompt controlling avatar body motion. Photo avatars only.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"output_format\": {\n \"default\": \"mp4\",\n \"description\": \"Output container. 'webm' returns a video with a transparent background (alpha channel); 'mp4' (default) returns a standard video. 'webm' requires an avatar that supports matting. When 'webm' is selected, any 'background' value is rejected and background removal is applied automatically — the caller does not need to set 'remove_background'.\",\n \"enum\": [\n \"mp4\",\n \"webm\"\n ],\n \"type\": \"string\"\n },\n \"remove_background\": {\n \"description\": \"Remove the avatar background. Video avatars must be trained with matting enabled.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"resolution\": {\n \"description\": \"Output video resolution. Avatar IV and Avatar V render the avatar at up to 1080p: with `4k`, the avatar is composited onto a 4K canvas rather than rendered natively. Native 4K output is available for Avatar III digital twins and studio avatars.\",\n \"enum\": [\n \"4k\",\n \"1080p\",\n \"720p\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"script\": {\n \"description\": \"Text script for the avatar to speak. Pair with voice_id, or omit voice_id when using avatar_id to use the avatar's default voice. Mutually exclusive with audio_url/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"Display title for the video in the HeyGen dashboard.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Must be 'image' for image-based video creation.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID for text-to-speech. Required when script is provided, unless avatar_id is set (the avatar's default voice is used as fallback).\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning parameters (speed, pitch, locale).\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Use the variant matching the engine backing the chosen voice (e.g. engine_type='elevenlabs' for ElevenLabs-backed voices). The request is rejected if the voice_id is not compatible with the selected engine.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3, stability must be 0, 0.5, or 1.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"default\": 0,\n \"description\": \"Pitch adjustment in semitones. -50 to +50.\",\n \"type\": \"number\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Playback speed multiplier. 0.5 (half speed) to 1.5 (1.5x speed).\",\n \"type\": \"number\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent. Useful when mixing spoken voice with background audio.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"watermark\": {\n \"description\": \"Custom watermark image to overlay on the video (PNG or JPEG). Available as a premium option for select Enterprise customers. To request access, please contact our support team.\",\n \"nullable\": true,\n \"properties\": {\n \"image\": {\n \"description\": \"Image asset to use as the watermark overlay (PNG or JPEG).\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"opacity\": {\n \"default\": 1,\n \"description\": \"Watermark opacity. 0.0 is fully transparent, 1.0 is fully opaque.\",\n \"type\": \"number\"\n },\n \"placement\": {\n \"description\": \"Watermark placement. Defaults to bottom-right with standard margins when omitted.\",\n \"nullable\": true,\n \"properties\": {\n \"offset_x\": {\n \"description\": \"Fine-tune horizontal position. Fraction of frame width; 0.05 shifts 5% rightward, -0.05 shifts 5% leftward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"offset_y\": {\n \"description\": \"Fine-tune vertical position. Fraction of frame height; 0.05 shifts 5% downward, -0.05 shifts 5% upward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"position\": {\n \"default\": \"bottom_right\",\n \"description\": \"Anchor corner for the watermark.\",\n \"enum\": [\n \"top_left\",\n \"top_right\",\n \"bottom_left\",\n \"bottom_right\"\n ],\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"scale\": {\n \"default\": 1,\n \"description\": \"Scale multiplier for the watermark image. 1.0 renders at native size.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"image\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"image\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Create a video from a text prompt plus avatar and asset references (Cinematic Avatar).\\n\\nCinematic Avatar generates a video from a natural-language ``prompt`` guided by\\nreference content: one to three avatar looks and optional reference assets\\n(images / videos / audio). Unlike the ``avatar`` and ``image`` modes there is\\nno script or voice — motion and speech are driven entirely by the prompt and\\nthe supplied references. Backed by the Seedance generation pipeline.\",\n \"properties\": {\n \"aspect_ratio\": {\n \"default\": \"16:9\",\n \"description\": \"Output aspect ratio. Supported for cinematic_avatar: '16:9', '9:16', '1:1'. Defaults to '16:9'.\",\n \"enum\": [\n \"16:9\",\n \"9:16\",\n \"1:1\"\n ],\n \"type\": \"string\"\n },\n \"auto_duration\": {\n \"default\": false,\n \"description\": \"Let the model choose the video length. When true, omit duration.\",\n \"type\": \"boolean\"\n },\n \"avatar_id\": {\n \"description\": \"Avatar look ID(s) used as visual references. Provide 1 to 3 look IDs.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"duration\": {\n \"description\": \"Video length in seconds (4–15). Defaults to 10. Omit when auto_duration is true.\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"enhance_prompt\": {\n \"default\": false,\n \"description\": \"Enable server-side prompt enhancement.\",\n \"type\": \"boolean\"\n },\n \"folder_id\": {\n \"description\": \"Destination folder ID in the caller's workspace. Omit, pass null, or pass an empty string to place the video at the workspace root. Supported only for single-video creation; batch video items cannot set their own destination folder.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"prompt\": {\n \"description\": \"Natural-language prompt describing the video to generate.\",\n \"type\": \"string\"\n },\n \"references\": {\n \"description\": \"Reference assets (images, videos, or audio) guiding the generation. Each accepts a URL, an asset_id, or inline base64. Combined limits: at most 3 videos and 9 images across avatars and references.\",\n \"items\": {\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"nullable\": true,\n \"type\": \"array\"\n },\n \"resolution\": {\n \"default\": \"720p\",\n \"description\": \"Output resolution. Supported for cinematic_avatar: '720p', '1080p'. Defaults to '720p'.\",\n \"enum\": [\n \"720p\",\n \"1080p\"\n ],\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"Display title for the video in the HeyGen dashboard.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Must be 'cinematic_avatar' for prompt-and-reference video creation.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"prompt\",\n \"avatar_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Create a single video by composing an ordered list of whole-frame scenes.\\n\\nThe server owns layout and center-crops each scene to the global output\\ncanvas. Output settings are global (one per request); a single video_id is\\nreturned and rendering is all-or-nothing. MP4 only in v1 — the output\\ncontainer is fixed and ``output_format`` is not exposed.\",\n \"properties\": {\n \"aspect_ratio\": {\n \"default\": \"16:9\",\n \"description\": \"Global output aspect ratio. Supported values: '16:9', '9:16', '4:5', '5:4', '1:1', 'auto'. Defaults to '16:9'. Each scene is center-cropped to this canvas.\",\n \"enum\": [\n \"16:9\",\n \"9:16\",\n \"4:5\",\n \"5:4\",\n \"1:1\",\n \"auto\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary ID controlling how custom terms are pronounced in generated speech (for example, saying 'HeyGen' as 'hey-jen'). Applies to every scene whose audio is synthesized from a script; scenes that supply their own audio URL or audio asset are unaffected. Pronunciation is applied to the synthesized audio only, so caption and subtitle text still show the original script wording. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"Caller-defined identifier echoed back in the webhook payload.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL to receive a POST notification when the video is ready.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"caption\": {\n \"description\": \"Caption generation settings. A sidecar subtitle file is always returned via subtitle_url; set 'style' to additionally burn captions into the rendered video.\",\n \"nullable\": true,\n \"properties\": {\n \"file_format\": {\n \"default\": \"srt\",\n \"description\": \"Output format for the sidecar caption file.\",\n \"enum\": [\n \"srt\"\n ],\n \"type\": \"string\"\n },\n \"style\": {\n \"description\": \"Visual style for burning captions into the rendered video. Omit for sidecar-only captions.\",\n \"enum\": [\n \"default\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"folder_id\": {\n \"description\": \"Destination folder ID in the caller's workspace. Omit, pass null, or pass an empty string to place the video at the workspace root. Supported only for single-video creation; batch video items cannot set their own destination folder.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"resolution\": {\n \"description\": \"Global output video resolution. Avatar IV and Avatar V render the avatar at up to 1080p: with `4k`, the avatar is composited onto a 4K canvas rather than rendered natively. Native 4K output is available for Avatar III digital twins and studio avatars.\",\n \"enum\": [\n \"4k\",\n \"1080p\",\n \"720p\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"scenes\": {\n \"description\": \"Ordered list of whole-frame scenes to concatenate. Each scene is one of 'avatar_video', 'image', or 'video'. Must contain 1 to 50 scenes.\",\n \"items\": {\n \"discriminator\": {\n \"mapping\": {\n \"avatar_video\": \"#/components/schemas/AvatarVideoScene\",\n \"image\": \"#/components/schemas/ImageScene\",\n \"video\": \"#/components/schemas/VideoScene\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"A whole-frame speaking scene backed by an avatar.\",\n \"properties\": {\n \"input\": {\n \"description\": \"Scene source ('type': 'avatar'): an avatar_id plus one audio source. The avatar_id accepts any avatar look — video avatars and photo avatars alike (pass a photo avatar's look id to get a talking photo). The scene duration is derived server-side from the audio.\",\n \"properties\": {\n \"audio_asset_id\": {\n \"description\": \"HeyGen asset ID of an uploaded audio file. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_url\": {\n \"description\": \"Public URL of an audio file to lip-sync. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"avatar_id\": {\n \"description\": \"HeyGen avatar ID (video avatar or photo avatar look ID).\",\n \"type\": \"string\"\n },\n \"background\": {\n \"description\": \"Optional scene background composited behind the avatar. Color-only in v1: pass {\\\"type\\\": \\\"color\\\", \\\"color\\\": \\\"#RRGGBB\\\"}. Other background types are not yet supported.\",\n \"discriminator\": {\n \"mapping\": {\n \"color\": \"#/components/schemas/StudioColorBackgroundInput\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Solid-color background for an ``avatar_video`` studio scene.\\n\\nStudio v1 supports solid-color backgrounds.\",\n \"properties\": {\n \"color\": {\n \"description\": \"Background color as a 6-digit hex string, e.g. '#1a2b3c'.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Background type discriminator. Must be 'color'.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"color\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"engine\": {\n \"description\": \"Engine configuration for video generation. Pass {\\\"type\\\": \\\"avatar_v\\\"} to enable cross-reference-driven animation for higher quality. Check supported_api_engines on the avatar look to confirm eligibility. Defaults to Avatar IV when omitted.\",\n \"discriminator\": {\n \"mapping\": {\n \"avatar_iii\": \"#/components/schemas/AvatarIIIEngineConfig\",\n \"avatar_iv\": \"#/components/schemas/AvatarIVEngineConfig\",\n \"avatar_v\": \"#/components/schemas/AvatarVEngineConfig\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Avatar V engine configuration with cross-reference-driven animation.\",\n \"properties\": {\n \"reference_look_id\": {\n \"description\": \"Optional look to use as the animation reference. When provided, it must be a `digital_twin` look accessible to your workspace and in the same avatar group as `avatar_id` (`studio_avatar` and `photo_avatar` looks are rejected). When omitted, video avatars self-reference and photo avatars select from their group's eligible candidates, preferring digital twins (ready first, then processing / upgrading), then curated public studio looks. A photo avatar whose group has no eligible reference renders directly from its image without one; motion_prompt is rejected in that case. A non-public `digital_twin` reference, whether provided or selected automatically, must also satisfy its group's consent requirements.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_v'. Check supported_api_engines on the avatar look to confirm eligibility.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Avatar IV engine configuration (default behavior).\",\n \"properties\": {\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_iv'.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Avatar III engine configuration.\\n\\nA single engine value that resolves to the right product by the avatar's\\nlook type (mirrors how ``avatar_iv`` already serves both photo and video\\navatars):\\n\\n- video avatar looks (``digital_twin``, ``studio_avatar``) -\\u003e Digital Twin\\n (supports 4K)\\n- ``photo_avatar`` look -\\u003e Photo Avatar (no 4K output)\\n\\nNot supported for raw image input (``type: \\\"image\\\"``).\\n``motion_prompt`` and ``expressiveness`` are not supported with this engine.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_iii'. Resolves to Digital Twin for video avatar looks (digital_twin, studio_avatar) and Photo Avatar for photo_avatar looks; not supported for raw image input. Check supported_api_engines on the avatar look to confirm eligibility.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"expressiveness\": {\n \"description\": \"Avatar expressiveness level. Photo avatars only. Defaults to 'low' when omitted. Avatar IV only; rejected when engine.type is 'avatar_v'.\",\n \"enum\": [\n \"high\",\n \"medium\",\n \"low\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"motion_prompt\": {\n \"description\": \"Natural-language prompt controlling avatar body motion and hand gestures. Supported for photo avatars on either engine, and for video avatars when engine.type is 'avatar_v'. Rejected for video avatars on the default Avatar IV engine, and for photo avatars on 'avatar_v' when the avatar's group has no animation reference (no digital twin or curated reference look).\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"script\": {\n \"description\": \"Text script for the avatar to speak. Pair with voice_id, or omit voice_id when using avatar_id to use the avatar's default voice. Mutually exclusive with audio_url/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Must be 'avatar' for an avatar-driven scene source.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID for text-to-speech. Required when script is provided, unless avatar_id is set (the avatar's default voice is used as fallback).\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning parameters (speed, pitch, locale).\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Use the variant matching the engine backing the chosen voice (e.g. engine_type='elevenlabs' for ElevenLabs-backed voices). The request is rejected if the voice_id is not compatible with the selected engine.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3, stability must be 0, 0.5, or 1.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"default\": 0,\n \"description\": \"Pitch adjustment in semitones. -50 to +50.\",\n \"type\": \"number\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Playback speed multiplier. 0.5 (half speed) to 1.5 (1.5x speed).\",\n \"type\": \"number\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent. Useful when mixing spoken voice with background audio.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"avatar_id\"\n ],\n \"type\": \"object\"\n },\n \"type\": {\n \"description\": \"Must be 'avatar_video' for an avatar speaking scene.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"input\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A whole-frame still-image scene: either silent (held for ``duration``) or narrated.\\n\\nExactly one mode must be chosen:\\n- silent: set ``duration`` (seconds) and no audio source.\\n- narrated: set exactly one audio source (script + voice_id, audio_url, or\\n audio_asset_id) and omit ``duration`` — the scene length follows the audio.\",\n \"properties\": {\n \"audio_asset_id\": {\n \"description\": \"Narrated mode: HeyGen asset ID of an uploaded audio file. Mutually exclusive with duration/script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_url\": {\n \"description\": \"Narrated mode: public URL of an audio file to play over the image. Mutually exclusive with duration/script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"duration\": {\n \"description\": \"Silent mode: hold the still image for this many seconds. Mutually exclusive with any audio source. Must be \\u003e 0 and \\u003c= 300.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"script\": {\n \"description\": \"Narrated mode: text to speak over the image. Pair with voice_id. Mutually exclusive with duration/audio_url/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"source\": {\n \"description\": \"Still image to display. Accepts URL, asset ID, or base64-encoded data.\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": {\n \"description\": \"Must be 'image' for a still-image scene.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID for text-to-speech. Required when script is provided.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning parameters (speed, pitch, locale).\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Use the variant matching the engine backing the chosen voice (e.g. engine_type='elevenlabs' for ElevenLabs-backed voices). The request is rejected if the voice_id is not compatible with the selected engine.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3, stability must be 0, 0.5, or 1.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"default\": 0,\n \"description\": \"Pitch adjustment in semitones. -50 to +50.\",\n \"type\": \"number\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Playback speed multiplier. 0.5 (half speed) to 1.5 (1.5x speed).\",\n \"type\": \"number\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent. Useful when mixing spoken voice with background audio.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"source\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A whole-frame scene backed by an existing video clip.\\n\\nOptional ``playback`` exposes the audio volume / mute capability; when\\nomitted the clip plays at its source volume.\\n\\nFor optional voiceover / narration, supply at most one audio source\\n(``script`` + ``voice_id``, ``audio_url``, or ``audio_asset_id``) — the *same*\\naudio inputs a narrated ``image`` scene accepts. When present, the narration\\ndrives the scene length and ``playback.mode`` controls whether the clip\\nfreezes, loops, or changes speed to fit that duration. When omitted the clip\\nplays full-length as before. The clip's own audio level is still governed by\\n``playback`` (the two compose).\",\n \"properties\": {\n \"audio_asset_id\": {\n \"description\": \"Optional voiceover: HeyGen asset ID of an uploaded audio file. Mutually exclusive with script/audio_url.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_url\": {\n \"description\": \"Optional voiceover: public URL of an audio file to play over the clip. Mutually exclusive with script/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"playback\": {\n \"description\": \"Optional playback capability: narrated-clip alignment 'mode', audio 'volume' (0.0–1.0), and 'mute'. Omit to use freeze alignment and keep the clip's source volume.\",\n \"nullable\": true,\n \"properties\": {\n \"mode\": {\n \"description\": \"How a narrated clip aligns to the voiceover-driven scene duration. 'freeze' plays once and holds the last frame; 'loop' repeats the clip; 'fit_to_scene' adjusts playback speed to exactly match the scene. Defaults to 'freeze' when omitted. Requires a video-scene voiceover.\",\n \"enum\": [\n \"freeze\",\n \"loop\",\n \"fit_to_scene\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"mute\": {\n \"default\": false,\n \"description\": \"If True, force the clip silent regardless of 'volume'.\",\n \"type\": \"boolean\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Clip audio volume. 1.0 = source level (default), 0.0 = silent.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"script\": {\n \"description\": \"Optional voiceover: text to speak over the clip. Pair with voice_id. Mutually exclusive with audio_url/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"source\": {\n \"description\": \"Video clip to include. Accepts URL, asset ID, or base64-encoded data.\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": {\n \"description\": \"Must be 'video' for a video-clip scene.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID for text-to-speech. Required when script is provided.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning parameters (speed, pitch, locale) for a script voiceover.\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Use the variant matching the engine backing the chosen voice (e.g. engine_type='elevenlabs' for ElevenLabs-backed voices). The request is rejected if the voice_id is not compatible with the selected engine.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3, stability must be 0, 0.5, or 1.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"default\": 0,\n \"description\": \"Pitch adjustment in semitones. -50 to +50.\",\n \"type\": \"number\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Playback speed multiplier. 0.5 (half speed) to 1.5 (1.5x speed).\",\n \"type\": \"number\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent. Useful when mixing spoken voice with background audio.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"source\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": \"array\"\n },\n \"title\": {\n \"description\": \"Display title for the video in the HeyGen dashboard.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Must be 'studio' for scene-composition video creation.\",\n \"type\": \"string\"\n },\n \"watermark\": {\n \"description\": \"Custom watermark image to overlay on the video (PNG or JPEG). Available as a premium option for select Enterprise customers. To request access, please contact our support team.\",\n \"nullable\": true,\n \"properties\": {\n \"image\": {\n \"description\": \"Image asset to use as the watermark overlay (PNG or JPEG).\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"opacity\": {\n \"default\": 1,\n \"description\": \"Watermark opacity. 0.0 is fully transparent, 1.0 is fully opaque.\",\n \"type\": \"number\"\n },\n \"placement\": {\n \"description\": \"Watermark placement. Defaults to bottom-right with standard margins when omitted.\",\n \"nullable\": true,\n \"properties\": {\n \"offset_x\": {\n \"description\": \"Fine-tune horizontal position. Fraction of frame width; 0.05 shifts 5% rightward, -0.05 shifts 5% leftward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"offset_y\": {\n \"description\": \"Fine-tune vertical position. Fraction of frame height; 0.05 shifts 5% downward, -0.05 shifts 5% upward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"position\": {\n \"default\": \"bottom_right\",\n \"description\": \"Anchor corner for the watermark.\",\n \"enum\": [\n \"top_left\",\n \"top_right\",\n \"bottom_left\",\n \"bottom_right\"\n ],\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"scale\": {\n \"default\": 1,\n \"description\": \"Scale multiplier for the watermark image. 1.0 renders at native size.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"image\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"scenes\"\n ],\n \"type\": \"object\"\n }\n ]\n}", + RequestSchema: "{\n \"description\": \"Discriminated union for POST /v3/videos request body.\",\n \"discriminator\": {\n \"mapping\": {\n \"avatar\": \"#/components/schemas/CreateVideoFromAvatar\",\n \"cinematic_avatar\": \"#/components/schemas/CreateVideoFromCinematicAvatar\",\n \"image\": \"#/components/schemas/CreateVideoFromImage\",\n \"studio\": \"#/components/schemas/CreateVideoFromStudio\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Create a video from a HeyGen avatar (video or photo avatar).\\n\\nProvide an avatar_id to use a previously created avatar. Supports all\\navatar types: studio_avatar, digital_twin, and photo_avatar. Optionally\\nset ``engine`` to select Avatar V for eligible avatars; when omitted, the\\nserver defaults to Avatar IV.\",\n \"properties\": {\n \"aspect_ratio\": {\n \"default\": \"16:9\",\n \"description\": \"Output video aspect ratio. Supported values: '16:9', '9:16', '4:5', '5:4', '1:1', 'auto'. Defaults to '16:9'. 'auto' preserves the source's aspect ratio (avatar source frames or uploaded image), short-edge anchored to the requested resolution and capped at the tier's long edge. Falls back to '16:9' when source dimensions can't be read.\",\n \"enum\": [\n \"16:9\",\n \"9:16\",\n \"4:5\",\n \"5:4\",\n \"1:1\",\n \"auto\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_asset_id\": {\n \"description\": \"HeyGen asset ID of an uploaded audio file. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_url\": {\n \"description\": \"Public URL of an audio file to lip-sync. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"avatar_id\": {\n \"description\": \"HeyGen avatar ID (video avatar or photo avatar look ID).\",\n \"type\": \"string\"\n },\n \"background\": {\n \"description\": \"Background settings for the video.\",\n \"nullable\": true,\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID of the background image. Used when type is 'image'. Mutually exclusive with url.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Background type. 'color' uses a solid hex color; 'image' uses an image from url or asset_id.\",\n \"enum\": [\n \"color\",\n \"image\"\n ],\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the background image. Used when type is 'image'. Mutually exclusive with asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"value\": {\n \"description\": \"Hex color code (e.g. '#ff0000'). Required when type is 'color'.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary ID controlling how custom terms are pronounced in generated speech (for example, saying 'HeyGen' as 'hey-jen'). Applies when the audio is synthesized from `script`; a caller-supplied `audio_url` or `audio_asset_id` is unaffected. Pronunciation is applied to the synthesized audio only, so caption and subtitle text still show the original script wording. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"Caller-defined identifier echoed back in the webhook payload.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL to receive a POST notification when the video is ready.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"caption\": {\n \"description\": \"Caption generation settings. A sidecar subtitle file is always returned via subtitle_url; set 'style' to additionally burn captions into the rendered video.\",\n \"nullable\": true,\n \"properties\": {\n \"file_format\": {\n \"default\": \"srt\",\n \"description\": \"Output format for the sidecar caption file.\",\n \"enum\": [\n \"srt\"\n ],\n \"type\": \"string\"\n },\n \"style\": {\n \"description\": \"Visual style for burning captions into the rendered video. Omit for sidecar-only captions.\",\n \"enum\": [\n \"default\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"engine\": {\n \"description\": \"Engine configuration for video generation. Pass {\\\"type\\\": \\\"avatar_v\\\"} to enable cross-reference-driven animation for higher quality. Check supported_api_engines on the avatar look to confirm eligibility. Defaults to Avatar IV when omitted.\",\n \"discriminator\": {\n \"mapping\": {\n \"avatar_iii\": \"#/components/schemas/AvatarIIIEngineConfig\",\n \"avatar_iv\": \"#/components/schemas/AvatarIVEngineConfig\",\n \"avatar_v\": \"#/components/schemas/AvatarVEngineConfig\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Avatar V engine configuration with cross-reference-driven animation.\",\n \"properties\": {\n \"reference_look_id\": {\n \"description\": \"Optional look to use as the animation reference. When provided, it must be a `digital_twin` look accessible to your workspace and in the same avatar group as `avatar_id` (`studio_avatar` and `photo_avatar` looks are rejected). When omitted, video avatars self-reference and photo avatars select from their group's eligible candidates, preferring digital twins (ready first, then processing / upgrading), then curated public studio looks. A photo avatar whose group has no eligible reference renders directly from its image without one; motion_prompt is rejected in that case. A non-public `digital_twin` reference, whether provided or selected automatically, must also satisfy its group's consent requirements.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_v'. Check supported_api_engines on the avatar look to confirm eligibility.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Avatar IV engine configuration (default behavior).\",\n \"properties\": {\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_iv'.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Avatar III engine configuration.\\n\\nA single engine value that resolves to the right product by the avatar's\\nlook type (mirrors how ``avatar_iv`` already serves both photo and video\\navatars):\\n\\n- video avatar looks (``digital_twin``, ``studio_avatar``) -\\u003e Digital Twin\\n (supports 4K)\\n- ``photo_avatar`` look -\\u003e Photo Avatar (no 4K output)\\n\\nNot supported for raw image input (``type: \\\"image\\\"``).\\n``motion_prompt`` and ``expressiveness`` are not supported with this engine.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_iii'. Resolves to Digital Twin for video avatar looks (digital_twin, studio_avatar) and Photo Avatar for photo_avatar looks; not supported for raw image input. Check supported_api_engines on the avatar look to confirm eligibility.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"expressiveness\": {\n \"description\": \"Avatar expressiveness level. Photo avatars only. Defaults to 'low' when omitted. Avatar IV only; rejected when engine.type is 'avatar_v'.\",\n \"enum\": [\n \"high\",\n \"medium\",\n \"low\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"fit\": {\n \"description\": \"How the subject is fitted to the output canvas. 'cover' scales to fill the frame (may crop edges). 'contain' scales to fit entirely within the frame (may show background). When omitted, the server picks the best option based on the source and canvas orientations.\",\n \"enum\": [\n \"contain\",\n \"cover\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"folder_id\": {\n \"description\": \"Destination folder ID in the caller's workspace, for example one returned by POST /v3/folders. Omit, pass null, or pass an empty string to place the result at the workspace root. The id must name a folder that is not in the trash and that the caller can write to; an unknown id, a folder in another workspace, a project of another kind such as a brand kit, or a trashed folder is rejected with 404 before anything is created. Supported only for single-video creation; batch video items cannot set their own destination folder.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"motion_prompt\": {\n \"description\": \"Natural-language prompt controlling avatar body motion and hand gestures. Supported for photo avatars on either engine, and for video avatars when engine.type is 'avatar_v'. Rejected for video avatars on the default Avatar IV engine, and for photo avatars on 'avatar_v' when the avatar's group has no animation reference (no digital twin or curated reference look).\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"output_format\": {\n \"default\": \"mp4\",\n \"description\": \"Output container. 'webm' returns a video with a transparent background (alpha channel); 'mp4' (default) returns a standard video. 'webm' requires an avatar that supports matting. When 'webm' is selected, any 'background' value is rejected and background removal is applied automatically — the caller does not need to set 'remove_background'.\",\n \"enum\": [\n \"mp4\",\n \"webm\"\n ],\n \"type\": \"string\"\n },\n \"remove_background\": {\n \"description\": \"Remove the avatar background. Video avatars must be trained with matting enabled.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"resolution\": {\n \"description\": \"Output video resolution. Avatar IV and Avatar V render the avatar at up to 1080p: with `4k`, the avatar is composited onto a 4K canvas rather than rendered natively. Native 4K output is available for Avatar III digital twins and studio avatars.\",\n \"enum\": [\n \"4k\",\n \"1080p\",\n \"720p\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"script\": {\n \"description\": \"Text script for the avatar to speak. Pair with voice_id, or omit voice_id when using avatar_id to use the avatar's default voice. Mutually exclusive with audio_url/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"Display title for the video in the HeyGen dashboard.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Must be 'avatar' for avatar-based video creation.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID for text-to-speech. Required when script is provided, unless avatar_id is set (the avatar's default voice is used as fallback).\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning parameters (speed, pitch, locale).\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Use the variant matching the engine backing the chosen voice (e.g. engine_type='elevenlabs' for ElevenLabs-backed voices). The request is rejected if the voice_id is not compatible with the selected engine.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3 or eleven_v4, stability may be any value in [0.0, 1.0].\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\",\n \"eleven_v4\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"default\": 0,\n \"description\": \"Pitch adjustment in semitones. -50 to +50.\",\n \"type\": \"number\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Playback speed multiplier. 0.5 (half speed) to 1.5 (1.5x speed).\",\n \"type\": \"number\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent. Useful when mixing spoken voice with background audio.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"watermark\": {\n \"description\": \"Custom watermark image to overlay on the video (PNG or JPEG). Available as a premium option for select Enterprise customers. To request access, please contact our support team.\",\n \"nullable\": true,\n \"properties\": {\n \"image\": {\n \"description\": \"Image asset to use as the watermark overlay (PNG or JPEG).\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"opacity\": {\n \"default\": 1,\n \"description\": \"Watermark opacity. 0.0 is fully transparent, 1.0 is fully opaque.\",\n \"type\": \"number\"\n },\n \"placement\": {\n \"description\": \"Watermark placement. Defaults to bottom-right with standard margins when omitted.\",\n \"nullable\": true,\n \"properties\": {\n \"offset_x\": {\n \"description\": \"Fine-tune horizontal position. Fraction of frame width; 0.05 shifts 5% rightward, -0.05 shifts 5% leftward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"offset_y\": {\n \"description\": \"Fine-tune vertical position. Fraction of frame height; 0.05 shifts 5% downward, -0.05 shifts 5% upward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"position\": {\n \"default\": \"bottom_right\",\n \"description\": \"Anchor corner for the watermark.\",\n \"enum\": [\n \"top_left\",\n \"top_right\",\n \"bottom_left\",\n \"bottom_right\"\n ],\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"scale\": {\n \"default\": 1,\n \"description\": \"Scale multiplier for the watermark image. 1.0 renders at native size.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"image\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"avatar_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Create a video by animating an arbitrary image.\\n\\nProvide an image via URL, asset ID, or inline base64. The image will be\\nanimated with lip-sync to the provided audio or generated speech.\",\n \"properties\": {\n \"aspect_ratio\": {\n \"default\": \"16:9\",\n \"description\": \"Output video aspect ratio. Supported values: '16:9', '9:16', '4:5', '5:4', '1:1', 'auto'. Defaults to '16:9'. 'auto' preserves the source's aspect ratio (avatar source frames or uploaded image), short-edge anchored to the requested resolution and capped at the tier's long edge. Falls back to '16:9' when source dimensions can't be read.\",\n \"enum\": [\n \"16:9\",\n \"9:16\",\n \"4:5\",\n \"5:4\",\n \"1:1\",\n \"auto\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_asset_id\": {\n \"description\": \"HeyGen asset ID of an uploaded audio file. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_url\": {\n \"description\": \"Public URL of an audio file to lip-sync. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"background\": {\n \"description\": \"Background settings for the video.\",\n \"nullable\": true,\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID of the background image. Used when type is 'image'. Mutually exclusive with url.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Background type. 'color' uses a solid hex color; 'image' uses an image from url or asset_id.\",\n \"enum\": [\n \"color\",\n \"image\"\n ],\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the background image. Used when type is 'image'. Mutually exclusive with asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"value\": {\n \"description\": \"Hex color code (e.g. '#ff0000'). Required when type is 'color'.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary ID controlling how custom terms are pronounced in generated speech (for example, saying 'HeyGen' as 'hey-jen'). Applies when the audio is synthesized from `script`; a caller-supplied `audio_url` or `audio_asset_id` is unaffected. Pronunciation is applied to the synthesized audio only, so caption and subtitle text still show the original script wording. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"Caller-defined identifier echoed back in the webhook payload.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL to receive a POST notification when the video is ready.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"caption\": {\n \"description\": \"Caption generation settings. A sidecar subtitle file is always returned via subtitle_url; set 'style' to additionally burn captions into the rendered video.\",\n \"nullable\": true,\n \"properties\": {\n \"file_format\": {\n \"default\": \"srt\",\n \"description\": \"Output format for the sidecar caption file.\",\n \"enum\": [\n \"srt\"\n ],\n \"type\": \"string\"\n },\n \"style\": {\n \"description\": \"Visual style for burning captions into the rendered video. Omit for sidecar-only captions.\",\n \"enum\": [\n \"default\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"expressiveness\": {\n \"description\": \"Avatar expressiveness level. Photo avatars only. Defaults to 'low' when omitted.\",\n \"enum\": [\n \"high\",\n \"medium\",\n \"low\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"fit\": {\n \"description\": \"How the subject is fitted to the output canvas. 'cover' scales to fill the frame (may crop edges). 'contain' scales to fit entirely within the frame (may show background). When omitted, the server picks the best option based on the source and canvas orientations.\",\n \"enum\": [\n \"contain\",\n \"cover\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"folder_id\": {\n \"description\": \"Destination folder ID in the caller's workspace, for example one returned by POST /v3/folders. Omit, pass null, or pass an empty string to place the result at the workspace root. The id must name a folder that is not in the trash and that the caller can write to; an unknown id, a folder in another workspace, a project of another kind such as a brand kit, or a trashed folder is rejected with 404 before anything is created. Supported only for single-video creation; batch video items cannot set their own destination folder.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"image\": {\n \"description\": \"Image to animate. Accepts URL, asset ID, or base64-encoded data.\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"motion_prompt\": {\n \"description\": \"Natural-language prompt controlling avatar body motion. Photo avatars only.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"output_format\": {\n \"default\": \"mp4\",\n \"description\": \"Output container. 'webm' returns a video with a transparent background (alpha channel); 'mp4' (default) returns a standard video. 'webm' requires an avatar that supports matting. When 'webm' is selected, any 'background' value is rejected and background removal is applied automatically — the caller does not need to set 'remove_background'.\",\n \"enum\": [\n \"mp4\",\n \"webm\"\n ],\n \"type\": \"string\"\n },\n \"remove_background\": {\n \"description\": \"Remove the avatar background. Video avatars must be trained with matting enabled.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"resolution\": {\n \"description\": \"Output video resolution. Avatar IV and Avatar V render the avatar at up to 1080p: with `4k`, the avatar is composited onto a 4K canvas rather than rendered natively. Native 4K output is available for Avatar III digital twins and studio avatars.\",\n \"enum\": [\n \"4k\",\n \"1080p\",\n \"720p\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"script\": {\n \"description\": \"Text script for the avatar to speak. Pair with voice_id, or omit voice_id when using avatar_id to use the avatar's default voice. Mutually exclusive with audio_url/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"Display title for the video in the HeyGen dashboard.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Must be 'image' for image-based video creation.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID for text-to-speech. Required when script is provided, unless avatar_id is set (the avatar's default voice is used as fallback).\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning parameters (speed, pitch, locale).\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Use the variant matching the engine backing the chosen voice (e.g. engine_type='elevenlabs' for ElevenLabs-backed voices). The request is rejected if the voice_id is not compatible with the selected engine.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3 or eleven_v4, stability may be any value in [0.0, 1.0].\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\",\n \"eleven_v4\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"default\": 0,\n \"description\": \"Pitch adjustment in semitones. -50 to +50.\",\n \"type\": \"number\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Playback speed multiplier. 0.5 (half speed) to 1.5 (1.5x speed).\",\n \"type\": \"number\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent. Useful when mixing spoken voice with background audio.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"watermark\": {\n \"description\": \"Custom watermark image to overlay on the video (PNG or JPEG). Available as a premium option for select Enterprise customers. To request access, please contact our support team.\",\n \"nullable\": true,\n \"properties\": {\n \"image\": {\n \"description\": \"Image asset to use as the watermark overlay (PNG or JPEG).\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"opacity\": {\n \"default\": 1,\n \"description\": \"Watermark opacity. 0.0 is fully transparent, 1.0 is fully opaque.\",\n \"type\": \"number\"\n },\n \"placement\": {\n \"description\": \"Watermark placement. Defaults to bottom-right with standard margins when omitted.\",\n \"nullable\": true,\n \"properties\": {\n \"offset_x\": {\n \"description\": \"Fine-tune horizontal position. Fraction of frame width; 0.05 shifts 5% rightward, -0.05 shifts 5% leftward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"offset_y\": {\n \"description\": \"Fine-tune vertical position. Fraction of frame height; 0.05 shifts 5% downward, -0.05 shifts 5% upward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"position\": {\n \"default\": \"bottom_right\",\n \"description\": \"Anchor corner for the watermark.\",\n \"enum\": [\n \"top_left\",\n \"top_right\",\n \"bottom_left\",\n \"bottom_right\"\n ],\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"scale\": {\n \"default\": 1,\n \"description\": \"Scale multiplier for the watermark image. 1.0 renders at native size.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"image\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"image\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Create a video from a text prompt plus avatar and asset references (Cinematic Avatar).\\n\\nCinematic Avatar generates a video from a natural-language ``prompt`` guided by\\nreference content: one to three avatar looks and optional reference assets\\n(images / videos / audio). Unlike the ``avatar`` and ``image`` modes there is\\nno script or voice — motion and speech are driven entirely by the prompt and\\nthe supplied references. Backed by the Seedance generation pipeline.\",\n \"properties\": {\n \"aspect_ratio\": {\n \"default\": \"16:9\",\n \"description\": \"Output aspect ratio. Supported for cinematic_avatar: '16:9', '9:16', '1:1'. Defaults to '16:9'.\",\n \"enum\": [\n \"16:9\",\n \"9:16\",\n \"1:1\"\n ],\n \"type\": \"string\"\n },\n \"auto_duration\": {\n \"default\": false,\n \"description\": \"Let the model choose the video length. When true, omit duration.\",\n \"type\": \"boolean\"\n },\n \"avatar_id\": {\n \"description\": \"Avatar look ID(s) used as visual references. Provide 1 to 3 look IDs.\",\n \"items\": {\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"duration\": {\n \"description\": \"Video length in seconds (4–15). Defaults to 10. Omit when auto_duration is true.\",\n \"nullable\": true,\n \"type\": \"integer\"\n },\n \"enhance_prompt\": {\n \"default\": false,\n \"description\": \"Enable server-side prompt enhancement.\",\n \"type\": \"boolean\"\n },\n \"folder_id\": {\n \"description\": \"Destination folder ID in the caller's workspace, for example one returned by POST /v3/folders. Omit, pass null, or pass an empty string to place the result at the workspace root. The id must name a folder that is not in the trash and that the caller can write to; an unknown id, a folder in another workspace, a project of another kind such as a brand kit, or a trashed folder is rejected with 404 before anything is created. Supported only for single-video creation; batch video items cannot set their own destination folder.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"prompt\": {\n \"description\": \"Natural-language prompt describing the video to generate.\",\n \"type\": \"string\"\n },\n \"references\": {\n \"description\": \"Reference assets (images, videos, or audio) guiding the generation. Each accepts a URL, an asset_id, or inline base64. Combined limits: at most 3 videos and 9 images across avatars and references.\",\n \"items\": {\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"nullable\": true,\n \"type\": \"array\"\n },\n \"resolution\": {\n \"default\": \"720p\",\n \"description\": \"Output resolution. Supported for cinematic_avatar: '720p', '1080p'. Defaults to '720p'.\",\n \"enum\": [\n \"720p\",\n \"1080p\"\n ],\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"Display title for the video in the HeyGen dashboard.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Must be 'cinematic_avatar' for prompt-and-reference video creation.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"prompt\",\n \"avatar_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Create a single video by composing an ordered list of whole-frame scenes.\\n\\nThe server owns layout and center-crops each scene to the global output\\ncanvas. Output settings are global (one per request); a single video_id is\\nreturned and rendering is all-or-nothing. MP4 only in v1 — the output\\ncontainer is fixed and ``output_format`` is not exposed.\",\n \"properties\": {\n \"aspect_ratio\": {\n \"default\": \"16:9\",\n \"description\": \"Global output aspect ratio. Supported values: '16:9', '9:16', '4:5', '5:4', '1:1', 'auto'. Defaults to '16:9'. Each scene is center-cropped to this canvas.\",\n \"enum\": [\n \"16:9\",\n \"9:16\",\n \"4:5\",\n \"5:4\",\n \"1:1\",\n \"auto\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary ID controlling how custom terms are pronounced in generated speech (for example, saying 'HeyGen' as 'hey-jen'). Applies to every scene whose audio is synthesized from a script; scenes that supply their own audio URL or audio asset are unaffected. Pronunciation is applied to the synthesized audio only, so caption and subtitle text still show the original script wording. Discover IDs via GET /v3/brand-glossaries.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_id\": {\n \"description\": \"Caller-defined identifier echoed back in the webhook payload.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"callback_url\": {\n \"description\": \"Webhook URL to receive a POST notification when the video is ready.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"caption\": {\n \"description\": \"Caption generation settings. A sidecar subtitle file is always returned via subtitle_url; set 'style' to additionally burn captions into the rendered video.\",\n \"nullable\": true,\n \"properties\": {\n \"file_format\": {\n \"default\": \"srt\",\n \"description\": \"Output format for the sidecar caption file.\",\n \"enum\": [\n \"srt\"\n ],\n \"type\": \"string\"\n },\n \"style\": {\n \"description\": \"Visual style for burning captions into the rendered video. Omit for sidecar-only captions.\",\n \"enum\": [\n \"default\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"folder_id\": {\n \"description\": \"Destination folder ID in the caller's workspace, for example one returned by POST /v3/folders. Omit, pass null, or pass an empty string to place the result at the workspace root. The id must name a folder that is not in the trash and that the caller can write to; an unknown id, a folder in another workspace, a project of another kind such as a brand kit, or a trashed folder is rejected with 404 before anything is created. Supported only for single-video creation; batch video items cannot set their own destination folder.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"resolution\": {\n \"description\": \"Global output video resolution. Avatar IV and Avatar V render the avatar at up to 1080p: with `4k`, the avatar is composited onto a 4K canvas rather than rendered natively. Native 4K output is available for Avatar III digital twins and studio avatars.\",\n \"enum\": [\n \"4k\",\n \"1080p\",\n \"720p\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"scenes\": {\n \"description\": \"Ordered list of whole-frame scenes to concatenate. Each scene is one of 'avatar_video', 'image', or 'video'. Must contain 1 to 50 scenes.\",\n \"items\": {\n \"discriminator\": {\n \"mapping\": {\n \"avatar_video\": \"#/components/schemas/AvatarVideoScene\",\n \"image\": \"#/components/schemas/ImageScene\",\n \"video\": \"#/components/schemas/VideoScene\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"A whole-frame speaking scene backed by an avatar.\",\n \"properties\": {\n \"input\": {\n \"description\": \"Scene source ('type': 'avatar'): an avatar_id plus one audio source. The avatar_id accepts any avatar look — video avatars and photo avatars alike (pass a photo avatar's look id to get a talking photo). The scene duration is derived server-side from the audio.\",\n \"properties\": {\n \"audio_asset_id\": {\n \"description\": \"HeyGen asset ID of an uploaded audio file. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_url\": {\n \"description\": \"Public URL of an audio file to lip-sync. Mutually exclusive with script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"avatar_id\": {\n \"description\": \"HeyGen avatar ID (video avatar or photo avatar look ID).\",\n \"type\": \"string\"\n },\n \"background\": {\n \"description\": \"Optional scene background composited behind the avatar. Color-only in v1: pass {\\\"type\\\": \\\"color\\\", \\\"color\\\": \\\"#RRGGBB\\\"}. Other background types are not yet supported.\",\n \"discriminator\": {\n \"mapping\": {\n \"color\": \"#/components/schemas/StudioColorBackgroundInput\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Solid-color background for an ``avatar_video`` studio scene.\\n\\nStudio v1 supports solid-color backgrounds.\",\n \"properties\": {\n \"color\": {\n \"description\": \"Background color as a 6-digit hex string, e.g. '#1a2b3c'.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Background type discriminator. Must be 'color'.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"color\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"engine\": {\n \"description\": \"Engine configuration for video generation. Pass {\\\"type\\\": \\\"avatar_v\\\"} to enable cross-reference-driven animation for higher quality. Check supported_api_engines on the avatar look to confirm eligibility. Defaults to Avatar IV when omitted.\",\n \"discriminator\": {\n \"mapping\": {\n \"avatar_iii\": \"#/components/schemas/AvatarIIIEngineConfig\",\n \"avatar_iv\": \"#/components/schemas/AvatarIVEngineConfig\",\n \"avatar_v\": \"#/components/schemas/AvatarVEngineConfig\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Avatar V engine configuration with cross-reference-driven animation.\",\n \"properties\": {\n \"reference_look_id\": {\n \"description\": \"Optional look to use as the animation reference. When provided, it must be a `digital_twin` look accessible to your workspace and in the same avatar group as `avatar_id` (`studio_avatar` and `photo_avatar` looks are rejected). When omitted, video avatars self-reference and photo avatars select from their group's eligible candidates, preferring digital twins (ready first, then processing / upgrading), then curated public studio looks. A photo avatar whose group has no eligible reference renders directly from its image without one; motion_prompt is rejected in that case. A non-public `digital_twin` reference, whether provided or selected automatically, must also satisfy its group's consent requirements.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_v'. Check supported_api_engines on the avatar look to confirm eligibility.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Avatar IV engine configuration (default behavior).\",\n \"properties\": {\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_iv'.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Avatar III engine configuration.\\n\\nA single engine value that resolves to the right product by the avatar's\\nlook type (mirrors how ``avatar_iv`` already serves both photo and video\\navatars):\\n\\n- video avatar looks (``digital_twin``, ``studio_avatar``) -\\u003e Digital Twin\\n (supports 4K)\\n- ``photo_avatar`` look -\\u003e Photo Avatar (no 4K output)\\n\\nNot supported for raw image input (``type: \\\"image\\\"``).\\n``motion_prompt`` and ``expressiveness`` are not supported with this engine.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Engine type discriminator. Must be 'avatar_iii'. Resolves to Digital Twin for video avatar looks (digital_twin, studio_avatar) and Photo Avatar for photo_avatar looks; not supported for raw image input. Check supported_api_engines on the avatar look to confirm eligibility.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"expressiveness\": {\n \"description\": \"Avatar expressiveness level. Photo avatars only. Defaults to 'low' when omitted. Avatar IV only; rejected when engine.type is 'avatar_v'.\",\n \"enum\": [\n \"high\",\n \"medium\",\n \"low\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"motion_prompt\": {\n \"description\": \"Natural-language prompt controlling avatar body motion and hand gestures. Supported for photo avatars on either engine, and for video avatars when engine.type is 'avatar_v'. Rejected for video avatars on the default Avatar IV engine, and for photo avatars on 'avatar_v' when the avatar's group has no animation reference (no digital twin or curated reference look).\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"script\": {\n \"description\": \"Text script for the avatar to speak. Pair with voice_id, or omit voice_id when using avatar_id to use the avatar's default voice. Mutually exclusive with audio_url/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Must be 'avatar' for an avatar-driven scene source.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID for text-to-speech. Required when script is provided, unless avatar_id is set (the avatar's default voice is used as fallback).\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning parameters (speed, pitch, locale).\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Use the variant matching the engine backing the chosen voice (e.g. engine_type='elevenlabs' for ElevenLabs-backed voices). The request is rejected if the voice_id is not compatible with the selected engine.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3 or eleven_v4, stability may be any value in [0.0, 1.0].\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\",\n \"eleven_v4\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"default\": 0,\n \"description\": \"Pitch adjustment in semitones. -50 to +50.\",\n \"type\": \"number\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Playback speed multiplier. 0.5 (half speed) to 1.5 (1.5x speed).\",\n \"type\": \"number\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent. Useful when mixing spoken voice with background audio.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"avatar_id\"\n ],\n \"type\": \"object\"\n },\n \"type\": {\n \"description\": \"Must be 'avatar_video' for an avatar speaking scene.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"input\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A whole-frame still-image scene: either silent (held for ``duration``) or narrated.\\n\\nExactly one mode must be chosen:\\n- silent: set ``duration`` (seconds) and no audio source.\\n- narrated: set exactly one audio source (script + voice_id, audio_url, or\\n audio_asset_id) and omit ``duration`` — the scene length follows the audio.\",\n \"properties\": {\n \"audio_asset_id\": {\n \"description\": \"Narrated mode: HeyGen asset ID of an uploaded audio file. Mutually exclusive with duration/script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_url\": {\n \"description\": \"Narrated mode: public URL of an audio file to play over the image. Mutually exclusive with duration/script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"duration\": {\n \"description\": \"Silent mode: hold the still image for this many seconds. Mutually exclusive with any audio source. Must be \\u003e 0 and \\u003c= 300.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"script\": {\n \"description\": \"Narrated mode: text to speak over the image. Pair with voice_id. Mutually exclusive with duration/audio_url/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"source\": {\n \"description\": \"Still image to display. Accepts URL, asset ID, or base64-encoded data.\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": {\n \"description\": \"Must be 'image' for a still-image scene.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID for text-to-speech. Required when script is provided.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning parameters (speed, pitch, locale).\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Use the variant matching the engine backing the chosen voice (e.g. engine_type='elevenlabs' for ElevenLabs-backed voices). The request is rejected if the voice_id is not compatible with the selected engine.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3 or eleven_v4, stability may be any value in [0.0, 1.0].\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\",\n \"eleven_v4\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"default\": 0,\n \"description\": \"Pitch adjustment in semitones. -50 to +50.\",\n \"type\": \"number\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Playback speed multiplier. 0.5 (half speed) to 1.5 (1.5x speed).\",\n \"type\": \"number\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent. Useful when mixing spoken voice with background audio.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"source\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A whole-frame scene backed by an existing video clip.\\n\\nOptional ``playback`` exposes the audio volume / mute capability; when\\nomitted the clip plays at its source volume.\\n\\nFor optional voiceover / narration, supply at most one audio source\\n(``script`` + ``voice_id``, ``audio_url``, or ``audio_asset_id``) — the *same*\\naudio inputs a narrated ``image`` scene accepts. When present, the narration\\ndrives the scene length and ``playback.mode`` controls whether the clip\\nfreezes, loops, or changes speed to fit that duration. When omitted the clip\\nplays full-length as before. The clip's own audio level is still governed by\\n``playback`` (the two compose).\",\n \"properties\": {\n \"audio_asset_id\": {\n \"description\": \"Optional voiceover: HeyGen asset ID of an uploaded audio file. Mutually exclusive with script/audio_url.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"audio_url\": {\n \"description\": \"Optional voiceover: public URL of an audio file to play over the clip. Mutually exclusive with script/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"playback\": {\n \"description\": \"Optional playback capability: narrated-clip alignment 'mode', audio 'volume' (0.0–1.0), and 'mute'. Omit to use freeze alignment and keep the clip's source volume.\",\n \"nullable\": true,\n \"properties\": {\n \"mode\": {\n \"description\": \"How a narrated clip aligns to the voiceover-driven scene duration. 'freeze' plays once and holds the last frame; 'loop' repeats the clip; 'fit_to_scene' adjusts playback speed to exactly match the scene. Defaults to 'freeze' when omitted. Requires a video-scene voiceover.\",\n \"enum\": [\n \"freeze\",\n \"loop\",\n \"fit_to_scene\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"mute\": {\n \"default\": false,\n \"description\": \"If True, force the clip silent regardless of 'volume'.\",\n \"type\": \"boolean\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Clip audio volume. 1.0 = source level (default), 0.0 = silent.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"script\": {\n \"description\": \"Optional voiceover: text to speak over the clip. Pair with voice_id. Mutually exclusive with audio_url/audio_asset_id.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"source\": {\n \"description\": \"Video clip to include. Accepts URL, asset ID, or base64-encoded data.\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": {\n \"description\": \"Must be 'video' for a video-clip scene.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID for text-to-speech. Required when script is provided.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning parameters (speed, pitch, locale) for a script voiceover.\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Use the variant matching the engine backing the chosen voice (e.g. engine_type='elevenlabs' for ElevenLabs-backed voices). The request is rejected if the voice_id is not compatible with the selected engine.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3 or eleven_v4, stability may be any value in [0.0, 1.0].\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\",\n \"eleven_v4\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"default\": 0,\n \"description\": \"Pitch adjustment in semitones. -50 to +50.\",\n \"type\": \"number\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Playback speed multiplier. 0.5 (half speed) to 1.5 (1.5x speed).\",\n \"type\": \"number\"\n },\n \"volume\": {\n \"default\": 1,\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent. Useful when mixing spoken voice with background audio.\",\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"source\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": \"array\"\n },\n \"title\": {\n \"description\": \"Display title for the video in the HeyGen dashboard.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Must be 'studio' for scene-composition video creation.\",\n \"type\": \"string\"\n },\n \"watermark\": {\n \"description\": \"Custom watermark image to overlay on the video (PNG or JPEG). Available as a premium option for select Enterprise customers. To request access, please contact our support team.\",\n \"nullable\": true,\n \"properties\": {\n \"image\": {\n \"description\": \"Image asset to use as the watermark overlay (PNG or JPEG).\",\n \"discriminator\": {\n \"mapping\": {\n \"asset_id\": \"#/components/schemas/AssetId\",\n \"base64\": \"#/components/schemas/AssetBase64\",\n \"url\": \"#/components/schemas/AssetUrl\"\n },\n \"propertyName\": \"type\"\n },\n \"oneOf\": [\n {\n \"description\": \"Asset input via publicly accessible HTTPS URL.\",\n \"properties\": {\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"Publicly accessible HTTPS URL for the asset\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"url\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via HeyGen asset ID from the asset upload endpoint.\",\n \"properties\": {\n \"asset_id\": {\n \"description\": \"HeyGen asset ID from the asset upload endpoint\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"asset_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Asset input via base64-encoded content.\",\n \"properties\": {\n \"data\": {\n \"description\": \"Base64-encoded file content\",\n \"type\": \"string\"\n },\n \"media_type\": {\n \"description\": \"MIME type of the encoded content (e.g. \\\"image/png\\\")\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"Input type discriminator\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"type\",\n \"media_type\",\n \"data\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"opacity\": {\n \"default\": 1,\n \"description\": \"Watermark opacity. 0.0 is fully transparent, 1.0 is fully opaque.\",\n \"type\": \"number\"\n },\n \"placement\": {\n \"description\": \"Watermark placement. Defaults to bottom-right with standard margins when omitted.\",\n \"nullable\": true,\n \"properties\": {\n \"offset_x\": {\n \"description\": \"Fine-tune horizontal position. Fraction of frame width; 0.05 shifts 5% rightward, -0.05 shifts 5% leftward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"offset_y\": {\n \"description\": \"Fine-tune vertical position. Fraction of frame height; 0.05 shifts 5% downward, -0.05 shifts 5% upward.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"position\": {\n \"default\": \"bottom_right\",\n \"description\": \"Anchor corner for the watermark.\",\n \"enum\": [\n \"top_left\",\n \"top_right\",\n \"bottom_left\",\n \"bottom_right\"\n ],\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"scale\": {\n \"default\": 1,\n \"description\": \"Scale multiplier for the watermark image. 1.0 renders at native size.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"image\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"type\",\n \"scenes\"\n ],\n \"type\": \"object\"\n }\n ]\n}", ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"properties\": {\n \"output_format\": {\n \"default\": \"mp4\",\n \"description\": \"Resolved output format for the video.\",\n \"enum\": [\n \"mp4\",\n \"webm\"\n ],\n \"type\": \"string\"\n },\n \"status\": {\n \"description\": \"Initial video status (e.g. 'waiting').\",\n \"type\": \"string\"\n },\n \"video_id\": {\n \"description\": \"Unique identifier for the created video.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"video_id\",\n \"status\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", Endpoint: "/v3/videos", Method: "POST", @@ -249,8 +249,8 @@ var VideoScenesGet = &command.Spec{ Group: "video", Name: "scenes get", Summary: "Get Video Scenes", - Description: "Returns the video's scenes together with the video-level context needed to use them. Describes the video as it stands now, including any edits made in the editor after it was created. The scene list is never paginated.\n\nEach scene splits by the role a thing plays: `background` fills the frame, `elements` are placed within it, and `script` is the audio delivered over it. A whole-frame image or clip lands in `background`, so code that reads only `elements` misses it.\n\nElement types are an open set: treat an unrecognized type as an element to skip rather than an error, and expect a type value to become more specific over time. Every element a scene places appears in `elements`, so the count is always truthful, but only `avatar`, `image` and `video` are described in full; `group` and `mask` carry their children; the rest carry an `id` and a `type` and nothing more.\n\n**What this does not describe.** A video may contain more than this response expresses, and a video rebuilt from it will differ in these respects: element geometry (position, size, opacity); the text inside a text element; scene and element animations and scene effects; per-scene caption styling, where only whether captions are enabled is reported; background audio, which is video-level and belongs to no scene, so a rebuild loses the music; and some per-avatar values, which this version does not return.", - ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"description\": \"The composite document: one video's context and every one of its scenes.\",\n \"properties\": {\n \"scenes\": {\n \"description\": \"Every scene in the video, in the video's own order, which is playback order unless the video branches. Never paginated.\",\n \"items\": {\n \"description\": \"One scene, in the video's own order.\\n\\nThat is playback order for a linear video, which is the ordinary case. A branching video plays as\\na walk over branch targets instead, so its scenes are still all here and still ordered, but the\\norder is not the sequence a viewer sees. Branching is not otherwise described by this version.\\n\\nThe three content fields split by the role a thing plays rather than by its kind:\\n``background`` fills the frame, ``elements`` are placed within it, ``script`` carries\\nthe speech delivered over it. The background is visual too, so ``elements`` means \\\"the\\nones the scene places\\\", not \\\"the visual ones\\\".\",\n \"properties\": {\n \"background\": {\n \"description\": \"What fills the frame behind this scene's elements.\",\n \"discriminator\": {\n \"mapping\": {\n \"color\": \"#/components/schemas/ColorBackground\",\n \"image\": \"#/components/schemas/ImageBackground\",\n \"video\": \"#/components/schemas/VideoBackground\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"A solid colour filling the frame behind the scene's elements.\",\n \"properties\": {\n \"color\": {\n \"description\": \"Hex colour, e.g. '#f6f6fc'.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"color\",\n \"description\": \"Background type discriminator.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"color\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A whole-frame image.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"image\",\n \"description\": \"Background type discriminator.\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the image. Present when the video records a link for this background and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A whole-frame video clip.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"playback\": {\n \"description\": \"How the clip is reconciled to the scene. Same values as a video element's playback.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"speed_multiplier\": {\n \"description\": \"Playback speed multiplier. 2.0 = twice as fast.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"trim\": {\n \"description\": \"Seconds trimmed from each end of the clip.\",\n \"nullable\": true,\n \"properties\": {\n \"end_offset\": {\n \"description\": \"Seconds trimmed from the end of the clip.\",\n \"type\": \"number\"\n },\n \"start_offset\": {\n \"description\": \"Seconds trimmed from the start of the clip.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"start_offset\",\n \"end_offset\"\n ],\n \"type\": \"object\"\n },\n \"type\": {\n \"default\": \"video\",\n \"description\": \"Background type discriminator.\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the clip. Present when the video records a link for this background and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"volume\": {\n \"description\": \"Clip audio volume. 1.0 = source level, 0.0 = silent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"elements\": {\n \"description\": \"Every element this scene places, in the video's own order.\",\n \"items\": {\n \"anyOf\": [\n {\n \"description\": \"An avatar performing the scene's speech.\\n\\nCarries what a v3 create request can set per avatar, so what is read back here is what can be\\nsent back. Fields are added as callers need them, and a new one is not a breaking change.\\n\\nOutput resolution is a whole-video setting, reported on the video rather than per avatar.\",\n \"properties\": {\n \"avatar_id\": {\n \"description\": \"The avatar look performing this scene.\",\n \"type\": \"string\"\n },\n \"engine\": {\n \"description\": \"Generation engine for this avatar. Absent when the video does not record one and none can be determined.\",\n \"enum\": [\n \"avatar_v\",\n \"avatar_iv\",\n \"avatar_iii\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"expressiveness\": {\n \"description\": \"Expressiveness level. Absent when left at the default.\",\n \"enum\": [\n \"high\",\n \"medium\",\n \"low\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"motion_prompt\": {\n \"description\": \"Motion description authored for this avatar.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"avatar\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\",\n \"avatar_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"An image placed in the scene.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"image\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the image. Present when the video records a link for this element and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed; the element is still reported either way, so an image with no URL is one this read could not link to rather than an element this API does not describe. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A video clip placed in the scene.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"playback\": {\n \"description\": \"How the clip is reconciled to the scene: 'freeze' holds the last frame, 'loop' repeats it, 'fit_to_scene' changes speed to match, 'full_video' plays it whole and drives the scene's length.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"speed_multiplier\": {\n \"description\": \"Playback speed multiplier. 2.0 = twice as fast.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"trim\": {\n \"description\": \"Seconds trimmed from each end of the clip.\",\n \"nullable\": true,\n \"properties\": {\n \"end_offset\": {\n \"description\": \"Seconds trimmed from the end of the clip.\",\n \"type\": \"number\"\n },\n \"start_offset\": {\n \"description\": \"Seconds trimmed from the start of the clip.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"start_offset\",\n \"end_offset\"\n ],\n \"type\": \"object\"\n },\n \"type\": {\n \"default\": \"video\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the clip. Present when the video records a link for this element and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"volume\": {\n \"description\": \"Clip audio volume. 1.0 = source level, 0.0 = silent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"An element this version names but does not describe.\\n\\nExists so a scene's ``elements`` array is never a filtered view presented as a complete one: a\\ncaller iterating it sees every element the scene places. Carries exactly two fields: which\\nelement this is, and what kind of thing it is.\\n\\n**Absence of content is not by itself proof that an element is undescribed.** A described\\ncategory can serialize with nothing but an id and a type when every optional field happens to be\\nunavailable, and one case is reachable today: an image with no usable link carries no ``url``,\\nleaving ``{id, type}``. Read the ``type`` to tell them apart.\\n``avatar``, ``image`` and ``video`` are described, so a bare one of those is a described element\\nmissing an optional value rather than an unexpanded node.\\n\\nThe ``id`` matters most here. Two masked elements in one scene are otherwise identical on the\\nwire, so without it a caller can see that the scene places two things and nothing else about\\neither.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"The element's category. An **open set**: new values may be added, so treat an unrecognized one as an element to skip rather than an error. Several kinds of element can share one value, so a category says what an element is rather than naming a single underlying kind. Every value is permanent except 'other', which means 'a kind of element this API does not yet categorise' and may be replaced by a more specific category later.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\",\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"An element that holds other elements: a group, or a mask.\\n\\nA scene lists the container, not its contents, so a scene holding one group of five images\\nlists a single element while five things render. Walking a scene's composition means recursing\\ninto ``children``.\\n\\nA mask matters more here than a group: a group conveys linkage this response does not express,\\nwhile a mask conveys clipping, so treating a masked image as a plain image asserts a\\ncomposition that renders differently in kind.\",\n \"properties\": {\n \"children\": {\n \"description\": \"The elements this container holds, each described exactly as a top-level element would be.\",\n \"items\": {\n \"anyOf\": [\n {\n \"description\": \"An avatar performing the scene's speech.\\n\\nCarries what a v3 create request can set per avatar, so what is read back here is what can be\\nsent back. Fields are added as callers need them, and a new one is not a breaking change.\\n\\nOutput resolution is a whole-video setting, reported on the video rather than per avatar.\",\n \"properties\": {\n \"avatar_id\": {\n \"description\": \"The avatar look performing this scene.\",\n \"type\": \"string\"\n },\n \"engine\": {\n \"description\": \"Generation engine for this avatar. Absent when the video does not record one and none can be determined.\",\n \"enum\": [\n \"avatar_v\",\n \"avatar_iv\",\n \"avatar_iii\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"expressiveness\": {\n \"description\": \"Expressiveness level. Absent when left at the default.\",\n \"enum\": [\n \"high\",\n \"medium\",\n \"low\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"motion_prompt\": {\n \"description\": \"Motion description authored for this avatar.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"avatar\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\",\n \"avatar_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"An image placed in the scene.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"image\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the image. Present when the video records a link for this element and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed; the element is still reported either way, so an image with no URL is one this read could not link to rather than an element this API does not describe. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A video clip placed in the scene.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"playback\": {\n \"description\": \"How the clip is reconciled to the scene: 'freeze' holds the last frame, 'loop' repeats it, 'fit_to_scene' changes speed to match, 'full_video' plays it whole and drives the scene's length.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"speed_multiplier\": {\n \"description\": \"Playback speed multiplier. 2.0 = twice as fast.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"trim\": {\n \"description\": \"Seconds trimmed from each end of the clip.\",\n \"nullable\": true,\n \"properties\": {\n \"end_offset\": {\n \"description\": \"Seconds trimmed from the end of the clip.\",\n \"type\": \"number\"\n },\n \"start_offset\": {\n \"description\": \"Seconds trimmed from the start of the clip.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"start_offset\",\n \"end_offset\"\n ],\n \"type\": \"object\"\n },\n \"type\": {\n \"default\": \"video\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the clip. Present when the video records a link for this element and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"volume\": {\n \"description\": \"Clip audio volume. 1.0 = source level, 0.0 = silent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"An element this version names but does not describe.\\n\\nExists so a scene's ``elements`` array is never a filtered view presented as a complete one: a\\ncaller iterating it sees every element the scene places. Carries exactly two fields: which\\nelement this is, and what kind of thing it is.\\n\\n**Absence of content is not by itself proof that an element is undescribed.** A described\\ncategory can serialize with nothing but an id and a type when every optional field happens to be\\nunavailable, and one case is reachable today: an image with no usable link carries no ``url``,\\nleaving ``{id, type}``. Read the ``type`` to tell them apart.\\n``avatar``, ``image`` and ``video`` are described, so a bare one of those is a described element\\nmissing an optional value rather than an unexpanded node.\\n\\nThe ``id`` matters most here. Two masked elements in one scene are otherwise identical on the\\nwire, so without it a caller can see that the scene places two things and nothing else about\\neither.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"The element's category. An **open set**: new values may be added, so treat an unrecognized one as an element to skip rather than an error. Several kinds of element can share one value, so a category says what an element is rather than naming a single underlying kind. Every value is permanent except 'other', which means 'a kind of element this API does not yet categorise' and may be replaced by a more specific category later.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\",\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Circular reference to ContainerElement\"\n }\n ]\n },\n \"type\": \"array\"\n },\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"The container's category, e.g. 'group' or 'mask'.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\",\n \"type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": \"array\"\n },\n \"id\": {\n \"description\": \"Identifier of this scene within the video.\",\n \"type\": \"string\"\n },\n \"script\": {\n \"description\": \"This scene's audio sources in order, each one synthesized speech, uploaded audio, or a silence. The create API's per-scene `script` is narrower: it is the text of a single one of these entries.\",\n \"items\": {\n \"anyOf\": [\n {\n \"description\": \"Synthesized speech: a script delivered by a voice.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this script entry within the video. Not to be parsed, sorted, or assumed to encode anything. The same id under two scenes means one entry is shared between them: its text belongs to both, and concatenating both scenes' scripts would synthesize the shared words twice.\",\n \"type\": \"string\"\n },\n \"text\": {\n \"description\": \"The script, verbatim, including inline markup. Empty when the scene was left unfinished.\",\n \"type\": \"string\"\n },\n \"trim_to_speech\": {\n \"description\": \"Whether leading and trailing silence is trimmed to the spoken window.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"type\": {\n \"default\": \"text\",\n \"description\": \"Script entry type discriminator.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"The voice delivering this script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning applied to this script.\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Present only when the video pins an engine this API exposes; absent when the engine is left for the server to pick.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3, stability must be 0, 0.5, or 1.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"description\": \"Pitch adjustment in semitones.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"speed\": {\n \"description\": \"Playback speed multiplier.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"volume\": {\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"id\",\n \"text\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Uploaded audio played as the scene's speech.\",\n \"properties\": {\n \"audio_url\": {\n \"description\": \"URL of the audio. Present when the video records a link for this entry and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"id\": {\n \"description\": \"Identifier of this script entry within the video.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"audio\",\n \"description\": \"Script entry type discriminator.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A held silence, with no words spoken over the scene.\",\n \"properties\": {\n \"duration\": {\n \"description\": \"How long the silence is held, in seconds. Absent where the number does not describe what renders — on a scene whose video clip plays in full, the clip's own length governs the scene.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"id\": {\n \"description\": \"Identifier of this script entry within the video.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"silence\",\n \"description\": \"Script entry type discriminator.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": \"array\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n \"type\": \"array\"\n },\n \"video\": {\n \"description\": \"Video-level context for the scenes.\",\n \"properties\": {\n \"aspect_ratio\": {\n \"description\": \"The video's aspect ratio.\",\n \"enum\": [\n \"16:9\",\n \"9:16\",\n \"4:5\",\n \"5:4\",\n \"1:1\",\n \"auto\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary applied to this video's speech.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"caption\": {\n \"description\": \"Caption configuration for the video.\",\n \"nullable\": true,\n \"properties\": {\n \"enabled\": {\n \"description\": \"Whether captions are enabled for this video.\",\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"enabled\"\n ],\n \"type\": \"object\"\n },\n \"resolution\": {\n \"description\": \"The video's output resolution, in the same values a create request accepts. Absent when the video's size matches none of them, which is a size a create request could not have asked for and cannot reproduce.\",\n \"enum\": [\n \"4k\",\n \"1080p\",\n \"720p\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"The video's current title. Editor saves update it.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"video_id\": {\n \"description\": \"The video these scenes belong to.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"video_id\",\n \"video\"\n ],\n \"type\": \"object\"\n },\n \"error\": {\n \"nullable\": true,\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", + Description: "Returns the video's scenes together with the video-level context needed to use them. Describes the video as it stands now, including any edits made in the editor after it was created. The scene list is never paginated. The response includes an opaque `edit_version` for optimistic concurrency. A video whose editor document is still being prepared returns `409 resource_not_ready`; retry after the video advances.\n\nEach scene splits by the role a thing plays: `background` fills the frame, `elements` are placed within it, and `script` is the audio delivered over it. A whole-frame image or clip lands in `background`, so code that reads only `elements` misses it.\n\nElement types are an open set: treat an unrecognized type as an element to skip rather than an error, and expect a type value to become more specific over time. Every element a scene places appears in `elements`, so the count is always truthful, but only `avatar`, `image` and `video` are described in full; `group` and `mask` carry their children; the rest carry an `id` and a `type` and nothing more.\n\n**What this does not describe.** A video may contain more than this response expresses, and a video rebuilt from it will differ in these respects: element geometry (position, size, opacity); the text inside a text element; scene and element animations and scene effects; per-scene caption styling, where only whether captions are enabled is reported; background audio, which is video-level and belongs to no scene, so a rebuild loses the music; and some per-avatar values, which this version does not return.", + ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"description\": \"The composite document: one video's context and every one of its scenes.\",\n \"properties\": {\n \"edit_version\": {\n \"description\": \"Opaque version of the editor document returned by this request. Pass it unchanged when requesting an edit so the server can reject a document that changed before submission. This submission-time check does not lock the document while the asynchronous agent applies the edit.\",\n \"type\": \"string\"\n },\n \"scenes\": {\n \"description\": \"Every scene in the video, in the video's own order, which is playback order unless the video branches. Never paginated.\",\n \"items\": {\n \"description\": \"One scene, in the video's own order.\\n\\nThat is playback order for a linear video, which is the ordinary case. A branching video plays as\\na walk over branch targets instead, so its scenes are still all here and still ordered, but the\\norder is not the sequence a viewer sees. Branching is not otherwise described by this version.\\n\\nThe three content fields split by the role a thing plays rather than by its kind:\\n``background`` fills the frame, ``elements`` are placed within it, ``script`` carries\\nthe speech delivered over it. The background is visual too, so ``elements`` means \\\"the\\nones the scene places\\\", not \\\"the visual ones\\\".\",\n \"properties\": {\n \"background\": {\n \"description\": \"What fills the frame behind this scene's elements.\",\n \"discriminator\": {\n \"mapping\": {\n \"color\": \"#/components/schemas/ColorBackground\",\n \"image\": \"#/components/schemas/ImageBackground\",\n \"video\": \"#/components/schemas/VideoBackground\"\n },\n \"propertyName\": \"type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"A solid colour filling the frame behind the scene's elements.\",\n \"properties\": {\n \"color\": {\n \"description\": \"Hex colour, e.g. '#f6f6fc'.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"color\",\n \"description\": \"Background type discriminator.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"color\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A whole-frame image.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"image\",\n \"description\": \"Background type discriminator.\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the image. Present when the video records a link for this background and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A whole-frame video clip.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"playback\": {\n \"description\": \"How the clip is reconciled to the scene. Same values as a video element's playback.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"speed_multiplier\": {\n \"description\": \"Playback speed multiplier. 2.0 = twice as fast.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"trim\": {\n \"description\": \"Seconds trimmed from each end of the clip.\",\n \"nullable\": true,\n \"properties\": {\n \"end_offset\": {\n \"description\": \"Seconds trimmed from the end of the clip.\",\n \"type\": \"number\"\n },\n \"start_offset\": {\n \"description\": \"Seconds trimmed from the start of the clip.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"start_offset\",\n \"end_offset\"\n ],\n \"type\": \"object\"\n },\n \"type\": {\n \"default\": \"video\",\n \"description\": \"Background type discriminator.\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the clip. Present when the video records a link for this background and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"volume\": {\n \"description\": \"Clip audio volume. 1.0 = source level, 0.0 = silent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"elements\": {\n \"description\": \"Every element this scene places, in the video's own order.\",\n \"items\": {\n \"anyOf\": [\n {\n \"description\": \"An avatar performing the scene's speech.\\n\\nCarries what a v3 create request can set per avatar, so what is read back here is what can be\\nsent back. Fields are added as callers need them, and a new one is not a breaking change.\\n\\nOutput resolution is a whole-video setting, reported on the video rather than per avatar.\",\n \"properties\": {\n \"avatar_id\": {\n \"description\": \"The avatar look performing this scene.\",\n \"type\": \"string\"\n },\n \"engine\": {\n \"description\": \"Generation engine for this avatar. Absent when the video does not record one and none can be determined.\",\n \"enum\": [\n \"avatar_v\",\n \"avatar_iv\",\n \"avatar_iii\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"expressiveness\": {\n \"description\": \"Expressiveness level. Absent when left at the default.\",\n \"enum\": [\n \"high\",\n \"medium\",\n \"low\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"motion_prompt\": {\n \"description\": \"Motion description authored for this avatar.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"avatar\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\",\n \"avatar_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"An image placed in the scene.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"image\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the image. Present when the video records a link for this element and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed; the element is still reported either way, so an image with no URL is one this read could not link to rather than an element this API does not describe. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A video clip placed in the scene.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"playback\": {\n \"description\": \"How the clip is reconciled to the scene: 'freeze' holds the last frame, 'loop' repeats it, 'fit_to_scene' changes speed to match, 'full_video' plays it whole and drives the scene's length.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"speed_multiplier\": {\n \"description\": \"Playback speed multiplier. 2.0 = twice as fast.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"trim\": {\n \"description\": \"Seconds trimmed from each end of the clip.\",\n \"nullable\": true,\n \"properties\": {\n \"end_offset\": {\n \"description\": \"Seconds trimmed from the end of the clip.\",\n \"type\": \"number\"\n },\n \"start_offset\": {\n \"description\": \"Seconds trimmed from the start of the clip.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"start_offset\",\n \"end_offset\"\n ],\n \"type\": \"object\"\n },\n \"type\": {\n \"default\": \"video\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the clip. Present when the video records a link for this element and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"volume\": {\n \"description\": \"Clip audio volume. 1.0 = source level, 0.0 = silent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"On-screen text, plain or styled, reported as the text it displays.\\n\\nStyling (font, size, colour, alignment) is not described: like geometry, it would need a\\ncontract this response does not have. Styled text is flattened to its characters with one\\nnewline between paragraphs, which is exactly the text a template variable is matched against.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"text\": {\n \"description\": \"The text as displayed. Paragraphs of styled text are joined with newlines.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"text\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\",\n \"text\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"An element this version names but does not describe.\\n\\nExists so a scene's ``elements`` array is never a filtered view presented as a complete one: a\\ncaller iterating it sees every element the scene places. Carries exactly two fields: which\\nelement this is, and what kind of thing it is.\\n\\n**Absence of content is not by itself proof that an element is undescribed.** A described\\ncategory can serialize with nothing but an id and a type when every optional field happens to be\\nunavailable, and one case is reachable today: an image with no usable link carries no ``url``,\\nleaving ``{id, type}``. Read the ``type`` to tell them apart.\\n``avatar``, ``image``, ``video`` and ``text`` are described, so a bare one of those is a described\\nelement missing an optional value rather than an unexpanded node.\\n\\nThe ``id`` matters most here. Two masked elements in one scene are otherwise identical on the\\nwire, so without it a caller can see that the scene places two things and nothing else about\\neither.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"The element's category. An **open set**: new values may be added, so treat an unrecognized one as an element to skip rather than an error. Several kinds of element can share one value, so a category says what an element is rather than naming a single underlying kind. Every value is permanent except 'other', which means 'a kind of element this API does not yet categorise' and may be replaced by a more specific category later.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\",\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"An element that holds other elements: a group, or a mask.\\n\\nA scene lists the container, not its contents, so a scene holding one group of five images\\nlists a single element while five things render. Walking a scene's composition means recursing\\ninto ``children``.\\n\\nA mask matters more here than a group: a group conveys linkage this response does not express,\\nwhile a mask conveys clipping, so treating a masked image as a plain image asserts a\\ncomposition that renders differently in kind.\",\n \"properties\": {\n \"children\": {\n \"description\": \"The elements this container holds, each described exactly as a top-level element would be.\",\n \"items\": {\n \"anyOf\": [\n {\n \"description\": \"An avatar performing the scene's speech.\\n\\nCarries what a v3 create request can set per avatar, so what is read back here is what can be\\nsent back. Fields are added as callers need them, and a new one is not a breaking change.\\n\\nOutput resolution is a whole-video setting, reported on the video rather than per avatar.\",\n \"properties\": {\n \"avatar_id\": {\n \"description\": \"The avatar look performing this scene.\",\n \"type\": \"string\"\n },\n \"engine\": {\n \"description\": \"Generation engine for this avatar. Absent when the video does not record one and none can be determined.\",\n \"enum\": [\n \"avatar_v\",\n \"avatar_iv\",\n \"avatar_iii\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"expressiveness\": {\n \"description\": \"Expressiveness level. Absent when left at the default.\",\n \"enum\": [\n \"high\",\n \"medium\",\n \"low\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"motion_prompt\": {\n \"description\": \"Motion description authored for this avatar.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"avatar\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\",\n \"avatar_id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"An image placed in the scene.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"image\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the image. Present when the video records a link for this element and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed; the element is still reported either way, so an image with no URL is one this read could not link to rather than an element this API does not describe. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A video clip placed in the scene.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"playback\": {\n \"description\": \"How the clip is reconciled to the scene: 'freeze' holds the last frame, 'loop' repeats it, 'fit_to_scene' changes speed to match, 'full_video' plays it whole and drives the scene's length.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"speed_multiplier\": {\n \"description\": \"Playback speed multiplier. 2.0 = twice as fast.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"trim\": {\n \"description\": \"Seconds trimmed from each end of the clip.\",\n \"nullable\": true,\n \"properties\": {\n \"end_offset\": {\n \"description\": \"Seconds trimmed from the end of the clip.\",\n \"type\": \"number\"\n },\n \"start_offset\": {\n \"description\": \"Seconds trimmed from the start of the clip.\",\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"start_offset\",\n \"end_offset\"\n ],\n \"type\": \"object\"\n },\n \"type\": {\n \"default\": \"video\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n },\n \"url\": {\n \"description\": \"URL of the clip. Present when the video records a link for this element and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"volume\": {\n \"description\": \"Clip audio volume. 1.0 = source level, 0.0 = silent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"On-screen text, plain or styled, reported as the text it displays.\\n\\nStyling (font, size, colour, alignment) is not described: like geometry, it would need a\\ncontract this response does not have. Styled text is flattened to its characters with one\\nnewline between paragraphs, which is exactly the text a template variable is matched against.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"text\": {\n \"description\": \"The text as displayed. Paragraphs of styled text are joined with newlines.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"text\",\n \"description\": \"Element type discriminator.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\",\n \"text\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"An element this version names but does not describe.\\n\\nExists so a scene's ``elements`` array is never a filtered view presented as a complete one: a\\ncaller iterating it sees every element the scene places. Carries exactly two fields: which\\nelement this is, and what kind of thing it is.\\n\\n**Absence of content is not by itself proof that an element is undescribed.** A described\\ncategory can serialize with nothing but an id and a type when every optional field happens to be\\nunavailable, and one case is reachable today: an image with no usable link carries no ``url``,\\nleaving ``{id, type}``. Read the ``type`` to tell them apart.\\n``avatar``, ``image``, ``video`` and ``text`` are described, so a bare one of those is a described\\nelement missing an optional value rather than an unexpanded node.\\n\\nThe ``id`` matters most here. Two masked elements in one scene are otherwise identical on the\\nwire, so without it a caller can see that the scene places two things and nothing else about\\neither.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"The element's category. An **open set**: new values may be added, so treat an unrecognized one as an element to skip rather than an error. Several kinds of element can share one value, so a category says what an element is rather than naming a single underlying kind. Every value is permanent except 'other', which means 'a kind of element this API does not yet categorise' and may be replaced by a more specific category later.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\",\n \"type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Circular reference to ContainerElement\"\n }\n ]\n },\n \"type\": \"array\"\n },\n \"id\": {\n \"description\": \"Opaque identifier of this element within the video. Not to be parsed, sorted, or assumed to encode anything. Stable across edits and across regeneration of the video.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"description\": \"The container's category, e.g. 'group' or 'mask'.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\",\n \"type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": \"array\"\n },\n \"id\": {\n \"description\": \"Identifier of this scene within the video.\",\n \"type\": \"string\"\n },\n \"script\": {\n \"description\": \"This scene's audio sources in order, each one synthesized speech, uploaded audio, or a silence. The create API's per-scene `script` is narrower: it is the text of a single one of these entries.\",\n \"items\": {\n \"anyOf\": [\n {\n \"description\": \"Synthesized speech: a script delivered by a voice.\",\n \"properties\": {\n \"id\": {\n \"description\": \"Opaque identifier of this script entry within the video. Not to be parsed, sorted, or assumed to encode anything. The same id under two scenes means one entry is shared between them: its text belongs to both, and concatenating both scenes' scripts would synthesize the shared words twice.\",\n \"type\": \"string\"\n },\n \"text\": {\n \"description\": \"The script, verbatim, including inline markup. Empty when the scene was left unfinished.\",\n \"type\": \"string\"\n },\n \"trim_to_speech\": {\n \"description\": \"Whether leading and trailing silence is trimmed to the spoken window.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n },\n \"type\": {\n \"default\": \"text\",\n \"description\": \"Script entry type discriminator.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"The voice delivering this script.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"voice_settings\": {\n \"description\": \"Voice tuning applied to this script.\",\n \"nullable\": true,\n \"properties\": {\n \"engine_settings\": {\n \"description\": \"Engine-specific voice tuning, discriminated by 'engine_type'. Present only when the video pins an engine this API exposes; absent when the engine is left for the server to pick.\",\n \"discriminator\": {\n \"mapping\": {\n \"elevenlabs\": \"#/components/schemas/ElevenLabsEngineSettings\",\n \"fish\": \"#/components/schemas/FishEngineSettings\",\n \"starfish\": \"#/components/schemas/StarfishEngineSettings\"\n },\n \"propertyName\": \"engine_type\"\n },\n \"nullable\": true,\n \"oneOf\": [\n {\n \"description\": \"Engine-specific voice settings for ElevenLabs-backed voices.\\n\\nSupports model, stability, similarity_boost, style, and use_speaker_boost.\\nWhen using eleven_v3 or eleven_v4, stability may be any value in [0.0, 1.0].\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'elevenlabs' for ElevenLabs-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"The model ID to use for ElevenLabs.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_turbo_v2_5\",\n \"eleven_flash_v2_5\",\n \"eleven_v3\",\n \"eleven_v4\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity_boost\": {\n \"description\": \"The similarity boost parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"The stability parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"style\": {\n \"description\": \"The style parameter for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"use_speaker_boost\": {\n \"description\": \"Whether to use speaker boost for ElevenLabs.\",\n \"nullable\": true,\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-specific voice settings for Fish Audio-backed voices.\\n\\nInherits Fish's tuning fields (model, stability, similarity).\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'fish' for Fish Audio-backed voices.\",\n \"type\": \"string\"\n },\n \"model\": {\n \"description\": \"Fish Audio model version (default 's1').\",\n \"enum\": [\n \"s1\",\n \"s2-pro\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"similarity\": {\n \"description\": \"Similarity parameter; how closely to match the source voice.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"stability\": {\n \"description\": \"Stability parameter; higher is more consistent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Engine-selection for Starfish-backed voices.\\n\\nStarfish has no user-tunable settings today; set ``engine_type='starfish'`` to force\\nStarfish routing on voices that support multiple engines.\",\n \"properties\": {\n \"engine_type\": {\n \"description\": \"Engine type discriminator. Must be 'starfish' for Starfish-backed voices.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"engine_type\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"locale\": {\n \"description\": \"Locale/accent hint for multi-lingual voices (e.g. 'en-US').\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"pitch\": {\n \"description\": \"Pitch adjustment in semitones.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"speed\": {\n \"description\": \"Playback speed multiplier.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"volume\": {\n \"description\": \"Voice audio volume. 1.0 = full, 0.0 = silent.\",\n \"nullable\": true,\n \"type\": \"number\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n }\n },\n \"required\": [\n \"id\",\n \"text\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"Uploaded audio played as the scene's speech.\",\n \"properties\": {\n \"audio_url\": {\n \"description\": \"URL of the audio. Present when the video records a link for this entry and, if that link is signed, it is not at or near its deadline. Absent when no link is recorded, or the only recorded link has lapsed. Only the deadline is checked, so a present URL is not a promise that it resolves: it may name an object that has since moved, or a host the customer supplied. Signed links expire, so download what you need rather than storing the URL.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"id\": {\n \"description\": \"Identifier of this script entry within the video.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"audio\",\n \"description\": \"Script entry type discriminator.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n {\n \"description\": \"A held silence, with no words spoken over the scene.\",\n \"properties\": {\n \"duration\": {\n \"description\": \"How long the silence is held, in seconds. Absent where the number does not describe what renders — on a scene whose video clip plays in full, the clip's own length governs the scene.\",\n \"nullable\": true,\n \"type\": \"number\"\n },\n \"id\": {\n \"description\": \"Identifier of this script entry within the video.\",\n \"type\": \"string\"\n },\n \"type\": {\n \"default\": \"silence\",\n \"description\": \"Script entry type discriminator.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n }\n ]\n },\n \"type\": \"array\"\n }\n },\n \"required\": [\n \"id\"\n ],\n \"type\": \"object\"\n },\n \"type\": \"array\"\n },\n \"video\": {\n \"description\": \"Video-level context for the scenes.\",\n \"properties\": {\n \"aspect_ratio\": {\n \"description\": \"The video's aspect ratio.\",\n \"enum\": [\n \"16:9\",\n \"9:16\",\n \"4:5\",\n \"5:4\",\n \"1:1\",\n \"auto\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"brand_glossary_id\": {\n \"description\": \"Brand glossary applied to this video's speech.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"caption\": {\n \"description\": \"Caption configuration for the video.\",\n \"nullable\": true,\n \"properties\": {\n \"enabled\": {\n \"description\": \"Whether captions are enabled for this video.\",\n \"type\": \"boolean\"\n }\n },\n \"required\": [\n \"enabled\"\n ],\n \"type\": \"object\"\n },\n \"resolution\": {\n \"description\": \"The video's output resolution, in the same values a create request accepts. Absent when the video's size matches none of them, which is a size a create request could not have asked for and cannot reproduce.\",\n \"enum\": [\n \"4k\",\n \"1080p\",\n \"720p\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"title\": {\n \"description\": \"The video's current title. Editor saves update it.\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n },\n \"video_id\": {\n \"description\": \"The video these scenes belong to.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"video_id\",\n \"edit_version\",\n \"video\"\n ],\n \"type\": \"object\"\n },\n \"error\": {\n \"nullable\": true,\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", Endpoint: "/v3/videos/{video_id}/scenes", Method: "GET", BodyEncoding: "", diff --git a/gen/voice.go b/gen/voice.go index a598365..508c81f 100644 --- a/gen/voice.go +++ b/gen/voice.go @@ -164,8 +164,8 @@ var VoiceList = &command.Spec{ Group: "voice", Name: "list", Summary: "List Voices", - Description: "Returns a paginated list of voices, filterable by type, engine, language, and gender. Use engine=starfish for voices compatible with the TTS endpoint.", - ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"items\": {\n \"description\": \"A single voice in the listing response.\",\n \"properties\": {\n \"gender\": {\n \"description\": \"Gender of the voice.\",\n \"type\": \"string\"\n },\n \"language\": {\n \"description\": \"Primary language of the voice.\",\n \"type\": \"string\"\n },\n \"name\": {\n \"description\": \"Display name of the voice.\",\n \"type\": \"string\"\n },\n \"preview_audio_url\": {\n \"description\": \"URL to a short audio preview of the voice.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"support_locale\": {\n \"description\": \"Whether the voice supports locale variants.\",\n \"type\": \"boolean\"\n },\n \"support_pause\": {\n \"description\": \"Whether the voice supports SSML pause/break tags.\",\n \"type\": \"boolean\"\n },\n \"type\": {\n \"description\": \"Whether this is a public or private voice.\",\n \"enum\": [\n \"public\",\n \"private\"\n ],\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Unique voice identifier.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"voice_id\",\n \"name\",\n \"language\",\n \"gender\",\n \"support_pause\",\n \"support_locale\",\n \"type\"\n ],\n \"type\": \"object\"\n },\n \"type\": \"array\"\n },\n \"has_more\": {\n \"description\": \"Whether more pages are available\",\n \"type\": \"boolean\"\n },\n \"next_token\": {\n \"description\": \"Opaque cursor for the next page\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", + Description: "Returns a paginated list of voices, filterable by type, engine, language, and gender. Speech generation supports Starfish, Orca, and ElevenLabs. Each voice includes its speech default and available speech engines.", + ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"items\": {\n \"properties\": {\n \"available_engines\": {\n \"description\": \"Engines this voice can use for speech generation after workspace plan and vendor restrictions. Language restrictions still apply.\",\n \"items\": {\n \"enum\": [\n \"starfish\",\n \"orca\",\n \"elevenlabs\",\n \"elevenlabs_v3\"\n ],\n \"type\": \"string\"\n },\n \"type\": \"array\"\n },\n \"default_engine\": {\n \"description\": \"Saved speech engine, including your preference. Starfish if no concrete default is saved; null if the default is unavailable for speech.\",\n \"enum\": [\n \"starfish\",\n \"orca\",\n \"elevenlabs\",\n \"elevenlabs_v3\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"gender\": {\n \"description\": \"Gender of the voice.\",\n \"type\": \"string\"\n },\n \"language\": {\n \"description\": \"Primary language of the voice.\",\n \"type\": \"string\"\n },\n \"name\": {\n \"description\": \"Display name of the voice.\",\n \"type\": \"string\"\n },\n \"preview_audio_url\": {\n \"description\": \"URL to a short audio preview of the voice.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"support_locale\": {\n \"description\": \"Whether the voice supports locale variants.\",\n \"type\": \"boolean\"\n },\n \"support_pause\": {\n \"description\": \"Whether the voice supports SSML pause/break tags.\",\n \"type\": \"boolean\"\n },\n \"type\": {\n \"description\": \"Whether this is a public or private voice.\",\n \"enum\": [\n \"public\",\n \"private\"\n ],\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Unique voice identifier.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"voice_id\",\n \"name\",\n \"language\",\n \"gender\",\n \"support_pause\",\n \"support_locale\",\n \"type\"\n ],\n \"type\": \"object\"\n },\n \"type\": \"array\"\n },\n \"has_more\": {\n \"description\": \"Whether more pages are available\",\n \"type\": \"boolean\"\n },\n \"next_token\": {\n \"description\": \"Opaque cursor for the next page\",\n \"nullable\": true,\n \"type\": \"string\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", Endpoint: "/v3/voices", Method: "GET", BodyEncoding: "", @@ -192,7 +192,7 @@ var VoiceList = &command.Spec{ Name: "engine", Type: "string", Default: "", - Help: "Filter by voice engine (e.g. 'starfish'). When set, only voices compatible with that engine are returned.", + Help: "Filter by stored engine mapping. Accepts 'starfish', 'orca', 'elevenlabs' or internal engine names. The deprecated elevenlabs_v3 input alias is also accepted. Available engines may also include engines prepared on first use.", Required: false, Enum: nil, Min: nil, @@ -255,9 +255,9 @@ var VoiceSpeechCreate = &command.Spec{ Group: "voice", Name: "speech create", Summary: "Generate Speech", - Description: "Synthesize speech audio from text using a specified voice. The voice must support the starfish engine — use GET /v3/voices?engine=starfish to find compatible voices. Supports plain text and SSML. Speed range: 0.5–2.0x. Returns a URL to the generated audio file along with duration and optional word-level timestamps.", - RequestSchema: "{\n \"description\": \"Request body for text-to-speech generation.\",\n \"properties\": {\n \"input_type\": {\n \"default\": \"text\",\n \"description\": \"Type of the input: 'text' for plain text, 'ssml' for SSML markup. Defaults to 'text'.\",\n \"type\": \"string\"\n },\n \"language\": {\n \"description\": \"Base language code (e.g. 'en', 'pt', 'zh'). Optional — auto-detected from text when omitted.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"locale\": {\n \"description\": \"BCP-47 locale tag (e.g. 'en-US', 'pt-BR'). When set, language is inferred from locale.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Speed multiplier (0.5-2.0).\",\n \"type\": \"number\"\n },\n \"text\": {\n \"description\": \"Text to synthesize (1-5000 characters). Break tags must express time in seconds (for example, \\u003cbreak time=\\\"0.35s\\\"/\\u003e); millisecond values are not supported.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID to use. The voice must support the starfish engine. Filter compatible voices by passing engine=starfish to the voice listing endpoint.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"text\",\n \"voice_id\"\n ],\n \"type\": \"object\"\n}", - ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"description\": \"Response payload for text-to-speech generation.\",\n \"properties\": {\n \"audio_url\": {\n \"description\": \"URL of the generated audio file.\",\n \"type\": \"string\"\n },\n \"duration\": {\n \"description\": \"Duration of the audio in seconds.\",\n \"type\": \"number\"\n },\n \"request_id\": {\n \"description\": \"Unique identifier for this generation request.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"word_timestamps\": {\n \"description\": \"Word-level timing data.\",\n \"items\": {\n \"description\": \"Word-level timing data from TTS generation.\",\n \"properties\": {\n \"end\": {\n \"description\": \"End time in seconds.\",\n \"type\": \"number\"\n },\n \"start\": {\n \"description\": \"Start time in seconds.\",\n \"type\": \"number\"\n },\n \"word\": {\n \"description\": \"The word.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"word\",\n \"start\",\n \"end\"\n ],\n \"type\": \"object\"\n },\n \"nullable\": true,\n \"type\": \"array\"\n }\n },\n \"required\": [\n \"audio_url\",\n \"duration\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", + Description: "Synthesize speech audio from text with a voice from the voice catalog. Supports Starfish, Orca, and ElevenLabs. The deprecated elevenlabs_v3 engine retains its existing behavior until callers and saved defaults migrate. Select an ElevenLabs model with engine: 'elevenlabs' and settings: {model_id: 'eleven_v4'} (or 'eleven_v3'); model selection does not change the engine. Omit engine to use the voice's saved default and settings, including your saved preference, or specify an engine for this request only. Voices without a concrete saved default retain Starfish. Unsupported selections return an error without switching engines. Breaking change: requests that omit engine now use the voice's saved default, which is Orca for most public catalog voices. Pass engine: 'starfish' to explicitly request Starfish, subject to voice and workspace restrictions. HTTP 400: the saved default or explicit engine is unsupported, not allowed on your plan, or incompatible with the requested locale or language; specify a compatible engine or language. HTTP 403 (ai_vendor_access_restricted): workspace vendor policy blocks ElevenLabs and all of its model variants; choose an allowed engine from List Voices or ask a workspace admin to allow the vendor. A professional HeyGen Voice clone is synthesized by POST /v3/models/audio/tts instead. Supports plain text and SSML. Speed range: 0.5–2.0x. Returns a URL to the generated audio file along with duration and optional word-level timestamps.", + RequestSchema: "{\n \"properties\": {\n \"engine\": {\n \"description\": \"Speech engine override for this request only. Omit to use the voice's saved default, including your saved preference. Voices without a concrete saved default retain Starfish. Unsupported defaults or overrides return an error; engines are never silently substituted. Orca is the HeyGen voice engine. ElevenLabs model variants are selected through settings.model_id. The deprecated elevenlabs_v3 engine retains its existing behavior while callers and saved defaults migrate.\",\n \"enum\": [\n \"starfish\",\n \"orca\",\n \"elevenlabs\",\n \"elevenlabs_v3\"\n ],\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"input_type\": {\n \"default\": \"text\",\n \"description\": \"Type of the input: 'text' for plain text, 'ssml' for SSML markup. Defaults to 'text'.\",\n \"type\": \"string\"\n },\n \"language\": {\n \"description\": \"Base language code (e.g. 'en', 'pt', 'zh'). Optional — auto-detected from text when omitted.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"locale\": {\n \"description\": \"Locale tag for Starfish (e.g. 'zh-HK'). ElevenLabs does not support locale selection; use language instead.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"settings\": {\n \"description\": \"Request-only ElevenLabs model settings. Applied after saved preferences; valid for elevenlabs or the deprecated elevenlabs_v3 compatibility engine.\",\n \"nullable\": true,\n \"properties\": {\n \"model_id\": {\n \"description\": \"ElevenLabs model for this request only. Omit settings to retain the voice's saved model preference.\",\n \"enum\": [\n \"eleven_multilingual_v2\",\n \"eleven_flash_v2\",\n \"eleven_flash_v2_5\",\n \"eleven_turbo_v2_5\",\n \"eleven_v3\",\n \"eleven_v4\"\n ],\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"model_id\"\n ],\n \"type\": \"object\"\n },\n \"speed\": {\n \"default\": 1,\n \"description\": \"Speed multiplier (0.5-2.0).\",\n \"type\": \"number\"\n },\n \"text\": {\n \"description\": \"Text to synthesize (1-5000 characters). Break tags must express time in seconds (for example, \\u003cbreak time=\\\"0.35s\\\"/\\u003e); millisecond values are not supported.\",\n \"type\": \"string\"\n },\n \"voice_id\": {\n \"description\": \"Voice ID from the voice catalog. Professional voice clones use model speech generation instead.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"text\",\n \"voice_id\"\n ],\n \"type\": \"object\"\n}", + ResponseSchema: "{\n \"properties\": {\n \"data\": {\n \"properties\": {\n \"audio_url\": {\n \"description\": \"URL of the generated audio file.\",\n \"type\": \"string\"\n },\n \"duration\": {\n \"description\": \"Duration of the audio in seconds.\",\n \"type\": \"number\"\n },\n \"engine\": {\n \"description\": \"Engine selected for this speech generation.\",\n \"enum\": [\n \"starfish\",\n \"orca\",\n \"elevenlabs\",\n \"elevenlabs_v3\"\n ],\n \"type\": \"string\"\n },\n \"request_id\": {\n \"description\": \"Unique identifier for this generation request.\",\n \"nullable\": true,\n \"type\": \"string\"\n },\n \"word_timestamps\": {\n \"description\": \"Word-level timing data.\",\n \"items\": {\n \"description\": \"Word-level timing data from TTS generation.\",\n \"properties\": {\n \"end\": {\n \"description\": \"End time in seconds.\",\n \"type\": \"number\"\n },\n \"start\": {\n \"description\": \"Start time in seconds.\",\n \"type\": \"number\"\n },\n \"word\": {\n \"description\": \"The word.\",\n \"type\": \"string\"\n }\n },\n \"required\": [\n \"word\",\n \"start\",\n \"end\"\n ],\n \"type\": \"object\"\n },\n \"nullable\": true,\n \"type\": \"array\"\n }\n },\n \"required\": [\n \"audio_url\",\n \"duration\",\n \"engine\"\n ],\n \"type\": \"object\"\n }\n },\n \"required\": [],\n \"type\": \"object\"\n}", Endpoint: "/v3/voices/speech", Method: "POST", BodyEncoding: "json", @@ -265,6 +265,18 @@ var VoiceSpeechCreate = &command.Spec{ "# Generate speech (requires starfish-engine voice)\n heygen voice speech create --text 'Hello world' --voice-id ", }, Flags: []command.FlagSpec{ + { + Name: "engine", + Type: "string", + Default: "", + Help: "Speech engine override for this request only. Omit to use the voice's saved default, including your saved preference. Voices without a concrete saved default retain Starfish. Unsupported defaults or overrides return an error; engines are never silently substituted. Orca is the HeyGen voice engine. ElevenLabs model variants are selected through settings.model_id. The deprecated elevenlabs_v3 engine retains its existing behavior while callers and saved defaults migrate.", + Required: false, + Enum: []string{"starfish", "orca", "elevenlabs", "elevenlabs_v3"}, + Min: nil, + Max: nil, + Source: "body", + JSONName: "engine", + }, { Name: "input-type", Type: "string", @@ -293,7 +305,7 @@ var VoiceSpeechCreate = &command.Spec{ Name: "locale", Type: "string", Default: "", - Help: "BCP-47 locale tag (e.g. 'en-US', 'pt-BR'). When set, language is inferred from locale.", + Help: "Locale tag for Starfish (e.g. 'zh-HK'). ElevenLabs does not support locale selection; use language instead.", Required: false, Enum: nil, Min: nil, @@ -329,7 +341,7 @@ var VoiceSpeechCreate = &command.Spec{ Name: "voice-id", Type: "string", Default: "", - Help: "Voice ID to use. The voice must support the starfish engine. Filter compatible voices by passing engine=starfish to the voice listing endpoint.", + Help: "Voice ID from the voice catalog. Professional voice clones use model speech generation instead.", Required: true, Enum: nil, Min: nil,