{"id":"ec377018-aab3-4aa5-b6a7-b12971ec936d","entityType":"agent","slug":"clawhub-yjx-research-controlfoley-audio-generator","name":"ControlFoley Audio Generator","canonicalUrl":"https://www.xpersona.co/agent/clawhub-yjx-research-controlfoley-audio-generator","canonicalPath":"/agent/clawhub-yjx-research-controlfoley-audio-generator","generatedAt":"2026-10-11T03:55:39.276Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-11T01:50:08.940Z","emptyReason":null},"description":"A multi-functional audio generation tool for SFX generation, video-to-audio and text-to-audio. 多功能音频生成工具，集成可控视频生成音频、文本生成音频等功能. Skill: ControlFoley Audio Generator Owner: yjx-research Summary: A multi-functional audio generation tool for SFX generation, video-to-audio and text-to-audio. 多功能音频生成工具，集成可控视频生成音频、文本生成音频等功能. Tags: latest:1.0.9 Version history: v1.0.9 | 2026-06-08T03:51:10.695Z | user - Removed the _meta.json file from the skill package. - No functional changes to user experience or features. - Internal metadata cleanup for this vers","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.2K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s1735syvcp8qrc0r38n7crw4pn859fn6:controlfoley-audio-generator","sourceUrl":"https://clawhub.ai/yjx-research/controlfoley-audio-generator","homepage":"https://clawhub.ai/yjx-research/skills/controlfoley-audio-generator","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/yjx-research/controlfoley-audio-generator","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/yjx-research/skills/controlfoley-audio-generator","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":62,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"A multi-functional audio generation tool for SFX generation, video-to-audio and text-to-audio. 多功能音频生成工具，集成可控视频生成音频、文本生成音频等功能. Skill: ControlFoley Audio Generat"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T01:50:08.940Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T01:50:08.940Z","emptyReason":null},"stars":null,"forks":null,"downloads":1200,"packageName":null,"latestVersion":"1.0.9","tractionLabel":"1.2K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T01:50:08.869Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T01:50:08.940Z","lastCrawledAt":"2026-10-11T01:50:08.869Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T01:50:08.869Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.9","createdAt":"2026-06-08T03:51:10.695Z","changelog":"- Removed the _meta.json file from the skill package. - No functional changes to user experience or features. - Internal metadata cleanup for this version.","fileCount":5,"zipByteSize":8724},{"version":"1.0.8","createdAt":"2026-04-23T07:10:40.026Z","changelog":"- Updated privacy and security section to add clear guidelines on data handling, processing, and user recommendations. - Removed duplicated and verbose API usage examples to streamline documentation. - Kept the API and CLI usage, parameters, and error handling instructions unchanged. - No functional or interface changes; documentation improvements only.","fileCount":5,"zipByteSize":8690},{"version":"1.0.7","createdAt":"2026-04-23T06:28:31.495Z","changelog":"Version 1.0.7 - Added _meta.json metadata file. - Updated API endpoint and documentation: now points to https://controlfoley.ai.xiaomi.com (was https://llmplus.ai.xiaomi.com). - Expanded and clarified SKILL.md usage instructions for both CLI and API, including new cURL examples and result retrieval methods. - Improved documentation for API parameters and response formats, including successful, processing, and pending task statuses. - Removed GitHub and ClawHub social promotion from documentation. - No breaking changes to model functionality.","fileCount":4,"zipByteSize":7578},{"version":"1.0.6","createdAt":"2026-04-22T02:07:36.244Z","changelog":"- Added a \"star us\" message with links to the project's GitHub and ClawHub pages. - No functional or API changes; documentation only.","fileCount":4,"zipByteSize":7222},{"version":"1.0.5","createdAt":"2026-04-21T09:04:45.027Z","changelog":"- No user-facing changes in this version. - No file changes detected; documentation and functionality remain the same.","fileCount":4,"zipByteSize":7130},{"version":"1.0.4","createdAt":"2026-04-21T08:46:51.412Z","changelog":"Version 1.0.4 - No file changes detected in this release. - No feature updates, bug fixes, or documentation changes.","fileCount":4,"zipByteSize":7130},{"version":"1.0.3","createdAt":"2026-04-21T08:38:30.592Z","changelog":"- Clarified that this documentation applies to the CLI version by updating the title to \"ControlFoley Audio Generator (CLI version)\". - No functional or API changes; documentation only. - Improved accuracy and clarity of documentation scope.","fileCount":4,"zipByteSize":7219},{"version":"1.0.2","createdAt":"2026-04-21T08:31:39.623Z","changelog":"- Updated the tool description to a shorter, more concise format in both English and Chinese. - No changes to code or functionality; documentation only.","fileCount":4,"zipByteSize":7207}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s1735syvcp8qrc0r38n7crw4pn859fn6:controlfoley-audio-generator","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-yjx-research-controlfoley-audio-generator/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-yjx-research-controlfoley-audio-generator/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-yjx-research-controlfoley-audio-generator/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-yjx-research-controlfoley-audio-generator/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-yjx-research-controlfoley-audio-generator/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-yjx-research-controlfoley-audio-generator/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T03:55:39.272Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-yjx-research-controlfoley-audio-generator/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-yjx-research-controlfoley-audio-generator/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-yjx-research-controlfoley-audio-generator/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-yjx-research-controlfoley-audio-generator/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-11T01:50:08.940Z","emptyReason":null},"readme":"Skill: ControlFoley Audio Generator\n\nOwner: yjx-research\n\nSummary: A multi-functional audio generation tool for SFX generation, video-to-audio and text-to-audio. 多功能音频生成工具，集成可控视频生成音频、文本生成音频等功能.\n\nTags: latest:1.0.9\n\nVersion history:\n\nv1.0.9 | 2026-06-08T03:51:10.695Z | user\n\n- Removed the _meta.json file from the skill package.\n- No functional changes to user experience or features.\n- Internal metadata cleanup for this version.\n\nv1.0.8 | 2026-04-23T07:10:40.026Z | user\n\n- Updated privacy and security section to add clear guidelines on data handling, processing, and user recommendations.\n- Removed duplicated and verbose API usage examples to streamline documentation.\n- Kept the API and CLI usage, parameters, and error handling instructions unchanged.\n- No functional or interface changes; documentation improvements only.\n\nv1.0.7 | 2026-04-23T06:28:31.495Z | user\n\nVersion 1.0.7\n\n- Added _meta.json metadata file.\n- Updated API endpoint and documentation: now points to https://controlfoley.ai.xiaomi.com (was https://llmplus.ai.xiaomi.com).\n- Expanded and clarified SKILL.md usage instructions for both CLI and API, including new cURL examples and result retrieval methods.\n- Improved documentation for API parameters and response formats, including successful, processing, and pending task statuses.\n- Removed GitHub and ClawHub social promotion from documentation.\n- No breaking changes to model functionality.\n\nv1.0.6 | 2026-04-22T02:07:36.244Z | user\n\n- Added a \"star us\" message with links to the project's GitHub and ClawHub pages.\n- No functional or API changes; documentation only.\n\nv1.0.5 | 2026-04-21T09:04:45.027Z | user\n\n- No user-facing changes in this version.\n- No file changes detected; documentation and functionality remain the same.\n\nv1.0.4 | 2026-04-21T08:46:51.412Z | user\n\nVersion 1.0.4\n\n- No file changes detected in this release.\n- No feature updates, bug fixes, or documentation changes.\n\nv1.0.3 | 2026-04-21T08:38:30.592Z | user\n\n- Clarified that this documentation applies to the CLI version by updating the title to \"ControlFoley Audio Generator (CLI version)\".\n- No functional or API changes; documentation only.\n- Improved accuracy and clarity of documentation scope.\n\nv1.0.2 | 2026-04-21T08:31:39.623Z | user\n\n- Updated the tool description to a shorter, more concise format in both English and Chinese.\n- No changes to code or functionality; documentation only.\n\nv1.0.1 | 2026-04-21T08:20:25.865Z | user\n\n- Removed the README.md and README_zh.md files.\n- No new features or functional changes; documentation files were cleaned up.\n- Skill description and core usage remain unchanged.\n\nv1.0.0 | 2026-04-21T07:54:05.934Z | user\n\nInitial release of ControlFoley Audio Generator.\n\n- Supports AI-generated audio and foley using the ControlFoley Audio Generator API.\n- Two core modes: Video-to-Audio (V2A) for syncing sounds to video, and Text-to-Audio (T2A) for generating audio from descriptions.\n- Additional capabilities: text-controlled video dubbing (TC-V2A), audio style transfer (AC-V2A).\n- Suitable for a wide range of sound effects: games, nature, animals, mechanical, ads, and more.\n- Includes CLI tool with configurable parameters (model, seed, negative prompts, duration, etc.).\n- Built-in workarounds for API quirks and network issues; no authentication required.\n\nArchive index:\n\nArchive v1.0.9: 5 files, 8724 bytes\n\nFiles: references/api-reference.md (3639b), scripts/foley.py (8071b), skill-card.md (2298b), SKILL.md (7728b), _meta.json (147b)\n\nFile v1.0.9:SKILL.md\n\n---\nname: controlfoley-audio-generator\ndescription: >\n  A multi-functional audio generation tool for SFX generation, video-to-audio and text-to-audio.\n  多功能音频生成工具，集成可控视频生成音频、文本生成音频等功能.\n\n---\n\n# ControlFoley Audio Generator\n\nA multi-functional audio generation tool powered by the ControlFoley model, integrating video sound effect (SFX) generation, video background music composition, text-to-audio and other functions to realize diversified creative audio generation. \n\nThis tool supports four modes: Video-to-Audio (V2A), Text-Controlled Video-to-Audio (TC-V2A), Audio-Controlled Video-to-Audio (AC-V2A), and Text-to-Audio (T2A).\n\n## Basic Info\n\n| Field | Value |\n|---|---|\n| Service Operator | Xiaomi LLM Plus Team |\n| API Endpoint | `https://controlfoley.ai.xiaomi.com` |\n| Open Source Repo | `https://github.com/xiaomi-research/controlfoley` |\n| Project Page | `https://yjx-research.github.io/ControlFoley_web_page/` |\n| Online Demo | `https://yjx-research.github.io/ControlFoley_web_page/#try-gen` |\n| Model Weights | `https://huggingface.co/YJX-Xiaomi/ControlFoley/` |\n| API Key | Not required |\n| Script Path | `scripts/foley.py` |\n\n## Prerequisites\n\n```bash\npython3 --version   # Python 3.x\ncurl --version      # curl for API submission\nffmpeg -version     # optional, for audio format conversion\n```\n\n## Modes\n\n| Mode | Command | Input | Output | Description |\n|------|---------|-------|--------|-------------|\n| **V2A** | `v2a video.mp4` | Video file | .mp4 + .flac | Generate audio matching the video content |\n| **TC-V2A** | `v2a video.mp4 --prompt \"text\"` | Video + text | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| **AC-V2A** | `v2a video.mp4 --ref-audio ref.wav` | Video + reference audio | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n| **T2A** | `t2a \"prompt\"` | Text description | .flac | Generate audio from text descriptions |\n\n\n## Usage (CLI version)\n\n### 1. Text-to-Audio (T2A, default 8s)\n\n```bash\npython3 scripts/foley.py t2a \"dog barking loudly in a park\"\n```\n\n### 2. Video-to-Audio (V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4\n```\n\n### 3. Text-Controlled Video-to-Audio (TC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --prompt \"footsteps on gravel with birds chirping\"\n```\n\n### 4. Audio-Controlled Video-to-Audio (AC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --ref-audio reference.wav\n```\n\n### 5. Specify duration\n\n```bash\npython3 scripts/foley.py t2a \"A mountain stream murmurs, its gentle current lapping against the pebbles.\" --duration 15\n```\n\n### 6. Generate multiple candidates\n\n```bash\npython3 scripts/foley.py t2a \"cat purring softly\" --count 3\n```\n\n### 7. Fixed seed (reproducible results)\n\n```bash\npython3 scripts/foley.py t2a \"rain on a tin roof\" --seed 42\n```\n\n### 8. List available models\n\n```bash\npython3 scripts/foley.py models\n```\n\n## Usage (API version)\n\n### POST\n\n```bash\ncurl -X POST \"https://controlfoley.ai.xiaomi.com/api/v1/v2a/submit\" -F \"file=@video_path\" -F \"prompt=footsteps on gravel with birds chirping\"\n```\n\nreturn \n\n```json\n{\"taskId\": \"xxx\", \"message\": \"Task submitted successfully\"}\n```\n\n### GET \n\n#### 1. Available Models\n\n```bash\ncurl -X GET \"https://controlfoley.ai.xiaomi.com/api/v1/v2a/models\" \n```\n\nreturn \n\n```json\n{\"models\":[{\"name\":\"ControlFoley\",\"enabled\":true}]}\n```\n\n#### 2. Status Inquiry \n```bash\ncurl -X GET \"https://controlfoley.ai.xiaomi.com/api/v1/v2a/status/{taskId}\" \n```\n\nreturn \n\n1. success：\n\n```json\n{\"urls\":[\"{Domain name}/ControlFoley_output/{taskId}/{filename}\"],\"status\":\"success\",\"done\":true}\n```\n\n2. processing：\n\n```json\n{\"status\":\"processing\",\"done\":false}\n```\n\n3. pending:\n```json \n{\"status\":\"pending\",\"queue_pos\":1,\"queue_position\":1,\"total_queue\":2,\"done\":false}\n```\n\n#### 3. Result Download\n\n```bash\ncurl -X GET \"https://controlfoley.ai.xiaomi.com/ControlFoley_output/{taskId}/{filename}\" --output ./output.flac\n```\n\n#### 4. Status Inquiry & Result Download\n\n```bash\ncurl -X GET \"https://controlfoley.ai.xiaomi.com/api/v1/v2a/status_download/{taskId}\" --output-dir ./output --output audio.zip\n```\n\n## Parameters\n\n### T2A (Text-to-Audio)\n\n| Parameter | Description | Default | Example |\n|-----------|-------------|---------|---------|\n| `prompt` | Audio description text (required) | — | `\"dog barking in park\"` |\n| `--model` | Model ID | `ControlFoley` | `--model ControlFoley` |\n| `--duration` | Audio length in seconds (max 30) | `8` | `--duration 15` |\n| `--negative` | Negative prompt to exclude unwanted sounds | — | `--negative \"noise, human voice\"` |\n| `--cfg` | CFG strength — higher = stricter prompt adherence | `4.5` | `--cfg 6.0` |\n| `--count` | Number of variants to generate (1–5) | `1` | `--count 3` |\n| `--seed` | Fixed random seed for reproducibility | — | `--seed 42` |\n| `-o/--outdir` | Output directory | `./output` | `-o ./my_audio` |\n\n### V2A (Video-to-Audio)\n\n| Parameter | Description | Default | Example |\n|-----------|-------------|---------|---------|\n| `video` | Input video path (required) | — | `input.mp4` |\n| `--model` | Model ID | `ControlFoley` | `--model ControlFoley` |\n| `--prompt` | Text prompt to guide audio generation (TC-V2A) | — | `--prompt \"keyboard tapping\"` |\n| `--negative` | Negative prompt to exclude unwanted sounds | — | `--negative \"music, noise\"` |\n| `--ref-audio` | Reference audio file for timbre control (AC-V2A) | — | `--ref-audio reference.wav` |\n| `--cfg` | CFG strength | `4.5` | `--cfg 7.0` |\n| `--count` | Number of variants to generate (1–5) | `1` | `--count 2` |\n| `--seed` | Fixed random seed (not forwarded to API currently) | — | `--seed 42` |\n| `-o/--outdir` | Output directory | `./output` | `-o ./results` |\n\n## Prompt Tips\n\n- **Be specific**: `\"cat footsteps on wooden floor\"` beats `\"cat sound\"`\n- **Use negative prompts**: `--negative \"human voice, music, noise\"` to filter unwanted audio\n- **CFG tuning**: high CFG (6.0–7.5) for precise control, low CFG (3.0–4.5) for creative freedom\n\n## Output & Post-Processing\n\n- **Audio**: `.flac` (44100 Hz, lossless)\n- **Video**: `.mp4` (original video + generated audio track)\n- Results saved to `--outdir`, paths printed to stdout\n\n**Convert to MP3 for sharing:**\n\n```bash\nffmpeg -i output.flac -codec:a libmp3lame -qscale:a 2 output.mp3\n```\n\n## Error Handling\n\n| Issue | Cause | Fix |\n|-------|-------|-----|\n| Internal URL inaccessible | Result URL uses `.xiaomi.srv` internal domain | Script auto-falls back to `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` |\n| Queue busy | Task is waiting | Script auto-polls up to ~5 min; check load via `curl $API_BASE/health` |\n| Model unavailable | Model not enabled | Run `foley.py models` to see available models |\n| Task timeout | Service overloaded | Resubmit the task |\n\n## API Reference\n\nSee ./references/api-reference.md for full endpoint documentation.\n\n## ⚠️ Privacy & Security\n\n- **Service Operator**: Cloud processing is operated by the Xiaomi LLM Plus Team at `https://controlfoley.ai.xiaomi.com`\n- **Data Upload**: V2A/TC-V2A/AC-V2A modes upload the full video file to the remote service for processing. Do not upload videos containing sensitive personal or identifiable information\n- **Data Processing**: Uploaded videos and audio are used solely for audio generation. Results are returned via URL. Refer to the Xiaomi LLM Plus Team's terms of service for data retention and access control policies\n- **No API Key Required**: The service requires no authentication — please use it responsibly to avoid unnecessary load\n- **Recommendation**: Before first use, validate with a small, non-sensitive test clip\n\nFile v1.0.9:_meta.json\n\n{\n  \"ownerId\": \"kn7ftwffrk08e278sse17ktw39858t8n\",\n  \"slug\": \"controlfoley-audio-generator\",\n  \"version\": \"1.0.9\",\n  \"publishedAt\": 1780890670695\n}\n\nFile v1.0.9:references/api-reference.md\n\n# ControlFoley Audio Generator API Reference\n\nBase URL: `https://controlfoley.ai.xiaomi.com`  \nAuth: None required\n\n**Known API quirks (from real testing):**\n- Submit response may return either `task_id` or `taskId` (camelCase); code handles both\n- Status success may return either `processed_urls` or `urls`; code checks both\n- Result URLs may use internal `.xiaomi.srv` domains; use download endpoint as fallback\n\n## Endpoints\n\n### POST `/api/v1/v2a/submit` — Submit Task\n\nContent-Type: `multipart/form-data`\n\n#### T2A (Text-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| prompt | string | Yes | — | Audio description (e.g. `dog barking in park`) |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| duration | float | Optional | `8.0` | Audio length in seconds (max 30) |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n#### V2A (Video-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| video | file | Yes | — | Video file (.mp4/.webm/.mov) |\n| prompt | string | Optional | — | Text prompt to guide audio generation; omitted if empty |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| ref_audio | file | Optional | — | Reference audio for AC-V2A task; omitted if file not found |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n> Note: `seed` is accepted in the V2A function signature but not currently forwarded to the API.\n\n**Response (202):**\n```json\n{\"task_id\": \"uuid\", \"status\": \"pending\", \"queue_pos\": 2, \"message\": \"Task submitted successfully\"}\n```\n> Note: field may appear as `taskId` (camelCase) — handle both.\n\n### GET `/api/v1/v2a/status/{task_id}` — Poll Status\n\n**Pending/Processing:**\n```json\n{\"status\": \"pending|processing\", \"queue_pos\": 3}\n```\n\n**Success:**\n```json\n{\n  \"status\": \"success\", \"done\": true,\n  \"processed_urls\": [\"http://.../uuid.flac\", \"http://.../uuid.mp4\"]\n}\n```\n> Note: result URLs field may appear as `urls` — handle both `processed_urls` and `urls`.\n\n**Failed:**\n```json\n{\"status\": \"failed\", \"done\": true, \"error\": \"reason\"}\n```\n\n### GET `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` — Download Result\n\nFallback download endpoint for internal URLs that are not directly accessible.  \nReturns binary file stream (.flac at 44100 Hz or .mp4).\n\n### GET `/api/v1/v2a/models` — List Models\n\n```json\n{\"models\": [{\"name\": \"ControlFoley\", \"enabled\": true}]}\n```\n\n### GET `/health` — Health Check\n\n```json\n{\"status\": \"ok\", \"queue_size\": 3}\n```\n\n## Models\n\n| Model | Use Case |\n|-------|----------|\n| ControlFoley | General T2A and V2A (default) |\n\n## Modes\n\n| Mode | Input | Output | Description |\n|------|-------|--------|-------------|\n| T2A | prompt | .flac | Generate audio from text descriptions |\n| V2A | video file | .mp4 + .flac | Generate audio that matches video content |\n| TC-V2A | prompt + video file | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| AC-V2A | reference_audio + video file | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n\nFile v1.0.9:skill-card.md\n\n## Description:\n\nA multi-functional audio generation tool for SFX generation, video-to-audio and text-to-audio.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[yjx-research](https://clawhub.ai/user/yjx-research)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and creators use this skill to generate sound effects, background audio, or Foley-style audio from text prompts, video files, and optional reference audio.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill uploads user media to a remote ControlFoley service for processing.\n\nMitigation: Use only non-sensitive media and validate behavior with a small test clip before broader use.\n\nRisk: The release security summary reports flaws that can expose local files or fetch untrusted URLs during normal use.\n\nMitigation: Review before installing, avoid prompt, model, or negative values beginning with @ or <, and prefer a patched version with native multipart upload, download URL validation, and size limits.\n\nRisk: Returned downloads come from service-provided URLs.\n\nMitigation: Treat generated downloads as untrusted files and inspect them before using them in downstream workflows.\n\n## Reference(s):\n\n- [ControlFoley Audio Generator API Reference](references/api-reference.md)\n- [ControlFoley project page](https://yjx-research.github.io/ControlFoley_web_page/)\n- [ControlFoley online demo](https://yjx-research.github.io/ControlFoley_web_page/#try-gen)\n- [ControlFoley model weights](https://huggingface.co/YJX-Xiaomi/ControlFoley/)\n- [ControlFoley source repository](https://github.com/xiaomi-research/controlfoley)\n\n## Skill Output:\n\n**Output Type(s):** [Files, Shell commands, API Calls, Guidance]\n\n**Output Format:** [Generated audio/video files with text status output]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Produces FLAC audio and, for video-to-audio workflows, MP4 video with generated audio.]\n\n## Skill Version(s):\n\n1.0.9 (source: release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.0.8: 5 files, 8690 bytes\n\nFiles: _meta.json (147b), references/api-reference.md (3639b), scripts/foley.py (8071b), skill-card.md (2346b), SKILL.md (7739b)\n\nFile v1.0.8:SKILL.md\n\n---\nname: controlfoley-audio-generator\ndescription: >\n  A multi-functional audio generation tool for SFX generation, video-to-audio and text-to-audio.\n  多功能音频生成工具，集成可控视频生成音频、文本生成音频等功能.\n\n---\n\n# ControlFoley Audio Generator\n\nA multi-functional audio generation tool powered by the ControlFoley model, integrating video sound effect (SFX) generation, video background music composition, text-to-audio and other functions to realize diversified creative audio generation. \n\nThis tool supports four modes: Video-to-Audio (V2A), Text-Controlled Video-to-Audio (TC-V2A), Audio-Controlled Video-to-Audio (AC-V2A), and Text-to-Audio (T2A).\n\n## Basic Info\n\n| Field | Value |\n|---|---|\n| Service Operator | Xiaomi LLM Plus Team |\n| API Endpoint | `https://controlfoley.ai.xiaomi.com` |\n| Open Source Repo | `https://github.com/xiaomi-research/controlfoley` |\n| Project Page | `https://yjx-research.github.io/ControlFoley_web_page/` |\n| Online Demo | `https://yjx-research.github.io/ControlFoley_web_page/#try-gen` |\n| Model Weights | `https://huggingface.co/YJX-Xiaomi/ControlFoley/` |\n| API Key | Not required |\n| Script Path | `scripts/foley.py` |\n\n## Prerequisites\n\n```bash\npython3 --version   # Python 3.x\ncurl --version      # curl for API submission\nffmpeg -version     # optional, for audio format conversion\n```\n\n## Modes\n\n| Mode | Command | Input | Output | Description |\n|------|---------|-------|--------|-------------|\n| **V2A** | `v2a video.mp4` | Video file | .mp4 + .flac | Generate audio matching the video content |\n| **TC-V2A** | `v2a video.mp4 --prompt \"text\"` | Video + text | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| **AC-V2A** | `v2a video.mp4 --ref-audio ref.wav` | Video + reference audio | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n| **T2A** | `t2a \"prompt\"` | Text description | .flac | Generate audio from text descriptions |\n\n\n## Usage (CLI version)\n\n### 1. Text-to-Audio (T2A, default 8s)\n\n```bash\npython3 scripts/foley.py t2a \"dog barking loudly in a park\"\n```\n\n### 2. Video-to-Audio (V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4\n```\n\n### 3. Text-Controlled Video-to-Audio (TC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --prompt \"footsteps on gravel with birds chirping\"\n```\n\n### 4. Audio-Controlled Video-to-Audio (AC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --ref-audio reference.wav\n```\n\n### 5. Specify duration\n\n```bash\npython3 scripts/foley.py t2a \"A mountain stream murmurs, its gentle current lapping against the pebbles.\" --duration 15\n```\n\n### 6. Generate multiple candidates\n\n```bash\npython3 scripts/foley.py t2a \"cat purring softly\" --count 3\n```\n\n### 7. Fixed seed (reproducible results)\n\n```bash\npython3 scripts/foley.py t2a \"rain on a tin roof\" --seed 42\n```\n\n### 8. List available models\n\n```bash\npython3 scripts/foley.py models\n```\n\n## Usage (API version)\n\n### POST\n\n```bash\ncurl -X POST \"https://controlfoley.ai.xiaomi.com/api/v1/v2a/submit\" -F \"file=@video_path\" -F \"prompt=footsteps on gravel with birds chirping\"\n```\n\nreturn \n\n```json\n{\"taskId\": \"xxx\", \"message\": \"Task submitted successfully\"}\n```\n\n### GET \n\n#### 1. Available Models\n\n```bash\ncurl -X GET \"https://controlfoley.ai.xiaomi.com/api/v1/v2a/models\" \n```\n\nreturn \n\n```json\n{\"models\":[{\"name\":\"ControlFoley\",\"enabled\":true}]}\n```\n\n#### 2. Status Inquiry \n```bash\ncurl -X GET \"https://controlfoley.ai.xiaomi.com/api/v1/v2a/status/{taskId}\" \n```\n\nreturn \n\n1. success：\n\n```json\n{\"urls\":[\"{Domain name}/ControlFoley_output/{taskId}/{filename}\"],\"status\":\"success\",\"done\":true}\n```\n\n2. processing：\n\n```json\n{\"status\":\"processing\",\"done\":false}\n```\n\n3. pending:\n```json \n{\"status\":\"pending\",\"queue_pos\":1,\"queue_position\":1,\"total_queue\":2,\"done\":false}\n```\n\n#### 3. Result Download\n\n```bash\ncurl -X GET \"https://controlfoley.ai.xiaomi.com/api/v1/v2a/ControlFoley_output/{taskId}/{filename}\" --output ./output.flac\n```\n\n#### 4. Status Inquiry & Result Download\n\n```bash\ncurl -X GET \"https://controlfoley.ai.xiaomi.com/api/v1/v2a/status_download/{taskId}\" --output-dir ./output --output audio.zip\n```\n\n## Parameters\n\n### T2A (Text-to-Audio)\n\n| Parameter | Description | Default | Example |\n|-----------|-------------|---------|---------|\n| `prompt` | Audio description text (required) | — | `\"dog barking in park\"` |\n| `--model` | Model ID | `ControlFoley` | `--model ControlFoley` |\n| `--duration` | Audio length in seconds (max 30) | `8` | `--duration 15` |\n| `--negative` | Negative prompt to exclude unwanted sounds | — | `--negative \"noise, human voice\"` |\n| `--cfg` | CFG strength — higher = stricter prompt adherence | `4.5` | `--cfg 6.0` |\n| `--count` | Number of variants to generate (1–5) | `1` | `--count 3` |\n| `--seed` | Fixed random seed for reproducibility | — | `--seed 42` |\n| `-o/--outdir` | Output directory | `./output` | `-o ./my_audio` |\n\n### V2A (Video-to-Audio)\n\n| Parameter | Description | Default | Example |\n|-----------|-------------|---------|---------|\n| `video` | Input video path (required) | — | `input.mp4` |\n| `--model` | Model ID | `ControlFoley` | `--model ControlFoley` |\n| `--prompt` | Text prompt to guide audio generation (TC-V2A) | — | `--prompt \"keyboard tapping\"` |\n| `--negative` | Negative prompt to exclude unwanted sounds | — | `--negative \"music, noise\"` |\n| `--ref-audio` | Reference audio file for timbre control (AC-V2A) | — | `--ref-audio reference.wav` |\n| `--cfg` | CFG strength | `4.5` | `--cfg 7.0` |\n| `--count` | Number of variants to generate (1–5) | `1` | `--count 2` |\n| `--seed` | Fixed random seed (not forwarded to API currently) | — | `--seed 42` |\n| `-o/--outdir` | Output directory | `./output` | `-o ./results` |\n\n## Prompt Tips\n\n- **Be specific**: `\"cat footsteps on wooden floor\"` beats `\"cat sound\"`\n- **Use negative prompts**: `--negative \"human voice, music, noise\"` to filter unwanted audio\n- **CFG tuning**: high CFG (6.0–7.5) for precise control, low CFG (3.0–4.5) for creative freedom\n\n## Output & Post-Processing\n\n- **Audio**: `.flac` (44100 Hz, lossless)\n- **Video**: `.mp4` (original video + generated audio track)\n- Results saved to `--outdir`, paths printed to stdout\n\n**Convert to MP3 for sharing:**\n\n```bash\nffmpeg -i output.flac -codec:a libmp3lame -qscale:a 2 output.mp3\n```\n\n## Error Handling\n\n| Issue | Cause | Fix |\n|-------|-------|-----|\n| Internal URL inaccessible | Result URL uses `.xiaomi.srv` internal domain | Script auto-falls back to `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` |\n| Queue busy | Task is waiting | Script auto-polls up to ~5 min; check load via `curl $API_BASE/health` |\n| Model unavailable | Model not enabled | Run `foley.py models` to see available models |\n| Task timeout | Service overloaded | Resubmit the task |\n\n## API Reference\n\nSee ./references/api-reference.md for full endpoint documentation.\n\n## ⚠️ Privacy & Security\n\n- **Service Operator**: Cloud processing is operated by the Xiaomi LLM Plus Team at `https://controlfoley.ai.xiaomi.com`\n- **Data Upload**: V2A/TC-V2A/AC-V2A modes upload the full video file to the remote service for processing. Do not upload videos containing sensitive personal or identifiable information\n- **Data Processing**: Uploaded videos and audio are used solely for audio generation. Results are returned via URL. Refer to the Xiaomi LLM Plus Team's terms of service for data retention and access control policies\n- **No API Key Required**: The service requires no authentication — please use it responsibly to avoid unnecessary load\n- **Recommendation**: Before first use, validate with a small, non-sensitive test clip\n\nFile v1.0.8:_meta.json\n\n{\n  \"ownerId\": \"kn7ftwffrk08e278sse17ktw39858t8n\",\n  \"slug\": \"controlfoley-audio-generator\",\n  \"version\": \"1.0.8\",\n  \"publishedAt\": 1776928240026\n}\n\nFile v1.0.8:references/api-reference.md\n\n# ControlFoley Audio Generator API Reference\n\nBase URL: `https://controlfoley.ai.xiaomi.com`  \nAuth: None required\n\n**Known API quirks (from real testing):**\n- Submit response may return either `task_id` or `taskId` (camelCase); code handles both\n- Status success may return either `processed_urls` or `urls`; code checks both\n- Result URLs may use internal `.xiaomi.srv` domains; use download endpoint as fallback\n\n## Endpoints\n\n### POST `/api/v1/v2a/submit` — Submit Task\n\nContent-Type: `multipart/form-data`\n\n#### T2A (Text-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| prompt | string | Yes | — | Audio description (e.g. `dog barking in park`) |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| duration | float | Optional | `8.0` | Audio length in seconds (max 30) |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n#### V2A (Video-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| video | file | Yes | — | Video file (.mp4/.webm/.mov) |\n| prompt | string | Optional | — | Text prompt to guide audio generation; omitted if empty |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| ref_audio | file | Optional | — | Reference audio for AC-V2A task; omitted if file not found |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n> Note: `seed` is accepted in the V2A function signature but not currently forwarded to the API.\n\n**Response (202):**\n```json\n{\"task_id\": \"uuid\", \"status\": \"pending\", \"queue_pos\": 2, \"message\": \"Task submitted successfully\"}\n```\n> Note: field may appear as `taskId` (camelCase) — handle both.\n\n### GET `/api/v1/v2a/status/{task_id}` — Poll Status\n\n**Pending/Processing:**\n```json\n{\"status\": \"pending|processing\", \"queue_pos\": 3}\n```\n\n**Success:**\n```json\n{\n  \"status\": \"success\", \"done\": true,\n  \"processed_urls\": [\"http://.../uuid.flac\", \"http://.../uuid.mp4\"]\n}\n```\n> Note: result URLs field may appear as `urls` — handle both `processed_urls` and `urls`.\n\n**Failed:**\n```json\n{\"status\": \"failed\", \"done\": true, \"error\": \"reason\"}\n```\n\n### GET `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` — Download Result\n\nFallback download endpoint for internal URLs that are not directly accessible.  \nReturns binary file stream (.flac at 44100 Hz or .mp4).\n\n### GET `/api/v1/v2a/models` — List Models\n\n```json\n{\"models\": [{\"name\": \"ControlFoley\", \"enabled\": true}]}\n```\n\n### GET `/health` — Health Check\n\n```json\n{\"status\": \"ok\", \"queue_size\": 3}\n```\n\n## Models\n\n| Model | Use Case |\n|-------|----------|\n| ControlFoley | General T2A and V2A (default) |\n\n## Modes\n\n| Mode | Input | Output | Description |\n|------|-------|--------|-------------|\n| T2A | prompt | .flac | Generate audio from text descriptions |\n| V2A | video file | .mp4 + .flac | Generate audio that matches video content |\n| TC-V2A | prompt + video file | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| AC-V2A | reference_audio + video file | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n\nFile v1.0.8:skill-card.md\n\n## Description: <br>\nA multi-functional audio generation tool for SFX generation, video-to-audio, and text-to-audio. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[yjx-research](https://clawhub.ai/user/yjx-research) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers, creators, and audio workflow users use this skill to submit text prompts, videos, and optional reference audio to ControlFoley and download generated sound effects, background audio, or synchronized video/audio outputs. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Prompts, videos, and optional reference audio are sent to Xiaomi's ControlFoley cloud service. <br>\nMitigation: Use non-sensitive test media first and review the service's privacy and retention terms before uploading confidential or identifiable recordings. <br>\nRisk: Generated files are downloaded from service-provided result URLs and saved locally. <br>\nMitigation: Choose an output directory you control and review generated .flac and .mp4 files before sharing or reusing them. <br>\n\n\n## Reference(s): <br>\n- [ControlFoley Audio Generator API Reference](references/api-reference.md) <br>\n- [ClawHub Release Page](https://clawhub.ai/yjx-research/controlfoley-audio-generator) <br>\n- [ControlFoley Project Page](https://yjx-research.github.io/ControlFoley_web_page/) <br>\n- [ControlFoley Open Source Repo](https://github.com/xiaomi-research/controlfoley) <br>\n- [ControlFoley Model Weights](https://huggingface.co/YJX-Xiaomi/ControlFoley/) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [Files, Shell commands, Guidance] <br>\n**Output Format:** [Markdown guidance with CLI/API commands and local .flac or .mp4 output files] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Submits prompts and media to a remote API, polls for completion, and saves generated results to the selected output directory.] <br>\n\n## Skill Version(s): <br>\n1.0.8 (source: server release metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.0.7: 4 files, 7578 bytes\n\nFiles: _meta.json (147b), references/api-reference.md (3639b), scripts/foley.py (8071b), SKILL.md (8804b)\n\nFile v1.0.7:SKILL.md\n\n---\nname: controlfoley-audio-generator\ndescription: >\n  A multi-functional audio generation tool for SFX generation, video-to-audio and text-to-audio.\n  多功能音频生成工具，集成可控视频生成音频、文本生成音频等功能.\n\n---\n\n# ControlFoley Audio Generator\n\nA multi-functional audio generation tool powered by the ControlFoley model, integrating video sound effect (SFX) generation, video background music composition, text-to-audio and other functions to realize diversified creative audio generation. \n\nThis tool supports four modes: Video-to-Audio (V2A), Text-Controlled Video-to-Audio (TC-V2A), Audio-Controlled Video-to-Audio (AC-V2A), and Text-to-Audio (T2A).\n\n## Basic Info\n\n| Field | Value |\n|---|---|\n| Service Operator | Xiaomi LLM Plus Team |\n| API Endpoint | `https://controlfoley.ai.xiaomi.com` |\n| Open Source Repo | `https://github.com/xiaomi-research/controlfoley` |\n| Project Page | `https://yjx-research.github.io/ControlFoley_web_page/` |\n| Online Demo | `https://yjx-research.github.io/ControlFoley_web_page/#try-gen` |\n| Model Weights | `https://huggingface.co/YJX-Xiaomi/ControlFoley/` |\n| API Key | Not required |\n| Script Path | `scripts/foley.py` |\n\n## Prerequisites\n\n```bash\npython3 --version   # Python 3.x\ncurl --version      # curl for API submission\nffmpeg -version     # optional, for audio format conversion\n```\n\n## Modes\n\n| Mode | Command | Input | Output | Description |\n|------|---------|-------|--------|-------------|\n| **V2A** | `v2a video.mp4` | Video file | .mp4 + .flac | Generate audio matching the video content |\n| **TC-V2A** | `v2a video.mp4 --prompt \"text\"` | Video + text | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| **AC-V2A** | `v2a video.mp4 --ref-audio ref.wav` | Video + reference audio | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n| **T2A** | `t2a \"prompt\"` | Text description | .flac | Generate audio from text descriptions |\n\n\n## Usage (CLI version)\n\n### 1. Text-to-Audio (T2A, default 8s)\n\n```bash\npython3 scripts/foley.py t2a \"dog barking loudly in a park\"\n```\n\n### 2. Video-to-Audio (V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4\n```\n\n### 3. Text-Controlled Video-to-Audio (TC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --prompt \"footsteps on gravel with birds chirping\"\n```\n\n### 4. Audio-Controlled Video-to-Audio (AC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --ref-audio reference.wav\n```\n\n### 5. Specify duration\n\n```bash\npython3 scripts/foley.py t2a \"A mountain stream murmurs, its gentle current lapping against the pebbles.\" --duration 15\n```\n\n### 6. Generate multiple candidates\n\n```bash\npython3 scripts/foley.py t2a \"cat purring softly\" --count 3\n```\n\n### 7. Fixed seed (reproducible results)\n\n```bash\npython3 scripts/foley.py t2a \"rain on a tin roof\" --seed 42\n```\n\n### 8. List available models\n\n```bash\npython3 scripts/foley.py models\n```\n\n## Usage (API version)\n\n### POST\n\n#### 1. Text-to-Audio (T2A, default 8s)\n\n```bash\ncurl -X POST \"https://controlfoley.ai.xiaomi.com/api/v1/v2a/submit\" -F \"prompt=dog barking in park\"\n```\n\n#### 2. Video-to-Audio (V2A)\n\n```bash\ncurl -X POST \"https://controlfoley.ai.xiaomi.com/api/v1/v2a/submit\" -F \"file=@video_path\"\n```\n\n#### 3. Text-Controlled Video-to-Audio (TC-V2A)\n\n```bash\ncurl -X POST \"https://controlfoley.ai.xiaomi.com/api/v1/v2a/submit\" -F \"file=@video_path\" -F \"prompt=footsteps on gravel with birds chirping\"\n```\n\n#### 4. Audio-Controlled Video-to-Audio (AC-V2A)\n\n```bash\ncurl -X POST \"https://controlfoley.ai.xiaomi.com/api/v1/v2a/submit\" -F \"file=@video_path\" -F \"ref_audio=@audio_path\"\n```\n\n#### 5. Specify duration\n\n```bash\ncurl -X POST \"https://controlfoley.ai.xiaomi.com/api/v1/v2a/submit\" -F \"prompt=A mountain stream murmurs, its gentle current lapping against the pebbles\" -F \"duration=15\"\n```\n\n#### 6. Generate multiple candidates\n\n```bash\ncurl -X POST \"https://controlfoley.ai.xiaomi.com/api/v1/v2a/submit\" -F \"prompt=cat purring softly\" -F \"count=3\"\n```\n\n#### 7. Fixed seed (reproducible results)\n\n```bash\ncurl -X POST \"https://controlfoley.ai.xiaomi.com/api/v1/v2a/submit\" -F \"prompt=rain on a tin roof\" -F \"seed=42\"\n```\n\nreturn \n\n```json\n{\"taskId\": \"xxx\", \"message\": \"Task submitted successfully\"}\n```\n\n### GET \n\n#### 1. Available Models\n\n```bash\ncurl -X GET \"https://controlfoley.ai.xiaomi.com/api/v1/v2a/models\" \n```\n\nreturn \n\n```json\n{\"models\":[{\"name\":\"ControlFoley\",\"enabled\":true}]}\n```\n\n#### 2. Status Inquiry \n```bash\ncurl -X GET \"https://controlfoley.ai.xiaomi.com/api/v1/v2a/status/{taskId}\" \n```\n\nreturn \n\n1. success：\n```json\n{\"urls\":[\"{Intranet domain name}/ControlFoley_output/{taskId}/{filename}\"],\"status\":\"success\",\"done\":true}\n```\n\n2. processing：\n\n```json\n{\"status\":\"processing\",\"done\":false}\n```\n\n3. pending:\n```json \n{\"status\":\"pending\",\"queue_pos\":1,\"queue_position\":1,\"total_queue\":2,\"done\":false}\n```\n\n#### 3. Result Download\n\n```bash\ncurl -X GET \"https://controlfoley.ai.xiaomi.com/api/v1/v2a/ControlFoley_output/{taskId}/{filename}\" --output ./output.flac\n```\n\n#### 4. Status Inquiry & Result Download\n\n```bash\ncurl -X GET \"https://controlfoley.ai.xiaomi.com/api/v1/v2a/status_download/{taskId}\" --output-dir ./output --output audio.zip\n```\n\n## Parameters\n\n### T2A (Text-to-Audio)\n\n| Parameter | Description | Default | Example |\n|-----------|-------------|---------|---------|\n| `prompt` | Audio description text (required) | — | `\"dog barking in park\"` |\n| `--model` | Model ID | `ControlFoley` | `--model ControlFoley` |\n| `--duration` | Audio length in seconds (max 30) | `8` | `--duration 15` |\n| `--negative` | Negative prompt to exclude unwanted sounds | — | `--negative \"noise, human voice\"` |\n| `--cfg` | CFG strength — higher = stricter prompt adherence | `4.5` | `--cfg 6.0` |\n| `--count` | Number of variants to generate (1–5) | `1` | `--count 3` |\n| `--seed` | Fixed random seed for reproducibility | — | `--seed 42` |\n| `-o/--outdir` | Output directory | `./output` | `-o ./my_audio` |\n\n### V2A (Video-to-Audio)\n\n| Parameter | Description | Default | Example |\n|-----------|-------------|---------|---------|\n| `video` | Input video path (required) | — | `input.mp4` |\n| `--model` | Model ID | `ControlFoley` | `--model ControlFoley` |\n| `--prompt` | Text prompt to guide audio generation (TC-V2A) | — | `--prompt \"keyboard tapping\"` |\n| `--negative` | Negative prompt to exclude unwanted sounds | — | `--negative \"music, noise\"` |\n| `--ref-audio` | Reference audio file for timbre control (AC-V2A) | — | `--ref-audio reference.wav` |\n| `--cfg` | CFG strength | `4.5` | `--cfg 7.0` |\n| `--count` | Number of variants to generate (1–5) | `1` | `--count 2` |\n| `--seed` | Fixed random seed (not forwarded to API currently) | — | `--seed 42` |\n| `-o/--outdir` | Output directory | `./output` | `-o ./results` |\n\n## Prompt Tips\n\n- **Be specific**: `\"cat footsteps on wooden floor\"` beats `\"cat sound\"`\n- **Use negative prompts**: `--negative \"human voice, music, noise\"` to filter unwanted audio\n- **CFG tuning**: high CFG (6.0–7.5) for precise control, low CFG (3.0–4.5) for creative freedom\n\n## Output & Post-Processing\n\n- **Audio**: `.flac` (44100 Hz, lossless)\n- **Video**: `.mp4` (original video + generated audio track)\n- Results saved to `--outdir`, paths printed to stdout\n\n**Convert to MP3 for sharing:**\n\n```bash\nffmpeg -i output.flac -codec:a libmp3lame -qscale:a 2 output.mp3\n```\n\n## Error Handling\n\n| Issue | Cause | Fix |\n|-------|-------|-----|\n| Internal URL inaccessible | Result URL uses `.xiaomi.srv` internal domain | Script auto-falls back to `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` |\n| Queue busy | Task is waiting | Script auto-polls up to ~5 min; check load via `curl $API_BASE/health` |\n| Model unavailable | Model not enabled | Run `foley.py models` to see available models |\n| Task timeout | Service overloaded | Resubmit the task |\n\n## API Reference\n\nSee ./references/api-reference.md for full endpoint documentation.\n\n## ⚠️ Privacy & Security\n\n- **Service Operator**: Cloud processing is operated by the Xiaomi LLM Plus Team at `https://controlfoley.ai.xiaomi.com`\n- **Data Upload**: V2A/TC-V2A/AC-V2A modes upload the full video file to the remote service for processing. Do not upload videos containing sensitive personal or identifiable information\n- **Data Processing**: Uploaded videos and audio are used solely for audio generation. Results are returned via URL. Refer to the Xiaomi LLM Plus Team's terms of service for data retention and access control policies\n- **No API Key Required**: The service requires no authentication — please use it responsibly to avoid unnecessary load\n- **Recommendation**: Before first use, validate with a small, non-sensitive test clip\n\nFile v1.0.7:_meta.json\n\n{\n  \"ownerId\": \"kn7ftwffrk08e278sse17ktw39858t8n\",\n  \"slug\": \"controlfoley-audio-generator\",\n  \"version\": \"1.0.7\",\n  \"publishedAt\": 1776925711495\n}\n\nFile v1.0.7:references/api-reference.md\n\n# ControlFoley Audio Generator API Reference\n\nBase URL: `https://controlfoley.ai.xiaomi.com`  \nAuth: None required\n\n**Known API quirks (from real testing):**\n- Submit response may return either `task_id` or `taskId` (camelCase); code handles both\n- Status success may return either `processed_urls` or `urls`; code checks both\n- Result URLs may use internal `.xiaomi.srv` domains; use download endpoint as fallback\n\n## Endpoints\n\n### POST `/api/v1/v2a/submit` — Submit Task\n\nContent-Type: `multipart/form-data`\n\n#### T2A (Text-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| prompt | string | Yes | — | Audio description (e.g. `dog barking in park`) |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| duration | float | Optional | `8.0` | Audio length in seconds (max 30) |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n#### V2A (Video-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| video | file | Yes | — | Video file (.mp4/.webm/.mov) |\n| prompt | string | Optional | — | Text prompt to guide audio generation; omitted if empty |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| ref_audio | file | Optional | — | Reference audio for AC-V2A task; omitted if file not found |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n> Note: `seed` is accepted in the V2A function signature but not currently forwarded to the API.\n\n**Response (202):**\n```json\n{\"task_id\": \"uuid\", \"status\": \"pending\", \"queue_pos\": 2, \"message\": \"Task submitted successfully\"}\n```\n> Note: field may appear as `taskId` (camelCase) — handle both.\n\n### GET `/api/v1/v2a/status/{task_id}` — Poll Status\n\n**Pending/Processing:**\n```json\n{\"status\": \"pending|processing\", \"queue_pos\": 3}\n```\n\n**Success:**\n```json\n{\n  \"status\": \"success\", \"done\": true,\n  \"processed_urls\": [\"http://.../uuid.flac\", \"http://.../uuid.mp4\"]\n}\n```\n> Note: result URLs field may appear as `urls` — handle both `processed_urls` and `urls`.\n\n**Failed:**\n```json\n{\"status\": \"failed\", \"done\": true, \"error\": \"reason\"}\n```\n\n### GET `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` — Download Result\n\nFallback download endpoint for internal URLs that are not directly accessible.  \nReturns binary file stream (.flac at 44100 Hz or .mp4).\n\n### GET `/api/v1/v2a/models` — List Models\n\n```json\n{\"models\": [{\"name\": \"ControlFoley\", \"enabled\": true}]}\n```\n\n### GET `/health` — Health Check\n\n```json\n{\"status\": \"ok\", \"queue_size\": 3}\n```\n\n## Models\n\n| Model | Use Case |\n|-------|----------|\n| ControlFoley | General T2A and V2A (default) |\n\n## Modes\n\n| Mode | Input | Output | Description |\n|------|-------|--------|-------------|\n| T2A | prompt | .flac | Generate audio from text descriptions |\n| V2A | video file | .mp4 + .flac | Generate audio that matches video content |\n| TC-V2A | prompt + video file | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| AC-V2A | reference_audio + video file | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n\nArchive v1.0.6: 4 files, 7222 bytes\n\nFiles: references/api-reference.md (3634b), scripts/foley.py (8066b), SKILL.md (6719b), _meta.json (147b)\n\nFile v1.0.6:SKILL.md\n\n---\nname: controlfoley-audio-generator\ndescription: >\n  A multi-functional audio generation tool for SFX generation, video-to-audio and text-to-audio.\n  多功能音频生成工具，集成可控视频生成音频、文本生成音频等功能.\n\n---\n\n# ControlFoley Audio Generator (CLI version)\n\nA multi-functional audio generation tool powered by the ControlFoley model, integrating video sound effect (SFX) generation, video background music composition, text-to-audio and other functions to realize diversified creative audio generation. \n\nThis tool supports four modes: Video-to-Audio (V2A), Text-Controlled Video-to-Audio (TC-V2A), Audio-Controlled Video-to-Audio (AC-V2A), and Text-to-Audio (T2A).\n\nIf you find this project useful, please consider giving a star ⭐️ on our [GitHub](https://github.com/xiaomi-research/controlfoley) and [ClawHub](https://clawhub.ai/yjx-research/controlfoley-audio-generator) pages ~\n\n## Basic Info\n\n| Field | Value |\n|---|---|\n| Service Operator | Xiaomi LLM Plus Team |\n| API Endpoint | `https://llmplus.ai.xiaomi.com` |\n| Open Source Repo | `https://github.com/xiaomi-research/controlfoley` |\n| Project Page | `https://yjx-research.github.io/ControlFoley_web_page/` |\n| Online Demo | `https://yjx-research.github.io/ControlFoley_web_page/#try-gen` |\n| Model Weights | `https://huggingface.co/YJX-Xiaomi/ControlFoley/` |\n| API Key | Not required |\n| Script Path | `scripts/foley.py` |\n\n## Prerequisites\n\n```bash\npython3 --version   # Python 3.x\ncurl --version      # curl for API submission\nffmpeg -version     # optional, for audio format conversion\n```\n\n## Modes\n\n| Mode | Command | Input | Output | Description |\n|------|---------|-------|--------|-------------|\n| **V2A** | `v2a video.mp4` | Video file | .mp4 + .flac | Generate audio matching the video content |\n| **TC-V2A** | `v2a video.mp4 --prompt \"text\"` | Video + text | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| **AC-V2A** | `v2a video.mp4 --ref-audio ref.wav` | Video + reference audio | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n| **T2A** | `t2a \"prompt\"` | Text description | .flac | Generate audio from text descriptions |\n\n\n## Usage\n\n### 1. Text-to-Audio (T2A, default 8s)\n\n```bash\npython3 scripts/foley.py t2a \"dog barking loudly in a park\"\n```\n\n### 2. Video-to-Audio (V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4\n```\n\n### 3. Text-Controlled Video-to-Audio (TC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --prompt \"footsteps on gravel with birds chirping\"\n```\n\n### 4. Audio-Controlled Video-to-Audio (AC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --ref-audio reference.wav\n```\n\n### 5. Specify duration\n\n```bash\npython3 scripts/foley.py t2a \"A mountain stream murmurs, its gentle current lapping against the pebbles.\" --duration 15\n```\n\n### 6. Generate multiple candidates\n\n```bash\npython3 scripts/foley.py t2a \"cat purring softly\" --count 3\n```\n\n### 7. Fixed seed (reproducible results)\n\n```bash\npython3 scripts/foley.py t2a \"rain on a tin roof\" --seed 42\n```\n\n### 8. List available models\n\n```bash\npython3 scripts/foley.py models\n```\n\n## Parameters\n\n### T2A (Text-to-Audio)\n\n| Parameter | Description | Default | Example |\n|-----------|-------------|---------|---------|\n| `prompt` | Audio description text (required) | — | `\"dog barking in park\"` |\n| `--model` | Model ID | `ControlFoley` | `--model ControlFoley` |\n| `--duration` | Audio length in seconds (max 30) | `8` | `--duration 15` |\n| `--negative` | Negative prompt to exclude unwanted sounds | — | `--negative \"noise, human voice\"` |\n| `--cfg` | CFG strength — higher = stricter prompt adherence | `4.5` | `--cfg 6.0` |\n| `--count` | Number of variants to generate (1–5) | `1` | `--count 3` |\n| `--seed` | Fixed random seed for reproducibility | — | `--seed 42` |\n| `-o/--outdir` | Output directory | `./output` | `-o ./my_audio` |\n\n### V2A (Video-to-Audio)\n\n| Parameter | Description | Default | Example |\n|-----------|-------------|---------|---------|\n| `video` | Input video path (required) | — | `input.mp4` |\n| `--model` | Model ID | `ControlFoley` | `--model ControlFoley` |\n| `--prompt` | Text prompt to guide audio generation (TC-V2A) | — | `--prompt \"keyboard tapping\"` |\n| `--negative` | Negative prompt to exclude unwanted sounds | — | `--negative \"music, noise\"` |\n| `--ref-audio` | Reference audio file for timbre control (AC-V2A) | — | `--ref-audio reference.wav` |\n| `--cfg` | CFG strength | `4.5` | `--cfg 7.0` |\n| `--count` | Number of variants to generate (1–5) | `1` | `--count 2` |\n| `--seed` | Fixed random seed (not forwarded to API currently) | — | `--seed 42` |\n| `-o/--outdir` | Output directory | `./output` | `-o ./results` |\n\n## Prompt Tips\n\n- **Be specific**: `\"cat footsteps on wooden floor\"` beats `\"cat sound\"`\n- **Use negative prompts**: `--negative \"human voice, music, noise\"` to filter unwanted audio\n- **CFG tuning**: high CFG (6.0–7.5) for precise control, low CFG (3.0–4.5) for creative freedom\n\n## Output & Post-Processing\n\n- **Audio**: `.flac` (44100 Hz, lossless)\n- **Video**: `.mp4` (original video + generated audio track)\n- Results saved to `--outdir`, paths printed to stdout\n\n**Convert to MP3 for sharing:**\n\n```bash\nffmpeg -i output.flac -codec:a libmp3lame -qscale:a 2 output.mp3\n```\n\n## Error Handling\n\n| Issue | Cause | Fix |\n|-------|-------|-----|\n| Internal URL inaccessible | Result URL uses `.xiaomi.srv` internal domain | Script auto-falls back to `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` |\n| Queue busy | Task is waiting | Script auto-polls up to ~5 min; check load via `curl $API_BASE/health` |\n| Model unavailable | Model not enabled | Run `foley.py models` to see available models |\n| Task timeout | Service overloaded | Resubmit the task |\n\n## API Reference\n\nSee ./references/api-reference.md for full endpoint documentation.\n\n## ⚠️ Privacy & Security\n\n- **Service Operator**: Cloud processing is operated by the Xiaomi LLM Plus Team at `https://llmplus.ai.xiaomi.com`\n- **Data Upload**: V2A/TC-V2A/AC-V2A modes upload the full video file to the remote service for processing. Do not upload videos containing sensitive personal or identifiable information\n- **Data Processing**: Uploaded videos and audio are used solely for audio generation. Results are returned via URL. Refer to the Xiaomi LLM Plus Team's terms of service for data retention and access control policies\n- **No API Key Required**: The service requires no authentication — please use it responsibly to avoid unnecessary load\n- **Recommendation**: Before first use, validate with a small, non-sensitive test clip\n\nFile v1.0.6:_meta.json\n\n{\n  \"ownerId\": \"kn7ftwffrk08e278sse17ktw39858t8n\",\n  \"slug\": \"controlfoley-audio-generator\",\n  \"version\": \"1.0.6\",\n  \"publishedAt\": 1776823656244\n}\n\nFile v1.0.6:references/api-reference.md\n\n# ControlFoley Audio Generator API Reference\n\nBase URL: `https://llmplus.ai.xiaomi.com`  \nAuth: None required\n\n**Known API quirks (from real testing):**\n- Submit response may return either `task_id` or `taskId` (camelCase); code handles both\n- Status success may return either `processed_urls` or `urls`; code checks both\n- Result URLs may use internal `.xiaomi.srv` domains; use download endpoint as fallback\n\n## Endpoints\n\n### POST `/api/v1/v2a/submit` — Submit Task\n\nContent-Type: `multipart/form-data`\n\n#### T2A (Text-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| prompt | string | Yes | — | Audio description (e.g. `dog barking in park`) |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| duration | float | Optional | `8.0` | Audio length in seconds (max 30) |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n#### V2A (Video-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| video | file | Yes | — | Video file (.mp4/.webm/.mov) |\n| prompt | string | Optional | — | Text prompt to guide audio generation; omitted if empty |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| ref_audio | file | Optional | — | Reference audio for AC-V2A task; omitted if file not found |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n> Note: `seed` is accepted in the V2A function signature but not currently forwarded to the API.\n\n**Response (202):**\n```json\n{\"task_id\": \"uuid\", \"status\": \"pending\", \"queue_pos\": 2, \"message\": \"Task submitted successfully\"}\n```\n> Note: field may appear as `taskId` (camelCase) — handle both.\n\n### GET `/api/v1/v2a/status/{task_id}` — Poll Status\n\n**Pending/Processing:**\n```json\n{\"status\": \"pending|processing\", \"queue_pos\": 3}\n```\n\n**Success:**\n```json\n{\n  \"status\": \"success\", \"done\": true,\n  \"processed_urls\": [\"http://.../uuid.flac\", \"http://.../uuid.mp4\"]\n}\n```\n> Note: result URLs field may appear as `urls` — handle both `processed_urls` and `urls`.\n\n**Failed:**\n```json\n{\"status\": \"failed\", \"done\": true, \"error\": \"reason\"}\n```\n\n### GET `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` — Download Result\n\nFallback download endpoint for internal URLs that are not directly accessible.  \nReturns binary file stream (.flac at 44100 Hz or .mp4).\n\n### GET `/api/v1/v2a/models` — List Models\n\n```json\n{\"models\": [{\"name\": \"ControlFoley\", \"enabled\": true}]}\n```\n\n### GET `/health` — Health Check\n\n```json\n{\"status\": \"ok\", \"queue_size\": 3}\n```\n\n## Models\n\n| Model | Use Case |\n|-------|----------|\n| ControlFoley | General T2A and V2A (default) |\n\n## Modes\n\n| Mode | Input | Output | Description |\n|------|-------|--------|-------------|\n| T2A | prompt | .flac | Generate audio from text descriptions |\n| V2A | video file | .mp4 + .flac | Generate audio that matches video content |\n| TC-V2A | prompt + video file | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| AC-V2A | reference_audio + video file | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n\nArchive v1.0.5: 4 files, 7130 bytes\n\nFiles: references/api-reference.md (3634b), scripts/foley.py (8066b), SKILL.md (6499b), _meta.json (147b)\n\nFile v1.0.5:SKILL.md\n\n---\nname: controlfoley-audio-generator\ndescription: >\n  A multi-functional audio generation tool for SFX generation, video-to-audio and text-to-audio.\n  多功能音频生成工具，集成可控视频生成音频、文本生成音频等功能.\n\n---\n\n# ControlFoley Audio Generator (CLI version)\n\nA multi-functional audio generation tool powered by the ControlFoley model, integrating video sound effect (SFX) generation, video background music composition, text-to-audio and other functions to realize diversified creative audio generation. \n\nThis tool supports four modes: Video-to-Audio (V2A), Text-Controlled Video-to-Audio (TC-V2A), Audio-Controlled Video-to-Audio (AC-V2A), and Text-to-Audio (T2A).\n\n## Basic Info\n\n| Field | Value |\n|---|---|\n| Service Operator | Xiaomi LLM Plus Team |\n| API Endpoint | `https://llmplus.ai.xiaomi.com` |\n| Open Source Repo | `https://github.com/xiaomi-research/controlfoley` |\n| Project Page | `https://yjx-research.github.io/ControlFoley_web_page/` |\n| Online Demo | `https://yjx-research.github.io/ControlFoley_web_page/#try-gen` |\n| Model Weights | `https://huggingface.co/YJX-Xiaomi/ControlFoley/` |\n| API Key | Not required |\n| Script Path | `scripts/foley.py` |\n\n## Prerequisites\n\n```bash\npython3 --version   # Python 3.x\ncurl --version      # curl for API submission\nffmpeg -version     # optional, for audio format conversion\n```\n\n## Modes\n\n| Mode | Command | Input | Output | Description |\n|------|---------|-------|--------|-------------|\n| **V2A** | `v2a video.mp4` | Video file | .mp4 + .flac | Generate audio matching the video content |\n| **TC-V2A** | `v2a video.mp4 --prompt \"text\"` | Video + text | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| **AC-V2A** | `v2a video.mp4 --ref-audio ref.wav` | Video + reference audio | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n| **T2A** | `t2a \"prompt\"` | Text description | .flac | Generate audio from text descriptions |\n\n\n## Usage\n\n### 1. Text-to-Audio (T2A, default 8s)\n\n```bash\npython3 scripts/foley.py t2a \"dog barking loudly in a park\"\n```\n\n### 2. Video-to-Audio (V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4\n```\n\n### 3. Text-Controlled Video-to-Audio (TC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --prompt \"footsteps on gravel with birds chirping\"\n```\n\n### 4. Audio-Controlled Video-to-Audio (AC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --ref-audio reference.wav\n```\n\n### 5. Specify duration\n\n```bash\npython3 scripts/foley.py t2a \"A mountain stream murmurs, its gentle current lapping against the pebbles.\" --duration 15\n```\n\n### 6. Generate multiple candidates\n\n```bash\npython3 scripts/foley.py t2a \"cat purring softly\" --count 3\n```\n\n### 7. Fixed seed (reproducible results)\n\n```bash\npython3 scripts/foley.py t2a \"rain on a tin roof\" --seed 42\n```\n\n### 8. List available models\n\n```bash\npython3 scripts/foley.py models\n```\n\n## Parameters\n\n### T2A (Text-to-Audio)\n\n| Parameter | Description | Default | Example |\n|-----------|-------------|---------|---------|\n| `prompt` | Audio description text (required) | — | `\"dog barking in park\"` |\n| `--model` | Model ID | `ControlFoley` | `--model ControlFoley` |\n| `--duration` | Audio length in seconds (max 30) | `8` | `--duration 15` |\n| `--negative` | Negative prompt to exclude unwanted sounds | — | `--negative \"noise, human voice\"` |\n| `--cfg` | CFG strength — higher = stricter prompt adherence | `4.5` | `--cfg 6.0` |\n| `--count` | Number of variants to generate (1–5) | `1` | `--count 3` |\n| `--seed` | Fixed random seed for reproducibility | — | `--seed 42` |\n| `-o/--outdir` | Output directory | `./output` | `-o ./my_audio` |\n\n### V2A (Video-to-Audio)\n\n| Parameter | Description | Default | Example |\n|-----------|-------------|---------|---------|\n| `video` | Input video path (required) | — | `input.mp4` |\n| `--model` | Model ID | `ControlFoley` | `--model ControlFoley` |\n| `--prompt` | Text prompt to guide audio generation (TC-V2A) | — | `--prompt \"keyboard tapping\"` |\n| `--negative` | Negative prompt to exclude unwanted sounds | — | `--negative \"music, noise\"` |\n| `--ref-audio` | Reference audio file for timbre control (AC-V2A) | — | `--ref-audio reference.wav` |\n| `--cfg` | CFG strength | `4.5` | `--cfg 7.0` |\n| `--count` | Number of variants to generate (1–5) | `1` | `--count 2` |\n| `--seed` | Fixed random seed (not forwarded to API currently) | — | `--seed 42` |\n| `-o/--outdir` | Output directory | `./output` | `-o ./results` |\n\n## Prompt Tips\n\n- **Be specific**: `\"cat footsteps on wooden floor\"` beats `\"cat sound\"`\n- **Use negative prompts**: `--negative \"human voice, music, noise\"` to filter unwanted audio\n- **CFG tuning**: high CFG (6.0–7.5) for precise control, low CFG (3.0–4.5) for creative freedom\n\n## Output & Post-Processing\n\n- **Audio**: `.flac` (44100 Hz, lossless)\n- **Video**: `.mp4` (original video + generated audio track)\n- Results saved to `--outdir`, paths printed to stdout\n\n**Convert to MP3 for sharing:**\n\n```bash\nffmpeg -i output.flac -codec:a libmp3lame -qscale:a 2 output.mp3\n```\n\n## Error Handling\n\n| Issue | Cause | Fix |\n|-------|-------|-----|\n| Internal URL inaccessible | Result URL uses `.xiaomi.srv` internal domain | Script auto-falls back to `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` |\n| Queue busy | Task is waiting | Script auto-polls up to ~5 min; check load via `curl $API_BASE/health` |\n| Model unavailable | Model not enabled | Run `foley.py models` to see available models |\n| Task timeout | Service overloaded | Resubmit the task |\n\n## API Reference\n\nSee ./references/api-reference.md for full endpoint documentation.\n\n## ⚠️ Privacy & Security\n\n- **Service Operator**: Cloud processing is operated by the Xiaomi LLM Plus Team at `https://llmplus.ai.xiaomi.com`\n- **Data Upload**: V2A/TC-V2A/AC-V2A modes upload the full video file to the remote service for processing. Do not upload videos containing sensitive personal or identifiable information\n- **Data Processing**: Uploaded videos and audio are used solely for audio generation. Results are returned via URL. Refer to the Xiaomi LLM Plus Team's terms of service for data retention and access control policies\n- **No API Key Required**: The service requires no authentication — please use it responsibly to avoid unnecessary load\n- **Recommendation**: Before first use, validate with a small, non-sensitive test clip\n\nFile v1.0.5:_meta.json\n\n{\n  \"ownerId\": \"kn7ftwffrk08e278sse17ktw39858t8n\",\n  \"slug\": \"controlfoley-audio-generator\",\n  \"version\": \"1.0.5\",\n  \"publishedAt\": 1776762285027\n}\n\nFile v1.0.5:references/api-reference.md\n\n# ControlFoley Audio Generator API Reference\n\nBase URL: `https://llmplus.ai.xiaomi.com`  \nAuth: None required\n\n**Known API quirks (from real testing):**\n- Submit response may return either `task_id` or `taskId` (camelCase); code handles both\n- Status success may return either `processed_urls` or `urls`; code checks both\n- Result URLs may use internal `.xiaomi.srv` domains; use download endpoint as fallback\n\n## Endpoints\n\n### POST `/api/v1/v2a/submit` — Submit Task\n\nContent-Type: `multipart/form-data`\n\n#### T2A (Text-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| prompt | string | Yes | — | Audio description (e.g. `dog barking in park`) |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| duration | float | Optional | `8.0` | Audio length in seconds (max 30) |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n#### V2A (Video-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| video | file | Yes | — | Video file (.mp4/.webm/.mov) |\n| prompt | string | Optional | — | Text prompt to guide audio generation; omitted if empty |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| ref_audio | file | Optional | — | Reference audio for AC-V2A task; omitted if file not found |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n> Note: `seed` is accepted in the V2A function signature but not currently forwarded to the API.\n\n**Response (202):**\n```json\n{\"task_id\": \"uuid\", \"status\": \"pending\", \"queue_pos\": 2, \"message\": \"Task submitted successfully\"}\n```\n> Note: field may appear as `taskId` (camelCase) — handle both.\n\n### GET `/api/v1/v2a/status/{task_id}` — Poll Status\n\n**Pending/Processing:**\n```json\n{\"status\": \"pending|processing\", \"queue_pos\": 3}\n```\n\n**Success:**\n```json\n{\n  \"status\": \"success\", \"done\": true,\n  \"processed_urls\": [\"http://.../uuid.flac\", \"http://.../uuid.mp4\"]\n}\n```\n> Note: result URLs field may appear as `urls` — handle both `processed_urls` and `urls`.\n\n**Failed:**\n```json\n{\"status\": \"failed\", \"done\": true, \"error\": \"reason\"}\n```\n\n### GET `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` — Download Result\n\nFallback download endpoint for internal URLs that are not directly accessible.  \nReturns binary file stream (.flac at 44100 Hz or .mp4).\n\n### GET `/api/v1/v2a/models` — List Models\n\n```json\n{\"models\": [{\"name\": \"ControlFoley\", \"enabled\": true}]}\n```\n\n### GET `/health` — Health Check\n\n```json\n{\"status\": \"ok\", \"queue_size\": 3}\n```\n\n## Models\n\n| Model | Use Case |\n|-------|----------|\n| ControlFoley | General T2A and V2A (default) |\n\n## Modes\n\n| Mode | Input | Output | Description |\n|------|-------|--------|-------------|\n| T2A | prompt | .flac | Generate audio from text descriptions |\n| V2A | video file | .mp4 + .flac | Generate audio that matches video content |\n| TC-V2A | prompt + video file | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| AC-V2A | reference_audio + video file | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n\nArchive v1.0.4: 4 files, 7130 bytes\n\nFiles: references/api-reference.md (3634b), scripts/foley.py (8066b), SKILL.md (6499b), _meta.json (147b)\n\nFile v1.0.4:SKILL.md\n\n---\nname: controlfoley-audio-generator\ndescription: >\n  A multi-functional audio generation tool for SFX generation, video-to-audio and text-to-audio.\n  多功能音频生成工具，集成可控视频生成音频、文本生成音频等功能.\n\n---\n\n# ControlFoley Audio Generator (CLI version)\n\nA multi-functional audio generation tool powered by the ControlFoley model, integrating video sound effect (SFX) generation, video background music composition, text-to-audio and other functions to realize diversified creative audio generation. \n\nThis tool supports four modes: Video-to-Audio (V2A), Text-Controlled Video-to-Audio (TC-V2A), Audio-Controlled Video-to-Audio (AC-V2A), and Text-to-Audio (T2A).\n\n## Basic Info\n\n| Field | Value |\n|---|---|\n| Service Operator | Xiaomi LLM Plus Team |\n| API Endpoint | `https://llmplus.ai.xiaomi.com` |\n| Open Source Repo | `https://github.com/xiaomi-research/controlfoley` |\n| Project Page | `https://yjx-research.github.io/ControlFoley_web_page/` |\n| Online Demo | `https://yjx-research.github.io/ControlFoley_web_page/#try-gen` |\n| Model Weights | `https://huggingface.co/YJX-Xiaomi/ControlFoley/` |\n| API Key | Not required |\n| Script Path | `scripts/foley.py` |\n\n## Prerequisites\n\n```bash\npython3 --version   # Python 3.x\ncurl --version      # curl for API submission\nffmpeg -version     # optional, for audio format conversion\n```\n\n## Modes\n\n| Mode | Command | Input | Output | Description |\n|------|---------|-------|--------|-------------|\n| **V2A** | `v2a video.mp4` | Video file | .mp4 + .flac | Generate audio matching the video content |\n| **TC-V2A** | `v2a video.mp4 --prompt \"text\"` | Video + text | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| **AC-V2A** | `v2a video.mp4 --ref-audio ref.wav` | Video + reference audio | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n| **T2A** | `t2a \"prompt\"` | Text description | .flac | Generate audio from text descriptions |\n\n\n## Usage\n\n### 1. Text-to-Audio (T2A, default 8s)\n\n```bash\npython3 scripts/foley.py t2a \"dog barking loudly in a park\"\n```\n\n### 2. Video-to-Audio (V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4\n```\n\n### 3. Text-Controlled Video-to-Audio (TC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --prompt \"footsteps on gravel with birds chirping\"\n```\n\n### 4. Audio-Controlled Video-to-Audio (AC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --ref-audio reference.wav\n```\n\n### 5. Specify duration\n\n```bash\npython3 scripts/foley.py t2a \"A mountain stream murmurs, its gentle current lapping against the pebbles.\" --duration 15\n```\n\n### 6. Generate multiple candidates\n\n```bash\npython3 scripts/foley.py t2a \"cat purring softly\" --count 3\n```\n\n### 7. Fixed seed (reproducible results)\n\n```bash\npython3 scripts/foley.py t2a \"rain on a tin roof\" --seed 42\n```\n\n### 8. List available models\n\n```bash\npython3 scripts/foley.py models\n```\n\n## Parameters\n\n### T2A (Text-to-Audio)\n\n| Parameter | Description | Default | Example |\n|-----------|-------------|---------|---------|\n| `prompt` | Audio description text (required) | — | `\"dog barking in park\"` |\n| `--model` | Model ID | `ControlFoley` | `--model ControlFoley` |\n| `--duration` | Audio length in seconds (max 30) | `8` | `--duration 15` |\n| `--negative` | Negative prompt to exclude unwanted sounds | — | `--negative \"noise, human voice\"` |\n| `--cfg` | CFG strength — higher = stricter prompt adherence | `4.5` | `--cfg 6.0` |\n| `--count` | Number of variants to generate (1–5) | `1` | `--count 3` |\n| `--seed` | Fixed random seed for reproducibility | — | `--seed 42` |\n| `-o/--outdir` | Output directory | `./output` | `-o ./my_audio` |\n\n### V2A (Video-to-Audio)\n\n| Parameter | Description | Default | Example |\n|-----------|-------------|---------|---------|\n| `video` | Input video path (required) | — | `input.mp4` |\n| `--model` | Model ID | `ControlFoley` | `--model ControlFoley` |\n| `--prompt` | Text prompt to guide audio generation (TC-V2A) | — | `--prompt \"keyboard tapping\"` |\n| `--negative` | Negative prompt to exclude unwanted sounds | — | `--negative \"music, noise\"` |\n| `--ref-audio` | Reference audio file for timbre control (AC-V2A) | — | `--ref-audio reference.wav` |\n| `--cfg` | CFG strength | `4.5` | `--cfg 7.0` |\n| `--count` | Number of variants to generate (1–5) | `1` | `--count 2` |\n| `--seed` | Fixed random seed (not forwarded to API currently) | — | `--seed 42` |\n| `-o/--outdir` | Output directory | `./output` | `-o ./results` |\n\n## Prompt Tips\n\n- **Be specific**: `\"cat footsteps on wooden floor\"` beats `\"cat sound\"`\n- **Use negative prompts**: `--negative \"human voice, music, noise\"` to filter unwanted audio\n- **CFG tuning**: high CFG (6.0–7.5) for precise control, low CFG (3.0–4.5) for creative freedom\n\n## Output & Post-Processing\n\n- **Audio**: `.flac` (44100 Hz, lossless)\n- **Video**: `.mp4` (original video + generated audio track)\n- Results saved to `--outdir`, paths printed to stdout\n\n**Convert to MP3 for sharing:**\n\n```bash\nffmpeg -i output.flac -codec:a libmp3lame -qscale:a 2 output.mp3\n```\n\n## Error Handling\n\n| Issue | Cause | Fix |\n|-------|-------|-----|\n| Internal URL inaccessible | Result URL uses `.xiaomi.srv` internal domain | Script auto-falls back to `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` |\n| Queue busy | Task is waiting | Script auto-polls up to ~5 min; check load via `curl $API_BASE/health` |\n| Model unavailable | Model not enabled | Run `foley.py models` to see available models |\n| Task timeout | Service overloaded | Resubmit the task |\n\n## API Reference\n\nSee ./references/api-reference.md for full endpoint documentation.\n\n## ⚠️ Privacy & Security\n\n- **Service Operator**: Cloud processing is operated by the Xiaomi LLM Plus Team at `https://llmplus.ai.xiaomi.com`\n- **Data Upload**: V2A/TC-V2A/AC-V2A modes upload the full video file to the remote service for processing. Do not upload videos containing sensitive personal or identifiable information\n- **Data Processing**: Uploaded videos and audio are used solely for audio generation. Results are returned via URL. Refer to the Xiaomi LLM Plus Team's terms of service for data retention and access control policies\n- **No API Key Required**: The service requires no authentication — please use it responsibly to avoid unnecessary load\n- **Recommendation**: Before first use, validate with a small, non-sensitive test clip\n\nFile v1.0.4:_meta.json\n\n{\n  \"ownerId\": \"kn7ftwffrk08e278sse17ktw39858t8n\",\n  \"slug\": \"controlfoley-audio-generator\",\n  \"version\": \"1.0.4\",\n  \"publishedAt\": 1776761211412\n}\n\nFile v1.0.4:references/api-reference.md\n\n# ControlFoley Audio Generator API Reference\n\nBase URL: `https://llmplus.ai.xiaomi.com`  \nAuth: None required\n\n**Known API quirks (from real testing):**\n- Submit response may return either `task_id` or `taskId` (camelCase); code handles both\n- Status success may return either `processed_urls` or `urls`; code checks both\n- Result URLs may use internal `.xiaomi.srv` domains; use download endpoint as fallback\n\n## Endpoints\n\n### POST `/api/v1/v2a/submit` — Submit Task\n\nContent-Type: `multipart/form-data`\n\n#### T2A (Text-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| prompt | string | Yes | — | Audio description (e.g. `dog barking in park`) |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| duration | float | Optional | `8.0` | Audio length in seconds (max 30) |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n#### V2A (Video-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| video | file | Yes | — | Video file (.mp4/.webm/.mov) |\n| prompt | string | Optional | — | Text prompt to guide audio generation; omitted if empty |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| ref_audio | file | Optional | — | Reference audio for AC-V2A task; omitted if file not found |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n> Note: `seed` is accepted in the V2A function signature but not currently forwarded to the API.\n\n**Response (202):**\n```json\n{\"task_id\": \"uuid\", \"status\": \"pending\", \"queue_pos\": 2, \"message\": \"Task submitted successfully\"}\n```\n> Note: field may appear as `taskId` (camelCase) — handle both.\n\n### GET `/api/v1/v2a/status/{task_id}` — Poll Status\n\n**Pending/Processing:**\n```json\n{\"status\": \"pending|processing\", \"queue_pos\": 3}\n```\n\n**Success:**\n```json\n{\n  \"status\": \"success\", \"done\": true,\n  \"processed_urls\": [\"http://.../uuid.flac\", \"http://.../uuid.mp4\"]\n}\n```\n> Note: result URLs field may appear as `urls` — handle both `processed_urls` and `urls`.\n\n**Failed:**\n```json\n{\"status\": \"failed\", \"done\": true, \"error\": \"reason\"}\n```\n\n### GET `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` — Download Result\n\nFallback download endpoint for internal URLs that are not directly accessible.  \nReturns binary file stream (.flac at 44100 Hz or .mp4).\n\n### GET `/api/v1/v2a/models` — List Models\n\n```json\n{\"models\": [{\"name\": \"ControlFoley\", \"enabled\": true}]}\n```\n\n### GET `/health` — Health Check\n\n```json\n{\"status\": \"ok\", \"queue_size\": 3}\n```\n\n## Models\n\n| Model | Use Case |\n|-------|----------|\n| ControlFoley | General T2A and V2A (default) |\n\n## Modes\n\n| Mode | Input | Output | Description |\n|------|-------|--------|-------------|\n| T2A | prompt | .flac | Generate audio from text descriptions |\n| V2A | video file | .mp4 + .flac | Generate audio that matches video content |\n| TC-V2A | prompt + video file | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| AC-V2A | reference_audio + video file | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n\nArchive v1.0.3: 4 files, 7219 bytes\n\nFiles: references/api-reference.md (3634b), scripts/foley.py (8342b), SKILL.md (6499b), _meta.json (147b)\n\nFile v1.0.3:SKILL.md\n\n---\nname: controlfoley-audio-generator\ndescription: >\n  A multi-functional audio generation tool for SFX generation, video-to-audio and text-to-audio.\n  多功能音频生成工具，集成可控视频生成音频、文本生成音频等功能.\n\n---\n\n# ControlFoley Audio Generator (CLI version)\n\nA multi-functional audio generation tool powered by the ControlFoley model, integrating video sound effect (SFX) generation, video background music composition, text-to-audio and other functions to realize diversified creative audio generation. \n\nThis tool supports four modes: Video-to-Audio (V2A), Text-Controlled Video-to-Audio (TC-V2A), Audio-Controlled Video-to-Audio (AC-V2A), and Text-to-Audio (T2A).\n\n## Basic Info\n\n| Field | Value |\n|---|---|\n| Service Operator | Xiaomi LLM Plus Team |\n| API Endpoint | `https://llmplus.ai.xiaomi.com` |\n| Open Source Repo | `https://github.com/xiaomi-research/controlfoley` |\n| Project Page | `https://yjx-research.github.io/ControlFoley_web_page/` |\n| Online Demo | `https://yjx-research.github.io/ControlFoley_web_page/#try-gen` |\n| Model Weights | `https://huggingface.co/YJX-Xiaomi/ControlFoley/` |\n| API Key | Not required |\n| Script Path | `scripts/foley.py` |\n\n## Prerequisites\n\n```bash\npython3 --version   # Python 3.x\ncurl --version      # curl for API submission\nffmpeg -version     # optional, for audio format conversion\n```\n\n## Modes\n\n| Mode | Command | Input | Output | Description |\n|------|---------|-------|--------|-------------|\n| **V2A** | `v2a video.mp4` | Video file | .mp4 + .flac | Generate audio matching the video content |\n| **TC-V2A** | `v2a video.mp4 --prompt \"text\"` | Video + text | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| **AC-V2A** | `v2a video.mp4 --ref-audio ref.wav` | Video + reference audio | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n| **T2A** | `t2a \"prompt\"` | Text description | .flac | Generate audio from text descriptions |\n\n\n## Usage\n\n### 1. Text-to-Audio (T2A, default 8s)\n\n```bash\npython3 scripts/foley.py t2a \"dog barking loudly in a park\"\n```\n\n### 2. Video-to-Audio (V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4\n```\n\n### 3. Text-Controlled Video-to-Audio (TC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --prompt \"footsteps on gravel with birds chirping\"\n```\n\n### 4. Audio-Controlled Video-to-Audio (AC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --ref-audio reference.wav\n```\n\n### 5. Specify duration\n\n```bash\npython3 scripts/foley.py t2a \"A mountain stream murmurs, its gentle current lapping against the pebbles.\" --duration 15\n```\n\n### 6. Generate multiple candidates\n\n```bash\npython3 scripts/foley.py t2a \"cat purring softly\" --count 3\n```\n\n### 7. Fixed seed (reproducible results)\n\n```bash\npython3 scripts/foley.py t2a \"rain on a tin roof\" --seed 42\n```\n\n### 8. List available models\n\n```bash\npython3 scripts/foley.py models\n```\n\n## Parameters\n\n### T2A (Text-to-Audio)\n\n| Parameter | Description | Default | Example |\n|-----------|-------------|---------|---------|\n| `prompt` | Audio description text (required) | — | `\"dog barking in park\"` |\n| `--model` | Model ID | `ControlFoley` | `--model ControlFoley` |\n| `--duration` | Audio length in seconds (max 30) | `8` | `--duration 15` |\n| `--negative` | Negative prompt to exclude unwanted sounds | — | `--negative \"noise, human voice\"` |\n| `--cfg` | CFG strength — higher = stricter prompt adherence | `4.5` | `--cfg 6.0` |\n| `--count` | Number of variants to generate (1–5) | `1` | `--count 3` |\n| `--seed` | Fixed random seed for reproducibility | — | `--seed 42` |\n| `-o/--outdir` | Output directory | `./output` | `-o ./my_audio` |\n\n### V2A (Video-to-Audio)\n\n| Parameter | Description | Default | Example |\n|-----------|-------------|---------|---------|\n| `video` | Input video path (required) | — | `input.mp4` |\n| `--model` | Model ID | `ControlFoley` | `--model ControlFoley` |\n| `--prompt` | Text prompt to guide audio generation (TC-V2A) | — | `--prompt \"keyboard tapping\"` |\n| `--negative` | Negative prompt to exclude unwanted sounds | — | `--negative \"music, noise\"` |\n| `--ref-audio` | Reference audio file for timbre control (AC-V2A) | — | `--ref-audio reference.wav` |\n| `--cfg` | CFG strength | `4.5` | `--cfg 7.0` |\n| `--count` | Number of variants to generate (1–5) | `1` | `--count 2` |\n| `--seed` | Fixed random seed (not forwarded to API currently) | — | `--seed 42` |\n| `-o/--outdir` | Output directory | `./output` | `-o ./results` |\n\n## Prompt Tips\n\n- **Be specific**: `\"cat footsteps on wooden floor\"` beats `\"cat sound\"`\n- **Use negative prompts**: `--negative \"human voice, music, noise\"` to filter unwanted audio\n- **CFG tuning**: high CFG (6.0–7.5) for precise control, low CFG (3.0–4.5) for creative freedom\n\n## Output & Post-Processing\n\n- **Audio**: `.flac` (44100 Hz, lossless)\n- **Video**: `.mp4` (original video + generated audio track)\n- Results saved to `--outdir`, paths printed to stdout\n\n**Convert to MP3 for sharing:**\n\n```bash\nffmpeg -i output.flac -codec:a libmp3lame -qscale:a 2 output.mp3\n```\n\n## Error Handling\n\n| Issue | Cause | Fix |\n|-------|-------|-----|\n| Internal URL inaccessible | Result URL uses `.xiaomi.srv` internal domain | Script auto-falls back to `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` |\n| Queue busy | Task is waiting | Script auto-polls up to ~5 min; check load via `curl $API_BASE/health` |\n| Model unavailable | Model not enabled | Run `foley.py models` to see available models |\n| Task timeout | Service overloaded | Resubmit the task |\n\n## API Reference\n\nSee ./references/api-reference.md for full endpoint documentation.\n\n## ⚠️ Privacy & Security\n\n- **Service Operator**: Cloud processing is operated by the Xiaomi LLM Plus Team at `https://llmplus.ai.xiaomi.com`\n- **Data Upload**: V2A/TC-V2A/AC-V2A modes upload the full video file to the remote service for processing. Do not upload videos containing sensitive personal or identifiable information\n- **Data Processing**: Uploaded videos and audio are used solely for audio generation. Results are returned via URL. Refer to the Xiaomi LLM Plus Team's terms of service for data retention and access control policies\n- **No API Key Required**: The service requires no authentication — please use it responsibly to avoid unnecessary load\n- **Recommendation**: Before first use, validate with a small, non-sensitive test clip\n\nFile v1.0.3:_meta.json\n\n{\n  \"ownerId\": \"kn7ftwffrk08e278sse17ktw39858t8n\",\n  \"slug\": \"controlfoley-audio-generator\",\n  \"version\": \"1.0.3\",\n  \"publishedAt\": 1776760710592\n}\n\nFile v1.0.3:references/api-reference.md\n\n# ControlFoley Audio Generator API Reference\n\nBase URL: `https://llmplus.ai.xiaomi.com`  \nAuth: None required\n\n**Known API quirks (from real testing):**\n- Submit response may return either `task_id` or `taskId` (camelCase); code handles both\n- Status success may return either `processed_urls` or `urls`; code checks both\n- Result URLs may use internal `.xiaomi.srv` domains; use download endpoint as fallback\n\n## Endpoints\n\n### POST `/api/v1/v2a/submit` — Submit Task\n\nContent-Type: `multipart/form-data`\n\n#### T2A (Text-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| prompt | string | Yes | — | Audio description (e.g. `dog barking in park`) |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| duration | float | Optional | `8.0` | Audio length in seconds (max 30) |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n#### V2A (Video-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| video | file | Yes | — | Video file (.mp4/.webm/.mov) |\n| prompt | string | Optional | — | Text prompt to guide audio generation; omitted if empty |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| ref_audio | file | Optional | — | Reference audio for AC-V2A task; omitted if file not found |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n> Note: `seed` is accepted in the V2A function signature but not currently forwarded to the API.\n\n**Response (202):**\n```json\n{\"task_id\": \"uuid\", \"status\": \"pending\", \"queue_pos\": 2, \"message\": \"Task submitted successfully\"}\n```\n> Note: field may appear as `taskId` (camelCase) — handle both.\n\n### GET `/api/v1/v2a/status/{task_id}` — Poll Status\n\n**Pending/Processing:**\n```json\n{\"status\": \"pending|processing\", \"queue_pos\": 3}\n```\n\n**Success:**\n```json\n{\n  \"status\": \"success\", \"done\": true,\n  \"processed_urls\": [\"http://.../uuid.flac\", \"http://.../uuid.mp4\"]\n}\n```\n> Note: result URLs field may appear as `urls` — handle both `processed_urls` and `urls`.\n\n**Failed:**\n```json\n{\"status\": \"failed\", \"done\": true, \"error\": \"reason\"}\n```\n\n### GET `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` — Download Result\n\nFallback download endpoint for internal URLs that are not directly accessible.  \nReturns binary file stream (.flac at 44100 Hz or .mp4).\n\n### GET `/api/v1/v2a/models` — List Models\n\n```json\n{\"models\": [{\"name\": \"ControlFoley\", \"enabled\": true}]}\n```\n\n### GET `/health` — Health Check\n\n```json\n{\"status\": \"ok\", \"queue_size\": 3}\n```\n\n## Models\n\n| Model | Use Case |\n|-------|----------|\n| ControlFoley | General T2A and V2A (default) |\n\n## Modes\n\n| Mode | Input | Output | Description |\n|------|-------|--------|-------------|\n| T2A | prompt | .flac | Generate audio from text descriptions |\n| V2A | video file | .mp4 + .flac | Generate audio that matches video content |\n| TC-V2A | prompt + video file | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| AC-V2A | reference_audio + video file | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n\nArchive v1.0.2: 4 files, 7207 bytes\n\nFiles: references/api-reference.md (3634b), scripts/foley.py (8342b), SKILL.md (6485b), _meta.json (147b)\n\nFile v1.0.2:SKILL.md\n\n---\nname: controlfoley-audio-generator\ndescription: >\n  A multi-functional audio generation tool for SFX generation, video-to-audio and text-to-audio.\n  多功能音频生成工具，集成可控视频生成音频、文本生成音频等功能.\n\n---\n\n# ControlFoley Audio Generator\n\nA multi-functional audio generation tool powered by the ControlFoley model, integrating video sound effect (SFX) generation, video background music composition, text-to-audio and other functions to realize diversified creative audio generation. \n\nThis tool supports four modes: Video-to-Audio (V2A), Text-Controlled Video-to-Audio (TC-V2A), Audio-Controlled Video-to-Audio (AC-V2A), and Text-to-Audio (T2A).\n\n## Basic Info\n\n| Field | Value |\n|---|---|\n| Service Operator | Xiaomi LLM Plus Team |\n| API Endpoint | `https://llmplus.ai.xiaomi.com` |\n| Open Source Repo | `https://github.com/xiaomi-research/controlfoley` |\n| Project Page | `https://yjx-research.github.io/ControlFoley_web_page/` |\n| Online Demo | `https://yjx-research.github.io/ControlFoley_web_page/#try-gen` |\n| Model Weights | `https://huggingface.co/YJX-Xiaomi/ControlFoley/` |\n| API Key | Not required |\n| Script Path | `scripts/foley.py` |\n\n## Prerequisites\n\n```bash\npython3 --version   # Python 3.x\ncurl --version      # curl for API submission\nffmpeg -version     # optional, for audio format conversion\n```\n\n## Modes\n\n| Mode | Command | Input | Output | Description |\n|------|---------|-------|--------|-------------|\n| **V2A** | `v2a video.mp4` | Video file | .mp4 + .flac | Generate audio matching the video content |\n| **TC-V2A** | `v2a video.mp4 --prompt \"text\"` | Video + text | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| **AC-V2A** | `v2a video.mp4 --ref-audio ref.wav` | Video + reference audio | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n| **T2A** | `t2a \"prompt\"` | Text description | .flac | Generate audio from text descriptions |\n\n\n## Usage\n\n### 1. Text-to-Audio (T2A, default 8s)\n\n```bash\npython3 scripts/foley.py t2a \"dog barking loudly in a park\"\n```\n\n### 2. Video-to-Audio (V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4\n```\n\n### 3. Text-Controlled Video-to-Audio (TC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --prompt \"footsteps on gravel with birds chirping\"\n```\n\n### 4. Audio-Controlled Video-to-Audio (AC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --ref-audio reference.wav\n```\n\n### 5. Specify duration\n\n```bash\npython3 scripts/foley.py t2a \"A mountain stream murmurs, its gentle current lapping against the pebbles.\" --duration 15\n```\n\n### 6. Generate multiple candidates\n\n```bash\npython3 scripts/foley.py t2a \"cat purring softly\" --count 3\n```\n\n### 7. Fixed seed (reproducible results)\n\n```bash\npython3 scripts/foley.py t2a \"rain on a tin roof\" --seed 42\n```\n\n### 8. List available models\n\n```bash\npython3 scripts/foley.py models\n```\n\n## Parameters\n\n### T2A (Text-to-Audio)\n\n| Parameter | Description | Default | Example |\n|-----------|-------------|---------|---------|\n| `prompt` | Audio description text (required) | — | `\"dog barking in park\"` |\n| `--model` | Model ID | `ControlFoley` | `--model ControlFoley` |\n| `--duration` | Audio length in seconds (max 30) | `8` | `--duration 15` |\n| `--negative` | Negative prompt to exclude unwanted sounds | — | `--negative \"noise, human voice\"` |\n| `--cfg` | CFG strength — higher = stricter prompt adherence | `4.5` | `--cfg 6.0` |\n| `--count` | Number of variants to generate (1–5) | `1` | `--count 3` |\n| `--seed` | Fixed random seed for reproducibility | — | `--seed 42` |\n| `-o/--outdir` | Output directory | `./output` | `-o ./my_audio` |\n\n### V2A (Video-to-Audio)\n\n| Parameter | Description | Default | Example |\n|-----------|-------------|---------|---------|\n| `video` | Input video path (required) | — | `input.mp4` |\n| `--model` | Model ID | `ControlFoley` | `--model ControlFoley` |\n| `--prompt` | Text prompt to guide audio generation (TC-V2A) | — | `--prompt \"keyboard tapping\"` |\n| `--negative` | Negative prompt to exclude unwanted sounds | — | `--negative \"music, noise\"` |\n| `--ref-audio` | Reference audio file for timbre control (AC-V2A) | — | `--ref-audio reference.wav` |\n| `--cfg` | CFG strength | `4.5` | `--cfg 7.0` |\n| `--count` | Number of variants to generate (1–5) | `1` | `--count 2` |\n| `--seed` | Fixed random seed (not forwarded to API currently) | — | `--seed 42` |\n| `-o/--outdir` | Output directory | `./output` | `-o ./results` |\n\n## Prompt Tips\n\n- **Be specific**: `\"cat footsteps on wooden floor\"` beats `\"cat sound\"`\n- **Use negative prompts**: `--negative \"human voice, music, noise\"` to filter unwanted audio\n- **CFG tuning**: high CFG (6.0–7.5) for precise control, low CFG (3.0–4.5) for creative freedom\n\n## Output & Post-Processing\n\n- **Audio**: `.flac` (44100 Hz, lossless)\n- **Video**: `.mp4` (original video + generated audio track)\n- Results saved to `--outdir`, paths printed to stdout\n\n**Convert to MP3 for sharing:**\n\n```bash\nffmpeg -i output.flac -codec:a libmp3lame -qscale:a 2 output.mp3\n```\n\n## Error Handling\n\n| Issue | Cause | Fix |\n|-------|-------|-----|\n| Internal URL inaccessible | Result URL uses `.xiaomi.srv` internal domain | Script auto-falls back to `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` |\n| Queue busy | Task is waiting | Script auto-polls up to ~5 min; check load via `curl $API_BASE/health` |\n| Model unavailable | Model not enabled | Run `foley.py models` to see available models |\n| Task timeout | Service overloaded | Resubmit the task |\n\n## API Reference\n\nSee ./references/api-reference.md for full endpoint documentation.\n\n## ⚠️ Privacy & Security\n\n- **Service Operator**: Cloud processing is operated by the Xiaomi LLM Plus Team at `https://llmplus.ai.xiaomi.com`\n- **Data Upload**: V2A/TC-V2A/AC-V2A modes upload the full video file to the remote service for processing. Do not upload videos containing sensitive personal or identifiable information\n- **Data Processing**: Uploaded videos and audio are used solely for audio generation. Results are returned via URL. Refer to the Xiaomi LLM Plus Team's terms of service for data retention and access control policies\n- **No API Key Required**: The service requires no authentication — please use it responsibly to avoid unnecessary load\n- **Recommendation**: Before first use, validate with a small, non-sensitive test clip\n\nFile v1.0.2:_meta.json\n\n{\n  \"ownerId\": \"kn7ftwffrk08e278sse17ktw39858t8n\",\n  \"slug\": \"controlfoley-audio-generator\",\n  \"version\": \"1.0.2\",\n  \"publishedAt\": 1776760299623\n}\n\nFile v1.0.2:references/api-reference.md\n\n# ControlFoley Audio Generator API Reference\n\nBase URL: `https://llmplus.ai.xiaomi.com`  \nAuth: None required\n\n**Known API quirks (from real testing):**\n- Submit response may return either `task_id` or `taskId` (camelCase); code handles both\n- Status success may return either `processed_urls` or `urls`; code checks both\n- Result URLs may use internal `.xiaomi.srv` domains; use download endpoint as fallback\n\n## Endpoints\n\n### POST `/api/v1/v2a/submit` — Submit Task\n\nContent-Type: `multipart/form-data`\n\n#### T2A (Text-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| prompt | string | Yes | — | Audio description (e.g. `dog barking in park`) |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| duration | float | Optional | `8.0` | Audio length in seconds (max 30) |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n#### V2A (Video-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| video | file | Yes | — | Video file (.mp4/.webm/.mov) |\n| prompt | string | Optional | — | Text prompt to guide audio generation; omitted if empty |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| ref_audio | file | Optional | — | Reference audio for AC-V2A task; omitted if file not found |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n> Note: `seed` is accepted in the V2A function signature but not currently forwarded to the API.\n\n**Response (202):**\n```json\n{\"task_id\": \"uuid\", \"status\": \"pending\", \"queue_pos\": 2, \"message\": \"Task submitted successfully\"}\n```\n> Note: field may appear as `taskId` (camelCase) — handle both.\n\n### GET `/api/v1/v2a/status/{task_id}` — Poll Status\n\n**Pending/Processing:**\n```json\n{\"status\": \"pending|processing\", \"queue_pos\": 3}\n```\n\n**Success:**\n```json\n{\n  \"status\": \"success\", \"done\": true,\n  \"processed_urls\": [\"http://.../uuid.flac\", \"http://.../uuid.mp4\"]\n}\n```\n> Note: result URLs field may appear as `urls` — handle both `processed_urls` and `urls`.\n\n**Failed:**\n```json\n{\"status\": \"failed\", \"done\": true, \"error\": \"reason\"}\n```\n\n### GET `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` — Download Result\n\nFallback download endpoint for internal URLs that are not directly accessible.  \nReturns binary file stream (.flac at 44100 Hz or .mp4).\n\n### GET `/api/v1/v2a/models` — List Models\n\n```json\n{\"models\": [{\"name\": \"ControlFoley\", \"enabled\": true}]}\n```\n\n### GET `/health` — Health Check\n\n```json\n{\"status\": \"ok\", \"queue_size\": 3}\n```\n\n## Models\n\n| Model | Use Case |\n|-------|----------|\n| ControlFoley | General T2A and V2A (default) |\n\n## Modes\n\n| Mode | Input | Output | Description |\n|------|-------|--------|-------------|\n| T2A | prompt | .flac | Generate audio from text descriptions |\n| V2A | video file | .mp4 + .flac | Generate audio that matches video content |\n| TC-V2A | prompt + video file | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| AC-V2A | reference_audio + video file | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n\nArchive v1.0.1: 4 files, 7369 bytes\n\nFiles: references/api-reference.md (3634b), scripts/foley.py (8342b), SKILL.md (7094b), _meta.json (147b)\n\nFile v1.0.1:SKILL.md\n\n---\nname: controlfoley-audio-generator\ndescription: >\n  A multi-functional audio generation tool powered by the ControlFoley model, integrating video sound effect (SFX) generation, video background music composition, text-to-audio and other functions to realize diversified creative audio generation. This tool supports four modes: Video-to-Audio (V2A), Text-Controlled Video-to-Audio (TC-V2A), Audio-Controlled Video-to-Audio (AC-V2A), and Text-to-Audio (T2A).\n  多功能音频生成工具，依托ControlFoley模型，集视频音效（SFX）生成、视频配乐、文本生成音频等功能于一体，实现多元化创意音频生成。该工具支持视频生成音频（V2A）、文本控制视频配音（TC-V2A）、参考音频控制视频配音（AC-V2A）和文本生成音频（T2A）四种模式。\n---\n\n# ControlFoley Audio Generator\n\nA multi-functional audio generation tool powered by the ControlFoley model, integrating video sound effect (SFX) generation, video background music composition, text-to-audio and other functions to realize diversified creative audio generation. \n\nThis tool supports four modes: Video-to-Audio (V2A), Text-Controlled Video-to-Audio (TC-V2A), Audio-Controlled Video-to-Audio (AC-V2A), and Text-to-Audio (T2A).\n\n## Basic Info\n\n| Field | Value |\n|---|---|\n| Service Operator | Xiaomi LLM Plus Team |\n| API Endpoint | `https://llmplus.ai.xiaomi.com` |\n| Open Source Repo | `https://github.com/xiaomi-research/controlfoley` |\n| Project Page | `https://yjx-research.github.io/ControlFoley_web_page/` |\n| Online Demo | `https://yjx-research.github.io/ControlFoley_web_page/#try-gen` |\n| Model Weights | `https://huggingface.co/YJX-Xiaomi/ControlFoley/` |\n| API Key | Not required |\n| Script Path | `scripts/foley.py` |\n\n## Prerequisites\n\n```bash\npython3 --version   # Python 3.x\ncurl --version      # curl for API submission\nffmpeg -version     # optional, for audio format conversion\n```\n\n## Modes\n\n| Mode | Command | Input | Output | Description |\n|------|---------|-------|--------|-------------|\n| **V2A** | `v2a video.mp4` | Video file | .mp4 + .flac | Generate audio matching the video content |\n| **TC-V2A** | `v2a video.mp4 --prompt \"text\"` | Video + text | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| **AC-V2A** | `v2a video.mp4 --ref-audio ref.wav` | Video + reference audio | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n| **T2A** | `t2a \"prompt\"` | Text description | .flac audio | Generate audio from text descriptions |\n\n\n## Usage\n\n### 1. Text-to-Audio (T2A, default 8s)\n\n```bash\npython3 scripts/foley.py t2a \"dog barking loudly in a park\"\n```\n\n### 2. Video-to-Audio (V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4\n```\n\n### 3. Text-Controlled Video-to-Audio (TC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --prompt \"footsteps on gravel with birds chirping\"\n```\n\n### 4. Audio-Controlled Video-to-Audio (AC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --ref-audio reference.wav\n```\n\n### 5. Specify duration\n\n```bash\npython3 scripts/foley.py t2a \"A mountain stream murmurs, its gentle current lapping against the pebbles.\" --duration 15\n```\n\n### 6. Generate multiple candidates\n\n```bash\npython3 scripts/foley.py t2a \"cat purring softly\" --count 3\n```\n\n### 7. Fixed seed (reproducible results)\n\n```bash\npython3 scripts/foley.py t2a \"rain on a tin roof\" --seed 42\n```\n\n### 8. List available models\n\n```bash\npython3 scripts/foley.py models\n```\n\n## Parameters\n\n### T2A (Text-to-Audio)\n\n| Parameter | Description | Default | Example |\n|-----------|-------------|---------|---------|\n| `prompt` | Audio description text (required) | — | `\"dog barking in park\"` |\n| `--model` | Model ID | `ControlFoley` | `--model ControlFoley` |\n| `--duration` | Audio length in seconds (max 30) | `8` | `--duration 15` |\n| `--negative` | Negative prompt to exclude unwanted sounds | — | `--negative \"noise, human voice\"` |\n| `--cfg` | CFG strength — higher = stricter prompt adherence | `4.5` | `--cfg 6.0` |\n| `--count` | Number of variants to generate (1–5) | `1` | `--count 3` |\n| `--seed` | Fixed random seed for reproducibility | — | `--seed 42` |\n| `-o/--outdir` | Output directory | `./output` | `-o ./my_audio` |\n\n### V2A (Video-to-Audio)\n\n| Parameter | Description | Default | Example |\n|-----------|-------------|---------|---------|\n| `video` | Input video path (required) | — | `input.mp4` |\n| `--model` | Model ID | `ControlFoley` | `--model ControlFoley` |\n| `--prompt` | Text prompt to guide audio generation (TC-V2A) | — | `--prompt \"keyboard tapping\"` |\n| `--negative` | Negative prompt to exclude unwanted sounds | — | `--negative \"music, noise\"` |\n| `--ref-audio` | Reference audio file for timbre control (AC-V2A) | — | `--ref-audio reference.wav` |\n| `--cfg` | CFG strength | `4.5` | `--cfg 7.0` |\n| `--count` | Number of variants to generate (1–5) | `1` | `--count 2` |\n| `--seed` | Fixed random seed (not forwarded to API currently) | — | `--seed 42` |\n| `-o/--outdir` | Output directory | `./output` | `-o ./results` |\n\n## Prompt Tips\n\n- **Be specific**: `\"cat footsteps on wooden floor\"` beats `\"cat sound\"`\n- **Use negative prompts**: `--negative \"human voice, music, noise\"` to filter unwanted audio\n- **CFG tuning**: high CFG (6.0–7.5) for precise control, low CFG (3.0–4.5) for creative freedom\n\n## Output & Post-Processing\n\n- **Audio**: `.flac` (44100 Hz, lossless)\n- **Video**: `.mp4` (original video + generated audio track)\n- Results saved to `--outdir`, paths printed to stdout\n\n**Convert to MP3 for sharing:**\n\n```bash\nffmpeg -i output.flac -codec:a libmp3lame -qscale:a 2 output.mp3\n```\n\n## Error Handling\n\n| Issue | Cause | Fix |\n|-------|-------|-----|\n| Internal URL inaccessible | Result URL uses `.xiaomi.srv` internal domain | Script auto-falls back to `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` |\n| Queue busy | Task is waiting | Script auto-polls up to ~5 min; check load via `curl $API_BASE/health` |\n| Model unavailable | Model not enabled | Run `foley.py models` to see available models |\n| Task timeout | Service overloaded | Resubmit the task |\n\n## API Reference\n\nSee [./references/api-reference.md](./references/api-reference.md) for full endpoint documentation.\n\n## ⚠️ Privacy & Security\n\n- **Service Operator**: Cloud processing is operated by the Xiaomi LLM Plus Team at `https://llmplus.ai.xiaomi.com`\n- **Data Upload**: V2A/TC-V2A/AC-V2A modes upload the full video file to the remote service for processing. Do not upload videos containing sensitive personal or identifiable information\n- **Data Processing**: Uploaded videos and audio are used solely for audio generation. Results are returned via URL. Refer to the Xiaomi LLM Plus Team's terms of service for data retention and access control policies\n- **No API Key Required**: The service requires no authentication — please use it responsibly to avoid unnecessary load\n- **Recommendation**: Before first use, validate with a small, non-sensitive test clip\n\nFile v1.0.1:_meta.json\n\n{\n  \"ownerId\": \"kn7ftwffrk08e278sse17ktw39858t8n\",\n  \"slug\": \"controlfoley-audio-generator\",\n  \"version\": \"1.0.1\",\n  \"publishedAt\": 1776759625865\n}\n\nFile v1.0.1:references/api-reference.md\n\n# ControlFoley Audio Generator API Reference\n\nBase URL: `https://llmplus.ai.xiaomi.com`  \nAuth: None required\n\n**Known API quirks (from real testing):**\n- Submit response may return either `task_id` or `taskId` (camelCase); code handles both\n- Status success may return either `processed_urls` or `urls`; code checks both\n- Result URLs may use internal `.xiaomi.srv` domains; use download endpoint as fallback\n\n## Endpoints\n\n### POST `/api/v1/v2a/submit` — Submit Task\n\nContent-Type: `multipart/form-data`\n\n#### T2A (Text-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| prompt | string | Yes | — | Audio description (e.g. `dog barking in park`) |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| duration | float | Optional | `8.0` | Audio length in seconds (max 30) |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n#### V2A (Video-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| video | file | Yes | — | Video file (.mp4/.webm/.mov) |\n| prompt | string | Optional | — | Text prompt to guide audio generation; omitted if empty |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| ref_audio | file | Optional | — | Reference audio for AC-V2A task; omitted if file not found |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n> Note: `seed` is accepted in the V2A function signature but not currently forwarded to the API.\n\n**Response (202):**\n```json\n{\"task_id\": \"uuid\", \"status\": \"pending\", \"queue_pos\": 2, \"message\": \"Task submitted successfully\"}\n```\n> Note: field may appear as `taskId` (camelCase) — handle both.\n\n### GET `/api/v1/v2a/status/{task_id}` — Poll Status\n\n**Pending/Processing:**\n```json\n{\"status\": \"pending|processing\", \"queue_pos\": 3}\n```\n\n**Success:**\n```json\n{\n  \"status\": \"success\", \"done\": true,\n  \"processed_urls\": [\"http://.../uuid.flac\", \"http://.../uuid.mp4\"]\n}\n```\n> Note: result URLs field may appear as `urls` — handle both `processed_urls` and `urls`.\n\n**Failed:**\n```json\n{\"status\": \"failed\", \"done\": true, \"error\": \"reason\"}\n```\n\n### GET `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` — Download Result\n\nFallback download endpoint for internal URLs that are not directly accessible.  \nReturns binary file stream (.flac at 44100 Hz or .mp4).\n\n### GET `/api/v1/v2a/models` — List Models\n\n```json\n{\"models\": [{\"name\": \"ControlFoley\", \"enabled\": true}]}\n```\n\n### GET `/health` — Health Check\n\n```json\n{\"status\": \"ok\", \"queue_size\": 3}\n```\n\n## Models\n\n| Model | Use Case |\n|-------|----------|\n| ControlFoley | General T2A and V2A (default) |\n\n## Modes\n\n| Mode | Input | Output | Description |\n|------|-------|--------|-------------|\n| T2A | prompt | .flac | Generate audio from text descriptions |\n| V2A | video file | .mp4 + .flac | Generate audio that matches video content |\n| TC-V2A | prompt + video file | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| AC-V2A | reference_audio + video file | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n\nArchive v1.0.0: 6 files, 12118 bytes\n\nFiles: README_zh.md (6112b), README.md (6298b), references/api-reference.md (3634b), scripts/foley.py (8342b), SKILL.md (4842b), _meta.json (147b)\n\nFile v1.0.0:SKILL.md\n\n---\nname: controlfoley-audio-generator\ndescription: >\n  AI audio and foley generation via ControlFoley Audio Generator API.\n  Two core modes: Video-to-Audio (V2A) generates matching audio/foley for video clips, \n  and Text-to-Audio (T2A) generates audio from text descriptions.\n  Also supports text-controlled video dubbing (TC-V2A) and audio style transfer (AC-V2A).\n  Use cases: game sound effects, pet video dubbing, short drama audio, ad foley,\n  environmental sounds, nature sounds, animal sounds, mechanical sounds.\n  Use when user wants to generate sound effects, foley, audio from text descriptions,\n  add audio to video, dub video with sound effects, or create environmental audio.\n  NOT for: speech/TTS or audio denoising.\n  Triggers: \"sound effect generation\", \"generate sound effect\", \"foley\", \"sound effect\", \"text to audio\",\n  \"video dubbing\", \"video sound effect\", \"ambient sound\", \"animal sound\", \"text to audio\", \"video to audio\",\n  \"generate sound\", \"add audio to video\", \"audio effect\", \"foley sound\", \"add sound effects\", \"SFX\".\n---\n\n# ControlFoley Audio Generator\n\ndiverse audio generation: video-to-audio and text-to-audio via ControlFoley Audio Generator API.\n\n## Quick Start\n\n```bash\n# Text → Audio (default 8s)\npython3 scripts/foley.py t2a \"dog barking loudly in a park\"\n\n# Longer duration (15s)\npython3 scripts/foley.py t2a \"A mountain stream murmurs, its gentle current lapping against the pebbles.\" --duration 15\n\n# Video → Audio\npython3 scripts/foley.py v2a input.mp4\n\n# Video + text guidance → Audio (TV2A)\npython3 scripts/foley.py v2a input.mp4 --prompt \"footsteps on gravel with birds chirping\"\n\n# Video + text control → Audio (TC-V2A)\npython3 scripts/foley.py v2a input.mp4 --prompt \"keyboard tapping sound\"\n\n# Video + audio control → Audio (AC-V2A)\npython3 scripts/foley.py v2a input.mp4 --ref-audio reference.wav\n\n# List available models\npython3 scripts/foley.py models\n```\n\n## Modes\n\n| Mode | Command | Input | Output |\n|------|---------|-------|--------|\n| **T2A** | `t2a \"prompt\"` | Text description | .flac audio |\n| **V2A** | `v2a video.mp4` | Video file | .mp4 + .flac |\n| **TC-V2A** | `v2a video.mp4 --prompt \"text\"` | Video + text | .mp4 + .flac |\n| **AC-V2A** | `v2a video.mp4 --ref-audio ref.wav` | Video + ref audio | .mp4 + .flac |\n\n## Key Parameters\n\n### T2A (Text-to-Audio)\n\n- `prompt`: Text prompt to guide audio generation\n- `--model`: Model ID (default `ControlFoley`)\n- `--negative`: Negative prompt to exclude unwanted sounds (e.g. \"noise, human voice\")\n- `--duration`: Audio length in seconds (default 8, max 30)\n- `--count`: Generate multiple variants (1–5)\n- `--seed`: Fixed seed for reproducible results\n- `--cfg`: CFG strength (default 4.5; higher = stricter prompt adherence)\n- `-o/--outdir`: Output directory (default `./output`)\n\n### V2A (Video-to-Audio)\n\n- `video`: Path of the video file to be dubbed\n- `--model`: Model ID (default `ControlFoley`)\n- `--prompt`: Text prompt to guide audio generation\n- `--negative`: Negative prompt to exclude unwanted sounds (e.g. \"noise, human voice\")\n- `--ref-audio`: Reference audio for timbre control (AC-V2A mode)\n- `--count`: Generate multiple variants (1–5)\n- `--seed`: Fixed seed (accepted but not forwarded to API)\n- `--cfg`: CFG strength (default 4.5; higher = stricter prompt adherence)\n- `-o/--outdir`: Output directory (default `./output`)\n\n## Prompt Tips\n\n- Be specific: \"cat footsteps on wooden floor\" > \"cat sound\"\n- Use negative prompts: `--negative \"human voice, music, noise\"`\n- Higher CFG (6.0–7.5) for precise control, lower (3.0–4.5) for creative freedom\n\n## Output & Post-Processing\n\n- Audio: .flac (44100 Hz, lossless)\n- Video: .mp4 (original + generated audio track)\n- Results saved to `--outdir`, paths printed to stdout\n\n**Convert for sharing** (flac → mp3):\n```bash\nffmpeg -i output.flac -codec:a libmp3lame -qscale:a 2 output.mp3\n```\n\n## Error Handling\n\n- **Internal URLs**: Result URLs may use internal domains (`.xiaomi.srv`); script auto-falls back to API download endpoint `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}`\n- **Queue busy**: Auto-polls up to ~5 min; check load via `curl $API_BASE/health`\n- **Model unavailable**: Run `foley.py models` to see which are enabled\n- **Timeout**: Retry the full flow once\n\n## Environment\n\n- `FOLEY_API_BASE`: Override API base URL (default: `https://llmplus.ai.xiaomi.com`)\n- No authentication required\n\n## Known Quirks (from real testing)\n\n1. Submit response may return `task_id` or `taskId` (camelCase) — script handles both\n2. Status success may return `processed_urls` or `urls` — script checks both\n3. Result URLs often point to internal `.xiaomi.srv` hosts — script falls back to public API download endpoint automatically\n\n## API Reference\n\nSee [references/api-reference.md](references/api-reference.md) for full endpoint docs.\n\nFile v1.0.0:README.md\n\n[中文阅读](./README_zh.md)\n\n# ControlFoley Audio Generator\n\nA multi-functional audio generation tool powered by the ControlFoley model, integrating video sound effect (SFX) generation, video background music composition, text-to-audio and other functions to realize diversified creative audio generation.\n\nThis tool supports four modes: Video-to-Audio (V2A), Text-Controlled Video-to-Audio (TC-V2A), Audio-Controlled Video-to-Audio (AC-V2A), and Text-to-Audio (T2A).\n\n## Basic Info\n\n| Field | Value |\n|---|---|\n| Service Operator | Xiaomi LLM Plus Team |\n| API Endpoint | `https://llmplus.ai.xiaomi.com` |\n| Open Source Repo | `https://github.com/xiaomi-research/controlfoley` |\n| Project Page | `https://yjx-research.github.io/ControlFoley_web_page/` |\n| Online Demo | `https://yjx-research.github.io/ControlFoley_web_page/#try-gen` |\n| Model Weights | `https://huggingface.co/YJX-Xiaomi/ControlFoley/` |\n| API Key | Not required |\n| Script Path | `scripts/foley.py` |\n\n## Prerequisites\n\n```bash\npython3 --version   # Python 3.x\ncurl --version      # curl for API submission\nffmpeg -version     # optional, for audio format conversion\n```\n\n## Modes\n\n| Mode | Command | Input | Output | Description |\n|------|---------|-------|--------|-------------|\n| **V2A** | `v2a video.mp4` | Video file | .mp4 + .flac | Generate audio matching the video content |\n| **TC-V2A** | `v2a video.mp4 --prompt \"text\"` | Video + text | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| **AC-V2A** | `v2a video.mp4 --ref-audio ref.wav` | Video + reference audio | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n| **T2A** | `t2a \"prompt\"` | Text description | .flac audio | Generate audio from text descriptions |\n\n## Usage\n\n### 1. Text-to-Audio (T2A, default 8s)\n\n```bash\npython3 scripts/foley.py t2a \"dog barking loudly in a park\"\n```\n\n### 2. Video-to-Audio (V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4\n```\n\n### 3. Text-Controlled Video-to-Audio (TC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --prompt \"footsteps on gravel with birds chirping\"\n```\n\n### 4. Audio-Controlled Video-to-Audio (AC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --ref-audio reference.wav\n```\n\n### 5. Specify duration\n\n```bash\npython3 scripts/foley.py t2a \"A mountain stream murmurs, its gentle current lapping against the pebbles.\" --duration 15\n```\n\n### 6. Generate multiple candidates\n\n```bash\npython3 scripts/foley.py t2a \"cat purring softly\" --count 3\n```\n\n### 7. Fixed seed (reproducible results)\n\n```bash\npython3 scripts/foley.py t2a \"rain on a tin roof\" --seed 42\n```\n\n### 8. List available models\n\n```bash\npython3 scripts/foley.py models\n```\n\n## Parameters\n\n### T2A (Text-to-Audio)\n\n| Parameter | Description | Default | Example |\n|-----------|-------------|---------|---------|\n| `prompt` | Audio description text (required) | — | `\"dog barking in park\"` |\n| `--model` | Model ID | `ControlFoley` | `--model ControlFoley` |\n| `--duration` | Audio length in seconds (max 30) | `8` | `--duration 15` |\n| `--negative` | Negative prompt to exclude unwanted sounds | — | `--negative \"noise, human voice\"` |\n| `--cfg` | CFG strength — higher = stricter prompt adherence | `4.5` | `--cfg 6.0` |\n| `--count` | Number of variants to generate (1–5) | `1` | `--count 3` |\n| `--seed` | Fixed random seed for reproducibility | — | `--seed 42` |\n| `-o/--outdir` | Output directory | `./output` | `-o ./my_audio` |\n\n### V2A (Video-to-Audio)\n\n| Parameter | Description | Default | Example |\n|-----------|-------------|---------|---------|\n| `video` | Input video path (required) | — | `input.mp4` |\n| `--model` | Model ID | `ControlFoley` | `--model ControlFoley` |\n| `--prompt` | Text prompt to guide audio generation (TC-V2A) | — | `--prompt \"keyboard tapping\"` |\n| `--negative` | Negative prompt to exclude unwanted sounds | — | `--negative \"music, noise\"` |\n| `--ref-audio` | Reference audio file for timbre control (AC-V2A) | — | `--ref-audio reference.wav` |\n| `--cfg` | CFG strength | `4.5` | `--cfg 7.0` |\n| `--count` | Number of variants to generate (1–5) | `1` | `--count 2` |\n| `--seed` | Fixed random seed (not forwarded to API currently) | — | `--seed 42` |\n| `-o/--outdir` | Output directory | `./output` | `-o ./results` |\n\n## Prompt Tips\n\n- **Be specific**: `\"cat footsteps on wooden floor\"` beats `\"cat sound\"`\n- **Use negative prompts**: `--negative \"human voice, music, noise\"` to filter unwanted audio\n- **CFG tuning**: high CFG (6.0–7.5) for precise control, low CFG (3.0–4.5) for creative freedom\n\n## Output & Post-Processing\n\n- **Audio**: `.flac` (44100 Hz, lossless)\n- **Video**: `.mp4` (original video + generated audio track)\n- Results saved to `--outdir`, paths printed to stdout\n\n**Convert to MP3 for sharing:**\n\n```bash\nffmpeg -i output.flac -codec:a libmp3lame -qscale:a 2 output.mp3\n```\n\n## Error Handling\n\n| Issue | Cause | Fix |\n|-------|-------|-----|\n| Internal URL inaccessible | Result URL uses `.xiaomi.srv` internal domain | Script auto-falls back to `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` |\n| Queue busy | Task is waiting | Script auto-polls up to ~5 min; check load via `curl $API_BASE/health` |\n| Model unavailable | Model not enabled | Run `foley.py models` to see available models |\n| Task timeout | Service overloaded | Resubmit the task |\n\n## API Reference\n\nSee [references/api-reference.md](references/api-reference.md) for full endpoint documentation.\n\n## ⚠️ Privacy & Security\n\n- **Service Operator**: Cloud processing is operated by the Xiaomi LLM Plus Team at `https://llmplus.ai.xiaomi.com`\n- **Data Upload**: V2A/TC-V2A/AC-V2A modes upload the full video file to the remote service for processing. Do not upload videos containing sensitive personal or identifiable information\n- **Data Processing**: Uploaded videos and audio are used solely for audio generation. Results are returned via URL. Refer to the Xiaomi LLM Plus Team's terms of service for data retention and access control policies\n- **No API Key Required**: The service requires no authentication — please use it responsibly to avoid unnecessary load\n- **Recommendation**: Before first use, validate with a small, non-sensitive test clip\n\nFile v1.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn7ftwffrk08e278sse17ktw39858t8n\",\n  \"slug\": \"controlfoley-audio-generator\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1776758045934\n}\n\nFile v1.0.0:references/api-reference.md\n\n# ControlFoley Audio Generator API Reference\n\nBase URL: `https://llmplus.ai.xiaomi.com`  \nAuth: None required\n\n**Known API quirks (from real testing):**\n- Submit response may return either `task_id` or `taskId` (camelCase); code handles both\n- Status success may return either `processed_urls` or `urls`; code checks both\n- Result URLs may use internal `.xiaomi.srv` domains; use download endpoint as fallback\n\n## Endpoints\n\n### POST `/api/v1/v2a/submit` — Submit Task\n\nContent-Type: `multipart/form-data`\n\n#### T2A (Text-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| prompt | string | Yes | — | Audio description (e.g. `dog barking in park`) |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| duration | float | Optional | `8.0` | Audio length in seconds (max 30) |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n#### V2A (Video-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| video | file | Yes | — | Video file (.mp4/.webm/.mov) |\n| prompt | string | Optional | — | Text prompt to guide audio generation; omitted if empty |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| ref_audio | file | Optional | — | Reference audio for AC-V2A task; omitted if file not found |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n> Note: `seed` is accepted in the V2A function signature but not currently forwarded to the API.\n\n**Response (202):**\n```json\n{\"task_id\": \"uuid\", \"status\": \"pending\", \"queue_pos\": 2, \"message\": \"Task submitted successfully\"}\n```\n> Note: field may appear as `taskId` (camelCase) — handle both.\n\n### GET `/api/v1/v2a/status/{task_id}` — Poll Status\n\n**Pending/Processing:**\n```json\n{\"status\": \"pending|processing\", \"queue_pos\": 3}\n```\n\n**Success:**\n```json\n{\n  \"status\": \"success\", \"done\": true,\n  \"processed_urls\": [\"http://.../uuid.flac\", \"http://.../uuid.mp4\"]\n}\n```\n> Note: result URLs field may appear as `urls` — handle both `processed_urls` and `urls`.\n\n**Failed:**\n```json\n{\"status\": \"failed\", \"done\": true, \"error\": \"reason\"}\n```\n\n### GET `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` — Download Result\n\nFallback download endpoint for internal URLs that are not directly accessible.  \nReturns binary file stream (.flac at 44100 Hz or .mp4).\n\n### GET `/api/v1/v2a/models` — List Models\n\n```json\n{\"models\": [{\"name\": \"ControlFoley\", \"enabled\": true}]}\n```\n\n### GET `/health` — Health Check\n\n```json\n{\"status\": \"ok\", \"queue_size\": 3}\n```\n\n## Models\n\n| Model | Use Case |\n|-------|----------|\n| ControlFoley | General T2A and V2A (default) |\n\n## Modes\n\n| Mode | Input | Output | Description |\n|------|-------|--------|-------------|\n| T2A | prompt | .flac | Generate audio from text descriptions |\n| V2A | video file | .mp4 + .flac | Generate audio that matches video content |\n| TC-V2A | prompt + video file | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| AC-V2A | reference_audio + video file | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n\nFile v1.0.0:README_zh.md\n\n[English](./README.md)\n\n# 音频生成器ControlFoley\n\n多功能音频生成工具，依托ControlFoley模型，集视频音效（SFX）生成、视频配乐、文本生成音频等功能于一体，实现多元化创意音频生成。\n\n该工具支持视频生成音频（V2A）、文本控制视频配音（TC-V2A）、参考音频控制视频配音（AC-V2A）和文本生成音频（T2A）四种模式。\n\n## 基本信息\n\n| 项 | 值 |\n|---|---|\n| 服务运营商 | 小米大模型Plus团队 |\n| 服务域名 | `https://llmplus.ai.xiaomi.com` |\n| 开源仓库 | `https://github.com/xiaomi-research/controlfoley` |\n| 项目主页 | `https://yjx-research.github.io/ControlFoley_web_page/` |\n| 在线推理 | `https://yjx-research.github.io/ControlFoley_web_page/#try-gen` |\n| 模型权重 | `https://huggingface.co/YJX-Xiaomi/ControlFoley/` |\n| API Key | 无需认证 |\n| 脚本路径 | `scripts/foley.py` |\n\n## 前置依赖\n\n```bash\npython3 --version   # Python 3.x\ncurl --version      # curl 用于 API 提交\nffmpeg -version     # ffmpeg 可选，用于音频格式转换\n```\n\n## 模式说明\n\n| 模式 | 命令 | 输入 | 输出 | 说明 |\n|------|------|------|------|------|\n| **V2A** | `v2a video.mp4` | 视频文件 | .mp4 + .flac | 生成与视频内容相匹配的音频 |\n| **TC-V2A** | `v2a video.mp4 --prompt \"text\"` | 视频 + 文本 | .mp4 + .flac | 根据文本提示生成对齐的音频，同时保持与视频同步 |\n| **AC-V2A** | `v2a video.mp4 --ref-audio ref.wav` | 视频 + 参考音频 | .mp4 + .flac | 生成与参考音频音色一致、视频时间同步的音频 |\n| **T2A** | `t2a \"text\"` | 文本 | .flac | 根据文本描述生成音频 |\n\n## 命令行用法\n\n### 1. 文本生成音频（T2A，默认 8 秒）\n\n```bash\npython3 scripts/foley.py t2a \"dog barking loudly in a park\"\n```\n\n### 2. 视频生成音频（V2A）\n\n```bash\npython3 scripts/foley.py v2a input.mp4\n```\n\n### 3. 文本控制下视频生成音频（TC-V2A）\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --prompt \"footsteps on gravel with birds chirping\"\n```\n\n### 4. 参考音频控制下视频生成音频（AC-V2A）\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --ref-audio reference.wav\n```\n\n### 5. 指定时长\n\n```bash\npython3 scripts/foley.py t2a \"A mountain stream murmurs, its gentle current lapping against the pebbles.\" --duration 15\n```\n\n### 6. 生成多个候选结果\n\n```bash\npython3 scripts/foley.py t2a \"cat purring softly\" --count 3\n```\n\n### 7. 固定随机种子（可复现结果）\n\n```bash\npython3 scripts/foley.py t2a \"rain on a tin roof\" --seed 42\n```\n\n### 8. 查看可用模型\n\n```bash\npython3 scripts/foley.py models\n```\n\n## 参数说明\n\n### T2A（文本转音频）\n\n| 参数 | 说明 | 默认值 | 示例 |\n|------|------|--------|------|\n| `prompt` | 音频描述文本（必填） | — | `\"dog barking in park\"` |\n| `--model` | 模型 ID | `ControlFoley` | `--model ControlFoley` |\n| `--duration` | 音频时长（秒，最大 30） | `8` | `--duration 15` |\n| `--negative` | 负向提示，排除不想要的声音 | — | `--negative \"noise, human voice\"` |\n| `--cfg` | CFG 强度，越高越严格遵循提示词 | `4.5` | `--cfg 6.0` |\n| `--count` | 生成变体数量（1–5） | `1` | `--count 3` |\n| `--seed` | 固定随机种子，保证可复现 | — | `--seed 42` |\n| `-o/--outdir` | 输出目录 | `./output` | `-o ./my_audio` |\n\n### V2A（视频转音频）\n\n| 参数 | 说明 | 默认值 | 示例 |\n|------|------|--------|------|\n| `video` | 输入视频路径（必填） | — | `input.mp4` |\n| `--model` | 模型 ID | `ControlFoley` | `--model ControlFoley` |\n| `--prompt` | 文本提示引导音频生成（TC-V2A） | — | `--prompt \"keyboard tapping\"` |\n| `--negative` | 负向提示，排除不想要的声音 | — | `--negative \"music, noise\"` |\n| `--ref-audio` | 参考音频文件（AC-V2A） | — | `--ref-audio reference.wav` |\n| `--cfg` | CFG 强度 | `4.5` | `--cfg 7.0` |\n| `--count` | 生成变体数量（1–5） | `1` | `--count 2` |\n| `--seed` | 固定随机种子（目前不转发至 API） | — | `--seed 42` |\n| `-o/--outdir` | 输出目录 | `./output` | `-o ./results` |\n\n## Prompt 技巧\n\n- **具体描述**：`\"cat footsteps on wooden floor\"` 优于 `\"cat sound\"`\n- **使用负向提示**：`--negative \"human voice, music, noise\"` 过滤杂音\n- **CFG 调优**：高 CFG（6.0–7.5）精确控制，低 CFG（3.0–4.5）创意自由度更高\n\n## 输出格式与后处理\n\n- **音频**：`.flac`（44100 Hz，无损）\n- **视频**：`.mp4`（原视频 + 生成音轨）\n- 结果保存至 `--outdir`，路径打印到 stdout\n\n**转换为 MP3 分享：**\n\n```bash\nffmpeg -i output.flac -codec:a libmp3lame -qscale:a 2 output.mp3\n```\n\n## 错误处理\n\n| 问题 | 原因 | 处理方式 |\n|------|------|----------|\n| 内部 URL 无法访问 | 结果 URL 使用 `.xiaomi.srv` 内部域名 | 脚本自动回退到 `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` |\n| 队列繁忙 | 任务等待中 | 脚本自动轮询最长约 5 分钟；可通过 `curl $API_BASE/health` 查看队列状态 |\n| 模型不可用 | 模型未启用 | 执行 `foley.py models` 查看可用模型 |\n| 任务超时 | 服务负载过高 | 重新提交任务 |\n\n## API 参考\n\n完整接口文档见 [references/api-reference.md](references/api-reference.md)。\n\n## ⚠️ 隐私与安全声明\n\n- **服务运营商**：本技能的云端处理服务由小米大模型Plus团队运营，服务端点为 `https://llmplus.ai.xiaomi.com`\n- **数据上传**：V2A/TC-V2A/AC-V2A 模式会将完整视频文件上传至上述远程服务进行处理。请勿上传包含敏感个人信息或可识别身份信息的视频内容\n- **数据处理**：上传的视频和音频仅用于音频生成处理，处理完成后结果通过 URL 返回。具体数据保留和访问控制政策请参阅小米大模型应用团队的服务条款\n- **无需 API Key**：当前服务无需认证，但请合理使用，避免对服务造成不必要的压力\n- **使用建议**：首次使用前，建议用小型、不涉及敏感内容的测试素材验证功能","readmeExcerpt":"Skill: ControlFoley Audio Generator Owner: yjx-research Summary: A multi-functional audio generation tool for SFX generation, video-to-audio and text-to-audio. 多功能音频生成工具，集成可控视频生成音频、文本生成音频等功能. Tags: latest:1.0.9 Version history: v1.0.9 | 2026-06-08T03:51:10.695Z | user - Removed the _meta.json file from the skill package. - No functional changes to user experience or features. - Internal metadata cleanup for this vers","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"curl --version      # curl for API submission"},{"language":"bash","snippet":"python3 --version   # Python 3.x\ncurl --version      # curl for API submission\nffmpeg -version     # optional, for audio format conversion"},{"language":"bash","snippet":"python3 scripts/foley.py t2a \"dog barking loudly in a park\""},{"language":"bash","snippet":"python3 scripts/foley.py v2a input.mp4"},{"language":"bash","snippet":"python3 scripts/foley.py v2a input.mp4 --prompt \"footsteps on gravel with birds chirping\""},{"language":"bash","snippet":"python3 scripts/foley.py v2a input.mp4 --ref-audio reference.wav"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: controlfoley-audio-generator\ndescription: >\n  A multi-functional audio generation tool for SFX generation, video-to-audio and text-to-audio.\n  多功能音频生成工具，集成可控视频生成音频、文本生成音频等功能.\n\n---\n\n# ControlFoley Audio Generator\n\nA multi-functional audio generation tool powered by the ControlFoley model, integrating video sound effect (SFX) generation, video background music composition, text-to-audio and other functions to realize diversified creative audio generation. \n\nThis tool supports four modes: Video-to-Audio (V2A), Text-Controlled Video-to-Audio (TC-V2A), Audio-Controlled Video-to-Audio (AC-V2A), and Text-to-Audio (T2A).\n\n## Basic Info\n\n| Field | Value |\n|---|---|\n| Service Operator | Xiaomi LLM Plus Team |\n| API Endpoint | `https://controlfoley.ai.xiaomi.com` |\n| Open Source Repo | `https://github.com/xiaomi-research/controlfoley` |\n| Project Page | `https://yjx-research.github.io/ControlFoley_web_page/` |\n| Online Demo | `https://yjx-research.github.io/ControlFoley_web_page/#try-gen` |\n| Model Weights | `https://huggingface.co/YJX-Xiaomi/ControlFoley/` |\n| API Key | Not required |\n| Script Path | `scripts/foley.py` |\n\n## Prerequisites\n\n```bash\npython3 --version   # Python 3.x\ncurl --version      # curl for API submission\nffmpeg -version     # optional, for audio format conversion\n```\n\n## Modes\n\n| Mode | Command | Input | Output | Description |\n|------|---------|-------|--------|-------------|\n| **V2A** | `v2a video.mp4` | Video file | .mp4 + .flac | Generate audio matching the video content |\n| **TC-V2A** | `v2a video.mp4 --prompt \"text\"` | Video + text | .mp4 + .flac | Generate audio aligned with text prompts while staying synchronized with the video |\n| **AC-V2A** | `v2a video.mp4 --ref-audio ref.wav` | Video + reference audio | .mp4 + .flac | Generate audio with timbre matching reference audio while staying synchronized with the video |\n| **T2A** | `t2a \"prompt\"` | Text description | .flac | Generate audio from text descriptions |\n\n\n## Usage (CLI version)\n\n### 1. Text-to-Audio (T2A, default 8s)\n\n```bash\npython3 scripts/foley.py t2a \"dog barking loudly in a park\"\n```\n\n### 2. Video-to-Audio (V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4\n```\n\n### 3. Text-Controlled Video-to-Audio (TC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --prompt \"footsteps on gravel with birds chirping\"\n```\n\n### 4. Audio-Controlled Video-to-Audio (AC-V2A)\n\n```bash\npython3 scripts/foley.py v2a input.mp4 --ref-audio reference.wav\n```\n\n### 5. Specify duration\n\n```bash\npython3 scripts/foley.py t2a \"A mountain stream murmurs, its gentle current lapping against the pebbles.\" --duration 15\n```\n\n### 6. Generate multiple candidates\n\n```bash\npython3 scripts/foley.py t2a \"cat purring softly\" --count 3\n```\n\n### 7. Fixed seed (reproducible results)\n\n```bash\npython3 scripts/foley.py t2a \"rain on a tin roof\" --seed 42\n```\n\n### 8. List available models\n\n```bash\npython3 scripts/foley.py models\n```\n\n## Usage (API version)\n\n### POST\n\n```bash\ncurl -X POST \"https://controlf"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7ftwffrk08e278sse17ktw39858t8n\",\n  \"slug\": \"controlfoley-audio-generator\",\n  \"version\": \"1.0.9\",\n  \"publishedAt\": 1780890670695\n}"},{"path":"references/api-reference.md","content":"# ControlFoley Audio Generator API Reference\n\nBase URL: `https://controlfoley.ai.xiaomi.com`  \nAuth: None required\n\n**Known API quirks (from real testing):**\n- Submit response may return either `task_id` or `taskId` (camelCase); code handles both\n- Status success may return either `processed_urls` or `urls`; code checks both\n- Result URLs may use internal `.xiaomi.srv` domains; use download endpoint as fallback\n\n## Endpoints\n\n### POST `/api/v1/v2a/submit` — Submit Task\n\nContent-Type: `multipart/form-data`\n\n#### T2A (Text-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| prompt | string | Yes | — | Audio description (e.g. `dog barking in park`) |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| duration | float | Optional | `8.0` | Audio length in seconds (max 30) |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n#### V2A (Video-to-Audio)\n\n| Param | Type | Required | Default | Description |\n|-------|------|----------|---------|-------------|\n| video | file | Yes | — | Video file (.mp4/.webm/.mov) |\n| prompt | string | Optional | — | Text prompt to guide audio generation; omitted if empty |\n| negative | string | Optional | — | What to avoid; omitted if empty |\n| ref_audio | file | Optional | — | Reference audio for AC-V2A task; omitted if file not found |\n| count | int | Optional | `1` | Number of results (1–5) |\n| seed | int | Optional | — | Fixed seed for reproducibility; omitted if not set |\n| cfg | float | Optional | `4.5` | CFG strength |\n| outdir | string | Optional | \"./output\" | Output directory |\n\n> Note: `seed` is accepted in the V2A function signature but not currently forwarded to the API.\n\n**Response (202):**\n```json\n{\"task_id\": \"uuid\", \"status\": \"pending\", \"queue_pos\": 2, \"message\": \"Task submitted successfully\"}\n```\n> Note: field may appear as `taskId` (camelCase) — handle both.\n\n### GET `/api/v1/v2a/status/{task_id}` — Poll Status\n\n**Pending/Processing:**\n```json\n{\"status\": \"pending|processing\", \"queue_pos\": 3}\n```\n\n**Success:**\n```json\n{\n  \"status\": \"success\", \"done\": true,\n  \"processed_urls\": [\"http://.../uuid.flac\", \"http://.../uuid.mp4\"]\n}\n```\n> Note: result URLs field may appear as `urls` — handle both `processed_urls` and `urls`.\n\n**Failed:**\n```json\n{\"status\": \"failed\", \"done\": true, \"error\": \"reason\"}\n```\n\n### GET `/api/v1/v2a/ControlFoley_output/{task_id}/{filename}` — Download Result\n\nFallback download endpoint for internal URLs that are not directly accessible.  \nReturns binary file stream (.flac at 44100 Hz or .mp4).\n\n### GET `/api/v1/v2a/models` — List Models\n\n```json\n{\"models\": [{\"name\": \"ControlFoley\", \"enabled\": true}]}\n```\n\n### GET `/health` — Health Check\n\n```json\n{\"status\": \"ok\", \"queue_size\": 3}\n```\n\n## Models\n\n| Model "},{"path":"skill-card.md","content":"## Description:\n\nA multi-functional audio generation tool for SFX generation, video-to-audio and text-to-audio.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[yjx-research](https://clawhub.ai/user/yjx-research)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and creators use this skill to generate sound effects, background audio, or Foley-style audio from text prompts, video files, and optional reference audio.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill uploads user media to a remote ControlFoley service for processing.\n\nMitigation: Use only non-sensitive media and validate behavior with a small test clip before broader use.\n\nRisk: The release security summary reports flaws that can expose local files or fetch untrusted URLs during normal use.\n\nMitigation: Review before installing, avoid prompt, model, or negative values beginning with @ or <, and prefer a patched version with native multipart upload, download URL validation, and size limits.\n\nRisk: Returned downloads come from service-provided URLs.\n\nMitigation: Treat generated downloads as untrusted files and inspect them before using them in downstream workflows.\n\n## Reference(s):\n\n- [ControlFoley Audio Generator API Reference](references/api-reference.md)\n- [ControlFoley project page](https://yjx-research.github.io/ControlFoley_web_page/)\n- [ControlFoley online demo](https://yjx-research.github.io/ControlFoley_web_page/#try-gen)\n- [ControlFoley model weights](https://huggingface.co/YJX-Xiaomi/ControlFoley/)\n- [ControlFoley source repository](https://github.com/xiaomi-research/controlfoley)\n\n## Skill Output:\n\n**Output Type(s):** [Files, Shell commands, API Calls, Guidance]\n\n**Output Format:** [Generated audio/video files with text status output]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Produces FLAC audio and, for video-to-audio workflows, MP4 video with generated audio.]\n\n## Skill Version(s):\n\n1.0.9 (source: release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"A multi-functional audio generation tool for SFX generation, video-to-audio and text-to-audio. 多功能音频生成工具，集成可控视频生成音频、文本生成音频等功能. Skill: ControlFoley Audio Generator Owner: yjx-research Summary: A multi-functional audio generation tool for SFX generation, video-to-audio and text-to-audio. 多功能音频生成工具，集成可控视频生成音频、文本生成音频等功能. Tags: latest:1.0.9 Version history: v1.0.9 | 2026-06-08T03:51:10.695Z | user - Removed the _meta.json file from the skill package. - No functional changes to user experience or features. - Internal metadata cleanup for this vers","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1237,"uniquenessScore":47,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T01:50:08.940Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T01:50:08.940Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T03:55:39.276Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}