{"id":"1550c817-ffb3-46ff-a1df-d46aeeb8c88a","entityType":"agent","slug":"clawhub-phosor-ai-phosor-ai-skills","name":"Phosor AI","canonicalUrl":"https://www.xpersona.co/agent/clawhub-phosor-ai-phosor-ai-skills","canonicalPath":"/agent/clawhub-phosor-ai-phosor-ai-skills","generatedAt":"2026-10-11T00:34:35.766Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T22:25:41.580Z","emptyReason":null},"description":"Generate AI videos, images and speech (text-to-video, image-to-video, reference-to-video, speech-to-video, animate, text-to-image, image-to-image, image edit, text-to-speech), bring your own LoRA models, and generate AI product/model photography for e-commerce (Image Studio) via the Phosor AI platform. Use when the user wants to create videos or images from text prompts, animate images, generate lip-synced video from audio, synthesize speech from text, generate images with a custom LoRA, generate product photography or model/clothing photography for e-commerce listings, or manage generation jobs.","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.2K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s179vhg33k8dh6daa97nftg9kn8dpdg5:phosor-ai-skills","sourceUrl":"https://clawhub.ai/phosor.ai/phosor-ai-skills","homepage":"https://clawhub.ai/phosor.ai/skills/phosor-ai-skills","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/phosor.ai/phosor-ai-skills","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/phosor.ai/skills/phosor-ai-skills","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":62,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Phosor AI technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T22:25:41.580Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T22:25:41.580Z","emptyReason":null},"stars":null,"forks":null,"downloads":1242,"packageName":null,"latestVersion":"1.3.2","tractionLabel":"1.2K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T22:25:41.512Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T22:25:41.580Z","lastCrawledAt":"2026-10-10T22:25:41.512Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T22:25:41.512Z","lastVerifiedAt":null,"highlights":[{"version":"1.3.2","createdAt":"2026-09-26T05:17:51.799Z","changelog":"## 1.3.2 — 2026-09-25 (API v1.2.2) `minimax/h3/reference-to-video/turbo` removed - Requests to `minimax/h3/reference-to-video/turbo` now return an unknown-model error - Use `minimax/h3/reference-to-video` instead — same inputs and pricing structure - `reference-to-video` (non-turbo) is unaffected - API contract version v1.2.2: a model id was removed; no other endpoints or parameters changed","fileCount":8,"zipByteSize":40765},{"version":"1.3.1","createdAt":"2026-09-24T14:05:11.213Z","changelog":"## 1.3.1 — 2026-09-24 (API v1.2.1) Reference-to-Video limits and pricing - Reference caps, both 480p and 768p: 9 images when sending only images, 5 images when videos or audio are also present, 3 reference videos, 3 reference audios - Total reference video length is 15.1s across all reference videos - `reference-to-video` now uses the same rates as the rest of H3: 480p $0.0045/s, 768p $0.012/s, reference images $0.0075 each after the first 5 - Bundled client validation and version string updated to match - API contract version v1.2.1: Reference-to-Video limits and pricing changed; no new endpoints or parameters ## 1.3.0 — 2026-09-22 (API v1.2.0) Reference metadata and H3 limits - Reference labels are forwarded in the same order as image, video, and audio URLs - 768p Ref2VA supports the same 9 image-only, 4 mixed-image, and 3 video caps as 480p - Correct the bundled client's stale 768p validation and 15-second video budget","fileCount":8,"zipByteSize":41056},{"version":"1.2.0","createdAt":"2026-09-22T12:06:08.927Z","changelog":"Reference-to-Video now accepts up to 15 seconds of combined reference video (turbo remains 10 seconds). Updated H3 pricing and reference-input billing documentation, added Reference-to-Video Turbo, and synchronized the bundled client and API reference.","fileCount":8,"zipByteSize":40980},{"version":"1.0.2","createdAt":"2026-03-23T02:29:39.047Z","changelog":"Phosor AI 1.0.0 – Initial Release - AI content generation platform, supporting Wan 2.2 14B text-to-video and image-to-video - Upload and use custom LoRA models for style customization - 16 CLI commands for job submission, uploads, status, results, and model listing - Supports preset resolutions (480p/720p/1080p), frame alignment rules, and usage quotas","fileCount":6,"zipByteSize":12394},{"version":"1.0.1","createdAt":"2026-03-23T01:58:01.008Z","changelog":"Phosor AI 1.0.1 - AI content generation platform, supporting Wan 2.2 14B text-to-video and image-to-video - Upload and use custom LoRA models for style customization - 16 CLI commands for job submission, uploads, status, results, and model listing - Supports preset resolutions (480p/720p/1080p), frame alignment rules, and usage quotas","fileCount":5,"zipByteSize":11318},{"version":"1.0.0","createdAt":"2026-03-23T01:06:59.932Z","changelog":"Initial release of phosor-ai 1.0.0 - AI content generation platform, supporting Wan 2.2 14B text-to-video and image-to-video - Upload and use custom LoRA models for style customization - 16 CLI commands for job submission, uploads, status, results, and model listing - Supports preset resolutions (480p/720p/1080p), frame alignment rules, and usage quotas","fileCount":5,"zipByteSize":11103}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s179vhg33k8dh6daa97nftg9kn8dpdg5:phosor-ai-skills","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s179vhg33k8dh6daa97nftg9kn8dpdg5:phosor-ai-skills` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/phosor.ai/phosor-ai-skills before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-phosor-ai-phosor-ai-skills/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-phosor-ai-phosor-ai-skills/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-phosor-ai-phosor-ai-skills/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-phosor-ai-phosor-ai-skills/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-phosor-ai-phosor-ai-skills/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-phosor-ai-phosor-ai-skills/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T00:34:35.760Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-phosor-ai-phosor-ai-skills/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-phosor-ai-phosor-ai-skills/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-phosor-ai-phosor-ai-skills/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-phosor-ai-phosor-ai-skills/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T22:25:41.580Z","emptyReason":null},"readme":"Skill: Phosor AI\n\nOwner: phosor.ai\n\nSummary: Generate AI videos, images and speech (text-to-video, image-to-video, reference-to-video, speech-to-video, animate, text-to-image, image-to-image, image edit, text-to-speech), bring your own LoRA models, and generate AI product/model photography for e-commerce (Image Studio) via the Phosor AI platform. Use when the user wants to create videos or images from text prompts, animate images, generate lip-synced video from audio, synthesize speech from text, generate images with a custom LoRA, generate product photography or model/clothing photography for e-commerce listings, or manage generation jobs.\n\nTags: latest:1.3.2\n\nVersion history:\n\nv1.3.2 | 2026-09-26T05:17:51.799Z | user\n\n## 1.3.2 — 2026-09-25 (API v1.2.2)\n\n`minimax/h3/reference-to-video/turbo` removed\n\n- Requests to `minimax/h3/reference-to-video/turbo` now return an unknown-model error\n- Use `minimax/h3/reference-to-video` instead — same inputs and pricing structure\n- `reference-to-video` (non-turbo) is unaffected\n- API contract version v1.2.2: a model id was removed; no other endpoints or parameters changed\n\nv1.3.1 | 2026-09-24T14:05:11.213Z | user\n\n## 1.3.1 — 2026-09-24 (API v1.2.1)\n\nReference-to-Video limits and pricing\n\n- Reference caps, both 480p and 768p: 9 images when sending only images, 5 images when\n  videos or audio are also present, 3 reference videos, 3 reference audios\n- Total reference video length is 15.1s across all reference videos\n- `reference-to-video` now uses the same rates as the rest of H3: 480p $0.0045/s,\n  768p $0.012/s, reference images $0.0075 each after the first 5\n- Bundled client validation and version string updated to match\n- API contract version v1.2.1: Reference-to-Video limits and pricing changed; no new endpoints or parameters\n\n## 1.3.0 — 2026-09-22 (API v1.2.0)\n\nReference metadata and H3 limits\n\n- Reference labels are forwarded in the same order as image, video, and audio URLs\n- 768p Ref2VA supports the same 9 image-only, 4 mixed-image, and 3 video caps as 480p\n- Correct the bundled client's stale 768p validation and 15-second video budget\n\nv1.2.0 | 2026-09-22T12:06:08.927Z | user\n\nReference-to-Video now accepts up to 15 seconds of combined reference video (turbo remains 10 seconds). Updated H3 pricing and reference-input billing documentation, added Reference-to-Video Turbo, and synchronized the bundled client and API reference.\n\nv1.0.2 | 2026-03-23T02:29:39.047Z | user\n\nPhosor AI 1.0.0 – Initial Release\n\n  - AI content generation platform, supporting Wan 2.2 14B text-to-video and image-to-video\n  - Upload and use custom LoRA models for style customization\n  - 16 CLI commands for job submission, uploads, status, results, and model listing\n  - Supports preset resolutions (480p/720p/1080p), frame alignment rules, and usage quotas\n\nv1.0.1 | 2026-03-23T01:58:01.008Z | user\n\nPhosor AI 1.0.1\n\n- AI content generation platform, supporting Wan 2.2 14B text-to-video and image-to-video\n- Upload and use custom LoRA models for style customization\n- 16 CLI commands for job submission, uploads, status, results, and model listing\n- Supports preset resolutions (480p/720p/1080p), frame alignment rules, and usage quotas\n\nv1.0.0 | 2026-03-23T01:06:59.932Z | user\n\nInitial release of phosor-ai 1.0.0                                                                                                                                \n                                                                                                                                                                  \n   - AI content generation platform, supporting Wan 2.2 14B text-to-video and image-to-video                                                                         \n  - Upload and use custom LoRA models for style customization                                                                                                       \n  - 16 CLI commands for job submission, uploads, status, results, and model listing                                                                                 \n  - Supports preset resolutions (480p/720p/1080p), frame alignment rules, and usage quotas\n\nArchive index:\n\nArchive v1.3.2: 8 files, 40765 bytes\n\nFiles: _meta.json (135b), CHANGELOG.md (5049b), README.md (1986b), references/api.md (25285b), scripts/phosor_client.py (92596b), skill-card.md (2041b), SKILL.md (24707b), VERSION (6b)\n\nFile v1.3.2:SKILL.md\n\n---\nname: phosor-ai-skills\ndescription: Generate AI videos, images and speech (text-to-video, image-to-video, reference-to-video, speech-to-video, animate, text-to-image, image-to-image, image edit, text-to-speech), bring your own LoRA models, and generate AI product/model photography for e-commerce (Image Studio) via the Phosor AI platform. Use when the user wants to create videos or images from text prompts, animate images, generate lip-synced video from audio, synthesize speech from text, generate images with a custom LoRA, generate product photography or model/clothing photography for e-commerce listings, or manage generation jobs.\nlicense: MIT-0\ncompatibility: Requires Python 3.7+ and network access to phosor.ai\nmetadata:\n  author: phosor.ai\n  version: \"1.3.2\"\n  api_version: \"v1.2.2\"\n  homepage: https://phosor.ai\n---\n\n# Phosor AI\n\nGenerate AI videos and images (text-to-video, image-to-video, speech-to-video, animate, text-to-image, image-to-image), bring your own LoRA models, and generate AI product/model photography for e-commerce (Image Studio) via the Phosor AI platform.\n\nFor detailed API endpoints, parameters, pricing, and limits, see [references/api.md](references/api.md).\n\n## Setup\n\nSet your API key:\n\n```bash\nexport PHOSOR_API_KEY=\"your-api-key-here\"\n```\n\nGet an API key at [phosor.ai](https://phosor.ai) → Settings → API Keys.\n\nThe CLI script is at `scripts/phosor_client.py`. All commands output JSON to stdout.\n\n## Base URL\n\nThe client talks to `https://phosor.ai` by default, over HTTPS. Get your key from phosor.ai → Settings → API Keys.\n\n```bash\npython3 scripts/phosor_client.py --api-key <your-key> check-key\n```\n\nTo point the same client at another Phosor endpoint, pass `--base-url <url>` (or set `PHOSOR_BASE_URL`).\nHTTPS is required; plain `http://` is accepted only for `localhost`, or when you explicitly add `--allow-http`.\n\nNote: `studio-analyze` / `studio-suite` `--image-url` must be the **full https S3 URL** returned by `upload-image`, not the bare S3 key path — a bare key errors with `unsupported URL scheme`.\n\n## Quick Start\n\n### MiniMax H3 — Text-to-Video\n\nH3 is **duration-based**, not frame-based: it ignores `--num-frames` / `--fps` (output is\nalways 24fps) and bills per output second. Pick the frame size with\n`--resolution-tier` + `--aspect` instead of `--width/--height`.\n\n```bash\npython3 scripts/phosor_client.py submit \"A cat walking on a beach at sunset\" \\\n  --model minimax/h3/text-to-video \\\n  --resolution-tier 768p --aspect 16:9 --duration 5\n```\n\n### MiniMax H3 — Image-to-Video\n\n```bash\n# Upload first (direct URLs are not accepted); then submit with the returned s3_key\npython3 scripts/phosor_client.py upload-image /path/to/first-frame.jpg\n\npython3 scripts/phosor_client.py submit \"The person starts dancing\" \\\n  --model minimax/h3/image-to-video \\\n  --image-url \"images/img-xxx.jpg\" \\\n  --end-image-url \"images/img-yyy.jpg\" \\\n  --resolution-tier 480p --aspect 9:16 --duration 6\n```\n\n`--end-image-url` is optional and pins the closing frame.\n\n### MiniMax H3 — Reference-to-Video (Ref2VA)\n\nFeed reference **images**, **videos**, and **audio** together; refer to them positionally in\nthe prompt as `<Picture 1>`, `<Picture 2>`, … At least one of `--reference-image-urls` /\n`--reference-video-urls` is required.\n\n```bash\npython3 scripts/phosor_client.py submit \\\n  \"Use <Picture 1> and <Picture 2> as sequential keyframes; slow push-in, cinematic 35mm look.\" \\\n  --model minimax/h3/reference-to-video \\\n  --reference-image-urls \"images/a.jpg,images/b.jpg\" \\\n  --reference-audio-urls \"audio/voice.mp3\" \\\n  --resolution-tier 768p --aspect 16:9 --duration 5\n```\n\nReference inputs are billed on top of the output — see **Ref2VA Pricing** below, and\n**Ref2VA Reference Limits** for the caps (the same at 480p and 768p).\n\n### Text-to-Video (Wan)\n\nWan is frame-based. Add `/turbo` to the model id for the fast, ~3x cheaper variant\n(it ignores `--steps` / `--guidance`).\n\n```bash\n# Submit T2V job (480p, 81 frames, 16fps)\npython3 scripts/phosor_client.py submit \"A cat walking on a beach at sunset\" \\\n  --width 854 --height 480 --num-frames 81 --fps 16\n\n# Check status\npython3 scripts/phosor_client.py status <request_id>\n\n# Get result (video URL)\npython3 scripts/phosor_client.py result <request_id>\n```\n\n### Image-to-Video\n\n**Two-step flow**: upload image first, then submit with the returned S3 key.\n\n```bash\n# Step 1: Upload image\npython3 scripts/phosor_client.py upload-image /path/to/photo.jpg\n# Returns: {\"file_id\": \"img-xxx\", \"s3_key\": \"images/img-xxx.jpg\", ...}\n\n# Step 2: Submit I2V job using the s3_key as image_url\npython3 scripts/phosor_client.py submit \"The person in the photo starts dancing\" \\\n  --image-url \"images/img-xxx.jpg\" --width 854 --height 480\n```\n\n### Text-to-Image\n\n```bash\n# Submit T2I job (1024x1024, default settings)\npython3 scripts/phosor_client.py submit \"A futuristic city skyline at dusk\" \\\n  --model qwen-image/v2512/text-to-image --width 1024 --height 1024\n\n# Generate multiple images at once (1-4)\npython3 scripts/phosor_client.py submit \"A futuristic city skyline at dusk\" \\\n  --model z-image/turbo/text-to-image --width 1024 --height 768 --num-images 4\n\n# Check status and get result (image URL)\npython3 scripts/phosor_client.py status <request_id>\npython3 scripts/phosor_client.py result <request_id>\n# Returns: {\"data\": {\"image\": {\"url\": \"...\"}, \"seed\": 12345}, ...}\n```\n\n### Image-to-Image\n\n**Two-step flow**: upload source image first, then submit with the returned S3 key.\n\n```bash\n# Step 1: Upload source image\npython3 scripts/phosor_client.py upload-image /path/to/photo.jpg\n# Returns: {\"file_id\": \"img-xxx\", \"s3_key\": \"images/img-xxx.jpg\", ...}\n\n# Step 2: Submit I2I job using the s3_key as image_url\npython3 scripts/phosor_client.py submit \"Transform into oil painting style\" \\\n  --model z-image/turbo/image-to-image --image-url \"images/img-xxx.jpg\" \\\n  --width 1024 --height 1024 --strength 0.7\n```\n\n### Image Edit (Multi-image Reference)\n\n**Two-step flow**: upload reference images first, then submit with S3 keys as `image_urls`.\n\n```bash\n# Step 1: Upload reference images (up to 3)\npython3 scripts/phosor_client.py upload-image /path/to/ref1.jpg\npython3 scripts/phosor_client.py upload-image /path/to/ref2.jpg\n\n# Step 2: Submit image-edit job\npython3 scripts/phosor_client.py submit \\\n  \"The girl in image 1 is wearing the outfit from image 2\" \\\n  --model qwen-image/v2511/image-edit \\\n  --image-urls '[\"images/img-ref1.jpg\",\"images/img-ref2.jpg\"]' \\\n  --width 1024 --height 1024\n\n# Turbo variant (faster, Lightning LoRA built-in)\npython3 scripts/phosor_client.py submit \\\n  \"The girl in image 1 is wearing the outfit from image 2\" \\\n  --model qwen-image/v2511/image-edit \\\n  --image-urls '[\"images/img-ref1.jpg\",\"images/img-ref2.jpg\"]'\n```\n\n### Speech-to-Video (S2V)\n\n**Two-step flow**: upload both audio and reference image first, then submit with the returned S3 keys.\n\n```bash\n# Step 1: Upload reference image\npython3 scripts/phosor_client.py upload-image /path/to/face.jpg\n# Returns: {\"file_id\": \"img-xxx\", \"s3_key\": \"images/img-xxx.jpg\", ...}\n\n# Step 2: Submit S2V job using the s3_key as image_url and audio URL as audio_url\npython3 scripts/phosor_client.py submit \"A person speaking naturally\" \\\n  --model wan/v2.2-a14b/speech-to-video \\\n  --image-url \"images/img-xxx.jpg\" --audio-url \"https://example.com/speech.wav\" \\\n  --width 854 --height 480\n```\n\n### Animate\n\n**Two-step flow**: upload both source video and reference image first, then submit.\n\n```bash\n# Step 1: Upload reference image\npython3 scripts/phosor_client.py upload-image /path/to/character.jpg\n# Returns: {\"file_id\": \"img-xxx\", \"s3_key\": \"images/img-xxx.jpg\", ...}\n\n# Step 2: Submit Animate job using the s3_key as image_url and video URL as video_url\npython3 scripts/phosor_client.py submit \"The character performs the dance moves\" \\\n  --model wan/v2.2-a14b/animate \\\n  --image-url \"images/img-xxx.jpg\" --video-url \"https://example.com/dance.mp4\" \\\n  --width 854 --height 480\n```\n\n### Text-to-Image (GPT Image 2)\n\nIts own resolution set and **always exactly 1 image** (`--num-images` is ignored).\n\n```bash\npython3 scripts/phosor_client.py submit \"A ceramic mug on a linen cloth, soft window light\" \\\n  --model openai/gpt-image-2/text-to-image --width 1024 --height 1024\n```\n\nAllowed sizes: **1024×1024 only**. Any other size is rejected with `400 Invalid parameters`.\nIt ignores `--num-images`, `--steps` and `--guidance` — the only parameters it accepts are\nprompt, model, width, height and seed.\n\n### Text-to-Image (FLUX.2-dev)\n\nFLUX.2-dev has its own resolution whitelist and is **fixed at 1 image per request**\n(`--num-images` and `--steps` are ignored).\n\n```bash\npython3 scripts/phosor_client.py submit \"Editorial product photo, soft window light\" \\\n  --model flux2/dev/text-to-image --width 2048 --height 1536\n```\n\n### Image Edit (FLUX.2-dev)\n\n```bash\npython3 scripts/phosor_client.py upload-image /path/to/source.jpg\n\npython3 scripts/phosor_client.py submit \"Replace the background with a marble surface\" \\\n  --model flux2/dev/image-edit --image-url \"images/img-xxx.jpg\" \\\n  --width 1024 --height 1024\n```\n\n### Text-to-Speech\n\nTTS takes `text` (not a prompt) and is billed per character with a minimum charge.\n\n```bash\npython3 scripts/phosor_client.py submit-tts \"Hello, welcome to Phosor AI.\" \\\n  --speaker Sohee --language English\n```\n\n### LoRA Upload (Custom Pre-trained)\n\n**Video LoRA** requires two .safetensors files (high_noise + low_noise). **Image LoRA** requires a single .safetensors file.\n\n```bash\n# Video LoRA: upload two .safetensors files\npython3 scripts/phosor_client.py upload-lora high_noise.safetensors low_noise.safetensors --name \"My Style\"\n\n# Image LoRA: import single .safetensors file via URL\npython3 scripts/phosor_client.py import-lora \\\n  \"https://example.com/my_lora.safetensors\" \\\n  --name \"My Image Style\"\n\n# Video LoRA: import two files via URL\npython3 scripts/phosor_client.py import-lora \\\n  \"https://example.com/high_noise.safetensors\" \\\n  \"https://example.com/low_noise.safetensors\" \\\n  --name \"My Video Style\"\n\n# Check status, then use\npython3 scripts/phosor_client.py lora-status <lora_id>\npython3 scripts/phosor_client.py submit \"A person walking\" --lora-id <lora_id>\n```\n\n## CLI Commands\n\n| Command | Description | Key Arguments |\n|---------|-------------|---------------|\n| `check-key` | Validate API key | — |\n| `submit` | Submit inference job (T2V/I2V/S2V/Animate/T2I/I2I) | `prompt`, `--width`, `--height`, `--num-frames`, `--fps`, `--steps`, `--guidance`, `--image-url`, `--audio-url`, `--video-url`, `--lora-id`, `--lora-scale`, `--loras`, `--seed`, `--negative-prompt`, `--model`, `--num-images`, `--strength`, `--output-format` |\n| `status` | Get job status | `request_id` |\n| `result` | Get job result (video or image URL) | `request_id` |\n| `poll` | Poll all pending jobs | — |\n| `list` | List locally tracked pending jobs | — |\n| `history` | Get job history | `--limit` |\n| `upload-image` | Upload image for I2V or I2I | `file` |\n| `import-image` | Import image from URL | `url`, `--filename` |\n| `upload-lora` | Upload LoRA (two .safetensors for video) | `high_noise_file`, `low_noise_file`, `--name` |\n| `import-lora` | Import LoRA from URLs (one or two files) | `high_noise_url`, `[low_noise_url]`, `--name` |\n| `loras` | List LoRA models | `--limit`, `--offset` |\n| `lora-status` | Get LoRA upload/import status | `lora_id` |\n| `save-lora` | Activate a LoRA (extends expiry to 7 days) | `lora_id`, `--name` |\n| `delete-lora` | Delete a LoRA model | `lora_id` |\n| `submit-tts` | Submit a text-to-speech job (Qwen3-TTS) — keys off `text`, not a prompt | `text`, `--speaker`, `--language`, `--seed`, `--temperature`, `--top-p`, `--top-k`, `--repetition-penalty` |\n| `models` | List available video/image models (static offline reference) | — |\n| `studio-features` | List Image Studio endpoints, fields, billing (static offline reference) | — |\n| `studio-pricing` | Get live Image Studio pricing | — |\n| `studio-analyze` | AI-analyze a product/garment image or reference URL (freemium) | `--target agent\\|product\\|model\\|reference`, `--image-url`, `--url`, `--prompt`, `--language` |\n| `studio-layouts` | List the layout template library (query, then select) — static asset on the **web front** (phosor.ai), not a `/api/v1` endpoint; needs no key; on a self-hosted deployment set `PHOSOR_WEB_BASE_URL` if the web front is not the gateway host | `--module` (product\\|clothing), `--type` (selling_point\\|aplus\\|white_bg\\|scene\\|closeup\\|size_chart) |\n| `studio-suite` | Generate a product image suite | `--image-url`, `--layout-types`, `--count-per-type`, `--custom-suggestions`, `--template-ids` (ids from `studio-layouts`, auto-expanded to custom_suggestions like the UI's manual pick — use this to get **text-callout selling-point / A+ layouts** and model templates), `--product-info`, `--aspect-ratio`, `--gen-language`, `--model`, `--same-style-reference` |\n| `studio-clothing-suite` | Generate a model/garment image suite | `--image-urls`, `--main-image-types`, `--aplus-types`, `--product-info`, `--brand-config`, `--aspect-ratio`, `--gen-language`, `--model`, `--same-style-reference` |\n| `studio-status` | Get Image Studio job status (separate id space, same `request_id` key) | `request_id` |\n| `studio-cancel` | Cancel a running generation — queued images refunded, already-generating ones charged | `request_id` |\n| `studio-my-works` | List past Image Studio generations | `--task-type`, `--limit`, `--offset` |\n| `studio-call` | Generic call for any other Image Studio endpoint (remove-bg, replace, inpaint, erase, handheld, translate, outpaint, recolor, enhance, upscale, scene-compose, scene-variation, real-model-swap, mannequin-swap, model-scene-swap, ai-outfit, pose-variation, ai-wearable) | `method`, `path`, `--json` |\n\n## Image Studio (Product & Model Photography)\n\nImage Studio is a separate product surface for e-commerce sellers — AI product photography and model/clothing photography — reached through the **same gateway and API key** as video/LoRA, under the `/api/v1/image-studio` prefix. It has its own async namespace - the same key name `request_id`, but a **separate id space**: an Image Studio `request_id` is not valid on `/api/v1/inference/status/...` and vice versa - and its own pricing (flat per-image rate + freemium analyze quota, not per-frame). Full endpoint/parameter reference: [references/api.md](references/api.md#image-studio-product--model-photography--separate-product-surface).\n\n### Quick Start: Product Suite\n\n```bash\n# 1. Upload the product photo\npython3 scripts/phosor_client.py upload-image /path/to/product.jpg\n# Returns: {\"file_id\": \"img-xxx\", \"s3_key\": \"images/img-xxx.jpg\", ...}\n\n# 2. (Optional) AI-analyze it first for richer generation context\npython3 scripts/phosor_client.py studio-analyze --target product --image-url \"images/img-xxx.jpg\"\n\n# 3. Generate a product image suite\npython3 scripts/phosor_client.py studio-suite --image-url \"images/img-xxx.jpg\" \\\n  --layout-types \"white_background,lifestyle_scene\" --count-per-type 2\n\n# 4. Poll for the result\npython3 scripts/phosor_client.py studio-status <request_id>\n```\n\n### Quick Start: Clothing/Model Suite\n\n```bash\npython3 scripts/phosor_client.py upload-image /path/to/garment.jpg\npython3 scripts/phosor_client.py studio-clothing-suite \\\n  --image-urls \"images/img-xxx.jpg\" \\\n  --main-image-types '{\"model_shot\":2,\"selling_point\":1}' \\\n  --aplus-types '{\"standard_aplus\":1}'\npython3 scripts/phosor_client.py studio-status <request_id>\n```\n\n### Quick Start: One-off Edits (remove-bg, inpaint, translate, etc.)\n\nThe long tail of single-purpose editing endpoints doesn't get a dedicated subcommand — use `studio-call` with the exact field names from [references/api.md](references/api.md#image-studio-product--model-photography--separate-product-surface):\n\n```bash\npython3 scripts/phosor_client.py studio-call POST /product/remove-bg \\\n  --json '{\"image_url\": \"images/img-xxx.jpg\", \"count\": 2}'\npython3 scripts/phosor_client.py studio-status <request_id>\n```\n\n### Key facts\n\n- **Every Image Studio call requires `X-API-Key`** (`PHOSOR_API_KEY`), including `GET /pricing` — there is no unauthenticated endpoint under this prefix.\n- **All generation/analyze endpoints are async**: POST returns `{\"request_id\": \"...\", \"status\": \"pending\"}`; poll `studio-status <request_id>` until `status` is `\"done\"`, `\"error\"` or `\"cancelled\"`. Earlier revisions of this skill said the key was `job_id` - it is not, and reading it yields `undefined`. Image Studio ids live in a **separate id space** from video/LoRA: `poll`/`status`/`result` will not accept an Image Studio `request_id`.\n- **Cancelling**: `POST /jobs/{request_id}/cancel` stops a running generation. Images still queued are refunded; images already generating are charged and cannot be stopped; images already delivered bill once through the normal path. The response reports the split as `refunded_queued`, `charged_running` and `already_done`, and the task then polls as `status: \"cancelled\"` - not an error.\n- **Pricing is per-image, not per-frame**: call `studio-pricing` for the live rate. Partial success (e.g. 3 of 5 images) bills only the successes.\n- **Analyze is freemium**: `agent/analyze`, `product/analyze`, `model/analyze` share a daily free quota before per-call billing kicks in.\n- **`model_attrs` matters for model-photography endpoints** (real-model-swap, mannequin-swap, ai-outfit, ai-wearable) — pass `{gender, age_group, ethnicity, skin_tone, hair_color}` explicitly; it is not reliably inferred from the source image alone.\n- Run `studio-features` for the full offline endpoint/field catalog without leaving the terminal.\n\n## Key Constraints\n\n### Video Resolutions (exact pairs only)\n\n| Preset | Width × Height | Max Frames (turbo) | Max Frames (standard) |\n|--------|---------------|-------------------|----------------------|\n| 480p landscape | 854 × 480 | 161 | 161 |\n| 480p portrait | 480 × 854 | 161 | 161 |\n| 720p landscape | 1280 × 720 | 161 | 161 |\n| 720p portrait | 720 × 1280 | 161 | 161 |\n| 1080p landscape | 1920 × 1080 | 153 | **81** |\n| 1080p portrait | 1080 × 1920 | 153 | **81** |\n\n> Standard (non-turbo) mode: 1080p is capped at 81 frames due to generation time limits.\n\n### S2V / Animate Video Resolutions (exact pairs only)\n\n| Preset | Width x Height | Max Frames |\n|--------|---------------|------------|\n| 480p landscape | 854 x 480 | 161 |\n| 480p portrait | 480 x 854 | 161 |\n| 512p square | 512 x 512 | 161 |\n| 720p landscape | 1280 x 720 | 161 |\n| 720p portrait | 720 x 1280 | 161 |\n\n### Image Resolutions (exact pairs only)\n\n| Preset | Width × Height |\n|--------|---------------|\n| Square small | 512 × 512 |\n| Square | 1024 × 1024 |\n| Landscape | 1024 × 768 |\n| Portrait | 768 × 1024 |\n| Wide landscape | 1280 × 768 |\n| Tall portrait | 768 × 1280 |\n\n### MiniMax H3 Frame Sizes (`--resolution-tier` + `--aspect`)\n\n| Tier | 16:9 | 4:3 | 1:1 | 3:4 | 9:16 |\n|------|------|-----|-----|-----|------|\n| 480p | 832 × 480 | 640 × 480 | 480 × 480 | 480 × 640 | 480 × 832 |\n| 768p | 1344 × 768 | 1024 × 768 | 768 × 768 | 768 × 1024 | 768 × 1344 |\n\n`duration` is 4–15 seconds (default 5). Output FPS is fixed at 24 and\n`frames_per_second` is ignored.\n\n### Ref2VA Reference Limits (`minimax/h3/reference-to-video`)\n\nCaps differ per tier, and the image cap is higher when you send **only** images:\n\n| Limit | 480p | 768p |\n|-------|------|------|\n| Reference images (with videos/audio present) | 5 | 5 |\n| Reference images (images only) | 9 | 9 |\n| Reference videos | 3 | 3 |\n| Reference audios | 3 | 3 |\n\n| Limit | Value |\n|-------|-------|\n| Total reference video length | 15.1 s (across all reference videos) |\n| Reference video FPS ceiling | 24 |\n| Reference audio length | 10 s each |\n| Reference image longest edge | 2048 px |\n| Reference image aspect ratio | ≤ 4.0 |\n\n### FLUX.2-dev Resolutions (exact pairs only)\n\n| Width × Height |\n|---------------|\n| 2048 × 1536 · 1536 × 2048 |\n| 2048 × 1152 · 1152 × 2048 |\n| 2048 × 2048 · 1024 × 1024 |\n\n> FLUX.2-dev does not accept the general image resolution list above, and always\n> returns exactly 1 image.\n\n### Frame Alignment (video only)\n\nFrames must follow `1 + 4*k` where `k >= 1` (e.g. 5, 9, 13, ... 81, 85, ...). Server auto-aligns down.\n\n### Video Inference Parameters\n\n| Parameter | Default | Range |\n|-----------|---------|-------|\n| `frames_per_second` | 16 | 4–60 |\n| `num_inference_steps` | 4 | 4–40 |\n| `guidance_scale` | 1.0 | 1.0–10.0 |\n\n### Image Inference Parameters\n\n| Parameter | Default | Range | Notes |\n|-----------|---------|-------|-------|\n| `num_images` | 1 | 1–4 | Number of images to generate |\n| `num_inference_steps` | varies | 1–4 (z-image turbo), 1–40 (qwen-image) | Model-dependent max |\n| `guidance_scale` | varies | 1.0–20.0 | — |\n| `strength` | — | 0.0–1.0 | Image-to-image only: how much to transform the source |\n| `output_format` | png | png, jpeg | Output file format |\n\n### Concurrency\n\nModel API inference jobs run concurrently up to a per-account cap; over it the submit\nreturns `429` and you retry after a job finishes.\n\n| Tier | Plan | Concurrent Model API jobs |\n|------|------|---------------------------|\n| Free | — | 1 |\n| 1 | Starter | 2 |\n| 2 | Standard | 4 |\n| 3 | Pro | 8 |\n\nImage Studio runs on its own pool and does **not** consume this quota — a suite and a\nvideo generation can run at the same time.\n\n### Multiple LoRAs\n\n```bash\npython3 scripts/phosor_client.py submit \"A person dancing\" \\\n  --loras '[{\"lora_id\": \"lora-abc\", \"lora_scale\": 0.8}, {\"lora_id\": \"lora-def\", \"lora_scale\": 0.5}]'\n```\n\n### Two-Step Upload Rule\n\nFiles must be uploaded before use — direct URLs are NOT supported in `submit --image-url`:\n\n1. **Image** → `upload-image` / `import-image` → returns `s3_key` → use as `--image-url`\n2. **LoRA** → `upload-lora` / `import-lora` → returns `lora_id` → use as `--lora-id`\n\n## Queue Flow\n\n```\nPENDING → PROCESSING → COMPLETED / FAILED\n```\n\nThe `poll` command checks all locally-tracked pending jobs and removes completed/failed ones.\n\n## MiniMax H3 Pricing (per output second)\n\nPer output second, regardless of clip length:\n\n| Model | 480p USD/sec | 768p USD/sec | 480p credits/sec | 768p credits/sec |\n|-------|--------------|--------------|------------------|------------------|\n| `text-to-video`, `image-to-video` | $0.0045 | $0.012 | 0.045 | 0.12 |\n| `reference-to-video` | $0.0045 | $0.012 | 0.045 | 0.12 |\n\n### Ref2VA inputs\n\nOutput **plus** reference inputs:\n\n```\ntotal = duration x base_rate\n      + max(0, reference_images - 5) x image_rate   # first 5 images free\n      + reference_video_seconds x base_rate         # reference audio is free\n```\n\n| Model | Tier | Base USD/sec | Reference image | Reference audio | Reference video USD/sec |\n|-------|------|--------------|-----------------|-----------------|-------------------------|\n| `reference-to-video` | 480p | $0.0045 | first 5 free, then $0.0075 each | free | $0.0045 |\n| `reference-to-video` | 768p | $0.012 | first 5 free, then $0.0075 each | free | $0.012 |\n\nExample — `reference-to-video`, 5s at 768p with 4 reference images: `5 x $0.012` = 0.6 credits.\n5s at 480p with 9 images: `5 x $0.0045 + 4 x $0.0075` = 0.525 credits.\n\n## Wan Video Pricing (per frame)\n\n| Tier | Standard | + LoRA | Turbo |\n|------|----------|--------|-------|\n| 480p | $0.0009375 | $0.00125 | $0.0003125 |\n| 512p | $0.0013125 | $0.001625 | $0.0004375 |\n| 720p | $0.001875 | $0.0021875 | $0.000625 |\n| 1080p | $0.0025 | $0.003 | $0.0010938 |\n\nLoRA multiplier on turbo: 1.2x.\n\n## S2V Pricing (per frame)\n\n| Tier | Cost |\n|------|------|\n| 480p | $0.0009375 |\n| 512p | $0.0013125 |\n| 720p | $0.001875 |\n\n## Animate Pricing (per frame)\n\n| Tier | Cost |\n|------|------|\n| 480p | $0.00125 |\n| 512p | $0.00175 |\n| 720p | $0.0025 |\n\n## Image Pricing (per image)\n\n| Model | USD | Credits |\n|-------|-----|---------|\n| GPT Image 2 T2I | $0.03 | 0.3 |\n| FLUX.2-dev T2I | $0.006 | 0.06 |\n| FLUX.2-dev Image Edit | $0.012 | 0.12 |\n| qwen-image T2I | $0.015 | 0.15 |\n| qwen-image T2I + LoRA | $0.018 | 0.18 |\n| qwen-image-edit | $0.003 | 0.03 |\n| z-image turbo (T2I / I2I) | $0.0025 | 0.025 |\n| z-image turbo + LoRA | $0.003 | 0.03 |\n\nFlat per image at every resolution. Multiply by `num_images` (1–4) where the model\nsupports it; FLUX.2-dev is fixed at 1 image per request.\n\n## Audio Pricing\n\n| Item | Cost |\n|------|------|\n| Qwen3-TTS | $0.00003 per character |\n| Minimum charge | $0.003 per request |\n\nExchange rate: 10 credits = $1 USD. Credits pre-deducted, auto-refunded on failure.\nLive rates: `GET /api/v1/pricing/config` — always prefer it over any table here.\n\nFile v1.3.2:README.md\n\n# Phosor AI — Agent Skill\n\n**Skill version 1.3.2** · API v1.2.2 · updated 2026-09-25\n\nTwo version numbers, on purpose: the skill version tracks this package\n(commands, docs, the bundled client), the API version tracks the gateway\ncontract. A skill release that only rewords docs or fixes the client does not\nmove the API version, and a gateway change does not force a skill release.\n\nVerify what you installed: `python3 scripts/phosor_client.py --version`\n\n## Quick Start\n\n```bash\nexport PHOSOR_API_KEY=\"your-key\"\n\n# Text-to-Video\npython3 scripts/phosor_client.py submit \"A cat walking on a beach\" --width 854 --height 480\n\n# Image-to-Video (two-step: upload then submit)\npython3 scripts/phosor_client.py upload-image photo.jpg\npython3 scripts/phosor_client.py submit \"The scene comes alive\" --image-url \"images/img-xxx.jpg\"\n\n# Check status / get result\npython3 scripts/phosor_client.py status <request_id>\npython3 scripts/phosor_client.py result <request_id>\n\n# LoRA Upload (custom pre-trained)\npython3 scripts/phosor_client.py upload-lora high_noise.safetensors low_noise.safetensors\n\npython3 scripts/phosor_client.py save-lora <lora_id>\n\n# Image Studio: AI product photography (separate product surface, same API key)\npython3 scripts/phosor_client.py upload-image product.jpg\npython3 scripts/phosor_client.py studio-suite --image-url \"<s3_url from upload>\" \\\n  --layout-types \"white_background,lifestyle_scene\" --count-per-type 2\npython3 scripts/phosor_client.py studio-status <job_id>\n```\n\nSee [SKILL.md](SKILL.md#image-studio-product--model-photography) for the full Image Studio quick start (clothing/model suite, one-off edits like remove-bg/inpaint/translate).\n\n## Requirements\n\n- Python 3.7+ (stdlib only, no pip install needed)\n- `PHOSOR_API_KEY` environment variable\n\n## Commands\n\nRun `python3 scripts/phosor_client.py --help` for all 31 commands (23 video/LoRA + 8 Image Studio).\n\n## Links\n\n- [Phosor AI](https://phosor.ai)\n- [API Documentation](https://docs.phosor.ai)\n\nFile v1.3.2:_meta.json\n\n{\n  \"ownerId\": \"kn7dkx5wmnapr9y2x91m6qnb8s82w04e\",\n  \"slug\": \"phosor-ai-skills\",\n  \"version\": \"1.3.2\",\n  \"publishedAt\": 1790399871799\n}\n\nFile v1.3.2:references/api.md\n\n# Phosor AI API Reference\n\nBase URL: `https://phosor.ai`\n\nAll endpoints require `X-API-Key` header unless noted otherwise.\n\n## Endpoints\n\n### Models\n\n| Method | Path | Auth | Description |\n|--------|------|------|-------------|\n| GET | `/api/v1/models` | None | List available models |\n\n### Inference\n\n| Method | Path | Description |\n|--------|------|-------------|\n| POST | `/api/v1/inference/submit` | Submit video or image generation job |\n| GET | `/api/v1/inference/status/{request_id}` | Get job status + progress |\n| GET | `/api/v1/inference/result/{request_id}` | Get completed result (video or image URL) |\n| GET | `/api/v1/inference/history` | Get user's job history |\n\n### Storage — Image / Audio\n\n| Method | Path | Description |\n|--------|------|-------------|\n| POST | `/api/v1/storage/image/upload` | Multipart image or audio upload (images: jpg/png/webp; audio for S2V: mp3/wav/flac/aac/ogg/m4a) |\n| POST | `/api/v1/storage/image/import` | Import from public URL |\n\n### Storage — LoRA\n\n| Method | Path | Description |\n|--------|------|-------------|\n| POST | `/api/v1/storage/lora/upload` | Upload two .safetensors files (video LoRA: high_noise + low_noise) |\n| POST | `/api/v1/storage/lora/import` | Import from HTTPS URLs (video: two files, image: single file) |\n\n### LoRA Management\n\n| Method | Path | Description |\n|--------|------|-------------|\n| GET | `/api/v1/loras` | List LoRA models |\n| GET | `/api/v1/loras/{lora_id}` | Get single LoRA details |\n| GET | `/api/v1/loras/{lora_id}/status` | Get processing status |\n| POST | `/api/v1/loras/{lora_id}/save` | Activate a LoRA (extend expiry to 7 days) |\n| DELETE | `/api/v1/loras/{lora_id}` | Soft delete |\n\n### Image Studio (product & model photography — separate product surface)\n\nAll paths below are relative to `/api/v1/image-studio` (e.g. the full path for `/product/suite` is `/api/v1/image-studio/product/suite`). **Every path requires `X-API-Key`, including `GET /pricing`** — there is no unauthenticated Image Studio endpoint.\n\nAll generation/analyze endpoints are async: POST returns `{\"request_id\": \"...\", \"status\": \"pending\"}` immediately; poll `GET /jobs/{request_id}` until `status` is `\"done\"`, `\"error\"` or `\"cancelled\"`.\n\n> **The async key is `request_id`, not `request_id`.** Earlier revisions of this document said\n> `request_id`; that key is never present in a response. Reading it yields `undefined`, and the\n> poll then never resolves. Image Studio's `request_id` is a separate namespace from the\n> video/LoRA `request_id` — do not pass one to the other's status endpoint.\n\n| Method | Path | Description |\n|--------|------|-------------|\n| POST | `/agent/analyze` | AI-analyze an image for the Agent-image workflow (freemium) |\n| POST | `/product/analyze` | AI-analyze a product image ahead of `product/suite` (freemium) |\n| POST | `/model/analyze` | AI-analyze a garment image ahead of `model/clothing-suite` (freemium) |\n| POST | `/product/reference/analyze` | Analyze a reference product page/image for style-matching (free) |\n| POST | `/product/suite` | Generate a product image suite (multiple layout types) |\n| POST | `/product/scene-compose` | Composite a product into a reference scene |\n| POST | `/product/scene-variation` | Generate scene/background variations of a product photo |\n| POST | `/product/remove-bg` | Remove background (transparent) |\n| POST | `/product/replace` | Replace/refresh the product in an existing scene composite |\n| POST | `/product/inpaint` | Masked region fill |\n| POST | `/product/erase` | Masked region removal |\n| POST | `/product/handheld` | Generate a hand-holding-product shot |\n| POST | `/product/translate` | Translate on-image text to other language(s) |\n| POST | `/product/outpaint` | Expand canvas beyond original image bounds |\n| POST | `/product/recolor` | Recolor product or region |\n| POST | `/product/enhance` | AI enhancement/upgrade pass |\n| POST | `/product/upscale` | Upscale to 1024×1024 |\n| POST | `/model/clothing-suite` | Generate a full garment/model image suite (model shots, selling points, size chart, A+ modules) |\n| POST | `/model/real-model-swap` | Swap in a real human model wearing the garment |\n| POST | `/model/mannequin-swap` | Swap in a mannequin wearing the garment |\n| POST | `/model/model-scene-swap` | Swap the scene/background behind an existing model photo |\n| POST | `/model/ai-outfit` | Dress a described model in the garment |\n| POST | `/model/ai-wearable` | Generate a model wearing/using an accessory |\n| POST | `/model/pose-variation` | Generate pose variations from a source model photo |\n| GET | `/jobs/{request_id}` | Poll job status/result (Image Studio's own async namespace) |\n| GET | `/pricing` | Live pricing: `per_image_credits`, `per_analyze_credits`, `analyze_daily_free_quota` |\n| GET | `/my-works` | List past Image Studio generations (paginated) |\n| GET | `/my-works/{request_id}` | One past generation with its inputs and outputs |\n| GET | `/my-works/{request_id}/download` | Download the generated images (single file or archive) |\n| DELETE | `/my-works/{request_id}` | Soft-delete a past generation |\n| POST | `/jobs/{request_id}/cancel` | Cancel a running generation. See \"Cancelling\" below |\n| POST | `/tasks/{task_id}/review` | Submit a per-image review (thumbs) for a finished task |\n| GET | `/tasks/{task_id}/review` | Read the review already submitted for a task |\n| POST | `/feedback` | Send feedback about a generation |\n\n#### Common parameters\n\n| Parameter | Applies to | Notes |\n|-----------|-----------|-------|\n| `image_url` | analyze + most edit endpoints | S3 key or URL of the source image; some endpoints alias this as `product_image_url` |\n| `model` | all generation endpoints | Optional. Accepted values: `openai:gpt-image-2` and `phosor:phosor-model-api` (Phosor's image model). Analyze endpoints do not take this parameter |\n| `aspect_ratio` | most generation endpoints | One of `1:1`, `3:4`, `4:3`, `9:16`, `16:9` (default varies per endpoint, e.g. `3:4` for clothing-suite, `1:1` for most product tools) |\n| `count` | most single-image-in-single-image-out endpoints | How many output images to generate, typically clamped 1–20 (varies per endpoint) |\n| `prompt` | analyze + most edit endpoints | Free-text hint appended to the internal template — not a full prompt replacement |\n| `same_style_reference` | `product/suite`, `model/clothing-suite` | Object from a prior `product/reference/analyze` call, used to match layout/style to a reference image |\n| `model_attrs` | model-photography endpoints (`real-model-swap`, `mannequin-swap`, `ai-outfit`, `ai-wearable`) | `{gender, age_group, ethnicity, skin_tone, hair_color}` — **required for accurate results**, not reliably inferred from the source image alone |\n\n#### `POST /product/suite` body\n\n| Field | Required | Notes |\n|-------|----------|-------|\n| `product_image_url` | Yes | (or `image_url`) |\n| `layout_types` | No | list[str] of layout type ids |\n| `count_per_type` | No | int, default 1 |\n| `custom_suggestions` | No | list |\n| `product_info` | No | free-text product description/facts (authoritative over what the AI guesses from the image) |\n| `same_style_reference` | No | dict from `product/reference/analyze` |\n| `brand_config` | No | dict — brand color/font/platform/tone settings |\n| `aspect_ratio` | No | `1:1`\\|`3:4`\\|`4:3`\\|`9:16`\\|`16:9` |\n| `gen_language` | No | language for any on-image text |\n| `model` | No | see Common parameters |\n\nBilled count = `len(layout_types) * count_per_type + len(custom_suggestions)` (min 1).\n\n#### `POST /model/clothing-suite` body\n\n| Field | Required | Notes |\n|-------|----------|-------|\n| `clothing_image_urls` | Yes | list[str], up to 5 used for generation |\n| `main_image_types` | No | dict of `{model_shot\\|grass_shot\\|selling_point\\|size_chart: count}` |\n| `aplus_types` | No | dict of `{standard_aplus\\|mobile_aplus\\|basic_aplus\\|custom_ratio: count}` |\n| `product_info` | No | free-text |\n| `brand_config` | No | dict |\n| `same_style_reference` | No | dict |\n| `aspect_ratio` | No | default `3:4` |\n| `gen_language` | No | |\n| `model` | No | |\n\nBilled count = sum of all `main_image_types`/`aplus_types` counts (min 1). At least one image type must be selected, or a 400 is returned. Job result includes `grouped_images: [{label, url}]` with a display label per image type (model photo, lifestyle, selling point, size chart, premium A+, mobile A+, standard A+, custom ratio; labels are returned in Chinese), plus `failed_labels` for any type that failed to generate (partial success is normal, not an error state).\n\n#### `GET /jobs/{request_id}` response shape\n\n- `status: \"pending\"` — includes `partial_images: [{url, label}]` for images completed so far (useful for progressive UIs).\n- `status: \"done\"` — includes `result` (shape varies by feature; generation results always have `images`, `expected_count`, `success_count`; `clothing-suite` also has `grouped_images`/`failed_labels`; analyze results include the AI's structured findings plus `remaining_free_quota`, `daily_free_quota`, `is_free`).\n- `status: \"error\"` — includes a user-facing `error` string (never a raw provider error).\n\n## Video Inference Submit Parameters\n\n| Parameter | Type | Required | Default | Range |\n|-----------|------|----------|---------|-------|\n| `prompt` | string | Yes | — | max 2000 chars |\n| `model` | string | No | `wan/v2.2-a14b/text-to-video` | See supported models |\n| `width` | int | No | 854 | See video resolutions |\n| `height` | int | No | 480 | See video resolutions |\n| `num_frames` | int | No | 81 | Auto-aligned to 1+4k |\n| `frames_per_second` | int | No | 16 | 4–60 |\n| `num_inference_steps` | int | No | 4 | 4–40 |\n| `guidance_scale` | float | No | 1.0 | 1.0–10.0 |\n| `seed` | int | No | random | — |\n| `negative_prompt` | string | No | \"\" | — |\n| `image_url` | string | No | — | S3 key from upload-image (required for I2V) |\n| `lora_id` | string | No | — | Single LoRA |\n| `lora_scale` | float | No | 1.0 | 0.0–1.0 |\n| `loras` | array | No | — | Multiple LoRAs: `[{\"lora_id\":\"...\",\"lora_scale\":1.0}]` |\n\n## S2V Inference Submit Parameters\n\nUses the same `/api/v1/inference/submit` endpoint. Set `model` to `wan/v2.2-a14b/speech-to-video`.\n\n| Parameter | Type | Required | Default | Range |\n|-----------|------|----------|---------|-------|\n| `prompt` | string | Yes | — | max 2000 chars |\n| `model` | string | Yes | — | `wan/v2.2-a14b/speech-to-video` |\n| `image_url` | string | Yes | — | S3 key from upload-image (reference face image) |\n| `audio_url` | string | Yes | — | S3 key from `/api/v1/storage/image/upload` (accepts mp3, wav, flac, aac, ogg, m4a) |\n| `width` | int | No | 854 | See S2V/Animate resolutions |\n| `height` | int | No | 480 | See S2V/Animate resolutions |\n| `num_frames` | int | No | 81 | Auto-aligned to 1+4k |\n| `frames_per_second` | int | No | 16 | 4–60 |\n| `num_inference_steps` | int | No | 4 | 4–40 |\n| `guidance_scale` | float | No | 1.0 | 1.0–10.0 |\n| `seed` | int | No | random | — |\n| `negative_prompt` | string | No | \"\" | — |\n\n## Animate Inference Submit Parameters\n\nUses the same `/api/v1/inference/submit` endpoint. Set `model` to `wan/v2.2-a14b/animate`.\n\n| Parameter | Type | Required | Default | Range |\n|-----------|------|----------|---------|-------|\n| `prompt` | string | Yes | — | max 2000 chars |\n| `model` | string | Yes | — | `wan/v2.2-a14b/animate` |\n| `image_url` | string | Yes | — | S3 key from upload-image (reference character image) |\n| `video_url` | string | Yes | — | URL to motion/pose reference video |\n| `width` | int | No | 854 | See S2V/Animate resolutions |\n| `height` | int | No | 480 | See S2V/Animate resolutions |\n| `num_frames` | int | No | 81 | Auto-aligned to 1+4k |\n| `frames_per_second` | int | No | 16 | 4–60 |\n| `num_inference_steps` | int | No | 4 | 4–40 |\n| `guidance_scale` | float | No | 1.0 | 1.0–10.0 |\n| `seed` | int | No | random | — |\n| `negative_prompt` | string | No | \"\" | — |\n\n## Image Inference Submit Parameters\n\nUses the same `/api/v1/inference/submit` endpoint. Set `model` to an image model ID.\n\n| Parameter | Type | Required | Default | Range |\n|-----------|------|----------|---------|-------|\n| `prompt` | string | Yes | — | max 2000 chars |\n| `model` | string | Yes | — | Must be an image model ID (see below) |\n| `width` | int | No | 1024 | See image resolutions |\n| `height` | int | No | 1024 | See image resolutions |\n| `num_images` | int | No | 1 | 1–4 |\n| `num_inference_steps` | int | No | varies | 1–8 (z-image turbo), 1–40 (qwen-image) |\n| `guidance_scale` | float | No | varies | 1.0–20.0 |\n| `seed` | int | No | random | — |\n| `negative_prompt` | string | No | \"\" | — |\n| `image_url` | string | No | — | S3 key from upload-image (required for I2I) |\n| `strength` | float | No | — | 0.0–1.0 (I2I only: transformation strength) |\n| `output_format` | string | No | \"png\" | \"png\" or \"jpeg\" |\n| `lora_id` | string | No | — | Single LoRA |\n| `lora_scale` | float | No | 1.0 | 0.0–1.0 |\n| `loras` | array | No | — | Multiple LoRAs: `[{\"lora_id\":\"...\",\"lora_scale\":1.0}]` |\n\n### Wan Video Model IDs\n\n| Model ID | Description |\n|----------|-------------|\n| `wan/v2.2-a14b/text-to-video` | Wan 2.2 T2V 14B — standard quality (`num_inference_steps`, `guidance_scale` apply) |\n| `wan/v2.2-a14b/text-to-video/turbo` | Wan 2.2 T2V turbo — fixed internal defaults, ~3x cheaper per frame |\n| `wan/v2.2-a14b/image-to-video` | Wan 2.2 I2V 14B — standard quality |\n| `wan/v2.2-a14b/image-to-video/turbo` | Wan 2.2 I2V turbo |\n\n> Turbo variants ignore `num_inference_steps` / `guidance_scale`. Standard mode caps\n> 1080p at 81 frames; turbo allows 153.\n\n### LTX-Video Model IDs\n\n| Model ID | Type | Required inputs |\n|----------|------|-----------------|\n| `ltx-video/v2.3/image-audio-to-video` | Image(+Audio)-to-Video | `prompt`, `image_url`; `audio_url` optional |\n\nFrame-based. `resolution_tier` + `aspect_ratio` (`480p`/`512p`/`720p`/`1080p` × `16:9`/`9:16`/`1:1`)\nor explicit `width`/`height` from: 854×480, 1280×720, 1920×1080, 480×854, 720×1280, 1080×1920,\n480×480, 720×720, 1080×1080. Default 1280×720, fps 24, duration 5s\n(`num_frames` = `duration × fps + 1`). Max frames: 480p 481, 720p 241, 1080p 121 —\n**not** subject to the Wan `1+4k` alignment.\n\nPricing per frame: 480p `$0.0008`, 512p `$0.0009`, 720p and above `$0.0012`.\n\n### MiniMax H3 Model IDs\n\nDuration-based (ignores `num_frames` / `frames_per_second`; output is fixed 24fps).\n\n| Model ID | Type | Required inputs |\n|----------|------|-----------------|\n| `minimax/h3/text-to-video` | Text-to-Video | `prompt` |\n| `minimax/h3/image-to-video` | Image-to-Video | `prompt`, `image_url` (optional `end_image_url`) |\n| `minimax/h3/reference-to-video` | Reference-to-Video | `prompt`, plus at least one of `reference_image_urls` / `reference_video_urls` |\n\n| Parameter | Default | Allowed values |\n|-----------|---------|----------------|\n| `resolution_tier` | `480p` | `480p`, `768p` |\n| `aspect_ratio` | `16:9` | `16:9`, `4:3`, `1:1`, `3:4`, `9:16` |\n| `duration` | `5` | 4–15 seconds |\n| `reference_audio_urls` | — | array; Ref2VA only |\n| `reference_image_labels` | — | array, same order as `reference_image_urls`; what each picture shows (e.g. `[\"Shen Yu - character\", \"A Man - character\", \"coop - scene\"]`) |\n| `reference_video_labels` | — | array, same order as `reference_video_urls` |\n| `reference_audio_labels` | — | array, same order as `reference_audio_urls` |\n| `use_ref_video_audio` | `false` | also use each reference video's own soundtrack |\n\nFrame sizes: 480p → 832×480 (16:9), 640×480 (4:3), 480×480 (1:1), 480×640 (3:4), 480×832 (9:16).\n768p → 1344×768, 1024×768, 768×768, 768×1024, 768×1344.\n\n**Naming your references.** Uploaded files are stored under generated ids, so a prompt\nthat says \"Shen Yu reaches into the coop\" cannot be matched to a picture by name on its own.\nPass `reference_image_labels` alongside `reference_image_urls` (same order) and the platform\nbinds each name in your prompt to the right picture while building the model prompt.\n**Write your prompt normally** — do not rewrite it to mention picture numbers.\n\n**Ref2VA reference caps** — differ per tier and per variant, and the image cap is\nhigher when sending images only:\n\n`minimax/h3/reference-to-video`\n\n| Limit | 480p | 768p |\n|-------|------|------|\n| Reference images (with videos/audio) | 5 | 5 |\n| Reference images (images only) | 9 | 9 |\n| Reference videos | 3 | 3 |\n| Reference audios | 3 | 3 |\n\nTotal reference video length 15.1s (all videos combined), reference video FPS ≤ 24,\nreference audio ≤ 10s each, reference image longest edge ≤ 2048px, aspect ratio ≤ 4.0.\n\n### Audio Model IDs\n\n| Model ID | Description |\n|----------|-------------|\n| `qwen3-tts/text-to-speech/1.7b` | Qwen3-TTS CustomVoice — takes `text` (≤500 chars), not `prompt` |\n\nParameters: `text` (required), `speaker` (default `Sohee`), `language` (default `Chinese`),\n`seed`, `temperature` (0.0–2.0, default 0.9), `top_p` (default 1.0), `top_k` (1–200, default 50),\n`repetition_penalty` (1.0–2.0, default 1.05).\n\n### S2V / Animate Model IDs\n\n| Model ID | Description |\n|----------|-------------|\n| `wan/v2.2-a14b/speech-to-video` | Wan 2.2 Speech-to-Video 14B (lip-sync from audio + face image) |\n| `wan/v2.2-a14b/animate` | Wan 2.2 Animate 14B (motion transfer from video + character image) |\n\n### Image Model IDs\n\n| Model ID | Description |\n|----------|-------------|\n| `openai/gpt-image-2/text-to-image` | GPT Image 2 Text-to-Image — 5 sizes: **1024×1024**, **1920×1072** / **1072×1920** (1080p), **2560×1440** / **1440×2560** (2K). Always 1 image; ignores `num_inference_steps` / `guidance_scale` / `output_format` |\n| `flux2/dev/text-to-image` | FLUX.2-dev Text-to-Image — own resolution whitelist, always 1 image |\n| `flux2/dev/image-edit` | FLUX.2-dev Image Edit |\n| `qwen-image/v2512/text-to-image` | Qwen Image 2512 Text-to-Image |\n| `qwen-image/v2512/text-to-image/lora` | Qwen Image 2512 T2I with LoRA |\n| `qwen-image/v2511/image-edit` | Qwen Image Edit 2511 — multi-image reference editing |\n| `z-image/turbo/text-to-image` | Z-Image Turbo Text-to-Image |\n| `z-image/turbo/text-to-image/lora` | Z-Image Turbo T2I with LoRA |\n| `z-image/turbo/image-to-image` | Z-Image Turbo Image-to-Image |\n| `z-image/turbo/image-to-image/lora` | Z-Image Turbo I2I with LoRA |\n\n## Supported Video Resolutions\n\n| Preset | Width | Height | Max Frames (turbo) | Max Frames (standard) |\n|--------|-------|--------|-------------------|----------------------|\n| 480p landscape | 854 | 480 | 161 | 161 |\n| 480p portrait | 480 | 854 | 161 | 161 |\n| 720p landscape | 1280 | 720 | 161 | 161 |\n| 720p portrait | 720 | 1280 | 161 | 161 |\n| 1080p landscape | 1920 | 1080 | 153 | 81 |\n| 1080p portrait | 1080 | 1920 | 153 | 81 |\n\nOnly these exact pairs are accepted. Frame alignment: `valid_frames = 1 + 4*k` where `k >= 1`.\n\n> Standard (non-turbo) model: 1080p max is 81 frames due to generation time constraints.\n\n## Supported S2V / Animate Resolutions\n\n| Preset | Width | Height | Max Frames |\n|--------|-------|--------|------------|\n| 480p landscape | 854 | 480 | 161 |\n| 480p portrait | 480 | 854 | 161 |\n| 512p square | 512 | 512 | 161 |\n| 720p landscape | 1280 | 720 | 161 |\n| 720p portrait | 720 | 1280 | 161 |\n\nOnly these exact pairs are accepted. Frame alignment: `valid_frames = 1 + 4*k` where `k >= 1`.\n\n## Supported Image Resolutions\n\n| Preset | Width | Height |\n|--------|-------|--------|\n| Square small | 512 | 512 |\n| Square | 1024 | 1024 |\n| Landscape | 1024 | 768 |\n| Portrait | 768 | 1024 |\n| Wide landscape | 1280 | 768 |\n| Tall portrait | 768 | 1280 |\n\nOnly these exact pairs are accepted.\n\n## Pricing\n\n> Video/image inference pricing below is a static reference and can drift — verify against support before relying on it for billing-sensitive integrations. Image Studio has its own pricing surface with a live endpoint (`GET /api/v1/image-studio/pricing`) — always prefer that over any static table for Image Studio.\n\n### Video Inference (10 credits = $1 USD)\n\n| Resolution | Per-Frame (USD) | Per-Frame (Credits) |\n|-----------|----------------|-------------------|\n| 480p | $0.0009375 | 0.009375 |\n| 720p | $0.001875 | 0.01875 |\n| 1080p | $0.0025 | 0.025 |\n\nLoRA multiplier: 1.2x (applied when any LoRA is specified).\n\n### MiniMax H3 (10 credits = $1 USD, per output second)\n\nPer output second, regardless of clip length:\n\n| Model | 480p USD/sec | 768p USD/sec | 480p credits/sec | 768p credits/sec |\n|-------|--------------|--------------|------------------|------------------|\n| `text-to-video`, `image-to-video` | $0.0045 | $0.012 | 0.045 | 0.12 |\n| `reference-to-video` | $0.0045 | $0.012 | 0.045 | 0.12 |\n\n### MiniMax H3 Ref2VA inputs (10 credits = $1 USD)\n\n```\ntotal = duration x base_rate\n      + max(0, reference_images - 5) x image_rate   # first 5 images free\n      + reference_video_seconds x base_rate         # reference audio is free\n```\n\n| Model | Resolution | Base USD/sec | Reference image | Reference audio | Reference video USD/sec |\n|-------|-----------|--------------|-----------------|-----------------|-------------------------|\n| `reference-to-video` | 480p | $0.0045 | first 5 free, then $0.0075 each | free | $0.0045 |\n| `reference-to-video` | 768p | $0.012 | first 5 free, then $0.0075 each | free | $0.012 |\n\n`GET /api/v1/pricing/config` publishes the live rates (incl. `reference_image_free_count`).\n\n### S2V Inference (10 credits = $1 USD)\n\n| Resolution | Per-Frame (USD) | Per-Frame (Credits) |\n|-----------|----------------|-------------------|\n| 480p | $0.0009375 | 0.009375 |\n| 512p | $0.0013125 | 0.013125 |\n| 720p | $0.001875 | 0.01875 |\n\n### Animate Inference (10 credits = $1 USD)\n\n| Resolution | Per-Frame (USD) | Per-Frame (Credits) |\n|-----------|----------------|-------------------|\n| 480p | $0.00125 | 0.0125 |\n| 512p | $0.00175 | 0.0175 |\n| 720p | $0.0025 | 0.025 |\n\n### Image Inference (10 credits = $1 USD)\n\n| Model | Per Image (USD) | Per Image (Credits) |\n|-------|----------------|-------------------|\n| GPT Image 2 T2I | $0.03 | 0.3 |\n| FLUX.2-dev T2I | $0.006 | 0.06 |\n| FLUX.2-dev Image Edit | $0.012 | 0.12 |\n| qwen-image T2I | $0.015 | 0.15 |\n| qwen-image T2I + LoRA | $0.018 | 0.18 |\n| qwen-image-edit | $0.003 | 0.03 |\n| z-image turbo | $0.0025 | 0.025 |\n| z-image turbo + LoRA | $0.003 | 0.03 |\n\nTotal cost = per-image price x `num_images` (1-4). LoRA pricing is built into the model-specific rate (no separate multiplier).\n\n### Audio (10 credits = $1 USD)\n\n| Item | Cost |\n|------|------|\n| Qwen3-TTS | $0.00003 per character |\n| Minimum charge | $0.003 per request |\n\n### Image Studio Pricing\n\nImage Studio (see below) charges a **flat per-image rate for every generation endpoint**, and a **freemium daily-quota rate for analyze endpoints**. Both numbers are computed live server-side — call `GET /api/v1/image-studio/pricing` (or `phosor_client.py studio-pricing`) to get the current `per_image_credits`, `per_analyze_credits`, and `analyze_daily_free_quota`. Partial success (e.g. 3 of 5 images generated) bills only the successes; the rest is auto-refunded.\n\n## Concurrency Limits\n\nModel API inference jobs run concurrently up to a per-account cap; going over returns `429`.\n\n| Tier | Plan | Concurrent Model API jobs |\n|------|------|---------------------------|\n| Free | — | 1 |\n| 1 | Starter | 2 |\n| 2 | Standard | 4 |\n| 3 | Pro | 8 |\n\n> Image Studio runs on its own pool and does not consume this quota.\n\n## Limits & Quotas\n\n| Resource | Limit |\n|----------|-------|\n| Rate limit | 1000 requests / 60 seconds per API key |\n| Concurrent jobs | Tier-based: Starter=1, Standard=5, Pro=20 |\n| Max API keys per user | 10 |\n| Max LoRAs per user (total) | 20 |\n| LoRA file format | `.safetensors` only (video: two files high_noise + low_noise; image: single file) |\n| Max LoRA file size | 2048 MB |\n| Uploaded LoRA expiry | 1 day (auto-cleaned) |\n| Saved LoRA expiry | 7 days from save (auto-cleaned) |\n| Max image file size | 20 MB |\n| Max video file size | 50 MB |\n| Image formats | JPEG, PNG, WebP |\n| Queue timeout | 3000s (auto-refund) |\n| Execution timeout | 2400s inference (auto-refund) |\n\n## Response Formats\n\n### Models Response\n```json\n{\"models\": [{\"model_id\": \"wan/v2.2-a14b/text-to-video\", \"description\": \"Wan 2.2 Text-to-Video 14B\", \"model_type\": \"video\", \"model_mode\": \"text-to-video\", \"metadata\": {}}]}\n```\n\n### Submit Response\n```json\n{\"request_id\": \"<request_id>\", \"status\": \"queued\"}\n```\n\n### Status Response\n```json\n{\"request_id\": \"...\", \"status\": \"queued|processing|completed|failed\", \"progress\": 0-100}\n```\n\n### Result Response (Video)\n```json\n{\"data\": {\"video\": {\"url\": \"...\"}, \"seed\": 12345}, \"request_id\": \"...\"}\n```\n\n### Result Response (Image)\n```json\n{\"data\": {\"image\": {\"url\": \"...\"}, \"images\": [{\"url\": \"...\"}, ...], \"seed\": 12345}, \"request_id\": \"...\"}\n```\n\n`data.images` is an array of all generated images. `data.image` is the first image (backward-compatible).\n\n### Error Codes\n| Code | Meaning |\n|------|---------|\n| 400 | Validation error |\n| 401 | Invalid API key or email not verified |\n| 402 | Insufficient credits |\n| 404 | Job not found |\n| 429 | Rate limit or concurrency limit exceeded |\n| 503 | Queue at capacity |\n\nFile v1.3.2:CHANGELOG.md\n\n# Changelog\n\nSkill package versions. The API contract version is separate and lives in\n`SKILL.md` as `metadata.api_version` — a skill release does not move it.\n\n## 1.3.2 — 2026-09-25 (API v1.2.2)\n\n`minimax/h3/reference-to-video/turbo` removed\n\n- Requests to `minimax/h3/reference-to-video/turbo` now return an unknown-model error\n- Use `minimax/h3/reference-to-video` instead — same inputs and pricing structure\n- `reference-to-video` (non-turbo) is unaffected\n- API contract version v1.2.2: a model id was removed; no other endpoints or parameters changed\n\n## 1.3.1 — 2026-09-24 (API v1.2.1)\n\nReference-to-Video limits and pricing\n\n- Reference caps, both 480p and 768p: 9 images when sending only images, 5 images when\n  videos or audio are also present, 3 reference videos, 3 reference audios\n- Total reference video length is 15.1s across all reference videos\n- `reference-to-video` now uses the same rates as the rest of H3: 480p $0.0045/s,\n  768p $0.012/s, reference images $0.0075 each after the first 5\n- Bundled client validation and version string updated to match\n- API contract version v1.2.1: Reference-to-Video limits and pricing changed; no new endpoints or parameters\n\n## 1.3.0 — 2026-09-22 (API v1.2.0)\n\nReference metadata and H3 limits\n\n- Reference labels are forwarded in the same order as image, video, and audio URLs\n- 768p Ref2VA supports the same 9 image-only, 4 mixed-image, and 3 video caps as 480p\n- Correct the bundled client's stale 768p validation and 15-second video budget\n\n## 1.2.0 — 2026-09-13 (API v1.2.0)\n\nReference-to-Video\n\n- Total reference video length raised from 10s to 15s, across all reference\n  videos combined. `reference-to-video/turbo` is unchanged at 10s.\n- No other caps moved, and pricing is unchanged.\n\n## 1.2.0 — 2026-09-10\n\nPricing\n\n- MiniMax H3 (T2V / I2V / Ref2VA / Ref2VA Turbo) now shares one per-output-second\n  rate regardless of clip length: 480p $0.0045/s, 768p $0.012/s (previously 0.02 / 0.04 USD\n  for T2V/I2V and 0.025 / 0.063 USD for Ref2VA)\n- Ref2VA reference inputs: the first 5 reference images are free, then $0.0075 each\n  (previously 0.01 USD from the first image); reference audio is free (previously 0.01 USD);\n  reference video stays at the output tier rate\n- `GET /api/v1/pricing/config` adds `reference_image_free_count`; the built-in\n  client's price table and estimate formula follow it\n\nReference-to-Video Turbo\n\n- `minimax/h3/reference-to-video/turbo` added to the model table with its own reference\n  caps and 768p usage guidance (stay at or below 3 images, or 2 images plus 1 video)\n\nClient and docs\n\n- Environment section and `--base-url` help use placeholders instead of a specific host\n- `studio-layouts` on a self-hosted deployment reads `PHOSOR_WEB_BASE_URL` for the web\n  front instead of guessing ports\n- Code comments in `phosor_client.py` are English throughout\n\n## 1.1.0 — 2026-09-03\n\nNew models\n\n- MiniMax H3: text-to-video, image-to-video with an optional closing frame,\n  and Reference-to-Video, which takes reference images, videos and audio in\n  one request. 480p and 768p, 4-15 seconds, billed per output second\n- GPT Image 2 text-to-image, at five sizes: 1024x1024, 1920x1072 / 1072x1920\n  and 2560x1440 / 1440x2560\n- FLUX.2-dev text-to-image and image edit\n- Z-Image Turbo text-to-image and image-to-image, with LoRA\n- Qwen3-TTS text-to-speech, billed per character, minimum $0.003 per request\n\nNew surface\n\n- Image Studio, for e-commerce product and model photography: product suites,\n  clothing and on-model suites, selling-point layouts, A+ content, white\n  background, scene variation, and one-off tools for background removal,\n  inpainting and image translation. Async throughout: POST returns a\n  request_id, then poll until done\n\nClient\n\n- Advertises all five GPT Image 2 sizes; the previous release listed only\n  1024x1024 and rejected the other four valid sizes\n- `--version` now prints the real package version\n\nPackaging\n\n- The package and its zip are named phosor-ai-skills, matching the folder it\n  unpacks into and this skill's slug\n- Two version numbers, on purpose: the skill version covers this package\n  (commands, docs, bundled client), `metadata.api_version` covers the gateway\n  contract. Either can move without forcing the other\n- A stray compiled .pyc no longer rides along inside the archive\n\nDocs\n\n- `references/api.md` covers every endpoint the skill calls, with limits and\n  live pricing pointers. Prices are never hardcoded — call the pricing\n  endpoints for current numbers\n\n## 1.0.2 / 1.0.1 — 2026-05\n\nPublished without changelog entries; both carried the 1.0.0 text. Recorded\nhere so the gap is visible rather than implied.\n\n## 1.0.0 — 2026-03-23 — Initial release\n\n- AI content generation platform, supporting Wan 2.2 14B text-to-video and\n  image-to-video\n- Upload and use custom LoRA models for style customization\n- 16 CLI commands for job submission, uploads, status, results, and model\n  listing\n- Supports preset resolutions (480p/720p/1080p), frame alignment rules, and\n  usage quotas\n\nFile v1.3.2:skill-card.md\n\n## Description:\n\nHelps agents generate videos, images, speech, and e-commerce photography with Phosor AI, and manage the resulting jobs and custom LoRA models.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[phosor.ai](https://clawhub.ai/user/phosor.ai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nCreators, e-commerce teams, and developers use this skill to generate and edit media, create product and model photography, work with custom LoRA models, and check generation jobs through Phosor AI.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Prompts, selected media, LoRA files, and generation metadata may be sent to Phosor AI.\n\nMitigation: Use only approved content and confirm what will be uploaded before running commands.\n\nRisk: Generation commands and broader Image Studio calls can consume account credits.\n\nMitigation: Review the requested operation and expected costs before submitting jobs.\n\nRisk: The client may use a PHOSOR_API_KEY from the OpenClaw workspace when one is not provided directly.\n\nMitigation: Confirm which account key will be used and keep it private.\n\n## Reference(s):\n\n- [Phosor AI API reference](references/api.md)\n- [Phosor AI documentation](https://docs.phosor.ai)\n- [Phosor AI](https://phosor.ai)\n- [Phosor AI skill on ClawHub](https://clawhub.ai/phosor.ai/skills/phosor-ai-skills)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Shell commands, JSON, Guidance]\n\n**Output Format:** [Markdown guidance and shell commands; client responses in JSON]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Generation results may include media URLs and job identifiers.]\n\n## Skill Version(s):\n\n1.3.2 (source: frontmatter, VERSION, CHANGELOG; released 2026-09-25)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.3.1: 8 files, 41056 bytes\n\nFiles: _meta.json (135b), CHANGELOG.md (4648b), README.md (1986b), references/api.md (26236b), scripts/phosor_client.py (92623b), skill-card.md (2647b), SKILL.md (24968b), VERSION (6b)\n\nFile v1.3.1:SKILL.md\n\n---\nname: phosor-ai-skills\ndescription: Generate AI videos, images and speech (text-to-video, image-to-video, reference-to-video, speech-to-video, animate, text-to-image, image-to-image, image edit, text-to-speech), bring your own LoRA models, and generate AI product/model photography for e-commerce (Image Studio) via the Phosor AI platform. Use when the user wants to create videos or images from text prompts, animate images, generate lip-synced video from audio, synthesize speech from text, generate images with a custom LoRA, generate product photography or model/clothing photography for e-commerce listings, or manage generation jobs.\nlicense: MIT-0\ncompatibility: Requires Python 3.7+ and network access to phosor.ai\nmetadata:\n  author: phosor.ai\n  version: \"1.3.1\"\n  api_version: \"v1.2.1\"\n  homepage: https://phosor.ai\n---\n\n# Phosor AI\n\nGenerate AI videos and images (text-to-video, image-to-video, speech-to-video, animate, text-to-image, image-to-image), bring your own LoRA models, and generate AI product/model photography for e-commerce (Image Studio) via the Phosor AI platform.\n\nFor detailed API endpoints, parameters, pricing, and limits, see [references/api.md](references/api.md).\n\n## Setup\n\nSet your API key:\n\n```bash\nexport PHOSOR_API_KEY=\"your-api-key-here\"\n```\n\nGet an API key at [phosor.ai](https://phosor.ai) → Settings → API Keys.\n\nThe CLI script is at `scripts/phosor_client.py`. All commands output JSON to stdout.\n\n## Base URL\n\nThe client talks to `https://phosor.ai` by default, over HTTPS. Get your key from phosor.ai → Settings → API Keys.\n\n```bash\npython3 scripts/phosor_client.py --api-key <your-key> check-key\n```\n\nTo point the same client at another Phosor endpoint, pass `--base-url <url>` (or set `PHOSOR_BASE_URL`).\nHTTPS is required; plain `http://` is accepted only for `localhost`, or when you explicitly add `--allow-http`.\n\nNote: `studio-analyze` / `studio-suite` `--image-url` must be the **full https S3 URL** returned by `upload-image`, not the bare S3 key path — a bare key errors with `unsupported URL scheme`.\n\n## Quick Start\n\n### MiniMax H3 — Text-to-Video\n\nH3 is **duration-based**, not frame-based: it ignores `--num-frames` / `--fps` (output is\nalways 24fps) and bills per output second. Pick the frame size with\n`--resolution-tier` + `--aspect` instead of `--width/--height`.\n\n```bash\npython3 scripts/phosor_client.py submit \"A cat walking on a beach at sunset\" \\\n  --model minimax/h3/text-to-video \\\n  --resolution-tier 768p --aspect 16:9 --duration 5\n```\n\n### MiniMax H3 — Image-to-Video\n\n```bash\n# Upload first (direct URLs are not accepted); then submit with the returned s3_key\npython3 scripts/phosor_client.py upload-image /path/to/first-frame.jpg\n\npython3 scripts/phosor_client.py submit \"The person starts dancing\" \\\n  --model minimax/h3/image-to-video \\\n  --image-url \"images/img-xxx.jpg\" \\\n  --end-image-url \"images/img-yyy.jpg\" \\\n  --resolution-tier 480p --aspect 9:16 --duration 6\n```\n\n`--end-image-url` is optional and pins the closing frame.\n\n### MiniMax H3 — Reference-to-Video (Ref2VA)\n\nFeed reference **images**, **videos**, and **audio** together; refer to them positionally in\nthe prompt as `<Picture 1>`, `<Picture 2>`, … At least one of `--reference-image-urls` /\n`--reference-video-urls` is required.\n\n```bash\npython3 scripts/phosor_client.py submit \\\n  \"Use <Picture 1> and <Picture 2> as sequential keyframes; slow push-in, cinematic 35mm look.\" \\\n  --model minimax/h3/reference-to-video \\\n  --reference-image-urls \"images/a.jpg,images/b.jpg\" \\\n  --reference-audio-urls \"audio/voice.mp3\" \\\n  --resolution-tier 768p --aspect 16:9 --duration 5\n```\n\nReference inputs are billed on top of the output — see **Ref2VA Pricing** below, and\n**Ref2VA Reference Limits** for the caps (the same at 480p and 768p).\n\n### Text-to-Video (Wan)\n\nWan is frame-based. Add `/turbo` to the model id for the fast, ~3x cheaper variant\n(it ignores `--steps` / `--guidance`).\n\n```bash\n# Submit T2V job (480p, 81 frames, 16fps)\npython3 scripts/phosor_client.py submit \"A cat walking on a beach at sunset\" \\\n  --width 854 --height 480 --num-frames 81 --fps 16\n\n# Check status\npython3 scripts/phosor_client.py status <request_id>\n\n# Get result (video URL)\npython3 scripts/phosor_client.py result <request_id>\n```\n\n### Image-to-Video\n\n**Two-step flow**: upload image first, then submit with the returned S3 key.\n\n```bash\n# Step 1: Upload image\npython3 scripts/phosor_client.py upload-image /path/to/photo.jpg\n# Returns: {\"file_id\": \"img-xxx\", \"s3_key\": \"images/img-xxx.jpg\", ...}\n\n# Step 2: Submit I2V job using the s3_key as image_url\npython3 scripts/phosor_client.py submit \"The person in the photo starts dancing\" \\\n  --image-url \"images/img-xxx.jpg\" --width 854 --height 480\n```\n\n### Text-to-Image\n\n```bash\n# Submit T2I job (1024x1024, default settings)\npython3 scripts/phosor_client.py submit \"A futuristic city skyline at dusk\" \\\n  --model qwen-image/v2512/text-to-image --width 1024 --height 1024\n\n# Generate multiple images at once (1-4)\npython3 scripts/phosor_client.py submit \"A futuristic city skyline at dusk\" \\\n  --model z-image/turbo/text-to-image --width 1024 --height 768 --num-images 4\n\n# Check status and get result (image URL)\npython3 scripts/phosor_client.py status <request_id>\npython3 scripts/phosor_client.py result <request_id>\n# Returns: {\"data\": {\"image\": {\"url\": \"...\"}, \"seed\": 12345}, ...}\n```\n\n### Image-to-Image\n\n**Two-step flow**: upload source image first, then submit with the returned S3 key.\n\n```bash\n# Step 1: Upload source image\npython3 scripts/phosor_client.py upload-image /path/to/photo.jpg\n# Returns: {\"file_id\": \"img-xxx\", \"s3_key\": \"images/img-xxx.jpg\", ...}\n\n# Step 2: Submit I2I job using the s3_key as image_url\npython3 scripts/phosor_client.py submit \"Transform into oil painting style\" \\\n  --model z-image/turbo/image-to-image --image-url \"images/img-xxx.jpg\" \\\n  --width 1024 --height 1024 --strength 0.7\n```\n\n### Image Edit (Multi-image Reference)\n\n**Two-step flow**: upload reference images first, then submit with S3 keys as `image_urls`.\n\n```bash\n# Step 1: Upload reference images (up to 3)\npython3 scripts/phosor_client.py upload-image /path/to/ref1.jpg\npython3 scripts/phosor_client.py upload-image /path/to/ref2.jpg\n\n# Step 2: Submit image-edit job\npython3 scripts/phosor_client.py submit \\\n  \"The girl in image 1 is wearing the outfit from image 2\" \\\n  --model qwen-image/v2511/image-edit \\\n  --image-urls '[\"images/img-ref1.jpg\",\"images/img-ref2.jpg\"]' \\\n  --width 1024 --height 1024\n\n# Turbo variant (faster, Lightning LoRA built-in)\npython3 scripts/phosor_client.py submit \\\n  \"The girl in image 1 is wearing the outfit from image 2\" \\\n  --model qwen-image/v2511/image-edit \\\n  --image-urls '[\"images/img-ref1.jpg\",\"images/img-ref2.jpg\"]'\n```\n\n### Speech-to-Video (S2V)\n\n**Two-step flow**: upload both audio and reference image first, then submit with the returned S3 keys.\n\n```bash\n# Step 1: Upload reference image\npython3 scripts/phosor_client.py upload-image /path/to/face.jpg\n# Returns: {\"file_id\": \"img-xxx\", \"s3_key\": \"images/img-xxx.jpg\", ...}\n\n# Step 2: Submit S2V job using the s3_key as image_url and audio URL as audio_url\npython3 scripts/phosor_client.py submit \"A person speaking naturally\" \\\n  --model wan/v2.2-a14b/speech-to-video \\\n  --image-url \"images/img-xxx.jpg\" --audio-url \"https://example.com/speech.wav\" \\\n  --width 854 --height 480\n```\n\n### Animate\n\n**Two-step flow**: upload both source video and reference image first, then submit.\n\n```bash\n# Step 1: Upload reference image\npython3 scripts/phosor_client.py upload-image /path/to/character.jpg\n# Returns: {\"file_id\": \"img-xxx\", \"s3_key\": \"images/img-xxx.jpg\", ...}\n\n# Step 2: Submit Animate job using the s3_key as image_url and video URL as video_url\npython3 scripts/phosor_client.py submit \"The character performs the dance moves\" \\\n  --model wan/v2.2-a14b/animate \\\n  --image-url \"images/img-xxx.jpg\" --video-url \"https://example.com/dance.mp4\" \\\n  --width 854 --height 480\n```\n\n### Text-to-Image (GPT Image 2)\n\nIts own resolution set and **always exactly 1 image** (`--num-images` is ignored).\n\n```bash\npython3 scripts/phosor_client.py submit \"A ceramic mug on a linen cloth, soft window light\" \\\n  --model openai/gpt-image-2/text-to-image --width 1024 --height 1024\n```\n\nAllowed sizes: **1024×1024 only**. Any other size is rejected with `400 Invalid parameters`.\nIt ignores `--num-images`, `--steps` and `--guidance` — the only parameters it accepts are\nprompt, model, width, height and seed.\n\n### Text-to-Image (FLUX.2-dev)\n\nFLUX.2-dev has its own resolution whitelist and is **fixed at 1 image per request**\n(`--num-images` and `--steps` are ignored).\n\n```bash\npython3 scripts/phosor_client.py submit \"Editorial product photo, soft window light\" \\\n  --model flux2/dev/text-to-image --width 2048 --height 1536\n```\n\n### Image Edit (FLUX.2-dev)\n\n```bash\npython3 scripts/phosor_client.py upload-image /path/to/source.jpg\n\npython3 scripts/phosor_client.py submit \"Replace the background with a marble surface\" \\\n  --model flux2/dev/image-edit --image-url \"images/img-xxx.jpg\" \\\n  --width 1024 --height 1024\n```\n\n### Text-to-Speech\n\nTTS takes `text` (not a prompt) and is billed per character with a minimum charge.\n\n```bash\npython3 scripts/phosor_client.py submit-tts \"Hello, welcome to Phosor AI.\" \\\n  --speaker Sohee --language English\n```\n\n### LoRA Upload (Custom Pre-trained)\n\n**Video LoRA** requires two .safetensors files (high_noise + low_noise). **Image LoRA** requires a single .safetensors file.\n\n```bash\n# Video LoRA: upload two .safetensors files\npython3 scripts/phosor_client.py upload-lora high_noise.safetensors low_noise.safetensors --name \"My Style\"\n\n# Image LoRA: import single .safetensors file via URL\npython3 scripts/phosor_client.py import-lora \\\n  \"https://example.com/my_lora.safetensors\" \\\n  --name \"My Image Style\"\n\n# Video LoRA: import two files via URL\npython3 scripts/phosor_client.py import-lora \\\n  \"https://example.com/high_noise.safetensors\" \\\n  \"https://example.com/low_noise.safetensors\" \\\n  --name \"My Video Style\"\n\n# Check status, then use\npython3 scripts/phosor_client.py lora-status <lora_id>\npython3 scripts/phosor_client.py submit \"A person walking\" --lora-id <lora_id>\n```\n\n## CLI Commands\n\n| Command | Description | Key Arguments |\n|---------|-------------|---------------|\n| `check-key` | Validate API key | — |\n| `submit` | Submit inference job (T2V/I2V/S2V/Animate/T2I/I2I) | `prompt`, `--width`, `--height`, `--num-frames`, `--fps`, `--steps`, `--guidance`, `--image-url`, `--audio-url`, `--video-url`, `--lora-id`, `--lora-scale`, `--loras`, `--seed`, `--negative-prompt`, `--model`, `--num-images`, `--strength`, `--output-format` |\n| `status` | Get job status | `request_id` |\n| `result` | Get job result (video or image URL) | `request_id` |\n| `poll` | Poll all pending jobs | — |\n| `list` | List locally tracked pending jobs | — |\n| `history` | Get job history | `--limit` |\n| `upload-image` | Upload image for I2V or I2I | `file` |\n| `import-image` | Import image from URL | `url`, `--filename` |\n| `upload-lora` | Upload LoRA (two .safetensors for video) | `high_noise_file`, `low_noise_file`, `--name` |\n| `import-lora` | Import LoRA from URLs (one or two files) | `high_noise_url`, `[low_noise_url]`, `--name` |\n| `loras` | List LoRA models | `--limit`, `--offset` |\n| `lora-status` | Get LoRA upload/import status | `lora_id` |\n| `save-lora` | Activate a LoRA (extends expiry to 7 days) | `lora_id`, `--name` |\n| `delete-lora` | Delete a LoRA model | `lora_id` |\n| `submit-tts` | Submit a text-to-speech job (Qwen3-TTS) — keys off `text`, not a prompt | `text`, `--speaker`, `--language`, `--seed`, `--temperature`, `--top-p`, `--top-k`, `--repetition-penalty` |\n| `models` | List available video/image models (static offline reference) | — |\n| `studio-features` | List Image Studio endpoints, fields, billing (static offline reference) | — |\n| `studio-pricing` | Get live Image Studio pricing | — |\n| `studio-analyze` | AI-analyze a product/garment image or reference URL (freemium) | `--target agent\\|product\\|model\\|reference`, `--image-url`, `--url`, `--prompt`, `--language` |\n| `studio-layouts` | List the layout template library (query, then select) — static asset on the **web front** (phosor.ai), not a `/api/v1` endpoint; needs no key; on a self-hosted deployment set `PHOSOR_WEB_BASE_URL` if the web front is not the gateway host | `--module` (product\\|clothing), `--type` (selling_point\\|aplus\\|white_bg\\|scene\\|closeup\\|size_chart) |\n| `studio-suite` | Generate a product image suite | `--image-url`, `--layout-types`, `--count-per-type`, `--custom-suggestions`, `--template-ids` (ids from `studio-layouts`, auto-expanded to custom_suggestions like the UI's manual pick — use this to get **text-callout selling-point / A+ layouts** and model templates), `--product-info`, `--aspect-ratio`, `--gen-language`, `--model`, `--same-style-reference` |\n| `studio-clothing-suite` | Generate a model/garment image suite | `--image-urls`, `--main-image-types`, `--aplus-types`, `--product-info`, `--brand-config`, `--aspect-ratio`, `--gen-language`, `--model`, `--same-style-reference` |\n| `studio-status` | Get Image Studio job status (separate id space, same `request_id` key) | `request_id` |\n| `studio-cancel` | Cancel a running generation — queued images refunded, already-generating ones charged | `request_id` |\n| `studio-my-works` | List past Image Studio generations | `--task-type`, `--limit`, `--offset` |\n| `studio-call` | Generic call for any other Image Studio endpoint (remove-bg, replace, inpaint, erase, handheld, translate, outpaint, recolor, enhance, upscale, scene-compose, scene-variation, real-model-swap, mannequin-swap, model-scene-swap, ai-outfit, pose-variation, ai-wearable) | `method`, `path`, `--json` |\n\n## Image Studio (Product & Model Photography)\n\nImage Studio is a separate product surface for e-commerce sellers — AI product photography and model/clothing photography — reached through the **same gateway and API key** as video/LoRA, under the `/api/v1/image-studio` prefix. It has its own async namespace - the same key name `request_id`, but a **separate id space**: an Image Studio `request_id` is not valid on `/api/v1/inference/status/...` and vice versa - and its own pricing (flat per-image rate + freemium analyze quota, not per-frame). Full endpoint/parameter reference: [references/api.md](references/api.md#image-studio-product--model-photography--separate-product-surface).\n\n### Quick Start: Product Suite\n\n```bash\n# 1. Upload the product photo\npython3 scripts/phosor_client.py upload-image /path/to/product.jpg\n# Returns: {\"file_id\": \"img-xxx\", \"s3_key\": \"images/img-xxx.jpg\", ...}\n\n# 2. (Optional) AI-analyze it first for richer generation context\npython3 scripts/phosor_client.py studio-analyze --target product --image-url \"images/img-xxx.jpg\"\n\n# 3. Generate a product image suite\npython3 scripts/phosor_client.py studio-suite --image-url \"images/img-xxx.jpg\" \\\n  --layout-types \"white_background,lifestyle_scene\" --count-per-type 2\n\n# 4. Poll for the result\npython3 scripts/phosor_client.py studio-status <request_id>\n```\n\n### Quick Start: Clothing/Model Suite\n\n```bash\npython3 scripts/phosor_client.py upload-image /path/to/garment.jpg\npython3 scripts/phosor_client.py studio-clothing-suite \\\n  --image-urls \"images/img-xxx.jpg\" \\\n  --main-image-types '{\"model_shot\":2,\"selling_point\":1}' \\\n  --aplus-types '{\"standard_aplus\":1}'\npython3 scripts/phosor_client.py studio-status <request_id>\n```\n\n### Quick Start: One-off Edits (remove-bg, inpaint, translate, etc.)\n\nThe long tail of single-purpose editing endpoints doesn't get a dedicated subcommand — use `studio-call` with the exact field names from [references/api.md](references/api.md#image-studio-product--model-photography--separate-product-surface):\n\n```bash\npython3 scripts/phosor_client.py studio-call POST /product/remove-bg \\\n  --json '{\"image_url\": \"images/img-xxx.jpg\", \"count\": 2}'\npython3 scripts/phosor_client.py studio-status <request_id>\n```\n\n### Key facts\n\n- **Every Image Studio call requires `X-API-Key`** (`PHOSOR_API_KEY`), including `GET /pricing` — there is no unauthenticated endpoint under this prefix.\n- **All generation/analyze endpoints are async**: POST returns `{\"request_id\": \"...\", \"status\": \"pending\"}`; poll `studio-status <request_id>` until `status` is `\"done\"`, `\"error\"` or `\"cancelled\"`. Earlier revisions of this skill said the key was `job_id` - it is not, and reading it yields `undefined`. Image Studio ids live in a **separate id space** from video/LoRA: `poll`/`status`/`result` will not accept an Image Studio `request_id`.\n- **Cancelling**: `POST /jobs/{request_id}/cancel` stops a running generation. Images still queued are refunded; images already generating are charged and cannot be stopped; images already delivered bill once through the normal path. The response reports the split as `refunded_queued`, `charged_running` and `already_done`, and the task then polls as `status: \"cancelled\"` - not an error.\n- **Pricing is per-image, not per-frame**: call `studio-pricing` for the live rate. Partial success (e.g. 3 of 5 images) bills only the successes.\n- **Analyze is freemium**: `agent/analyze`, `product/analyze`, `model/analyze` share a daily free quota before per-call billing kicks in.\n- **`model_attrs` matters for model-photography endpoints** (real-model-swap, mannequin-swap, ai-outfit, ai-wearable) — pass `{gender, age_group, ethnicity, skin_tone, hair_color}` explicitly; it is not reliably inferred from the source image alone.\n- Run `studio-features` for the full offline endpoint/field catalog without leaving the terminal.\n\n## Key Constraints\n\n### Video Resolutions (exact pairs only)\n\n| Preset | Width × Height | Max Frames (turbo) | Max Frames (standard) |\n|--------|---------------|-------------------|----------------------|\n| 480p landscape | 854 × 480 | 161 | 161 |\n| 480p portrait | 480 × 854 | 161 | 161 |\n| 720p landscape | 1280 × 720 | 161 | 161 |\n| 720p portrait | 720 × 1280 | 161 | 161 |\n| 1080p landscape | 1920 × 1080 | 153 | **81** |\n| 1080p portrait | 1080 × 1920 | 153 | **81** |\n\n> Standard (non-turbo) mode: 1080p is capped at 81 frames due to generation time limits.\n\n### S2V / Animate Video Resolutions (exact pairs only)\n\n| Preset | Width x Height | Max Frames |\n|--------|---------------|------------|\n| 480p landscape | 854 x 480 | 161 |\n| 480p portrait | 480 x 854 | 161 |\n| 512p square | 512 x 512 | 161 |\n| 720p landscape | 1280 x 720 | 161 |\n| 720p portrait | 720 x 1280 | 161 |\n\n### Image Resolutions (exact pairs only)\n\n| Preset | Width × Height |\n|--------|---------------|\n| Square small | 512 × 512 |\n| Square | 1024 × 1024 |\n| Landscape | 1024 × 768 |\n| Portrait | 768 × 1024 |\n| Wide landscape | 1280 × 768 |\n| Tall portrait | 768 × 1280 |\n\n### MiniMax H3 Frame Sizes (`--resolution-tier` + `--aspect`)\n\n| Tier | 16:9 | 4:3 | 1:1 | 3:4 | 9:16 |\n|------|------|-----|-----|-----|------|\n| 480p | 832 × 480 | 640 × 480 | 480 × 480 | 480 × 640 | 480 × 832 |\n| 768p | 1344 × 768 | 1024 × 768 | 768 × 768 | 768 × 1024 | 768 × 1344 |\n\n`duration` is 4–15 seconds (default 5). Output FPS is fixed at 24 and\n`frames_per_second` is ignored.\n\n### Ref2VA Reference Limits (`minimax/h3/reference-to-video`)\n\nCaps differ per tier, and the image cap is higher when you send **only** images:\n\n| Limit | 480p | 768p |\n|-------|------|------|\n| Reference images (with videos/audio present) | 5 | 5 |\n| Reference images (images only) | 9 | 9 |\n| Reference videos | 3 | 3 |\n| Reference audios | 3 | 3 |\n\n| Limit | Value |\n|-------|-------|\n| Total reference video length | 15.1 s (across all reference videos) |\n| Reference video FPS ceiling | 24 |\n| Reference audio length | 10 s each |\n| Reference image longest edge | 2048 px |\n| Reference image aspect ratio | ≤ 4.0 |\n\n### FLUX.2-dev Resolutions (exact pairs only)\n\n| Width × Height |\n|---------------|\n| 2048 × 1536 · 1536 × 2048 |\n| 2048 × 1152 · 1152 × 2048 |\n| 2048 × 2048 · 1024 × 1024 |\n\n> FLUX.2-dev does not accept the general image resolution list above, and always\n> returns exactly 1 image.\n\n### Frame Alignment (video only)\n\nFrames must follow `1 + 4*k` where `k >= 1` (e.g. 5, 9, 13, ... 81, 85, ...). Server auto-aligns down.\n\n### Video Inference Parameters\n\n| Parameter | Default | Range |\n|-----------|---------|-------|\n| `frames_per_second` | 16 | 4–60 |\n| `num_inference_steps` | 4 | 4–40 |\n| `guidance_scale` | 1.0 | 1.0–10.0 |\n\n### Image Inference Parameters\n\n| Parameter | Default | Range | Notes |\n|-----------|---------|-------|-------|\n| `num_images` | 1 | 1–4 | Number of images to generate |\n| `num_inference_steps` | varies | 1–4 (z-image turbo), 1–40 (qwen-image) | Model-dependent max |\n| `guidance_scale` | varies | 1.0–20.0 | — |\n| `strength` | — | 0.0–1.0 | Image-to-image only: how much to transform the source |\n| `output_format` | png | png, jpeg | Output file format |\n\n### Concurrency\n\nModel API inference jobs run concurrently up to a per-account cap; over it the submit\nreturns `429` and you retry after a job finishes.\n\n| Tier | Plan | Concurrent Model API jobs |\n|------|------|---------------------------|\n| Free | — | 1 |\n| 1 | Starter | 2 |\n| 2 | Standard | 4 |\n| 3 | Pro | 8 |\n\nImage Studio runs on its own pool and does **not** consume this quota — a suite and a\nvideo generation can run at the same time.\n\n### Multiple LoRAs\n\n```bash\npython3 scripts/phosor_client.py submit \"A person dancing\" \\\n  --loras '[{\"lora_id\": \"lora-abc\", \"lora_scale\": 0.8}, {\"lora_id\": \"lora-def\", \"lora_scale\": 0.5}]'\n```\n\n### Two-Step Upload Rule\n\nFiles must be uploaded before use — direct URLs are NOT supported in `submit --image-url`:\n\n1. **Image** → `upload-image` / `import-image` → returns `s3_key` → use as `--image-url`\n2. **LoRA** → `upload-lora` / `import-lora` → returns `lora_id` → use as `--lora-id`\n\n## Queue Flow\n\n```\nPENDING → PROCESSING → COMPLETED / FAILED\n```\n\nThe `poll` command checks all locally-tracked pending jobs and removes completed/failed ones.\n\n## MiniMax H3 Pricing (per output second)\n\nPer output second, regardless of clip length:\n\n| Model | 480p USD/sec | 768p USD/sec | 480p credits/sec | 768p credits/sec |\n|-------|--------------|--------------|------------------|------------------|\n| `text-to-video`, `image-to-video` | $0.0045 | $0.012 | 0.045 | 0.12 |\n| `reference-to-video` | $0.0045 | $0.012 | 0.045 | 0.12 |\n| `reference-to-video/turbo` | $0.0045 | $0.012 | 0.045 | 0.12 |\n\n### Ref2VA inputs\n\nOutput **plus** reference inputs:\n\n```\ntotal = duration x base_rate\n      + max(0, reference_images - 5) x image_rate   # first 5 images free\n      + reference_video_seconds x base_rate         # reference audio is free\n```\n\n| Model | Tier | Base USD/sec | Reference image | Reference audio | Reference video USD/sec |\n|-------|------|--------------|-----------------|-----------------|-------------------------|\n| `reference-to-video` | 480p | $0.0045 | first 5 free, then $0.0075 each | free | $0.0045 |\n| `reference-to-video` | 768p | $0.012 | first 5 free, then $0.0075 each | free | $0.012 |\n| `reference-to-video/turbo` | 480p | $0.0045 | first 5 free, then $0.0075 each | free | $0.0045 |\n| `reference-to-video/turbo` | 768p | $0.012 | first 5 free, then $0.0075 each | free | $0.012 |\n\nExample — `reference-to-video`, 5s at 768p with 4 reference images: `5 x $0.012` = 0.6 credits.\n5s at 480p with 9 images: `5 x $0.0045 + 4 x $0.0075` = 0.525 credits.\n\n## Wan Video Pricing (per frame)\n\n| Tier | Standard | + LoRA | Turbo |\n|------|----------|--------|-------|\n| 480p | $0.0009375 | $0.00125 | $0.0003125 |\n| 512p | $0.0013125 | $0.001625 | $0.0004375 |\n| 720p | $0.001875 | $0.0021875 | $0.000625 |\n| 1080p | $0.0025 | $0.003 | $0.0010938 |\n\nLoRA multiplier on turbo: 1.2x.\n\n## S2V Pricing (per frame)\n\n| Tier | Cost |\n|------|------|\n| 480p | $0.0009375 |\n| 512p | $0.0013125 |\n| 720p | $0.001875 |\n\n## Animate Pricing (per frame)\n\n| Tier | Cost |\n|------|------|\n| 480p | $0.00125 |\n| 512p | $0.00175 |\n| 720p | $0.0025 |\n\n## Image Pricing (per image)\n\n| Model | USD | Credits |\n|-------|-----|---------|\n| GPT Image 2 T2I | $0.03 | 0.3 |\n| FLUX.2-dev T2I | $0.006 | 0.06 |\n| FLUX.2-dev Image Edit | $0.012 | 0.12 |\n| qwen-image T2I | $0.015 | 0.15 |\n| qwen-image T2I + LoRA | $0.018 | 0.18 |\n| qwen-image-edit | $0.003 | 0.03 |\n| z-image turbo (T2I / I2I) | $0.0025 | 0.025 |\n| z-image turbo + LoRA | $0.003 | 0.03 |\n\nFlat per image at every resolution. Multiply by `num_images` (1–4) where the model\nsupports it; FLUX.2-dev is fixed at 1 image per request.\n\n## Audio Pricing\n\n| Item | Cost |\n|------|------|\n| Qwen3-TTS | $0.00003 per character |\n| Minimum charge | $0.003 per request |\n\nExchange rate: 10 credits = $1 USD. Credits pre-deducted, auto-refunded on failure.\nLive rates: `GET /api/v1/pricing/config` — always prefer it over any table here.\n\nFile v1.3.1:README.md\n\n# Phosor AI — Agent Skill\n\n**Skill version 1.3.1** · API v1.2.1 · updated 2026-09-24\n\nTwo version numbers, on purpose: the skill version tracks this package\n(commands, docs, the bundled client), the API version tracks the gateway\ncontract. A skill release that only rewords docs or fixes the client does not\nmove the API version, and a gateway change does not force a skill release.\n\nVerify what you installed: `python3 scripts/phosor_client.py --version`\n\n## Quick Start\n\n```bash\nexport PHOSOR_API_KEY=\"your-key\"\n\n# Text-to-Video\npython3 scripts/phosor_client.py submit \"A cat walking on a beach\" --width 854 --height 480\n\n# Image-to-Video (two-step: upload then submit)\npython3 scripts/phosor_client.py upload-image photo.jpg\npython3 scripts/phosor_client.py submit \"The scene comes alive\" --image-url \"images/img-xxx.jpg\"\n\n# Check status / get result\npython3 scripts/phosor_client.py status <request_id>\npython3 scripts/phosor_client.py result <request_id>\n\n# LoRA Upload (custom pre-trained)\npython3 scripts/phosor_client.py upload-lora high_noise.safetensors low_noise.safetensors\n\npython3 scripts/phosor_client.py save-lora <lora_id>\n\n# Image Studio: AI product photography (separate product surface, same API key)\npython3 scripts/phosor_client.py upload-image product.jpg\npython3 scripts/phosor_client.py studio-suite --image-url \"<s3_url from upload>\" \\\n  --layout-types \"white_background,lifestyle_scene\" --count-per-type 2\npython3 scripts/phosor_client.py studio-status <job_id>\n```\n\nSee [SKILL.md](SKILL.md#image-studio-product--model-photography) for the full Image Studio quick start (clothing/model suite, one-off edits like remove-bg/inpaint/translate).\n\n## Requirements\n\n- Python 3.7+ (stdlib only, no pip install needed)\n- `PHOSOR_API_KEY` environment variable\n\n## Commands\n\nRun `python3 scripts/phosor_client.py --help` for all 31 commands (23 video/LoRA + 8 Image Studio).\n\n## Links\n\n- [Phosor AI](https://phosor.ai)\n- [API Documentation](https://docs.phosor.ai)\n\nFile v1.3.1:_meta.json\n\n{\n  \"ownerId\": \"kn7dkx5wmnapr9y2x91m6qnb8s82w04e\",\n  \"slug\": \"phosor-ai-skills\",\n  \"version\": \"1.3.1\",\n  \"publishedAt\": 1790258711213\n}\n\nFile v1.3.1:references/api.md\n\n# Phosor AI API Reference\n\nBase URL: `https://phosor.ai`\n\nAll endpoints require `X-API-Key` header unless noted otherwise.\n\n## Endpoints\n\n### Models\n\n| Method | Path | Auth | Description |\n|--------|------|------|-------------|\n| GET | `/api/v1/models` | None | List available models |\n\n### Inference\n\n| Method | Path | Description |\n|--------|------|-------------|\n| POST | `/api/v1/inference/submit` | Submit video or image generation job |\n| GET | `/api/v1/inference/status/{request_id}` | Get job status + progress |\n| GET | `/api/v1/inference/result/{request_id}` | Get completed result (video or image URL) |\n| GET | `/api/v1/inference/history` | Get user's job history |\n\n### Storage — Image / Audio\n\n| Method | Path | Description |\n|--------|------|-------------|\n| POST | `/api/v1/storage/image/upload` | Multipart image or audio upload (images: jpg/png/webp; audio for S2V: mp3/wav/flac/aac/ogg/m4a) |\n| POST | `/api/v1/storage/image/import` | Import from public URL |\n\n### Storage — LoRA\n\n| Method | Path | Description |\n|--------|------|-------------|\n| POST | `/api/v1/storage/lora/upload` | Upload two .safetensors files (video LoRA: high_noise + low_noise) |\n| POST | `/api/v1/storage/lora/import` | Import from HTTPS URLs (video: two files, image: single file) |\n\n### LoRA Management\n\n| Method | Path | Description |\n|--------|------|-------------|\n| GET | `/api/v1/loras` | List LoRA models |\n| GET | `/api/v1/loras/{lora_id}` | Get single LoRA details |\n| GET | `/api/v1/loras/{lora_id}/status` | Get processing status |\n| POST | `/api/v1/loras/{lora_id}/save` | Activate a LoRA (extend expiry to 7 days) |\n| DELETE | `/api/v1/loras/{lora_id}` | Soft delete |\n\n### Image Studio (product & model photography — separate product surface)\n\nAll paths below are relative to `/api/v1/image-studio` (e.g. the full path for `/product/suite` is `/api/v1/image-studio/product/suite`). **Every path requires `X-API-Key`, including `GET /pricing`** — there is no unauthenticated Image Studio endpoint.\n\nAll generation/analyze endpoints are async: POST returns `{\"request_id\": \"...\", \"status\": \"pending\"}` immediately; poll `GET /jobs/{request_id}` until `status` is `\"done\"`, `\"error\"` or `\"cancelled\"`.\n\n> **The async key is `request_id`, not `request_id`.** Earlier revisions of this document said\n> `request_id`; that key is never present in a response. Reading it yields `undefined`, and the\n> poll then never resolves. Image Studio's `request_id` is a separate namespace from the\n> video/LoRA `request_id` — do not pass one to the other's status endpoint.\n\n| Method | Path | Description |\n|--------|------|-------------|\n| POST | `/agent/analyze` | AI-analyze an image for the Agent-image workflow (freemium) |\n| POST | `/product/analyze` | AI-analyze a product image ahead of `product/suite` (freemium) |\n| POST | `/model/analyze` | AI-analyze a garment image ahead of `model/clothing-suite` (freemium) |\n| POST | `/product/reference/analyze` | Analyze a reference product page/image for style-matching (free) |\n| POST | `/product/suite` | Generate a product image suite (multiple layout types) |\n| POST | `/product/scene-compose` | Composite a product into a reference scene |\n| POST | `/product/scene-variation` | Generate scene/background variations of a product photo |\n| POST | `/product/remove-bg` | Remove background (transparent) |\n| POST | `/product/replace` | Replace/refresh the product in an existing scene composite |\n| POST | `/product/inpaint` | Masked region fill |\n| POST | `/product/erase` | Masked region removal |\n| POST | `/product/handheld` | Generate a hand-holding-product shot |\n| POST | `/product/translate` | Translate on-image text to other language(s) |\n| POST | `/product/outpaint` | Expand canvas beyond original image bounds |\n| POST | `/product/recolor` | Recolor product or region |\n| POST | `/product/enhance` | AI enhancement/upgrade pass |\n| POST | `/product/upscale` | Upscale to 1024×1024 |\n| POST | `/model/clothing-suite` | Generate a full garment/model image suite (model shots, selling points, size chart, A+ modules) |\n| POST | `/model/real-model-swap` | Swap in a real human model wearing the garment |\n| POST | `/model/mannequin-swap` | Swap in a mannequin wearing the garment |\n| POST | `/model/model-scene-swap` | Swap the scene/background behind an existing model photo |\n| POST | `/model/ai-outfit` | Dress a described model in the garment |\n| POST | `/model/ai-wearable` | Generate a model wearing/using an accessory |\n| POST | `/model/pose-variation` | Generate pose variations from a source model photo |\n| GET | `/jobs/{request_id}` | Poll job status/result (Image Studio's own async namespace) |\n| GET | `/pricing` | Live pricing: `per_image_credits`, `per_analyze_credits`, `analyze_daily_free_quota` |\n| GET | `/my-works` | List past Image Studio generations (paginated) |\n| GET | `/my-works/{request_id}` | One past generation with its inputs and outputs |\n| GET | `/my-works/{request_id}/download` | Download the generated images (single file or archive) |\n| DELETE | `/my-works/{request_id}` | Soft-delete a past generation |\n| POST | `/jobs/{request_id}/cancel` | Cancel a running generation. See \"Cancelling\" below |\n| POST | `/tasks/{task_id}/review` | Submit a per-image review (thumbs) for a finished task |\n| GET | `/tasks/{task_id}/review` | Read the review already submitted for a task |\n| POST | `/feedback` | Send feedback about a generation |\n\n#### Common parameters\n\n| Parameter | Applies to | Notes |\n|-----------|-----------|-------|\n| `image_url` | analyze + most edit endpoints | S3 key or URL of the source image; some endpoints alias this as `product_image_url` |\n| `model` | all generation endpoints | Optional. Accepted values: `openai:gpt-image-2` and `phosor:phosor-model-api` (Phosor's image model). Analyze endpoints do not take this parameter |\n| `aspect_ratio` | most generation endpoints | One of `1:1`, `3:4`, `4:3`, `9:16`, `16:9` (default varies per endpoint, e.g. `3:4` for clothing-suite, `1:1` for most product tools) |\n| `count` | most single-image-in-single-image-out endpoints | How many output images to generate, typically clamped 1–20 (varies per endpoint) |\n| `prompt` | analyze + most edit endpoints | Free-text hint appended to the internal template — not a full prompt replacement |\n| `same_style_reference` | `product/suite`, `model/clothing-suite` | Object from a prior `product/reference/analyze` call, used to match layout/style to a reference image |\n| `model_attrs` | model-photography endpoints (`real-model-swap`, `mannequin-swap`, `ai-outfit`, `ai-wearable`) | `{gender, age_group, ethnicity, skin_tone, hair_color}` — **required for accurate results**, not reliably inferred from the source image alone |\n\n#### `POST /product/suite` body\n\n| Field | Required | Notes |\n|-------|----------|-------|\n| `product_image_url` | Yes | (or `image_url`) |\n| `layout_types` | No | list[str] of layout type ids |\n| `count_per_type` | No | int, default 1 |\n| `custom_suggestions` | No | list |\n| `product_info` | No | free-text product description/facts (authoritative over what the AI guesses from the image) |\n| `same_style_reference` | No | dict from `product/reference/analyze` |\n| `brand_config` | No | dict — brand color/font/platform/tone settings |\n| `aspect_ratio` | No | `1:1`\\|`3:4`\\|`4:3`\\|`9:16`\\|`16:9` |\n| `gen_language` | No | language for any on-image text |\n| `model` | No | see Common parameters |\n\nBilled count = `len(layout_types) * count_per_type + len(custom_suggestions)` (min 1).\n\n#### `POST /model/clothing-suite` body\n\n| Field | Required | Notes |\n|-------|----------|-------|\n| `clothing_image_urls` | Yes | list[str], up to 5 used for generation |\n| `main_image_types` | No | dict of `{model_shot\\|grass_shot\\|selling_point\\|size_chart: count}` |\n| `aplus_types` | No | dict of `{standard_aplus\\|mobile_aplus\\|basic_aplus\\|custom_ratio: count}` |\n| `product_info` | No | free-text |\n| `brand_config` | No | dict |\n| `same_style_reference` | No | dict |\n| `aspect_ratio` | No | default `3:4` |\n| `gen_language` | No | |\n| `model` | No | |\n\nBilled count = sum of all `main_image_types`/`aplus_types` counts (min 1). At least one image type must be selected, or a 400 is returned. Job result includes `grouped_images: [{label, url}]` with a display label per image type (model photo, lifestyle, selling point, size chart, premium A+, mobile A+, standard A+, custom ratio; labels are returned in Chinese), plus `failed_labels` for any type that failed to generate (partial success is normal, not an error state).\n\n#### `GET /jobs/{request_id}` response shape\n\n- `status: \"pending\"` — includes `partial_images: [{url, label}]` for images completed so far (useful for progressive UIs).\n- `status: \"done\"` — includes `result` (shape varies by feature; generation results always have `images`, `expected_count`, `success_count`; `clothing-suite` also has `grouped_images`/`failed_labels`; analyze results include the AI's structured findings plus `remaining_free_quota`, `daily_free_quota`, `is_free`).\n- `status: \"error\"` — includes a user-facing `error` string (never a raw provider error).\n\n## Video Inference Submit Parameters\n\n| Parameter | Type | Required | Default | Range |\n|-----------|------|----------|---------|-------|\n| `prompt` | string | Yes | — | max 2000 chars |\n| `model` | string | No | `wan/v2.2-a14b/text-to-video` | See supported models |\n| `width` | int | No | 854 | See video resolutions |\n| `height` | int | No | 480 | See video resolutions |\n| `num_frames` | int | No | 81 | Auto-aligned to 1+4k |\n| `frames_per_second` | int | No | 16 | 4–60 |\n| `num_inference_steps` | int | No | 4 | 4–40 |\n| `guidance_scale` | float | No | 1.0 | 1.0–10.0 |\n| `seed` | int | No | random | — |\n| `negative_prompt` | string | No | \"\" | — |\n| `image_url` | string | No | — | S3 key from upload-image (required for I2V) |\n| `lora_id` | string | No | — | Single LoRA |\n| `lora_scale` | float | No | 1.0 | 0.0–1.0 |\n| `loras` | array | No | — | Multiple LoRAs: `[{\"lora_id\":\"...\",\"lora_scale\":1.0}]` |\n\n## S2V Inference Submit Parameters\n\nUses the same `/api/v1/inference/submit` endpoint. Set `model` to `wan/v2.2-a14b/speech-to-video`.\n\n| Parameter | Type | Required | Default | Range |\n|-----------|------|----------|---------|-------|\n| `prompt` | string | Yes | — | max 2000 chars |\n| `model` | string | Yes | — | `wan/v2.2-a14b/speech-to-video` |\n| `image_url` | string | Yes | — | S3 key from upload-image (reference face image) |\n| `audio_url` | string | Yes | — | S3 key from `/api/v1/storage/image/upload` (accepts mp3, wav, flac, aac, ogg, m4a) |\n| `width` | int | No | 854 | See S2V/Animate resolutions |\n| `height` | int | No | 480 | See S2V/Animate resolutions |\n| `num_frames` | int | No | 81 | Auto-aligned to 1+4k |\n| `frames_per_second` | int | No | 16 | 4–60 |\n| `num_inference_steps` | int | No | 4 | 4–40 |\n| `guidance_scale` | float | No | 1.0 | 1.0–10.0 |\n| `seed` | int | No | random | — |\n| `negative_prompt` | string | No | \"\" | — |\n\n## Animate Inference Submit Parameters\n\nUses the same `/api/v1/inference/submit` endpoint. Set `model` to `wan/v2.2-a14b/animate`.\n\n| Parameter | Type | Required | Default | Range |\n|-----------|------|----------|---------|-------|\n| `prompt` | string | Yes | — | max 2000 chars |\n| `model` | string | Yes | — | `wan/v2.2-a14b/animate` |\n| `image_url` | string | Yes | — | S3 key from upload-image (reference character image) |\n| `video_url` | string | Yes | — | URL to motion/pose reference video |\n| `width` | int | No | 854 | See S2V/Animate resolutions |\n| `height` | int | No | 480 | See S2V/Animate resolutions |\n| `num_frames` | int | No | 81 | Auto-aligned to 1+4k |\n| `frames_per_second` | int | No | 16 | 4–60 |\n| `num_inference_steps` | int | No | 4 | 4–40 |\n| `guidance_scale` | float | No | 1.0 | 1.0–10.0 |\n| `seed` | int | No | random | — |\n| `negative_prompt` | string | No | \"\" | — |\n\n## Image Inference Submit Parameters\n\nUses the same `/api/v1/inference/submit` endpoint. Set `model` to an image model ID.\n\n| Parameter | Type | Required | Default | Range |\n|-----------|------|----------|---------|-------|\n| `prompt` | string | Yes | — | max 2000 chars |\n| `model` | string | Yes | — | Must be an image model ID (see below) |\n| `width` | int | No | 1024 | See image resolutions |\n| `height` | int | No | 1024 | See image resolutions |\n| `num_images` | int | No | 1 | 1–4 |\n| `num_inference_steps` | int | No | varies | 1–8 (z-image turbo), 1–40 (qwen-image) |\n| `guidance_scale` | float | No | varies | 1.0–20.0 |\n| `seed` | int | No | random | — |\n| `negative_prompt` | string | No | \"\" | — |\n| `image_url` | string | No | — | S3 key from upload-image (required for I2I) |\n| `strength` | float | No | — | 0.0–1.0 (I2I only: transformation strength) |\n| `output_format` | string | No | \"png\" | \"png\" or \"jpeg\" |\n| `lora_id` | string | No | — | Single LoRA |\n| `lora_scale` | float | No | 1.0 | 0.0–1.0 |\n| `loras` | array | No | — | Multiple LoRAs: `[{\"lora_id\":\"...\",\"lora_scale\":1.0}]` |\n\n### Wan Video Model IDs\n\n| Model ID | Description |\n|----------|-------------|\n| `wan/v2.2-a14b/text-to-video` | Wan 2.2 T2V 14B — standard quality (`num_inference_steps`, `guidance_scale` apply) |\n| `wan/v2.2-a14b/text-to-video/turbo` | Wan 2.2 T2V turbo — fixed internal defaults, ~3x cheaper per frame |\n| `wan/v2.2-a14b/image-to-video` | Wan 2.2 I2V 14B — standard quality |\n| `wan/v2.2-a14b/image-to-video/turbo` | Wan 2.2 I2V turbo |\n\n> Turbo variants ignore `num_inference_steps` / `guidance_scale`. Standard mode caps\n> 1080p at 81 frames; turbo allows 153.\n\n### LTX-Video Model IDs\n\n| Model ID | Type | Required inputs |\n|----------|------|-----------------|\n| `ltx-video/v2.3/image-audio-to-video` | Image(+Audio)-to-Video | `prompt`, `image_url`; `audio_url` optional |\n\nFrame-based. `resolution_tier` + `aspect_ratio` (`480p`/`512p`/`720p`/`1080p` × `16:9`/`9:16`/`1:1`)\nor explicit `width`/`height` from: 854×480, 1280×720, 1920×1080, 480×854, 720×1280, 1080×1920,\n480×480, 720×720, 1080×1080. Default 1280×720, fps 24, duration 5s\n(`num_frames` = `duration × fps + 1`). Max frames: 480p 481, 720p 241, 1080p 121 —\n**not** subject to the Wan `1+4k` alignment.\n\nPricing per frame: 480p `$0.0008`, 512p `$0.0009`, 720p and above `$0.0012`.\n\n### MiniMax H3 Model IDs\n\nDuration-based (ignores `num_frames` / `frames_per_second`; output is fixed 24fps).\n\n| Model ID | Type | Required inputs |\n|----------|------|-----------------|\n| `minimax/h3/text-to-video` | Text-to-Video | `prompt` |\n| `minimax/h3/image-to-video` | Image-to-Video | `prompt`, `image_url` (optional `end_image_url`) |\n| `minimax/h3/reference-to-video` | Reference-to-Video | `prompt`, plus at least one of `reference_image_urls` / `reference_video_urls` |\n| `minimax/h3/reference-to-video/turbo` | Reference-to-Video (Turbo workflow) | same as above |\n\n| Parameter | Default | Allowed values |\n|-----------|---------|----------------|\n| `resolution_tier` | `480p` | `480p`, `768p` |\n| `aspect_ratio` | `16:9` | `16:9`, `4:3`, `1:1`, `3:4`, `9:16` |\n| `duration` | `5` | 4–15 seconds |\n| `reference_audio_urls` | — | array; Ref2VA only |\n| `reference_image_labels` | — | array, same order as `reference_image_urls`; what each picture shows (e.g. `[\"Shen Yu - character\", \"A Man - character\", \"coop - scene\"]`) |\n| `reference_video_labels` | — | array, same order as `reference_video_urls` |\n| `reference_audio_labels` | — | array, same order as `reference_audio_urls` |\n| `use_ref_video_audio` | `false` | also use each reference video's own soundtrack |\n\nFrame sizes: 480p → 832×480 (16:9), 640×480 (4:3), 480×480 (1:1), 480×640 (3:4), 480×832 (9:16).\n768p → 1344×768, 1024×768, 768×768, 768×1024, 768×1344.\n\n**Naming your references.** Uploaded files are stored under generated ids, so a prompt\nthat says \"Shen Yu reaches into the coop\" cannot be matched to a picture by name on its own.\nPass `reference_image_labels` alongside `reference_image_urls` (same order) and the platform\nbinds each name in your prompt to the right picture while building the model prompt.\n**Write your prompt normally** — do not rewrite it to mention picture numbers.\n\n**Ref2VA reference caps** — differ per tier and per variant, and the image cap is\nhigher when sending images only:\n\n`minimax/h3/reference-to-video`\n\n| Limit | 480p | 768p |\n|-------|------|------|\n| Reference images (with videos/audio) | 5 | 5 |\n| Reference images (images only) | 9 | 9 |\n| Reference videos | 3 | 3 |\n| Reference audios | 3 | 3 |\n\nTotal reference video length 15.1s (all videos combined), reference video FPS ≤ 24,\nreference audio ≤ 10s each, reference image longest edge ≤ 2048px, aspect ratio ≤ 4.0.\n\n`minimax/h3/reference-to-video/turbo`\n\n| Limit | 480p | 768p |\n|-------|------|------|\n| Reference images (with videos/audio) | 5 | 5 |\n| Reference images (images only) | 9 | 9 |\n| Reference videos | 3 | 3 |\n| Reference audios | 3 | 3 |\n\nTotal reference video length 10s, reference video FPS ≤ 24, reference audio ≤ 10s each,\nreference image longest edge ≤ 4096px, aspect ratio ≤ 4.0.\n\n> ⚠️ **768p Turbo: not every combination inside the caps above is accepted.** Keep 768p Turbo\n> requests at or below **3 reference images, or 2 images plus 1 video**. 480p has far more headroom.\n\n### Audio Model IDs\n\n| Model ID | Description |\n|----------|-------------|\n| `qwen3-tts/text-to-speech/1.7b` | Qwen3-TTS CustomVoice — takes `text` (≤500 chars), not `prompt` |\n\nParameters: `text` (required), `speaker` (default `Sohee`), `language` (default `Chinese`),\n`seed`, `temperature` (0.0–2.0, default 0.9), `top_p` (default 1.0), `top_k` (1–200, default 50),\n`repetition_penalty` (1.0–2.0, default 1.05).\n\n### S2V / Animate Model IDs\n\n| Model ID | Description |\n|----------|-------------|\n| `wan/v2.2-a14b/speech-to-video` | Wan 2.2 Speech-to-Video 14B (lip-sync from audio + face image) |\n| `wan/v2.2-a14b/animate` | Wan 2.2 Animate 14B (motion transfer from video + character image) |\n\n### Image Model IDs\n\n| Model ID | Description |\n|----------|-------------|\n| `openai/gpt-image-2/text-to-image` | GPT Image 2 Text-to-Image — 5 sizes: **1024×1024**, **1920×1072** / **1072×1920** (1080p), **2560×1440** / **1440×2560** (2K). Always 1 image; ignores `num_inference_steps` / `guidance_scale` / `output_format` |\n| `flux2/dev/text-to-image` | FLUX.2-dev Text-to-Image — own resolution whitelist, always 1 image |\n| `flux2/dev/image-edit` | FLUX.2-dev Image Edit |\n| `qwen-image/v2512/text-to-image` | Qwen Image 2512 Text-to-Image |\n| `qwen-image/v2512/text-to-image/lora` | Qwen Image 2512 T2I with LoRA |\n| `qwen-image/v2511/image-edit` | Qwen Image Edit 2511 — multi-image reference editing |\n| `z-image/turbo/text-to-image` | Z-Image Turbo Text-to-Image |\n| `z-image/turbo/text-to-image/lora` | Z-Image Turbo T2I with LoRA |\n| `z-image/turbo/image-to-image` | Z-Image Turbo Image-to-Image |\n| `z-image/turbo/image-to-image/lora` | Z-Image Turbo I2I with LoRA |\n\n## Supported Video Resolutions\n\n| Preset | Width | Height | Max Frames (turbo) | Max Frames (standard) |\n|--------|-------|--------|-------------------|----------------------|\n| 480p landscape | 854 | 480 | 161 | 161 |\n| 480p portrait | 480 | 854 | 161 | 161 |\n| 720p landscape | 1280 | 720 | 161 | 161 |\n| 720p portrait | 720 | 1280 | 161 | 161 |\n| 1080p landscape | 1920 | 1080 | 153 | 81 |\n| 1080p portrait | 1080 | 1920 | 153 | 81 |\n\nOnly these exact pairs are accepted. Frame alignment: `valid_frames = 1 + 4*k` where `k >= 1`.\n\n> Standard (non-turbo) model: 1080p max is 81 frames due to generation time constraints.\n\n## Supported S2V / Animate Resolutions\n\n| Preset | Width | Height | Max Frames |\n|--------|-------|--------|------------|\n| 480p landscape | 854 | 480 | 161 |\n| 480p portrait | 480 | 854 | 161 |\n| 512p square | 512 | 512 | 161 |\n| 720p landscape | 1280 | 720 | 161 |\n| 720p portrait | 720 | 1280 | 161 |\n\nOnly these exact pairs are accepted. Frame alignment: `valid_frames = 1 + 4*k` where `k >= 1`.\n\n## Supported Image Resolutions\n\n| Preset | Width | Height |\n|--------|-------|--------|\n| Square small | 512 | 512 |\n| Square | 1024 | 1024 |\n| Landscape | 1024 | 768 |\n| Portrait | 768 | 1024 |\n| Wide landscape | 1280 | 768 |\n| Tall portrait | 768 | 1280 |\n\nOnly these exact pairs are accepted.\n\n## Pricing\n\n> Video/image inference pricing below is a static reference and can drift — verify against support before relying on it for billing-sensitive integrations. Image Studio has its own pricing surface with a live endpoint (`GET /api/v1/image-studio/pricing`) — always prefer that over any static table for Image Studio.\n\n### Video Inference (10 credits = $1 USD)\n\n| Resolution | Per-Frame (USD) | Per-Frame (Credits) |\n|-----------|----------------|-------------------|\n| 480p | $0.0009375 | 0.009375 |\n| 720p | $0.001875 | 0.01875 |\n| 1080p | $0.0025 | 0.025 |\n\nLoRA multiplier: 1.2x (applied when any LoRA is specified).\n\n### MiniMax H3 (10 credits = $1 USD, per output second)\n\nPer output second, regardless of clip length:\n\n| Model | 480p USD/sec | 768p USD/sec | 480p credits/sec | 768p credits/sec |\n|-------|--------------|--------------|------------------|------------------|\n| `text-to-video`, `image-to-video` | $0.0045 | $0.012 | 0.045 | 0.12 |\n| `reference-to-video` | $0.0045 | $0.012 | 0.045 | 0.12 |\n| `reference-to-video/turbo` | $0.0045 | $0.012 | 0.045 | 0.12 |\n\n### MiniMax H3 Ref2VA inputs (10 credits = $1 USD)\n\n```\ntotal = duration x base_rate\n      + max(0, reference_images - 5) x image_rate   # first 5 images free\n      + reference_video_seconds x base_rate         # reference audio is free\n```\n\n| Model | Resolution | Base USD/sec | Reference image | Reference audio | Reference video USD/sec |\n|-------|-----------|--------------|-----------------|-----------------|-------------------------|\n| `reference-to-video` | 480p | $0.0045 | first 5 free, then $0.0075 each | free | $0.0045 |\n| `reference-to-video` | 768p | $0.012 | first 5 free, then $0.0075 each | free | $0.012 |\n| `reference-to-video/turbo` | 480p | $0.0045 | first 5 free, then $0.0075 each | free | $0.0045 |\n| `reference-to-video/turbo` | 768p | $0.012 | first 5 free, then $0.0075 each | free | $0.012 |\n\n`GET /api/v1/pricing/config` publishes the live rates (incl. `reference_image_free_count`).\n\n### S2V Inference (10 credits = $1 USD)\n\n| Resolution | Per-Frame (USD) | Per-Frame (Credits) |\n|-----------|----------------|-------------------|\n| 480p | $0.0009375 | 0.009375 |\n| 512p | $0.0013125 | 0.013125 |\n| 720p | $0.001875 | 0.01875 |\n\n### Animate Inference (10 credits = $1 USD)\n\n| Resolution | Per-Frame (USD) | Per-Frame (Credits) |\n|-----------|----------------|-------------------|\n| 480p | $0.00125 | 0.0125 |\n| 512p | $0.00175 | 0.0175 |\n| 720p | $0.0025 | 0.025 |\n\n### Image Inference (10 credits = $1 USD)\n\n| Model | Per Image (USD) | Per Image (Credits) |\n|-------|----------------|-------------------|\n| GPT Image 2 T2I | $0.03 | 0.3 |\n| FLUX.2-dev T2I | $0.006 | 0.06 |\n| FLUX.2-dev Image Edit | $0.012 | 0.12 |\n| qwen-image T2I | $0.015 | 0.15 |\n| qwen-image T2I + LoRA | $0.018 | 0.18 |\n| qwen-image-edit | $0.003 | 0.03 |\n| z-image turbo | $0.0025 | 0.025 |\n| z-image turbo + LoRA | $0.003 | 0.03 |\n\nTotal cost = per-image price x `num_images` (1-4). LoRA pricing is built into the model-specific rate (no separate multiplier).\n\n### Audio (10 credits = $1 USD)\n\n| Item | Cost |\n|------|------|\n| Qwen3-TTS | $0.00003 per character |\n| Minimum charge | $0.003 per request |\n\n### Image Studio Pricing\n\nImage Studio (see below) charges a **flat per-image rate for every generation endpoint**, and a **freemium daily-quota rate for analyze endpoints**. Both numbers are computed live server-side — call `GET /api/v1/image-studio/pricing` (or `phosor_client.py studio-pricing`) to get the current `per_image_credits`, `per_analyze_credits`, and `analyze_daily_free_quota`. Partial success (e.g. 3 of 5 images generated) bills only the successes; the rest is auto-refunded.\n\n## Concurrency Limits\n\nModel API inference jobs run concurrently up to a per-account cap; going over returns `429`.\n\n| Tier | Plan | Concurrent Model API jobs |\n|------|------|---------------------------|\n| Free | — | 1 |\n| 1 | Starter | 2 |\n| 2 | Standard | 4 |\n| 3 | Pro | 8 |\n\n> Image Studio runs on its own pool and does not consume this quota.\n\n## Limits & Quotas\n\n| Resource | Limit |\n|----------|-------|\n| Rate limit | 1000 requests / 60 seconds per API key |\n| Concurrent jobs | Tier-based: Starter=1, Standard=5, Pro=20 |\n| Max API keys per user | 10 |\n| Max LoRAs per user (total) | 20 |\n| LoRA file format | `.safetensors` only (video: two files high_noise + low_noise; image: single file) |\n| Max LoRA file size | 2048 MB |\n| Uploaded LoRA expiry | 1 day (auto-cleaned) |\n| Saved LoRA expiry | 7 days from save (auto-cleaned) |\n| Max image file size | 20 MB |\n| Max video file size | 50 MB |\n| Image formats | JPEG, PNG, WebP |\n| Queue timeout | 3000s (auto-refund) |\n| Execution timeout | 2400s inference (auto-refund) |\n\n## Response Formats\n\n### Models Response\n```json\n{\"models\": [{\"model_id\": \"wan/v2.2-a14b/text-to-video\", \"description\": \"Wan 2.2 Text-to-Video 14B\", \"model_type\": \"video\", \"model_mode\": \"text-to-video\", \"metadata\": {}}]}\n```\n\n### Submit Response\n```json\n{\"request_id\": \"<request_id>\", \"status\": \"queued\"}\n```\n\n### Status Response\n```json\n{\"request_id\": \"...\", \"status\": \"queued|processing|completed|failed\", \"progress\": 0-100}\n```\n\n### Result Response (Video)\n```json\n{\"data\": {\"video\": {\"url\": \"...\"}, \"seed\": 12345}, \"request_id\": \"...\"}\n```\n\n### Result Response (Image)\n```json\n{\"data\": {\"image\": {\"url\": \"...\"}, \"images\": [{\"url\": \"...\"}, ...], \"seed\": 12345}, \"request_id\": \"...\"}\n```\n\n`data.images` is an array of all generated images. `data.image` is the first image (backward-compatible).\n\n### Error Codes\n| Code | Meaning |\n|------|---------|\n| 400 | Validation error |\n| 401 | Invalid API key or email not verified |\n| 402 | Insufficient credits |\n| 404 | Job not found |\n| 429 | Rate limit or concurrency limit exceeded |\n| 503 | Queue at capacity |\n\nFile v1.3.1:CHANGELOG.md\n\n# Changelog\n\nSkill package versions. The API contract version is separate and lives in\n`SKILL.md` as `metadata.api_version` — a skill release does not move it.\n\n## 1.3.1 — 2026-09-24 (API v1.2.1)\n\nReference-to-Video limits and pricing\n\n- Reference caps, both 480p and 768p: 9 images when sending only images, 5 images when\n  videos or audio are also present, 3 reference videos, 3 reference audios\n- Total reference video length is 15.1s across all reference videos\n- `reference-to-video` now uses the same rates as the rest of H3: 480p $0.0045/s,\n  768p $0.012/s, reference images $0.0075 each after the first 5\n- Bundled client validation and version string updated to match\n- API contract version v1.2.1: Reference-to-Video limits and pricing changed; no new endpoints or parameters\n\n## 1.3.0 — 2026-09-22 (API v1.2.0)\n\nReference metadata and H3 limits\n\n- Reference labels are forwarded in the same order as image, video, and audio URLs\n- 768p Ref2VA supports the same 9 image-only, 4 mixed-image, and 3 video caps as 480p\n- Correct the bundled client's stale 768p validation and 15-second video budget\n\n## 1.2.0 — 2026-09-13 (API v1.2.0)\n\nReference-to-Video\n\n- Total reference video length raised from 10s to 15s, across all reference\n  videos combined. `reference-to-video/turbo` is unchanged at 10s.\n- No other caps moved, and pricing is unchanged.\n\n## 1.2.0 — 2026-09-10\n\nPricing\n\n- MiniMax H3 (T2V / I2V / Ref2VA / Ref2VA Turbo) now shares one per-output-second\n  rate regardless of clip length: 480p $0.0045/s, 768p $0.012/s (previously 0.02 / 0.04 USD\n  for T2V/I2V and 0.025 / 0.063 USD for Ref2VA)\n- Ref2VA reference inputs: the first 5 reference images are free, then $0.0075 each\n  (previously 0.01 USD from the first image); reference audio is free (previously 0.01 USD);\n  reference video stays at the output tier rate\n- `GET /api/v1/pricing/config` adds `reference_image_free_count`; the built-in\n  client's price table and estimate formula follow it\n\nReference-to-Video Turbo\n\n- `minimax/h3/reference-to-video/turbo` added to the model table with its own reference\n  caps and 768p usage guidance (stay at or below 3 images, or 2 images plus 1 video)\n\nClient and docs\n\n- Environment section and `--base-url` help use placeholders instead of a specific host\n- `studio-layouts` on a self-hosted deployment reads `PHOSOR_WEB_BASE_URL` for the web\n  front instead of guessing ports\n- Code comments in `phosor_client.py` are English throughout\n\n## 1.1.0 — 2026-09-03\n\nNew models\n\n- MiniMax H3: text-to-video, image-to-video with an optional closing frame,\n  and Reference-to-Video, which takes reference images, videos and audio in\n  one request. 480p and 768p, 4-15 seconds, billed per output second\n- GPT Image 2 text-to-image, at five sizes: 1024x1024, 1920x1072 / 1072x1920\n  and 2560x1440 / 1440x2560\n- FLUX.2-dev text-to-image and image edit\n- Z-Image Turbo text-to-image and image-to-image, with LoRA\n- Qwen3-TTS text-to-speech, billed per character, minimum $0.003 per request\n\nNew surface\n\n- Image Studio, for e-commerce product and model photography: product suites,\n  clothing and on-model suites, selling-point layouts, A+ content, white\n  background, scene variation, and one-off tools for background removal,\n  inpainting and image translation. Async throughout: POST returns a\n  request_id, then poll until done\n\nClient\n\n- Advertises all five GPT Image 2 sizes; the previous release listed only\n  1024x1024 and rejected the other four valid sizes\n- `--version` now prints the real package version\n\nPackaging\n\n- The package and its zip are named phosor-ai-skills, matching the folder it\n  unpacks into and this skill's slug\n- Two version numbers, on purpose: the skill version covers this package\n  (commands, docs, bundled client), `metadata.api_version` covers the gateway\n  contract. Either can move without forcing the other\n- A stray compiled .pyc no longer rides along inside the archive\n\nDocs\n\n- `references/api.md` covers every endpoint the skill calls, with limits and\n  live pricing pointers. Prices are never hardcoded — call the pricing\n  endpoints for current numbers\n\n## 1.0.2 / 1.0.1 — 2026-05\n\nPublished without changelog entries; both carried the 1.0.0 text. Recorded\nhere so the gap is visible rather than implied.\n\n## 1.0.0 — 2026-03-23 — Initial release\n\n- AI content generation platform, supporting Wan 2.2 14B text-to-video and\n  image-to-video\n- Upload and use custom LoRA models for style customization\n- 16 CLI commands for job submission, uploads, status, results, and model\n  listing\n- Supports preset resolutions (480p/720p/1080p), frame alignment rules, and\n  usage quotas\n\nFile v1.3.1:skill-card.md\n\n## Description:\n\nGenerate AI videos, images and speech (text-to-video, image-to-video, reference-to-video, speech-to-video, animate, text-to-image, image-to-image, image edit, text-to-speech), bring your own LoRA models, and generate AI product/model photography for e-commerce (Image Studio) via the Phosor AI platform.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[phosor.ai](https://clawhub.ai/user/phosor.ai)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers, creators, and e-commerce operators use this skill to submit and manage Phosor AI media generation jobs, upload input assets, manage LoRA files, synthesize speech, and generate product or model photography through Image Studio.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Prompts, uploaded images, audio, video, LoRA files, and product or model photography data are sent to Phosor AI under the user's API key.\n\nMitigation: Use the skill only with data that may be uploaded to Phosor AI, and avoid passing private files unless that transfer is intended.\n\nRisk: Generation, upload, LoRA, and Image Studio commands can affect service usage and billing.\n\nMitigation: Review billing-sensitive commands, counts, model choices, and pricing commands before execution.\n\nRisk: The client supports a base URL override, which could send requests to an unintended endpoint.\n\nMitigation: Confirm that PHOSOR_BASE_URL or --base-url points to the intended Phosor endpoint before submitting jobs or uploading files.\n\n## Reference(s):\n\n- [Phosor AI API Reference](references/api.md)\n- [Phosor AI](https://phosor.ai)\n- [Phosor AI API Documentation](https://docs.phosor.ai)\n- [ClawHub Skill Page](https://clawhub.ai/phosor.ai/skills/phosor-ai-skills)\n- [ClawHub Publisher Profile](https://clawhub.ai/user/phosor.ai)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance, JSON]\n\n**Output Format:** [Markdown guidance with inline shell commands; CLI commands return JSON from Phosor AI.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Requires a Phosor API key and network access to phosor.ai; generated media URLs, job status, and billing-sensitive results are returned by the remote service.]\n\n## Skill Version(s):\n\n1.3.1 (source: server evidence, frontmatter, README, CHANGELOG, VERSION)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.2.0: 8 files, 40980 bytes\n\nFiles: _meta.json (135b), CHANGELOG.md (3917b), README.md (1986b), references/api.md (25602b), scripts/phosor_client.py (92278b), skill-card.md (2428b), SKILL.md (25774b), VERSION (6b)\n\nFile v1.2.0:SKILL.md\n\n---\nname: phosor-ai-skills\ndescription: Generate AI videos, images and speech (text-to-video, image-to-video, reference-to-video, speech-to-video, animate, text-to-image, image-to-image, image edit, text-to-speech), bring your own LoRA models, and generate AI product/model photography for e-commerce (Image Studio) via the Phosor AI platform. Use when the user wants to create videos or images from text prompts, animate images, generate lip-synced video from audio, synthesize speech from text, generate images with a custom LoRA, generate product photography or model/clothing photography for e-commerce listings, or manage generation jobs.\nlicense: MIT-0\ncompatibility: Requires Python 3.7+ and network access to phosor.ai\nmetadata:\n  author: phosor.ai\n  version: \"1.2.0\"\n  api_version: \"v1.2.0\"\n  homepage: https://phosor.ai\n---\n\n# Phosor AI\n\nGenerate AI videos and images (text-to-video, image-to-video, speech-to-video, animate, text-to-image, image-to-image), bring your own LoRA models, and generate AI product/model photography for e-commerce (Image Studio) via the Phosor AI platform.\n\nFor detailed API endpoints, parameters, pricing, and limits, see [references/api.md](references/api.md).\n\n## Setup\n\nSet your API key:\n\n```bash\nexport PHOSOR_API_KEY=\"your-api-key-here\"\n```\n\nGet an API key at [phosor.ai](https://phosor.ai) → Settings → API Keys.\n\nThe CLI script is at `scripts/phosor_client.py`. All commands output JSON to stdout.\n\n## Environments (dev vs prod — same API, only the base URL differs)\n\nIt is **one API**. Endpoints, parameters and payloads are identical across environments — **only the base URL (host:port) and scheme change**. Do not fork the client or the skill per environment; just point the same client at a different base URL.\n\n| Env | Base URL | How to target it |\n|-----|----------|------------------|\n| **Production** (default) | `https://phosor.ai` (fixed) | nothing to set; key from phosor.ai → Settings → API Keys |\n| **Dev / staging** | a deployment you operate yourself — plain http is allowed only when you opt in (e.g. `http://<dev-host>:<port>`) | `--base-url http://<dev-host:port> --allow-http` (or env `PHOSOR_BASE_URL=... PHOSOR_ALLOW_HTTP=1`). `localhost`/`127.0.0.1` needs no flag. Dev key from the dev site. |\n\n```bash\n# dev\npython3 scripts/phosor_client.py --base-url http://<dev-host>:<port> --allow-http --api-key <dev-key> check-key\n# prod (default — HTTPS enforced)\npython3 scripts/phosor_client.py --api-key <prod-key> check-key\n```\n\nNote: `studio-analyze` / `studio-suite` `--image-url` must be the **full https S3 URL** returned by `upload-image`, not the bare S3 key path — a bare key errors with `unsupported URL scheme`.\n\n## Quick Start\n\n### MiniMax H3 — Text-to-Video\n\nH3 is **duration-based**, not frame-based: it ignores `--num-frames` / `--fps` (output is\nalways 24fps) and bills per output second. Pick the frame size with\n`--resolution-tier` + `--aspect` instead of `--width/--height`.\n\n```bash\npython3 scripts/phosor_client.py submit \"A cat walking on a beach at sunset\" \\\n  --model minimax/h3/text-to-video \\\n  --resolution-tier 768p --aspect 16:9 --duration 5\n```\n\n### MiniMax H3 — Image-to-Video\n\n```bash\n# Upload first (direct URLs are not accepted); then submit with the returned s3_key\npython3 scripts/phosor_client.py upload-image /path/to/first-frame.jpg\n\npython3 scripts/phosor_client.py submit \"The person starts dancing\" \\\n  --model minimax/h3/image-to-video \\\n  --image-url \"images/img-xxx.jpg\" \\\n  --end-image-url \"images/img-yyy.jpg\" \\\n  --resolution-tier 480p --aspect 9:16 --duration 6\n```\n\n`--end-image-url` is optional and pins the closing frame.\n\n### MiniMax H3 — Reference-to-Video (Ref2VA)\n\nFeed reference **images**, **videos**, and **audio** together; refer to them positionally in\nthe prompt as `<Picture 1>`, `<Picture 2>`, … At least one of `--reference-image-urls` /\n`--reference-video-urls` is required.\n\n```bash\npython3 scripts/phosor_client.py submit \\\n  \"Use <Picture 1> and <Picture 2> as sequential keyframes; slow push-in, cinematic 35mm look.\" \\\n  --model minimax/h3/reference-to-video \\\n  --reference-image-urls \"images/a.jpg,images/b.jpg\" \\\n  --reference-audio-urls \"audio/voice.mp3\" \\\n  --resolution-tier 768p --aspect 16:9 --duration 5\n```\n\nReference inputs are billed on top of the output — see **Ref2VA Pricing** below, and\n**Ref2VA Reference Limits** for the per-tier caps (they differ between 480p and 768p).\n\n### Text-to-Video (Wan)\n\nWan is frame-based. Add `/turbo` to the model id for the fast, ~3x cheaper variant\n(it ignores `--steps` / `--guidance`).\n\n```bash\n# Submit T2V job (480p, 81 frames, 16fps)\npython3 scripts/phosor_client.py submit \"A cat walking on a beach at sunset\" \\\n  --width 854 --height 480 --num-frames 81 --fps 16\n\n# Check status\npython3 scripts/phosor_client.py status <request_id>\n\n# Get result (video URL)\npython3 scripts/phosor_client.py result <request_id>\n```\n\n### Image-to-Video\n\n**Two-step flow**: upload image first, then submit with the returned S3 key.\n\n```bash\n# Step 1: Upload image\npython3 scripts/phosor_client.py upload-image /path/to/photo.jpg\n# Returns: {\"file_id\": \"img-xxx\", \"s3_key\": \"images/img-xxx.jpg\", ...}\n\n# Step 2: Submit I2V job using the s3_key as image_url\npython3 scripts/phosor_client.py submit \"The person in the photo starts dancing\" \\\n  --image-url \"images/img-xxx.jpg\" --width 854 --height 480\n```\n\n### Text-to-Image\n\n```bash\n# Submit T2I job (1024x1024, default settings)\npython3 scripts/phosor_client.py submit \"A futuristic city skyline at dusk\" \\\n  --model qwen-image/v2512/text-to-image --width 1024 --height 1024\n\n# Generate multiple images at once (1-4)\npython3 scripts/phosor_client.py submit \"A futuristic city skyline at dusk\" \\\n  --model z-image/turbo/text-to-image --width 1024 --height 768 --num-images 4\n\n# Check status and get result (image URL)\npython3 scripts/phosor_client.py status <request_id>\npython3 scripts/phosor_client.py result <request_id>\n# Returns: {\"data\": {\"image\": {\"url\": \"...\"}, \"seed\": 12345}, ...}\n```\n\n### Image-to-Image\n\n**Two-step flow**: upload source image first, then submit with the returned S3 key.\n\n```bash\n# Step 1: Upload source image\npython3 scripts/phosor_client.py upload-image /path/to/photo.jpg\n# Returns: {\"file_id\": \"img-xxx\", \"s3_key\": \"images/img-xxx.jpg\", ...}\n\n# Step 2: Submit I2I job using the s3_key as image_url\npython3 scripts/phosor_client.py submit \"Transform into oil painting style\" \\\n  --model z-image/turbo/image-to-image --image-url \"images/img-xxx.jpg\" \\\n  --width 1024 --height 1024 --strength 0.7\n```\n\n### Image Edit (Multi-image Reference)\n\n**Two-step flow**: upload reference images first, then submit with S3 keys as `image_urls`.\n\n```bash\n# Step 1: Upload reference images (up to 3)\npython3 scripts/phosor_client.py upload-image /path/to/ref1.jpg\npython3 scripts/phosor_client.py upload-image /path/to/ref2.jpg\n\n# Step 2: Submit image-edit job\npython3 scripts/phosor_client.py submit \\\n  \"The girl in image 1 is wearing the outfit from image 2\" \\\n  --model qwen-image/v2511/image-edit \\\n  --image-urls '[\"images/img-ref1.jpg\",\"images/img-ref2.jpg\"]' \\\n  --width 1024 --height 1024\n\n# Turbo variant (faster, Lightning LoRA built-in)\npython3 scripts/phosor_client.py submit \\\n  \"The girl in image 1 is wearing the outfit from image 2\" \\\n  --model qwen-image/v2511/image-edit \\\n  --image-urls '[\"images/img-ref1.jpg\",\"images/img-ref2.jpg\"]'\n```\n\n### Speech-to-Video (S2V)\n\n**Two-step flow**: upload both audio and reference image first, then submit with the returned S3 keys.\n\n```bash\n# Step 1: Upload reference image\npython3 scripts/phosor_client.py upload-image /path/to/face.jpg\n# Returns: {\"file_id\": \"img-xxx\", \"s3_key\": \"images/img-xxx.jpg\", ...}\n\n# Step 2: Submit S2V job using the s3_key as image_url and audio URL as audio_url\npython3 scripts/phosor_client.py submit \"A person speaking naturally\" \\\n  --model wan/v2.2-a14b/speech-to-video \\\n  --image-url \"images/img-xxx.jpg\" --audio-url \"https://example.com/speech.wav\" \\\n  --width 854 --height 480\n```\n\n### Animate\n\n**Two-step flow**: upload both source video and reference image first, then submit.\n\n```bash\n# Step 1: Upload reference image\npython3 scripts/phosor_client.py upload-image /path/to/character.jpg\n# Returns: {\"file_id\": \"img-xxx\", \"s3_key\": \"images/img-xxx.jpg\", ...}\n\n# Step 2: Submit Animate job using the s3_key as image_url and video URL as video_url\npython3 scripts/phosor_client.py submit \"The character performs the dance moves\" \\\n  --model wan/v2.2-a14b/animate \\\n  --image-url \"images/img-xxx.jpg\" --video-url \"https://example.com/dance.mp4\" \\\n  --width 854 --height 480\n```\n\n### Text-to-Image (GPT Image 2)\n\nIts own resolution set and **always exactly 1 image** (`--num-images` is ignored).\n\n```bash\npython3 scripts/phosor_client.py submit \"A ceramic mug on a linen cloth, soft window light\" \\\n  --model openai/gpt-image-2/text-to-image --width 1024 --height 1024\n```\n\nAllowed sizes: **1024×1024 only**. Any other size is rejected with `400 Invalid parameters`.\nIt ignores `--num-images`, `--steps` and `--guidance` — the only parameters it accepts are\nprompt, model, width, height and seed.\n\n### Text-to-Image (FLUX.2-dev)\n\nFLUX.2-dev has its own resolution whitelist and is **fixed at 1 image per request**\n(`--num-images` and `--steps` are ignored).\n\n```bash\npython3 scripts/phosor_client.py submit \"Editorial product photo, soft window light\" \\\n  --model flux2/dev/text-to-image --width 2048 --height 1536\n```\n\n### Image Edit (FLUX.2-dev)\n\n```bash\npython3 scripts/phosor_client.py upload-image /path/to/source.jpg\n\npython3 scripts/phosor_client.py submit \"Replace the background with a marble surface\" \\\n  --model flux2/dev/image-edit --image-url \"images/img-xxx.jpg\" \\\n  --width 1024 --height 1024\n```\n\n### Text-to-Speech\n\nTTS takes `text` (not a prompt) and is billed per character with a minimum charge.\n\n```bash\npython3 scripts/phosor_client.py submit-tts \"Hello, welcome to Phosor AI.\" \\\n  --speaker Sohee --language English\n```\n\n### LoRA Upload (Custom Pre-trained)\n\n**Video LoRA** requires two .safetensors files (high_noise + low_noise). **Image LoRA** requires a single .safetensors file.\n\n```bash\n# Video LoRA: upload two .safetensors files\npython3 scripts/phosor_client.py upload-lora high_noise.safetensors low_noise.safetensors --name \"My Style\"\n\n# Image LoRA: import single .safetensors file via URL\npython3 scripts/phosor_client.py import-lora \\\n  \"https://example.com/my_lora.safetensors\" \\\n  --name \"My Image Style\"\n\n# Video LoRA: import two files via URL\npython3 scripts/phosor_client.py import-lora \\\n  \"https://example.com/high_noise.safetensors\" \\\n  \"https://example.com/low_noise.safetensors\" \\\n  --name \"My Video Style\"\n\n# Check status, then use\npython3 scripts/phosor_client.py lora-status <lora_id>\npython3 scripts/phosor_client.py submit \"A person walking\" --lora-id <lora_id>\n```\n\n## CLI Commands\n\n| Command | Description | Key Arguments |\n|---------|-------------|---------------|\n| `check-key` | Validate API key | — |\n| `submit` | Submit inference job (T2V/I2V/S2V/Animate/T2I/I2I) | `prompt`, `--width`, `--height`, `--num-frames`, `--fps`, `--steps`, `--guidance`, `--image-url`, `--audio-url`, `--video-url`, `--lora-id`, `--lora-scale`, `--loras`, `--seed`, `--negative-prompt`, `--model`, `--num-images`, `--strength`, `--output-format` |\n| `status` | Get job status | `request_id` |\n| `result` | Get job result (video or image URL) | `request_id` |\n| `poll` | Poll all pending jobs | — |\n| `list` | List locally tracked pending jobs | — |\n| `history` | Get job history | `--limit` |\n| `upload-image` | Upload image for I2V or I2I | `file` |\n| `import-image` | Import image from URL | `url`, `--filename` |\n| `upload-lora` | Upload LoRA (two .safetensors for video) | `high_noise_file`, `low_noise_file`, `--name` |\n| `import-lora` | Import LoRA from URLs (one or two files) | `high_noise_url`, `[low_noise_url]`, `--name` |\n| `loras` | List LoRA models | `--limit`, `--offset` |\n| `lora-status` | Get LoRA upload/import status | `lora_id` |\n| `save-lora` | Activate a LoRA (extends expiry to 7 days) | `lora_id`, `--name` |\n| `delete-lora` | Delete a LoRA model | `lora_id` |\n| `submit-tts` | Submit a text-to-speech job (Qwen3-TTS) — keys off `text`, not a prompt | `text`, `--speaker`, `--language`, `--seed`, `--temperature`, `--top-p`, `--top-k`, `--repetition-penalty` |\n| `models` | List available video/image models (static offline reference) | — |\n| `studio-features` | List Image Studio endpoints, fields, billing (static offline reference) | — |\n| `studio-pricing` | Get live Image Studio pricing | — |\n| `studio-analyze` | AI-analyze a product/garment image or reference URL (freemium) | `--target agent\\|product\\|model\\|reference`, `--image-url`, `--url`, `--prompt`, `--language` |\n| `studio-layouts` | List the layout template library (query, then select) — static asset on the **web front** (phosor.ai), not a `/api/v1` endpoint; needs no key; on a self-hosted deployment set `PHOSOR_WEB_BASE_URL` if the web front is not the gateway host | `--module` (product\\|clothing), `--type` (selling_point\\|aplus\\|white_bg\\|scene\\|closeup\\|size_chart) |\n| `studio-suite` | Generate a product image suite | `--image-url`, `--layout-types`, `--count-per-type`, `--custom-suggestions`, `--template-ids` (ids from `studio-layouts`, auto-expanded to custom_suggestions like the UI's manual pick — use this to get **text-callout selling-point / A+ layouts** and model templates), `--product-info`, `--aspect-ratio`, `--gen-language`, `--model`, `--same-style-reference` |\n| `studio-clothing-suite` | Generate a model/garment image suite | `--image-urls`, `--main-image-types`, `--aplus-types`, `--product-info`, `--brand-config`, `--aspect-ratio`, `--gen-language`, `--model`, `--same-style-reference` |\n| `studio-status` | Get Image Studio job status (separate id space, same `request_id` key) | `request_id` |\n| `studio-cancel` | Cancel a running generation — queued images refunded, already-generating ones charged | `request_id` |\n| `studio-my-works` | List past Image Studio generations | `--task-type`, `--limit`, `--offset` |\n| `studio-call` | Generic call for any other Image Studio endpoint (remove-bg, replace, inpaint, erase, handheld, translate, outpaint, recolor, enhance, upscale, scene-compose, scene-variation, real-model-swap, mannequin-swap, model-scene-swap, ai-outfit, pose-variation, ai-wearable) | `method`, `path`, `--json` |\n\n## Image Studio (Product & Model Photography)\n\nImage Studio is a separate product surface for e-commerce sellers — AI product photography and model/clothing photography — reached through the **same gateway and API key** as video/LoRA, under the `/api/v1/image-studio` prefix. It has its own async namespace - the same key name `request_id`, but a **separate id space**: an Image Studio `request_id` is not valid on `/api/v1/inference/status/...` and vice versa - and its own pricing (flat per-image rate + freemium analyze quota, not per-frame). Full endpoint/parameter reference: [references/api.md](references/api.md#image-studio-product--model-photography--separate-product-surface).\n\n### Quick Start: Product Suite\n\n```bash\n# 1. Upload the product photo\npython3 scripts/phosor_client.py upload-image /path/to/product.jpg\n# Returns: {\"file_id\": \"img-xxx\", \"s3_key\": \"images/img-xxx.jpg\", ...}\n\n# 2. (Optional) AI-analyze it first for richer generation context\npython3 scripts/phosor_client.py studio-analyze --target product --image-url \"images/img-xxx.jpg\"\n\n# 3. Generate a product image suite\npython3 scripts/phosor_client.py studio-suite --image-url \"images/img-xxx.jpg\" \\\n  --layout-types \"white_background,lifestyle_scene\" --count-per-type 2\n\n# 4. Poll for the result\npython3 scripts/phosor_client.py studio-status <request_id>\n```\n\n### Quick Start: Clothing/Model Suite\n\n```bash\npython3 scripts/phosor_client.py upload-image /path/to/garment.jpg\npython3 scripts/phosor_client.py studio-clothing-suite \\\n  --image-urls \"images/img-xxx.jpg\" \\\n  --main-image-types '{\"model_shot\":2,\"selling_point\":1}' \\\n  --aplus-types '{\"standard_aplus\":1}'\npython3 scripts/phosor_client.py studio-status <request_id>\n```\n\n### Quick Start: One-off Edits (remove-bg, inpaint, translate, etc.)\n\nThe long tail of single-purpose editing endpoints doesn't get a dedicated subcommand — use `studio-call` with the exact field names from [references/api.md](references/api.md#image-studio-product--model-photography--separate-product-surface):\n\n```bash\npython3 scripts/phosor_client.py studio-call POST /product/remove-bg \\\n  --json '{\"image_url\": \"images/img-xxx.jpg\", \"count\": 2}'\npython3 scripts/phosor_client.py studio-status <request_id>\n```\n\n### Key facts\n\n- **Every Image Studio call requires `X-API-Key`** (`PHOSOR_API_KEY`), including `GET /pricing` — there is no unauthenticated endpoint under this prefix.\n- **All generation/analyze endpoints are async**: POST returns `{\"request_id\": \"...\", \"status\": \"pending\"}`; poll `studio-status <request_id>` until `status` is `\"done\"`, `\"error\"` or `\"cancelled\"`. Earlier revisions of this skill said the key was `job_id` - it is not, and reading it yields `undefined`. Image Studio ids live in a **separate id space** from video/LoRA: `poll`/`status`/`result` will not accept an Image Studio `request_id`.\n- **Cancelling**: `POST /jobs/{request_id}/cancel` stops a running generation. Images still queued are refunded; images already generating are charged and cannot be stopped; images already delivered bill once through the normal path. The response reports the split as `refunded_queued`, `charged_running` and `already_done`, and the task then polls as `status: \"cancelled\"` - not an error.\n- **Pricing is per-image, not per-frame**: call `studio-pricing` for the live rate. Partial success (e.g. 3 of 5 images) bills only the successes.\n- **Analyze is freemium**: `agent/analyze`, `product/analyze`, `model/analyze` share a daily free quota before per-call billing kicks in.\n- **`model_attrs` matters for model-photography endpoints** (real-model-swap, mannequin-swap, ai-outfit, ai-wearable) — pass `{gender, age_group, ethnicity, skin_tone, hair_color}` explicitly; it is not reliably inferred from the source image alone.\n- Run `studio-features` for the full offline endpoint/field catalog without leaving the terminal.\n\n## Key Constraints\n\n### Video Resolutions (exact pairs only)\n\n| Preset | Width × Height | Max Frames (turbo) | Max Frames (standard) |\n|--------|---------------|-------------------|----------------------|\n| 480p landscape | 854 × 480 | 161 | 161 |\n| 480p portrait | 480 × 854 | 161 | 161 |\n| 720p landscape | 1280 × 720 | 161 | 161 |\n| 720p portrait | 720 × 1280 | 161 | 161 |\n| 1080p landscape | 1920 × 1080 | 153 | **81** |\n| 1080p portrait | 1080 × 1920 | 153 | **81** |\n\n> Standard (non-turbo) mode: 1080p is capped at 81 frames due to generation time limits.\n\n### S2V / Animate Video Resolutions (exact pairs only)\n\n| Preset | Width x Height | Max Frames |\n|--------|---------------|------------|\n| 480p landscape | 854 x 480 | 161 |\n| 480p portrait | 480 x 854 | 161 |\n| 512p square | 512 x 512 | 161 |\n| 720p landscape | 1280 x 720 | 161 |\n| 720p portrait | 720 x 1280 | 161 |\n\n### Image Resolutions (exact pairs only)\n\n| Preset | Width × Height |\n|--------|---------------|\n| Square small | 512 × 512 |\n| Square | 1024 × 1024 |\n| Landscape | 1024 × 768 |\n| Portrait | 768 × 1024 |\n| Wide landscape | 1280 × 768 |\n| Tall portrait | 768 × 1280 |\n\n### MiniMax H3 Frame Sizes (`--resolution-tier` + `--aspect`)\n\n| Tier | 16:9 | 4:3 | 1:1 | 3:4 | 9:16 |\n|------|------|-----|-----|-----|------|\n| 480p | 832 × 480 | 640 × 480 | 480 × 480 | 480 × 640 | 480 × 832 |\n| 768p | 1344 × 768 | 1024 × 768 | 768 × 768 | 768 × 1024 | 768 × 1344 |\n\n`duration` is 4–15 seconds (default 5). Output FPS is fixed at 24 and\n`frames_per_second` is ignored.\n\n### Ref2VA Reference Limits (`minimax/h3/reference-to-video`)\n\nCaps differ per tier, and the image cap is higher when you send **only** images:\n\n| Limit | 480p | 768p |\n|-------|------|------|\n| Reference images (with videos/audio present) | 4 | 4 |\n| Reference images (images only) | 9 | 9 |\n| Reference videos | 3 | 3 |\n| Reference audios | 3 | 3 |\n\n| Limit | Value |\n|-------|-------|\n| Total reference video length | 15 s (across all reference videos) |\n| Reference video FPS ceiling | 24 |\n| Reference audio length | 10 s each |\n| Reference image longest edge | 2048 px |\n| Reference image aspect ratio | ≤ 4.0 |\n\n### FLUX.2-dev Resolutions (exact pairs only)\n\n| Width × Height |\n|---------------|\n| 2048 × 1536 · 1536 × 2048 |\n| 2048 × 1152 · 1152 × 2048 |\n| 2048 × 2048 · 1024 × 1024 |\n\n> FLUX.2-dev does not accept the general image resolution list above, and always\n> returns exactly 1 image.\n\n### Frame Alignment (video only)\n\nFrames must follow `1 + 4*k` where `k >= 1` (e.g. 5, 9, 13, ... 81, 85, ...). Server auto-aligns down.\n\n### Video Inference Parameters\n\n| Parame\n\nArchive v1.0.2: 6 files, 12394 bytes\n\nFiles: README.md (1088b), references/api.md (4317b), scripts/phosor_client.py (28214b), skill-card.md (2548b), SKILL.md (5738b), _meta.json (135b)\n\nArchive v1.0.1: 5 files, 11318 bytes\n\nFiles: README.md (1088b), references/api.md (4317b), scripts/phosor_client.py (28643b), SKILL.md (5746b), _meta.json (135b)\n\nArchive v1.0.0: 5 files, 11103 bytes\n\nFiles: README.md (1088b), references/api.md (4317b), scripts/phosor_client.py (28643b), SKILL.md (5335b), _meta.json (135b)","readmeExcerpt":"Skill: Phosor AI Owner: phosor.ai Summary: Generate AI videos, images and speech (text-to-video, image-to-video, reference-to-video, speech-to-video, animate, text-to-image, image-to-image, image edit, text-to-speech), bring your own LoRA models, and generate AI product/model photography for e-commerce (Image Studio) via the Phosor AI platform. Use when the user wants to create videos or images from text prompts, ani","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"export PHOSOR_API_KEY=\"your-api-key-here\""},{"language":"bash","snippet":"python3 scripts/phosor_client.py --api-key <your-key> check-key"},{"language":"bash","snippet":"python3 scripts/phosor_client.py submit \"A cat walking on a beach at sunset\" \\\n  --model minimax/h3/text-to-video \\\n  --resolution-tier 768p --aspect 16:9 --duration 5"},{"language":"bash","snippet":"# Upload first (direct URLs are not accepted); then submit with the returned s3_key\npython3 scripts/phosor_client.py upload-image /path/to/first-frame.jpg\n\npython3 scripts/phosor_client.py submit \"The person starts dancing\" \\\n  --model minimax/h3/image-to-video \\\n  --image-url \"images/img-xxx.jpg\" \\\n  --end-image-url \"images/img-yyy.jpg\" \\\n  --resolution-tier 480p --aspect 9:16 --duration 6"},{"language":"bash","snippet":"python3 scripts/phosor_client.py submit \\\n  \"Use <Picture 1> and <Picture 2> as sequential keyframes; slow push-in, cinematic 35mm look.\" \\\n  --model minimax/h3/reference-to-video \\\n  --reference-image-urls \"images/a.jpg,images/b.jpg\" \\\n  --reference-audio-urls \"audio/voice.mp3\" \\\n  --resolution-tier 768p --aspect 16:9 --duration 5"},{"language":"bash","snippet":"# Submit T2V job (480p, 81 frames, 16fps)\npython3 scripts/phosor_client.py submit \"A cat walking on a beach at sunset\" \\\n  --width 854 --height 480 --num-frames 81 --fps 16\n\n# Check status\npython3 scripts/phosor_client.py status <request_id>\n\n# Get result (video URL)\npython3 scripts/phosor_client.py result <request_id>"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: phosor-ai-skills\ndescription: Generate AI videos, images and speech (text-to-video, image-to-video, reference-to-video, speech-to-video, animate, text-to-image, image-to-image, image edit, text-to-speech), bring your own LoRA models, and generate AI product/model photography for e-commerce (Image Studio) via the Phosor AI platform. Use when the user wants to create videos or images from text prompts, animate images, generate lip-synced video from audio, synthesize speech from text, generate images with a custom LoRA, generate product photography or model/clothing photography for e-commerce listings, or manage generation jobs.\nlicense: MIT-0\ncompatibility: Requires Python 3.7+ and network access to phosor.ai\nmetadata:\n  author: phosor.ai\n  version: \"1.3.2\"\n  api_version: \"v1.2.2\"\n  homepage: https://phosor.ai\n---\n\n# Phosor AI\n\nGenerate AI videos and images (text-to-video, image-to-video, speech-to-video, animate, text-to-image, image-to-image), bring your own LoRA models, and generate AI product/model photography for e-commerce (Image Studio) via the Phosor AI platform.\n\nFor detailed API endpoints, parameters, pricing, and limits, see [references/api.md](references/api.md).\n\n## Setup\n\nSet your API key:\n\n```bash\nexport PHOSOR_API_KEY=\"your-api-key-here\"\n```\n\nGet an API key at [phosor.ai](https://phosor.ai) → Settings → API Keys.\n\nThe CLI script is at `scripts/phosor_client.py`. All commands output JSON to stdout.\n\n## Base URL\n\nThe client talks to `https://phosor.ai` by default, over HTTPS. Get your key from phosor.ai → Settings → API Keys.\n\n```bash\npython3 scripts/phosor_client.py --api-key <your-key> check-key\n```\n\nTo point the same client at another Phosor endpoint, pass `--base-url <url>` (or set `PHOSOR_BASE_URL`).\nHTTPS is required; plain `http://` is accepted only for `localhost`, or when you explicitly add `--allow-http`.\n\nNote: `studio-analyze` / `studio-suite` `--image-url` must be the **full https S3 URL** returned by `upload-image`, not the bare S3 key path — a bare key errors with `unsupported URL scheme`.\n\n## Quick Start\n\n### MiniMax H3 — Text-to-Video\n\nH3 is **duration-based**, not frame-based: it ignores `--num-frames` / `--fps` (output is\nalways 24fps) and bills per output second. Pick the frame size with\n`--resolution-tier` + `--aspect` instead of `--width/--height`.\n\n```bash\npython3 scripts/phosor_client.py submit \"A cat walking on a beach at sunset\" \\\n  --model minimax/h3/text-to-video \\\n  --resolution-tier 768p --aspect 16:9 --duration 5\n```\n\n### MiniMax H3 — Image-to-Video\n\n```bash\n# Upload first (direct URLs are not accepted); then submit with the returned s3_key\npython3 scripts/phosor_client.py upload-image /path/to/first-frame.jpg\n\npython3 scripts/phosor_client.py submit \"The person starts dancing\" \\\n  --model minimax/h3/image-to-video \\\n  --image-url \"images/img-xxx.jpg\" \\\n  --end-image-url \"images/img-yyy.jpg\" \\\n  --resolution-tier 480p --aspect 9:16 --duration 6\n```\n\n`--end-image-url` is optional and pins the cl"},{"path":"README.md","content":"# Phosor AI — Agent Skill\n\n**Skill version 1.3.2** · API v1.2.2 · updated 2026-09-25\n\nTwo version numbers, on purpose: the skill version tracks this package\n(commands, docs, the bundled client), the API version tracks the gateway\ncontract. A skill release that only rewords docs or fixes the client does not\nmove the API version, and a gateway change does not force a skill release.\n\nVerify what you installed: `python3 scripts/phosor_client.py --version`\n\n## Quick Start\n\n```bash\nexport PHOSOR_API_KEY=\"your-key\"\n\n# Text-to-Video\npython3 scripts/phosor_client.py submit \"A cat walking on a beach\" --width 854 --height 480\n\n# Image-to-Video (two-step: upload then submit)\npython3 scripts/phosor_client.py upload-image photo.jpg\npython3 scripts/phosor_client.py submit \"The scene comes alive\" --image-url \"images/img-xxx.jpg\"\n\n# Check status / get result\npython3 scripts/phosor_client.py status <request_id>\npython3 scripts/phosor_client.py result <request_id>\n\n# LoRA Upload (custom pre-trained)\npython3 scripts/phosor_client.py upload-lora high_noise.safetensors low_noise.safetensors\n\npython3 scripts/phosor_client.py save-lora <lora_id>\n\n# Image Studio: AI product photography (separate product surface, same API key)\npython3 scripts/phosor_client.py upload-image product.jpg\npython3 scripts/phosor_client.py studio-suite --image-url \"<s3_url from upload>\" \\\n  --layout-types \"white_background,lifestyle_scene\" --count-per-type 2\npython3 scripts/phosor_client.py studio-status <job_id>\n```\n\nSee [SKILL.md](SKILL.md#image-studio-product--model-photography) for the full Image Studio quick start (clothing/model suite, one-off edits like remove-bg/inpaint/translate).\n\n## Requirements\n\n- Python 3.7+ (stdlib only, no pip install needed)\n- `PHOSOR_API_KEY` environment variable\n\n## Commands\n\nRun `python3 scripts/phosor_client.py --help` for all 31 commands (23 video/LoRA + 8 Image Studio).\n\n## Links\n\n- [Phosor AI](https://phosor.ai)\n- [API Documentation](https://docs.phosor.ai)"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7dkx5wmnapr9y2x91m6qnb8s82w04e\",\n  \"slug\": \"phosor-ai-skills\",\n  \"version\": \"1.3.2\",\n  \"publishedAt\": 1790399871799\n}"},{"path":"references/api.md","content":"# Phosor AI API Reference\n\nBase URL: `https://phosor.ai`\n\nAll endpoints require `X-API-Key` header unless noted otherwise.\n\n## Endpoints\n\n### Models\n\n| Method | Path | Auth | Description |\n|--------|------|------|-------------|\n| GET | `/api/v1/models` | None | List available models |\n\n### Inference\n\n| Method | Path | Description |\n|--------|------|-------------|\n| POST | `/api/v1/inference/submit` | Submit video or image generation job |\n| GET | `/api/v1/inference/status/{request_id}` | Get job status + progress |\n| GET | `/api/v1/inference/result/{request_id}` | Get completed result (video or image URL) |\n| GET | `/api/v1/inference/history` | Get user's job history |\n\n### Storage — Image / Audio\n\n| Method | Path | Description |\n|--------|------|-------------|\n| POST | `/api/v1/storage/image/upload` | Multipart image or audio upload (images: jpg/png/webp; audio for S2V: mp3/wav/flac/aac/ogg/m4a) |\n| POST | `/api/v1/storage/image/import` | Import from public URL |\n\n### Storage — LoRA\n\n| Method | Path | Description |\n|--------|------|-------------|\n| POST | `/api/v1/storage/lora/upload` | Upload two .safetensors files (video LoRA: high_noise + low_noise) |\n| POST | `/api/v1/storage/lora/import` | Import from HTTPS URLs (video: two files, image: single file) |\n\n### LoRA Management\n\n| Method | Path | Description |\n|--------|------|-------------|\n| GET | `/api/v1/loras` | List LoRA models |\n| GET | `/api/v1/loras/{lora_id}` | Get single LoRA details |\n| GET | `/api/v1/loras/{lora_id}/status` | Get processing status |\n| POST | `/api/v1/loras/{lora_id}/save` | Activate a LoRA (extend expiry to 7 days) |\n| DELETE | `/api/v1/loras/{lora_id}` | Soft delete |\n\n### Image Studio (product & model photography — separate product surface)\n\nAll paths below are relative to `/api/v1/image-studio` (e.g. the full path for `/product/suite` is `/api/v1/image-studio/product/suite`). **Every path requires `X-API-Key`, including `GET /pricing`** — there is no unauthenticated Image Studio endpoint.\n\nAll generation/analyze endpoints are async: POST returns `{\"request_id\": \"...\", \"status\": \"pending\"}` immediately; poll `GET /jobs/{request_id}` until `status` is `\"done\"`, `\"error\"` or `\"cancelled\"`.\n\n> **The async key is `request_id`, not `request_id`.** Earlier revisions of this document said\n> `request_id`; that key is never present in a response. Reading it yields `undefined`, and the\n> poll then never resolves. Image Studio's `request_id` is a separate namespace from the\n> video/LoRA `request_id` — do not pass one to the other's status endpoint.\n\n| Method | Path | Description |\n|--------|------|-------------|\n| POST | `/agent/analyze` | AI-analyze an image for the Agent-image workflow (freemium) |\n| POST | `/product/analyze` | AI-analyze a product image ahead of `product/suite` (freemium) |\n| POST | `/model/analyze` | AI-analyze a garment image ahead of `model/clothing-suite` (freemium) |\n| POST | `/product/reference/analyze` | Analyze a reference product page/image for s"},{"path":"CHANGELOG.md","content":"# Changelog\n\nSkill package versions. The API contract version is separate and lives in\n`SKILL.md` as `metadata.api_version` — a skill release does not move it.\n\n## 1.3.2 — 2026-09-25 (API v1.2.2)\n\n`minimax/h3/reference-to-video/turbo` removed\n\n- Requests to `minimax/h3/reference-to-video/turbo` now return an unknown-model error\n- Use `minimax/h3/reference-to-video` instead — same inputs and pricing structure\n- `reference-to-video` (non-turbo) is unaffected\n- API contract version v1.2.2: a model id was removed; no other endpoints or parameters changed\n\n## 1.3.1 — 2026-09-24 (API v1.2.1)\n\nReference-to-Video limits and pricing\n\n- Reference caps, both 480p and 768p: 9 images when sending only images, 5 images when\n  videos or audio are also present, 3 reference videos, 3 reference audios\n- Total reference video length is 15.1s across all reference videos\n- `reference-to-video` now uses the same rates as the rest of H3: 480p $0.0045/s,\n  768p $0.012/s, reference images $0.0075 each after the first 5\n- Bundled client validation and version string updated to match\n- API contract version v1.2.1: Reference-to-Video limits and pricing changed; no new endpoints or parameters\n\n## 1.3.0 — 2026-09-22 (API v1.2.0)\n\nReference metadata and H3 limits\n\n- Reference labels are forwarded in the same order as image, video, and audio URLs\n- 768p Ref2VA supports the same 9 image-only, 4 mixed-image, and 3 video caps as 480p\n- Correct the bundled client's stale 768p validation and 15-second video budget\n\n## 1.2.0 — 2026-09-13 (API v1.2.0)\n\nReference-to-Video\n\n- Total reference video length raised from 10s to 15s, across all reference\n  videos combined. `reference-to-video/turbo` is unchanged at 10s.\n- No other caps moved, and pricing is unchanged.\n\n## 1.2.0 — 2026-09-10\n\nPricing\n\n- MiniMax H3 (T2V / I2V / Ref2VA / Ref2VA Turbo) now shares one per-output-second\n  rate regardless of clip length: 480p $0.0045/s, 768p $0.012/s (previously 0.02 / 0.04 USD\n  for T2V/I2V and 0.025 / 0.063 USD for Ref2VA)\n- Ref2VA reference inputs: the first 5 reference images are free, then $0.0075 each\n  (previously 0.01 USD from the first image); reference audio is free (previously 0.01 USD);\n  reference video stays at the output tier rate\n- `GET /api/v1/pricing/config` adds `reference_image_free_count`; the built-in\n  client's price table and estimate formula follow it\n\nReference-to-Video Turbo\n\n- `minimax/h3/reference-to-video/turbo` added to the model table with its own reference\n  caps and 768p usage guidance (stay at or below 3 images, or 2 images plus 1 video)\n\nClient and docs\n\n- Environment section and `--base-url` help use placeholders instead of a specific host\n- `studio-layouts` on a self-hosted deployment reads `PHOSOR_WEB_BASE_URL` for the web\n  front instead of guessing ports\n- Code comments in `phosor_client.py` are English throughout\n\n## 1.1.0 — 2026-09-03\n\nNew models\n\n- MiniMax H3: text-to-video, image-to-video with an optional closing frame,\n  and Reference-to-Video, which tak"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":2204,"uniquenessScore":31,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T22:25:41.580Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T22:25:41.580Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T00:34:35.766Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}