{"id":"af1bd0e2-7bbf-4009-8d1a-916dc06f2951","entityType":"agent","slug":"clawhub-chrischall-gemini-mcp","name":"gemini-mcp","canonicalUrl":"https://www.xpersona.co/agent/clawhub-chrischall-gemini-mcp","canonicalPath":"/agent/clawhub-chrischall-gemini-mcp","generatedAt":"2026-10-10T00:15:04.391Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T10:26:00.303Z","emptyReason":null},"description":"Generate and edit images, video, and music with Google Gemini models via MCP. Use when the user asks to generate, create, or edit images (Gemini / Nano Banana), produce a consistent set of images, compose/blend multiple images, generate a short video (text→video or image→video, via the omni model), or generate music/audio clips (via Lyria). Triggers on phrases like \"generate an image of\", \"edit this image with Gemini\", \"create a set of consistent images\", \"make a video of\", \"generate a video\", \"generate music\", \"make a song/audio clip\", \"use Nano Banana to make\", or any request to produce images, video, or music via the Gemini API. Requires the @chrischall/gemini-mcp package installed and the gemini server registered (see Setup below).","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 3K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s17cjx1a349nz5apaqp02vgz4h85728z:gemini-mcp","sourceUrl":"https://clawhub.ai/chrischall/gemini-mcp","homepage":"https://clawhub.ai/chrischall/skills/gemini-mcp","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/chrischall/gemini-mcp","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/chrischall/skills/gemini-mcp","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":69,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"gemini-mcp technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T10:26:00.303Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T10:26:00.303Z","emptyReason":null},"stars":null,"forks":null,"downloads":2976,"packageName":null,"latestVersion":"2.3.3","tractionLabel":"3K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T10:26:00.303Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T10:26:00.303Z","lastCrawledAt":"2026-10-09T10:26:00.303Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T10:26:00.303Z","lastVerifiedAt":null,"highlights":[{"version":"2.3.3","createdAt":"2026-10-07T13:37:12.402Z","changelog":"gemini-mcp 2.3.3 - Documentation updated in SKILL.md. - Removed skill-card.md file.","fileCount":3,"zipByteSize":11183},{"version":"2.3.2","createdAt":"2026-10-05T02:50:01.686Z","changelog":"- Removed the file skill-card.md from the project. - No other changes to functionality or configuration.","fileCount":3,"zipByteSize":11097},{"version":"2.3.1","createdAt":"2026-10-03T01:40:19.628Z","changelog":"- Removed the file: skill-card.md. - No other user-facing changes in this version.","fileCount":3,"zipByteSize":11090},{"version":"2.3.0","createdAt":"2026-09-25T15:51:02.036Z","changelog":"- Removed the skill-card.md file. - No other functional or documented changes in this version.","fileCount":3,"zipByteSize":11133},{"version":"2.2.0","createdAt":"2026-09-24T15:12:52.611Z","changelog":"## gemini-mcp 2.2.0 - Added confirmation token parameters (`confirmToken?`) for file uploads (`gemini_upload_file`) and file deletion (`gemini_delete_file`). - Updated documentation in SKILL.md to reflect new confirmation flow for file operations. - Removed `skill-card.md` file.","fileCount":3,"zipByteSize":11326},{"version":"2.1.4","createdAt":"2026-09-23T21:39:43.416Z","changelog":"gemini-mcp v2.1.4 changelog - Removed the file skill-card.md from the project. - No changes to core functionality or usage.","fileCount":3,"zipByteSize":10864},{"version":"2.1.3","createdAt":"2026-09-23T15:48:05.793Z","changelog":"gemini-mcp 2.1.3 - Removed the file: skill-card.md. - No changes made to functionality or configuration. - Documentation and setup instructions remain unchanged.","fileCount":3,"zipByteSize":10867},{"version":"2.1.2","createdAt":"2026-09-21T05:03:04.560Z","changelog":"gemini-mcp v2.1.2 - Removed the file: skill-card.md - No functional or API changes in this release.","fileCount":3,"zipByteSize":10941}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17cjx1a349nz5apaqp02vgz4h85728z:gemini-mcp","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s17cjx1a349nz5apaqp02vgz4h85728z:gemini-mcp` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/chrischall/gemini-mcp before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chrischall-gemini-mcp/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chrischall-gemini-mcp/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chrischall-gemini-mcp/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-chrischall-gemini-mcp/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-chrischall-gemini-mcp/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-chrischall-gemini-mcp/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T00:15:04.384Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chrischall-gemini-mcp/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chrischall-gemini-mcp/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chrischall-gemini-mcp/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chrischall-gemini-mcp/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T10:26:00.303Z","emptyReason":null},"readme":"Skill: gemini-mcp\n\nOwner: chrischall\n\nSummary: Generate and edit images, video, and music with Google Gemini models via MCP. Use when the user asks to generate, create, or edit images (Gemini / Nano Banana), produce a consistent set of images, compose/blend multiple images, generate a short video (text→video or image→video, via the omni model), or generate music/audio clips (via Lyria). Triggers on phrases like \"generate an image of\", \"edit this image with Gemini\", \"create a set of consistent images\", \"make a video of\", \"generate a video\", \"generate music\", \"make a song/audio clip\", \"use Nano Banana to make\", or any request to produce images, video, or music via the Gemini API. Requires the @chrischall/gemini-mcp package installed and the gemini server registered (see Setup below).\n\nTags: latest:2.3.3\n\nVersion history:\n\nv2.3.3 | 2026-10-07T13:37:12.402Z | auto\n\ngemini-mcp 2.3.3\n\n- Documentation updated in SKILL.md.\n- Removed skill-card.md file.\n\nv2.3.2 | 2026-10-05T02:50:01.686Z | auto\n\n- Removed the file skill-card.md from the project.\n- No other changes to functionality or configuration.\n\nv2.3.1 | 2026-10-03T01:40:19.628Z | auto\n\n- Removed the file: skill-card.md.\n- No other user-facing changes in this version.\n\nv2.3.0 | 2026-09-25T15:51:02.036Z | auto\n\n- Removed the skill-card.md file.\n- No other functional or documented changes in this version.\n\nv2.2.0 | 2026-09-24T15:12:52.611Z | auto\n\n## gemini-mcp 2.2.0\n\n- Added confirmation token parameters (`confirmToken?`) for file uploads (`gemini_upload_file`) and file deletion (`gemini_delete_file`).\n- Updated documentation in SKILL.md to reflect new confirmation flow for file operations.\n- Removed `skill-card.md` file.\n\nv2.1.4 | 2026-09-23T21:39:43.416Z | auto\n\ngemini-mcp v2.1.4 changelog\n\n- Removed the file skill-card.md from the project.\n- No changes to core functionality or usage.\n\nv2.1.3 | 2026-09-23T15:48:05.793Z | auto\n\ngemini-mcp 2.1.3\n\n- Removed the file: skill-card.md.\n- No changes made to functionality or configuration.\n- Documentation and setup instructions remain unchanged.\n\nv2.1.2 | 2026-09-21T05:03:04.560Z | auto\n\ngemini-mcp v2.1.2\n\n- Removed the file: skill-card.md\n- No functional or API changes in this release.\n\nv2.1.1 | 2026-09-21T04:14:09.681Z | auto\n\n- Removed the sample skill-card.md file.\n- No changes to functionality or user-facing features.\n\nv2.1.0 | 2026-09-20T02:51:58.760Z | auto\n\n## gemini-mcp 2.1.0\n\n- Removed the file: `skill-card.md`\n- No other functional or API changes noted in this version.\n\nv2.0.0 | 2026-09-19T11:19:36.642Z | auto\n\ngemini-mcp 2.0.0\n\n- Removed the skill-card.md file from the project.\n- No user-facing feature or behavior changes.\n\nv1.14.2 | 2026-09-14T14:16:53.514Z | auto\n\n- Removed the file: skill-card.md\n- No functional or feature changes—documentation or metadata update only.\n\nv1.14.1 | 2026-09-10T17:53:59.038Z | auto\n\n- Removed the skill overview/readme file skill-card.md.\n- No behavioral changes; documentation file cleanup only.\n\nv1.14.0 | 2026-09-09T19:21:17.568Z | auto\n\ngemini-mcp v1.14.0\n\n- Updated documentation in SKILL.md with clarifications about context cost and behavior for `images_base64` reference images.\n- Added details on first-sighting upload and recommended flow for `images_base64` (switch to `images_file_uris` after initial use).\n- Removed the file skill-card.md from the project.\n\nv1.13.0 | 2026-09-04T22:22:17.165Z | auto\n\n- Removed the file: skill-card.md\n- No other changes to skill description or functionality.\n\nv1.12.0 | 2026-08-31T00:22:19.904Z | auto\n\n- Removed the file skill-card.md.\n- No functional or feature changes in this version.\n\nv1.11.1 | 2026-08-28T21:07:26.786Z | auto\n\n- Removed the skill-card.md file.\n- No changes to core functionality or documentation content.\n- Maintenance update only; no user-facing differences.\n\nv1.11.0 | 2026-08-28T11:34:32.315Z | auto\n\n- Removed the skill-card.md file.\n- No other changes.\n\nv1.10.0 | 2026-08-26T00:29:04.958Z | auto\n\n- Removed the file: skill-card.md\n- No changes to functionality or features; only documentation cleanup.\n\nv1.9.0 | 2026-08-24T02:08:35.621Z | auto\n\ngemini-mcp 1.9.0 Changelog\n\n- Removed the file: skill-card.md.\n- No other user-facing changes.\n\nv1.8.0 | 2026-08-22T12:24:36.214Z | auto\n\n- Removed the file: skill-card.md\n- No changes to functionality or configuration; only documentation cleanup.\n\nv1.7.0 | 2026-08-19T02:54:33.502Z | auto\n\n- Removed the file: skill-card.md\n- No changes to skill functionality or documentation—maintenance update only.\n\nv1.6.2 | 2026-08-18T02:14:37.347Z | auto\n\n- Removed the file: skill-card.md\n- No other changes to functionality or documentation\n- Minor cleanup of repository contents\n\nv1.6.1 | 2026-08-08T02:47:42.191Z | auto\n\n- Removed the skill-card.md file from the project.\n- No changes to functionality or configuration.\n\nv1.6.0 | 2026-08-07T14:49:06.310Z | auto\n\ngemini-mcp v1.6.0\n\n- Updated documentation in SKILL.md with minor clarifications and typo fixes.\n- Updated Files API section: \"hosted connector\" phrasing replaced with \"hosted deployment\" and route example updated.\n- Removed skill-card.md file from the project.\n\nv1.5.1 | 2026-08-01T13:07:23.152Z | auto\n\ngemini-mcp 1.5.1\n\n- Removed the sample file \"skill-card.md\" from the repository.\n- No changes to functionality or usage.\n\nv1.5.0 | 2026-08-01T12:35:27.612Z | auto\n\ngemini-mcp 1.5.0\n\n- Removed the skill-card.md file from the project.\n- No changes to functionality or APIs.\n\nv1.4.0 | 2026-07-30T14:46:52.745Z | auto\n\n- skill-card.md file removed to simplify skill packaging.\n- No changes to tool behavior or APIs.\n- Documentation and usage remain unchanged.\n\nv1.3.0 | 2026-07-29T13:35:19.632Z | auto\n\n- Removed the sample file skill-card.md.\n\nv1.2.0 | 2026-07-29T10:53:36.054Z | auto\n\ngemini-mcp v1.2.0\n\n- Updated documentation in SKILL.md.\n- Removed the skill-card.md file.\n- Clarified reference image usage applies to all relevant tools (including video/music generation).\n- No functional changes to the skill itself; this is a documentation and cleanup release.\n\nv1.1.0 | 2026-07-29T03:37:48.430Z | auto\n\n- Added support for new reference image parameters: images_url and images_file_uris (and respective master_images_*), with docs on the four ways to provide reference images.\n- Expanded Files API documentation, including upload, listing, and delete tools, plus guidance for hosted connectors.\n- Updated tool parameter lists to reflect new image upload methods and clarify best practices (especially costs/limits of images_base64).\n- Removed the file skill-card.md.\n\nv1.0.4 | 2026-07-27T02:56:59.387Z | auto\n\n## gemini-mcp 1.0.4 Changelog\n\n- Removed the `skill-card.md` file from the project.\n- No user-facing functionality changes; documentation and usage remain unchanged.\n\nv1.0.3 | 2026-07-19T16:49:37.374Z | auto\n\ngemini-mcp 1.0.3\n\n- Removed the `skill-card.md` file from the project.\n- No functional or interface changes to core tools or documentation.\n\nv1.0.2 | 2026-07-19T12:46:33.591Z | auto\n\n- Removed the skill-card.md file.\n- No other changes to functionality or documentation.\n\nv1.0.1 | 2026-07-14T10:41:05.284Z | auto\n\n- Removed sample file: skill-card.md\n- No functional or user-facing changes; documentation and configuration remain unchanged\n\nv1.0.0 | 2026-07-08T15:14:31.167Z | auto\n\nMedia generation expanded: now supports images, video, and music via Google Gemini (image, omni, and Lyria models).\n\n- Updated triggers and description to include video (text→video, image→video) and music/audio (Lyria) alongside images.\n- Added details and workflow examples for video and music generation tools: gemini_video_generate and gemini_music_generate.\n- Detailed async usage, idempotency, and result-fetching for long-running generations.\n- All image tools now named consistently (e.g., gemini_image_generate, gemini_image_edit).\n- Expanded tool documentation to cover video aspect ratios, output formats, model options, and session/interaction handling.\n\nv0.9.0 | 2026-07-08T13:49:24.269Z | auto\n\nVersion 0.9.0 of gemini-mcp\n\n- No file changes detected in this release.\n- Documentation and usage instructions remain unchanged.\n- No new features, enhancements, or bug fixes introduced in this version.\n\nv0.8.0 | 2026-07-07T23:41:12.867Z | auto\n\nNo user-facing changes detected in this release.  \n- Version bump to 0.8.0 with no file changes.\n\nv0.7.2 | 2026-07-06T14:14:49.174Z | auto\n\nVersion 0.7.2 of gemini-mcp\n\n- No file changes detected in this release.\n- Functionality, documentation, and setup instructions remain unchanged from the previous version.\n\nv0.7.1 | 2026-07-06T12:33:17.974Z | auto\n\nNo changes detected in this version.\n\n- No file changes were found between this version and the previous one.\n- Behavior, documentation, and features remain the same as the prior release.\n\nv0.7.0 | 2026-07-06T07:08:47.943Z | auto\n\ngemini-mcp v0.7.0\n\n- Updated default image model to gemini-3.1-flash-image.\n- Added a detailed model selection guide, including new reference limits and capabilities for each Gemini model.\n- Documented new `continue_last` and `search_types` arguments for the Interactions API, enabling seamless iterative refinement workflows.\n- Improved tool descriptions for clarity and up-to-date best practices.\n- Updated environment variable defaults and usage.\n- Removed deprecated skill-card.md file.\n\nv0.6.1 | 2026-07-05T23:00:48.547Z | auto\n\n- Removed the file skill-card.md.\n- No user-facing features or interface changes.\n\nv0.6.0 | 2026-06-13T00:54:26.916Z | auto\n\ngemini-mcp v0.6.0 changelog\n\n- Added support for video input via new video_path parameter in gemini_generate_image and gemini_interact tools.\n- Updated documentation for gemini_generate_image and gemini_interact to reflect video_path capability.\n- Removed redundant skill-card.md file.\n\nv0.5.0 | 2026-06-10T02:09:57.824Z | auto\n\n- Removed the skill-card.md file.\n- No changes to core functionality or documentation.\n\nv0.4.0 | 2026-06-08T14:43:16.193Z | auto\n\n- Added support for the GEMINI_INPUT_DIR environment variable to resolve bare input-image filenames (e.g. for Cowork uploads).\n- Updated SKILL.md with workflow tips for handling chat-pasted or attached images, clarifying limitations and new built-in solutions.\n- Removed the obsolete skill-card.md file.\n- Expanded guidance on using from_clipboard: true and input directories for easier reference image handling, especially in Cowork and similar hosts.\n\nv0.3.0 | 2026-06-08T13:10:32.042Z | auto\n\n**Adds multi-turn conversational editing and expands input options.**\n\n- Introduced multi-turn image interaction via the new beta tool `gemini_interact`, enabling iterative refinement with `interaction_id` context.\n- Expanded input image support: now accepts both file paths (`images`) and direct base64/data URL values (`images_base64`) for both generation and editing.\n- Added options for video-based image generation (`video_url`) and live Google Search grounding (`google_search`).\n- Updated tool parameters, adding controls like `seed`, `filename`, `thinking_level`, and fine-tuned result metadata (e.g., reproducibility, caption text).\n- Updated documentation to reflect new APIs, workflows (including conversational iterations), and additional notes, such as edit limitations and determinism caveats.\n- Removed redundant file (`skill-card.md`).\n\nv0.2.0 | 2026-06-08T01:11:54.509Z | auto\n\ngemini-mcp 0.2.0\n\n- Added detailed usage instructions, setup steps, and API key guidance in SKILL.md.\n- Documented available models, tool commands, and parameters for image generation and editing.\n- Explained workflows for generating, editing, and composing single images, variations, or consistent sets.\n- Listed supported environment variables, options, and output behaviors.\n- Clarified new prompt triggers and scenarios for using the skill.\n\nArchive index:\n\nArchive v2.3.3: 3 files, 11183 bytes\n\nFiles: skill-card.md (1920b), SKILL.md (23734b), _meta.json (129b)\n\nFile v2.3.3:SKILL.md\n\n---\nname: gemini-mcp\ndescription: Generate and edit images, video, and music with Google Gemini models via MCP. Use when the user asks to generate, create, or edit images (Gemini / Nano Banana), produce a consistent set of images, compose/blend multiple images, generate a short video (text→video or image→video, via the omni model), or generate music/audio clips (via Lyria). Triggers on phrases like \"generate an image of\", \"edit this image with Gemini\", \"create a set of consistent images\", \"make a video of\", \"generate a video\", \"generate music\", \"make a song/audio clip\", \"use Nano Banana to make\", or any request to produce images, video, or music via the Gemini API. Requires the @chrischall/gemini-mcp package installed and the gemini server registered (see Setup below).\n---\n\n# gemini-mcp\n\nMCP server for Google Gemini media generation — natural-language **image**, **video**, and **music** creation via the Gemini API (Nano Banana / Nano Banana Pro images, omni video, Lyria music).\n\n- **npm:** [npmjs.com/package/@chrischall/gemini-mcp](https://www.npmjs.com/package/@chrischall/gemini-mcp)\n- **Source:** [github.com/chrischall/gemini-mcp](https://github.com/chrischall/gemini-mcp)\n\n## Setup\n\n### Option A — npx (recommended)\n\nAdd to `.mcp.json` in your project or `~/.claude/mcp.json`:\n\n```json\n{\n  \"mcpServers\": {\n    \"gemini\": {\n      \"command\": \"npx\",\n      \"args\": [\"-y\", \"@chrischall/gemini-mcp\"],\n      \"env\": {\n        \"GEMINI_API_KEY\": \"your-api-key-here\"\n      }\n    }\n  }\n}\n```\n\n### Option B — from source\n\n```bash\ngit clone https://github.com/chrischall/gemini-mcp\ncd gemini-mcp\nnpm install && npm run build\n```\n\nThen add to `.mcp.json`:\n\n```json\n{\n  \"mcpServers\": {\n    \"gemini\": {\n      \"command\": \"node\",\n      \"args\": [\"/path/to/gemini-mcp/dist/index.js\"],\n      \"env\": {\n        \"GEMINI_API_KEY\": \"your-api-key-here\"\n      }\n    }\n  }\n}\n```\n\nOr use a `.env` file in the project directory with `GEMINI_API_KEY=<value>`.\n\n### Getting your API key\n\n1. Go to [aistudio.google.com/apikey](https://aistudio.google.com/apikey)\n2. Create an API key (requires a Google account)\n3. Copy the key and set it as `GEMINI_API_KEY`\n\nNote: Image generation requires a billing-enabled Google Cloud project.\n\n## Environment Variables\n\n| Variable | Required | Description |\n|---|---|---|\n| `GEMINI_API_KEY` | Yes | Your Google Gemini API key |\n| `GEMINI_IMAGE_MODEL` | No | Override the default image model (default: `gemini-3.1-flash-image`) |\n| `GEMINI_OUTPUT_DIR` | No | Default directory for saved images (default: current working directory) |\n| `GEMINI_INPUT_DIR` | No | Directory to resolve bare input-image filenames against (e.g. point at Cowork's `uploads/` folder so `images: [\"house.jpg\"]` works) |\n\n## Tools\n\n### Models\n| Tool | Description |\n|------|-------------|\n| `gemini_list_models` | List available Gemini image models and the current default |\n\n**Which model to pick** (per-call `model`, or `GEMINI_IMAGE_MODEL`):\n\n| Model | When to use | Reference-image caps (of 14 max) |\n|-------|-------------|----------------------------------|\n| `gemini-3.1-flash-image` (Nano Banana 2) | The versatile generalist workhorse for all tasks — balances speed with state-of-the-art 4K generation, world knowledge, and reliable text rendering; excels at multi-reference-image processing and consistency. Only model with video input + image_search grounding | 10 objects + 4 characters + 3 style refs |\n| `gemini-3-pro-image` (Nano Banana Pro) | The premium choice for the most complex visual tasks — highest world knowledge, advanced localization, accurate brand consistency, precision creative control | 6 objects + 5 characters |\n| `gemini-3.1-flash-lite-image` (Nano Banana 2 Lite) | The fastest/cheapest for simple tasks — 1K output only, no Google Search grounding | 14 objects (no character consistency) |\n\n### Image Generation\n| Tool | Description |\n|------|-------------|\n| `gemini_image_generate(prompt, count?, images_url?, images_file_uris?, images?, images_base64?, video_url?, video_path?, google_search?, seed?, filename?, model?, aspect_ratio?, image_size?, thinking_level?, output_dir?, inline?)` | Generate image(s) from a text prompt (optionally image-conditioned — see **Reference images** below — or video-conditioned via `video_url`/`video_path`) |\n| `gemini_image_edit(prompt, images_url?, images_file_uris?, images?, images_base64?, google_search?, seed?, filename?, model?, aspect_ratio?, image_size?, thinking_level?, output_dir?, inline?)` | Edit or compose input image(s) with a text instruction. Requires ≥1 input from any of the four reference forms |\n| `gemini_image_set(master_prompt, scenes? \\| count?, reference_mode?, master_images_url?, master_images_file_uris?, master_images?, master_images_base64?, google_search?, seed?, basename?, model?, thinking_level?, ...)` | Master image (optionally seeded from a reference photo) plus N consistent images referencing it. `master_images_url` / `master_images_file_uris` are resolved **once** and passed to the master *and* every scene |\n\n### Reference images — four ways in, one that costs context\n\nThese apply to **every** tool that takes a reference image: the four image tools, plus\n`gemini_video_generate` (reference stills) and `gemini_music_generate`.\n\n| Parameter | Where the bytes travel | Context cost |\n|---|---|---|\n| `images_url` (`master_images_url`) | the **server** downloads the https URL | none |\n| `images_file_uris` (`master_images_file_uris`) | a `files/<id>` reference from `gemini_upload_file` | none |\n| `images` | read off local disk (stdio builds only) | none |\n| `images_base64` | **through the tool-call JSON** | **~14k tokens per JPEG** |\n\n**Reach for `images_base64` last.** It costs ~14k tokens per modest photo, and a truncated file\nread produces base64 that still *looks* valid — so the corruption surfaces as a bad generation,\nnot an error.\n\n- `images_url` accepts public `https://` URLs only (private/loopback/link-local hosts refused,\n  every redirect revalidated), must be `Content-Type: image/*`, capped at 15MB. Errors name the\n  failing URL. Over 6MB is auto-uploaded to the Files API instead of inlined.\n- `images_file_uris` accepts `files/<id>` or the full uri. Retained **~48h**; reusable across\n  any number of calls until then.\n- On stdio, an `images` path referenced **more than once in a session** is auto-uploaded to the\n  Files API (keyed on path + mtime + size) so the bytes stop being re-sent.\n- `images_base64` is uploaded on the **first** sighting, not the second — the tokens are already\n  spent by then. The result reports it under `image_inputs.base64_uploaded[].file_uri`: pass that\n  to `images_file_uris` on the next call instead of pasting the bytes again.\n\n### Files API\n| Tool | Description |\n|------|-------------|\n| `gemini_upload_file(url? \\| data_base64? \\| path?, mime_type?, display_name?, confirmToken?)` | Upload once, get a reusable `files/<id>`. Exactly one source. `url` is fetched by the server (image/video/audio, ≤100MB); `path` is stdio-only and confirmed first (see Notes); `data_base64` is the last resort |\n| `gemini_list_files(page_size?)` | List current uploads with MIME types and expiry |\n| `gemini_delete_file(file_uri, confirmToken?)` | Delete an upload before its ~48h expiry (confirmed first — see Notes) |\n\nOn the **hosted hosted deployment** there is also `POST /upload`, behind the same OAuth token as\n`/mcp` — the zero-base64 path for an agent with a shell:\n\n```bash\ncurl -X POST https://<hosted deployment>/upload \\\n  -H \"Authorization: Bearer $ACCESS_TOKEN\" \\\n  -H \"Content-Type: image/jpeg\" \\\n  --data-binary @photo.jpg\n# → {\"file_uri\":\"files/abc123\", ...}   then: images_file_uris: [\"files/abc123\"]\n```\n\n### Multi-turn (Interactions API)\n| Tool | Description |\n|------|-------------|\n| `gemini_interact(input, previous_interaction_id?, continue_last?, images_url?, images_file_uris?, images?, images_base64?, video_url?, video_path?, google_search?, search_types?, model?, aspect_ratio?, image_size?, thinking_level?, filename?, output_dir?, inline?)` | **Preferred tool for iterative refinement.** Generate/edit via Gemini's **Interactions API**. Returns an `interaction_id`; pass it back as `previous_interaction_id` (or set `continue_last: true` to reuse the session's most recent one) to **iteratively refine the same image** conversationally — do NOT start a new interaction or re-upload the image per tweak. Output is **JPEG**. |\n\n### Video & Music (preview — funded account)\n| Tool | Description |\n|------|-------------|\n| `gemini_video_generate(prompt, aspect_ratio?, resolution?, task?, images_url?, images_file_uris?, images?, images_base64?, from_clipboard?, previous_interaction_id?, continue_last?, model?, filename?, output_dir?, timeout_ms?, idempotency_key?, async?)` | Generate a short video (~10s) via the Gemini omni model: `text_to_video` (default), `image_to_video` / `reference_to_video` (supply reference image[s]), interpolation (pass first frame then last frame as `images`), or `edit` / `extend` (with `previous_interaction_id` / `continue_last`; extensions add ~3-10s each, to ~40s). `aspect_ratio` is `16:9` or `9:16`; `resolution` is `360p`/`720p` (default)/`1080p`/`4k` and is the cost lever — a 10s 360p clip runs about a third of 720p. Written to disk as **MP4**. Runs long — use `async: true` + `gemini_get_result`, or raise `timeout_ms`. |\n| `gemini_music_generate(prompt, model?, images_url?, images_file_uris?, images?, images_base64?, from_clipboard?, filename?, output_dir?, inline?, timeout_ms?, idempotency_key?, async?)` | Generate music from a prompt (mood/genre/instruments/structure/lyrics) via a Lyria model: `lyria-3-clip-preview` (30s instrumental, default, $0.04), `lyria-3.5` (full-length song with vocals, $0.08) or `lyria-3-pro-preview` ($0.08). Output is MP3. **Single-turn** — there is no follow-up refinement, so put the whole brief in the prompt. Written to disk, or returned inline. |\n\n### Async / idempotency (any generation tool)\n| Tool | Description |\n|------|-------------|\n| `gemini_get_result(job_id)` | Fetch a generation started with `async: true`. Returns `running` while in flight, then the normal result on completion. Lets a long video/music/image generation outlive a host's `tools/call` timeout. All generation tools also accept `idempotency_key` — a repeat call with the same key returns the recorded result (`reused: true`) instead of billing again. |\n\n## Workflows\n\n**Generate a single image:**\n```\ngemini_image_generate(prompt: \"a red maple leaf on white background, studio photo\")\n→ returns path to saved PNG\n```\n\n**Generate multiple variations:**\n```\ngemini_image_generate(prompt: \"a cartoon fox\", count: 4, output_dir: \"/tmp/foxes\")\n→ returns paths to 4 PNG files\n```\n\n**Edit an existing image:**\n```\ngemini_image_edit(prompt: \"make the background blue\", images: [\"/path/to/image.png\"])\n→ returns path to edited PNG\n```\n\n**Edit an image you only have a URL for (nothing downloads into the conversation):**\n```\ngemini_image_edit(prompt: \"make the background blue\", images_url: [\"https://example.com/photo.jpg\"])\n→ the server fetches the URL itself; returns path to edited PNG\n```\n\n**Reuse one photo across many generations (hosted hosted deployment, agent with a shell):**\n```\n$ curl -X POST https://<hosted deployment>/upload -H \"Authorization: Bearer $TOKEN\" \\\n    -H \"Content-Type: image/jpeg\" --data-binary @photo.jpg\n  → {\"file_uri\": \"files/abc123\"}\n\ngemini_image_edit(prompt: \"make it winter\",  images_file_uris: [\"files/abc123\"])\ngemini_image_edit(prompt: \"make it sunrise\", images_file_uris: [\"files/abc123\"])\n→ two edits, one upload, zero image bytes in context. Valid ~48h.\n```\n\n**Generate a consistent set (master + scenes):**\n```\ngemini_image_set(\n  master_prompt: \"a cartoon fox named Rusty, orange fur, blue scarf\",\n  scenes: [\"Rusty waving hello\", \"Rusty eating an apple\", \"Rusty sleeping\"]\n)\n→ returns paths to master + 3 scene images, all consistent\n```\n\n**Generate variations of a concept:**\n```\ngemini_image_set(\n  master_prompt: \"minimalist logo for a coffee shop\",\n  count: 5\n)\n→ returns master + 5 variations\n```\n\n**Use a reference photo by value (when you have the bytes):**\n```\ngemini_image_edit(\n  prompt: \"place this house on a vintage travel-poster background\",\n  images_base64: [\"data:image/jpeg;base64,/9j/4AAQ...\"]   // or raw base64\n)\n→ returns path to the edited image\n```\n`images_base64` is for bytes you actually have — a file you `Read`/encode, a URL\nyou fetch, or a `data:` URI the user pastes as **text**. Send them once: the result's\n`image_inputs.base64_uploaded[].file_uri` is a `files/<id>` to reuse via `images_file_uris`.\n\n**Iterate on ONE image conversationally (multi-turn):**\n```\nr1 = gemini_interact(input: \"a cozy reading nook, watercolor\")\n   → { images: [...], interaction_id: \"v1_abc…\" }\nr2 = gemini_interact(input: \"add a sleeping cat on the chair\",\n                     previous_interaction_id: r1.interaction_id)\n   → refined image that preserves r1; returns a NEW interaction_id\nr3 = gemini_interact(input: \"warmer lighting\", continue_last: true)\n   → same chain, without threading the id (uses the session's most recent interaction)\n```\nPrefer this over re-running `gemini_image_edit` when you're making a *series* of incremental edits — the model keeps the prior result in context. Every result echoes `interaction_id` (and `previous_interaction_id` when chaining) plus a `hint` with the exact follow-up call.\n\n**Generate a video (runs long, use async):**\n```\njob = gemini_video_generate(prompt: \"a paper boat sailing down a rain gutter, cinematic\",\n                            aspect_ratio: \"16:9\", resolution: \"360p\", async: true)\n   → { job_id, status: \"running\" }   (returns immediately — no host timeout)\ngemini_get_result(job_id: job.job_id)\n   → \"running\" until done, then the MP4 path on disk\n# Draft at 360p, re-run the keeper at 1080p — video bills per output token, so the\n# resolution is roughly the price.\n# Animate a still instead: gemini_video_generate(prompt: \"…\", task: \"image_to_video\", images: [\"/path/still.png\"])\n# Interpolate between two stills: images: [\"/path/first.png\", \"/path/last.png\"]\n# Extend a clip: gemini_video_generate(prompt: \"…\", task: \"extend\", continue_last: true)\n```\n\n**Generate music:**\n```\ngemini_music_generate(prompt: \"warm lo-fi hip hop, mellow Rhodes, vinyl crackle, 70bpm\")\n   → 30s MP3 on disk (lyria-3-clip-preview)\n# Full-length song with vocals: gemini_music_generate(prompt: \"…\", model: \"lyria-3.5\")\n```\n**⚠️ Chat-pasted/attached images can't be fed to these tools directly.** A pasted\nimage reaches the assistant as a *vision* block — the assistant can SEE it but\nnever receives the original bytes, and the host doesn't write it to disk. So\nneither `images` (no file exists) nor `images_base64` (the bytes can't be\nreconstructed from a downscaled vision rendering) is obtainable from a paste.\nTo use a real reference photo, the **user** must make the bytes available: save\nthe file and give its **path** (→ `images`), drop it into the project dir, paste\nit as a **`data:` URI in text**, or host it at a **URL** (fetch → base64 →\n`images_base64`). This is a host/Cowork limitation, not an MCP one.\n\nTwo built-in ways to get past the unreachable-paste problem without any manual\nextraction:\n- **`from_clipboard: true`** (macOS) — the tool reads the image off the system\n  clipboard itself (osascript), downscales it, and uses it. The user just needs\n  to **copy** the image (⌘C — distinct from pasting it inline into chat, which\n  doesn't keep it on the clipboard). Works on every image tool:\n  `gemini_image_edit(prompt: \"…\", from_clipboard: true)`.\n- **`GEMINI_INPUT_DIR`** — point it at a folder (e.g. Cowork's `uploads/`); then a\n  **bare filename** resolves against it: `gemini_image_edit(prompt: \"…\", images: [\"house.jpg\"])`.\n\n## Prompting playbook\n\nCondensed from Google's official Nano Banana prompting guide. Core rule:\n**describe the scene, don't list keywords** — narrative sentences beat tag soups.\n\n**Best practices:**\n- **Be hyper-specific.** \"Ornate elven plate armor, etched with silver leaf patterns\" beats \"fantasy armor\".\n- **Give context & intent.** \"Create a logo for a high-end minimalist skincare brand\" beats \"create a logo\".\n- **Iterate conversationally** (`gemini_interact`): \"warmer lighting\", \"same, but more serious expression\".\n- **Step-by-step for complex scenes.** \"First, a misty forest background. Then a stone altar in the foreground. Finally a glowing sword on the altar.\"\n- **Semantic negatives.** Describe what you want positively — \"an empty, deserted street\" — instead of \"no cars\".\n- **Camera language controls composition.** Wide-angle / macro / low-angle perspective / 85mm portrait lens / three-point softbox lighting.\n\n**Generation templates (abbreviated):**\n- *Photorealistic:* `photorealistic [shot type] of [subject] in [setting], [lighting], shot from [angle] with [lens]`\n- *Sticker/illustration:* `[style] sticker of [subject] doing [activity], bold outlines, cel-shading, [palette], white background`\n- *Text in image:* `create a [type] for [brand] with the text \"[exact text]\" in a [font style]` — Gemini renders text well; Pro is best for professional assets. Tip: generate the wording first, then ask for the image containing it.\n- *Product shot:* `high-resolution studio-lit photo of [product] on [surface], [lighting setup], [angle], sharp focus on [detail]`\n- *Minimalist/negative space:* `single [subject] in [frame position], vast empty [color] background` — for text-overlay backgrounds.\n- *Comic/storyboard:* `make a 3 panel comic in [style]; put the character in [scene]` (Pro or 3.1 Flash).\n\n**Editing templates (abbreviated):**\n- *Add/remove:* `using the provided image of [subject], [add/remove] [element]; match the original style/lighting/perspective`\n- *Inpaint (semantic mask):* `change only the [element] to [new element]; keep everything else exactly the same`\n- *Style transfer:* `transform the photo of [subject] into the style of [artist/style]; preserve composition`\n- *Compose:* `take the [element from image 1] and place it with [element from image 2]; adjust lighting/shadows to match`\n- *Detail preservation:* describe the critical element (face, logo) in detail and say it must \"remain completely unchanged\"\n- *Sketch → finished:* `turn this rough sketch of [subject] into a [style] photo; keep [features], add [details]`\n- *Character 360°:* iterate angles via `gemini_interact` (\"in profile looking right\"), feeding prior outputs back for consistency\n\n## Notes\n\n- **Confirmations.** Local file inputs (`images`, `master_images`, `video_path`, `gemini_upload_file`'s `path`) and every delete are confirmed before anything is sent: a confirmation prompt where the client supports one (unless the server sets `MCP_CONFIRM_ELICITATION=off`); otherwise the first call does nothing and returns `status: \"confirmation-required\"` with a `preview` (resolved paths, MIME types, sizes — or the method/path being deleted) and a `confirmToken`. Show the preview to the user, and only after they approve call again with the **same arguments** plus `confirmToken`. The token is single-use, expires (default 10 min), and is bound to those exact arguments — change anything and the call is refused (`DRAFT_CHANGED`, with a fresh preview and token). `MCP_CONFIRM_MODE` on the server picks `ask-user` (default), `auto` (the model may approve after reviewing the preview) or `refuse`. Text prompts, URLs, `files/` references and base64 inputs are not gated.\n- **Input images** accept either file **paths** (`images` / `master_images`) or **base64/data-URI values** (`images_base64` / `master_images_base64`).\n- **`seed`** makes a result reproducible; it's echoed in the result metadata (a random one is chosen + echoed when omitted). `count>1` uses `seed, seed+1, …` so the images differ. Determinism isn't fully guaranteed by the model.\n- **`filename`/`basename`** set the output name (extension stripped); names never overwrite (a `-2`, `-3` suffix is added). The result echoes the absolute path(s), `model`, `seed`, and aspect/size.\n- **No edit-strength control.** Gemini exposes no denoise/strength knob, and Nano Banana over-preserves the input — big structural edits (\"move/remove/shrink\", add a mat border) are often ignored. Workarounds: reroll with a different `seed`, raise `thinking_level` to `high`, use forceful wording, do layout changes (padding/borders) externally, or use `gemini_interact` multi-turn.\n- **`thinking_level`** (`minimal`/`high`, Gemini 3 models) controls reasoning depth — `high` can improve complex compositions/edits at higher latency/cost.\n- **Model text.** When the model returns a caption/explanation (mostly Gemini 3 **Pro**), it's surfaced as `text` in the result metadata.\n- **`google_search: true`** grounds the image in live Google Search (current events, weather, real data — great for infographics). The result metadata includes `grounding` with the `queries` run and the `sources` (`{uri, title}`) used. (`gemini_interact` surfaces `grounding.queries` — the Interactions API returns no clean source list.)\n- **`search_types`** (`gemini_interact` only): `[\"web_search\", \"image_search\"]` picks the grounding search types (setting it implies `google_search`). **`image_search`** (gemini-3.1-flash-image only) pulls web images via Google Image Search as *visual* references — useful for real-world subjects (a specific butterfly species, a landmark, a product). ⚠️ Two catches: Google ToS require **displaying the returned `grounding.search_suggestions` HTML chips** to the user, and image_search won't depict real people from web images.\n- **`video_url`** (a public YouTube URL, on `gemini_image_generate` / `gemini_interact`) generates an image from a video reference — **requires a Flash model** (e.g. `model: \"gemini-3.1-flash-image\"`). For a **local video file**, use **`video_path`** instead: the file is uploaded to the Gemini Files API (streamed from disk, 2 GB max), waited to `ACTIVE`, and referenced by its `files/…` uri. The result metadata echoes `video_file` (`{uri, name, expires}`, ~48h retention) — reuse that uri as `video_url` in later calls to skip re-uploading.\n- **`gemini_interact`** is the multi-turn path: it returns an `interaction_id`; thread it back via `previous_interaction_id` for conversational refinement. Output is **JPEG only**. (The Interactions API is GA as of 2026-07; it uses a different request shape than the `generate`/`edit`/`set` tools.)\n- `output_dir` per-call overrides `$GEMINI_OUTPUT_DIR` overrides cwd. `inline: true` returns bytes (with a metadata text block) instead of writing.\n- `count` and `scenes` are mutually exclusive in `gemini_image_set`; `reference_mode: \"chain\"` references the previous image instead of the master.\n- Aspect ratios: `1:1`, `16:9`, `9:16`, `4:3`, `3:4`, `2:3`, `3:2`, … · Image sizes: `512` (0.5K, Flash only), `1K`, `2K`, `4K`. `4K` is the max native output — true 18×24 in @ 300 DPI (5400×7200) needs an external upscale step.\n- All generated images carry a **SynthID** watermark (Google).\n- The model can mis-render text/Roman numerals (e.g. years) — verify any text in the output; it's a model limitation, not a tool setting.\n- **Best-performance languages:** EN, plus ar, de, es-MX, fr, hi, id, it, ja, ko, pt-BR, ru, ua, vi, zh-CN — prefer prompting in one of these.\n- Asking the *model* for \"N images\" in one prompt is unreliable (documented limitation) — use the `count` parameter instead; it makes N independent calls with distinct seeds.\n- Server logs to stderr only — stdout is reserved for JSON-RPC.\n\nFile v2.3.3:_meta.json\n\n{\n  \"ownerId\": \"kn700jq4sjtf2anb0rk3ft4p7n856872\",\n  \"slug\": \"gemini-mcp\",\n  \"version\": \"2.3.3\",\n  \"publishedAt\": 1791380232402\n}\n\nFile v2.3.3:skill-card.md\n\n## Description:\n\nHelps agents generate and edit images, create short videos, and produce music with Google Gemini models through an MCP server.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[chrischall](https://clawhub.ai/user/chrischall)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal users and creative teams use this skill to direct an agent to generate or refine images, videos, and music from prompts and supplied media.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Prompts, media, clipboard images, URLs, and base64 data may be sent to Gemini or a hosted service.\n\nMitigation: Use only content you are authorized to share; avoid sensitive or regulated data without consent and a review of service retention.\n\nRisk: Use requires a Gemini API key and media generation may incur charges.\n\nMitigation: Install only from a trusted package source, safeguard the API key, and review billing before generation.\n\n## Reference(s):\n\n- [gemini-mcp ClawHub release](https://clawhub.ai/chrischall/skills/gemini-mcp)\n- [gemini-mcp npm package](https://www.npmjs.com/package/@chrischall/gemini-mcp)\n- [Google AI Studio API key](https://aistudio.google.com/apikey)\n\n## Skill Output:\n\n**Output Type(s):** [Files, Text]\n\n**Output Format:** [PNG or JPEG images, MP4 video, MP3 audio, and text responses with file paths and metadata]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Generated files may be saved locally or returned inline; responses can include job IDs for asynchronous generation.]\n\n## Skill Version(s):\n\n2.3.3 (source: server-resolved ClawHub release)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v2.3.2: 3 files, 11097 bytes\n\nFiles: skill-card.md (1832b), SKILL.md (23679b), _meta.json (129b)\n\nFile v2.3.2:SKILL.md\n\n---\nname: gemini-mcp\ndescription: Generate and edit images, video, and music with Google Gemini models via MCP. Use when the user asks to generate, create, or edit images (Gemini / Nano Banana), produce a consistent set of images, compose/blend multiple images, generate a short video (text→video or image→video, via the omni model), or generate music/audio clips (via Lyria). Triggers on phrases like \"generate an image of\", \"edit this image with Gemini\", \"create a set of consistent images\", \"make a video of\", \"generate a video\", \"generate music\", \"make a song/audio clip\", \"use Nano Banana to make\", or any request to produce images, video, or music via the Gemini API. Requires the @chrischall/gemini-mcp package installed and the gemini server registered (see Setup below).\n---\n\n# gemini-mcp\n\nMCP server for Google Gemini media generation — natural-language **image**, **video**, and **music** creation via the Gemini API (Nano Banana / Nano Banana Pro images, omni video, Lyria music).\n\n- **npm:** [npmjs.com/package/@chrischall/gemini-mcp](https://www.npmjs.com/package/@chrischall/gemini-mcp)\n- **Source:** [github.com/chrischall/gemini-mcp](https://github.com/chrischall/gemini-mcp)\n\n## Setup\n\n### Option A — npx (recommended)\n\nAdd to `.mcp.json` in your project or `~/.claude/mcp.json`:\n\n```json\n{\n  \"mcpServers\": {\n    \"gemini\": {\n      \"command\": \"npx\",\n      \"args\": [\"-y\", \"@chrischall/gemini-mcp\"],\n      \"env\": {\n        \"GEMINI_API_KEY\": \"your-api-key-here\"\n      }\n    }\n  }\n}\n```\n\n### Option B — from source\n\n```bash\ngit clone https://github.com/chrischall/gemini-mcp\ncd gemini-mcp\nnpm install && npm run build\n```\n\nThen add to `.mcp.json`:\n\n```json\n{\n  \"mcpServers\": {\n    \"gemini\": {\n      \"command\": \"node\",\n      \"args\": [\"/path/to/gemini-mcp/dist/index.js\"],\n      \"env\": {\n        \"GEMINI_API_KEY\": \"your-api-key-here\"\n      }\n    }\n  }\n}\n```\n\nOr use a `.env` file in the project directory with `GEMINI_API_KEY=<value>`.\n\n### Getting your API key\n\n1. Go to [aistudio.google.com/apikey](https://aistudio.google.com/apikey)\n2. Create an API key (requires a Google account)\n3. Copy the key and set it as `GEMINI_API_KEY`\n\nNote: Image generation requires a billing-enabled Google Cloud project.\n\n## Environment Variables\n\n| Variable | Required | Description |\n|---|---|---|\n| `GEMINI_API_KEY` | Yes | Your Google Gemini API key |\n| `GEMINI_IMAGE_MODEL` | No | Override the default image model (default: `gemini-3.1-flash-image`) |\n| `GEMINI_OUTPUT_DIR` | No | Default directory for saved images (default: current working directory) |\n| `GEMINI_INPUT_DIR` | No | Directory to resolve bare input-image filenames against (e.g. point at Cowork's `uploads/` folder so `images: [\"house.jpg\"]` works) |\n\n## Tools\n\n### Models\n| Tool | Description |\n|------|-------------|\n| `gemini_list_models` | List available Gemini image models and the current default |\n\n**Which model to pick** (per-call `model`, or `GEMINI_IMAGE_MODEL`):\n\n| Model | When to use | Reference-image caps (of 14 max) |\n|-------|-------------|----------------------------------|\n| `gemini-3.1-flash-image` (Nano Banana 2) | The versatile generalist workhorse for all tasks — balances speed with state-of-the-art 4K generation, world knowledge, and reliable text rendering; excels at multi-reference-image processing and consistency. Only model with video input + image_search grounding | 10 objects + 4 characters + 3 style refs |\n| `gemini-3-pro-image` (Nano Banana Pro) | The premium choice for the most complex visual tasks — highest world knowledge, advanced localization, accurate brand consistency, precision creative control | 6 objects + 5 characters |\n| `gemini-3.1-flash-lite-image` (Nano Banana 2 Lite) | The fastest/cheapest for simple tasks — 1K output only, no Google Search grounding | 14 objects (no character consistency) |\n\n### Image Generation\n| Tool | Description |\n|------|-------------|\n| `gemini_image_generate(prompt, count?, images_url?, images_file_uris?, images?, images_base64?, video_url?, video_path?, google_search?, seed?, filename?, model?, aspect_ratio?, image_size?, thinking_level?, output_dir?, inline?)` | Generate image(s) from a text prompt (optionally image-conditioned — see **Reference images** below — or video-conditioned via `video_url`/`video_path`) |\n| `gemini_image_edit(prompt, images_url?, images_file_uris?, images?, images_base64?, google_search?, seed?, filename?, model?, aspect_ratio?, image_size?, thinking_level?, output_dir?, inline?)` | Edit or compose input image(s) with a text instruction. Requires ≥1 input from any of the four reference forms |\n| `gemini_image_set(master_prompt, scenes? \\| count?, reference_mode?, master_images_url?, master_images_file_uris?, master_images?, master_images_base64?, google_search?, seed?, basename?, model?, thinking_level?, ...)` | Master image (optionally seeded from a reference photo) plus N consistent images referencing it. `master_images_url` / `master_images_file_uris` are resolved **once** and passed to the master *and* every scene |\n\n### Reference images — four ways in, one that costs context\n\nThese apply to **every** tool that takes a reference image: the four image tools, plus\n`gemini_video_generate` (reference stills) and `gemini_music_generate`.\n\n| Parameter | Where the bytes travel | Context cost |\n|---|---|---|\n| `images_url` (`master_images_url`) | the **server** downloads the https URL | none |\n| `images_file_uris` (`master_images_file_uris`) | a `files/<id>` reference from `gemini_upload_file` | none |\n| `images` | read off local disk (stdio builds only) | none |\n| `images_base64` | **through the tool-call JSON** | **~14k tokens per JPEG** |\n\n**Reach for `images_base64` last.** It costs ~14k tokens per modest photo, and a truncated file\nread produces base64 that still *looks* valid — so the corruption surfaces as a bad generation,\nnot an error.\n\n- `images_url` accepts public `https://` URLs only (private/loopback/link-local hosts refused,\n  every redirect revalidated), must be `Content-Type: image/*`, capped at 15MB. Errors name the\n  failing URL. Over 6MB is auto-uploaded to the Files API instead of inlined.\n- `images_file_uris` accepts `files/<id>` or the full uri. Retained **~48h**; reusable across\n  any number of calls until then.\n- On stdio, an `images` path referenced **more than once in a session** is auto-uploaded to the\n  Files API (keyed on path + mtime + size) so the bytes stop being re-sent.\n- `images_base64` is uploaded on the **first** sighting, not the second — the tokens are already\n  spent by then. The result reports it under `image_inputs.base64_uploaded[].file_uri`: pass that\n  to `images_file_uris` on the next call instead of pasting the bytes again.\n\n### Files API\n| Tool | Description |\n|------|-------------|\n| `gemini_upload_file(url? \\| data_base64? \\| path?, mime_type?, display_name?, confirmToken?)` | Upload once, get a reusable `files/<id>`. Exactly one source. `url` is fetched by the server (image/video/audio, ≤100MB); `path` is stdio-only and confirmed first (see Notes); `data_base64` is the last resort |\n| `gemini_list_files(page_size?)` | List current uploads with MIME types and expiry |\n| `gemini_delete_file(file_uri, confirmToken?)` | Delete an upload before its ~48h expiry (confirmed first — see Notes) |\n\nOn the **hosted hosted deployment** there is also `POST /upload`, behind the same OAuth token as\n`/mcp` — the zero-base64 path for an agent with a shell:\n\n```bash\ncurl -X POST https://<hosted deployment>/upload \\\n  -H \"Authorization: Bearer $ACCESS_TOKEN\" \\\n  -H \"Content-Type: image/jpeg\" \\\n  --data-binary @photo.jpg\n# → {\"file_uri\":\"files/abc123\", ...}   then: images_file_uris: [\"files/abc123\"]\n```\n\n### Multi-turn (Interactions API)\n| Tool | Description |\n|------|-------------|\n| `gemini_interact(input, previous_interaction_id?, continue_last?, images_url?, images_file_uris?, images?, images_base64?, video_url?, video_path?, google_search?, search_types?, model?, aspect_ratio?, image_size?, thinking_level?, filename?, output_dir?, inline?)` | **Preferred tool for iterative refinement.** Generate/edit via Gemini's **Interactions API**. Returns an `interaction_id`; pass it back as `previous_interaction_id` (or set `continue_last: true` to reuse the session's most recent one) to **iteratively refine the same image** conversationally — do NOT start a new interaction or re-upload the image per tweak. Output is **JPEG**. |\n\n### Video & Music (preview — funded account)\n| Tool | Description |\n|------|-------------|\n| `gemini_video_generate(prompt, aspect_ratio?, resolution?, task?, images_url?, images_file_uris?, images?, images_base64?, from_clipboard?, previous_interaction_id?, continue_last?, model?, filename?, output_dir?, timeout_ms?, idempotency_key?, async?)` | Generate a short video (~10s) via the Gemini omni model: `text_to_video` (default), `image_to_video` / `reference_to_video` (supply reference image[s]), interpolation (pass first frame then last frame as `images`), or `edit` / `extend` (with `previous_interaction_id` / `continue_last`; extensions add ~3-10s each, to ~40s). `aspect_ratio` is `16:9` or `9:16`; `resolution` is `360p`/`720p` (default)/`1080p`/`4k` and is the cost lever — a 10s 360p clip runs about a third of 720p. Written to disk as **MP4**. Runs long — use `async: true` + `gemini_get_result`, or raise `timeout_ms`. |\n| `gemini_music_generate(prompt, model?, images_url?, images_file_uris?, images?, images_base64?, from_clipboard?, filename?, output_dir?, inline?, timeout_ms?, idempotency_key?, async?)` | Generate music from a prompt (mood/genre/instruments/structure/lyrics) via a Lyria model: `lyria-3-clip-preview` (30s instrumental, default, $0.04), `lyria-3.5` (full-length song with vocals, $0.08) or `lyria-3-pro-preview` ($0.08). Output is MP3. **Single-turn** — there is no follow-up refinement, so put the whole brief in the prompt. Written to disk, or returned inline. |\n\n### Async / idempotency (any generation tool)\n| Tool | Description |\n|------|-------------|\n| `gemini_get_result(job_id)` | Fetch a generation started with `async: true`. Returns `running` while in flight, then the normal result on completion. Lets a long video/music/image generation outlive a host's `tools/call` timeout. All generation tools also accept `idempotency_key` — a repeat call with the same key returns the recorded result (`reused: true`) instead of billing again. |\n\n## Workflows\n\n**Generate a single image:**\n```\ngemini_image_generate(prompt: \"a red maple leaf on white background, studio photo\")\n→ returns path to saved PNG\n```\n\n**Generate multiple variations:**\n```\ngemini_image_generate(prompt: \"a cartoon fox\", count: 4, output_dir: \"/tmp/foxes\")\n→ returns paths to 4 PNG files\n```\n\n**Edit an existing image:**\n```\ngemini_image_edit(prompt: \"make the background blue\", images: [\"/path/to/image.png\"])\n→ returns path to edited PNG\n```\n\n**Edit an image you only have a URL for (nothing downloads into the conversation):**\n```\ngemini_image_edit(prompt: \"make the background blue\", images_url: [\"https://example.com/photo.jpg\"])\n→ the server fetches the URL itself; returns path to edited PNG\n```\n\n**Reuse one photo across many generations (hosted hosted deployment, agent with a shell):**\n```\n$ curl -X POST https://<hosted deployment>/upload -H \"Authorization: Bearer $TOKEN\" \\\n    -H \"Content-Type: image/jpeg\" --data-binary @photo.jpg\n  → {\"file_uri\": \"files/abc123\"}\n\ngemini_image_edit(prompt: \"make it winter\",  images_file_uris: [\"files/abc123\"])\ngemini_image_edit(prompt: \"make it sunrise\", images_file_uris: [\"files/abc123\"])\n→ two edits, one upload, zero image bytes in context. Valid ~48h.\n```\n\n**Generate a consistent set (master + scenes):**\n```\ngemini_image_set(\n  master_prompt: \"a cartoon fox named Rusty, orange fur, blue scarf\",\n  scenes: [\"Rusty waving hello\", \"Rusty eating an apple\", \"Rusty sleeping\"]\n)\n→ returns paths to master + 3 scene images, all consistent\n```\n\n**Generate variations of a concept:**\n```\ngemini_image_set(\n  master_prompt: \"minimalist logo for a coffee shop\",\n  count: 5\n)\n→ returns master + 5 variations\n```\n\n**Use a reference photo by value (when you have the bytes):**\n```\ngemini_image_edit(\n  prompt: \"place this house on a vintage travel-poster background\",\n  images_base64: [\"data:image/jpeg;base64,/9j/4AAQ...\"]   // or raw base64\n)\n→ returns path to the edited image\n```\n`images_base64` is for bytes you actually have — a file you `Read`/encode, a URL\nyou fetch, or a `data:` URI the user pastes as **text**. Send them once: the result's\n`image_inputs.base64_uploaded[].file_uri` is a `files/<id>` to reuse via `images_file_uris`.\n\n**Iterate on ONE image conversationally (multi-turn):**\n```\nr1 = gemini_interact(input: \"a cozy reading nook, watercolor\")\n   → { images: [...], interaction_id: \"v1_abc…\" }\nr2 = gemini_interact(input: \"add a sleeping cat on the chair\",\n                     previous_interaction_id: r1.interaction_id)\n   → refined image that preserves r1; returns a NEW interaction_id\nr3 = gemini_interact(input: \"warmer lighting\", continue_last: true)\n   → same chain, without threading the id (uses the session's most recent interaction)\n```\nPrefer this over re-running `gemini_image_edit` when you're making a *series* of incremental edits — the model keeps the prior result in context. Every result echoes `interaction_id` (and `previous_interaction_id` when chaining) plus a `hint` with the exact follow-up call.\n\n**Generate a video (runs long, use async):**\n```\njob = gemini_video_generate(prompt: \"a paper boat sailing down a rain gutter, cinematic\",\n                            aspect_ratio: \"16:9\", resolution: \"360p\", async: true)\n   → { job_id, status: \"running\" }   (returns immediately — no host timeout)\ngemini_get_result(job_id: job.job_id)\n   → \"running\" until done, then the MP4 path on disk\n# Draft at 360p, re-run the keeper at 1080p — video bills per output token, so the\n# resolution is roughly the price.\n# Animate a still instead: gemini_video_generate(prompt: \"…\", task: \"image_to_video\", images: [\"/path/still.png\"])\n# Interpolate between two stills: images: [\"/path/first.png\", \"/path/last.png\"]\n# Extend a clip: gemini_video_generate(prompt: \"…\", task: \"extend\", continue_last: true)\n```\n\n**Generate music:**\n```\ngemini_music_generate(prompt: \"warm lo-fi hip hop, mellow Rhodes, vinyl crackle, 70bpm\")\n   → 30s MP3 on disk (lyria-3-clip-preview)\n# Full-length song with vocals: gemini_music_generate(prompt: \"…\", model: \"lyria-3.5\")\n```\n**⚠️ Chat-pasted/attached images can't be fed to these tools directly.** A pasted\nimage reaches the assistant as a *vision* block — the assistant can SEE it but\nnever receives the original bytes, and the host doesn't write it to disk. So\nneither `images` (no file exists) nor `images_base64` (the bytes can't be\nreconstructed from a downscaled vision rendering) is obtainable from a paste.\nTo use a real reference photo, the **user** must make the bytes available: save\nthe file and give its **path** (→ `images`), drop it into the project dir, paste\nit as a **`data:` URI in text**, or host it at a **URL** (fetch → base64 →\n`images_base64`). This is a host/Cowork limitation, not an MCP one.\n\nTwo built-in ways to get past the unreachable-paste problem without any manual\nextraction:\n- **`from_clipboard: true`** (macOS) — the tool reads the image off the system\n  clipboard itself (osascript), downscales it, and uses it. The user just needs\n  to **copy** the image (⌘C — distinct from pasting it inline into chat, which\n  doesn't keep it on the clipboard). Works on every image tool:\n  `gemini_image_edit(prompt: \"…\", from_clipboard: true)`.\n- **`GEMINI_INPUT_DIR`** — point it at a folder (e.g. Cowork's `uploads/`); then a\n  **bare filename** resolves against it: `gemini_image_edit(prompt: \"…\", images: [\"house.jpg\"])`.\n\n## Prompting playbook\n\nCondensed from Google's official Nano Banana prompting guide. Core rule:\n**describe the scene, don't list keywords** — narrative sentences beat tag soups.\n\n**Best practices:**\n- **Be hyper-specific.** \"Ornate elven plate armor, etched with silver leaf patterns\" beats \"fantasy armor\".\n- **Give context & intent.** \"Create a logo for a high-end minimalist skincare brand\" beats \"create a logo\".\n- **Iterate conversationally** (`gemini_interact`): \"warmer lighting\", \"same, but more serious expression\".\n- **Step-by-step for complex scenes.** \"First, a misty forest background. Then a stone altar in the foreground. Finally a glowing sword on the altar.\"\n- **Semantic negatives.** Describe what you want positively — \"an empty, deserted street\" — instead of \"no cars\".\n- **Camera language controls composition.** Wide-angle / macro / low-angle perspective / 85mm portrait lens / three-point softbox lighting.\n\n**Generation templates (abbreviated):**\n- *Photorealistic:* `photorealistic [shot type] of [subject] in [setting], [lighting], shot from [angle] with [lens]`\n- *Sticker/illustration:* `[style] sticker of [subject] doing [activity], bold outlines, cel-shading, [palette], white background`\n- *Text in image:* `create a [type] for [brand] with the text \"[exact text]\" in a [font style]` — Gemini renders text well; Pro is best for professional assets. Tip: generate the wording first, then ask for the image containing it.\n- *Product shot:* `high-resolution studio-lit photo of [product] on [surface], [lighting setup], [angle], sharp focus on [detail]`\n- *Minimalist/negative space:* `single [subject] in [frame position], vast empty [color] background` — for text-overlay backgrounds.\n- *Comic/storyboard:* `make a 3 panel comic in [style]; put the character in [scene]` (Pro or 3.1 Flash).\n\n**Editing templates (abbreviated):**\n- *Add/remove:* `using the provided image of [subject], [add/remove] [element]; match the original style/lighting/perspective`\n- *Inpaint (semantic mask):* `change only the [element] to [new element]; keep everything else exactly the same`\n- *Style transfer:* `transform the photo of [subject] into the style of [artist/style]; preserve composition`\n- *Compose:* `take the [element from image 1] and place it with [element from image 2]; adjust lighting/shadows to match`\n- *Detail preservation:* describe the critical element (face, logo) in detail and say it must \"remain completely unchanged\"\n- *Sketch → finished:* `turn this rough sketch of [subject] into a [style] photo; keep [features], add [details]`\n- *Character 360°:* iterate angles via `gemini_interact` (\"in profile looking right\"), feeding prior outputs back for consistency\n\n## Notes\n\n- **Confirmations.** Local file inputs (`images`, `master_images`, `video_path`, `gemini_upload_file`'s `path`) and every delete are confirmed before anything is sent: a confirmation prompt where the client supports one; otherwise the first call does nothing and returns `status: \"confirmation-required\"` with a `preview` (resolved paths, MIME types, sizes — or the method/path being deleted) and a `confirmToken`. Show the preview to the user, and only after they approve call again with the **same arguments** plus `confirmToken`. The token is single-use, expires (default 10 min), and is bound to those exact arguments — change anything and the call is refused (`DRAFT_CHANGED`, with a fresh preview and token). `MCP_CONFIRM_MODE` on the server picks `ask-user` (default), `auto` (the model may approve after reviewing the preview) or `refuse`. Text prompts, URLs, `files/` references and base64 inputs are not gated.\n- **Input images** accept either file **paths** (`images` / `master_images`) or **base64/data-URI values** (`images_base64` / `master_images_base64`).\n- **`seed`** makes a result reproducible; it's echoed in the result metadata (a random one is chosen + echoed when omitted). `count>1` uses `seed, seed+1, …` so the images differ. Determinism isn't fully guaranteed by the model.\n- **`filename`/`basename`** set the output name (extension stripped); names never overwrite (a `-2`, `-3` suffix is added). The result echoes the absolute path(s), `model`, `seed`, and aspect/size.\n- **No edit-strength control.** Gemini exposes no denoise/strength knob, and Nano Banana over-preserves the input — big structural edits (\"move/remove/shrink\", add a mat border) are often ignored. Workarounds: reroll with a different `seed`, raise `thinking_level` to `high`, use forceful wording, do layout changes (padding/borders) externally, or use `gemini_interact` multi-turn.\n- **`thinking_level`** (`minimal`/`high`, Gemini 3 models) controls reasoning depth — `high` can improve complex compositions/edits at higher latency/cost.\n- **Model text.** When the model returns a caption/explanation (mostly Gemini 3 **Pro**), it's surfaced as `text` in the result metadata.\n- **`google_search: true`** grounds the image in live Google Search (current events, weather, real data — great for infographics). The result metadata includes `grounding` with the `queries` run and the `sources` (`{uri, title}`) used. (`gemini_interact` surfaces `grounding.queries` — the Interactions API returns no clean source list.)\n- **`search_types`** (`gemini_interact` only): `[\"web_search\", \"image_search\"]` picks the grounding search types (setting it implies `google_search`). **`image_search`** (gemini-3.1-flash-image only) pulls web images via Google Image Search as *visual* references — useful for real-world subjects (a specific butterfly species, a landmark, a product). ⚠️ Two catches: Google ToS require **displaying the returned `grounding.search_suggestions` HTML chips** to the user, and image_search won't depict real people from web images.\n- **`video_url`** (a public YouTube URL, on `gemini_image_generate` / `gemini_interact`) generates an image from a video reference — **requires a Flash model** (e.g. `model: \"gemini-3.1-flash-image\"`). For a **local video file**, use **`video_path`** instead: the file is uploaded to the Gemini Files API (streamed from disk, 2 GB max), waited to `ACTIVE`, and referenced by its `files/…` uri. The result metadata echoes `video_file` (`{uri, name, expires}`, ~48h retention) — reuse that uri as `video_url` in later calls to skip re-uploading.\n- **`gemini_interact`** is the multi-turn path: it returns an `interaction_id`; thread it back via `previous_interaction_id` for conversational refinement. Output is **JPEG only**. (The Interactions API is GA as of 2026-07; it uses a different request shape than the `generate`/`edit`/`set` tools.)\n- `output_dir` per-call overrides `$GEMINI_OUTPUT_DIR` overrides cwd. `inline: true` returns bytes (with a metadata text block) instead of writing.\n- `count` and `scenes` are mutually exclusive in `gemini_image_set`; `reference_mode: \"chain\"` references the previous image instead of the master.\n- Aspect ratios: `1:1`, `16:9`, `9:16`, `4:3`, `3:4`, `2:3`, `3:2`, … · Image sizes: `512` (0.5K, Flash only), `1K`, `2K`, `4K`. `4K` is the max native output — true 18×24 in @ 300 DPI (5400×7200) needs an external upscale step.\n- All generated images carry a **SynthID** watermark (Google).\n- The model can mis-render text/Roman numerals (e.g. years) — verify any text in the output; it's a model limitation, not a tool setting.\n- **Best-performance languages:** EN, plus ar, de, es-MX, fr, hi, id, it, ja, ko, pt-BR, ru, ua, vi, zh-CN — prefer prompting in one of these.\n- Asking the *model* for \"N images\" in one prompt is unreliable (documented limitation) — use the `count` parameter instead; it makes N independent calls with distinct seeds.\n- Server logs to stderr only — stdout is reserved for JSON-RPC.\n\nFile v2.3.2:_meta.json\n\n{\n  \"ownerId\": \"kn700jq4sjtf2anb0rk3ft4p7n856872\",\n  \"slug\": \"gemini-mcp\",\n  \"version\": \"2.3.2\",\n  \"publishedAt\": 1791168601686\n}\n\nFile v2.3.2:skill-card.md\n\n## Description:\n\nGuides agents in generating and editing images, videos, and music through a Google Gemini MCP server.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[chrischall](https://clawhub.ai/user/chrischall)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nCreators and developers use this skill to direct an agent to generate or edit images, create short videos, and produce music with Gemini models.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Prompts and selected media are sent to external services, with potential retention and billing implications.\n\nMitigation: Use only media you are comfortable sharing under the provider's terms, and review expected costs before generation.\n\nRisk: Uploading local files or deleting uploaded media can affect user data.\n\nMitigation: Review the file or deletion preview and obtain user approval before confirming the action.\n\n## Reference(s):\n\n- [ClawHub gemini-mcp release](https://clawhub.ai/chrischall/skills/gemini-mcp)\n- [gemini-mcp npm package](https://www.npmjs.com/package/@chrischall/gemini-mcp)\n- [Google AI Studio API key setup](https://aistudio.google.com/apikey)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Configuration instructions, MCP tool calls]\n\n**Output Format:** [Markdown with configuration and tool-call examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [The connected service can save generated images, videos, and audio files.]\n\n## Skill Version(s):\n\n2.3.2 (source: ClawHub release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v2.3.1: 3 files, 11090 bytes\n\nFiles: skill-card.md (1808b), SKILL.md (23679b), _meta.json (129b)\n\nFile v2.3.1:SKILL.md\n\n---\nname: gemini-mcp\ndescription: Generate and edit images, video, and music with Google Gemini models via MCP. Use when the user asks to generate, create, or edit images (Gemini / Nano Banana), produce a consistent set of images, compose/blend multiple images, generate a short video (text→video or image→video, via the omni model), or generate music/audio clips (via Lyria). Triggers on phrases like \"generate an image of\", \"edit this image with Gemini\", \"create a set of consistent images\", \"make a video of\", \"generate a video\", \"generate music\", \"make a song/audio clip\", \"use Nano Banana to make\", or any request to produce images, video, or music via the Gemini API. Requires the @chrischall/gemini-mcp package installed and the gemini server registered (see Setup below).\n---\n\n# gemini-mcp\n\nMCP server for Google Gemini media generation — natural-language **image**, **video**, and **music** creation via the Gemini API (Nano Banana / Nano Banana Pro images, omni video, Lyria music).\n\n- **npm:** [npmjs.com/package/@chrischall/gemini-mcp](https://www.npmjs.com/package/@chrischall/gemini-mcp)\n- **Source:** [github.com/chrischall/gemini-mcp](https://github.com/chrischall/gemini-mcp)\n\n## Setup\n\n### Option A — npx (recommended)\n\nAdd to `.mcp.json` in your project or `~/.claude/mcp.json`:\n\n```json\n{\n  \"mcpServers\": {\n    \"gemini\": {\n      \"command\": \"npx\",\n      \"args\": [\"-y\", \"@chrischall/gemini-mcp\"],\n      \"env\": {\n        \"GEMINI_API_KEY\": \"your-api-key-here\"\n      }\n    }\n  }\n}\n```\n\n### Option B — from source\n\n```bash\ngit clone https://github.com/chrischall/gemini-mcp\ncd gemini-mcp\nnpm install && npm run build\n```\n\nThen add to `.mcp.json`:\n\n```json\n{\n  \"mcpServers\": {\n    \"gemini\": {\n      \"command\": \"node\",\n      \"args\": [\"/path/to/gemini-mcp/dist/index.js\"],\n      \"env\": {\n        \"GEMINI_API_KEY\": \"your-api-key-here\"\n      }\n    }\n  }\n}\n```\n\nOr use a `.env` file in the project directory with `GEMINI_API_KEY=<value>`.\n\n### Getting your API key\n\n1. Go to [aistudio.google.com/apikey](https://aistudio.google.com/apikey)\n2. Create an API key (requires a Google account)\n3. Copy the key and set it as `GEMINI_API_KEY`\n\nNote: Image generation requires a billing-enabled Google Cloud project.\n\n## Environment Variables\n\n| Variable | Required | Description |\n|---|---|---|\n| `GEMINI_API_KEY` | Yes | Your Google Gemini API key |\n| `GEMINI_IMAGE_MODEL` | No | Override the default image model (default: `gemini-3.1-flash-image`) |\n| `GEMINI_OUTPUT_DIR` | No | Default directory for saved images (default: current working directory) |\n| `GEMINI_INPUT_DIR` | No | Directory to resolve bare input-image filenames against (e.g. point at Cowork's `uploads/` folder so `images: [\"house.jpg\"]` works) |\n\n## Tools\n\n### Models\n| Tool | Description |\n|------|-------------|\n| `gemini_list_models` | List available Gemini image models and the current default |\n\n**Which model to pick** (per-call `model`, or `GEMINI_IMAGE_MODEL`):\n\n| Model | When to use | Reference-image caps (of 14 max) |\n|-------|-------------|----------------------------------|\n| `gemini-3.1-flash-image` (Nano Banana 2) | The versatile generalist workhorse for all tasks — balances speed with state-of-the-art 4K generation, world knowledge, and reliable text rendering; excels at multi-reference-image processing and consistency. Only model with video input + image_search grounding | 10 objects + 4 characters + 3 style refs |\n| `gemini-3-pro-image` (Nano Banana Pro) | The premium choice for the most complex visual tasks — highest world knowledge, advanced localization, accurate brand consistency, precision creative control | 6 objects + 5 characters |\n| `gemini-3.1-flash-lite-image` (Nano Banana 2 Lite) | The fastest/cheapest for simple tasks — 1K output only, no Google Search grounding | 14 objects (no character consistency) |\n\n### Image Generation\n| Tool | Description |\n|------|-------------|\n| `gemini_image_generate(prompt, count?, images_url?, images_file_uris?, images?, images_base64?, video_url?, video_path?, google_search?, seed?, filename?, model?, aspect_ratio?, image_size?, thinking_level?, output_dir?, inline?)` | Generate image(s) from a text prompt (optionally image-conditioned — see **Reference images** below — or video-conditioned via `video_url`/`video_path`) |\n| `gemini_image_edit(prompt, images_url?, images_file_uris?, images?, images_base64?, google_search?, seed?, filename?, model?, aspect_ratio?, image_size?, thinking_level?, output_dir?, inline?)` | Edit or compose input image(s) with a text instruction. Requires ≥1 input from any of the four reference forms |\n| `gemini_image_set(master_prompt, scenes? \\| count?, reference_mode?, master_images_url?, master_images_file_uris?, master_images?, master_images_base64?, google_search?, seed?, basename?, model?, thinking_level?, ...)` | Master image (optionally seeded from a reference photo) plus N consistent images referencing it. `master_images_url` / `master_images_file_uris` are resolved **once** and passed to the master *and* every scene |\n\n### Reference images — four ways in, one that costs context\n\nThese apply to **every** tool that takes a reference image: the four image tools, plus\n`gemini_video_generate` (reference stills) and `gemini_music_generate`.\n\n| Parameter | Where the bytes travel | Context cost |\n|---|---|---|\n| `images_url` (`master_images_url`) | the **server** downloads the https URL | none |\n| `images_file_uris` (`master_images_file_uris`) | a `files/<id>` reference from `gemini_upload_file` | none |\n| `images` | read off local disk (stdio builds only) | none |\n| `images_base64` | **through the tool-call JSON** | **~14k tokens per JPEG** |\n\n**Reach for `images_base64` last.** It costs ~14k tokens per modest photo, and a truncated file\nread produces base64 that still *looks* valid — so the corruption surfaces as a bad generation,\nnot an error.\n\n- `images_url` accepts public `https://` URLs only (private/loopback/link-local hosts refused,\n  every redirect revalidated), must be `Content-Type: image/*`, capped at 15MB. Errors name the\n  failing URL. Over 6MB is auto-uploaded to the Files API instead of inlined.\n- `images_file_uris` accepts `files/<id>` or the full uri. Retained **~48h**; reusable across\n  any number of calls until then.\n- On stdio, an `images` path referenced **more than once in a session** is auto-uploaded to the\n  Files API (keyed on path + mtime + size) so the bytes stop being re-sent.\n- `images_base64` is uploaded on the **first** sighting, not the second — the tokens are already\n  spent by then. The result reports it under `image_inputs.base64_uploaded[].file_uri`: pass that\n  to `images_file_uris` on the next call instead of pasting the bytes again.\n\n### Files API\n| Tool | Description |\n|------|-------------|\n| `gemini_upload_file(url? \\| data_base64? \\| path?, mime_type?, display_name?, confirmToken?)` | Upload once, get a reusable `files/<id>`. Exactly one source. `url` is fetched by the server (image/video/audio, ≤100MB); `path` is stdio-only and confirmed first (see Notes); `data_base64` is the last resort |\n| `gemini_list_files(page_size?)` | List current uploads with MIME types and expiry |\n| `gemini_delete_file(file_uri, confirmToken?)` | Delete an upload before its ~48h expiry (confirmed first — see Notes) |\n\nOn the **hosted hosted deployment** there is also `POST /upload`, behind the same OAuth token as\n`/mcp` — the zero-base64 path for an agent with a shell:\n\n```bash\ncurl -X POST https://<hosted deployment>/upload \\\n  -H \"Authorization: Bearer $ACCESS_TOKEN\" \\\n  -H \"Content-Type: image/jpeg\" \\\n  --data-binary @photo.jpg\n# → {\"file_uri\":\"files/abc123\", ...}   then: images_file_uris: [\"files/abc123\"]\n```\n\n### Multi-turn (Interactions API)\n| Tool | Description |\n|------|-------------|\n| `gemini_interact(input, previous_interaction_id?, continue_last?, images_url?, images_file_uris?, images?, images_base64?, video_url?, video_path?, google_search?, search_types?, model?, aspect_ratio?, image_size?, thinking_level?, filename?, output_dir?, inline?)` | **Preferred tool for iterative refinement.** Generate/edit via Gemini's **Interactions API**. Returns an `interaction_id`; pass it back as `previous_interaction_id` (or set `continue_last: true` to reuse the session's most recent one) to **iteratively refine the same image** conversationally — do NOT start a new interaction or re-upload the image per tweak. Output is **JPEG**. |\n\n### Video & Music (preview — funded account)\n| Tool | Description |\n|------|-------------|\n| `gemini_video_generate(prompt, aspect_ratio?, resolution?, task?, images_url?, images_file_uris?, images?, images_base64?, from_clipboard?, previous_interaction_id?, continue_last?, model?, filename?, output_dir?, timeout_ms?, idempotency_key?, async?)` | Generate a short video (~10s) via the Gemini omni model: `text_to_video` (default), `image_to_video` / `reference_to_video` (supply reference image[s]), interpolation (pass first frame then last frame as `images`), or `edit` / `extend` (with `previous_interaction_id` / `continue_last`; extensions add ~3-10s each, to ~40s). `aspect_ratio` is `16:9` or `9:16`; `resolution` is `360p`/`720p` (default)/`1080p`/`4k` and is the cost lever — a 10s 360p clip runs about a third of 720p. Written to disk as **MP4**. Runs long — use `async: true` + `gemini_get_result`, or raise `timeout_ms`. |\n| `gemini_music_generate(prompt, model?, images_url?, images_file_uris?, images?, images_base64?, from_clipboard?, filename?, output_dir?, inline?, timeout_ms?, idempotency_key?, async?)` | Generate music from a prompt (mood/genre/instruments/structure/lyrics) via a Lyria model: `lyria-3-clip-preview` (30s instrumental, default, $0.04), `lyria-3.5` (full-length song with vocals, $0.08) or `lyria-3-pro-preview` ($0.08). Output is MP3. **Single-turn** — there is no follow-up refinement, so put the whole brief in the prompt. Written to disk, or returned inline. |\n\n### Async / idempotency (any generation tool)\n| Tool | Description |\n|------|-------------|\n| `gemini_get_result(job_id)` | Fetch a generation started with `async: true`. Returns `running` while in flight, then the normal result on completion. Lets a long video/music/image generation outlive a host's `tools/call` timeout. All generation tools also accept `idempotency_key` — a repeat call with the same key returns the recorded result (`reused: true`) instead of billing again. |\n\n## Workflows\n\n**Generate a single image:**\n```\ngemini_image_generate(prompt: \"a red maple leaf on white background, studio photo\")\n→ returns path to saved PNG\n```\n\n**Generate multiple variations:**\n```\ngemini_image_generate(prompt: \"a cartoon fox\", count: 4, output_dir: \"/tmp/foxes\")\n→ returns paths to 4 PNG files\n```\n\n**Edit an existing image:**\n```\ngemini_image_edit(prompt: \"make the background blue\", images: [\"/path/to/image.png\"])\n→ returns path to edited PNG\n```\n\n**Edit an image you only have a URL for (nothing downloads into the conversation):**\n```\ngemini_image_edit(prompt: \"make the background blue\", images_url: [\"https://example.com/photo.jpg\"])\n→ the server fetches the URL itself; returns path to edited PNG\n```\n\n**Reuse one photo across many generations (hosted hosted deployment, agent with a shell):**\n```\n$ curl -X POST https://<hosted deployment>/upload -H \"Authorization: Bearer $TOKEN\" \\\n    -H \"Content-Type: image/jpeg\" --data-binary @photo.jpg\n  → {\"file_uri\": \"files/abc123\"}\n\ngemini_image_edit(prompt: \"make it winter\",  images_file_uris: [\"files/abc123\"])\ngemini_image_edit(prompt: \"make it sunrise\", images_file_uris: [\"files/abc123\"])\n→ two edits, one upload, zero image bytes in context. Valid ~48h.\n```\n\n**Generate a consistent set (master + scenes):**\n```\ngemini_image_set(\n  master_prompt: \"a cartoon fox named Rusty, orange fur, blue scarf\",\n  scenes: [\"Rusty waving hello\", \"Rusty eating an apple\", \"Rusty sleeping\"]\n)\n→ returns paths to master + 3 scene images, all consistent\n```\n\n**Generate variations of a concept:**\n```\ngemini_image_set(\n  master_prompt: \"minimalist logo for a coffee shop\",\n  count: 5\n)\n→ returns master + 5 variations\n```\n\n**Use a reference photo by value (when you have the bytes):**\n```\ngemini_image_edit(\n  prompt: \"place this house on a vintage travel-poster background\",\n  images_base64: [\"data:image/jpeg;base64,/9j/4AAQ...\"]   // or raw base64\n)\n→ returns path to the edited image\n```\n`images_base64` is for bytes you actually have — a file you `Read`/encode, a URL\nyou fetch, or a `data:` URI the user pastes as **text**. Send them once: the result's\n`image_inputs.base64_uploaded[].file_uri` is a `files/<id>` to reuse via `images_file_uris`.\n\n**Iterate on ONE image conversationally (multi-turn):**\n```\nr1 = gemini_interact(input: \"a cozy reading nook, watercolor\")\n   → { images: [...], interaction_id: \"v1_abc…\" }\nr2 = gemini_interact(input: \"add a sleeping cat on the chair\",\n                     previous_interaction_id: r1.interaction_id)\n   → refined image that preserves r1; returns a NEW interaction_id\nr3 = gemini_interact(input: \"warmer lighting\", continue_last: true)\n   → same chain, without threading the id (uses the session's most recent interaction)\n```\nPrefer this over re-running `gemini_image_edit` when you're making a *series* of incremental edits — the model keeps the prior result in context. Every result echoes `interaction_id` (and `previous_interaction_id` when chaining) plus a `hint` with the exact follow-up call.\n\n**Generate a video (runs long, use async):**\n```\njob = gemini_video_generate(prompt: \"a paper boat sailing down a rain gutter, cinematic\",\n                            aspect_ratio: \"16:9\", resolution: \"360p\", async: true)\n   → { job_id, status: \"running\" }   (returns immediately — no host timeout)\ngemini_get_result(job_id: job.job_id)\n   → \"running\" until done, then the MP4 path on disk\n# Draft at 360p, re-run the keeper at 1080p — video bills per output token, so the\n# resolution is roughly the price.\n# Animate a still instead: gemini_video_generate(prompt: \"…\", task: \"image_to_video\", images: [\"/path/still.png\"])\n# Interpolate between two stills: images: [\"/path/first.png\", \"/path/last.png\"]\n# Extend a clip: gemini_video_generate(prompt: \"…\", task: \"extend\", continue_last: true)\n```\n\n**Generate music:**\n```\ngemini_music_generate(prompt: \"warm lo-fi hip hop, mellow Rhodes, vinyl crackle, 70bpm\")\n   → 30s MP3 on disk (lyria-3-clip-preview)\n# Full-length song with vocals: gemini_music_generate(prompt: \"…\", model: \"lyria-3.5\")\n```\n**⚠️ Chat-pasted/attached images can't be fed to these tools directly.** A pasted\nimage reaches the assistant as a *vision* block — the assistant can SEE it but\nnever receives the original bytes, and the host doesn't write it to disk. So\nneither `images` (no file exists) nor `images_base64` (the bytes can't be\nreconstructed from a downscaled vision rendering) is obtainable from a paste.\nTo use a real reference photo, the **user** must make the bytes available: save\nthe file and give its **path** (→ `images`), drop it into the project dir, paste\nit as a **`data:` URI in text**, or host it at a **URL** (fetch → base64 →\n`images_base64`). This is a host/Cowork limitation, not an MCP one.\n\nTwo built-in ways to get past the unreachable-paste problem without any manual\nextraction:\n- **`from_clipboard: true`** (macOS) — the tool reads the image off the system\n  clipboard itself (osascript), downscales it, and uses it. The user just needs\n  to **copy** the image (⌘C — distinct from pasting it inline into chat, which\n  doesn't keep it on the clipboard). Works on every image tool:\n  `gemini_image_edit(prompt: \"…\", from_clipboard: true)`.\n- **`GEMINI_INPUT_DIR`** — point it at a folder (e.g. Cowork's `uploads/`); then a\n  **bare filename** resolves against it: `gemini_image_edit(prompt: \"…\", images: [\"house.jpg\"])`.\n\n## Prompting playbook\n\nCondensed from Google's official Nano Banana prompting guide. Core rule:\n**describe the scene, don't list keywords** — narrative sentences beat tag soups.\n\n**Best practices:**\n- **Be hyper-specific.** \"Ornate elven plate armor, etched with silver leaf patterns\" beats \"fantasy armor\".\n- **Give context & intent.** \"Create a logo for a high-end minimalist skincare brand\" beats \"create a logo\".\n- **Iterate conversationally** (`gemini_interact`): \"warmer lighting\", \"same, but more serious expression\".\n- **Step-by-step for complex scenes.** \"First, a misty forest background. Then a stone altar in the foreground. Finally a glowing sword on the altar.\"\n- **Semantic negatives.** Describe what you want positively — \"an empty, deserted street\" — instead of \"no cars\".\n- **Camera language controls composition.** Wide-angle / macro / low-angle perspective / 85mm portrait lens / three-point softbox lighting.\n\n**Generation templates (abbreviated):**\n- *Photorealistic:* `photorealistic [shot type] of [subject] in [setting], [lighting], shot from [angle] with [lens]`\n- *Sticker/illustration:* `[style] sticker of [subject] doing [activity], bold outlines, cel-shading, [palette], white background`\n- *Text in image:* `create a [type] for [brand] with the text \"[exact text]\" in a [font style]` — Gemini renders text well; Pro is best for professional assets. Tip: generate the wording first, then ask for the image containing it.\n- *Product shot:* `high-resolution studio-lit photo of [product] on [surface], [lighting setup], [angle], sharp focus on [detail]`\n- *Minimalist/negative space:* `single [subject] in [frame position], vast empty [color] background` — for text-overlay backgrounds.\n- *Comic/storyboard:* `make a 3 panel comic in [style]; put the character in [scene]` (Pro or 3.1 Flash).\n\n**Editing templates (abbreviated):**\n- *Add/remove:* `using the provided image of [subject], [add/remove] [element]; match the original style/lighting/perspective`\n- *Inpaint (semantic mask):* `change only the [element] to [new element]; keep everything else exactly the same`\n- *Style transfer:* `transform the photo of [subject] into the style of [artist/style]; preserve composition`\n- *Compose:* `take the [element from image 1] and place it with [element from image 2]; adjust lighting/shadows to match`\n- *Detail preservation:* describe the critical element (face, logo) in detail and say it must \"remain completely unchanged\"\n- *Sketch → finished:* `turn this rough sketch of [subject] into a [style] photo; keep [features], add [details]`\n- *Character 360°:* iterate angles via `gemini_interact` (\"in profile looking right\"), feeding prior outputs back for consistency\n\n## Notes\n\n- **Confirmations.** Local file inputs (`images`, `master_images`, `video_path`, `gemini_upload_file`'s `path`) and every delete are confirmed before anything is sent: a confirmation prompt where the client supports one; otherwise the first call does nothing and returns `status: \"confirmation-required\"` with a `preview` (resolved paths, MIME types, sizes — or the method/path being deleted) and a `confirmToken`. Show the preview to the user, and only after they approve call again with the **same arguments** plus `confirmToken`. The token is single-use, expires (default 10 min), and is bound to those exact arguments — change anything and the call is refused (`DRAFT_CHANGED`, with a fresh preview and token). `MCP_CONFIRM_MODE` on the server picks `ask-user` (default), `auto` (the model may approve after reviewing the preview) or `refuse`. Text prompts, URLs, `files/` references and base64 inputs are not gated.\n- **Input images** accept either file **paths** (`images` / `master_images`) or **base64/data-URI values** (`images_base64` / `master_images_base64`).\n- **`seed`** makes a result reproducible; it's echoed in the result metadata (a random one is chosen + echoed when omitted). `count>1` uses `seed, seed+1, …` so the images differ. Determinism isn't fully guaranteed by the model.\n- **`filename`/`basename`** set the output name (extension stripped); names never overwrite (a `-2`, `-3` suffix is added). The result echoes the absolute path(s), `model`, `seed`, and aspect/size.\n- **No edit-strength control.** Gemini exposes no denoise/strength knob, and Nano Banana over-preserves the input — big structural edits (\"move/remove/shrink\", add a mat border) are often ignored. Workarounds: reroll with a different `seed`, raise `thinking_level` to `high`, use forceful wording, do layout changes (padding/borders) externally, or use `gemini_interact` multi-turn.\n- **`thinking_level`** (`minimal`/`high`, Gemini 3 models) controls reasoning depth — `high` can improve complex compositions/edits at higher latency/cost.\n- **Model text.** When the model returns a caption/explanation (mostly Gemini 3 **Pro**), it's surfaced as `text` in the result metadata.\n- **`google_search: true`** grounds the image in live Google Search (current events, weather, real data — great for infographics). The result metadata includes `grounding` with the `queries` run and the `sources` (`{uri, title}`) used. (`gemini_interact` surfaces `grounding.queries` — the Interactions API returns no clean source list.)\n- **`search_types`** (`gemini_interact` only): `[\"web_search\", \"image_search\"]` picks the grounding search types (setting it implies `google_search`). **`image_search`** (gemini-3.1-flash-image only) pulls web images via Google Image Search as *visual* references — useful for real-world subjects (a specific butterfly species, a landmark, a product). ⚠️ Two catches: Google ToS require **displaying the returned `grounding.search_suggestions` HTML chips** to the user, and image_search won't depict real people from web images.\n- **`video_url`** (a public YouTube URL, on `gemini_image_generate` / `gemini_interact`) generates an image from a video reference — **requires a Flash model** (e.g. `model: \"gemini-3.1-flash-image\"`). For a **local video file**, use **`video_path`** instead: the file is uploaded to the Gemini Files API (streamed from disk, 2 GB max), waited to `ACTIVE`, and referenced by its `files/…` uri. The result metadata echoes `video_file` (`{uri, name, expires}`, ~48h retention) — reuse that uri as `video_url` in later calls to skip re-uploading.\n- **`gemini_interact`** is the multi-turn path: it returns an `interaction_id`; thread it back via `previous_interaction_id` for conversational refinement. Output is **JPEG only**. (The Interactions API is GA as of 2026-07; it uses a different request shape than the `generate`/`edit`/`set` tools.)\n- `output_dir` per-call overrides `$GEMINI_OUTPUT_DIR` overrides cwd. `inline: true` returns bytes (with a metadata text block) instead of writing.\n- `count` and `scenes` are mutually exclusive in `gemini_image_set`; `reference_mode: \"chain\"` references the previous image instead of the master.\n- Aspect ratios: `1:1`, `16:9`, `9:16`, `4:3`, `3:4`, `2:3`, `3:2`, … · Image sizes: `512` (0.5K, Flash only), `1K`, `2K`, `4K`. `4K` is the max native output — true 18×24 in @ 300 DPI (5400×7200) needs an external upscale step.\n- All generated images carry a **SynthID** watermark (Google).\n- The model can mis-render text/Roman numerals (e.g. years) — verify any text in the output; it's a model limitation, not a tool setting.\n- **Best-performance languages:** EN, plus ar, de, es-MX, fr, hi, id, it, ja, ko, pt-BR, ru, ua, vi, zh-CN — prefer prompting in one of these.\n- Asking the *model* for \"N images\" in one prompt is unreliable (documented limitation) — use the `count` parameter instead; it makes N independent calls with distinct seeds.\n- Server logs to stderr only — stdout is reserved for JSON-RPC.\n\nFile v2.3.1:_meta.json\n\n{\n  \"ownerId\": \"kn700jq4sjtf2anb0rk3ft4p7n856872\",\n  \"slug\": \"gemini-mcp\",\n  \"version\": \"2.3.1\",\n  \"publishedAt\": 1790991619628\n}\n\nFile v2.3.1:skill-card.md\n\n## Description:\n\nHelps agents generate and edit images, create short videos, and compose music with Google Gemini media tools.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[chrischall](https://clawhub.ai/user/chrischall)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nCreators and developers use this skill to direct an agent to generate or refine images, video clips, and music from prompts or supplied media.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Supplied prompts and media may be sent to Gemini or a hosted deployment.\n\nMitigation: Use non-sensitive test media first and approve local-file or clipboard uploads explicitly.\n\nRisk: The required third-party package and Gemini API key introduce software and credential exposure.\n\nMitigation: Review the package before running it and keep the API key scoped and protected.\n\n## Reference(s):\n\n- [ClawHub skill release](https://clawhub.ai/chrischall/skills/gemini-mcp)\n- [gemini-mcp package](https://www.npmjs.com/package/@chrischall/gemini-mcp)\n- [Google AI Studio API key setup](https://aistudio.google.com/apikey)\n\n## Skill Output:\n\n**Output Type(s):** [Guidance, Configuration instructions, API calls]\n\n**Output Format:** [Text guidance and MCP tool calls; generated image, video, or audio files]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Media may be saved to disk or returned inline, depending on the tool and settings.]\n\n## Skill Version(s):\n\n2.3.1 (source: ClawHub release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v2.3.0: 3 files, 11133 bytes\n\nFiles: skill-card.md (1867b), SKILL.md (23679b), _meta.json (129b)\n\nFile v2.3.0:SKILL.md\n\n---\nname: gemini-mcp\ndescription: Generate and edit images, video, and music with Google Gemini models via MCP. Use when the user asks to generate, create, or edit images (Gemini / Nano Banana), produce a consistent set of images, compose/blend multiple images, generate a short video (text→video or image→video, via the omni model), or generate music/audio clips (via Lyria). Triggers on phrases like \"generate an image of\", \"edit this image with Gemini\", \"create a set of consistent images\", \"make a video of\", \"generate a video\", \"generate music\", \"make a song/audio clip\", \"use Nano Banana to make\", or any request to produce images, video, or music via the Gemini API. Requires the @chrischall/gemini-mcp package installed and the gemini server registered (see Setup below).\n---\n\n# gemini-mcp\n\nMCP server for Google Gemini media generation — natural-language **image**, **video**, and **music** creation via the Gemini API (Nano Banana / Nano Banana Pro images, omni video, Lyria music).\n\n- **npm:** [npmjs.com/package/@chrischall/gemini-mcp](https://www.npmjs.com/package/@chrischall/gemini-mcp)\n- **Source:** [github.com/chrischall/gemini-mcp](https://github.com/chrischall/gemini-mcp)\n\n## Setup\n\n### Option A — npx (recommended)\n\nAdd to `.mcp.json` in your project or `~/.claude/mcp.json`:\n\n```json\n{\n  \"mcpServers\": {\n    \"gemini\": {\n      \"command\": \"npx\",\n      \"args\": [\"-y\", \"@chrischall/gemini-mcp\"],\n      \"env\": {\n        \"GEMINI_API_KEY\": \"your-api-key-here\"\n      }\n    }\n  }\n}\n```\n\n### Option B — from source\n\n```bash\ngit clone https://github.com/chrischall/gemini-mcp\ncd gemini-mcp\nnpm install && npm run build\n```\n\nThen add to `.mcp.json`:\n\n```json\n{\n  \"mcpServers\": {\n    \"gemini\": {\n      \"command\": \"node\",\n      \"args\": [\"/path/to/gemini-mcp/dist/index.js\"],\n      \"env\": {\n        \"GEMINI_API_KEY\": \"your-api-key-here\"\n      }\n    }\n  }\n}\n```\n\nOr use a `.env` file in the project directory with `GEMINI_API_KEY=<value>`.\n\n### Getting your API key\n\n1. Go to [aistudio.google.com/apikey](https://aistudio.google.com/apikey)\n2. Create an API key (requires a Google account)\n3. Copy the key and set it as `GEMINI_API_KEY`\n\nNote: Image generation requires a billing-enabled Google Cloud project.\n\n## Environment Variables\n\n| Variable | Required | Description |\n|---|---|---|\n| `GEMINI_API_KEY` | Yes | Your Google Gemini API key |\n| `GEMINI_IMAGE_MODEL` | No | Override the default image model (default: `gemini-3.1-flash-image`) |\n| `GEMINI_OUTPUT_DIR` | No | Default directory for saved images (default: current working directory) |\n| `GEMINI_INPUT_DIR` | No | Directory to resolve bare input-image filenames against (e.g. point at Cowork's `uploads/` folder so `images: [\"house.jpg\"]` works) |\n\n## Tools\n\n### Models\n| Tool | Description |\n|------|-------------|\n| `gemini_list_models` | List available Gemini image models and the current default |\n\n**Which model to pick** (per-call `model`, or `GEMINI_IMAGE_MODEL`):\n\n| Model | When to use | Reference-image caps (of 14 max) |\n|-------|-------------|----------------------------------|\n| `gemini-3.1-flash-image` (Nano Banana 2) | The versatile generalist workhorse for all tasks — balances speed with state-of-the-art 4K generation, world knowledge, and reliable text rendering; excels at multi-reference-image processing and consistency. Only model with video input + image_search grounding | 10 objects + 4 characters + 3 style refs |\n| `gemini-3-pro-image` (Nano Banana Pro) | The premium choice for the most complex visual tasks — highest world knowledge, advanced localization, accurate brand consistency, precision creative control | 6 objects + 5 characters |\n| `gemini-3.1-flash-lite-image` (Nano Banana 2 Lite) | The fastest/cheapest for simple tasks — 1K output only, no Google Search grounding | 14 objects (no character consistency) |\n\n### Image Generation\n| Tool | Description |\n|------|-------------|\n| `gemini_image_generate(prompt, count?, images_url?, images_file_uris?, images?, images_base64?, video_url?, video_path?, google_search?, seed?, filename?, model?, aspect_ratio?, image_size?, thinking_level?, output_dir?, inline?)` | Generate image(s) from a text prompt (optionally image-conditioned — see **Reference images** below — or video-conditioned via `video_url`/`video_path`) |\n| `gemini_image_edit(prompt, images_url?, images_file_uris?, images?, images_base64?, google_search?, seed?, filename?, model?, aspect_ratio?, image_size?, thinking_level?, output_dir?, inline?)` | Edit or compose input image(s) with a text instruction. Requires ≥1 input from any of the four reference forms |\n| `gemini_image_set(master_prompt, scenes? \\| count?, reference_mode?, master_images_url?, master_images_file_uris?, master_images?, master_images_base64?, google_search?, seed?, basename?, model?, thinking_level?, ...)` | Master image (optionally seeded from a reference photo) plus N consistent images referencing it. `master_images_url` / `master_images_file_uris` are resolved **once** and passed to the master *and* every scene |\n\n### Reference images — four ways in, one that costs context\n\nThese apply to **every** tool that takes a reference image: the four image tools, plus\n`gemini_video_generate` (reference stills) and `gemini_music_generate`.\n\n| Parameter | Where the bytes travel | Context cost |\n|---|---|---|\n| `images_url` (`master_images_url`) | the **server** downloads the https URL | none |\n| `images_file_uris` (`master_images_file_uris`) | a `files/<id>` reference from `gemini_upload_file` | none |\n| `images` | read off local disk (stdio builds only) | none |\n| `images_base64` | **through the tool-call JSON** | **~14k tokens per JPEG** |\n\n**Reach for `images_base64` last.** It costs ~14k tokens per modest photo, and a truncated file\nread produces base64 that still *looks* valid — so the corruption surfaces as a bad generation,\nnot an error.\n\n- `images_url` accepts public `https://` URLs only (private/loopback/link-local hosts refused,\n  every redirect revalidated), must be `Content-Type: image/*`, capped at 15MB. Errors name the\n  failing URL. Over 6MB is auto-uploaded to the Files API instead of inlined.\n- `images_file_uris` accepts `files/<id>` or the full uri. Retained **~48h**; reusable across\n  any number of calls until then.\n- On stdio, an `images` path referenced **more than once in a session** is auto-uploaded to the\n  Files API (keyed on path + mtime + size) so the bytes stop being re-sent.\n- `images_base64` is uploaded on the **first** sighting, not the second — the tokens are already\n  spent by then. The result reports it under `image_inputs.base64_uploaded[].file_uri`: pass that\n  to `images_file_uris` on the next call instead of pasting the bytes again.\n\n### Files API\n| Tool | Description |\n|------|-------------|\n| `gemini_upload_file(url? \\| data_base64? \\| path?, mime_type?, display_name?, confirmToken?)` | Upload once, get a reusable `files/<id>`. Exactly one source. `url` is fetched by the server (image/video/audio, ≤100MB); `path` is stdio-only and confirmed first (see Notes); `data_base64` is the last resort |\n| `gemini_list_files(page_size?)` | List current uploads with MIME types and expiry |\n| `gemini_delete_file(file_uri, confirmToken?)` | Delete an upload before its ~48h expiry (confirmed first — see Notes) |\n\nOn the **hosted hosted deployment** there is also `POST /upload`, behind the same OAuth token as\n`/mcp` — the zero-base64 path for an agent with a shell:\n\n```bash\ncurl -X POST https://<hosted deployment>/upload \\\n  -H \"Authorization: Bearer $ACCESS_TOKEN\" \\\n  -H \"Content-Type: image/jpeg\" \\\n  --data-binary @photo.jpg\n# → {\"file_uri\":\"files/abc123\", ...}   then: images_file_uris: [\"files/abc123\"]\n```\n\n### Multi-turn (Interactions API)\n| Tool | Description |\n|------|-------------|\n| `gemini_interact(input, previous_interaction_id?, continue_last?, images_url?, images_file_uris?, images?, images_base64?, video_url?, video_path?, google_search?, search_types?, model?, aspect_ratio?, image_size?, thinking_level?, filename?, output_dir?, inline?)` | **Preferred tool for iterative refinement.** Generate/edit via Gemini's **Interactions API**. Returns an `interaction_id`; pass it back as `previous_interaction_id` (or set `continue_last: true` to reuse the session's most recent one) to **iteratively refine the same image** conversationally — do NOT start a new interaction or re-upload the image per tweak. Output is **JPEG**. |\n\n### Video & Music (preview — funded account)\n| Tool | Description |\n|------|-------------|\n| `gemini_video_generate(prompt, aspect_ratio?, resolution?, task?, images_url?, images_file_uris?, images?, images_base64?, from_clipboard?, previous_interaction_id?, continue_last?, model?, filename?, output_dir?, timeout_ms?, idempotency_key?, async?)` | Generate a short video (~10s) via the Gemini omni model: `text_to_video` (default), `image_to_video` / `reference_to_video` (supply reference image[s]), interpolation (pass first frame then last frame as `images`), or `edit` / `extend` (with `previous_interaction_id` / `continue_last`; extensions add ~3-10s each, to ~40s). `aspect_ratio` is `16:9` or `9:16`; `resolution` is `360p`/`720p` (default)/`1080p`/`4k` and is the cost lever — a 10s 360p clip runs about a third of 720p. Written to disk as **MP4**. Runs long — use `async: true` + `gemini_get_result`, or raise `timeout_ms`. |\n| `gemini_music_generate(prompt, model?, images_url?, images_file_uris?, images?, images_base64?, from_clipboard?, filename?, output_dir?, inline?, timeout_ms?, idempotency_key?, async?)` | Generate music from a prompt (mood/genre/instruments/structure/lyrics) via a Lyria model: `lyria-3-clip-preview` (30s instrumental, default, $0.04), `lyria-3.5` (full-length song with vocals, $0.08) or `lyria-3-pro-preview` ($0.08). Output is MP3. **Single-turn** — there is no follow-up refinement, so put the whole brief in the prompt. Written to disk, or returned inline. |\n\n### Async / idempotency (any generation tool)\n| Tool | Description |\n|------|-------------|\n| `gemini_get_result(job_id)` | Fetch a generation started with `async: true`. Returns `running` while in flight, then the normal result on completion. Lets a long video/music/image generation outlive a host's `tools/call` timeout. All generation tools also accept `idempotency_key` — a repeat call with the same key returns the recorded result (`reused: true`) instead of billing again. |\n\n## Workflows\n\n**Generate a single image:**\n```\ngemini_image_generate(prompt: \"a red maple leaf on white background, studio photo\")\n→ returns path to saved PNG\n```\n\n**Generate multiple variations:**\n```\ngemini_image_generate(prompt: \"a cartoon fox\", count: 4, output_dir: \"/tmp/foxes\")\n→ returns paths to 4 PNG files\n```\n\n**Edit an existing image:**\n```\ngemini_image_edit(prompt: \"make the background blue\", images: [\"/path/to/image.png\"])\n→ returns path to edited PNG\n```\n\n**Edit an image you only have a URL for (nothing downloads into the conversation):**\n```\ngemini_image_edit(prompt: \"make the background blue\", images_url: [\"https://example.com/photo.jpg\"])\n→ the server fetches the URL itself; returns path to edited PNG\n```\n\n**Reuse one photo across many generations (hosted hosted deployment, agent with a shell):**\n```\n$ curl -X POST https://<hosted deployment>/upload -H \"Authorization: Bearer $TOKEN\" \\\n    -H \"Content-Type: image/jpeg\" --data-binary @photo.jpg\n  → {\"file_uri\": \"files/abc123\"}\n\ngemini_image_edit(prompt: \"make it winter\",  images_file_uris: [\"files/abc123\"])\ngemini_image_edit(prompt: \"make it sunrise\", images_file_uris: [\"files/abc123\"])\n→ two edits, one upload, zero image bytes in context. Valid ~48h.\n```\n\n**Generate a consistent set (master + scenes):**\n```\ngemini_image_set(\n  master_prompt: \"a cartoon fox named Rusty, orange fur, blue scarf\",\n  scenes: [\"Rusty waving hello\", \"Rusty eating an apple\", \"Rusty sleeping\"]\n)\n→ returns paths to master + 3 scene images, all consistent\n```\n\n**Generate variations of a concept:**\n```\ngemini_image_set(\n  master_prompt: \"minimalist logo for a coffee shop\",\n  count: 5\n)\n→ returns master + 5 variations\n```\n\n**Use a reference photo by value (when you have the bytes):**\n```\ngemini_image_edit(\n  prompt: \"place this house on a vintage travel-poster background\",\n  images_base64: [\"data:image/jpeg;base64,/9j/4AAQ...\"]   // or raw base64\n)\n→ returns path to the edited image\n```\n`images_base64` is for bytes you actually have — a file you `Read`/encode, a URL\nyou fetch, or a `data:` URI the user pastes as **text**. Send them once: the result's\n`image_inputs.base64_uploaded[].file_uri` is a `files/<id>` to reuse via `images_file_uris`.\n\n**Iterate on ONE image conversationally (multi-turn):**\n```\nr1 = gemini_interact(input: \"a cozy reading nook, watercolor\")\n   → { images: [...], interaction_id: \"v1_abc…\" }\nr2 = gemini_interact(input: \"add a sleeping cat on the chair\",\n                     previous_interaction_id: r1.interaction_id)\n   → refined image that preserves r1; returns a NEW interaction_id\nr3 = gemini_interact(input: \"warmer lighting\", continue_last: true)\n   → same chain, without threading the id (uses the session's most recent interaction)\n```\nPrefer this over re-running `gemini_image_edit` when you're making a *series* of incremental edits — the model keeps the prior result in context. Every result echoes `interaction_id` (and `previous_interaction_id` when chaining) plus a `hint` with the exact follow-up call.\n\n**Generate a video (runs long, use async):**\n```\njob = gemini_video_generate(prompt: \"a paper boat sailing down a rain gutter, cinematic\",\n                            aspect_ratio: \"16:9\", resolution: \"360p\", async: true)\n   → { job_id, status: \"running\" }   (returns immediately — no host timeout)\ngemini_get_result(job_id: job.job_id)\n   → \"running\" until done, then the MP4 path on disk\n# Draft at 360p, re-run the keeper at 1080p — video bills per output token, so the\n# resolution is roughly the price.\n# Animate a still instead: gemini_video_generate(prompt: \"…\", task: \"image_to_video\", images: [\"/path/still.png\"])\n# Interpolate between two stills: images: [\"/path/first.png\", \"/path/last.png\"]\n# Extend a clip: gemini_video_generate(prompt: \"…\", task: \"extend\", continue_last: true)\n```\n\n**Generate music:**\n```\ngemini_music_generate(prompt: \"warm lo-fi hip hop, mellow Rhodes, vinyl crackle, 70bpm\")\n   → 30s MP3 on disk (lyria-3-clip-preview)\n# Full-length song with vocals: gemini_music_generate(prompt: \"…\", model: \"lyria-3.5\")\n```\n**⚠️ Chat-pasted/attached images can't be fed to these tools directly.** A pasted\nimage reaches the assistant as a *vision* block — the assistant can SEE it but\nnever receives the original bytes, and the host doesn't write it to disk. So\nneither `images` (no file exists) nor `images_base64` (the bytes can't be\nreconstructed from a downscaled vision rendering) is obtainable from a paste.\nTo use a real reference photo, the **user** must make the bytes available: save\nthe file and give its **path** (→ `images`), drop it into the project dir, paste\nit as a **`data:` URI in text**, or host it at a **URL** (fetch → base64 →\n`images_base64`). This is a host/Cowork limitation, not an MCP one.\n\nTwo built-in ways to get past the unreachable-paste problem without any manual\nextraction:\n- **`from_clipboard: true`** (macOS) — the tool reads the image off the system\n  clipboard itself (osascript), downscales it, and uses it. The user just needs\n  to **copy** the image (⌘C — distinct from pasting it inline into chat, which\n  doesn't keep it on the clipboard). Works on every image tool:\n  `gemini_image_edit(prompt: \"…\", from_clipboard: true)`.\n- **`GEMINI_INPUT_DIR`** — point it at a folder (e.g. Cowork's `uploads/`); then a\n  **bare filename** resolves against it: `gemini_image_edit(prompt: \"…\", images: [\"house.jpg\"])`.\n\n## Prompting playbook\n\nCondensed from Google's official Nano Banana prompting guide. Core rule:\n**describe the scene, don't list keywords** — narrative sentences beat tag soups.\n\n**Best practices:**\n- **Be hyper-specific.** \"Ornate elven plate armor, etched with silver leaf patterns\" beats \"fantasy armor\".\n- **Give context & intent.** \"Create a logo for a high-end minimalist skincare brand\" beats \"create a logo\".\n- **Iterate conversationally** (`gemini_interact`): \"warmer lighting\", \"same, but more serious expression\".\n- **Step-by-step for complex scenes.** \"First, a misty forest background. Then a stone altar in the foreground. Finally a glowing sword on the altar.\"\n- **Semantic negatives.** Describe what you want positively — \"an empty, deserted street\" — instead of \"no cars\".\n- **Camera language controls composition.** Wide-angle / macro / low-angle perspective / 85mm portrait lens / three-point softbox lighting.\n\n**Generation templates (abbreviated):**\n- *Photorealistic:* `photorealistic [shot type] of [subject] in [setting], [lighting], shot from [angle] with [lens]`\n- *Sticker/illustration:* `[style] sticker of [subject] doing [activity], bold outlines, cel-shading, [palette], white background`\n- *Text in image:* `create a [type] for [brand] with the text \"[exact text]\" in a [font style]` — Gemini renders text well; Pro is best for professional assets. Tip: generate the wording first, then ask for the image containing it.\n- *Product shot:* `high-resolution studio-lit photo of [product] on [surface], [lighting setup], [angle], sharp focus on [detail]`\n- *Minimalist/negative space:* `single [subject] in [frame position], vast empty [color] background` — for text-overlay backgrounds.\n- *Comic/storyboard:* `make a 3 panel comic in [style]; put the character in [scene]` (Pro or 3.1 Flash).\n\n**Editing templates (abbreviated):**\n- *Add/remove:* `using the provided image of [subject], [add/remove] [element]; match the original style/lighting/perspective`\n- *Inpaint (semantic mask):* `change only the [element] to [new element]; keep everything else exactly the same`\n- *Style transfer:* `transform the photo of [subject] into the style of [artist/style]; preserve composition`\n- *Compose:* `take the [element from image 1] and place it with [element from image 2]; adjust lighting/shadows to match`\n- *Detail preservation:* describe the critical element (face, logo) in detail and say it must \"remain completely unchanged\"\n- *Sketch → finished:* `turn this rough sketch of [subject] into a [style] photo; keep [features], add [details]`\n- *Character 360°:* iterate angles via `gemini_interact` (\"in profile looking right\"), feeding prior outputs back for consistency\n\n## Notes\n\n- **Confirmations.** Local file inputs (`images`, `master_images`, `video_path`, `gemini_upload_file`'s `path`) and every delete are confirmed before anything is sent: a confirmation prompt where the client supports one; otherwise the first call does nothing and returns `status: \"confirmation-required\"` with a `preview` (resolved paths, MIME types, sizes — or the method/path being deleted) and a `confirmToken`. Show the preview to the user, and only after they approve call again with the **same arguments** plus `confirmToken`. The token is single-use, expires (default 10 min), and is bound to those exact arguments — change anything and the call is refused (`DRAFT_CHANGED`, with a fresh preview and token). `MCP_CONFIRM_MODE` on the server picks `ask-user` (default), `auto` (the model may approve after reviewing the preview) or `refuse`. Text prompts, URLs, `files/` references and base64 inputs are not gated.\n- **Input images** accept either file **paths** (`images` / `master_images`) or **base64/data-URI values** (`images_base64` / `master_images_base64`).\n- **`seed`** makes a result reproducible; it's echoed in the result metadata (a random one is chosen + echoed when omitted). `count>1` uses `seed, seed+1, …` so the images differ. Determinism isn't fully guaranteed by the model.\n- **`filename`/`basename`** set the output name (extension stripped); names never overwrite (a `-2`, `-3` suffix is added). The result echoes the absolute path(s), `model`, `seed`, and aspect/size.\n- **No edit-strength control.** Gemini exposes no denoise/strength knob, and Nano Banana over-preserves the input — big structural edits (\"move/remove/shrink\", add a mat border) are often ignored. Workarounds: reroll with a different `seed`, raise `thinking_level` to `high`, use forceful wording, do layout changes (padding/borders) externally, or use `gemini_interact` multi-turn.\n- **`thinking_level`** (`minimal`/`high`, Gemini 3 models) controls reasoning depth — `high` can improve complex compositions/edits at higher latency/cost.\n- **Model text.** When the model returns a caption/explanation (mostly Gemini 3 **Pro**), it's surfaced as `text` in the result metadata.\n- **`google_search: true`** grounds the image in live Google Search (current events, weather, real data — great for infographics). The result metadata includes `grounding` with the `queries` run and the `sources` (`{uri, title}`) used. (`gemini_interact` surfaces `grounding.queries` — the Interactions API returns no clean source list.)\n- **`search_types`** (`gemini_interact` only): `[\"web_search\", \"image_search\"]` picks the grounding search types (setting it implies `google_search`). **`image_search`** (gemini-3.1-flash-image only) pulls web images via Google Image Search as *visual* references — useful for real-world subjects (a specific butterfly species, a landmark, a product). ⚠️ Two catches: Google ToS require **displaying the returned `grounding.search_suggestions` HTML chips** to the user, and image_search won't depict real people from web images.\n- **`video_url`** (a public YouTube URL, on `gemini_image_generate` / `gemini_interact`) generates an image from a video reference — **requires a Flash model** (e.g. `model: \"gemini-3.1-flash-image\"`). For a **local video file**, use **`video_path`** instead: the file is uploaded to the Gemini Files API (streamed from disk, 2 GB max), waited to `ACTIVE`, and referenced by its `files/…` uri. The result metadata echoes `video_file` (`{uri, name, expires}`, ~48h retention) — reuse that uri as `video_url` in later calls to skip re-uploading.\n- **`gemini_interact`** is the multi-turn path: it returns an `interaction_id`; thread it back via `previous_interaction_id` for conversational refinement. Output is **JPEG only**. (The Interactions API is GA as of 2026-07; it uses a different request shape than the `generate`/`edit`/`set` tools.)\n- `output_dir` per-call overrides `$GEMINI_OUTPUT_DIR` overrides cwd. `inline: true` returns bytes (with a metadata text block) instead of writing.\n- `count` and `scenes` are mutually exclusive in `gemini_image_set`; `reference_mode: \"chain\"` references the previous image instead of the master.\n- Aspect ratios: `1:1`, `16:9`, `9:16`, `4:3`, `3:4`, `2:3`, `3:2`, … · Image sizes: `512` (0.5K, Flash only), `1K`, `2K`, `4K`. `4K` is the max native output — true 18×24 in @ 300 DPI (5400×7200) needs an external upscale step.\n- All generated images carry a **SynthID** watermark (Google).\n- The model can mis-render text/Roman numerals (e.g. years) — verify any text in the output; it's a model limitation, not a tool setting.\n- **Best-performance languages:** EN, plus ar, de, es-MX, fr, hi, id, it, ja, ko, pt-BR, ru, ua, vi, zh-CN — prefer prompting in one of these.\n- Asking the *model* for \"N images\" in one prompt is unreliable (documented limitation) — use the `count` parameter instead; it makes N independent calls with distinct seeds.\n- Server logs to stderr only — stdout is reserved for JSON-RPC.\n\nFile v2.3.0:_meta.json\n\n{\n  \"ownerId\": \"kn700jq4sjtf2anb0rk3ft4p7n856872\",\n  \"slug\": \"gemini-mcp\",\n  \"version\": \"2.3.0\",\n  \"publishedAt\": 1790351462036\n}\n\nFile v2.3.0:skill-card.md\n\n## Description:\n\nHelps agents generate and edit images, create short videos, and compose music with Google Gemini through an MCP server.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[chrischall](https://clawhub.ai/user/chrischall)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and other agent users can generate, revise, and reuse image, video, and music assets through Gemini-backed tools, including workflows that use reference media.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Prompts and selected media are sent to Gemini or a hosted deployment.\n\nMitigation: Avoid sensitive photos or documents and review confirmation previews before uploading local files.\n\nRisk: An API credential and reusable uploaded files need care.\n\nMitigation: Use a dedicated Gemini API key and delete reusable uploads when no longer needed.\n\nRisk: Generated media may be saved in an unexpected local directory.\n\nMitigation: Set GEMINI_OUTPUT_DIR to a directory you expect.\n\n## Reference(s):\n\n- [ClawHub gemini-mcp release](https://clawhub.ai/chrischall/skills/gemini-mcp)\n- [gemini-mcp npm package](https://www.npmjs.com/package/@chrischall/gemini-mcp)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Configuration instructions, Guidance]\n\n**Output Format:** [Markdown with MCP tool-call and configuration examples]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Tool results can provide local paths to generated image, video, and audio files.]\n\n## Skill Version(s):\n\n2.3.0 (source: server-resolved release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v2.2.0: 3 files, 11326 bytes\n\nFiles: skill-card.md (2438b), SKILL.md (23679b), _meta.json (129b)\n\nFile v2.2.0:SKILL.md\n\n---\nname: gemini-mcp\ndescription: Generate and edit images, video, and music with Google Gemini models via MCP. Use when the user asks to generate, create, or edit images (Gemini / Nano Banana), produce a consistent set of images, compose/blend multiple images, generate a short video (text→video or image→video, via the omni model), or generate music/audio clips (via Lyria). Triggers on phrases like \"generate an image of\", \"edit this image with Gemini\", \"create a set of consistent images\", \"make a video of\", \"generate a video\", \"generate music\", \"make a song/audio clip\", \"use Nano Banana to make\", or any request to produce images, video, or music via the Gemini API. Requires the @chrischall/gemini-mcp package installed and the gemini server registered (see Setup below).\n---\n\n# gemini-mcp\n\nMCP server for Google Gemini media generation — natural-language **image**, **video**, and **music** creation via the Gemini API (Nano Banana / Nano Banana Pro images, omni video, Lyria music).\n\n- **npm:** [npmjs.com/package/@chrischall/gemini-mcp](https://www.npmjs.com/package/@chrischall/gemini-mcp)\n- **Source:** [github.com/chrischall/gemini-mcp](https://github.com/chrischall/gemini-mcp)\n\n## Setup\n\n### Option A — npx (recommended)\n\nAdd to `.mcp.json` in your project or `~/.claude/mcp.json`:\n\n```json\n{\n  \"mcpServers\": {\n    \"gemini\": {\n      \"command\": \"npx\",\n      \"args\": [\"-y\", \"@chrischall/gemini-mcp\"],\n      \"env\": {\n        \"GEMINI_API_KEY\": \"your-api-key-here\"\n      }\n    }\n  }\n}\n```\n\n### Option B — from source\n\n```bash\ngit clone https://github.com/chrischall/gemini-mcp\ncd gemini-mcp\nnpm install && npm run build\n```\n\nThen add to `.mcp.json`:\n\n```json\n{\n  \"mcpServers\": {\n    \"gemini\": {\n      \"command\": \"node\",\n      \"args\": [\"/path/to/gemini-mcp/dist/index.js\"],\n      \"env\": {\n        \"GEMINI_API_KEY\": \"your-api-key-here\"\n      }\n    }\n  }\n}\n```\n\nOr use a `.env` file in the project directory with `GEMINI_API_KEY=<value>`.\n\n### Getting your API key\n\n1. Go to [aistudio.google.com/apikey](https://aistudio.google.com/apikey)\n2. Create an API key (requires a Google account)\n3. Copy the key and set it as `GEMINI_API_KEY`\n\nNote: Image generation requires a billing-enabled Google Cloud project.\n\n## Environment Variables\n\n| Variable | Required | Description |\n|---|---|---|\n| `GEMINI_API_KEY` | Yes | Your Google Gemini API key |\n| `GEMINI_IMAGE_MODEL` | No | Override the default image model (default: `gemini-3.1-flash-image`) |\n| `GEMINI_OUTPUT_DIR` | No | Default directory for saved images (default: current working directory) |\n| `GEMINI_INPUT_DIR` | No | Directory to resolve bare input-image filenames against (e.g. point at Cowork's `uploads/` folder so `images: [\"house.jpg\"]` works) |\n\n## Tools\n\n### Models\n| Tool | Description |\n|------|-------------|\n| `gemini_list_models` | List available Gemini image models and the current default |\n\n**Which model to pick** (per-call `model`, or `GEMINI_IMAGE_MODEL`):\n\n| Model | When to use | Reference-image caps (of 14 max) |\n|-------|-------------|----------------------------------|\n| `gemini-3.1-flash-image` (Nano Banana 2) | The versatile generalist workhorse for all tasks — balances speed with state-of-the-art 4K generation, world knowledge, and reliable text rendering; excels at multi-reference-image processing and consistency. Only model with video input + image_search grounding | 10 objects + 4 characters + 3 style refs |\n| `gemini-3-pro-image` (Nano Banana Pro) | The premium choice for the most complex visual tasks — highest world knowledge, advanced localization, accurate brand consistency, precision creative control | 6 objects + 5 characters |\n| `gemini-3.1-flash-lite-image` (Nano Banana 2 Lite) | The fastest/cheapest for simple tasks — 1K output only, no Google Search grounding | 14 objects (no character consistency) |\n\n### Image Generation\n| Tool | Description |\n|------|-------------|\n| `gemini_image_generate(prompt, count?, images_url?, images_file_uris?, images?, images_base64?, video_url?, video_path?, google_search?, seed?, filename?, model?, aspect_ratio?, image_size?, thinking_level?, output_dir?, inline?)` | Generate image(s) from a text prompt (optionally image-conditioned — see **Reference images** below — or video-conditioned via `video_url`/`video_path`) |\n| `gemini_image_edit(prompt, images_url?, images_file_uris?, images?, images_base64?, google_search?, seed?, filename?, model?, aspect_ratio?, image_size?, thinking_level?, output_dir?, inline?)` | Edit or compose input image(s) with a text instruction. Requires ≥1 input from any of the four reference forms |\n| `gemini_image_set(master_prompt, scenes? \\| count?, reference_mode?, master_images_url?, master_images_file_uris?, master_images?, master_images_base64?, google_search?, seed?, basename?, model?, thinking_level?, ...)` | Master image (optionally seeded from a reference photo) plus N consistent images referencing it. `master_images_url` / `master_images_file_uris` are resolved **once** and passed to the master *and* every scene |\n\n### Reference images — four ways in, one that costs context\n\nThese apply to **every** tool that takes a reference image: the four image tools, plus\n`gemini_video_generate` (reference stills) and `gemini_music_generate`.\n\n| Parameter | Where the bytes travel | Context cost |\n|---|---|---|\n| `images_url` (`master_images_url`) | the **server** downloads the https URL | none |\n| `images_file_uris` (`master_images_file_uris`) | a `files/<id>` reference from `gemini_upload_file` | none |\n| `images` | read off local disk (stdio builds only) | none |\n| `images_base64` | **through the tool-call JSON** | **~14k tokens per JPEG** |\n\n**Reach for `images_base64` last.** It costs ~14k tokens per modest photo, and a truncated file\nread produces base64 that still *looks* valid — so the corruption surfaces as a bad generation,\nnot an error.\n\n- `images_url` accepts public `https://` URLs only (private/loopback/link-local hosts refused,\n  every redirect revalidated), must be `Content-Type: image/*`, capped at 15MB. Errors name the\n  failing URL. Over 6MB is auto-uploaded to the Files API instead of inlined.\n- `images_file_uris` accepts `files/<id>` or the full uri. Retained **~48h**; reusable across\n  any number of calls until then.\n- On stdio, an `images` path referenced **more than once in a session** is auto-uploaded to the\n  Files API (keyed on path + mtime + size) so the bytes stop being re-sent.\n- `images_base64` is uploaded on the **first** sighting, not the second — the tokens are already\n  spent by then. The result reports it under `image_inputs.base64_uploaded[].file_uri`: pass that\n  to `images_file_uris` on the next call instead of pasting the bytes again.\n\n### Files API\n| Tool | Description |\n|------|-------------|\n| `gemini_upload_file(url? \\| data_base64? \\| path?, mime_type?, display_name?, confirmToken?)` | Upload once, get a reusable `files/<id>`. Exactly one source. `url` is fetched by the server (image/video/audio, ≤100MB); `path` is stdio-only and confirmed first (see Notes); `data_base64` is the last resort |\n| `gemini_list_files(page_size?)` | List current uploads with MIME types and expiry |\n| `gemini_delete_file(file_uri, confirmToken?)` | Delete an upload before its ~48h expiry (confirmed first — see Notes) |\n\nOn the **hosted hosted deployment** there is also `POST /upload`, behind the same OAuth token as\n`/mcp` — the zero-base64 path for an agent with a shell:\n\n```bash\ncurl -X POST https://<hosted deployment>/upload \\\n  -H \"Authorization: Bearer $ACCESS_TOKEN\" \\\n  -H \"Content-Type: image/jpeg\" \\\n  --data-binary @photo.jpg\n# → {\"file_uri\":\"files/abc123\", ...}   then: images_file_uris: [\"files/abc123\"]\n```\n\n### Multi-turn (Interactions API)\n| Tool | Description |\n|------|-------------|\n| `gemini_interact(input, previous_interaction_id?, continue_last?, images_url?, images_file_uris?, images?, images_base64?, video_url?, video_path?, google_search?, search_types?, model?, aspect_ratio?, image_size?, thinking_level?, filename?, output_dir?, inline?)` | **Preferred tool for iterative refinement.** Generate/edit via Gemini's **Interactions API**. Returns an `interaction_id`; pass it back as `previous_interaction_id` (or set `continue_last: true` to reuse the session's most recent one) to **iteratively refine the same image** conversationally — do NOT start a new interaction or re-upload the image per tweak. Output is **JPEG**. |\n\n### Video & Music (preview — funded account)\n| Tool | Description |\n|------|-------------|\n| `gemini_video_generate(prompt, aspect_ratio?, resolution?, task?, images_url?, images_file_uris?, images?, images_base64?, from_clipboard?, previous_interaction_id?, continue_last?, model?, filename?, output_dir?, timeout_ms?, idempotency_key?, async?)` | Generate a short video (~10s) via the Gemini omni model: `text_to_video` (default), `image_to_video` / `reference_to_video` (supply reference image[s]), interpolation (pass first frame then last frame as `images`), or `edit` / `extend` (with `previous_interaction_id` / `continue_last`; extensions add ~3-10s each, to ~40s). `aspect_ratio` is `16:9` or `9:16`; `resolution` is `360p`/`720p` (default)/`1080p`/`4k` and is the cost lever — a 10s 360p clip runs about a third of 720p. Written to disk as **MP4**. Runs long — use `async: true` + `gemini_get_result`, or raise `timeout_ms`. |\n| `gemini_music_generate(prompt, model?, images_url?, images_file_uris?, images?, images_base64?, from_clipboard?, filename?, output_dir?, inline?, timeout_ms?, idempotency_key?, async?)` | Generate music from a prompt (mood/genre/instruments/structure/lyrics) via a Lyria model: `lyria-3-clip-preview` (30s instrumental, default, $0.04), `lyria-3.5` (full-length song with vocals, $0.08) or `lyria-3-pro-preview` ($0.08). Output is MP3. **Single-turn** — there is no follow-up refinement, so put the whole brief in the prompt. Written to disk, or returned inline. |\n\n### Async / idempotency (any generation tool)\n| Tool | Description |\n|------|-------------|\n| `gemini_get_result(job_id)` | Fetch a generation started with `async: true`. Returns `running` while in flight, then the normal result on completion. Lets a long video/music/image generation outlive a host's `tools/call` timeout. All generation tools also accept `idempotency_key` — a repeat call with the same key returns the recorded result (`reused: true`) instead of billing again. |\n\n## Workflows\n\n**Generate a single image:**\n```\ngemini_image_generate(prompt: \"a red maple leaf on white background, studio photo\")\n→ returns path to saved PNG\n```\n\n**Generate multiple variations:**\n```\ngemini_image_generate(prompt: \"a cartoon fox\", count: 4, output_dir: \"/tmp/foxes\")\n→ returns paths to 4 PNG files\n```\n\n**Edit an existing image:**\n```\ngemini_image_edit(prompt: \"make the background blue\", images: [\"/path/to/image.png\"])\n→ returns path to edited PNG\n```\n\n**Edit an image you only have a URL for (nothing downloads into the conversation):**\n```\ngemini_image_edit(prompt: \"make the background blue\", images_url: [\"https://example.com/photo.jpg\"])\n→ the server fetches the URL itself; returns path to edited PNG\n```\n\n**Reuse one photo across many generations (hosted hosted deployment, agent with a shell):**\n```\n$ curl -X POST https://<hosted deployment>/upload -H \"Authorization: Bearer $TOKEN\" \\\n    -H \"Content-Type: image/jpeg\" --data-binary @photo.jpg\n  → {\"file_uri\": \"files/abc123\"}\n\ngemini_image_edit(prompt: \"make it winter\",  images_file_uris: [\"files/abc123\"])\ngemini_image_edit(prompt: \"make it sunrise\", images_file_uris: [\"files/abc123\"])\n→ two edits, one upload, zero image bytes in context. Valid ~48h.\n```\n\n**Generate a consistent set (master + scenes):**\n```\ngemini_image_set(\n  master_prompt: \"a cartoon fox named Rusty, orange fur, blue scarf\",\n  scenes: [\"Rusty waving hello\", \"Rusty eating an apple\", \"Rusty sleeping\"]\n)\n→ returns paths to master + 3 scene images, all consistent\n```\n\n**Generate variations of a concept:**\n```\ngemini_image_set(\n  master_prompt: \"minimalist logo for a coffee shop\",\n  count: 5\n)\n→ returns master + 5 variations\n```\n\n**Use a reference photo by value (when you have the bytes):**\n```\ngemini_image_edit(\n  prompt: \"place this house on a vintage travel-poster background\",\n  images_base64: [\"data:image/jpeg;base64,/9j/4AAQ...\"]   // or raw base64\n)\n→ returns path to the edited image\n```\n`images_base64` is for bytes you actually have — a file you `Read`/encode, a URL\nyou fetch, or a `data:` URI the user pastes as **text**. Send them once: the result's\n`image_inputs.base64_uploaded[].file_uri` is a `files/<id>` to reuse via `images_file_uris`.\n\n**Iterate on ONE image conversationally (multi-turn):**\n```\nr1 = gemini_interact(input: \"a cozy reading nook, watercolor\")\n   → { images: [...], interaction_id: \"v1_abc…\" }\nr2 = gemini_interact(input: \"add a sleeping cat on the chair\",\n                     previous_interaction_id: r1.interaction_id)\n   → refined image that preserves r1; returns a NEW interaction_id\nr3 = gemini_interact(input: \"warmer lighting\", continue_last: true)\n   → same chain, without threading the id (uses the session's most recent interaction)\n```\nPrefer this over re-running `gemini_image_edit` when you're making a *series* of incremental edits — the model keeps the prior result in context. Every result echoes `interaction_id` (and `previous_interaction_id` when chaining) plus a `hint` with the exact follow-up call.\n\n**Generate a video (runs long, use async):**\n```\njob = gemini_video_generate(prompt: \"a paper boat sailing down a rain gutter, cinematic\",\n                            aspect_ratio: \"16:9\", resolution: \"360p\", async: true)\n   → { job_id, status: \"running\" }   (returns immediately — no host timeout)\ngemini_get_result(job_id: job.job_id)\n   → \"running\" until done, then the MP4 path on disk\n# Draft at 360p, re-run the keeper at 1080p — video bills per output token, so the\n# resolution is roughly the price.\n# Animate a still instead: gemini_video_generate(prompt: \"…\", task: \"image_to_video\", images: [\"/path/still.png\"])\n# Interpolate between two stills: images: [\"/path/first.png\", \"/path/last.png\"]\n# Extend a clip: gemini_video_generate(prompt: \"…\", task: \"extend\", continue_last: true)\n```\n\n**Generate music:**\n```\ngemini_music_generate(prompt: \"warm lo-fi hip hop, mellow Rhodes, vinyl crackle, 70bpm\")\n   → 30s MP3 on disk (lyria-3-clip-preview)\n# Full-length song with vocals: gemini_music_generate(prompt: \"…\", model: \"lyria-3.5\")\n```\n**⚠️ Chat-pasted/attached images can't be fed to these tools directly.** A pasted\nimage reaches the assistant as a *vision* block — the assistant can SEE it but\nnever receives the original bytes, and the host doesn't write it to disk. So\nneither `images` (no file exists) nor `images_base64` (the bytes can't be\nreconstructed from a downscaled vision rendering) is obtainable from a paste.\nTo use a real reference photo, the **user** must make the bytes available: save\nthe file and give its **path** (→ `images`), drop it into the project dir, paste\nit as a **`data:` URI in text**, or host it at a **URL** (fetch → base64 →\n`images_base64`). This is a host/Cowork limitation, not an MCP one.\n\nTwo built-in ways to get past the unreachable-paste problem without any manual\nextraction:\n- **`from_clipboard: true`** (macOS) — the tool reads the image off the system\n  clipboard itself (osascript), downscales it, and uses it. The user just needs\n  to **copy** the image (⌘C — distinct from pasting it inline into chat, which\n  doesn't keep it on the clipboard). Works on every image tool:\n  `gemini_image_edit(prompt: \"…\", from_clipboard: true)`.\n- **`GEMINI_INPUT_DIR`** — point it at a folder (e.g. Cowork's `uploads/`); then a\n  **bare filename** resolves against it: `gemini_image_edit(prompt: \"…\", images: [\"house.jpg\"])`.\n\n## Prompting playbook\n\nCondensed from Google's official Nano Banana prompting guide. Core rule:\n**describe the scene, don't list keywords** — narrative sentences beat tag soups.\n\n**Best practices:**\n- **Be hyper-specific.** \"Ornate elven plate armor, etched with silver leaf patterns\" beats \"fantasy armor\".\n- **Give context & intent.** \"Create a logo for a high-end minimalist skincare brand\" beats \"create a logo\".\n- **Iterate conversationally** (`gemini_interact`): \"warmer lighting\", \"same, but more serious expression\".\n- **Step-by-step for complex scenes.** \"First, a misty forest background. Then a stone altar in the foreground. Finally a glowing sword on the altar.\"\n- **Semantic negatives.** Describe what you want positively — \"an empty, deserted street\" — instead of \"no cars\".\n- **Camera language controls composition.** Wide-angle / macro / low-angle perspective / 85mm portrait lens / three-point softbox lighting.\n\n**Generation templates (abbreviated):**\n- *Photorealistic:* `photorealistic [shot type] of [subject] in [setting], [lighting], shot from [angle] with [lens]`\n- *Sticker/illustration:* `[style] sticker of [subject] doing [activity], bold outlines, cel-shading, [palette], white background`\n- *Text in image:* `create a [type] for [brand] with the text \"[exact text]\" in a [font style]` — Gemini renders text well; Pro is best for professional assets. Tip: generate the wording first, then ask for the image containing it.\n- *Product shot:* `high-resolution studio-lit photo of [product] on [surface], [lighting setup], [angle], sharp focus on [detail]`\n- *Minimalist/negative space:* `single [subject] in [frame position], vast empty [color] background` — for text-overlay backgrounds.\n- *Comic/storyboard:* `make a 3 panel comic in [style]; put the character in [scene]` (Pro or 3.1 Flash).\n\n**Editing templates (abbreviated):**\n- *Add/remove:* `using the provided image of [subject], [add/remove] [element]; match the original style/lighting/perspective`\n- *Inpaint (semantic mask):* `change only the [element] to [new element]; keep everything else exactly the same`\n- *Style transfer:* `transform the photo of [subject] into the style of [artist/style]; preserve composition`\n- *Compose:* `take the [element from image 1] and place it with [element from image 2]; adjust lighting/shadows to match`\n- *Detail preservation:* describe the critical element (face, logo) in detail and say it must \"remain completely unchanged\"\n- *Sketch → finished:* `turn this rough sketch of [subject] into a [style] photo; keep [features], add [details]`\n- *Character 360°:* iterate angles via `gemini_interact` (\"in profile looking right\"), feeding prior outputs back for consistency\n\n## Notes\n\n- **Confirmations.** Local file inputs (`images`, `master_images`, `video_path`, `gemini_upload_file`'s `path`) and every delete are confirmed before anything is sent: a confirmation prompt where the client supports one; otherwise the first call does nothing and returns `status: \"confirmation-required\"` with a `preview` (resolved paths, MIME types, sizes — or the method/path being deleted) and a `confirmToken`. Show the preview to the user, and only after they approve call again with the **same arguments** plus `confirmToken`. The token is single-use, expires (default 10 min), and is bound to those exact arguments — change anything and the call is refused (`DRAFT_CHANGED`, with a fresh preview and token). `MCP_CONFIRM_MODE` on the server picks `ask-user` (default), `auto` (the model may approve after reviewing the preview) or `refuse`. Text prompts, URLs, `files/` references and base64 inputs are not gated.\n- **Input images** accept either file **paths** (`images` / `master_images`) or **base64/data-URI values** (`images_base64` / `master_images_base64`).\n- **`seed`** makes a result reproducible; it's echoed in the result metadata (a random one is chosen + echoed when omitted). `count>1` uses `seed, seed+1, …` so the images differ. Determinism isn't fully guaranteed by the model.\n- **`filename`/`basename`** set the output name (extension stripped); names never overwrite (a `-2`, `-3` suffix is added). The result echoes the absolute path(s), `model`, `seed`, and aspect/size.\n- **No edit-strength control.** Gemini exposes no denoise/strength knob, and Nano Banana over-preserves the input — big structural edits (\"move/remove/shrink\", add a mat border) are often ignored. Workarounds: reroll with a different `seed`, raise `thinking_level` to `high`, use forceful wording, do layout changes (padding/borders) externally, or use `gemini_interact` multi-turn.\n- **`thinking_level`** (`minimal`/`high`, Gemini 3 models) controls reasoning depth — `high` can improve complex compositions/edits at higher latency/cost.\n- **Model text.** When the model returns a caption/explanation (mostly Gemini 3 **Pro**), it's surfaced as `text` in the result metadata.\n- **`google_search: true`** grounds the image in live Google Search (current events, weather, real data — great for infographics). The result metadata includes `grounding` with the `queries` run and the `sources` (`{uri, title}`) used. (`gemini_interact` surfaces `grounding.queries` — the Interactions API returns no clean source list.)\n- **`search_types`** (`gemini_interact` only): `[\"web_search\", \"image_search\"]` picks the grounding search types (setting it implies `google_search`). **`image_search`** (gemini-3.1-flash-image only) pulls web images via Google Image Search as *visual* references — useful for real-world subjects (a specific butterfly species, a landmark, a product). ⚠️ Two catches: Google ToS require **displaying the returned `grounding.search_suggestions` HTML chips** to the user, and image_search won't depict real people from web images.\n- **`video_url`** (a public YouTube URL, on `gemini_image_generate` / `gemini_interact`) generates an image from a video reference — **requires a Flash model** (e.g. `model: \"gemini-3.1-flash-image\"`). For a **local video file**, use **`video_path`** instead: the file is uploaded to the Gemini Files API (streamed from disk, 2 GB max), waited to `ACTIVE`, and referenced by its `files/…` uri. The result metadata echoes `video_file` (`{uri, name, expires}`, ~48h retention) — reuse that uri as `video_url` in later calls to skip re-uploading.\n- **`gemini_interact`** is the multi-turn path: it returns an `interaction_id`; thread it back via `previous_interaction_id` for conversational refinement. Output is **JPEG only**. (The Interactions API is GA as of 2026-07; it uses a different request shape than the `generate`/`edit`/`set` tools.)\n- `output_dir` per-call overrides `$GEMINI_OUTPUT_DIR` overrides cwd. `inline: true` returns bytes (with a metadata text block) instead of writing.\n- `count` and `scenes` are mutually exclusive in `gemini_image_set`; `reference_mode: \"chain\"` references the previous image instead of the master.\n- Aspect ratios: `1:1`, `16:9`, `9:16`, `4:3`, `3:4`, `2:3`, `3:2`, … · Image sizes: `512` (0.5K, Flash only), `1K`, `2K`, `4K`. `4K` is the max native output — true 18×24 in @ 300 DPI (5400×7200) needs an external upscale step.\n- All generated images carry a **SynthID** watermark (Google).\n- The model can mis-render text/Roman numerals (e.g. years) — verify any text in the output; it's a model limitation, not a tool setting.\n- **Best-performance languages:** EN, plus ar, de, es-MX, fr, hi, id, it, ja, ko, pt-BR, ru, ua, vi, zh-CN — prefer prompting in one of these.\n- Asking the *model* for \"N images\" in one prompt is unreliable (documented limitation) — use the `count` parameter instead; it makes N independent calls with distinct seeds.\n- Server logs to stderr only — stdout is reserved for JSON-RPC.\n\nFile v2.2.0:_meta.json\n\n{\n  \"ownerId\": \"kn700jq4sjtf2anb0rk3ft4p7n856872\",\n  \"slug\": \"gemini-mcp\",\n  \"version\": \"2.2.0\",\n  \"publishedAt\": 1790262772611\n}\n\nFile v2.2.0:skill-card.md\n\n## Description:\n\ngemini-mcp helps agents generate and edit images, video, and music with Google Gemini models through an MCP server.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[chrischall](https://clawhub.ai/user/chrischall)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and external users use this skill to configure and call a Gemini MCP server for media generation workflows, including image creation and editing, video generation, and music generation.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill requires a Gemini API key and may send selected prompts or media to Gemini or a configured hosted service.\n\nMitigation: Install only when that data flow is acceptable, avoid sensitive prompts or media unless approved, and use appropriately scoped credentials.\n\nRisk: File uploads, deletion, local file paths, URLs, and clipboard inputs can expose unintended media or files.\n\nMitigation: Use explicit file paths or URLs, review confirmation previews before approving file operations, and avoid clipboard input unless the current clipboard contents are known.\n\nRisk: The runtime package is installed from npm or source outside NVIDIA control.\n\nMitigation: Pin or review the npm package or source before deployment, and follow the security guidance from the release evidence.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/chrischall/skills/gemini-mcp)\n- [npm package: @chrischall/gemini-mcp](https://www.npmjs.com/package/@chrischall/gemini-mcp)\n- [Source repository declared by artifact](https://github.com/chrischall/gemini-mcp)\n- [Google AI Studio API key setup](https://aistudio.google.com/apikey)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, Shell commands, Configuration, Guidance, Files]\n\n**Output Format:** [Markdown guidance with JSON configuration snippets, tool-call examples, shell commands, and generated media file paths]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May produce image, video, and audio files through the configured Gemini MCP server.]\n\n## Skill Version(s):\n\n2.2.0 (source: server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v2.1.4: 3 files, 10864 bytes\n\nFiles: skill-card.md (2263b), SKILL.md (22697b), _meta.json (129b)\n\nFile v2.1.4:SKILL.md\n\n---\nname: gemini-mcp\ndescription: Generate and edit images, video, and music with Google Gemini models via MCP. Use when the user asks to generate, create, or edit images (Gemini / Nano Banana), produce a consistent set of images, compose/blend multiple images, generate a short video (text→video or image→video, via the omni model), or generate music/audio clips (via Lyria). Triggers on phrases like \"generate an image of\", \"edit this image with Gemini\", \"create a set of consistent images\", \"make a video of\", \"generate a video\", \"generate music\", \"make a song/audio clip\", \"use Nano Banana to make\", or any request to produce images, video, or music via the Gemini API. Requires the @chrischall/gemini-mcp package installed and the gemini server registered (see Setup below).\n---\n\n# gemini-mcp\n\nMCP server for Google Gemini media generation — natural-language **image**, **video**, and **music** creation via the Gemini API (Nano Banana / Nano Banana Pro images, omni video, Lyria music).\n\n- **npm:** [npmjs.com/package/@chrischall/gemini-mcp](https://www.npmjs.com/package/@chrischall/gemini-mcp)\n- **Source:** [github.com/chrischall/gemini-mcp](https://github.com/chrischall/gemini-mcp)\n\n## Setup\n\n### Option A — npx (recommended)\n\nAdd to `.mcp.json` in your project or `~/.claude/mcp.json`:\n\n```json\n{\n  \"mcpServers\": {\n    \"gemini\": {\n      \"command\": \"npx\",\n      \"args\": [\"-y\", \"@chrischall/gemini-mcp\"],\n      \"env\": {\n        \"GEMINI_API_KEY\": \"your-api-key-here\"\n      }\n    }\n  }\n}\n```\n\n### Option B — from source\n\n```bash\ngit clone https://github.com/chrischall/gemini-mcp\ncd gemini-mcp\nnpm install && npm run build\n```\n\nThen add to `.mcp.json`:\n\n```json\n{\n  \"mcpServers\": {\n    \"gemini\": {\n      \"command\": \"node\",\n      \"args\": [\"/path/to/gemini-mcp/dist/index.js\"],\n      \"env\": {\n        \"GEMINI_API_KEY\": \"your-api-key-here\"\n      }\n    }\n  }\n}\n```\n\nOr use a `.env` file in the project directory with `GEMINI_API_KEY=<value>`.\n\n### Getting your API key\n\n1. Go to [aistudio.google.com/apikey](https://aistudio.google.com/apikey)\n2. Create an API key (requires a Google account)\n3. Copy the key and set it as `GEMINI_API_KEY`\n\nNote: Image generation requires a billing-enabled Google Cloud project.\n\n## Environment Variables\n\n| Variable | Required | Description |\n|---|---|---|\n| `GEMINI_API_KEY` | Yes | Your Google Gemini API key |\n| `GEMINI_IMAGE_MODEL` | No | Override the default image model (default: `gemini-3.1-flash-image`) |\n| `GEMINI_OUTPUT_DIR` | No | Default directory for saved images (default: current working directory) |\n| `GEMINI_INPUT_DIR` | No | Directory to resolve bare input-image filenames against (e.g. point at Cowork's `uploads/` folder so `images: [\"house.jpg\"]` works) |\n\n## Tools\n\n### Models\n| Tool | Description |\n|------|-------------|\n| `gemini_list_models` | List available Gemini image models and the current default |\n\n**Which model to pick** (per-call `model`, or `GEMINI_IMAGE_MODEL`):\n\n| Model | When to use | Reference-image caps (of 14 max) |\n|-------|-------------|----------------------------------|\n| `gemini-3.1-flash-image` (Nano Banana 2) | The versatile generalist workhorse for all tasks — balances speed with state-of-the-art 4K generation, world knowledge, and reliable text rendering; excels at multi-reference-image processing and consistency. Only model with video input + image_search grounding | 10 objects + 4 characters + 3 style refs |\n| `gemini-3-pro-image` (Nano Banana Pro) | The premium choice for the most complex visual tasks — highest world knowledge, advanced localization, accurate brand consistency, precision creative control | 6 objects + 5 characters |\n| `gemini-3.1-flash-lite-image` (Nano Banana 2 Lite) | The fastest/cheapest for simple tasks — 1K output only, no Google Search grounding | 14 objects (no character consistency) |\n\n### Image Generation\n| Tool | Description |\n|------|-------------|\n| `gemini_image_generate(prompt, count?, images_url?, images_file_uris?, images?, images_base64?, video_url?, video_path?, google_search?, seed?, filename?, model?, aspect_ratio?, image_size?, thinking_level?, output_dir?, inline?)` | Generate image(s) from a text prompt (optionally image-conditioned — see **Reference images** below — or video-conditioned via `video_url`/`video_path`) |\n| `gemini_image_edit(prompt, images_url?, images_file_uris?, images?, images_base64?, google_search?, seed?, filename?, model?, aspect_ratio?, image_size?, thinking_level?, output_dir?, inline?)` | Edit or compose input image(s) with a text instruction. Requires ≥1 input from any of the four reference forms |\n| `gemini_image_set(master_prompt, scenes? \\| count?, reference_mode?, master_images_url?, master_images_file_uris?, master_images?, master_images_base64?, google_search?, seed?, basename?, model?, thinking_level?, ...)` | Master image (optionally seeded from a reference photo) plus N consistent images referencing it. `master_images_url` / `master_images_file_uris` are resolved **once** and passed to the master *and* every scene |\n\n### Reference images — four ways in, one that costs context\n\nThese apply to **every** tool that takes a reference image: the four image tools, plus\n`gemini_video_generate` (reference stills) and `gemini_music_generate`.\n\n| Parameter | Where the bytes travel | Context cost |\n|---|---|---|\n| `images_url` (`master_images_url`) | the **server** downloads the https URL | none |\n| `images_file_uris` (`master_images_file_uris`) | a `files/<id>` reference from `gemini_upload_file` | none |\n| `images` | read off local disk (stdio builds only) | none |\n| `images_base64` | **through the tool-call JSON** | **~14k tokens per JPEG** |\n\n**Reach for `images_base64` last.** It costs ~14k tokens per modest photo, and a truncated file\nread produces base64 that still *looks* valid — so the corruption surfaces as a bad generation,\nnot an error.\n\n- `images_url` accepts public `https://` URLs only (private/loopback/link-local hosts refused,\n  every redirect revalidated), must be `Content-Type: image/*`, capped at 15MB. Errors name the\n  failing URL. Over 6MB is auto-uploaded to the Files API instead of inlined.\n- `images_file_uris` accepts `files/<id>` or the full uri. Retained **~48h**; reusable across\n  any number of calls until then.\n- On stdio, an `images` path referenced **more than once in a session** is auto-uploaded to the\n  Files API (keyed on path + mtime + size) so the bytes stop being re-sent.\n- `images_base64` is uploaded on the **first** sighting, not the second — the tokens are already\n  spent by then. The result reports it under `image_inputs.base64_uploaded[].file_uri`: pass that\n  to `images_file_uris` on the next call instead of pasting the bytes again.\n\n### Files API\n| Tool | Description |\n|------|-------------|\n| `gemini_upload_file(url? \\| data_base64? \\| path?, mime_type?, display_name?, confirm?)` | Upload once, get a reusable `files/<id>`. Exactly one source. `url` is fetched by the server (image/video/audio, ≤100MB); `path` is stdio-only and confirm-gated; `data_base64` is the last resort |\n| `gemini_list_files(page_size?)` | List current uploads with MIME types and expiry |\n| `gemini_delete_file(file_uri, confirm)` | Delete an upload before its ~48h expiry |\n\nOn the **hosted hosted deployment** there is also `POST /upload`, behind the same OAuth token as\n`/mcp` — the zero-base64 path for an agent with a shell:\n\n```bash\ncurl -X POST https://<hosted deployment>/upload \\\n  -H \"Authorization: Bearer $ACCESS_TOKEN\" \\\n  -H \"Content-Type: image/jpeg\" \\\n  --data-binary @photo.jpg\n# → {\"file_uri\":\"files/abc123\", ...}   then: images_file_uris: [\"files/abc123\"]\n```\n\n### Multi-turn (Interactions API)\n| Tool | Description |\n|------|-------------|\n| `gemini_interact(input, previous_interaction_id?, continue_last?, images_url?, images_file_uris?, images?, images_base64?, video_url?, video_path?, google_search?, search_types?, model?, aspect_ratio?, image_size?, thinking_level?, filename?, output_dir?, inline?)` | **Preferred tool for iterative refinement.** Generate/edit via Gemini's **Interactions API**. Returns an `interaction_id`; pass it back as `previous_interaction_id` (or set `continue_last: true` to reuse the session's most recent one) to **iteratively refine the same image** conversationally — do NOT start a new interaction or re-upload the image per tweak. Output is **JPEG**. |\n\n### Video & Music (preview — funded account)\n| Tool | Description |\n|------|-------------|\n| `gemini_video_generate(prompt, aspect_ratio?, resolution?, task?, images_url?, images_file_uris?, images?, images_base64?, from_clipboard?, previous_interaction_id?, continue_last?, model?, filename?, output_dir?, timeout_ms?, idempotency_key?, async?)` | Generate a short video (~10s) via the Gemini omni model: `text_to_video` (default), `image_to_video` / `reference_to_video` (supply reference image[s]), interpolation (pass first frame then last frame as `images`), or `edit` / `extend` (with `previous_interaction_id` / `continue_last`; extensions add ~3-10s each, to ~40s). `aspect_ratio` is `16:9` or `9:16`; `resolution` is `360p`/`720p` (default)/`1080p`/`4k` and is the cost lever — a 10s 360p clip runs about a third of 720p. Written to disk as **MP4**. Runs long — use `async: true` + `gemini_get_result`, or raise `timeout_ms`. |\n| `gemini_music_generate(prompt, model?, images_url?, images_file_uris?, images?, images_base64?, from_clipboard?, filename?, output_dir?, inline?, timeout_ms?, idempotency_key?, async?)` | Generate music from a prompt (mood/genre/instruments/structure/lyrics) via a Lyria model: `lyria-3-clip-preview` (30s instrumental, default, $0.04), `lyria-3.5` (full-length song with vocals, $0.08) or `lyria-3-pro-preview` ($0.08). Output is MP3. **Single-turn** — there is no follow-up refinement, so put the whole brief in the prompt. Written to disk, or returned inline. |\n\n### Async / idempotency (any generation tool)\n| Tool | Description |\n|------|-------------|\n| `gemini_get_result(job_id)` | Fetch a generation started with `async: true`. Returns `running` while in flight, then the normal result on completion. Lets a long video/music/image generation outlive a host's `tools/call` timeout. All generation tools also accept `idempotency_key` — a repeat call with the same key returns the recorded result (`reused: true`) instead of billing again. |\n\n## Workflows\n\n**Generate a single image:**\n```\ngemini_image_generate(prompt: \"a red maple leaf on white background, studio photo\")\n→ returns path to saved PNG\n```\n\n**Generate multiple variations:**\n```\ngemini_image_generate(prompt: \"a cartoon fox\", count: 4, output_dir: \"/tmp/foxes\")\n→ returns paths to 4 PNG files\n```\n\n**Edit an existing image:**\n```\ngemini_image_edit(prompt: \"make the background blue\", images: [\"/path/to/image.png\"])\n→ returns path to edited PNG\n```\n\n**Edit an image you only have a URL for (nothing downloads into the conversation):**\n```\ngemini_image_edit(prompt: \"make the background blue\", images_url: [\"https://example.com/photo.jpg\"])\n→ the server fetches the URL itself; returns path to edited PNG\n```\n\n**Reuse one photo across many generations (hosted hosted deployment, agent with a shell):**\n```\n$ curl -X POST https://<hosted deployment>/upload -H \"Authorization: Bearer $TOKEN\" \\\n    -H \"Content-Type: image/jpeg\" --data-binary @photo.jpg\n  → {\"file_uri\": \"files/abc123\"}\n\ngemini_image_edit(prompt: \"make it winter\",  images_file_uris: [\"files/abc123\"])\ngemini_image_edit(prompt: \"make it sunrise\", images_file_uris: [\"files/abc123\"])\n→ two edits, one upload, zero image bytes in context. Valid ~48h.\n```\n\n**Generate a consistent set (master + scenes):**\n```\ngemini_image_set(\n  master_prompt: \"a cartoon fox named Rusty, orange fur, blue scarf\",\n  scenes: [\"Rusty waving hello\", \"Rusty eating an apple\", \"Rusty sleeping\"]\n)\n→ returns paths to master + 3 scene images, all consistent\n```\n\n**Generate variations of a conce\n\nArchive v2.1.3: 3 files, 10867 bytes\n\nFiles: skill-card.md (2235b), SKILL.md (22697b), _meta.json (129b)\n\nArchive v2.1.2: 3 files, 10941 bytes\n\nFiles: skill-card.md (2406b), SKILL.md (22697b), _meta.json (129b)\n\nArchive v2.1.1: 3 files, 10825 bytes\n\nFiles: skill-card.md (2195b), SKILL.md (22697b), _meta.json (129b)\n\nArchive v2.1.0: 3 files, 10984 bytes\n\nFiles: skill-card.md (2526b), SKILL.md (22697b), _meta.json (129b)","readmeExcerpt":"Skill: gemini-mcp Owner: chrischall Summary: Generate and edit images, video, and music with Google Gemini models via MCP. Use when the user asks to generate, create, or edit images (Gemini / Nano Banana), produce a consistent set of images, compose/blend multiple images, generate a short video (text→video or image→video, via the omni model), or generate music/audio clips (via Lyria). Triggers on phrases like \"genera","codeSnippets":[],"executableExamples":[{"language":"json","snippet":"{\n  \"mcpServers\": {\n    \"gemini\": {\n      \"command\": \"npx\",\n      \"args\": [\"-y\", \"@chrischall/gemini-mcp\"],\n      \"env\": {\n        \"GEMINI_API_KEY\": \"your-api-key-here\"\n      }\n    }\n  }\n}"},{"language":"bash","snippet":"git clone https://github.com/chrischall/gemini-mcp\ncd gemini-mcp\nnpm install && npm run build"},{"language":"json","snippet":"{\n  \"mcpServers\": {\n    \"gemini\": {\n      \"command\": \"node\",\n      \"args\": [\"/path/to/gemini-mcp/dist/index.js\"],\n      \"env\": {\n        \"GEMINI_API_KEY\": \"your-api-key-here\"\n      }\n    }\n  }\n}"},{"language":"bash","snippet":"curl -X POST https://<hosted deployment>/upload \\\n  -H \"Authorization: Bearer $ACCESS_TOKEN\" \\\n  -H \"Content-Type: image/jpeg\" \\\n  --data-binary @photo.jpg"},{"language":"bash","snippet":"curl -X POST https://<hosted deployment>/upload \\\n  -H \"Authorization: Bearer $ACCESS_TOKEN\" \\\n  -H \"Content-Type: image/jpeg\" \\\n  --data-binary @photo.jpg\n# → {\"file_uri\":\"files/abc123\", ...}   then: images_file_uris: [\"files/abc123\"]"},{"language":"text","snippet":"gemini_image_generate(prompt: \"a red maple leaf on white background, studio photo\")\n→ returns path to saved PNG"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: gemini-mcp\ndescription: Generate and edit images, video, and music with Google Gemini models via MCP. Use when the user asks to generate, create, or edit images (Gemini / Nano Banana), produce a consistent set of images, compose/blend multiple images, generate a short video (text→video or image→video, via the omni model), or generate music/audio clips (via Lyria). Triggers on phrases like \"generate an image of\", \"edit this image with Gemini\", \"create a set of consistent images\", \"make a video of\", \"generate a video\", \"generate music\", \"make a song/audio clip\", \"use Nano Banana to make\", or any request to produce images, video, or music via the Gemini API. Requires the @chrischall/gemini-mcp package installed and the gemini server registered (see Setup below).\n---\n\n# gemini-mcp\n\nMCP server for Google Gemini media generation — natural-language **image**, **video**, and **music** creation via the Gemini API (Nano Banana / Nano Banana Pro images, omni video, Lyria music).\n\n- **npm:** [npmjs.com/package/@chrischall/gemini-mcp](https://www.npmjs.com/package/@chrischall/gemini-mcp)\n- **Source:** [github.com/chrischall/gemini-mcp](https://github.com/chrischall/gemini-mcp)\n\n## Setup\n\n### Option A — npx (recommended)\n\nAdd to `.mcp.json` in your project or `~/.claude/mcp.json`:\n\n```json\n{\n  \"mcpServers\": {\n    \"gemini\": {\n      \"command\": \"npx\",\n      \"args\": [\"-y\", \"@chrischall/gemini-mcp\"],\n      \"env\": {\n        \"GEMINI_API_KEY\": \"your-api-key-here\"\n      }\n    }\n  }\n}\n```\n\n### Option B — from source\n\n```bash\ngit clone https://github.com/chrischall/gemini-mcp\ncd gemini-mcp\nnpm install && npm run build\n```\n\nThen add to `.mcp.json`:\n\n```json\n{\n  \"mcpServers\": {\n    \"gemini\": {\n      \"command\": \"node\",\n      \"args\": [\"/path/to/gemini-mcp/dist/index.js\"],\n      \"env\": {\n        \"GEMINI_API_KEY\": \"your-api-key-here\"\n      }\n    }\n  }\n}\n```\n\nOr use a `.env` file in the project directory with `GEMINI_API_KEY=<value>`.\n\n### Getting your API key\n\n1. Go to [aistudio.google.com/apikey](https://aistudio.google.com/apikey)\n2. Create an API key (requires a Google account)\n3. Copy the key and set it as `GEMINI_API_KEY`\n\nNote: Image generation requires a billing-enabled Google Cloud project.\n\n## Environment Variables\n\n| Variable | Required | Description |\n|---|---|---|\n| `GEMINI_API_KEY` | Yes | Your Google Gemini API key |\n| `GEMINI_IMAGE_MODEL` | No | Override the default image model (default: `gemini-3.1-flash-image`) |\n| `GEMINI_OUTPUT_DIR` | No | Default directory for saved images (default: current working directory) |\n| `GEMINI_INPUT_DIR` | No | Directory to resolve bare input-image filenames against (e.g. point at Cowork's `uploads/` folder so `images: [\"house.jpg\"]` works) |\n\n## Tools\n\n### Models\n| Tool | Description |\n|------|-------------|\n| `gemini_list_models` | List available Gemini image models and the current default |\n\n**Which model to pick** (per-call `model`, or `GEMINI_IMAGE_MODEL`):\n\n| Model | When to use | Reference-image caps (of 14 max) "},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn700jq4sjtf2anb0rk3ft4p7n856872\",\n  \"slug\": \"gemini-mcp\",\n  \"version\": \"2.3.3\",\n  \"publishedAt\": 1791380232402\n}"},{"path":"skill-card.md","content":"## Description:\n\nHelps agents generate and edit images, create short videos, and produce music with Google Gemini models through an MCP server.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[chrischall](https://clawhub.ai/user/chrischall)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal users and creative teams use this skill to direct an agent to generate or refine images, videos, and music from prompts and supplied media.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Prompts, media, clipboard images, URLs, and base64 data may be sent to Gemini or a hosted service.\n\nMitigation: Use only content you are authorized to share; avoid sensitive or regulated data without consent and a review of service retention.\n\nRisk: Use requires a Gemini API key and media generation may incur charges.\n\nMitigation: Install only from a trusted package source, safeguard the API key, and review billing before generation.\n\n## Reference(s):\n\n- [gemini-mcp ClawHub release](https://clawhub.ai/chrischall/skills/gemini-mcp)\n- [gemini-mcp npm package](https://www.npmjs.com/package/@chrischall/gemini-mcp)\n- [Google AI Studio API key](https://aistudio.google.com/apikey)\n\n## Skill Output:\n\n**Output Type(s):** [Files, Text]\n\n**Output Format:** [PNG or JPEG images, MP4 video, MP3 audio, and text responses with file paths and metadata]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Generated files may be saved locally or returned inline; responses can include job IDs for asynchronous generation.]\n\n## Skill Version(s):\n\n2.3.3 (source: server-resolved ClawHub release)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":1233,"uniquenessScore":41,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T10:26:00.303Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T10:26:00.303Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T00:15:04.391Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}