{"id":"c8d7adad-76f5-49b4-b76e-fb264d0e05b0","entityType":"agent","slug":"clawhub-vedantsingh60-prompt-performance-tester","name":"Prompt Performance Tester - UnisAI","canonicalUrl":"https://www.xpersona.co/agent/clawhub-vedantsingh60-prompt-performance-tester","canonicalPath":"/agent/clawhub-vedantsingh60-prompt-performance-tester","generatedAt":"2026-10-10T02:22:18.291Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"description":"Test prompts across Claude, GPT, and Gemini models and get detailed latency, cost, quality, consistency, and error metrics with smart recommendations.","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.9K downloads reported by the source. Last updated 4/15/2026.","installCommand":"clawhub skill install kn77yjs5esft2kgsd6dpz9c92n80dgsy:prompt-performance-tester","sourceUrl":"https://clawhub.ai/vedantsingh60/prompt-performance-tester","homepage":"https://clawhub.ai/vedantsingh60/prompt-performance-tester","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/vedantsingh60/prompt-performance-tester","kind":"source"}],"safetyScore":84,"overallRank":62,"popularityScore":66,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Prompt Performance Tester - UnisAI technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"stars":null,"forks":null,"downloads":1927,"packageName":null,"latestVersion":"1.1.9","tractionLabel":"1.9K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-02-28T19:43:01.667Z","emptyReason":null},"lastUpdatedAt":"2026-04-15T00:45:39.800Z","lastCrawledAt":"2026-02-28T19:43:01.667Z","lastIndexedAt":null,"nextCrawlAt":"2026-03-01T19:43:01.667Z","lastVerifiedAt":null,"highlights":[{"version":"1.1.9","createdAt":"2026-02-27T17:27:39.522Z","changelog":"- Updated provider/model lists to include more example models and the latest pricing. - Expanded cost comparison examples to include DeepSeek Chat. - Added clarification that unlisted models are supported, but cost is shown as $0.00 with a warning. - Improved explanation of quality, cost, and performance metrics for broader clarity. - Enhanced recommendation and real-world example sections to better showcase DeepSeek and new models.","fileCount":5,"zipByteSize":17508},{"version":"1.1.8","createdAt":"2026-02-27T17:27:27.786Z","changelog":"- Updated provider/model lists to include more example models and the latest pricing. - Expanded cost comparison examples to include DeepSeek Chat. - Added clarification that unlisted models are supported, but cost is shown as $0.00 with a warning. - Improved explanation of quality, cost, and performance metrics for broader clarity. - Enhanced recommendation and real-world example sections to better showcase DeepSeek and new models.","fileCount":5,"zipByteSize":17509},{"version":"1.1.7","createdAt":"2026-02-27T16:50:17.543Z","changelog":"**Expanded to support model-agnostic benchmarking across 9 major LLM providers.** - Now supports any model ID with automatic provider detection based on name prefix—no hardcoded model list. - Added support for Anthropic, OpenAI, Google, Mistral, DeepSeek, xAI (Grok), MiniMax, Qwen, and Meta Llama via OpenRouter. - Documentation updated with supported model prefixes, API keys, and latest known per-token pricing for each provider. - All previous features (latency, cost, quality, consistency, recommendations) remain and apply to all supported models.","fileCount":null,"zipByteSize":null},{"version":"1.1.6","createdAt":"2026-02-27T16:49:57.000Z","changelog":"**Expanded to support model-agnostic benchmarking across 9 major LLM providers.** - Now supports any model ID with automatic provider detection based on name prefix—no hardcoded model list. - Added support for Anthropic, OpenAI, Google, Mistral, DeepSeek, xAI (Grok), MiniMax, Qwen, and Meta Llama via OpenRouter. - Documentation updated with supported model prefixes, API keys, and latest known per-token pricing for each provider. - All previous features (latency, cost, quality, consistency, recommendations) remain and apply to all supported models. - No code changes in this release; update is documentation-only.","fileCount":null,"zipByteSize":null},{"version":"1.1.5","createdAt":"2026-02-16T20:07:53.120Z","changelog":"- Documentation reformatted with minor cleanups for clarity and consistency. - No functional or code changes in this release. - All prompts, features, and instructions remain unchanged.","fileCount":null,"zipByteSize":null},{"version":"1.1.4","createdAt":"2026-02-02T03:45:04.686Z","changelog":"**1.1.4 is a documentation cleanup and simplification release.** - SKILL.md significantly condensed for clarity and brevity. - Streamlined value proposition and example cost calculations. - Updated real-world model comparison table to use more recent models and prices. - Removed marketing language and focus on data-driven cost/quality analysis. - Reorganized use case and getting started sections for easier navigation. - No code or logic changes; documentation only.","fileCount":null,"zipByteSize":null},{"version":"1.1.3","createdAt":"2026-02-02T03:15:28.352Z","changelog":"- Expanded support to 10 leading AI models, including latest Claude 4.5, GPT-5.2, and Gemini 3 series. - Updated model selection and pricing details to reflect 2026 releases and current rates. - Quick Start and plan descriptions now accommodate 10 models per test. - Enhanced marketing copy with up-to-date benchmarks and role-based benefits. - Real-world example and recommendations remain for user clarity.","fileCount":null,"zipByteSize":null},{"version":"1.1.2","createdAt":"2026-02-02T03:03:35.305Z","changelog":"Version 1.1.2 - No code changes detected; documentation updated only. - SKILL.md rewritten for clarity and to emphasize product benefits. - Feature descriptions, real-world examples, and pricing explanations improved. - Enhanced summary of supported models and use cases. - Quick Start, API setup, and output formatting instructions are clearer and more actionable.","fileCount":null,"zipByteSize":null}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install kn77yjs5esft2kgsd6dpz9c92n80dgsy:prompt-performance-tester","setupComplexity":"low","setupSteps":["Install using `clawhub skill install kn77yjs5esft2kgsd6dpz9c92n80dgsy:prompt-performance-tester` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/vedantsingh60/prompt-performance-tester before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-vedantsingh60-prompt-performance-tester/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-vedantsingh60-prompt-performance-tester/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-vedantsingh60-prompt-performance-tester/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-vedantsingh60-prompt-performance-tester/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-vedantsingh60-prompt-performance-tester/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-vedantsingh60-prompt-performance-tester/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T02:22:18.290Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-vedantsingh60-prompt-performance-tester/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-vedantsingh60-prompt-performance-tester/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-vedantsingh60-prompt-performance-tester/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-vedantsingh60-prompt-performance-tester/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":null},"readme":"Skill: Prompt Performance Tester - UnisAI\n\nOwner: vedantsingh60\n\nSummary: Test prompts across Claude, GPT, and Gemini models and get detailed latency, cost, quality, consistency, and error metrics with smart recommendations.\n\nTags: Latest:1.1.4, ai-testing:1.0.1, ai-testing multi-provider prompt-optimization cost-analysis llm-benchmarking claude gpt gemini performance-testing api-comparison multi-model:1.1.2, claude-api:1.0.1, cost-analysis:1.0.1, latest:1.1.9, llm-benchmarking:1.0.1, openai-api:1.0.1, prompt-optimization:1.0.1\n\nVersion history:\n\nv1.1.9 | 2026-02-27T17:27:39.522Z | user\n\n- Updated provider/model lists to include more example models and the latest pricing.\n- Expanded cost comparison examples to include DeepSeek Chat.\n- Added clarification that unlisted models are supported, but cost is shown as $0.00 with a warning.\n- Improved explanation of quality, cost, and performance metrics for broader clarity.\n- Enhanced recommendation and real-world example sections to better showcase DeepSeek and new models.\n\nv1.1.8 | 2026-02-27T17:27:27.786Z | user\n\n- Updated provider/model lists to include more example models and the latest pricing.\n- Expanded cost comparison examples to include DeepSeek Chat.\n- Added clarification that unlisted models are supported, but cost is shown as $0.00 with a warning.\n- Improved explanation of quality, cost, and performance metrics for broader clarity.\n- Enhanced recommendation and real-world example sections to better showcase DeepSeek and new models.\n\nv1.1.7 | 2026-02-27T16:50:17.543Z | user\n\n**Expanded to support model-agnostic benchmarking across 9 major LLM providers.**\n\n- Now supports any model ID with automatic provider detection based on name prefix—no hardcoded model list.\n- Added support for Anthropic, OpenAI, Google, Mistral, DeepSeek, xAI (Grok), MiniMax, Qwen, and Meta Llama via OpenRouter.\n- Documentation updated with supported model prefixes, API keys, and latest known per-token pricing for each provider.\n- All previous features (latency, cost, quality, consistency, recommendations) remain and apply to all supported models.\n\nv1.1.6 | 2026-02-27T16:49:57.000Z | user\n\n**Expanded to support model-agnostic benchmarking across 9 major LLM providers.**\n\n- Now supports any model ID with automatic provider detection based on name prefix—no hardcoded model list.\n- Added support for Anthropic, OpenAI, Google, Mistral, DeepSeek, xAI (Grok), MiniMax, Qwen, and Meta Llama via OpenRouter.\n- Documentation updated with supported model prefixes, API keys, and latest known per-token pricing for each provider.\n- All previous features (latency, cost, quality, consistency, recommendations) remain and apply to all supported models.\n- No code changes in this release; update is documentation-only.\n\nv1.1.5 | 2026-02-16T20:07:53.120Z | user\n\n- Documentation reformatted with minor cleanups for clarity and consistency.\n- No functional or code changes in this release.\n- All prompts, features, and instructions remain unchanged.\n\nv1.1.4 | 2026-02-02T03:45:04.686Z | user\n\n**1.1.4 is a documentation cleanup and simplification release.**\n\n- SKILL.md significantly condensed for clarity and brevity.\n- Streamlined value proposition and example cost calculations.\n- Updated real-world model comparison table to use more recent models and prices.\n- Removed marketing language and focus on data-driven cost/quality analysis.\n- Reorganized use case and getting started sections for easier navigation.\n- No code or logic changes; documentation only.\n\nv1.1.3 | 2026-02-02T03:15:28.352Z | user\n\n- Expanded support to 10 leading AI models, including latest Claude 4.5, GPT-5.2, and Gemini 3 series.\n- Updated model selection and pricing details to reflect 2026 releases and current rates.\n- Quick Start and plan descriptions now accommodate 10 models per test.\n- Enhanced marketing copy with up-to-date benchmarks and role-based benefits.\n- Real-world example and recommendations remain for user clarity.\n\nv1.1.2 | 2026-02-02T03:03:35.305Z | user\n\nVersion 1.1.2\n\n- No code changes detected; documentation updated only.\n- SKILL.md rewritten for clarity and to emphasize product benefits.\n- Feature descriptions, real-world examples, and pricing explanations improved.\n- Enhanced summary of supported models and use cases.\n- Quick Start, API setup, and output formatting instructions are clearer and more actionable.\n\nv1.1.1 | 2026-02-02T02:55:31.272Z | user\n\n- Major upgrade: Now supports prompt testing across Claude, OpenAI GPT, and Google Gemini, not just Claude.\n- Added cross-provider cost and quality comparison for 9 different LLM models.\n- New reporting shows latency, cost, and quality side-by-side for Anthropic, OpenAI, and Google models.\n- Recommendations now include fastest, cheapest, and best quality models across all providers.\n- Expanded use cases and provider-specific examples in documentation.\n\nv1.1.0 | 2026-02-02T02:52:52.200Z | user\n\nVersion \"1.1.0\":\n    - \"✨ Multi-provider support: Claude, GPT, and Gemini\"\n    - \"✨ 9 LLM models supported across 3 providers\"\n    - \"✨ Cross-provider cost comparison engine\"\n    - \"✨ Provider-specific API optimizations\"\n    - \"✨ Enhanced recommendations with multi-provider insights\"\n    - \"✨ Rebranded from Prompt Migrator to UniAI\"\n    - \"🏷️ Updated tags for better discoverability (14 tags)\"\n    - \"📊 Improved cost calculation accuracy\"\n    - \"🔧 Added OpenAI and Google API integrations\"\n    - \"📝 Updated documentation with multi-provider examples\"\n\nv1.0.1 | 2026-02-02T02:33:38.204Z | user\n\n## Version 1.0.1\n\n- Removed the file `IP_PROTECTION_GUIDE.md`.\n- Updated documentation links and support contact information to use unisai.vercel.app addresses.\n\nv1.0.0 | 2026-02-02T02:22:51.962Z | user\n\nInitial release - Multi-model prompt testing across OpenAI, Claude Haiku, Sonnet, and Opus with latency, cost, and quality metrics\n\nArchive index:\n\nArchive v1.1.9: 5 files, 17508 bytes\n\nFiles: LICENSE.md (4799b), manifest.yaml (7490b), prompt_performance_tester.py (22862b), SKILL.md (17974b), _meta.json (144b)\n\nFile v1.1.9:SKILL.md\n\n# Prompt Performance Tester\n\n**Model-agnostic prompt benchmarking across 9 providers.**\n\nPass any model ID — provider auto-detected. Compare latency, cost, quality, and consistency across Claude, GPT, Gemini, DeepSeek, Grok, MiniMax, Qwen, Llama, and Mistral.\n\n---\n\n## 🚀 Why This Skill?\n\n### Problem Statement\nComparing LLM models across providers requires manual testing:\n- No systematic way to measure performance across models\n- Cost differences are significant but not easily comparable\n- Quality varies by use case and provider\n- Manual API testing is time-consuming and error-prone\n\n### The Solution\nTest prompts across any model from any supported provider simultaneously. Get performance metrics and recommendations based on latency, cost, and quality.\n\n### Example Cost Comparison\nFor 10,000 requests/day with average 28 input + 115 output tokens:\n- Claude Opus 4.6: ~$30.15/day ($903/month)\n- Gemini 2.5 Flash-Lite: ~$0.05/day ($1.50/month)\n- DeepSeek Chat: ~$0.14/day ($4.20/month)\n- Monthly cost difference (Opus vs Flash-Lite): $901.50\n\n---\n\n## ✨ What You Get\n\n### Model-Agnostic Multi-Provider Testing\nPass any model ID — provider is auto-detected from the model name prefix.\nNo hardcoded list; new models work without code changes.\n\n| Provider | Example Models | Prefix | Required Key |\n|----------|---------------|--------|--------------|\n| **Anthropic** | claude-opus-4-6, claude-sonnet-4-6, claude-haiku-4-5-20251001 | `claude-` | ANTHROPIC_API_KEY |\n| **OpenAI** | gpt-5.2-pro, gpt-5.2, gpt-5.1 | `gpt-`, `o1`, `o3` | OPENAI_API_KEY |\n| **Google** | gemini-2.5-pro, gemini-2.5-flash, gemini-2.5-flash-lite | `gemini-` | GOOGLE_API_KEY |\n| **Mistral** | mistral-large-latest, mistral-small-latest | `mistral-`, `mixtral-` | MISTRAL_API_KEY |\n| **DeepSeek** | deepseek-chat, deepseek-reasoner | `deepseek-` | DEEPSEEK_API_KEY |\n| **xAI** | grok-4-1-fast, grok-3-beta | `grok-` | XAI_API_KEY |\n| **MiniMax** | MiniMax-M2.1 | `MiniMax`, `minimax` | MINIMAX_API_KEY |\n| **Qwen** | qwen3.5-plus, qwen3-max-instruct | `qwen` | DASHSCOPE_API_KEY |\n| **Meta Llama** | meta-llama/llama-4-maverick, meta-llama/llama-3.3-70b-instruct | `meta-llama/`, `llama-` | OPENROUTER_API_KEY |\n\n### Known Pricing (per 1M tokens)\n\n| Model | Input | Output |\n|-------|-------|--------|\n| claude-opus-4-6 | $15.00 | $75.00 |\n| claude-sonnet-4-6 | $3.00 | $15.00 |\n| claude-haiku-4-5-20251001 | $1.00 | $5.00 |\n| gpt-5.2-pro | $21.00 | $168.00 |\n| gpt-5.2 | $1.75 | $14.00 |\n| gpt-5.1 | $2.00 | $8.00 |\n| gemini-2.5-pro | $1.25 | $10.00 |\n| gemini-2.5-flash | $0.30 | $2.50 |\n| gemini-2.5-flash-lite | $0.10 | $0.40 |\n| mistral-large-latest | $2.00 | $6.00 |\n| mistral-small-latest | $0.10 | $0.30 |\n| deepseek-chat | $0.27 | $1.10 |\n| deepseek-reasoner | $0.55 | $2.19 |\n| grok-4-1-fast | $5.00 | $25.00 |\n| grok-3-beta | $3.00 | $15.00 |\n| MiniMax-M2.1 | $0.40 | $1.60 |\n| qwen3.5-plus | $0.57 | $2.29 |\n| qwen3-max-instruct | $1.60 | $6.40 |\n| meta-llama/llama-4-maverick | $0.20 | $0.60 |\n| meta-llama/llama-3.3-70b-instruct | $0.59 | $0.79 |\n\n> **Note:** Unlisted models still work — cost calculation returns $0.00 with a warning. Pricing table is for reference only, not a validation gate.\n\n### Performance Metrics\n\nEvery test measures:\n- ⚡ **Latency** — Response time in milliseconds\n- 💰 **Cost** — Exact API cost per request (input + output tokens)\n- 🎯 **Quality** — Response quality score (0–100)\n- 📊 **Token Usage** — Input and output token counts\n- 🔄 **Consistency** — Variance across multiple test runs\n- ❌ **Error Tracking** — API failures, timeouts, rate limits\n\n### Smart Recommendations\n\nGet instant answers to:\n- Which model is **fastest** for your prompt?\n- Which is most **cost-effective**?\n- Which produces **best quality** responses?\n- How much can you **save** by switching providers?\n\n---\n\n## 📊 Real-World Example\n\n```\nPROMPT: \"Write a professional customer service response about a delayed shipment\"\n\n┌─────────────────────────────────────────────────────────────────┐\n│ GEMINI 2.5 FLASH-LITE (Google) 💰 MOST AFFORDABLE              │\n├─────────────────────────────────────────────────────────────────┤\n│ Latency:  523ms                                                 │\n│ Cost:     $0.000025                                             │\n│ Quality:  65/100                                                │\n│ Tokens:   28 in / 87 out                                        │\n└─────────────────────────────────────────────────────────────────┘\n\n┌─────────────────────────────────────────────────────────────────┐\n│ DEEPSEEK CHAT (DeepSeek) 💡 BUDGET PICK                        │\n├─────────────────────────────────────────────────────────────────┤\n│ Latency:  710ms                                                 │\n│ Cost:     $0.000048                                             │\n│ Quality:  70/100                                                │\n│ Tokens:   28 in / 92 out                                        │\n└─────────────────────────────────────────────────────────────────┘\n\n┌─────────────────────────────────────────────────────────────────┐\n│ CLAUDE HAIKU 4.5 (Anthropic) 🚀 BALANCED PERFORMER             │\n├─────────────────────────────────────────────────────────────────┤\n│ Latency:  891ms                                                 │\n│ Cost:     $0.000145                                             │\n│ Quality:  78/100                                                │\n│ Tokens:   28 in / 102 out                                       │\n└─────────────────────────────────────────────────────────────────┘\n\n┌─────────────────────────────────────────────────────────────────┐\n│ GPT-5.2 (OpenAI) 💡 EXCELLENT QUALITY                          │\n├─────────────────────────────────────────────────────────────────┤\n│ Latency:  645ms                                                 │\n│ Cost:     $0.000402                                             │\n│ Quality:  88/100                                                │\n│ Tokens:   28 in / 98 out                                        │\n└─────────────────────────────────────────────────────────────────┘\n\n┌─────────────────────────────────────────────────────────────────┐\n│ CLAUDE OPUS 4.6 (Anthropic) 🏆 HIGHEST QUALITY                 │\n├─────────────────────────────────────────────────────────────────┤\n│ Latency:  1,234ms                                               │\n│ Cost:     $0.001875                                             │\n│ Quality:  94/100                                                │\n│ Tokens:   28 in / 125 out                                       │\n└─────────────────────────────────────────────────────────────────┘\n\n🎯 RECOMMENDATIONS:\n1. Most cost-effective: Gemini 2.5 Flash-Lite ($0.000025/request) — 99.98% cheaper than Opus\n2. Budget pick: DeepSeek Chat ($0.000048/request) — strong quality at low cost\n3. Best quality: Claude Opus 4.6 (94/100) — state-of-the-art reasoning & analysis\n4. Smart pick: Claude Haiku 4.5 ($0.000145/request) — 81% cheaper, 83% quality match\n5. Speed + Quality: GPT-5.2 ($0.000402/request) — excellent quality at mid-range cost\n\n💡 Potential monthly savings (10,000 requests/day, 28 input + 115 output tokens avg):\n   - Using Gemini 2.5 Flash-Lite vs Opus: $903/month saved ($1.44 vs $904.50)\n   - Using DeepSeek Chat vs Opus: $899/month saved ($4.50 vs $904.50)\n   - Using Claude Haiku vs Opus: $731/month saved ($173.40 vs $904.50)\n```\n\n---\n\n## Use Cases\n\n### Production Deployment\n- Evaluate models before production selection\n- Compare cost vs quality tradeoffs\n- Benchmark API latency across providers\n\n### Prompt Development\n- Test prompt variations across models\n- Measure quality scores consistently\n- Compare performance metrics\n\n### Cost Analysis\n- Analyze LLM API spending by model\n- Compare provider pricing structures\n- Identify cost-efficient alternatives\n\n### Performance Testing\n- Measure latency and response times\n- Test consistency across multiple runs\n- Evaluate quality scores\n\n---\n\n## 🚀 Quick Start\n\n### 1. Subscribe to Skill\nClick \"Subscribe\" on ClawhHub to get access.\n\n### 2. Set API Keys\nAdd keys for the providers you want to test:\n\n```bash\n# Anthropic (Claude models)\nexport ANTHROPIC_API_KEY=\"sk-ant-...\"\n\n# OpenAI (GPT models)\nexport OPENAI_API_KEY=\"sk-...\"\n\n# Google (Gemini models)\nexport GOOGLE_API_KEY=\"AI...\"\n\n# DeepSeek\nexport DEEPSEEK_API_KEY=\"...\"\n\n# xAI (Grok models)\nexport XAI_API_KEY=\"...\"\n\n# MiniMax\nexport MINIMAX_API_KEY=\"...\"\n\n# Alibaba (Qwen models)\nexport DASHSCOPE_API_KEY=\"...\"\n\n# OpenRouter (Meta Llama models)\nexport OPENROUTER_API_KEY=\"...\"\n\n# Mistral\nexport MISTRAL_API_KEY=\"...\"\n```\n\nYou only need keys for the providers you plan to test.\n\n### 3. Install Dependencies\n\n```bash\n# Install only what you need\npip install anthropic          # Claude\npip install openai             # GPT, DeepSeek, xAI, MiniMax, Qwen, Llama\npip install google-generativeai  # Gemini\npip install mistralai          # Mistral\n\n# Or install everything\npip install anthropic openai google-generativeai mistralai\n```\n\n### 4. Run Your First Test\n\n**Option A: Python**\n```python\nimport os\nfrom prompt_performance_tester import PromptPerformanceTester\n\ntester = PromptPerformanceTester()  # reads API keys from environment\n\nresults = tester.test_prompt(\n    prompt_text=\"Write a professional email apologizing for a delayed shipment\",\n    models=[\n        \"claude-haiku-4-5-20251001\",\n        \"gpt-5.2\",\n        \"gemini-2.5-flash\",\n        \"deepseek-chat\",\n    ],\n    num_runs=3,\n    max_tokens=500\n)\n\nprint(tester.format_results(results))\nprint(f\"🏆 Best quality:  {results.best_model}\")\nprint(f\"💰 Cheapest:      {results.cheapest_model}\")\nprint(f\"⚡ Fastest:       {results.fastest_model}\")\n```\n\n**Option B: CLI**\n```bash\n# Test across multiple models\nprompt-tester test \"Your prompt here\" \\\n  --models claude-haiku-4-5-20251001 gpt-5.2 gemini-2.5-flash deepseek-chat \\\n  --runs 3\n\n# Export results\nprompt-tester test \"Your prompt here\" --export results.json\n```\n\n---\n\n## 🔒 Security & Privacy\n\n### API Key Safety\n- Keys stored in environment variables only — never hardcoded or logged\n- Never transmitted to UnisAI servers\n- HTTPS encryption for all provider API calls\n\n### Data Privacy\n- Your prompts are sent only to the AI providers you select for testing\n- Each provider has their own data retention policy (see their privacy pages)\n- No data stored on UnisAI infrastructure\n\n---\n\n## 📚 Technical Details\n\n### System Requirements\n- **Python**: 3.9+\n- **Dependencies**: `anthropic`, `openai`, `google-generativeai`, `mistralai` (install only what you need)\n- **Platform**: macOS, Linux, Windows\n\n### Architecture\n- **Lazy client initialization** — SDK clients only loaded for providers actually tested\n- **Prefix-based routing** — `PROVIDER_MAP` detects provider from model name; no hardcoded whitelist\n- **OpenAI-compat path** — DeepSeek, xAI, MiniMax, Qwen, and OpenRouter all use the `openai` SDK with a custom `base_url`\n- **Pricing table** — used for cost calculation only; unknown models get `cost=0` with a warning\n\n### Metrics Collected\nEvery test captures:\n- **Latency**: Total response time (ms)\n- **Cost**: Input + output cost based on known pricing (USD)\n- **Quality**: Heuristic response score based on length, completeness (0–100)\n- **Tokens**: Exact input/output token counts per provider\n- **Consistency**: Standard deviation across multiple runs\n- **Errors**: Timeouts, rate limits, API failures\n\n---\n\n## ❓ Frequently Asked Questions\n\n**Q: Do I need API keys for all 9 providers?**\nA: No. You only need keys for the providers you want to test. If you only test Claude models, you only need `ANTHROPIC_API_KEY`.\n\n**Q: Who pays for the API costs?**\nA: You do. You provide your own API keys and pay each provider directly. This skill has no per-request fees.\n\n**Q: How accurate are the cost calculations?**\nA: Costs are calculated from the known pricing table using actual token counts. Models not in the pricing table return `$0.00` — the model still runs, the cost just won't be shown.\n\n**Q: Can I test models not in the pricing table?**\nA: Yes. Any model whose name starts with a supported prefix will run. Cost will show as $0.00 for unlisted models.\n\n**Q: Can I test prompts in non-English languages?**\nA: Yes. All supported providers handle multiple languages.\n\n**Q: Can I use this in production/CI/CD?**\nA: Yes. Import `PromptPerformanceTester` directly from Python or call via CLI.\n\n**Q: What if my prompt is very long?**\nA: Set `max_tokens` appropriately. The skill passes your prompt as-is to each provider's API.\n\n---\n\n## 🗺️ Roadmap\n\n### ✅ Current Release (v1.1.8)\n- Model-agnostic architecture — any model ID works via prefix detection\n- 9 providers, 20 known models with pricing\n- DeepSeek, xAI Grok, MiniMax, Qwen, Meta Llama as first-class providers\n- Claude 4.6 series (opus-4-6, sonnet-4-6)\n- Lazy client initialization — only loads SDKs for providers actually used\n- Fixed UnisAI branding throughout\n\n### 🚧 Coming Soon (v1.3)\n- **Batch testing**: Test 100+ prompts simultaneously\n- **Historical tracking**: Track model performance over time\n- **Webhook integrations**: Slack, Discord, email notifications\n\n### 🔮 Future (v1.3+)\n- **A/B testing framework**: Scientific prompt experimentation\n- **Fine-tuning insights**: Which models to fine-tune for your use case\n- **Custom benchmarks**: Create your own evaluation criteria\n- **Auto-optimization**: AI-powered prompt improvement suggestions\n\n---\n\n## 📞 Support\n\n- **Email**: support@unisai.vercel.app\n- **Website**: https://unisai.vercel.app\n- **Bug Reports**: support@unisai.vercel.app\n\n---\n\n## 📄 License & Terms\n\nThis skill is distributed via ClawhHub under the following terms.\n\n### ✅ You CAN:\n- Use for your own business and projects\n- Test prompts for internal applications\n- Modify source code for personal use\n\n### ❌ You CANNOT:\n- Redistribute outside ClawhHub registry\n- Resell or sublicense\n- Use UnisAI trademark without permission\n\n**Full Terms**: See [LICENSE.md](LICENSE.md)\n\n---\n\n## 📝 Changelog\n\n### [1.1.8] - 2026-02-27\n\n#### Fixes & Polish\n- Bumped version to 1.1.8\n- SKILL.md fully rewritten — cleaned up formatting, removed stale content\n- Removed old IP watermark reference (`PROPRIETARY_SKILL_VEDANT_2024`) from docs\n- Corrected watermark to `PROPRIETARY_SKILL_UNISAI_2026_MULTI_PROVIDER` throughout\n- Fixed all UnisAI branding (was UniAI in v1.1.0 changelog)\n- Updated pricing table to include all 20 known models\n- Cleaned up FAQ, Quick Start, and Use Cases sections\n\n### [1.1.6] - 2026-02-27\n\n#### 🏗️ Model-Agnostic Architecture\n- Provider auto-detected from model name prefix — no hardcoded whitelist\n- Any new model works automatically without code changes\n- Added DeepSeek, xAI Grok, MiniMax, Qwen, Meta Llama as first-class providers (9 total)\n- Updated Claude to 4.6 series (claude-opus-4-6, claude-sonnet-4-6)\n- Lazy client initialization — only loads SDKs for providers actually tested\n- Unified OpenAI-compat path for DeepSeek, xAI, MiniMax, Qwen, OpenRouter\n\n### [1.1.5] - 2026-02-01\n\n#### 🚀 Latest Models Update\n- GPT-5.2 Series — Added Instant, Thinking, and Pro variants\n- Gemini 2.5 Series — Updated to 2.5 Pro, Flash, and Flash-Lite\n- Claude 4.5 pricing updates\n- 10 total models across 3 providers\n\n### [1.1.0] - 2026-01-15\n\n#### ✨ Major Features\n- Multi-provider support — Claude, GPT, Gemini\n- Cross-provider cost comparison\n- Enhanced recommendations engine\n- Rebranded to UnisAI\n\n### [1.0.0] - 2024-02-02\n\n#### Initial Release\n- Claude-only prompt testing (Haiku, Sonnet, Opus)\n- Performance metrics: latency, cost, quality, consistency\n- Basic recommendations engine\n\n---\n\n**Last Updated**: February 27, 2026\n**Current Version**: 1.1.8\n**Status**: Active & Maintained\n\n© 2026 UnisAI. All rights reserved.\n\nFile v1.1.9:_meta.json\n\n{\n  \"ownerId\": \"kn77yjs5esft2kgsd6dpz9c92n80dgsy\",\n  \"slug\": \"prompt-performance-tester\",\n  \"version\": \"1.1.9\",\n  \"publishedAt\": 1772213259522\n}\n\nFile v1.1.9:LICENSE.md\n\n# UniAI Skills - Proprietary License\n\n**Version 1.0 | Effective Date: February 2, 2024**\n\n## 1. GRANT OF LICENSE\n\nUniAI (\"Licensor\") grants you (\"Licensee\") a limited, non-exclusive, non-transferable, revocable license to use the ClawhHub Skills (\"Software\") solely in accordance with the terms of this license agreement.\n\n## 2. LICENSE RESTRICTIONS\n\nYou may NOT:\n- Reverse engineer, decompile, or disassemble the Software\n- Modify, alter, or create derivative works of the Software\n- Remove, obscure, or alter any proprietary notices or labels on the Software\n- Share, distribute, or sublicense the Software to any third party\n- Use the Software for commercial purposes without a commercial license\n- Access or use the Software beyond the scope of your subscription tier\n- Attempt to circumvent licensing controls or API rate limits\n\n## 3. INTELLECTUAL PROPERTY RIGHTS\n\nAll intellectual property rights in and to the Software are retained by Licensor. This includes:\n- Source code and object code\n- Algorithms and methodologies\n- Performance optimization techniques\n- Quality scoring mechanisms\n- Proprietary data structures\n- Trade secrets and confidential information\n\n## 4. PERMITTED USES\n\nYou may only:\n- Use the Software as provided through the ClawhHub platform\n- Access features available in your subscription tier\n- Create test results and reports for internal use\n- Share results with your team (if on a team plan)\n- Provide feedback to improve the Software\n\n## 5. SUBSCRIPTION TIERS\n\n### Starter (Free)\n- 5 tests per month\n- 2 models per test\n- Basic features\n- Personal use only\n\n### Professional ($29/month)\n- Unlimited tests\n- All models supported\n- Advanced analytics\n- API access\n- Commercial use permitted\n\n### Enterprise ($99/month)\n- Team collaboration\n- White-label option\n- Custom integrations\n- Dedicated support\n- SLA guarantees\n\n## 6. API KEY AND CREDENTIALS\n\n- You are responsible for keeping your API keys confidential\n- Do not share your license key with others\n- One license per person/organization\n- License keys are non-transferable\n- Unauthorized sharing may result in account termination\n\n## 7. DATA PRIVACY\n\n- We do not retain your test data by default\n- Free tier: 30-day retention\n- Paid tiers: 90-day retention\n- You can request data deletion anytime\n- See Privacy Policy for full details\n\n## 8. WARRANTY DISCLAIMER\n\nTHE SOFTWARE IS PROVIDED \"AS-IS\" WITHOUT ANY WARRANTIES. LICENSOR DISCLAIMS ALL WARRANTIES, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO:\n- Merchantability\n- Fitness for a particular purpose\n- Non-infringement\n- Accuracy of results\n\n## 9. LIMITATION OF LIABILITY\n\nIN NO EVENT SHALL LICENSOR BE LIABLE FOR:\n- Any indirect, incidental, special, or consequential damages\n- Loss of data, revenue, or profits\n- Business interruption\n- Even if advised of the possibility of such damages\n\n## 10. TERMINATION\n\nLicensor may terminate your license if you:\n- Violate any terms of this agreement\n- Fail to pay subscription fees\n- Attempt to reverse engineer the Software\n- Share your license key with others\n- Use the Software unlawfully\n\nUpon termination:\n- Your access to the Software is immediately revoked\n- You must destroy all copies of the Software in your possession\n- Any refunds are subject to our refund policy\n\n## 11. COMMERCIAL LICENSE\n\nTo use the Software for commercial purposes:\n- Starter tier: Personal use only\n- Professional tier: Commercial use permitted\n- Enterprise tier: Team commercial use permitted\n\nFor commercial use with Starter tier, contact: hello@unisai.vercel.app\n\n## 12. THIRD-PARTY SERVICES\n\nThe Software uses third-party services (e.g., Anthropic API). Your use is also subject to their terms of service:\n- Anthropic: https://www.anthropic.com/terms\n- OpenAI: https://openai.com/terms (if applicable)\n\n## 13. MODIFICATIONS TO SOFTWARE\n\nLicensor reserves the right to:\n- Update the Software at any time\n- Add or remove features\n- Change pricing (with 30 days notice)\n- Discontinue the Software (with 60 days notice)\n\n## 14. COMPLIANCE\n\nYou agree to comply with all applicable laws and regulations in your jurisdiction when using the Software.\n\n## 15. DISPUTE RESOLUTION\n\nAny disputes arising from this agreement shall be:\n- Resolved through binding arbitration\n- Governed by California law\n- Conducted in English\n\n## 16. ENTIRE AGREEMENT\n\nThis agreement, along with our Privacy Policy and Terms of Service, constitutes the entire agreement between you and Licensor regarding the Software.\n\n## 17. CONTACT\n\nFor licensing inquiries or support:\n- Email: hello@unisai.vercel.app\n- Website: https://unisai.vercel.app\n- Support: vedxnts@gmail.com\n- X: vedxnts\n\n---\n\n**By using the Software, you acknowledge that you have read, understood, and agree to be bound by this License Agreement.**\n\n© 2026 UniAI. All rights reserved.\n\nFile v1.1.9:manifest.yaml\n\nname: \"Prompt Performance Tester\"\nid: \"prompt-performance-tester\"\nversion: \"1.1.8\"\ndescription: \"Model-agnostic prompt benchmarking across 9 providers. Pass any model ID from Claude, GPT, Gemini, DeepSeek, Grok, MiniMax, Qwen, Llama, Mistral — provider auto-detected. Measures latency, cost, quality, and consistency.\"\n\nhomepage: \"https://unisai.vercel.app\"\nrepository: \"https://github.com/vedantsingh60/prompt-performance-tester\"\nsource: \"included\"\n\nintellectual_property:\n  license: \"free-to-use\"\n  license_file: \"LICENSE.md\"\n  copyright: \"© 2026 UnisAI. All rights reserved.\"\n  distribution: \"via-clawhub-only\"\n  source_code_access: \"included\"\n  modification: \"personal-use-only\"\n  reverse_engineering: \"allowed-for-security-audit\"\n\nauthor:\n  company: \"UnisAI\"\n  contact: \"hello@unisai.vercel.app\"\n  website: \"https://unisai.vercel.app\"\n\ncategory: \"ai-testing\"\ntags:\n  - \"prompt-testing\"\n  - \"performance-analysis\"\n  - \"cost-optimization\"\n  - \"multi-llm\"\n  - \"quality-assurance\"\n  - \"benchmarking\"\n  - \"llm-comparison\"\n  - \"ai-testing\"\n\npricing:\n  model: \"free\"\n\nruntime: \"local\"\nexecution: \"python\"\n\nrequired_env_vars:\n  - \"ANTHROPIC_API_KEY\"   # Required if testing Claude models\n  - \"OPENAI_API_KEY\"      # Required if testing GPT models\n  - \"GOOGLE_API_KEY\"      # Required if testing Gemini models\n  - \"MISTRAL_API_KEY\"     # Required if testing Mistral models\n  - \"DEEPSEEK_API_KEY\"    # Required if testing DeepSeek models\n  - \"XAI_API_KEY\"         # Required if testing Grok/xAI models\n  - \"MINIMAX_API_KEY\"     # Required if testing MiniMax models\n  - \"DASHSCOPE_API_KEY\"   # Required if testing Qwen/Alibaba models\n  - \"OPENROUTER_API_KEY\"  # Required if testing Llama/OpenRouter models\nprimary_credential: \"At least ONE provider API key is required per provider you want to test\"\n\ndependencies:\n  python: \">=3.9\"\n  packages:\n    - \"anthropic>=0.40.0\"\n    - \"openai>=1.60.0\"\n    - \"google-generativeai>=0.8.0\"\n    - \"mistralai>=1.3.0\"\n  install_all: \"pip install anthropic openai google-generativeai mistralai\"\n  install_selective: |\n    pip install anthropic          # Claude\n    pip install openai             # GPT, DeepSeek, xAI, MiniMax, Qwen, Llama (OpenAI-compat)\n    pip install google-generativeai  # Gemini\n    pip install mistralai          # Mistral\n  note: \"Install only the SDKs for the providers you plan to test. DeepSeek, xAI, MiniMax, Qwen, and Llama all use the openai package with a custom base URL.\"\n  requirements_file: \"requirements.txt\"\n\nsecurity:\n  data_retention: \"0 days\"\n  data_flow: \"prompts-sent-to-chosen-ai-providers\"\n  third_party_data_sharing: |\n    WARNING: This skill sends your prompts to whichever AI providers you select for testing.\n    Each provider has their own data retention and privacy policies:\n    - Anthropic: https://www.anthropic.com/legal/privacy\n    - OpenAI: https://openai.com/policies/privacy-policy\n    - Google: https://ai.google.dev/gemini-api/terms\n    - Mistral: https://mistral.ai/terms/\n    - DeepSeek: https://www.deepseek.com/privacy_policy\n    - xAI: https://x.ai/privacy\n    - OpenRouter: https://openrouter.ai/privacy\n  api_key_storage: \"Environment variables only — never hardcoded or logged\"\n  network_access: \"Required to call chosen AI provider APIs\"\n\ncapabilities:\n  functions:\n    - name: \"testPrompt\"\n      description: \"Test a prompt across multiple LLM models and providers\"\n      parameters:\n        prompt_text:\n          type: \"string\"\n          description: \"The prompt to benchmark\"\n          required: true\n        models:\n          type: \"array\"\n          description: \"List of model IDs to test — any model matching a supported prefix works\"\n          items:\n            type: \"string\"\n          examples:\n            - \"claude-sonnet-4-6\"\n            - \"gpt-5.2\"\n            - \"deepseek-chat\"\n            - \"grok-4-1-fast\"\n            - \"gemini-2.5-flash\"\n          required: false\n        num_runs:\n          type: \"number\"\n          description: \"Number of runs per model for consistency testing\"\n          default: 1\n          range: [1, 10]\n        system_prompt:\n          type: \"string\"\n          description: \"Optional system prompt\"\n        max_tokens:\n          type: \"number\"\n          description: \"Maximum response tokens\"\n          default: 1000\n          range: [100, 4000]\n\nenvironment_variables:\n  ANTHROPIC_API_KEY:\n    description: \"Anthropic API key — required for any claude-* model\"\n    required_for_prefix: \"claude-\"\n  OPENAI_API_KEY:\n    description: \"OpenAI API key — required for any gpt-*, o1*, o3* model\"\n    required_for_prefix: \"gpt-, o1, o3\"\n  GOOGLE_API_KEY:\n    description: \"Google AI API key — required for any gemini-* model\"\n    required_for_prefix: \"gemini-\"\n  MISTRAL_API_KEY:\n    description: \"Mistral API key — required for mistral-*, mixtral-* models\"\n    required_for_prefix: \"mistral-, mixtral-\"\n  DEEPSEEK_API_KEY:\n    description: \"DeepSeek API key — required for any deepseek-* model\"\n    required_for_prefix: \"deepseek-\"\n  XAI_API_KEY:\n    description: \"xAI API key — required for any grok-* model\"\n    required_for_prefix: \"grok-\"\n  MINIMAX_API_KEY:\n    description: \"MiniMax API key — required for minimax* or MiniMax* models\"\n    required_for_prefix: \"minimax, MiniMax\"\n  DASHSCOPE_API_KEY:\n    description: \"Alibaba DashScope API key — required for any qwen* model\"\n    required_for_prefix: \"qwen\"\n  OPENROUTER_API_KEY:\n    description: \"OpenRouter API key — required for meta-llama/* or llama-* models\"\n    required_for_prefix: \"meta-llama/, llama-\"\n\nsupport:\n  support_email: \"support@unisai.vercel.app\"\n  website: \"https://unisai.vercel.app\"\n  github: \"https://github.com/vedantsingh60/prompt-performance-tester\"\n  documentation: \"See SKILL.md in this package\"\n  response_time: \"Best effort — community supported\"\n\nrestrictions:\n  - \"No redistribution outside ClawhHub registry\"\n  - \"No resale or sublicensing\"\n  - \"No trademark usage without permission\"\n  - \"Modifications allowed for personal use only\"\n\nchangelog:\n  \"1.1.8\":\n    - \"🏗️ Model-agnostic architecture — provider auto-detected from model name prefix, no hardcoded whitelist\"\n    - \"✨ Added DeepSeek, xAI Grok, MiniMax, Qwen as first-class providers (9 total)\"\n    - \"✨ Updated Claude to 4.6 series (claude-opus-4-6, claude-sonnet-4-6)\"\n    - \"✨ Any future model works automatically without code changes\"\n    - \"🔧 Lazy client initialization — only loads SDKs for providers actually used\"\n    - \"🔧 Unified OpenAI-compat path for DeepSeek, xAI, MiniMax, Qwen, OpenRouter\"\n    - \"📝 Fixed UnisAI branding (was UniAI)\"\n    - \"💰 Updated pricing table with 20 models across 9 providers\"\n  \"1.1.5\":\n    - \"🚀 Updated to latest 2026 models\"\n    - \"✨ GPT-5.2 series (Instant, Thinking, Pro)\"\n    - \"✨ Gemini 3 Pro and 2.5 series\"\n    - \"✨ Claude 4.5 pricing updates\"\n    - \"✨ 10 total models across 3 providers\"\n  \"1.1.0\":\n    - \"✨ Multi-provider support (Claude, GPT, Gemini)\"\n    - \"✨ Cross-provider cost comparison\"\n    - \"✨ Enhanced recommendations engine\"\n  \"1.0.0\":\n    - \"Initial release with Claude-only support\"\n    - \"Performance metrics: latency, cost, quality, consistency\"\n\nmetadata:\n  status: \"active\"\n  created_at: \"2024-02-02T00:00:00Z\"\n  updated_at: \"2026-02-27T00:00:00Z\"\n  maturity: \"production\"\n  maintenance: \"actively-maintained\"\n  compatibility:\n    - \"OpenClaw v1.0+\"\n    - \"Claude Code\"\n    - \"ClawhHub v2.0+\"\n  security_audit: \"Source code included for security review and transparency\"\n\nArchive v1.1.8: 5 files, 17509 bytes\n\nFiles: LICENSE.md (4799b), manifest.yaml (7490b), prompt_performance_tester.py (22862b), SKILL.md (17974b), _meta.json (144b)\n\nFile v1.1.8:SKILL.md\n\n# Prompt Performance Tester\n\n**Model-agnostic prompt benchmarking across 9 providers.**\n\nPass any model ID — provider auto-detected. Compare latency, cost, quality, and consistency across Claude, GPT, Gemini, DeepSeek, Grok, MiniMax, Qwen, Llama, and Mistral.\n\n---\n\n## 🚀 Why This Skill?\n\n### Problem Statement\nComparing LLM models across providers requires manual testing:\n- No systematic way to measure performance across models\n- Cost differences are significant but not easily comparable\n- Quality varies by use case and provider\n- Manual API testing is time-consuming and error-prone\n\n### The Solution\nTest prompts across any model from any supported provider simultaneously. Get performance metrics and recommendations based on latency, cost, and quality.\n\n### Example Cost Comparison\nFor 10,000 requests/day with average 28 input + 115 output tokens:\n- Claude Opus 4.6: ~$30.15/day ($903/month)\n- Gemini 2.5 Flash-Lite: ~$0.05/day ($1.50/month)\n- DeepSeek Chat: ~$0.14/day ($4.20/month)\n- Monthly cost difference (Opus vs Flash-Lite): $901.50\n\n---\n\n## ✨ What You Get\n\n### Model-Agnostic Multi-Provider Testing\nPass any model ID — provider is auto-detected from the model name prefix.\nNo hardcoded list; new models work without code changes.\n\n| Provider | Example Models | Prefix | Required Key |\n|----------|---------------|--------|--------------|\n| **Anthropic** | claude-opus-4-6, claude-sonnet-4-6, claude-haiku-4-5-20251001 | `claude-` | ANTHROPIC_API_KEY |\n| **OpenAI** | gpt-5.2-pro, gpt-5.2, gpt-5.1 | `gpt-`, `o1`, `o3` | OPENAI_API_KEY |\n| **Google** | gemini-2.5-pro, gemini-2.5-flash, gemini-2.5-flash-lite | `gemini-` | GOOGLE_API_KEY |\n| **Mistral** | mistral-large-latest, mistral-small-latest | `mistral-`, `mixtral-` | MISTRAL_API_KEY |\n| **DeepSeek** | deepseek-chat, deepseek-reasoner | `deepseek-` | DEEPSEEK_API_KEY |\n| **xAI** | grok-4-1-fast, grok-3-beta | `grok-` | XAI_API_KEY |\n| **MiniMax** | MiniMax-M2.1 | `MiniMax`, `minimax` | MINIMAX_API_KEY |\n| **Qwen** | qwen3.5-plus, qwen3-max-instruct | `qwen` | DASHSCOPE_API_KEY |\n| **Meta Llama** | meta-llama/llama-4-maverick, meta-llama/llama-3.3-70b-instruct | `meta-llama/`, `llama-` | OPENROUTER_API_KEY |\n\n### Known Pricing (per 1M tokens)\n\n| Model | Input | Output |\n|-------|-------|--------|\n| claude-opus-4-6 | $15.00 | $75.00 |\n| claude-sonnet-4-6 | $3.00 | $15.00 |\n| claude-haiku-4-5-20251001 | $1.00 | $5.00 |\n| gpt-5.2-pro | $21.00 | $168.00 |\n| gpt-5.2 | $1.75 | $14.00 |\n| gpt-5.1 | $2.00 | $8.00 |\n| gemini-2.5-pro | $1.25 | $10.00 |\n| gemini-2.5-flash | $0.30 | $2.50 |\n| gemini-2.5-flash-lite | $0.10 | $0.40 |\n| mistral-large-latest | $2.00 | $6.00 |\n| mistral-small-latest | $0.10 | $0.30 |\n| deepseek-chat | $0.27 | $1.10 |\n| deepseek-reasoner | $0.55 | $2.19 |\n| grok-4-1-fast | $5.00 | $25.00 |\n| grok-3-beta | $3.00 | $15.00 |\n| MiniMax-M2.1 | $0.40 | $1.60 |\n| qwen3.5-plus | $0.57 | $2.29 |\n| qwen3-max-instruct | $1.60 | $6.40 |\n| meta-llama/llama-4-maverick | $0.20 | $0.60 |\n| meta-llama/llama-3.3-70b-instruct | $0.59 | $0.79 |\n\n> **Note:** Unlisted models still work — cost calculation returns $0.00 with a warning. Pricing table is for reference only, not a validation gate.\n\n### Performance Metrics\n\nEvery test measures:\n- ⚡ **Latency** — Response time in milliseconds\n- 💰 **Cost** — Exact API cost per request (input + output tokens)\n- 🎯 **Quality** — Response quality score (0–100)\n- 📊 **Token Usage** — Input and output token counts\n- 🔄 **Consistency** — Variance across multiple test runs\n- ❌ **Error Tracking** — API failures, timeouts, rate limits\n\n### Smart Recommendations\n\nGet instant answers to:\n- Which model is **fastest** for your prompt?\n- Which is most **cost-effective**?\n- Which produces **best quality** responses?\n- How much can you **save** by switching providers?\n\n---\n\n## 📊 Real-World Example\n\n```\nPROMPT: \"Write a professional customer service response about a delayed shipment\"\n\n┌─────────────────────────────────────────────────────────────────┐\n│ GEMINI 2.5 FLASH-LITE (Google) 💰 MOST AFFORDABLE              │\n├─────────────────────────────────────────────────────────────────┤\n│ Latency:  523ms                                                 │\n│ Cost:     $0.000025                                             │\n│ Quality:  65/100                                                │\n│ Tokens:   28 in / 87 out                                        │\n└─────────────────────────────────────────────────────────────────┘\n\n┌─────────────────────────────────────────────────────────────────┐\n│ DEEPSEEK CHAT (DeepSeek) 💡 BUDGET PICK                        │\n├─────────────────────────────────────────────────────────────────┤\n│ Latency:  710ms                                                 │\n│ Cost:     $0.000048                                             │\n│ Quality:  70/100                                                │\n│ Tokens:   28 in / 92 out                                        │\n└─────────────────────────────────────────────────────────────────┘\n\n┌─────────────────────────────────────────────────────────────────┐\n│ CLAUDE HAIKU 4.5 (Anthropic) 🚀 BALANCED PERFORMER             │\n├─────────────────────────────────────────────────────────────────┤\n│ Latency:  891ms                                                 │\n│ Cost:     $0.000145                                             │\n│ Quality:  78/100                                                │\n│ Tokens:   28 in / 102 out                                       │\n└─────────────────────────────────────────────────────────────────┘\n\n┌─────────────────────────────────────────────────────────────────┐\n│ GPT-5.2 (OpenAI) 💡 EXCELLENT QUALITY                          │\n├─────────────────────────────────────────────────────────────────┤\n│ Latency:  645ms                                                 │\n│ Cost:     $0.000402                                             │\n│ Quality:  88/100                                                │\n│ Tokens:   28 in / 98 out                                        │\n└─────────────────────────────────────────────────────────────────┘\n\n┌─────────────────────────────────────────────────────────────────┐\n│ CLAUDE OPUS 4.6 (Anthropic) 🏆 HIGHEST QUALITY                 │\n├─────────────────────────────────────────────────────────────────┤\n│ Latency:  1,234ms                                               │\n│ Cost:     $0.001875                                             │\n│ Quality:  94/100                                                │\n│ Tokens:   28 in / 125 out                                       │\n└─────────────────────────────────────────────────────────────────┘\n\n🎯 RECOMMENDATIONS:\n1. Most cost-effective: Gemini 2.5 Flash-Lite ($0.000025/request) — 99.98% cheaper than Opus\n2. Budget pick: DeepSeek Chat ($0.000048/request) — strong quality at low cost\n3. Best quality: Claude Opus 4.6 (94/100) — state-of-the-art reasoning & analysis\n4. Smart pick: Claude Haiku 4.5 ($0.000145/request) — 81% cheaper, 83% quality match\n5. Speed + Quality: GPT-5.2 ($0.000402/request) — excellent quality at mid-range cost\n\n💡 Potential monthly savings (10,000 requests/day, 28 input + 115 output tokens avg):\n   - Using Gemini 2.5 Flash-Lite vs Opus: $903/month saved ($1.44 vs $904.50)\n   - Using DeepSeek Chat vs Opus: $899/month saved ($4.50 vs $904.50)\n   - Using Claude Haiku vs Opus: $731/month saved ($173.40 vs $904.50)\n```\n\n---\n\n## Use Cases\n\n### Production Deployment\n- Evaluate models before production selection\n- Compare cost vs quality tradeoffs\n- Benchmark API latency across providers\n\n### Prompt Development\n- Test prompt variations across models\n- Measure quality scores consistently\n- Compare performance metrics\n\n### Cost Analysis\n- Analyze LLM API spending by model\n- Compare provider pricing structures\n- Identify cost-efficient alternatives\n\n### Performance Testing\n- Measure latency and response times\n- Test consistency across multiple runs\n- Evaluate quality scores\n\n---\n\n## 🚀 Quick Start\n\n### 1. Subscribe to Skill\nClick \"Subscribe\" on ClawhHub to get access.\n\n### 2. Set API Keys\nAdd keys for the providers you want to test:\n\n```bash\n# Anthropic (Claude models)\nexport ANTHROPIC_API_KEY=\"sk-ant-...\"\n\n# OpenAI (GPT models)\nexport OPENAI_API_KEY=\"sk-...\"\n\n# Google (Gemini models)\nexport GOOGLE_API_KEY=\"AI...\"\n\n# DeepSeek\nexport DEEPSEEK_API_KEY=\"...\"\n\n# xAI (Grok models)\nexport XAI_API_KEY=\"...\"\n\n# MiniMax\nexport MINIMAX_API_KEY=\"...\"\n\n# Alibaba (Qwen models)\nexport DASHSCOPE_API_KEY=\"...\"\n\n# OpenRouter (Meta Llama models)\nexport OPENROUTER_API_KEY=\"...\"\n\n# Mistral\nexport MISTRAL_API_KEY=\"...\"\n```\n\nYou only need keys for the providers you plan to test.\n\n### 3. Install Dependencies\n\n```bash\n# Install only what you need\npip install anthropic          # Claude\npip install openai             # GPT, DeepSeek, xAI, MiniMax, Qwen, Llama\npip install google-generativeai  # Gemini\npip install mistralai          # Mistral\n\n# Or install everything\npip install anthropic openai google-generativeai mistralai\n```\n\n### 4. Run Your First Test\n\n**Option A: Python**\n```python\nimport os\nfrom prompt_performance_tester import PromptPerformanceTester\n\ntester = PromptPerformanceTester()  # reads API keys from environment\n\nresults = tester.test_prompt(\n    prompt_text=\"Write a professional email apologizing for a delayed shipment\",\n    models=[\n        \"claude-haiku-4-5-20251001\",\n        \"gpt-5.2\",\n        \"gemini-2.5-flash\",\n        \"deepseek-chat\",\n    ],\n    num_runs=3,\n    max_tokens=500\n)\n\nprint(tester.format_results(results))\nprint(f\"🏆 Best quality:  {results.best_model}\")\nprint(f\"💰 Cheapest:      {results.cheapest_model}\")\nprint(f\"⚡ Fastest:       {results.fastest_model}\")\n```\n\n**Option B: CLI**\n```bash\n# Test across multiple models\nprompt-tester test \"Your prompt here\" \\\n  --models claude-haiku-4-5-20251001 gpt-5.2 gemini-2.5-flash deepseek-chat \\\n  --runs 3\n\n# Export results\nprompt-tester test \"Your prompt here\" --export results.json\n```\n\n---\n\n## 🔒 Security & Privacy\n\n### API Key Safety\n- Keys stored in environment variables only — never hardcoded or logged\n- Never transmitted to UnisAI servers\n- HTTPS encryption for all provider API calls\n\n### Data Privacy\n- Your prompts are sent only to the AI providers you select for testing\n- Each provider has their own data retention policy (see their privacy pages)\n- No data stored on UnisAI infrastructure\n\n---\n\n## 📚 Technical Details\n\n### System Requirements\n- **Python**: 3.9+\n- **Dependencies**: `anthropic`, `openai`, `google-generativeai`, `mistralai` (install only what you need)\n- **Platform**: macOS, Linux, Windows\n\n### Architecture\n- **Lazy client initialization** — SDK clients only loaded for providers actually tested\n- **Prefix-based routing** — `PROVIDER_MAP` detects provider from model name; no hardcoded whitelist\n- **OpenAI-compat path** — DeepSeek, xAI, MiniMax, Qwen, and OpenRouter all use the `openai` SDK with a custom `base_url`\n- **Pricing table** — used for cost calculation only; unknown models get `cost=0` with a warning\n\n### Metrics Collected\nEvery test captures:\n- **Latency**: Total response time (ms)\n- **Cost**: Input + output cost based on known pricing (USD)\n- **Quality**: Heuristic response score based on length, completeness (0–100)\n- **Tokens**: Exact input/output token counts per provider\n- **Consistency**: Standard deviation across multiple runs\n- **Errors**: Timeouts, rate limits, API failures\n\n---\n\n## ❓ Frequently Asked Questions\n\n**Q: Do I need API keys for all 9 providers?**\nA: No. You only need keys for the providers you want to test. If you only test Claude models, you only need `ANTHROPIC_API_KEY`.\n\n**Q: Who pays for the API costs?**\nA: You do. You provide your own API keys and pay each provider directly. This skill has no per-request fees.\n\n**Q: How accurate are the cost calculations?**\nA: Costs are calculated from the known pricing table using actual token counts. Models not in the pricing table return `$0.00` — the model still runs, the cost just won't be shown.\n\n**Q: Can I test models not in the pricing table?**\nA: Yes. Any model whose name starts with a supported prefix will run. Cost will show as $0.00 for unlisted models.\n\n**Q: Can I test prompts in non-English languages?**\nA: Yes. All supported providers handle multiple languages.\n\n**Q: Can I use this in production/CI/CD?**\nA: Yes. Import `PromptPerformanceTester` directly from Python or call via CLI.\n\n**Q: What if my prompt is very long?**\nA: Set `max_tokens` appropriately. The skill passes your prompt as-is to each provider's API.\n\n---\n\n## 🗺️ Roadmap\n\n### ✅ Current Release (v1.1.8)\n- Model-agnostic architecture — any model ID works via prefix detection\n- 9 providers, 20 known models with pricing\n- DeepSeek, xAI Grok, MiniMax, Qwen, Meta Llama as first-class providers\n- Claude 4.6 series (opus-4-6, sonnet-4-6)\n- Lazy client initialization — only loads SDKs for providers actually used\n- Fixed UnisAI branding throughout\n\n### 🚧 Coming Soon (v1.3)\n- **Batch testing**: Test 100+ prompts simultaneously\n- **Historical tracking**: Track model performance over time\n- **Webhook integrations**: Slack, Discord, email notifications\n\n### 🔮 Future (v1.3+)\n- **A/B testing framework**: Scientific prompt experimentation\n- **Fine-tuning insights**: Which models to fine-tune for your use case\n- **Custom benchmarks**: Create your own evaluation criteria\n- **Auto-optimization**: AI-powered prompt improvement suggestions\n\n---\n\n## 📞 Support\n\n- **Email**: support@unisai.vercel.app\n- **Website**: https://unisai.vercel.app\n- **Bug Reports**: support@unisai.vercel.app\n\n---\n\n## 📄 License & Terms\n\nThis skill is distributed via ClawhHub under the following terms.\n\n### ✅ You CAN:\n- Use for your own business and projects\n- Test prompts for internal applications\n- Modify source code for personal use\n\n### ❌ You CANNOT:\n- Redistribute outside ClawhHub registry\n- Resell or sublicense\n- Use UnisAI trademark without permission\n\n**Full Terms**: See [LICENSE.md](LICENSE.md)\n\n---\n\n## 📝 Changelog\n\n### [1.1.8] - 2026-02-27\n\n#### Fixes & Polish\n- Bumped version to 1.1.8\n- SKILL.md fully rewritten — cleaned up formatting, removed stale content\n- Removed old IP watermark reference (`PROPRIETARY_SKILL_VEDANT_2024`) from docs\n- Corrected watermark to `PROPRIETARY_SKILL_UNISAI_2026_MULTI_PROVIDER` throughout\n- Fixed all UnisAI branding (was UniAI in v1.1.0 changelog)\n- Updated pricing table to include all 20 known models\n- Cleaned up FAQ, Quick Start, and Use Cases sections\n\n### [1.1.6] - 2026-02-27\n\n#### 🏗️ Model-Agnostic Architecture\n- Provider auto-detected from model name prefix — no hardcoded whitelist\n- Any new model works automatically without code changes\n- Added DeepSeek, xAI Grok, MiniMax, Qwen, Meta Llama as first-class providers (9 total)\n- Updated Claude to 4.6 series (claude-opus-4-6, claude-sonnet-4-6)\n- Lazy client initialization — only loads SDKs for providers actually tested\n- Unified OpenAI-compat path for DeepSeek, xAI, MiniMax, Qwen, OpenRouter\n\n### [1.1.5] - 2026-02-01\n\n#### 🚀 Latest Models Update\n- GPT-5.2 Series — Added Instant, Thinking, and Pro variants\n- Gemini 2.5 Series — Updated to 2.5 Pro, Flash, and Flash-Lite\n- Claude 4.5 pricing updates\n- 10 total models across 3 providers\n\n### [1.1.0] - 2026-01-15\n\n#### ✨ Major Features\n- Multi-provider support — Claude, GPT, Gemini\n- Cross-provider cost comparison\n- Enhanced recommendations engine\n- Rebranded to UnisAI\n\n### [1.0.0] - 2024-02-02\n\n#### Initial Release\n- Claude-only prompt testing (Haiku, Sonnet, Opus)\n- Performance metrics: latency, cost, quality, consistency\n- Basic recommendations engine\n\n---\n\n**Last Updated**: February 27, 2026\n**Current Version**: 1.1.8\n**Status**: Active & Maintained\n\n© 2026 UnisAI. All rights reserved.\n\nFile v1.1.8:_meta.json\n\n{\n  \"ownerId\": \"kn77yjs5esft2kgsd6dpz9c92n80dgsy\",\n  \"slug\": \"prompt-performance-tester\",\n  \"version\": \"1.1.8\",\n  \"publishedAt\": 1772213247786\n}\n\nFile v1.1.8:LICENSE.md\n\n# UniAI Skills - Proprietary License\n\n**Version 1.0 | Effective Date: February 2, 2024**\n\n## 1. GRANT OF LICENSE\n\nUniAI (\"Licensor\") grants you (\"Licensee\") a limited, non-exclusive, non-transferable, revocable license to use the ClawhHub Skills (\"Software\") solely in accordance with the terms of this license agreement.\n\n## 2. LICENSE RESTRICTIONS\n\nYou may NOT:\n- Reverse engineer, decompile, or disassemble the Software\n- Modify, alter, or create derivative works of the Software\n- Remove, obscure, or alter any proprietary notices or labels on the Software\n- Share, distribute, or sublicense the Software to any third party\n- Use the Software for commercial purposes without a commercial license\n- Access or use the Software beyond the scope of your subscription tier\n- Attempt to circumvent licensing controls or API rate limits\n\n## 3. INTELLECTUAL PROPERTY RIGHTS\n\nAll intellectual property rights in and to the Software are retained by Licensor. This includes:\n- Source code and object code\n- Algorithms and methodologies\n- Performance optimization techniques\n- Quality scoring mechanisms\n- Proprietary data structures\n- Trade secrets and confidential information\n\n## 4. PERMITTED USES\n\nYou may only:\n- Use the Software as provided through the ClawhHub platform\n- Access features available in your subscription tier\n- Create test results and reports for internal use\n- Share results with your team (if on a team plan)\n- Provide feedback to improve the Software\n\n## 5. SUBSCRIPTION TIERS\n\n### Starter (Free)\n- 5 tests per month\n- 2 models per test\n- Basic features\n- Personal use only\n\n### Professional ($29/month)\n- Unlimited tests\n- All models supported\n- Advanced analytics\n- API access\n- Commercial use permitted\n\n### Enterprise ($99/month)\n- Team collaboration\n- White-label option\n- Custom integrations\n- Dedicated support\n- SLA guarantees\n\n## 6. API KEY AND CREDENTIALS\n\n- You are responsible for keeping your API keys confidential\n- Do not share your license key with others\n- One license per person/organization\n- License keys are non-transferable\n- Unauthorized sharing may result in account termination\n\n## 7. DATA PRIVACY\n\n- We do not retain your test data by default\n- Free tier: 30-day retention\n- Paid tiers: 90-day retention\n- You can request data deletion anytime\n- See Privacy Policy for full details\n\n## 8. WARRANTY DISCLAIMER\n\nTHE SOFTWARE IS PROVIDED \"AS-IS\" WITHOUT ANY WARRANTIES. LICENSOR DISCLAIMS ALL WARRANTIES, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO:\n- Merchantability\n- Fitness for a particular purpose\n- Non-infringement\n- Accuracy of results\n\n## 9. LIMITATION OF LIABILITY\n\nIN NO EVENT SHALL LICENSOR BE LIABLE FOR:\n- Any indirect, incidental, special, or consequential damages\n- Loss of data, revenue, or profits\n- Business interruption\n- Even if advised of the possibility of such damages\n\n## 10. TERMINATION\n\nLicensor may terminate your license if you:\n- Violate any terms of this agreement\n- Fail to pay subscription fees\n- Attempt to reverse engineer the Software\n- Share your license key with others\n- Use the Software unlawfully\n\nUpon termination:\n- Your access to the Software is immediately revoked\n- You must destroy all copies of the Software in your possession\n- Any refunds are subject to our refund policy\n\n## 11. COMMERCIAL LICENSE\n\nTo use the Software for commercial purposes:\n- Starter tier: Personal use only\n- Professional tier: Commercial use permitted\n- Enterprise tier: Team commercial use permitted\n\nFor commercial use with Starter tier, contact: hello@unisai.vercel.app\n\n## 12. THIRD-PARTY SERVICES\n\nThe Software uses third-party services (e.g., Anthropic API). Your use is also subject to their terms of service:\n- Anthropic: https://www.anthropic.com/terms\n- OpenAI: https://openai.com/terms (if applicable)\n\n## 13. MODIFICATIONS TO SOFTWARE\n\nLicensor reserves the right to:\n- Update the Software at any time\n- Add or remove features\n- Change pricing (with 30 days notice)\n- Discontinue the Software (with 60 days notice)\n\n## 14. COMPLIANCE\n\nYou agree to comply with all applicable laws and regulations in your jurisdiction when using the Software.\n\n## 15. DISPUTE RESOLUTION\n\nAny disputes arising from this agreement shall be:\n- Resolved through binding arbitration\n- Governed by California law\n- Conducted in English\n\n## 16. ENTIRE AGREEMENT\n\nThis agreement, along with our Privacy Policy and Terms of Service, constitutes the entire agreement between you and Licensor regarding the Software.\n\n## 17. CONTACT\n\nFor licensing inquiries or support:\n- Email: hello@unisai.vercel.app\n- Website: https://unisai.vercel.app\n- Support: vedxnts@gmail.com\n- X: vedxnts\n\n---\n\n**By using the Software, you acknowledge that you have read, understood, and agree to be bound by this License Agreement.**\n\n© 2026 UniAI. All rights reserved.\n\nFile v1.1.8:manifest.yaml\n\nname: \"Prompt Performance Tester\"\nid: \"prompt-performance-tester\"\nversion: \"1.1.8\"\ndescription: \"Model-agnostic prompt benchmarking across 9 providers. Pass any model ID from Claude, GPT, Gemini, DeepSeek, Grok, MiniMax, Qwen, Llama, Mistral — provider auto-detected. Measures latency, cost, quality, and consistency.\"\n\nhomepage: \"https://unisai.vercel.app\"\nrepository: \"https://github.com/vedantsingh60/prompt-performance-tester\"\nsource: \"included\"\n\nintellectual_property:\n  license: \"free-to-use\"\n  license_file: \"LICENSE.md\"\n  copyright: \"© 2026 UnisAI. All rights reserved.\"\n  distribution: \"via-clawhub-only\"\n  source_code_access: \"included\"\n  modification: \"personal-use-only\"\n  reverse_engineering: \"allowed-for-security-audit\"\n\nauthor:\n  company: \"UnisAI\"\n  contact: \"hello@unisai.vercel.app\"\n  website: \"https://unisai.vercel.app\"\n\ncategory: \"ai-testing\"\ntags:\n  - \"prompt-testing\"\n  - \"performance-analysis\"\n  - \"cost-optimization\"\n  - \"multi-llm\"\n  - \"quality-assurance\"\n  - \"benchmarking\"\n  - \"llm-comparison\"\n  - \"ai-testing\"\n\npricing:\n  model: \"free\"\n\nruntime: \"local\"\nexecution: \"python\"\n\nrequired_env_vars:\n  - \"ANTHROPIC_API_KEY\"   # Required if testing Claude models\n  - \"OPENAI_API_KEY\"      # Required if testing GPT models\n  - \"GOOGLE_API_KEY\"      # Required if testing Gemini models\n  - \"MISTRAL_API_KEY\"     # Required if testing Mistral models\n  - \"DEEPSEEK_API_KEY\"    # Required if testing DeepSeek models\n  - \"XAI_API_KEY\"         # Required if testing Grok/xAI models\n  - \"MINIMAX_API_KEY\"     # Required if testing MiniMax models\n  - \"DASHSCOPE_API_KEY\"   # Required if testing Qwen/Alibaba models\n  - \"OPENROUTER_API_KEY\"  # Required if testing Llama/OpenRouter models\nprimary_credential: \"At least ONE provider API key is required per provider you want to test\"\n\ndependencies:\n  python: \">=3.9\"\n  packages:\n    - \"anthropic>=0.40.0\"\n    - \"openai>=1.60.0\"\n    - \"google-generativeai>=0.8.0\"\n    - \"mistralai>=1.3.0\"\n  install_all: \"pip install anthropic openai google-generativeai mistralai\"\n  install_selective: |\n    pip install anthropic          # Claude\n    pip install openai             # GPT, DeepSeek, xAI, MiniMax, Qwen, Llama (OpenAI-compat)\n    pip install google-generativeai  # Gemini\n    pip install mistralai          # Mistral\n  note: \"Install only the SDKs for the providers you plan to test. DeepSeek, xAI, MiniMax, Qwen, and Llama all use the openai package with a custom base URL.\"\n  requirements_file: \"requirements.txt\"\n\nsecurity:\n  data_retention: \"0 days\"\n  data_flow: \"prompts-sent-to-chosen-ai-providers\"\n  third_party_data_sharing: |\n    WARNING: This skill sends your prompts to whichever AI providers you select for testing.\n    Each provider has their own data retention and privacy policies:\n    - Anthropic: https://www.anthropic.com/legal/privacy\n    - OpenAI: https://openai.com/policies/privacy-policy\n    - Google: https://ai.google.dev/gemini-api/terms\n    - Mistral: https://mistral.ai/terms/\n    - DeepSeek: https://www.deepseek.com/privacy_policy\n    - xAI: https://x.ai/privacy\n    - OpenRouter: https://openrouter.ai/privacy\n  api_key_storage: \"Environment variables only — never hardcoded or logged\"\n  network_access: \"Required to call chosen AI provider APIs\"\n\ncapabilities:\n  functions:\n    - name: \"testPrompt\"\n      description: \"Test a prompt across multiple LLM models and providers\"\n      parameters:\n        prompt_text:\n          type: \"string\"\n          description: \"The prompt to benchmark\"\n          required: true\n        models:\n          type: \"array\"\n          description: \"List of model IDs to test — any model matching a supported prefix works\"\n          items:\n            type: \"string\"\n          examples:\n            - \"claude-sonnet-4-6\"\n            - \"gpt-5.2\"\n            - \"deepseek-chat\"\n            - \"grok-4-1-fast\"\n            - \"gemini-2.5-flash\"\n          required: false\n        num_runs:\n          type: \"number\"\n          description: \"Number of runs per model for consistency testing\"\n          default: 1\n          range: [1, 10]\n        system_prompt:\n          type: \"string\"\n          description: \"Optional system prompt\"\n        max_tokens:\n          type: \"number\"\n          description: \"Maximum response tokens\"\n          default: 1000\n          range: [100, 4000]\n\nenvironment_variables:\n  ANTHROPIC_API_KEY:\n    description: \"Anthropic API key — required for any claude-* model\"\n    required_for_prefix: \"claude-\"\n  OPENAI_API_KEY:\n    description: \"OpenAI API key — required for any gpt-*, o1*, o3* model\"\n    required_for_prefix: \"gpt-, o1, o3\"\n  GOOGLE_API_KEY:\n    description: \"Google AI API key — required for any gemini-* model\"\n    required_for_prefix: \"gemini-\"\n  MISTRAL_API_KEY:\n    description: \"Mistral API key — required for mistral-*, mixtral-* models\"\n    required_for_prefix: \"mistral-, mixtral-\"\n  DEEPSEEK_API_KEY:\n    description: \"DeepSeek API key — required for any deepseek-* model\"\n    required_for_prefix: \"deepseek-\"\n  XAI_API_KEY:\n    description: \"xAI API key — required for any grok-* model\"\n    required_for_prefix: \"grok-\"\n  MINIMAX_API_KEY:\n    description: \"MiniMax API key — required for minimax* or MiniMax* models\"\n    required_for_prefix: \"minimax, MiniMax\"\n  DASHSCOPE_API_KEY:\n    description: \"Alibaba DashScope API key — required for any qwen* model\"\n    required_for_prefix: \"qwen\"\n  OPENROUTER_API_KEY:\n    description: \"OpenRouter API key — required for meta-llama/* or llama-* models\"\n    required_for_prefix: \"meta-llama/, llama-\"\n\nsupport:\n  support_email: \"support@unisai.vercel.app\"\n  website: \"https://unisai.vercel.app\"\n  github: \"https://github.com/vedantsingh60/prompt-performance-tester\"\n  documentation: \"See SKILL.md in this package\"\n  response_time: \"Best effort — community supported\"\n\nrestrictions:\n  - \"No redistribution outside ClawhHub registry\"\n  - \"No resale or sublicensing\"\n  - \"No trademark usage without permission\"\n  - \"Modifications allowed for personal use only\"\n\nchangelog:\n  \"1.1.8\":\n    - \"🏗️ Model-agnostic architecture — provider auto-detected from model name prefix, no hardcoded whitelist\"\n    - \"✨ Added DeepSeek, xAI Grok, MiniMax, Qwen as first-class providers (9 total)\"\n    - \"✨ Updated Claude to 4.6 series (claude-opus-4-6, claude-sonnet-4-6)\"\n    - \"✨ Any future model works automatically without code changes\"\n    - \"🔧 Lazy client initialization — only loads SDKs for providers actually used\"\n    - \"🔧 Unified OpenAI-compat path for DeepSeek, xAI, MiniMax, Qwen, OpenRouter\"\n    - \"📝 Fixed UnisAI branding (was UniAI)\"\n    - \"💰 Updated pricing table with 20 models across 9 providers\"\n  \"1.1.5\":\n    - \"🚀 Updated to latest 2026 models\"\n    - \"✨ GPT-5.2 series (Instant, Thinking, Pro)\"\n    - \"✨ Gemini 3 Pro and 2.5 series\"\n    - \"✨ Claude 4.5 pricing updates\"\n    - \"✨ 10 total models across 3 providers\"\n  \"1.1.0\":\n    - \"✨ Multi-provider support (Claude, GPT, Gemini)\"\n    - \"✨ Cross-provider cost comparison\"\n    - \"✨ Enhanced recommendations engine\"\n  \"1.0.0\":\n    - \"Initial release with Claude-only support\"\n    - \"Performance metrics: latency, cost, quality, consistency\"\n\nmetadata:\n  status: \"active\"\n  created_at: \"2024-02-02T00:00:00Z\"\n  updated_at: \"2026-02-27T00:00:00Z\"\n  maturity: \"production\"\n  maintenance: \"actively-maintained\"\n  compatibility:\n    - \"OpenClaw v1.0+\"\n    - \"Claude Code\"\n    - \"ClawhHub v2.0+\"\n  security_audit: \"Source code included for security review and transparency\"","readmeExcerpt":"Skill: Prompt Performance Tester - UnisAI Owner: vedantsingh60 Summary: Test prompts across Claude, GPT, and Gemini models and get detailed latency, cost, quality, consistency, and error metrics with smart recommendations. Tags: Latest:1.1.4, ai-testing:1.0.1, ai-testing multi-provider prompt-optimization cost-analysis llm-benchmarking claude gpt gemini performance-testing api-comparison multi-model:1.1.2, claude-api","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"PROMPT: \"Write a professional customer service response about a delayed shipment\"\n\n┌─────────────────────────────────────────────────────────────────┐\n│ GEMINI 2.5 FLASH-LITE (Google) 💰 MOST AFFORDABLE              │\n├─────────────────────────────────────────────────────────────────┤\n│ Latency:  523ms                                                 │\n│ Cost:     $0.000025                                             │\n│ Quality:  65/100                                                │\n│ Tokens:   28 in / 87 out                                        │\n└─────────────────────────────────────────────────────────────────┘\n\n┌─────────────────────────────────────────────────────────────────┐\n│ DEEPSEEK CHAT (DeepSeek) 💡 BUDGET PICK                        │\n├─────────────────────────────────────────────────────────────────┤\n│ Latency:  710ms                                                 │\n│ Cost:     $0.000048                                             │\n│ Quality:  70/100                                                │\n│ Tokens:   28 in / 92 out                                        │\n└─────────────────────────────────────────────────────────────────┘\n\n┌─────────────────────────────────────────────────────────────────┐\n│ CLAUDE HAIKU 4.5 (Anthropic) 🚀 BALANCED PERFORMER             │\n├─────────────────────────────────────────────────────────────────┤\n│ Latency:  891ms                                                 │\n│ Cost:     $0.000145                                             │\n│ Quality:  78/100                                                │\n│ Tokens:   28 in / 102 out                                       │\n└─────────────────────────────────────────────────────────────────┘\n\n┌─────────────────────────────────────────────────────────────────┐\n│ GPT-5.2 (OpenAI) 💡 EXCELLENT QUALITY                          │\n├─────────────────────────────────────────────────────────────────┤\n│ Latency:  645ms                                                 │\n│ Cost:     $0"},{"language":"bash","snippet":"# Anthropic (Claude models)\nexport ANTHROPIC_API_KEY=\"sk-ant-...\"\n\n# OpenAI (GPT models)\nexport OPENAI_API_KEY=\"sk-...\"\n\n# Google (Gemini models)\nexport GOOGLE_API_KEY=\"AI...\"\n\n# DeepSeek\nexport DEEPSEEK_API_KEY=\"...\"\n\n# xAI (Grok models)\nexport XAI_API_KEY=\"...\"\n\n# MiniMax\nexport MINIMAX_API_KEY=\"...\"\n\n# Alibaba (Qwen models)\nexport DASHSCOPE_API_KEY=\"...\"\n\n# OpenRouter (Meta Llama models)\nexport OPENROUTER_API_KEY=\"...\"\n\n# Mistral\nexport MISTRAL_API_KEY=\"...\""},{"language":"bash","snippet":"# Install only what you need\npip install anthropic          # Claude\npip install openai             # GPT, DeepSeek, xAI, MiniMax, Qwen, Llama\npip install google-generativeai  # Gemini\npip install mistralai          # Mistral\n\n# Or install everything\npip install anthropic openai google-generativeai mistralai"},{"language":"python","snippet":"import os\nfrom prompt_performance_tester import PromptPerformanceTester\n\ntester = PromptPerformanceTester()  # reads API keys from environment\n\nresults = tester.test_prompt(\n    prompt_text=\"Write a professional email apologizing for a delayed shipment\",\n    models=[\n        \"claude-haiku-4-5-20251001\",\n        \"gpt-5.2\",\n        \"gemini-2.5-flash\",\n        \"deepseek-chat\",\n    ],\n    num_runs=3,\n    max_tokens=500\n)\n\nprint(tester.format_results(results))\nprint(f\"🏆 Best quality:  {results.best_model}\")\nprint(f\"💰 Cheapest:      {results.cheapest_model}\")\nprint(f\"⚡ Fastest:       {results.fastest_model}\")"},{"language":"bash","snippet":"# Test across multiple models\nprompt-tester test \"Your prompt here\" \\\n  --models claude-haiku-4-5-20251001 gpt-5.2 gemini-2.5-flash deepseek-chat \\\n  --runs 3\n\n# Export results\nprompt-tester test \"Your prompt here\" --export results.json"},{"language":"text","snippet":"PROMPT: \"Write a professional customer service response about a delayed shipment\"\n\n┌─────────────────────────────────────────────────────────────────┐\n│ GEMINI 2.5 FLASH-LITE (Google) 💰 MOST AFFORDABLE              │\n├─────────────────────────────────────────────────────────────────┤\n│ Latency:  523ms                                                 │\n│ Cost:     $0.000025                                             │\n│ Quality:  65/100                                                │\n│ Tokens:   28 in / 87 out                                        │\n└─────────────────────────────────────────────────────────────────┘\n\n┌─────────────────────────────────────────────────────────────────┐\n│ DEEPSEEK CHAT (DeepSeek) 💡 BUDGET PICK                        │\n├─────────────────────────────────────────────────────────────────┤\n│ Latency:  710ms                                                 │\n│ Cost:     $0.000048                                             │\n│ Quality:  70/100                                                │\n│ Tokens:   28 in / 92 out                                        │\n└─────────────────────────────────────────────────────────────────┘\n\n┌─────────────────────────────────────────────────────────────────┐\n│ CLAUDE HAIKU 4.5 (Anthropic) 🚀 BALANCED PERFORMER             │\n├─────────────────────────────────────────────────────────────────┤\n│ Latency:  891ms                                                 │\n│ Cost:     $0.000145                                             │\n│ Quality:  78/100                                                │\n│ Tokens:   28 in / 102 out                                       │\n└─────────────────────────────────────────────────────────────────┘\n\n┌─────────────────────────────────────────────────────────────────┐\n│ GPT-5.2 (OpenAI) 💡 EXCELLENT QUALITY                          │\n├─────────────────────────────────────────────────────────────────┤\n│ Latency:  645ms                                                 │\n│ Cost:     $0"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"# Prompt Performance Tester\n\n**Model-agnostic prompt benchmarking across 9 providers.**\n\nPass any model ID — provider auto-detected. Compare latency, cost, quality, and consistency across Claude, GPT, Gemini, DeepSeek, Grok, MiniMax, Qwen, Llama, and Mistral.\n\n---\n\n## 🚀 Why This Skill?\n\n### Problem Statement\nComparing LLM models across providers requires manual testing:\n- No systematic way to measure performance across models\n- Cost differences are significant but not easily comparable\n- Quality varies by use case and provider\n- Manual API testing is time-consuming and error-prone\n\n### The Solution\nTest prompts across any model from any supported provider simultaneously. Get performance metrics and recommendations based on latency, cost, and quality.\n\n### Example Cost Comparison\nFor 10,000 requests/day with average 28 input + 115 output tokens:\n- Claude Opus 4.6: ~$30.15/day ($903/month)\n- Gemini 2.5 Flash-Lite: ~$0.05/day ($1.50/month)\n- DeepSeek Chat: ~$0.14/day ($4.20/month)\n- Monthly cost difference (Opus vs Flash-Lite): $901.50\n\n---\n\n## ✨ What You Get\n\n### Model-Agnostic Multi-Provider Testing\nPass any model ID — provider is auto-detected from the model name prefix.\nNo hardcoded list; new models work without code changes.\n\n| Provider | Example Models | Prefix | Required Key |\n|----------|---------------|--------|--------------|\n| **Anthropic** | claude-opus-4-6, claude-sonnet-4-6, claude-haiku-4-5-20251001 | `claude-` | ANTHROPIC_API_KEY |\n| **OpenAI** | gpt-5.2-pro, gpt-5.2, gpt-5.1 | `gpt-`, `o1`, `o3` | OPENAI_API_KEY |\n| **Google** | gemini-2.5-pro, gemini-2.5-flash, gemini-2.5-flash-lite | `gemini-` | GOOGLE_API_KEY |\n| **Mistral** | mistral-large-latest, mistral-small-latest | `mistral-`, `mixtral-` | MISTRAL_API_KEY |\n| **DeepSeek** | deepseek-chat, deepseek-reasoner | `deepseek-` | DEEPSEEK_API_KEY |\n| **xAI** | grok-4-1-fast, grok-3-beta | `grok-` | XAI_API_KEY |\n| **MiniMax** | MiniMax-M2.1 | `MiniMax`, `minimax` | MINIMAX_API_KEY |\n| **Qwen** | qwen3.5-plus, qwen3-max-instruct | `qwen` | DASHSCOPE_API_KEY |\n| **Meta Llama** | meta-llama/llama-4-maverick, meta-llama/llama-3.3-70b-instruct | `meta-llama/`, `llama-` | OPENROUTER_API_KEY |\n\n### Known Pricing (per 1M tokens)\n\n| Model | Input | Output |\n|-------|-------|--------|\n| claude-opus-4-6 | $15.00 | $75.00 |\n| claude-sonnet-4-6 | $3.00 | $15.00 |\n| claude-haiku-4-5-20251001 | $1.00 | $5.00 |\n| gpt-5.2-pro | $21.00 | $168.00 |\n| gpt-5.2 | $1.75 | $14.00 |\n| gpt-5.1 | $2.00 | $8.00 |\n| gemini-2.5-pro | $1.25 | $10.00 |\n| gemini-2.5-flash | $0.30 | $2.50 |\n| gemini-2.5-flash-lite | $0.10 | $0.40 |\n| mistral-large-latest | $2.00 | $6.00 |\n| mistral-small-latest | $0.10 | $0.30 |\n| deepseek-chat | $0.27 | $1.10 |\n| deepseek-reasoner | $0.55 | $2.19 |\n| grok-4-1-fast | $5.00 | $25.00 |\n| grok-3-beta | $3.00 | $15.00 |\n| MiniMax-M2.1 | $0.40 | $1.60 |\n| qwen3.5-plus | $0.57 | $2.29 |\n| qwen3-max-instruct | $1.60 | $6.40 |\n| meta-llama/llama-4-maverick | $0.20 | $0.60 |\n| meta-llama/l"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn77yjs5esft2kgsd6dpz9c92n80dgsy\",\n  \"slug\": \"prompt-performance-tester\",\n  \"version\": \"1.1.9\",\n  \"publishedAt\": 1772213259522\n}"},{"path":"LICENSE.md","content":"# UniAI Skills - Proprietary License\n\n**Version 1.0 | Effective Date: February 2, 2024**\n\n## 1. GRANT OF LICENSE\n\nUniAI (\"Licensor\") grants you (\"Licensee\") a limited, non-exclusive, non-transferable, revocable license to use the ClawhHub Skills (\"Software\") solely in accordance with the terms of this license agreement.\n\n## 2. LICENSE RESTRICTIONS\n\nYou may NOT:\n- Reverse engineer, decompile, or disassemble the Software\n- Modify, alter, or create derivative works of the Software\n- Remove, obscure, or alter any proprietary notices or labels on the Software\n- Share, distribute, or sublicense the Software to any third party\n- Use the Software for commercial purposes without a commercial license\n- Access or use the Software beyond the scope of your subscription tier\n- Attempt to circumvent licensing controls or API rate limits\n\n## 3. INTELLECTUAL PROPERTY RIGHTS\n\nAll intellectual property rights in and to the Software are retained by Licensor. This includes:\n- Source code and object code\n- Algorithms and methodologies\n- Performance optimization techniques\n- Quality scoring mechanisms\n- Proprietary data structures\n- Trade secrets and confidential information\n\n## 4. PERMITTED USES\n\nYou may only:\n- Use the Software as provided through the ClawhHub platform\n- Access features available in your subscription tier\n- Create test results and reports for internal use\n- Share results with your team (if on a team plan)\n- Provide feedback to improve the Software\n\n## 5. SUBSCRIPTION TIERS\n\n### Starter (Free)\n- 5 tests per month\n- 2 models per test\n- Basic features\n- Personal use only\n\n### Professional ($29/month)\n- Unlimited tests\n- All models supported\n- Advanced analytics\n- API access\n- Commercial use permitted\n\n### Enterprise ($99/month)\n- Team collaboration\n- White-label option\n- Custom integrations\n- Dedicated support\n- SLA guarantees\n\n## 6. API KEY AND CREDENTIALS\n\n- You are responsible for keeping your API keys confidential\n- Do not share your license key with others\n- One license per person/organization\n- License keys are non-transferable\n- Unauthorized sharing may result in account termination\n\n## 7. DATA PRIVACY\n\n- We do not retain your test data by default\n- Free tier: 30-day retention\n- Paid tiers: 90-day retention\n- You can request data deletion anytime\n- See Privacy Policy for full details\n\n## 8. WARRANTY DISCLAIMER\n\nTHE SOFTWARE IS PROVIDED \"AS-IS\" WITHOUT ANY WARRANTIES. LICENSOR DISCLAIMS ALL WARRANTIES, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO:\n- Merchantability\n- Fitness for a particular purpose\n- Non-infringement\n- Accuracy of results\n\n## 9. LIMITATION OF LIABILITY\n\nIN NO EVENT SHALL LICENSOR BE LIABLE FOR:\n- Any indirect, incidental, special, or consequential damages\n- Loss of data, revenue, or profits\n- Business interruption\n- Even if advised of the possibility of such damages\n\n## 10. TERMINATION\n\nLicensor may terminate your license if you:\n- Violate any terms of this agreement\n- Fail to pay subscription fees\n- Attempt to reverse engine"},{"path":"manifest.yaml","content":"name: \"Prompt Performance Tester\"\nid: \"prompt-performance-tester\"\nversion: \"1.1.8\"\ndescription: \"Model-agnostic prompt benchmarking across 9 providers. Pass any model ID from Claude, GPT, Gemini, DeepSeek, Grok, MiniMax, Qwen, Llama, Mistral — provider auto-detected. Measures latency, cost, quality, and consistency.\"\n\nhomepage: \"https://unisai.vercel.app\"\nrepository: \"https://github.com/vedantsingh60/prompt-performance-tester\"\nsource: \"included\"\n\nintellectual_property:\n  license: \"free-to-use\"\n  license_file: \"LICENSE.md\"\n  copyright: \"© 2026 UnisAI. All rights reserved.\"\n  distribution: \"via-clawhub-only\"\n  source_code_access: \"included\"\n  modification: \"personal-use-only\"\n  reverse_engineering: \"allowed-for-security-audit\"\n\nauthor:\n  company: \"UnisAI\"\n  contact: \"hello@unisai.vercel.app\"\n  website: \"https://unisai.vercel.app\"\n\ncategory: \"ai-testing\"\ntags:\n  - \"prompt-testing\"\n  - \"performance-analysis\"\n  - \"cost-optimization\"\n  - \"multi-llm\"\n  - \"quality-assurance\"\n  - \"benchmarking\"\n  - \"llm-comparison\"\n  - \"ai-testing\"\n\npricing:\n  model: \"free\"\n\nruntime: \"local\"\nexecution: \"python\"\n\nrequired_env_vars:\n  - \"ANTHROPIC_API_KEY\"   # Required if testing Claude models\n  - \"OPENAI_API_KEY\"      # Required if testing GPT models\n  - \"GOOGLE_API_KEY\"      # Required if testing Gemini models\n  - \"MISTRAL_API_KEY\"     # Required if testing Mistral models\n  - \"DEEPSEEK_API_KEY\"    # Required if testing DeepSeek models\n  - \"XAI_API_KEY\"         # Required if testing Grok/xAI models\n  - \"MINIMAX_API_KEY\"     # Required if testing MiniMax models\n  - \"DASHSCOPE_API_KEY\"   # Required if testing Qwen/Alibaba models\n  - \"OPENROUTER_API_KEY\"  # Required if testing Llama/OpenRouter models\nprimary_credential: \"At least ONE provider API key is required per provider you want to test\"\n\ndependencies:\n  python: \">=3.9\"\n  packages:\n    - \"anthropic>=0.40.0\"\n    - \"openai>=1.60.0\"\n    - \"google-generativeai>=0.8.0\"\n    - \"mistralai>=1.3.0\"\n  install_all: \"pip install anthropic openai google-generativeai mistralai\"\n  install_selective: |\n    pip install anthropic          # Claude\n    pip install openai             # GPT, DeepSeek, xAI, MiniMax, Qwen, Llama (OpenAI-compat)\n    pip install google-generativeai  # Gemini\n    pip install mistralai          # Mistral\n  note: \"Install only the SDKs for the providers you plan to test. DeepSeek, xAI, MiniMax, Qwen, and Llama all use the openai package with a custom base URL.\"\n  requirements_file: \"requirements.txt\"\n\nsecurity:\n  data_retention: \"0 days\"\n  data_flow: \"prompts-sent-to-chosen-ai-providers\"\n  third_party_data_sharing: |\n    WARNING: This skill sends your prompts to whichever AI providers you select for testing.\n    Each provider has their own data retention and privacy policies:\n    - Anthropic: https://www.anthropic.com/legal/privacy\n    - OpenAI: https://openai.com/policies/privacy-policy\n    - Google: https://ai.google.dev/gemini-api/terms\n    - Mistral: https://mistral.ai/terms/\n    - DeepSeek: https://www.deepseek"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":2048,"uniquenessScore":38,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-04-15T00:45:39.800Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T02:22:18.291Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}