{"id":"d151f050-091e-49b7-8c7d-30ef5fdd8b94","entityType":"agent","slug":"clawhub-tsag1-model-throughput-tester","name":"Model Throughput Tester","canonicalUrl":"https://www.xpersona.co/agent/clawhub-tsag1-model-throughput-tester","canonicalPath":"/agent/clawhub-tsag1-model-throughput-tester","generatedAt":"2026-10-11T21:51:31.598Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T19:49:18.188Z","emptyReason":null},"description":"Automation skill for Model Throughput Tester.","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s1759n54sysfe5jhqj320wvtzh875km6:model-throughput-tester","sourceUrl":"https://clawhub.ai/tsag1/model-throughput-tester","homepage":"https://clawhub.ai/tsag1/skills/model-throughput-tester","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/tsag1/model-throughput-tester","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/tsag1/skills/model-throughput-tester","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":60,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Model Throughput Tester technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T19:49:18.188Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T19:49:18.188Z","emptyReason":null},"stars":null,"forks":null,"downloads":1001,"likes":null,"task":null,"library":null,"packageName":null,"latestVersion":"1.0.8","tractionLabel":"1K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T19:49:18.121Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T19:49:18.188Z","lastCrawledAt":"2026-10-11T19:49:18.121Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T19:49:18.121Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.8","createdAt":"2026-07-06T03:27:36.837Z","changelog":"Exclude cold-start iteration from summary and disable thinking for pure inference throughput","fileCount":8,"zipByteSize":18209},{"version":"1.0.7","createdAt":"2026-07-02T00:24:00.127Z","changelog":"v1.0.7: throughput.py 性能优化与稳定性修复","fileCount":8,"zipByteSize":16947},{"version":"1.0.6","createdAt":"2026-06-10T12:53:34.651Z","changelog":"Added Chinese trigger words and tags for search discovery; converted all logs to English; enriched frontmatter metadata; added openclaw.requires.bins","fileCount":4,"zipByteSize":8753},{"version":"1.0.5","createdAt":"2026-06-08T03:57:35.786Z","changelog":"SKILL.md converted to English; added README.md and README.zh.md","fileCount":6,"zipByteSize":12200},{"version":"1.0.4","createdAt":"2026-06-08T00:28:22.680Z","changelog":"安全修复：收窄触发词，改为明确请求制；移除自动执行指令；同步清理 description 中的宽泛触发词","fileCount":5,"zipByteSize":9555},{"version":"1.0.3","createdAt":"2026-06-07T19:41:56.002Z","changelog":"新增触发规则：提到吞吐率/测速时默认使用Auto模式自动测试当前session模型","fileCount":5,"zipByteSize":9507},{"version":"1.0.2","createdAt":"2026-06-07T19:34:04.543Z","changelog":"新增Auto模式：无需API Key，通过openclaw infer自动测试当前模型吞吐率；默认英文prompt提高token估算精度；支持自动检测当前session模型","fileCount":5,"zipByteSize":9336},{"version":"1.0.1","createdAt":"2026-06-07T18:47:52.493Z","changelog":"优化描述，加入 benchmark、模型评测、延迟测试等搜索关键词；修复吞吐率计时 bug（urlopen 响应时间）和 reasoning_content 读取；添加 cache hit 检测；随机 prompt 后缀防缓存","fileCount":5,"zipByteSize":7633}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s1759n54sysfe5jhqj320wvtzh875km6:model-throughput-tester","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s1759n54sysfe5jhqj320wvtzh875km6:model-throughput-tester` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/tsag1/model-throughput-tester before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tsag1-model-throughput-tester/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tsag1-model-throughput-tester/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tsag1-model-throughput-tester/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-tsag1-model-throughput-tester/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-tsag1-model-throughput-tester/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-tsag1-model-throughput-tester/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T21:51:31.595Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tsag1-model-throughput-tester/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tsag1-model-throughput-tester/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tsag1-model-throughput-tester/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tsag1-model-throughput-tester/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T19:49:18.188Z","emptyReason":null},"readme":"Skill: Model Throughput Tester\n\nOwner: tsag1\n\nSummary: Automation skill for Model Throughput Tester.\n\nTags: latest:1.0.8\n\nVersion history:\n\nv1.0.8 | 2026-07-06T03:27:36.837Z | user\n\nExclude cold-start iteration from summary and disable thinking for pure inference throughput\n\nv1.0.7 | 2026-07-02T00:24:00.127Z | user\n\nv1.0.7: throughput.py 性能优化与稳定性修复\n\nv1.0.6 | 2026-06-10T12:53:34.651Z | user\n\nAdded Chinese trigger words and tags for search discovery; converted all logs to English; enriched frontmatter metadata; added openclaw.requires.bins\n\nv1.0.5 | 2026-06-08T03:57:35.786Z | user\n\nSKILL.md converted to English; added README.md and README.zh.md\n\nv1.0.4 | 2026-06-08T00:28:22.680Z | user\n\n安全修复：收窄触发词，改为明确请求制；移除自动执行指令；同步清理 description 中的宽泛触发词\n\nv1.0.3 | 2026-06-07T19:41:56.002Z | user\n\n新增触发规则：提到吞吐率/测速时默认使用Auto模式自动测试当前session模型\n\nv1.0.2 | 2026-06-07T19:34:04.543Z | user\n\n新增Auto模式：无需API Key，通过openclaw infer自动测试当前模型吞吐率；默认英文prompt提高token估算精度；支持自动检测当前session模型\n\nv1.0.1 | 2026-06-07T18:47:52.493Z | user\n\n优化描述，加入 benchmark、模型评测、延迟测试等搜索关键词；修复吞吐率计时 bug（urlopen 响应时间）和 reasoning_content 读取；添加 cache hit 检测；随机 prompt 后缀防缓存\n\nv1.0.0 | 2026-06-07T17:13:37.012Z | user\n\n初始版本：测试 OpenAI 兼容 API 吞吐率，多模型批量、Markdown+CSV 报告\n\nArchive index:\n\nArchive v1.0.8: 8 files, 18209 bytes\n\nFiles: _meta.json (142b), README.md (3085b), README.zh.md (3682b), skill-card.md (2274b), SKILL.md (7666b), tests/test_throughput.py (13332b), throughput-report.md (892b), throughput.py (18261b)\n\nFile v1.0.8:SKILL.md\n\n---\nname: model-throughput-tester\nname_zh: 吞吐率 测试 · 模型速度对比\ntags: [model-throughput-tester, 吞吐率测试, 模型速度对比]\ndescription: Benchmark LLM model throughput — measure tokens/s, latency, and output speed. Supports auto mode (no API key needed) via openclaw infer, or direct API mode for OpenAI-compatible endpoints. Trigger: throughput test, tokens/s, latency test, benchmark, speed test, model test.\ndescription_zh: AI 模型速度对比工具。一句话测出哪个模型更快、延迟更低、吞吐率更高。支持无 Key 的 auto 模式（openclaw infer）和 OpenAI 兼容 API 直连。对比多个模型的 tokens/s、响应延迟、输出速度，生成可视化报告。换模型前先跑个基线，不花冤枉钱。\ntriggerWords:\n  - 模型 哪个快\n  - 模型 速度 测试\n  - tokens/s 对比\n  - 吞吐率 测试\n  - 延迟 测试\n  - 模型 换哪个\n  - 测一下 模型 速度\n  - throughput test\n  - tokens/s\n  - speed test\n  - latency test\n  - model test\n  - 测速\n  - benchmark\nmetadata:\n  openclaw:\n    requires:\n      bins: [python3]\n    tags: [model-benchmark, throughput, tokens-per-second, latency, AI-speed, LLM, model-comparison, performance, speed-test, benchmark, 速度测试, 吞吐率, 模型对比, AI性能, 延迟测试, tokens/s, LLM测速, 模型测速]\n    permissions:\n      file:\n        read: [\"~/.openclaw/workspace/skills/model-throughput-tester/**\"]\n        write: [\"~/.openclaw/workspace/skills/model-throughput-tester/**\"]\n---\n\n# Model Throughput Tester\n\nBenchmark LLM model throughput (tokens/s). Two modes available:\n\n- **Auto Mode**: Test current model via `openclaw infer model run`, **no API key required**\n- **API Mode**: Direct call to OpenAI-compatible API, requires URL and Key\n\n## When to Use\n\n**Use when:** User explicitly requests a model throughput test.\n\n**Trigger words:**\n- throughput test, tokens/s, speed test, benchmark\n- model speed, latency test, model test\n\n**Do NOT trigger:** Broad performance discussion terms (e.g. \"model performance\", standalone \"benchmark\") should not auto-trigger execution.\n\n**Auto Mode (no API key):**\n```bash\npython3 throughput.py --auto --model \"<current session model>\"\n```\n\n## Core Features\n\n### 1. Auto Mode (No Key, Recommended)\n\n```bash\npython3 throughput.py --auto\n```\n\nTest a specific model:\n```bash\npython3 throughput.py --auto --model \"zai/glm-5-turbo\"\n```\n\n### 2. API Mode (Direct API Call)\n\n```bash\npython3 throughput.py \\\n  --url https://api.example.com/v1 \\\n  --key sk-xxx \\\n  --models gpt-4o-mini,gpt-4o\n```\n\n### 3. Common Parameters\n\n| Parameter | Default | Description |\n|-----------|---------|-------------|\n| `--iterations` | `3` | Test iterations per model |\n| `--max-tokens` | `512` | Max output tokens |\n| `--test-prompt` | English prose (summer field) | Test prompt |\n| `--timeout` | `60` | Single request timeout (seconds) |\n| `--output` | `throughput-report.md` | Output report filename |\n| `--csv` | false | Also generate CSV |\n\n## Workflow\n\n### Auto Mode Flow\n\n```\n1. Read current session model from openclaw.json (provider/model)\n2. Send test prompt via openclaw infer model run\n3. Timer: command start → output complete\n4. Estimate token count from response text (English: 0.75 word/token, Chinese: 1.5 chars/token)\n5. Calculate tokens/s\n6. Generate summary report\n```\n\n### API Mode Flow\n\n```\n1. Build /v1/chat/completions request\n2. Timer: request start → last token received\n3. Extract usage.completion_tokens from response (precise)\n4. Calculate tokens/s, error rate\n5. Generate summary report\n```\n\n### Metrics\n\n| Metric | Description |\n|--------|-------------|\n| **Tokens/s** | Throughput = Output Tokens / Elapsed Time |\n| **Avg Latency** | Average single request latency |\n| **Avg Output Tokens** | Average output token count |\n| **Error Rate** | Failed request ratio |\n\n## Output Example\n\n```markdown\n# Model Throughput Report\n**Mode:** Auto (openclaw infer)\n**Iterations:** 3\n\n## Summary\n| Model | Avg Tokens/s | Avg Latency(s) | Avg Output Tokens | Error Rate |\n|-------|-------------|----------------|-------------------|------------|\n| zai/glm-5-turbo | 57.9 | 20.6 | 979.0 | 0.0% |\n\n## Detail\n### zai/glm-5-turbo\n| Iter | Latency(s) | Output Tokens | Tokens/s | Status |\n|------|------------|--------------|---------|--------|\n| 1 | 19.5 | 950 | 48.7 | ✅ |\n| 2 | 21.3 | 1010 | 47.4 | ✅ |\n| 3 | 20.9 | 977 | 46.7 | ✅ |\n```\n\n## Error Handling\n\n| Scenario | Auto Mode | API Mode |\n|----------|-----------|----------|\n| openclaw not installed | cli_error | — |\n| Model not found | api_error | http_404 |\n| Network timeout | timeout | timeout |\n| Token estimation | English 0.75 word/token, Chinese 1.5 chars/token | Precise from API |\n\n## Usage Examples\n\n### Quick Test After Install (Auto Mode)\n\n```bash\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto --model \"<current session model>\"\n\n# Or auto-detect (may not match session override)\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto\n```\n\n### Test Multiple Models (API Mode)\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.openai.com/v1\" \\\n  --key \"sk-xxx\" \\\n  --models \"gpt-4o-mini,gpt-4o\" \\\n  --iterations 5\n```\n\n### Custom Prompt\n\n```bash\npython3 throughput.py --auto \\\n  --test-prompt \"Explain quantum computing in detail.\" \\\n  --iterations 5\n```\n\n## Technical Details\n\n- **Auto Mode**: `openclaw infer model run --json`, Python `subprocess` call\n- **API Mode**: `urllib` (Python built-in), OpenAI-compatible `/v1/chat/completions`\n- **Timer Precision**: `time.perf_counter()` nanosecond-level\n- **Token Counting**: API mode uses `usage.completion_tokens` (precise), Auto mode estimates by character count\n- **URL Handling**: Smart detection of `/v1`, `/v4`, `/chat/completions` paths\n\n## Notes\n\n- Auto mode throughput includes gateway routing overhead, slightly lower than direct API (~1-3%)\n- Auto mode token count is estimated, API mode is precise\n- English prompts recommended for more accurate token estimation\n- Anti-cache: random seed suffix appended to each iteration\n\n## Limitations\n\n**Auto mode cannot cap output length.** `openclaw infer model run` does not expose a `--max-tokens` flag (verified via `--help`), so the CLI argument `--max-tokens` is silently ignored when running with `--auto`. This means:\n\n- A slow model (e.g. <30 tokens/s) may exceed `--timeout` on long outputs and be marked `timeout` even though it is healthy.\n- The `--max-tokens` CLI arg only applies to **API mode** (direct OpenAI-compatible call).\n- To test slow models reliably with `--auto`, use the medium-length default prompt (`Write three short paragraphs about summer meadows.` — ~200-300 tokens output), raise `--timeout` (e.g. 120-180s) for slow models, or override `--test-prompt` to suit the model.\n- For precise max_tokens control and accurate `usage.completion_tokens`, use **API mode** (`--url`, `--key`, `--models`).\n\n**Warmup is enabled by default.** The first request to a model often pays a cold-start cost (KV cache init, weight load) that is not representative of steady-state throughput. A warmup request is fired and discarded before the timed iterations begin. Pass `--no-warmup` to disable (e.g. when measuring cold-start latency deliberately).\n\n## Testing\n\n```bash\npython3 tests/test_throughput.py\n```\n\n26 unit tests covering: token estimation (EN/ZH/mixed/empty), language detection threshold, provider/model parsing, URL handling (`/v1`, `/v4`, full path, trailing slash), markdown report generation (ok/timeout/cache_hit/multi-model/empty), CSV output, and config loading. **No network requests** — tests cover pure logic only.\n\nFile v1.0.8:README.md\n\n# Model Throughput Tester\n\nBenchmark LLM model throughput — measure tokens/s, latency, and output speed for any language model.\n\n## Features\n\n- **Auto Mode**: Test your current session model via `openclaw infer`, no API key needed\n- **API Mode**: Direct benchmark against any OpenAI-compatible endpoint\n- **Flexible**: Custom prompts, iteration counts, timeout controls\n- **Reports**: Markdown + CSV output with per-iteration details\n\n## Quick Start\n\n```bash\n# Auto mode — test current session model\npython3 throughput.py --auto\n\n# Test a specific model\npython3 throughput.py --auto --model \"gpt-4o\"\n\n# API mode — test against an endpoint\npython3 throughput.py \\\n  --url \"https://api.openai.com/v1\" \\\n  --key \"sk-xxx\" \\\n  --models \"gpt-4o-mini,gpt-4o\" \\\n  --iterations 5\n```\n\n## Parameters\n\n| Parameter | Default | Description |\n|-----------|---------|-------------|\n| `--auto` | off | Enable auto mode (uses openclaw infer) |\n| `--model` | auto-detect | Model identifier |\n| `--url` | — | API base URL (API mode) |\n| `--key` | — | API key (API mode) |\n| `--models` | — | Comma-separated model list (API mode) |\n| `--iterations` | `3` | Test iterations per model |\n| `--max-tokens` | `512` | Max output tokens |\n| `--test-prompt` | built-in | Custom test prompt |\n| `--timeout` | `60` | Request timeout (seconds) |\n| `--output` | `throughput-report.md` | Output report filename |\n| `--csv` | false | Also generate CSV output |\n\n## Metrics\n\n| Metric | Description |\n|--------|-------------|\n| **Tokens/s** | Throughput = Output Tokens / Elapsed Time |\n| **Avg Latency** | Average single-request latency |\n| **Avg Output Tokens** | Average output token count |\n| **Error Rate** | Failed request ratio |\n\n## Example Output\n\n```\n📊 Model Throughput Report\nMode: Auto (openclaw infer) | Iterations: 3\n\nSummary\n| Model             | Avg Tokens/s | Latency(s) | Output Tokens | Error |\n|-------------------|-------------|------------|----------------|-------|\n| zai/glm-5-turbo   | 57.9        | 20.6       | 979            | 0.0%  |\n```\n\n## How It Works\n\n**Auto Mode**: Sends a test prompt via `openclaw infer model run`, measures wall-clock time from start to last token, then estimates token count from output text.\n\n**API Mode**: Calls `/v1/chat/completions` with streaming disabled, reads `usage.completion_tokens` for precise token counts.\n\n## Notes\n\n- Auto mode throughput includes gateway routing overhead (~1-3% lower than direct API)\n- Auto mode token counts are estimates; API mode uses precise values\n- English prompts yield more accurate token estimates in auto mode\n- Anti-cache: random seed suffix appended per iteration\n\n## Prerequisites\n\n- Python 3 (built-in on macOS)\n- `openclaw` CLI (for auto mode)\n\n## File Structure\n\n```\n~/.openclaw/workspace/skills/model-throughput-tester/\n├── SKILL.md           # Agent trigger & execution guide\n├── README.md          # This file\n├── README.zh.md       # Chinese version\n├── throughput.py      # Main script\n└── throughput-report.md  # Generated reports (on demand)\n```\n\n## License\n\nMIT-0\n\nFile v1.0.8:_meta.json\n\n{\n  \"ownerId\": \"kn7dzk85gz4pz5c569zky2pssn875nwe\",\n  \"slug\": \"model-throughput-tester\",\n  \"version\": \"1.0.8\",\n  \"publishedAt\": 1783308456837\n}\n\nFile v1.0.8:README.zh.md\n\n# 模型吞吐率测试器\n\n测试 LLM 模型的吞吐率（tokens/s）。支持两种模式：\n\n- **Auto 模式**：通过 `openclaw infer model run` 测试当前模型，**无需 API Key**\n- **API 模式**：直接调用 OpenAI 兼容 API，需要 URL 和 Key\n\n## 触发规则\n\n**适用场景：** 用户明确要求测试模型吞吐率时使用。\n\n**推荐触发词：**\n- 测一下吞吐率、测速、模型测速、tokens/s\n- 跑个 benchmark、吞吐率测试、模型测试\n\n**不适用：** 宽泛的性能讨论词（如「模型性能」「benchmark」单独出现）不应自动触发执行。\n\n## 核心能力\n\n### 1. Auto 模式（无 Key，推荐）\n\n自动检测当前 session 的模型并测试吞吐率，无需任何配置。\n\n```bash\npython3 throughput.py --auto\n```\n\n指定模型测试：\n```bash\npython3 throughput.py --auto --model \"zai/glm-5-turbo\"\n```\n\n### 2. API 模式（直接调用 API）\n\n```bash\npython3 throughput.py \\\n  --url https://api.example.com/v1 \\\n  --key sk-xxx \\\n  --models gpt-4o-mini,gpt-4o\n```\n\n### 3. 通用参数\n\n| 参数 | 默认值 | 说明 |\n|------|--------|------|\n| `--iterations` | `3` | 每个模型测试次数 |\n| `--max-tokens` | `512` | 最大输出 token 数 |\n| `--test-prompt` | 英文散文（夏天的田野） | 测试提示词 |\n| `--timeout` | `60` | 单次请求超时（秒） |\n| `--output` | `throughput-report.md` | 输出报告文件名 |\n| `--csv` | false | 同时生成 CSV |\n\n## 工作流程\n\n### Auto 模式\n\n```\n1. 从 openclaw.json 读取当前 session 模型（provider/model）\n2. 通过 openclaw infer model run 发送测试 prompt\n3. 计时：命令开始 → 输出完成\n4. 从返回文本估算 token 数（英文 0.75 word/token，中文 1.5 字/token）\n5. 计算 tokens/s\n6. 汇总输出报告\n```\n\n### API 模式\n\n```\n1. 构造 /v1/chat/completions 请求\n2. 计时：请求开始 → 最后一个 token\n3. 从响应中提取 usage.completion_tokens（精确）\n4. 计算 tokens/s、错误率\n5. 汇总输出报告\n```\n\n## 指标说明\n\n| 指标 | 说明 |\n|------|------|\n| **Tokens/s** | 吞吐率 = Output Tokens / Elapsed Time |\n| **Avg Latency** | 平均单次请求延迟 |\n| **Avg Output Tokens** | 平均输出 token 数 |\n| **Error Rate** | 错误请求占比 |\n\n## 使用示例\n\n### 安装后立即测试（Auto 模式）\n\n```bash\n# agent 触发时应传入当前模型\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto --model \"<当前session模型>\"\n\n# 或使用自动检测（可能不是 session 覆盖的模型）\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto\n```\n\n### 测试多个模型（API 模式）\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.openai.com/v1\" \\\n  --key \"sk-xxx\" \\\n  --models \"gpt-4o-mini,gpt-4o\" \\\n  --iterations 5\n```\n\n### 自定义提示词\n\n```bash\npython3 throughput.py --auto \\\n  --test-prompt \"Explain quantum computing in detail.\" \\\n  --iterations 5\n```\n\n## 技术实现\n\n- **Auto 模式**：`openclaw infer model run --json`，Python `subprocess` 调用\n- **API 模式**：`urllib`（Python 内置），OpenAI 兼容 `/v1/chat/completions`\n- **计时精度**：`time.perf_counter()` 纳秒级精度\n- **Token 计数**：API 模式优先 `usage.completion_tokens`（精确），Auto 模式按字符估算\n- **URL 拼接**：智能检测 `/v1`、`/v4`、`/chat/completions` 路径\n\n## 注意事项\n\n- Auto 模式的吞吐率包含网关路由开销，会比直接 API 略低（约 1-3%）\n- Auto 模式 Token 数为估算值，API 模式为精确值\n- 建议使用英文 prompt 以获得更准确的 token 估算\n- 防缓存：每次迭代自动附加随机 seed 后缀\n\nFile v1.0.8:skill-card.md\n\n## Description:\n\nBenchmark LLM model throughput by measuring tokens per second, latency, and output speed through auto mode with OpenClaw or direct API mode for OpenAI-compatible endpoints.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[tsag1](https://clawhub.ai/user/tsag1)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and engineers use this skill to benchmark and compare LLM response throughput, latency, token output, and error rate before selecting or switching models.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: API mode can send API keys and benchmark prompts to any configured endpoint, including insecure HTTP endpoints.\n\nMitigation: Prefer auto mode when possible; in API mode, use trusted HTTPS endpoints and avoid placing real keys directly in shell history.\n\nRisk: Generated reports can store the benchmark prompt and API URL locally.\n\nMitigation: Use non-sensitive benchmark prompts and review generated Markdown or CSV reports before sharing them.\n\nRisk: Auto mode estimates token counts and cannot enforce the max token limit, which can make slow models appear to fail by timeout.\n\nMitigation: Use API mode for precise completion-token counts and max-token control, or raise the timeout and use a shorter prompt for slow models in auto mode.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/tsag1/skills/model-throughput-tester)\n- [README](artifact/README.md)\n- [Skill definition](artifact/SKILL.md)\n- [Release evidence](evidence.json)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, shell commands, configuration]\n\n**Output Format:** [Terminal summary text plus a Markdown throughput report, with optional CSV output.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Auto mode estimates token counts and excludes warmup from summary by default; API mode uses reported completion tokens when available.]\n\n## Skill Version(s):\n\n1.0.8 (source: server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.0.8:throughput-report.md\n\n# Model Throughput Report\n**Generated:** 2026-07-06 09:19:44\n**Mode:** Auto (openclaw infer)\n**API URL:** http://192.10.27.211:54000/v1\n**Iterations:** 3\n**Test Prompt:** Write three short paragraphs about summer meadows. No headings, no paragraph breaks, continuous prose only.\n**Token Note:** Estimated (EN ~0.75 word/token, ZH ~1.5 chars/token)\n**Note:** Summary excludes 1st iteration (cold-start removal)\n\n## Summary\n| Model | Avg Tokens/s | Avg Latency(s) | Avg Output Tokens | Error Rate |\n|-------|-------------|----------------|-------------------|------------|\n| gpt/006463-ALL55 | 0.0 | 0.000 | 0.0 | 100.0% |\n\n## Detail\n\n### gpt/006463-ALL55\n| Iter | Latency(s) | Output Tokens | Tokens/s | Status |\n|------|------------|--------------|---------|--------|\n| 1 | 36.713 | 0 | 0.0 | ❌ cli_error |\n| 2 | 36.501 | 0 | 0.0 | ❌ cli_error |\n| 3 | 36.722 | 0 | 0.0 | ❌ cli_error |\n\nArchive v1.0.7: 8 files, 16947 bytes\n\nFiles: _meta.json (142b), README.md (3085b), README.zh.md (3682b), skill-card.md (2324b), SKILL.md (6484b), tests/test_throughput.py (13332b), throughput-report.md (663b), throughput.py (16394b)\n\nFile v1.0.7:SKILL.md\n\n---\nname: model-throughput-tester\nname_zh: 吞吐率 测试 · 模型速度对比\ntags: [model-throughput-tester, 吞吐率测试, 模型速度对比]\ndescription: Benchmark LLM model throughput — measure tokens/s, latency, and output speed. Supports auto mode (no API key needed) via openclaw infer, or direct API mode for OpenAI-compatible endpoints. Trigger: throughput test, tokens/s, latency test, benchmark, speed test, model test.\ndescription_zh: AI 模型速度对比工具。一句话测出哪个模型更快、延迟更低、吞吐率更高。支持无 Key 的 auto 模式（openclaw infer）和 OpenAI 兼容 API 直连。对比多个模型的 tokens/s、响应延迟、输出速度，生成可视化报告。换模型前先跑个基线，不花冤枉钱。\ntriggerWords:\n  - 模型 哪个快\n  - 模型 速度 测试\n  - tokens/s 对比\n  - 吞吐率 测试\n  - 延迟 测试\n  - 模型 换哪个\n  - 测一下 模型 速度\n  - throughput test\n  - tokens/s\n  - speed test\n  - latency test\n  - model test\n  - 测速\n  - benchmark\nmetadata:\n  openclaw:\n    requires:\n      bins: [python3]\n    tags: [model-benchmark, throughput, tokens-per-second, latency, AI-speed, LLM, model-comparison, performance, speed-test, benchmark, 速度测试, 吞吐率, 模型对比, AI性能, 延迟测试, tokens/s, LLM测速, 模型测速]\n    permissions:\n      file:\n        read: [\"~/.openclaw/workspace/skills/model-throughput-tester/**\"]\n        write: [\"~/.openclaw/workspace/skills/model-throughput-tester/**\"]\n---\n\n# Model Throughput Tester\n\nBenchmark LLM model throughput (tokens/s). Two modes available:\n\n- **Auto Mode**: Test current model via `openclaw infer model run`, **no API key required**\n- **API Mode**: Direct call to OpenAI-compatible API, requires URL and Key\n\n## When to Use\n\n**Use when:** User explicitly requests a model throughput test.\n\n**Trigger words:**\n- throughput test, tokens/s, speed test, benchmark\n- model speed, latency test, model test\n\n**Do NOT trigger:** Broad performance discussion terms (e.g. \"model performance\", standalone \"benchmark\") should not auto-trigger execution.\n\n**Auto Mode (no API key):**\n```bash\npython3 throughput.py --auto --model \"<current session model>\"\n```\n\n## Core Features\n\n### 1. Auto Mode (No Key, Recommended)\n\n```bash\npython3 throughput.py --auto\n```\n\nTest a specific model:\n```bash\npython3 throughput.py --auto --model \"zai/glm-5-turbo\"\n```\n\n### 2. API Mode (Direct API Call)\n\n```bash\npython3 throughput.py \\\n  --url https://api.example.com/v1 \\\n  --key sk-xxx \\\n  --models gpt-4o-mini,gpt-4o\n```\n\n### 3. Common Parameters\n\n| Parameter | Default | Description |\n|-----------|---------|-------------|\n| `--iterations` | `3` | Test iterations per model |\n| `--max-tokens` | `512` | Max output tokens |\n| `--test-prompt` | English prose (summer field) | Test prompt |\n| `--timeout` | `60` | Single request timeout (seconds) |\n| `--output` | `throughput-report.md` | Output report filename |\n| `--csv` | false | Also generate CSV |\n\n## Workflow\n\n### Auto Mode Flow\n\n```\n1. Read current session model from openclaw.json (provider/model)\n2. Send test prompt via openclaw infer model run\n3. Timer: command start → output complete\n4. Estimate token count from response text (English: 0.75 word/token, Chinese: 1.5 chars/token)\n5. Calculate tokens/s\n6. Generate summary report\n```\n\n### API Mode Flow\n\n```\n1. Build /v1/chat/completions request\n2. Timer: request start → last token received\n3. Extract usage.completion_tokens from response (precise)\n4. Calculate tokens/s, error rate\n5. Generate summary report\n```\n\n### Metrics\n\n| Metric | Description |\n|--------|-------------|\n| **Tokens/s** | Throughput = Output Tokens / Elapsed Time |\n| **Avg Latency** | Average single request latency |\n| **Avg Output Tokens** | Average output token count |\n| **Error Rate** | Failed request ratio |\n\n## Output Example\n\n```markdown\n# Model Throughput Report\n**Mode:** Auto (openclaw infer)\n**Iterations:** 3\n\n## Summary\n| Model | Avg Tokens/s | Avg Latency(s) | Avg Output Tokens | Error Rate |\n|-------|-------------|----------------|-------------------|------------|\n| zai/glm-5-turbo | 57.9 | 20.6 | 979.0 | 0.0% |\n\n## Detail\n### zai/glm-5-turbo\n| Iter | Latency(s) | Output Tokens | Tokens/s | Status |\n|------|------------|--------------|---------|--------|\n| 1 | 19.5 | 950 | 48.7 | ✅ |\n| 2 | 21.3 | 1010 | 47.4 | ✅ |\n| 3 | 20.9 | 977 | 46.7 | ✅ |\n```\n\n## Error Handling\n\n| Scenario | Auto Mode | API Mode |\n|----------|-----------|----------|\n| openclaw not installed | cli_error | — |\n| Model not found | api_error | http_404 |\n| Network timeout | timeout | timeout |\n| Token estimation | English 0.75 word/token, Chinese 1.5 chars/token | Precise from API |\n\n## Usage Examples\n\n### Quick Test After Install (Auto Mode)\n\n```bash\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto --model \"<current session model>\"\n\n# Or auto-detect (may not match session override)\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto\n```\n\n### Test Multiple Models (API Mode)\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.openai.com/v1\" \\\n  --key \"sk-xxx\" \\\n  --models \"gpt-4o-mini,gpt-4o\" \\\n  --iterations 5\n```\n\n### Custom Prompt\n\n```bash\npython3 throughput.py --auto \\\n  --test-prompt \"Explain quantum computing in detail.\" \\\n  --iterations 5\n```\n\n## Technical Details\n\n- **Auto Mode**: `openclaw infer model run --json`, Python `subprocess` call\n- **API Mode**: `urllib` (Python built-in), OpenAI-compatible `/v1/chat/completions`\n- **Timer Precision**: `time.perf_counter()` nanosecond-level\n- **Token Counting**: API mode uses `usage.completion_tokens` (precise), Auto mode estimates by character count\n- **URL Handling**: Smart detection of `/v1`, `/v4`, `/chat/completions` paths\n\n## Notes\n\n- Auto mode throughput includes gateway routing overhead, slightly lower than direct API (~1-3%)\n- Auto mode token count is estimated, API mode is precise\n- English prompts recommended for more accurate token estimation\n- Anti-cache: random seed suffix appended to each iteration\n\n## Testing\n\n```bash\npython3 tests/test_throughput.py\n```\n\n26 unit tests covering: token estimation (EN/ZH/mixed/empty), language detection threshold, provider/model parsing, URL handling (`/v1`, `/v4`, full path, trailing slash), markdown report generation (ok/timeout/cache_hit/multi-model/empty), CSV output, and config loading. **No network requests** — tests cover pure logic only.\n\nFile v1.0.7:README.md\n\n# Model Throughput Tester\n\nBenchmark LLM model throughput — measure tokens/s, latency, and output speed for any language model.\n\n## Features\n\n- **Auto Mode**: Test your current session model via `openclaw infer`, no API key needed\n- **API Mode**: Direct benchmark against any OpenAI-compatible endpoint\n- **Flexible**: Custom prompts, iteration counts, timeout controls\n- **Reports**: Markdown + CSV output with per-iteration details\n\n## Quick Start\n\n```bash\n# Auto mode — test current session model\npython3 throughput.py --auto\n\n# Test a specific model\npython3 throughput.py --auto --model \"gpt-4o\"\n\n# API mode — test against an endpoint\npython3 throughput.py \\\n  --url \"https://api.openai.com/v1\" \\\n  --key \"sk-xxx\" \\\n  --models \"gpt-4o-mini,gpt-4o\" \\\n  --iterations 5\n```\n\n## Parameters\n\n| Parameter | Default | Description |\n|-----------|---------|-------------|\n| `--auto` | off | Enable auto mode (uses openclaw infer) |\n| `--model` | auto-detect | Model identifier |\n| `--url` | — | API base URL (API mode) |\n| `--key` | — | API key (API mode) |\n| `--models` | — | Comma-separated model list (API mode) |\n| `--iterations` | `3` | Test iterations per model |\n| `--max-tokens` | `512` | Max output tokens |\n| `--test-prompt` | built-in | Custom test prompt |\n| `--timeout` | `60` | Request timeout (seconds) |\n| `--output` | `throughput-report.md` | Output report filename |\n| `--csv` | false | Also generate CSV output |\n\n## Metrics\n\n| Metric | Description |\n|--------|-------------|\n| **Tokens/s** | Throughput = Output Tokens / Elapsed Time |\n| **Avg Latency** | Average single-request latency |\n| **Avg Output Tokens** | Average output token count |\n| **Error Rate** | Failed request ratio |\n\n## Example Output\n\n```\n📊 Model Throughput Report\nMode: Auto (openclaw infer) | Iterations: 3\n\nSummary\n| Model             | Avg Tokens/s | Latency(s) | Output Tokens | Error |\n|-------------------|-------------|------------|----------------|-------|\n| zai/glm-5-turbo   | 57.9        | 20.6       | 979            | 0.0%  |\n```\n\n## How It Works\n\n**Auto Mode**: Sends a test prompt via `openclaw infer model run`, measures wall-clock time from start to last token, then estimates token count from output text.\n\n**API Mode**: Calls `/v1/chat/completions` with streaming disabled, reads `usage.completion_tokens` for precise token counts.\n\n## Notes\n\n- Auto mode throughput includes gateway routing overhead (~1-3% lower than direct API)\n- Auto mode token counts are estimates; API mode uses precise values\n- English prompts yield more accurate token estimates in auto mode\n- Anti-cache: random seed suffix appended per iteration\n\n## Prerequisites\n\n- Python 3 (built-in on macOS)\n- `openclaw` CLI (for auto mode)\n\n## File Structure\n\n```\n~/.openclaw/workspace/skills/model-throughput-tester/\n├── SKILL.md           # Agent trigger & execution guide\n├── README.md          # This file\n├── README.zh.md       # Chinese version\n├── throughput.py      # Main script\n└── throughput-report.md  # Generated reports (on demand)\n```\n\n## License\n\nMIT-0\n\nFile v1.0.7:_meta.json\n\n{\n  \"ownerId\": \"kn7dzk85gz4pz5c569zky2pssn875nwe\",\n  \"slug\": \"model-throughput-tester\",\n  \"version\": \"1.0.7\",\n  \"publishedAt\": 1782951840127\n}\n\nFile v1.0.7:README.zh.md\n\n# 模型吞吐率测试器\n\n测试 LLM 模型的吞吐率（tokens/s）。支持两种模式：\n\n- **Auto 模式**：通过 `openclaw infer model run` 测试当前模型，**无需 API Key**\n- **API 模式**：直接调用 OpenAI 兼容 API，需要 URL 和 Key\n\n## 触发规则\n\n**适用场景：** 用户明确要求测试模型吞吐率时使用。\n\n**推荐触发词：**\n- 测一下吞吐率、测速、模型测速、tokens/s\n- 跑个 benchmark、吞吐率测试、模型测试\n\n**不适用：** 宽泛的性能讨论词（如「模型性能」「benchmark」单独出现）不应自动触发执行。\n\n## 核心能力\n\n### 1. Auto 模式（无 Key，推荐）\n\n自动检测当前 session 的模型并测试吞吐率，无需任何配置。\n\n```bash\npython3 throughput.py --auto\n```\n\n指定模型测试：\n```bash\npython3 throughput.py --auto --model \"zai/glm-5-turbo\"\n```\n\n### 2. API 模式（直接调用 API）\n\n```bash\npython3 throughput.py \\\n  --url https://api.example.com/v1 \\\n  --key sk-xxx \\\n  --models gpt-4o-mini,gpt-4o\n```\n\n### 3. 通用参数\n\n| 参数 | 默认值 | 说明 |\n|------|--------|------|\n| `--iterations` | `3` | 每个模型测试次数 |\n| `--max-tokens` | `512` | 最大输出 token 数 |\n| `--test-prompt` | 英文散文（夏天的田野） | 测试提示词 |\n| `--timeout` | `60` | 单次请求超时（秒） |\n| `--output` | `throughput-report.md` | 输出报告文件名 |\n| `--csv` | false | 同时生成 CSV |\n\n## 工作流程\n\n### Auto 模式\n\n```\n1. 从 openclaw.json 读取当前 session 模型（provider/model）\n2. 通过 openclaw infer model run 发送测试 prompt\n3. 计时：命令开始 → 输出完成\n4. 从返回文本估算 token 数（英文 0.75 word/token，中文 1.5 字/token）\n5. 计算 tokens/s\n6. 汇总输出报告\n```\n\n### API 模式\n\n```\n1. 构造 /v1/chat/completions 请求\n2. 计时：请求开始 → 最后一个 token\n3. 从响应中提取 usage.completion_tokens（精确）\n4. 计算 tokens/s、错误率\n5. 汇总输出报告\n```\n\n## 指标说明\n\n| 指标 | 说明 |\n|------|------|\n| **Tokens/s** | 吞吐率 = Output Tokens / Elapsed Time |\n| **Avg Latency** | 平均单次请求延迟 |\n| **Avg Output Tokens** | 平均输出 token 数 |\n| **Error Rate** | 错误请求占比 |\n\n## 使用示例\n\n### 安装后立即测试（Auto 模式）\n\n```bash\n# agent 触发时应传入当前模型\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto --model \"<当前session模型>\"\n\n# 或使用自动检测（可能不是 session 覆盖的模型）\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto\n```\n\n### 测试多个模型（API 模式）\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.openai.com/v1\" \\\n  --key \"sk-xxx\" \\\n  --models \"gpt-4o-mini,gpt-4o\" \\\n  --iterations 5\n```\n\n### 自定义提示词\n\n```bash\npython3 throughput.py --auto \\\n  --test-prompt \"Explain quantum computing in detail.\" \\\n  --iterations 5\n```\n\n## 技术实现\n\n- **Auto 模式**：`openclaw infer model run --json`，Python `subprocess` 调用\n- **API 模式**：`urllib`（Python 内置），OpenAI 兼容 `/v1/chat/completions`\n- **计时精度**：`time.perf_counter()` 纳秒级精度\n- **Token 计数**：API 模式优先 `usage.completion_tokens`（精确），Auto 模式按字符估算\n- **URL 拼接**：智能检测 `/v1`、`/v4`、`/chat/completions` 路径\n\n## 注意事项\n\n- Auto 模式的吞吐率包含网关路由开销，会比直接 API 略低（约 1-3%）\n- Auto 模式 Token 数为估算值，API 模式为精确值\n- 建议使用英文 prompt 以获得更准确的 token 估算\n- 防缓存：每次迭代自动附加随机 seed 后缀\n\nFile v1.0.7:skill-card.md\n\n## Description: <br>\nBenchmark LLM model throughput by measuring tokens per second, latency, output tokens, and error rate through auto mode or OpenAI-compatible API mode. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[tsag1](https://clawhub.ai/user/tsag1) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and engineers use this skill to benchmark LLM throughput and compare latency, tokens per second, output tokens, and error rate before selecting or changing models. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Benchmark prompts may be sent to the selected model provider and may spend tokens during repeated tests. <br>\nMitigation: Use non-sensitive test prompts, keep iteration and max-token settings appropriate, and confirm the target provider before running a benchmark. <br>\nRisk: API mode requires a provider URL and API key, and command-line keys may be exposed through shell history or process listings. <br>\nMitigation: Use trusted HTTPS API URLs and avoid placing real API keys directly in shell commands when possible. <br>\nRisk: Generated reports can include prompt text, provider details, model names, and benchmark results. <br>\nMitigation: Review generated Markdown or CSV reports before sharing them outside the intended audience. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/tsag1/skills/model-throughput-tester) <br>\n- [README](README.md) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, shell commands, configuration] <br>\n**Output Format:** [Console text plus a Markdown throughput report, with optional CSV output] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Reports include measured latency, output tokens, tokens per second, status, and error rate; auto mode estimates token counts while API mode uses provider usage data when available.] <br>\n\n## Skill Version(s): <br>\n1.0.7 (source: server release metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.0.7:throughput-report.md\n\n# Model Throughput Report\n**Generated:** 2026-06-23 23:16:10\n**Mode:** Auto (openclaw infer)\n**API URL:** https://open.bigmodel.cn/api/coding/paas/v4\n**Iterations:** 1\n**Test Prompt:** Say hello.\n**Token Note:** Estimated (EN ~0.75 word/token, ZH ~1.5 chars/token)\n\n## Summary\n| Model | Avg Tokens/s | Avg Latency(s) | Avg Output Tokens | Error Rate |\n|-------|-------------|----------------|-------------------|------------|\n| zai/glm-5-turbo | 0.0 | 0.000 | 0.0 | 100.0% |\n\n## Detail\n\n### zai/glm-5-turbo\n| Iter | Latency(s) | Output Tokens | Tokens/s | Status |\n|------|------------|--------------|---------|--------|\n| 1 | 4.736 | 0 | 0.0 | ❌ json_error |\n\nArchive v1.0.6: 4 files, 8753 bytes\n\nFiles: skill-card.md (1962b), SKILL.md (6390b), throughput.py (15585b), _meta.json (142b)\n\nFile v1.0.6:SKILL.md\n\n---\nname: model-throughput-tester\nname_zh: 吞吐率 测试 · 模型速度对比\ntags: [model-throughput-tester, 吞吐率测试, 模型速度对比]\ndescription: Benchmark LLM model throughput — measure tokens/s, latency, and output speed. Supports auto mode (no API key needed) via openclaw infer, or direct API mode for OpenAI-compatible endpoints. Trigger: throughput test, tokens/s, latency test, benchmark, speed test, model test.\ndescription_zh: AI 模型速度对比工具。一句话测出哪个模型更快、延迟更低、吞吐率更高。支持无 Key 的 auto 模式（openclaw infer）和 OpenAI 兼容 API 直连。对比多个模型的 tokens/s、响应延迟、输出速度，生成可视化报告。换模型前先跑个基线，不花冤枉钱。\ntriggerWords:\n  - 模型 哪个快\n  - 模型 速度 测试\n  - 模型 性能 对比\n  - AI 速度 测评\n  - LLM 速度 测试\n  - token 速度\n  - tokens/s 对比\n  - 吞吐率 测试\n  - 延迟 测试\n  - 响应 速度 对比\n  - 模型 延迟 排行\n  - DeepSeek GPT 哪个快\n  - AI 模型 基线\n  - 模型 换哪个\n  - 测一下 模型 速度\n  - 跑个 测试\n  - throughput test\n  - tokens/s\n  - speed test\n  - latency test\n  - model test\n  - 测速\n  - benchmark\nmetadata:\n  openclaw:\n    requires:\n      bins: [python3]\n    tags: [model-benchmark, throughput, tokens-per-second, latency, AI-speed, LLM, model-comparison, performance, speed-test, benchmark, 速度测试, 吞吐率, 模型对比, AI性能, 延迟测试, tokens/s, LLM测速, 模型测速]\n    permissions:\n      file:\n        read: [\"~/.openclaw/workspace/skills/model-throughput-tester/**\"]\n        write: [\"~/.openclaw/workspace/skills/model-throughput-tester/**\"]\n---\n\n# Model Throughput Tester\n\nBenchmark LLM model throughput (tokens/s). Two modes available:\n\n- **Auto Mode**: Test current model via `openclaw infer model run`, **no API key required**\n- **API Mode**: Direct call to OpenAI-compatible API, requires URL and Key\n\n## When to Use\n\n**Use when:** User explicitly requests a model throughput test.\n\n**Trigger words:**\n- throughput test, tokens/s, speed test, benchmark\n- model speed, latency test, model test\n\n**Do NOT trigger:** Broad performance discussion terms (e.g. \"model performance\", standalone \"benchmark\") should not auto-trigger execution.\n\n**Auto Mode (no API key):**\n```bash\npython3 throughput.py --auto --model \"<current session model>\"\n```\n\n## Core Features\n\n### 1. Auto Mode (No Key, Recommended)\n\nAuto-detects the current session model and benchmarks throughput, zero configuration needed.\n\n```bash\npython3 throughput.py --auto\n```\n\nTest a specific model:\n```bash\npython3 throughput.py --auto --model \"zai/glm-5-turbo\"\n```\n\n### 2. API Mode (Direct API Call)\n\n```bash\npython3 throughput.py \\\n  --url https://api.example.com/v1 \\\n  --key sk-xxx \\\n  --models gpt-4o-mini,gpt-4o\n```\n\n### 3. Common Parameters\n\n| Parameter | Default | Description |\n|-----------|---------|-------------|\n| `--iterations` | `3` | Test iterations per model |\n| `--max-tokens` | `512` | Max output tokens |\n| `--test-prompt` | English prose (summer field) | Test prompt |\n| `--timeout` | `60` | Single request timeout (seconds) |\n| `--output` | `throughput-report.md` | Output report filename |\n| `--csv` | false | Also generate CSV |\n\n## Workflow\n\n### Auto Mode Flow\n\n```\n1. Read current session model from openclaw.json (provider/model)\n2. Send test prompt via openclaw infer model run\n3. Timer: command start → output complete\n4. Estimate token count from response text (English: 0.75 word/token, Chinese: 1.5 chars/token)\n5. Calculate tokens/s\n6. Generate summary report\n```\n\n### API Mode Flow\n\n```\n1. Build /v1/chat/completions request\n2. Timer: request start → last token received\n3. Extract usage.completion_tokens from response (precise)\n4. Calculate tokens/s, error rate\n5. Generate summary report\n```\n\n### Metrics\n\n| Metric | Description |\n|--------|-------------|\n| **Tokens/s** | Throughput = Output Tokens / Elapsed Time |\n| **Avg Latency** | Average single request latency |\n| **Avg Output Tokens** | Average output token count |\n| **Error Rate** | Failed request ratio |\n\n## Output Example\n\n```markdown\n# Model Throughput Report\n**Mode:** Auto (openclaw infer)\n**Iterations:** 3\n\n## Summary\n| Model | Avg Tokens/s | Avg Latency(s) | Avg Output Tokens | Error Rate |\n|-------|-------------|----------------|-------------------|------------|\n| zai/glm-5-turbo | 57.9 | 20.6 | 979.0 | 0.0% |\n\n## Detail\n### zai/glm-5-turbo\n| Iter | Latency(s) | Output Tokens | Tokens/s | Status |\n|------|------------|--------------|---------|--------|\n| 1 | 19.5 | 950 | 48.7 | ✅ |\n| 2 | 21.3 | 1010 | 47.4 | ✅ |\n| 3 | 20.9 | 977 | 46.7 | ✅ |\n```\n\n## Error Handling\n\n| Scenario | Auto Mode | API Mode |\n|----------|-----------|----------|\n| openclaw not installed | cli_error | — |\n| Model not found | api_error | http_404 |\n| Network timeout | timeout | timeout |\n| Token estimation | English 0.75 word/token, Chinese 1.5 chars/token | Precise from API |\n\n## Usage Examples\n\n### Quick Test After Install (Auto Mode)\n\n```bash\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto --model \"<current session model>\"\n\n# Or auto-detect (may not match session override)\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto\n```\n\n### Test Multiple Models (API Mode)\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.openai.com/v1\" \\\n  --key \"sk-xxx\" \\\n  --models \"gpt-4o-mini,gpt-4o\" \\\n  --iterations 5\n```\n\n### Custom Prompt\n\n```bash\npython3 throughput.py --auto \\\n  --test-prompt \"Explain quantum computing in detail.\" \\\n  --iterations 5\n```\n\n## Technical Details\n\n- **Auto Mode**: `openclaw infer model run --json`, Python `subprocess` call\n- **API Mode**: `urllib` (Python built-in), OpenAI-compatible `/v1/chat/completions`\n- **Timer Precision**: `time.perf_counter()` nanosecond-level\n- **Token Counting**: API mode uses `usage.completion_tokens` (precise), Auto mode estimates by character count\n- **URL Handling**: Smart detection of `/v1`, `/v4`, `/chat/completions` paths\n\n## Notes\n\n- Auto mode throughput includes gateway routing overhead, slightly lower than direct API (~1-3%)\n- Auto mode token count is estimated, API mode is precise\n- English prompts recommended for more accurate token estimation\n- Anti-cache: random seed suffix appended to each iteration\n\nFile v1.0.6:_meta.json\n\n{\n  \"ownerId\": \"kn7dzk85gz4pz5c569zky2pssn875nwe\",\n  \"slug\": \"model-throughput-tester\",\n  \"version\": \"1.0.6\",\n  \"publishedAt\": 1781096014651\n}\n\nFile v1.0.6:skill-card.md\n\n## Description: <br>\nBenchmarks LLM throughput by measuring tokens per second, latency, output speed, and error rate in OpenClaw auto mode or OpenAI-compatible API mode. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[tsag1](https://clawhub.ai/user/tsag1) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and AI practitioners use this skill to compare model throughput and latency before selecting or changing LLM endpoints. It supports quick OpenClaw auto-mode tests and direct OpenAI-compatible API benchmarks across one or more models. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: API mode can send prompts to a user-provided endpoint with a user-provided API key. <br>\nMitigation: Use trusted endpoints, use appropriately scoped keys, and prefer auto mode when an API key is not needed. <br>\nRisk: Generated reports can include the model, API URL, test prompt, and benchmark results. <br>\nMitigation: Use non-sensitive test prompts and review or redact reports before sharing them. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/tsag1/model-throughput-tester) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, shell commands] <br>\n**Output Format:** [Markdown benchmark report with terminal status output and optional CSV] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Reports include model, API URL when supplied, test prompt, latency, output token counts or estimates, tokens per second, and error rate.] <br>\n\n## Skill Version(s): <br>\n1.0.6 (source: server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.0.5: 6 files, 12200 bytes\n\nFiles: README.md (3085b), README.zh.md (3682b), skill-card.md (2410b), SKILL.md (5000b), throughput.py (16224b), _meta.json (142b)\n\nFile v1.0.5:SKILL.md\n\n---\nname: model_throughput_tester\ndescription: Benchmark LLM model throughput — measure tokens/s, latency, and output speed. Supports auto mode (no API key needed) via openclaw infer, or direct API mode for OpenAI-compatible endpoints. Trigger: throughput test, tokens/s, latency test, benchmark, speed test, model test.\n---\n\n# Model Throughput Tester\n\nBenchmark LLM model throughput (tokens/s). Two modes available:\n\n- **Auto Mode**: Test current model via `openclaw infer model run`, **no API key required**\n- **API Mode**: Direct call to OpenAI-compatible API, requires URL and Key\n\n## Trigger Rules\n\n**Use when:** User explicitly requests a model throughput test.\n\n**Trigger words:**\n- throughput test, tokens/s, speed test, benchmark\n- model speed, latency test, model test\n\n**Do NOT trigger:** Broad performance discussion terms (e.g. \"model performance\", standalone \"benchmark\") should not auto-trigger execution.\n\n**Auto Mode (no API key):**\n```bash\npython3 throughput.py --auto --model \"<current session model>\"\n```\n\n## Core Features\n\n### 1. Auto Mode (No Key, Recommended)\n\nAuto-detects the current session model and benchmarks throughput, zero configuration needed.\n\n```bash\npython3 throughput.py --auto\n```\n\nTest a specific model:\n```bash\npython3 throughput.py --auto --model \"zai/glm-5-turbo\"\n```\n\n### 2. API Mode (Direct API Call)\n\n```bash\npython3 throughput.py \\\n  --url https://api.example.com/v1 \\\n  --key sk-xxx \\\n  --models gpt-4o-mini,gpt-4o\n```\n\n### 3. Common Parameters\n\n| Parameter | Default | Description |\n|-----------|---------|-------------|\n| `--iterations` | `3` | Test iterations per model |\n| `--max-tokens` | `512` | Max output tokens |\n| `--test-prompt` | English prose (summer field) | Test prompt |\n| `--timeout` | `60` | Single request timeout (seconds) |\n| `--output` | `throughput-report.md` | Output report filename |\n| `--csv` | false | Also generate CSV |\n\n## Workflow\n\n### Auto Mode Flow\n\n```\n1. Read current session model from openclaw.json (provider/model)\n2. Send test prompt via openclaw infer model run\n3. Timer: command start → output complete\n4. Estimate token count from response text (English: 0.75 word/token, Chinese: 1.5 chars/token)\n5. Calculate tokens/s\n6. Generate summary report\n```\n\n### API Mode Flow\n\n```\n1. Build /v1/chat/completions request\n2. Timer: request start → last token received\n3. Extract usage.completion_tokens from response (precise)\n4. Calculate tokens/s, error rate\n5. Generate summary report\n```\n\n### Metrics\n\n| Metric | Description |\n|--------|-------------|\n| **Tokens/s** | Throughput = Output Tokens / Elapsed Time |\n| **Avg Latency** | Average single request latency |\n| **Avg Output Tokens** | Average output token count |\n| **Error Rate** | Failed request ratio |\n\n## Output Example\n\n```markdown\n# Model Throughput Report\n**Mode:** Auto (openclaw infer)\n**Iterations:** 3\n\n## Summary\n| Model | Avg Tokens/s | Avg Latency(s) | Avg Output Tokens | Error Rate |\n|-------|-------------|----------------|-------------------|------------|\n| zai/glm-5-turbo | 57.9 | 20.6 | 979.0 | 0.0% |\n\n## Detail\n### zai/glm-5-turbo\n| Iter | Latency(s) | Output Tokens | Tokens/s | Status |\n|------|------------|--------------|---------|--------|\n| 1 | 19.5 | 950 | 48.7 | ✅ |\n| 2 | 21.3 | 1010 | 47.4 | ✅ |\n| 3 | 20.9 | 977 | 46.7 | ✅ |\n```\n\n## Error Handling\n\n| Scenario | Auto Mode | API Mode |\n|----------|-----------|----------|\n| openclaw not installed | cli_error | — |\n| Model not found | api_error | http_404 |\n| Network timeout | timeout | timeout |\n| Token estimation | English 0.75 word/token, Chinese 1.5 chars/token | Precise from API |\n\n## Usage Examples\n\n### Quick Test After Install (Auto Mode)\n\n```bash\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto --model \"<current session model>\"\n\n# Or auto-detect (may not match session override)\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto\n```\n\n### Test Multiple Models (API Mode)\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.openai.com/v1\" \\\n  --key \"sk-xxx\" \\\n  --models \"gpt-4o-mini,gpt-4o\" \\\n  --iterations 5\n```\n\n### Custom Prompt\n\n```bash\npython3 throughput.py --auto \\\n  --test-prompt \"Explain quantum computing in detail.\" \\\n  --iterations 5\n```\n\n## Technical Details\n\n- **Auto Mode**: `openclaw infer model run --json`, Python `subprocess` call\n- **API Mode**: `urllib` (Python built-in), OpenAI-compatible `/v1/chat/completions`\n- **Timer Precision**: `time.perf_counter()` nanosecond-level\n- **Token Counting**: API mode uses `usage.completion_tokens` (precise), Auto mode estimates by character count\n- **URL Handling**: Smart detection of `/v1`, `/v4`, `/chat/completions` paths\n\n## Notes\n\n- Auto mode throughput includes gateway routing overhead, slightly lower than direct API (~1-3%)\n- Auto mode token count is estimated, API mode is precise\n- English prompts recommended for more accurate token estimation\n- Anti-cache: random seed suffix appended to each iteration\n\nFile v1.0.5:README.md\n\n# Model Throughput Tester\n\nBenchmark LLM model throughput — measure tokens/s, latency, and output speed for any language model.\n\n## Features\n\n- **Auto Mode**: Test your current session model via `openclaw infer`, no API key needed\n- **API Mode**: Direct benchmark against any OpenAI-compatible endpoint\n- **Flexible**: Custom prompts, iteration counts, timeout controls\n- **Reports**: Markdown + CSV output with per-iteration details\n\n## Quick Start\n\n```bash\n# Auto mode — test current session model\npython3 throughput.py --auto\n\n# Test a specific model\npython3 throughput.py --auto --model \"gpt-4o\"\n\n# API mode — test against an endpoint\npython3 throughput.py \\\n  --url \"https://api.openai.com/v1\" \\\n  --key \"sk-xxx\" \\\n  --models \"gpt-4o-mini,gpt-4o\" \\\n  --iterations 5\n```\n\n## Parameters\n\n| Parameter | Default | Description |\n|-----------|---------|-------------|\n| `--auto` | off | Enable auto mode (uses openclaw infer) |\n| `--model` | auto-detect | Model identifier |\n| `--url` | — | API base URL (API mode) |\n| `--key` | — | API key (API mode) |\n| `--models` | — | Comma-separated model list (API mode) |\n| `--iterations` | `3` | Test iterations per model |\n| `--max-tokens` | `512` | Max output tokens |\n| `--test-prompt` | built-in | Custom test prompt |\n| `--timeout` | `60` | Request timeout (seconds) |\n| `--output` | `throughput-report.md` | Output report filename |\n| `--csv` | false | Also generate CSV output |\n\n## Metrics\n\n| Metric | Description |\n|--------|-------------|\n| **Tokens/s** | Throughput = Output Tokens / Elapsed Time |\n| **Avg Latency** | Average single-request latency |\n| **Avg Output Tokens** | Average output token count |\n| **Error Rate** | Failed request ratio |\n\n## Example Output\n\n```\n📊 Model Throughput Report\nMode: Auto (openclaw infer) | Iterations: 3\n\nSummary\n| Model             | Avg Tokens/s | Latency(s) | Output Tokens | Error |\n|-------------------|-------------|------------|----------------|-------|\n| zai/glm-5-turbo   | 57.9        | 20.6       | 979            | 0.0%  |\n```\n\n## How It Works\n\n**Auto Mode**: Sends a test prompt via `openclaw infer model run`, measures wall-clock time from start to last token, then estimates token count from output text.\n\n**API Mode**: Calls `/v1/chat/completions` with streaming disabled, reads `usage.completion_tokens` for precise token counts.\n\n## Notes\n\n- Auto mode throughput includes gateway routing overhead (~1-3% lower than direct API)\n- Auto mode token counts are estimates; API mode uses precise values\n- English prompts yield more accurate token estimates in auto mode\n- Anti-cache: random seed suffix appended per iteration\n\n## Prerequisites\n\n- Python 3 (built-in on macOS)\n- `openclaw` CLI (for auto mode)\n\n## File Structure\n\n```\n~/.openclaw/workspace/skills/model-throughput-tester/\n├── SKILL.md           # Agent trigger & execution guide\n├── README.md          # This file\n├── README.zh.md       # Chinese version\n├── throughput.py      # Main script\n└── throughput-report.md  # Generated reports (on demand)\n```\n\n## License\n\nMIT-0\n\nFile v1.0.5:_meta.json\n\n{\n  \"ownerId\": \"kn7dzk85gz4pz5c569zky2pssn875nwe\",\n  \"slug\": \"model-throughput-tester\",\n  \"version\": \"1.0.5\",\n  \"publishedAt\": 1780891055786\n}\n\nFile v1.0.5:README.zh.md\n\n# 模型吞吐率测试器\n\n测试 LLM 模型的吞吐率（tokens/s）。支持两种模式：\n\n- **Auto 模式**：通过 `openclaw infer model run` 测试当前模型，**无需 API Key**\n- **API 模式**：直接调用 OpenAI 兼容 API，需要 URL 和 Key\n\n## 触发规则\n\n**适用场景：** 用户明确要求测试模型吞吐率时使用。\n\n**推荐触发词：**\n- 测一下吞吐率、测速、模型测速、tokens/s\n- 跑个 benchmark、吞吐率测试、模型测试\n\n**不适用：** 宽泛的性能讨论词（如「模型性能」「benchmark」单独出现）不应自动触发执行。\n\n## 核心能力\n\n### 1. Auto 模式（无 Key，推荐）\n\n自动检测当前 session 的模型并测试吞吐率，无需任何配置。\n\n```bash\npython3 throughput.py --auto\n```\n\n指定模型测试：\n```bash\npython3 throughput.py --auto --model \"zai/glm-5-turbo\"\n```\n\n### 2. API 模式（直接调用 API）\n\n```bash\npython3 throughput.py \\\n  --url https://api.example.com/v1 \\\n  --key sk-xxx \\\n  --models gpt-4o-mini,gpt-4o\n```\n\n### 3. 通用参数\n\n| 参数 | 默认值 | 说明 |\n|------|--------|------|\n| `--iterations` | `3` | 每个模型测试次数 |\n| `--max-tokens` | `512` | 最大输出 token 数 |\n| `--test-prompt` | 英文散文（夏天的田野） | 测试提示词 |\n| `--timeout` | `60` | 单次请求超时（秒） |\n| `--output` | `throughput-report.md` | 输出报告文件名 |\n| `--csv` | false | 同时生成 CSV |\n\n## 工作流程\n\n### Auto 模式\n\n```\n1. 从 openclaw.json 读取当前 session 模型（provider/model）\n2. 通过 openclaw infer model run 发送测试 prompt\n3. 计时：命令开始 → 输出完成\n4. 从返回文本估算 token 数（英文 0.75 word/token，中文 1.5 字/token）\n5. 计算 tokens/s\n6. 汇总输出报告\n```\n\n### API 模式\n\n```\n1. 构造 /v1/chat/completions 请求\n2. 计时：请求开始 → 最后一个 token\n3. 从响应中提取 usage.completion_tokens（精确）\n4. 计算 tokens/s、错误率\n5. 汇总输出报告\n```\n\n## 指标说明\n\n| 指标 | 说明 |\n|------|------|\n| **Tokens/s** | 吞吐率 = Output Tokens / Elapsed Time |\n| **Avg Latency** | 平均单次请求延迟 |\n| **Avg Output Tokens** | 平均输出 token 数 |\n| **Error Rate** | 错误请求占比 |\n\n## 使用示例\n\n### 安装后立即测试（Auto 模式）\n\n```bash\n# agent 触发时应传入当前模型\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto --model \"<当前session模型>\"\n\n# 或使用自动检测（可能不是 session 覆盖的模型）\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto\n```\n\n### 测试多个模型（API 模式）\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.openai.com/v1\" \\\n  --key \"sk-xxx\" \\\n  --models \"gpt-4o-mini,gpt-4o\" \\\n  --iterations 5\n```\n\n### 自定义提示词\n\n```bash\npython3 throughput.py --auto \\\n  --test-prompt \"Explain quantum computing in detail.\" \\\n  --iterations 5\n```\n\n## 技术实现\n\n- **Auto 模式**：`openclaw infer model run --json`，Python `subprocess` 调用\n- **API 模式**：`urllib`（Python 内置），OpenAI 兼容 `/v1/chat/completions`\n- **计时精度**：`time.perf_counter()` 纳秒级精度\n- **Token 计数**：API 模式优先 `usage.completion_tokens`（精确），Auto 模式按字符估算\n- **URL 拼接**：智能检测 `/v1`、`/v4`、`/chat/completions` 路径\n\n## 注意事项\n\n- Auto 模式的吞吐率包含网关路由开销，会比直接 API 略低（约 1-3%）\n- Auto 模式 Token 数为估算值，API 模式为精确值\n- 建议使用英文 prompt 以获得更准确的 token 估算\n- 防缓存：每次迭代自动附加随机 seed 后缀\n\nFile v1.0.5:skill-card.md\n\n## Description: <br>\nBenchmarks LLM model throughput by measuring tokens per second, latency, output tokens, and error rate through auto mode or OpenAI-compatible API endpoints. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[tsag1](https://clawhub.ai/user/tsag1) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and engineers use this skill to benchmark language model throughput after configuring a current OpenClaw model or an OpenAI-compatible endpoint. It helps compare model latency, output speed, token volume, and error rate across repeated test runs. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: API mode requires credentials and can expose API keys through shell history or process arguments. <br>\nMitigation: Prefer auto mode when possible and pass credentials through safer mechanisms such as environment variables or a protected local configuration. <br>\nRisk: Benchmark prompts and generated responses are sent to the configured model provider. <br>\nMitigation: Use non-sensitive test prompts and avoid benchmarking with confidential or personal data. <br>\nRisk: Generated throughput reports may include endpoint, model, prompt, status, and error details that are not intended for broad sharing. <br>\nMitigation: Review Markdown and CSV reports before publishing or attaching them to tickets. <br>\n\n\n## Reference(s): <br>\n- [ClawHub Skill Page](https://clawhub.ai/tsag1/model-throughput-tester) <br>\n- [README.md](artifact/README.md) <br>\n- [SKILL.md](artifact/SKILL.md) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [Text, Markdown, Shell commands, Configuration, CSV] <br>\n**Output Format:** [Markdown guidance with shell command examples; generated benchmark reports are Markdown with optional CSV output.] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Reports include per-model averages and per-iteration latency, output token count, tokens per second, status, and error rate.] <br>\n\n## Skill Version(s): <br>\n1.0.5 (source: server release metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.0.4: 5 files, 9555 bytes\n\nFiles: _meta.json (142b), skill-card.md (2226b), SKILL.md (4994b), throughput-report.md (806b), throughput.py (16224b)\n\nFile v1.0.4:SKILL.md\n\n---\nname: model_throughput_tester\ndescription: Benchmark LLM model throughput — measure tokens/s, latency, and output speed. Supports auto mode (no API key needed) via openclaw infer, or direct API mode for OpenAI-compatible endpoints. Trigger: 测一下吞吐率, 测速, 模型测速, tokens/s, 吞吐率测试, 模型测试.\n---\n\n# Model Throughput Tester\n\n测试 LLM 模型的吞吐率（tokens/s）。支持两种模式：\n\n- **Auto 模式**：通过 `openclaw infer model run` 测试当前模型，**无需 API Key**\n- **API 模式**：直接调用 OpenAI 兼容 API，需要 URL 和 Key\n\n## 触发规则\n\n**适用场景：** 用户明确要求测试模型吞吐率时使用。\n\n**推荐触发词：**\n- 测一下吞吐率、测速、模型测速、tokens/s\n- 跑个 benchmark、吞吐率测试、模型测试\n\n**不适用：** 宽泛的性能讨论词（如「模型性能」「benchmark」单独出现）不应自动触发执行。\n\n**Auto 模式（无需 API Key）：**\n```bash\npython3 throughput.py --auto --model \"<当前session模型>\"\n```\n\n## 核心能力\n\n### 1. Auto 模式（无 Key，推荐）\n\n自动检测当前 session 的模型并测试吞吐率，无需任何配置。\n\n```bash\npython3 throughput.py --auto\n```\n\n指定模型测试：\n```bash\npython3 throughput.py --auto --model \"zai/glm-5-turbo\"\n```\n\n### 2. API 模式（直接调用 API）\n\n```bash\npython3 throughput.py \\\n  --url https://api.example.com/v1 \\\n  --key sk-xxx \\\n  --models gpt-4o-mini,gpt-4o\n```\n\n### 3. 通用参数\n\n| 参数 | 默认值 | 说明 |\n|------|--------|------|\n| `--iterations` | `3` | 每个模型测试次数 |\n| `--max-tokens` | `512` | 最大输出 token 数 |\n| `--test-prompt` | 英文散文（夏天的田野） | 测试提示词 |\n| `--timeout` | `60` | 单次请求超时（秒） |\n| `--output` | `throughput-report.md` | 输出报告文件名 |\n| `--csv` | false | 同时生成 CSV |\n\n## Workflow\n\n### Auto 模式流程\n\n```\n1. 从 openclaw.json 读取当前 session 模型（provider/model）\n2. 通过 openclaw infer model run 发送测试 prompt\n3. 计时：命令开始 → 输出完成\n4. 从返回文本估算 token 数（英文 0.75 word/token，中文 1.5 字/token）\n5. 计算 tokens/s\n6. 汇总输出报告\n```\n\n### API 模式流程\n\n```\n1. 构造 /v1/chat/completions 请求\n2. 计时：请求开始 → 最后一个 token\n3. 从响应中提取 usage.completion_tokens（精确）\n4. 计算 tokens/s、错误率\n5. 汇总输出报告\n```\n\n### 指标定义\n\n| 指标 | 说明 |\n|------|------|\n| **Tokens/s** | 吞吐率 = Output Tokens / Elapsed Time |\n| **Avg Latency** | 平均单次请求延迟 |\n| **Avg Output Tokens** | 平均输出 token 数 |\n| **Error Rate** | 错误请求占比 |\n\n## 输出示例\n\n```markdown\n# Model Throughput Report\n**Mode:** Auto (openclaw infer)\n**Iterations:** 3\n\n## Summary\n| Model | Avg Tokens/s | Avg Latency(s) | Avg Output Tokens | Error Rate |\n|-------|-------------|----------------|-------------------|------------|\n| zai/glm-5-turbo | 57.9 | 20.6 | 979.0 | 0.0% |\n\n## Detail\n### zai/glm-5-turbo\n| Iter | Latency(s) | Output Tokens | Tokens/s | Status |\n|------|------------|--------------|---------|--------|\n| 1 | 19.5 | 950 | 48.7 | ✅ |\n| 2 | 21.3 | 1010 | 47.4 | ✅ |\n| 3 | 20.9 | 977 | 46.7 | ✅ |\n```\n\n## 错误处理\n\n| 场景 | Auto 模式 | API 模式 |\n|------|----------|---------|\n| 未安装 openclaw | cli_error | — |\n| 模型不存在 | api_error | http_404 |\n| 网络超时 | timeout | timeout |\n| Token 估算 | 英文 0.75 word/token，中文 1.5 字/token | API 返回精确值 |\n\n## 使用示例\n\n### 安装后立即测试（Auto 模式）\n\n```bash\n# agent 触发时应传入当前模型\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto --model \"<当前session模型>\"\n\n# 或使用自动检测（可能不是 session 覆盖的模型）\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto\n```\n\n### 测试多个模型（API 模式）\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.openai.com/v1\" \\\n  --key \"sk-xxx\" \\\n  --models \"gpt-4o-mini,gpt-4o\" \\\n  --iterations 5\n```\n\n### 自定义提示词\n\n```bash\npython3 throughput.py --auto \\\n  --test-prompt \"Explain quantum computing in detail.\" \\\n  --iterations 5\n```\n\n## 技术实现\n\n- **Auto 模式**：`openclaw infer model run --json`，Python `subprocess` 调用\n- **API 模式**：`urllib`（Python 内置），OpenAI 兼容 `/v1/chat/completions`\n- **计时精度**：`time.perf_counter()` 纳秒级精度\n- **Token 计数**：API 模式优先 `usage.completion_tokens`（精确），Auto 模式按字符估算\n- **URL 拼接**：智能检测 `/v1`、`/v4`、`/chat/completions` 路径\n\n## 注意事项\n\n- Auto 模式的吞吐率包含网关路由开销，会比直接 API 略低（约 1-3%）\n- Auto 模式 Token 数为估算值，API 模式为精确值\n- 建议使用英文 prompt 以获得更准确的 token 估算\n- 防缓存：每次迭代自动附加随机 seed 后缀\n\nFile v1.0.4:_meta.json\n\n{\n  \"ownerId\": \"kn7dzk85gz4pz5c569zky2pssn875nwe\",\n  \"slug\": \"model-throughput-tester\",\n  \"version\": \"1.0.4\",\n  \"publishedAt\": 1780878502680\n}\n\nFile v1.0.4:skill-card.md\n\n## Description: <br>\nBenchmarks LLM model throughput by measuring tokens per second, latency, output tokens, and error rate in OpenClaw auto mode or direct OpenAI-compatible API mode. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[tsag1](https://clawhub.ai/user/tsag1) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and engineers use this skill to compare LLM response throughput for the current OpenClaw session model or configured OpenAI-compatible endpoints before selecting or tuning a model. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: API mode sends the benchmark prompt and bearer token to the endpoint the user provides. <br>\nMitigation: Use trusted HTTPS endpoints and avoid sensitive prompts when running API-mode benchmarks. <br>\nRisk: The skill writes local report files that may overwrite or retain benchmark details. <br>\nMitigation: Choose an explicit output path, review generated reports, and remove files that should not be retained. <br>\nRisk: Auto mode estimates output tokens, so throughput values may be approximate. <br>\nMitigation: Use consistent prompts for comparisons, and prefer API mode when precise completion token counts are required. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/tsag1/model-throughput-tester) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [Markdown, CSV, Shell commands, Text] <br>\n**Output Format:** [Markdown report with summary and detail tables, optional CSV rows, and terminal text] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Auto mode uses the current OpenClaw model configuration; API mode accepts endpoint URL, bearer key, model list, iteration count, token limit, prompt, timeout, output path, and optional CSV output.] <br>\n\n## Skill Version(s): <br>\n1.0.4 (source: server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.0.4:throughput-report.md\n\n# Model Throughput Report\n\n**Generated:** 2026-06-08 03:31:24\n**Mode:** Auto (openclaw infer)\n**API URL:** https://open.bigmodel.cn/api/coding/paas/v4\n**Iterations:** 2\n**Test Prompt:** Write a detailed essay about summer meadows. No headings, no paragraph breaks, continuous prose only.\n**Token 估算:** 估算值（英文 ≈ 0.75 word/token，中文 ≈ 1.5 字/token）\n\n## Summary\n\n| Model | Avg Tokens/s | Avg Latency(s) | Avg Output Tokens | Error Rate |\n|-------|-------------|----------------|-------------------|------------|\n| zai/glm-5-turbo | 52.1 | 24.088 | 1252.0 | 0.0% |\n\n## Detail\n\n### zai/glm-5-turbo\n| Iter | Latency(s) | Output Tokens | Tokens/s | Status |\n|------|------------|--------------|---------|--------|\n| 1 | 22.465 | 1199 | 53.4 | ✅ |\n| 2 | 25.712 | 1305 | 50.8 | ✅ |\n\nArchive v1.0.3: 5 files, 9507 bytes\n\nFiles: _meta.json (142b), skill-card.md (2166b), SKILL.md (5034b), throughput-report.md (806b), throughput.py (16224b)\n\nFile v1.0.3:SKILL.md\n\n---\nname: model_throughput_tester\ndescription: Benchmark LLM model throughput — measure tokens/s, latency, and output speed. Supports auto mode (no API key needed) via openclaw infer, or direct API mode for OpenAI-compatible endpoints. Trigger: 模型吞吐率, tokens/s, 延迟测试, benchmark, 吞吐率测试, 模型性能, 模型测速, 测一下吞吐率, 测速.\n---\n\n# Model Throughput Tester\n\n测试 LLM 模型的吞吐率（tokens/s）。支持两种模式：\n\n- **Auto 模式**：通过 `openclaw infer model run` 测试当前模型，**无需 API Key**\n- **API 模式**：直接调用 OpenAI 兼容 API，需要 URL 和 Key\n\n## 触发规则\n\n当用户提到以下关键词时，**默认使用 Auto 模式**测试当前 session 模型，无需 API Key：\n\n- 吞吐率、tokens/s、测速、模型速度、模型性能、benchmark\n- 模型测试、模型吞吐率测试、测一下吞吐率\n\n**Agent 触发时应自动执行：**\n```\npython3 throughput.py --auto --model \"<当前session模型>\"\n```\n不要询问用户是否需要 Key 或其他参数，直接用 Auto 模式开测。\n\n## 核心能力\n\n### 1. Auto 模式（无 Key，推荐）\n\n自动检测当前 session 的模型并测试吞吐率，无需任何配置。\n\n```bash\npython3 throughput.py --auto\n```\n\n指定模型测试：\n```bash\npython3 throughput.py --auto --model \"zai/glm-5-turbo\"\n```\n\n### 2. API 模式（直接调用 API）\n\n```bash\npython3 throughput.py \\\n  --url https://api.example.com/v1 \\\n  --key sk-xxx \\\n  --models gpt-4o-mini,gpt-4o\n```\n\n### 3. 通用参数\n\n| 参数 | 默认值 | 说明 |\n|------|--------|------|\n| `--iterations` | `3` | 每个模型测试次数 |\n| `--max-tokens` | `512` | 最大输出 token 数 |\n| `--test-prompt` | 英文散文（夏天的田野） | 测试提示词 |\n| `--timeout` | `60` | 单次请求超时（秒） |\n| `--output` | `throughput-report.md` | 输出报告文件名 |\n| `--csv` | false | 同时生成 CSV |\n\n## Workflow\n\n### Auto 模式流程\n\n```\n1. 从 openclaw.json 读取当前 session 模型（provider/model）\n2. 通过 openclaw infer model run 发送测试 prompt\n3. 计时：命令开始 → 输出完成\n4. 从返回文本估算 token 数（英文 0.75 word/token，中文 1.5 字/token）\n5. 计算 tokens/s\n6. 汇总输出报告\n```\n\n### API 模式流程\n\n```\n1. 构造 /v1/chat/completions 请求\n2. 计时：请求开始 → 最后一个 token\n3. 从响应中提取 usage.completion_tokens（精确）\n4. 计算 tokens/s、错误率\n5. 汇总输出报告\n```\n\n### 指标定义\n\n| 指标 | 说明 |\n|------|------|\n| **Tokens/s** | 吞吐率 = Output Tokens / Elapsed Time |\n| **Avg Latency** | 平均单次请求延迟 |\n| **Avg Output Tokens** | 平均输出 token 数 |\n| **Error Rate** | 错误请求占比 |\n\n## 输出示例\n\n```markdown\n# Model Throughput Report\n**Mode:** Auto (openclaw infer)\n**Iterations:** 3\n\n## Summary\n| Model | Avg Tokens/s | Avg Latency(s) | Avg Output Tokens | Error Rate |\n|-------|-------------|----------------|-------------------|------------|\n| zai/glm-5-turbo | 57.9 | 20.6 | 979.0 | 0.0% |\n\n## Detail\n### zai/glm-5-turbo\n| Iter | Latency(s) | Output Tokens | Tokens/s | Status |\n|------|------------|--------------|---------|--------|\n| 1 | 19.5 | 950 | 48.7 | ✅ |\n| 2 | 21.3 | 1010 | 47.4 | ✅ |\n| 3 | 20.9 | 977 | 46.7 | ✅ |\n```\n\n## 错误处理\n\n| 场景 | Auto 模式 | API 模式 |\n|------|----------|---------|\n| 未安装 openclaw | cli_error | — |\n| 模型不存在 | api_error | http_404 |\n| 网络超时 | timeout | timeout |\n| Token 估算 | 英文 0.75 word/token，中文 1.5 字/token | API 返回精确值 |\n\n## 使用示例\n\n### 安装后立即测试（Auto 模式）\n\n```bash\n# agent 触发时应传入当前模型\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto --model \"<当前session模型>\"\n\n# 或使用自动检测（可能不是 session 覆盖的模型）\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto\n```\n\n### 测试多个模型（API 模式）\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.openai.com/v1\" \\\n  --key \"sk-xxx\" \\\n  --models \"gpt-4o-mini,gpt-4o\" \\\n  --iterations 5\n```\n\n### 自定义提示词\n\n```bash\npython3 throughput.py --auto \\\n  --test-prompt \"Explain quantum computing in detail.\" \\\n  --iterations 5\n```\n\n## 技术实现\n\n- **Auto 模式**：`openclaw infer model run --json`，Python `subprocess` 调用\n- **API 模式**：`urllib`（Python 内置），OpenAI 兼容 `/v1/chat/completions`\n- **计时精度**：`time.perf_counter()` 纳秒级精度\n- **Token 计数**：API 模式优先 `usage.completion_tokens`（精确），Auto 模式按字符估算\n- **URL 拼接**：智能检测 `/v1`、`/v4`、`/chat/completions` 路径\n\n## 注意事项\n\n- Auto 模式的吞吐率包含网关路由开销，会比直接 API 略低（约 1-3%）\n- Auto 模式 Token 数为估算值，API 模式为精确值\n- 建议使用英文 prompt 以获得更准确的 token 估算\n- 防缓存：每次迭代自动附加随机 seed 后缀\n\nFile v1.0.3:_meta.json\n\n{\n  \"ownerId\": \"kn7dzk85gz4pz5c569zky2pssn875nwe\",\n  \"slug\": \"model-throughput-tester\",\n  \"version\": \"1.0.3\",\n  \"publishedAt\": 1780861316002\n}\n\nFile v1.0.3:skill-card.md\n\n## Description: <br>\nBenchmark LLM model throughput by measuring tokens per second, latency, and output speed using OpenClaw auto mode or OpenAI-compatible API mode. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[tsag1](https://clawhub.ai/user/tsag1) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and engineers use this skill to benchmark LLM response throughput, latency, output tokens, and error rate for a current OpenClaw session model or configured OpenAI-compatible endpoint. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Broad performance-related triggers can cause the agent to run local benchmark commands and contact the configured model provider without clear confirmation. <br>\nMitigation: Use the skill after an explicit throughput-test request and review the command before execution. <br>\nRisk: API mode can use sensitive API credentials and send prompts to a configured OpenAI-compatible endpoint. <br>\nMitigation: Confirm the API URL and key source before API-mode runs, and prefer auto mode when no direct endpoint credentials are needed. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/tsag1/model-throughput-tester) <br>\n- [Publisher profile](https://clawhub.ai/user/tsag1) <br>\n- [Skill source](artifact/SKILL.md) <br>\n- [Example throughput report](artifact/throughput-report.md) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, shell commands, code, configuration] <br>\n**Output Format:** [Markdown reports with optional CSV files and shell command output] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Reports summarize tokens per second, latency, output tokens, and error rate.] <br>\n\n## Skill Version(s): <br>\n1.0.3 (source: server-resolved release metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.0.3:throughput-report.md\n\n# Model Throughput Report\n\n**Generated:** 2026-06-08 03:31:24\n**Mode:** Auto (openclaw infer)\n**API URL:** https://open.bigmodel.cn/api/coding/paas/v4\n**Iterations:** 2\n**Test Prompt:** Write a detailed essay about summer meadows. No headings, no paragraph breaks, continuous prose only.\n**Token 估算:** 估算值（英文 ≈ 0.75 word/token，中文 ≈ 1.5 字/token）\n\n## Summary\n\n| Model | Avg Tokens/s | Avg Latency(s) | Avg Output Tokens | Error Rate |\n|-------|-------------|----------------|-------------------|------------|\n| zai/glm-5-turbo | 52.1 | 24.088 | 1252.0 | 0.0% |\n\n## Detail\n\n### zai/glm-5-turbo\n| Iter | Latency(s) | Output Tokens | Tokens/s | Status |\n|------|------------|--------------|---------|--------|\n| 1 | 22.465 | 1199 | 53.4 | ✅ |\n| 2 | 25.712 | 1305 | 50.8 | ✅ |\n\nArchive v1.0.2: 5 files, 9336 bytes\n\nFiles: _meta.json (142b), skill-card.md (2014b), SKILL.md (4577b), throughput-report.md (806b), throughput.py (16224b)\n\nFile v1.0.2:SKILL.md\n\n---\nname: model_throughput_tester\ndescription: Benchmark LLM model throughput — measure tokens/s, latency, and output speed. Supports auto mode (no API key needed) via openclaw infer, or direct API mode for OpenAI-compatible endpoints. Trigger: 模型吞吐率, tokens/s, 延迟测试, benchmark, 吞吐率测试, 模型性能, 模型测速, 测一下吞吐率, 测速.\n---\n\n# Model Throughput Tester\n\n测试 LLM 模型的吞吐率（tokens/s）。支持两种模式：\n\n- **Auto 模式**：通过 `openclaw infer model run` 测试当前模型，**无需 API Key**\n- **API 模式**：直接调用 OpenAI 兼容 API，需要 URL 和 Key\n\n## 核心能力\n\n### 1. Auto 模式（无 Key，推荐）\n\n自动检测当前 session 的模型并测试吞吐率，无需任何配置。\n\n```bash\npython3 throughput.py --auto\n```\n\n指定模型测试：\n```bash\npython3 throughput.py --auto --model \"zai/glm-5-turbo\"\n```\n\n### 2. API 模式（直接调用 API）\n\n```bash\npython3 throughput.py \\\n  --url https://api.example.com/v1 \\\n  --key sk-xxx \\\n  --models gpt-4o-mini,gpt-4o\n```\n\n### 3. 通用参数\n\n| 参数 | 默认值 | 说明 |\n|------|--------|------|\n| `--iterations` | `3` | 每个模型测试次数 |\n| `--max-tokens` | `512` | 最大输出 token 数 |\n| `--test-prompt` | 英文散文（夏天的田野） | 测试提示词 |\n| `--timeout` | `60` | 单次请求超时（秒） |\n| `--output` | `throughput-report.md` | 输出报告文件名 |\n| `--csv` | false | 同时生成 CSV |\n\n## Workflow\n\n### Auto 模式流程\n\n```\n1. 从 openclaw.json 读取当前 session 模型（provider/model）\n2. 通过 openclaw infer model run 发送测试 prompt\n3. 计时：命令开始 → 输出完成\n4. 从返回文本估算 token 数（英文 0.75 word/token，中文 1.5 字/token）\n5. 计算 tokens/s\n6. 汇总输出报告\n```\n\n### API 模式流程\n\n```\n1. 构造 /v1/chat/completions 请求\n2. 计时：请求开始 → 最后一个 token\n3. 从响应中提取 usage.completion_tokens（精确）\n4. 计算 tokens/s、错误率\n5. 汇总输出报告\n```\n\n### 指标定义\n\n| 指标 | 说明 |\n|------|------|\n| **Tokens/s** | 吞吐率 = Output Tokens / Elapsed Time |\n| **Avg Latency** | 平均单次请求延迟 |\n| **Avg Output Tokens** | 平均输出 token 数 |\n| **Error Rate** | 错误请求占比 |\n\n## 输出示例\n\n```markdown\n# Model Throughput Report\n**Mode:** Auto (openclaw infer)\n**Iterations:** 3\n\n## Summary\n| Model | Avg Tokens/s | Avg Latency(s) | Avg Output Tokens | Error Rate |\n|-------|-------------|----------------|-------------------|------------|\n| zai/glm-5-turbo | 57.9 | 20.6 | 979.0 | 0.0% |\n\n## Detail\n### zai/glm-5-turbo\n| Iter | Latency(s) | Output Tokens | Tokens/s | Status |\n|------|------------|--------------|---------|--------|\n| 1 | 19.5 | 950 | 48.7 | ✅ |\n| 2 | 21.3 | 1010 | 47.4 | ✅ |\n| 3 | 20.9 | 977 | 46.7 | ✅ |\n```\n\n## 错误处理\n\n| 场景 | Auto 模式 | API 模式 |\n|------|----------|---------|\n| 未安装 openclaw | cli_error | — |\n| 模型不存在 | api_error | http_404 |\n| 网络超时 | timeout | timeout |\n| Token 估算 | 英文 0.75 word/token，中文 1.5 字/token | API 返回精确值 |\n\n## 使用示例\n\n### 安装后立即测试（Auto 模式）\n\n```bash\n# agent 触发时应传入当前模型\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto --model \"<当前session模型>\"\n\n# 或使用自动检测（可能不是 session 覆盖的模型）\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto\n```\n\n### 测试多个模型（API 模式）\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.openai.com/v1\" \\\n  --key \"sk-xxx\" \\\n  --models \"gpt-4o-mini,gpt-4o\" \\\n  --iterations 5\n```\n\n### 自定义提示词\n\n```bash\npython3 throughput.py --auto \\\n  --test-prompt \"Explain quantum computing in detail.\" \\\n  --iterations 5\n```\n\n## 技术实现\n\n- **Auto 模式**：`openclaw infer model run --json`，Python `subprocess` 调用\n- **API 模式**：`urllib`（Python 内置），OpenAI 兼容 `/v1/chat/completions`\n- **计时精度**：`time.perf_counter()` 纳秒级精度\n- **Token 计数**：API 模式优先 `usage.completion_tokens`（精确），Auto 模式按字符估算\n- **URL 拼接**：智能检测 `/v1`、`/v4`、`/chat/completions` 路径\n\n## 注意事项\n\n- Auto 模式的吞吐率包含网关路由开销，会比直接 API 略低（约 1-3%）\n- Auto 模式 Token 数为估算值，API 模式为精确值\n- 建议使用英文 prompt 以获得更准确的 token 估算\n- 防缓存：每次迭代自动附加随机 seed 后缀\n\nFile v1.0.2:_meta.json\n\n{\n  \"ownerId\": \"kn7dzk85gz4pz5c569zky2pssn875nwe\",\n  \"slug\": \"model-throughput-tester\",\n  \"version\": \"1.0.2\",\n  \"publishedAt\": 1780860844543\n}\n\nFile v1.0.2:skill-card.md\n\n## Description: <br>\nBenchmarks LLM throughput by measuring tokens per second, latency, output tokens, and error rate through OpenClaw auto mode or OpenAI-compatible API calls. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[tsag1](https://clawhub.ai/user/tsag1) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and AI engineers use this skill to benchmark LLM serving performance for a current OpenClaw session model or specified OpenAI-compatible endpoints. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: API mode requires users to provide credentials for an OpenAI-compatible endpoint. <br>\nMitigation: Use API keys only with trusted endpoints and avoid placing long-lived or broadly scoped credentials in shared command history. <br>\nRisk: Benchmark prompts, model outputs, and endpoint details may be written into local Markdown or CSV reports. <br>\nMitigation: Use non-sensitive benchmark prompts and review generated report files before sharing them. <br>\nRisk: The default report path may overwrite an existing throughput-report.md file in the working directory. <br>\nMitigation: Set an explicit output path when preserving previous benchmark reports matters. <br>\n\n\n## Reference(s): <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, shell commands, configuration, guidance] <br>\n**Output Format:** [Markdown report, optional CSV, and terminal text] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Writes throughput-report.md by default and can write a matching CSV when requested.] <br>\n\n## Skill Version(s): <br>\n1.0.2 (source: server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.0.2:throughput-report.md\n\n# Model Throughput Report\n\n**Generated:** 2026-06-08 03:31:24\n**Mode:** Auto (openclaw infer)\n**API URL:** https://open.bigmodel.cn/api/coding/paas/v4\n**Iterations:** 2\n**Test Prompt:** Write a detailed essay about summer meadows. No headings, no paragraph breaks, continuous prose only.\n**Token 估算:** 估算值（英文 ≈ 0.75 word/token，中文 ≈ 1.5 字/token）\n\n## Summary\n\n| Model | Avg Tokens/s | Avg Latency(s) | Avg Output Tokens | Error Rate |\n|-------|-------------|----------------|-------------------|------------|\n| zai/glm-5-turbo | 52.1 | 24.088 | 1252.0 | 0.0% |\n\n## Detail\n\n### zai/glm-5-turbo\n| Iter | Latency(s) | Output Tokens | Tokens/s | Status |\n|------|------------|--------------|---------|--------|\n| 1 | 22.465 | 1199 | 53.4 | ✅ |\n| 2 | 25.712 | 1305 | 50.8 | ✅ |\n\nArchive v1.0.1: 5 files, 7633 bytes\n\nFiles: _meta.json (142b), skill-card.md (2351b), SKILL.md (4857b), throughput-report.md (527b), throughput.py (8385b)\n\nFile v1.0.1:SKILL.md\n\n---\nname: model_throughput_tester\ndescription: Benchmark any LLM API endpoint — measure tokens/s, latency, and throughput for OpenAI-compatible models. Compare performance across providers, detect rate limits, and export reports. Trigger: 模型吞吐率, API 性能测试, tokens/s, 延迟测试, benchmark, 模型对比, 模型评测, 基准测试。\n...\n\n# Model Throughput Tester\n\n测试 OpenAI 兼容 API 的模型吞吐率，帮助你了解不同模型的响应速度和吞吐量。\n\n## Quick Start\n\n直接运行核心脚本：\n\n```\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py \\\n  --url https://api.example.com/v1 \\\n  --key sk-xxx \\\n  --models gpt-4o-mini,gpt-4o\n```\n\n## Workflow\n\n### 流程说明\n\n```\n1. 解析参数（API URL、Key、模型列表、迭代次数）\n2. 对每个模型按顺序执行测试：\n   a. 构造 /v1/chat/completions 请求\n   b. 计时：请求开始 → 最后一个 token\n   c. 从响应中提取：总耗时、输出 token 数\n   d. 计算：tokens/s、错误率\n3. 汇总所有模型的测试结果\n4. 输出 Markdown 报告 + 可选 CSV\n```\n\n### 指标定义\n\n| 指标 | 说明 |\n|------|------|\n| **Total Time** | 请求总耗时（秒） |\n| **Output Tokens** | 模型输出的 token 数量 |\n| **Tokens/s** | 吞吐率 = Output Tokens / Total Time |\n| **Error Rate** | 错误请求 / 总请求数 |\n| **Avg Latency** | 平均单次请求延迟（秒） |\n\n## 参数说明\n\n| 参数 | 必填 | 默认值 | 说明 |\n|------|------|--------|------|\n| `--url` | ✅ | — | API 基础 URL（不含 /v1/chat/completions 路径） |\n| `--key` | ✅ | — | API Key |\n| `--models` | ✅ | — | 模型名，逗号分隔或 JSON 数组，如 `[\"gpt-4o\",\"gpt-4o-mini\"]` |\n| `--iterations` | ❌ | `3` | 每个模型测试迭代次数 |\n| `--max-tokens` | ❌ | `512` | 最大输出 token 数 |\n| `--system-prompt` | ❌ | `\"You are a helpful assistant.\"` | 系统提示词 |\n| `--test-prompt` | ❌ | `\"Write a short poem about the sea.\"` | 测试用用户提示词 |\n| `--timeout` | ❌ | `60` | 单次请求超时（秒） |\n| `--stream` | ❌ | `false` | 是否使用 streaming 模式 |\n| `--output` | ❌ | `throughput-report.md` | 输出报告文件名 |\n| `--csv` | ❌ | `false` | 是否同时生成 CSV 文件 |\n\n## 输出格式\n\n### Markdown 报告\n\n```markdown\n# Model Throughput Report\n\n**Generated:** 2026-06-08 00:15:00\n**API URL:** https://api.example.com/v1\n**Iterations:** 3\n\n## Summary\n\n| Model | Avg Tokens/s | Avg Latency(s) | Avg Output Tokens | Error Rate |\n|-------|-------------|----------------|-------------------|------------|\n| gpt-4o-mini | 45.2 | 1.85 | 83.7 | 0.0% |\n| gpt-4o | 38.1 | 2.56 | 98.2 | 0.0% |\n\n## Detail\n\n### gpt-4o-mini\n| Iter | Latency(s) | Output Tokens | Tokens/s | Status |\n|------|------------|--------------|---------|--------|\n| 1 | 1.78 | 82 | 46.1 | ✅ |\n| 2 | 1.89 | 85 | 45.0 | ✅ |\n| 3 | 1.88 | 84 | 44.7 | ✅ |\n```\n\n### CSV 格式\n\n```csv\nmodel,iter,elapsed_s,output_tokens,tokens_per_s,status,error\ngpt-4o-mini,1,1.78,82,46.1,ok,\ngpt-4o-mini,2,1.89,85,45.0,ok,\n...\n```\n\n## 使用示例\n\n### 基本用法\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.openai.com/v1\" \\\n  --key \"sk-xxx\" \\\n  --models \"gpt-4o-mini,gpt-4o\"\n```\n\n### 批量测试（JSON 数组格式）\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.example.com/v1\" \\\n  --key \"sk-key\" \\\n  --models '[\"gpt-4o-mini\",\"gpt-4o\",\"claude-3-haiku\"]' \\\n  --iterations 5 \\\n  --max-tokens 256\n```\n\n### Streaming 模式\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.example.com/v1\" \\\n  --key \"sk-key\" \\\n  --models \"gpt-4o-mini\" \\\n  --stream \\\n  --iterations 3\n```\n\n### 自定义提示词\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.example.com/v1\" \\\n  --key \"sk-key\" \\\n  --models \"gpt-4o-mini\" \\\n  --test-prompt \"Explain quantum computing in 3 sentences.\" \\\n  --system-prompt \"You are a physics expert.\"\n```\n\n### 完整参数示例\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.example.com/v1\" \\\n  --key \"sk-key\" \\\n  --models \"gpt-4o-mini,gpt-4o\" \\\n  --iterations 5 \\\n  --max-tokens 512 \\\n  --timeout 120 \\\n  --stream false \\\n  --output \"my-throughput-report.md\" \\\n  --csv\n```\n\n## 错误处理\n\n| 场景 | 处理方式 |\n|------|---------|\n| API Key 为空 | 输出错误并退出（exit code 1）|\n| URL 无响应 | 标记为 error，记录错误，继续下一个 |\n| 模型不存在 | 标记为 http_404，记录错误 |\n| 网络超时 | 标记为 timeout，继续下一个 |\n| 部分失败 | 汇总中显示 Error Rate |\n\n## 技术实现\n\n- **HTTP 库**：`urllib`（Python 内置，无需额外安装）\n- **计时精度**：`time.perf_counter()` 纳秒级精度\n- **Token 计数**：优先用 `usage.completion_tokens`，fallback 为字符数估算\n- **Streaming**：支持 `--stream` 模式\n- **并发**：当前版本为顺序执行\n\nFile v1.0.1:_meta.json\n\n{\n  \"ownerId\": \"kn7dzk85gz4pz5c569zky2pssn875nwe\",\n  \"slug\": \"model-throughput-tester\",\n  \"version\": \"1.0.1\",\n  \"publishedAt\": 1780858072493\n}\n\nFile v1.0.1:skill-card.md\n\n## Description: <br>\nBenchmarks OpenAI-compatible LLM API endpoints to measure latency, tokens per second, throughput, cache hits, and error rates, then exports Markdown and optional CSV reports. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[tsag1](https://clawhub.ai/user/tsag1) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and engineers use this skill to benchmark user-selected OpenAI-compatible model endpoints, compare latency and throughput across models or providers, and produce local Markdown or CSV performance reports. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Benchmark prompts and generated responses are sent to the configured provider endpoint. <br>\nMitigation: Use test prompts, avoid confidential or regulated data, and run against providers approved for the intended data. <br>\nRisk: API keys are passed as command-line arguments and may be exposed through shell history or process listings. <br>\nMitigation: Use a limited-scope API key and avoid running the command in shared shells or environments where process arguments are visible. <br>\nRisk: Throughput and latency results depend on endpoint behavior, model availability, rate limits, cache hits, network conditions, and prompt choice. <br>\nMitigation: Run repeated tests with representative non-sensitive prompts and review error rates and cache-hit status before comparing providers. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/tsag1/model-throughput-tester) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance] <br>\n**Output Format:** [Markdown reports with optional CSV files and terminal status output] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Writes a local report file, can also write a CSV file, and sends benchmark prompts to the configured API endpoint.] <br>\n\n## Skill Version(s): <br>\n1.0.1 (source: server release metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.0.1:throughput-report.md\n\n# Model Throughput Report\n\n**Generated:** 2026-06-08 01:20:04\n**API URL:** https://open.bigmodel.cn\n**Iterations:** 1\n**Test Prompt:** hi\n\n## Summary\n\n| Model | Avg Tokens/s | Avg Latency(s) | Avg Output Tokens | Error Rate |\n|-------|-------------|----------------|-------------------|------------|\n| glm-5-turbo | 0.0 | 0.000 | 0.0 | 100.0% |\n\n## Detail\n\n### glm-5-turbo\n| Iter | Latency(s) | Output Tokens | Tokens/s | Status |\n|------|------------|--------------|---------|--------|\n| 1 | 0.131 | 0 | 0.0 | ❌ http_405 |\n\nArchive v1.0.0: 4 files, 6211 bytes\n\nFiles: _meta.json (142b), skill-card.md (1623b), SKILL.md (4813b), throughput.py (7076b)\n\nFile v1.0.0:SKILL.md\n\n---\nname: model_throughput_tester\ndescription: 测试 OpenAI 兼容 API 的模型吞吐率。测量每个模型的延迟、tokens/s、错误率，支持单次或批量测试，输出 Markdown 表格和 CSV 报告。触发词：模型吞吐率、API 性能测试、throughput、tokens/s 测试、模型对比。\n...\n\n# Model Throughput Tester\n\n测试 OpenAI 兼容 API 的模型吞吐率，帮助你了解不同模型的响应速度和吞吐量。\n\n## Quick Start\n\n直接运行核心脚本：\n\n```\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py \\\n  --url https://api.example.com/v1 \\\n  --key sk-xxx \\\n  --models gpt-4o-mini,gpt-4o\n```\n\n## Workflow\n\n### 流程说明\n\n```\n1. 解析参数（API URL、Key、模型列表、迭代次数）\n2. 对每个模型按顺序执行测试：\n   a. 构造 /v1/chat/completions 请求\n   b. 计时：请求开始 → 最后一个 token\n   c. 从响应中提取：总耗时、输出 token 数\n   d. 计算：tokens/s、错误率\n3. 汇总所有模型的测试结果\n4. 输出 Markdown 报告 + 可选 CSV\n```\n\n### 指标定义\n\n| 指标 | 说明 |\n|------|------|\n| **Total Time** | 请求总耗时（秒） |\n| **Output Tokens** | 模型输出的 token 数量 |\n| **Tokens/s** | 吞吐率 = Output Tokens / Total Time |\n| **Error Rate** | 错误请求 / 总请求数 |\n| **Avg Latency** | 平均单次请求延迟（秒） |\n\n## 参数说明\n\n| 参数 | 必填 | 默认值 | 说明 |\n|------|------|--------|------|\n| `--url` | ✅ | — | API 基础 URL（不含 /v1/chat/completions 路径） |\n| `--key` | ✅ | — | API Key |\n| `--models` | ✅ | — | 模型名，逗号分隔或 JSON 数组，如 `[\"gpt-4o\",\"gpt-4o-mini\"]` |\n| `--iterations` | ❌ | `3` | 每个模型测试迭代次数 |\n| `--max-tokens` | ❌ | `512` | 最大输出 token 数 |\n| `--system-prompt` | ❌ | `\"You are a helpful assistant.\"` | 系统提示词 |\n| `--test-prompt` | ❌ | `\"Write a short poem about the sea.\"` | 测试用用户提示词 |\n| `--timeout` | ❌ | `60` | 单次请求超时（秒） |\n| `--stream` | ❌ | `false` | 是否使用 streaming 模式 |\n| `--output` | ❌ | `throughput-report.md` | 输出报告文件名 |\n| `--csv` | ❌ | `false` | 是否同时生成 CSV 文件 |\n\n## 输出格式\n\n### Markdown 报告\n\n```markdown\n# Model Throughput Report\n\n**Generated:** 2026-06-08 00:15:00\n**API URL:** https://api.example.com/v1\n**Iterations:** 3\n\n## Summary\n\n| Model | Avg Tokens/s | Avg Latency(s) | Avg Output Tokens | Error Rate |\n|-------|-------------|----------------|-------------------|------------|\n| gpt-4o-mini | 45.2 | 1.85 | 83.7 | 0.0% |\n| gpt-4o | 38.1 | 2.56 | 98.2 | 0.0% |\n\n## Detail\n\n### gpt-4o-mini\n| Iter | Latency(s) | Output Tokens | Tokens/s | Status |\n|------|------------|--------------|---------|--------|\n| 1 | 1.78 | 82 | 46.1 | ✅ |\n| 2 | 1.89 | 85 | 45.0 | ✅ |\n| 3 | 1.88 | 84 | 44.7 | ✅ |\n```\n\n### CSV 格式\n\n```csv\nmodel,iter,elapsed_s,output_tokens,tokens_per_s,status,error\ngpt-4o-mini,1,1.78,82,46.1,ok,\ngpt-4o-mini,2,1.89,85,45.0,ok,\n...\n```\n\n## 使用示例\n\n### 基本用法\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.openai.com/v1\" \\\n  --key \"sk-xxx\" \\\n  --models \"gpt-4o-mini,gpt-4o\"\n```\n\n### 批量测试（JSON 数组格式）\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.example.com/v1\" \\\n  --key \"sk-key\" \\\n  --models '[\"gpt-4o-mini\",\"gpt-4o\",\"claude-3-haiku\"]' \\\n  --iterations 5 \\\n  --max-tokens 256\n```\n\n### Streaming 模式\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.example.com/v1\" \\\n  --key \"sk-key\" \\\n  --models \"gpt-4o-mini\" \\\n  --stream \\\n  --iterations 3\n```\n\n### 自定义提示词\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.example.com/v1\" \\\n  --key \"sk-key\" \\\n  --models \"gpt-4o-mini\" \\\n  --test-prompt \"Explain quantum computing in 3 sentences.\" \\\n  --system-prompt \"You are a physics expert.\"\n```\n\n### 完整参数示例\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.example.com/v1\" \\\n  --key \"sk-key\" \\\n  --models \"gpt-4o-mini,gpt-4o\" \\\n  --iterations 5 \\\n  --max-tokens 512 \\\n  --timeout 120 \\\n  --stream false \\\n  --output \"my-throughput-report.md\" \\\n  --csv\n```\n\n## 错误处理\n\n| 场景 | 处理方式 |\n|------|---------|\n| API Key 为空 | 输出错误并退出（exit code 1）|\n| URL 无响应 | 标记为 error，记录错误，继续下一个 |\n| 模型不存在 | 标记为 http_404，记录错误 |\n| 网络超时 | 标记为 timeout，继续下一个 |\n| 部分失败 | 汇总中显示 Error Rate |\n\n## 技术实现\n\n- **HTTP 库**：`urllib`（Python 内置，无需额外安装）\n- **计时精度**：`time.perf_counter()` 纳秒级精度\n- **Token 计数**：优先用 `usage.completion_tokens`，fallback 为字符数估算\n- **Streaming**：支持 `--stream` 模式\n- **并发**：当前版本为顺序执行\n\nFile v1.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn7dzk85gz4pz5c569zky2pssn875nwe\",\n  \"slug\": \"model-throughput-tester\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1780852417012\n}\n\nFile v1.0.0:skill-card.md\n\n## Description: <br>\nTests OpenAI-compatible chat completion APIs across one or more models and reports latency, throughput, output tokens, and error rates. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[tsag1](https://clawhub.ai/user/tsag1) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and engineers use this skill to compare throughput and reliability of OpenAI-compatible model endpoints before selecting models or checking API performance. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Review before execution as proposals could introduce incorrect or misleading guidance into skills. <br>\nMitigation: Review and scan skill before deployment. <br>\n\n## Reference(s): <br>\n- [ClawHub release page](https://clawhub.ai/tsag1/model-throughput-tester) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [Text, Markdown, CSV, Shell commands, Guidance] <br>\n**Output Format:** [Terminal output plus a Markdown report, with optional CSV output.] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Requires a user-provided API base URL, API key, and model list; use a limited-scope test key and avoid sensitive prompts.] <br>\n\n## Skill Version(s): <br>\n1.0.0 (source: server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>","readmeExcerpt":"Skill: Model Throughput Tester Owner: tsag1 Summary: Automation skill for Model Throughput Tester. Tags: latest:1.0.8 Version history: v1.0.8 | 2026-07-06T03:27:36.837Z | user Exclude cold-start iteration from summary and disable thinking for pure inference throughput v1.0.7 | 2026-07-02T00:24:00.127Z | user v1.0.7: throughput.py 性能优化与稳定性修复 v1.0.6 | 2026-06-10T12:53:34.651Z | user Added Chinese trigger words and tags","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"python3 throughput.py --auto --model \"<current session model>\""},{"language":"bash","snippet":"python3 throughput.py --auto"},{"language":"bash","snippet":"python3 throughput.py --auto --model \"zai/glm-5-turbo\""},{"language":"bash","snippet":"python3 throughput.py \\\n  --url https://api.example.com/v1 \\\n  --key sk-xxx \\\n  --models gpt-4o-mini,gpt-4o"},{"language":"text","snippet":"1. Read current session model from openclaw.json (provider/model)\n2. Send test prompt via openclaw infer model run\n3. Timer: command start → output complete\n4. Estimate token count from response text (English: 0.75 word/token, Chinese: 1.5 chars/token)\n5. Calculate tokens/s\n6. Generate summary report"},{"language":"text","snippet":"1. Build /v1/chat/completions request\n2. Timer: request start → last token received\n3. Extract usage.completion_tokens from response (precise)\n4. Calculate tokens/s, error rate\n5. Generate summary report"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: model-throughput-tester\nname_zh: 吞吐率 测试 · 模型速度对比\ntags: [model-throughput-tester, 吞吐率测试, 模型速度对比]\ndescription: Benchmark LLM model throughput — measure tokens/s, latency, and output speed. Supports auto mode (no API key needed) via openclaw infer, or direct API mode for OpenAI-compatible endpoints. Trigger: throughput test, tokens/s, latency test, benchmark, speed test, model test.\ndescription_zh: AI 模型速度对比工具。一句话测出哪个模型更快、延迟更低、吞吐率更高。支持无 Key 的 auto 模式（openclaw infer）和 OpenAI 兼容 API 直连。对比多个模型的 tokens/s、响应延迟、输出速度，生成可视化报告。换模型前先跑个基线，不花冤枉钱。\ntriggerWords:\n  - 模型 哪个快\n  - 模型 速度 测试\n  - tokens/s 对比\n  - 吞吐率 测试\n  - 延迟 测试\n  - 模型 换哪个\n  - 测一下 模型 速度\n  - throughput test\n  - tokens/s\n  - speed test\n  - latency test\n  - model test\n  - 测速\n  - benchmark\nmetadata:\n  openclaw:\n    requires:\n      bins: [python3]\n    tags: [model-benchmark, throughput, tokens-per-second, latency, AI-speed, LLM, model-comparison, performance, speed-test, benchmark, 速度测试, 吞吐率, 模型对比, AI性能, 延迟测试, tokens/s, LLM测速, 模型测速]\n    permissions:\n      file:\n        read: [\"~/.openclaw/workspace/skills/model-throughput-tester/**\"]\n        write: [\"~/.openclaw/workspace/skills/model-throughput-tester/**\"]\n---\n\n# Model Throughput Tester\n\nBenchmark LLM model throughput (tokens/s). Two modes available:\n\n- **Auto Mode**: Test current model via `openclaw infer model run`, **no API key required**\n- **API Mode**: Direct call to OpenAI-compatible API, requires URL and Key\n\n## When to Use\n\n**Use when:** User explicitly requests a model throughput test.\n\n**Trigger words:**\n- throughput test, tokens/s, speed test, benchmark\n- model speed, latency test, model test\n\n**Do NOT trigger:** Broad performance discussion terms (e.g. \"model performance\", standalone \"benchmark\") should not auto-trigger execution.\n\n**Auto Mode (no API key):**\n```bash\npython3 throughput.py --auto --model \"<current session model>\"\n```\n\n## Core Features\n\n### 1. Auto Mode (No Key, Recommended)\n\n```bash\npython3 throughput.py --auto\n```\n\nTest a specific model:\n```bash\npython3 throughput.py --auto --model \"zai/glm-5-turbo\"\n```\n\n### 2. API Mode (Direct API Call)\n\n```bash\npython3 throughput.py \\\n  --url https://api.example.com/v1 \\\n  --key sk-xxx \\\n  --models gpt-4o-mini,gpt-4o\n```\n\n### 3. Common Parameters\n\n| Parameter | Default | Description |\n|-----------|---------|-------------|\n| `--iterations` | `3` | Test iterations per model |\n| `--max-tokens` | `512` | Max output tokens |\n| `--test-prompt` | English prose (summer field) | Test prompt |\n| `--timeout` | `60` | Single request timeout (seconds) |\n| `--output` | `throughput-report.md` | Output report filename |\n| `--csv` | false | Also generate CSV |\n\n## Workflow\n\n### Auto Mode Flow\n\n```\n1. Read current session model from openclaw.json (provider/model)\n2. Send test prompt via openclaw infer model run\n3. Timer: command start → output complete\n4. Estimate token count from response text (English: 0.75 word/token, Chinese: 1.5 chars/token)\n5. Calculate tokens/s\n6. Generate summary report\n```\n\n#"},{"path":"README.md","content":"# Model Throughput Tester\n\nBenchmark LLM model throughput — measure tokens/s, latency, and output speed for any language model.\n\n## Features\n\n- **Auto Mode**: Test your current session model via `openclaw infer`, no API key needed\n- **API Mode**: Direct benchmark against any OpenAI-compatible endpoint\n- **Flexible**: Custom prompts, iteration counts, timeout controls\n- **Reports**: Markdown + CSV output with per-iteration details\n\n## Quick Start\n\n```bash\n# Auto mode — test current session model\npython3 throughput.py --auto\n\n# Test a specific model\npython3 throughput.py --auto --model \"gpt-4o\"\n\n# API mode — test against an endpoint\npython3 throughput.py \\\n  --url \"https://api.openai.com/v1\" \\\n  --key \"sk-xxx\" \\\n  --models \"gpt-4o-mini,gpt-4o\" \\\n  --iterations 5\n```\n\n## Parameters\n\n| Parameter | Default | Description |\n|-----------|---------|-------------|\n| `--auto` | off | Enable auto mode (uses openclaw infer) |\n| `--model` | auto-detect | Model identifier |\n| `--url` | — | API base URL (API mode) |\n| `--key` | — | API key (API mode) |\n| `--models` | — | Comma-separated model list (API mode) |\n| `--iterations` | `3` | Test iterations per model |\n| `--max-tokens` | `512` | Max output tokens |\n| `--test-prompt` | built-in | Custom test prompt |\n| `--timeout` | `60` | Request timeout (seconds) |\n| `--output` | `throughput-report.md` | Output report filename |\n| `--csv` | false | Also generate CSV output |\n\n## Metrics\n\n| Metric | Description |\n|--------|-------------|\n| **Tokens/s** | Throughput = Output Tokens / Elapsed Time |\n| **Avg Latency** | Average single-request latency |\n| **Avg Output Tokens** | Average output token count |\n| **Error Rate** | Failed request ratio |\n\n## Example Output\n\n```\n📊 Model Throughput Report\nMode: Auto (openclaw infer) | Iterations: 3\n\nSummary\n| Model             | Avg Tokens/s | Latency(s) | Output Tokens | Error |\n|-------------------|-------------|------------|----------------|-------|\n| zai/glm-5-turbo   | 57.9        | 20.6       | 979            | 0.0%  |\n```\n\n## How It Works\n\n**Auto Mode**: Sends a test prompt via `openclaw infer model run`, measures wall-clock time from start to last token, then estimates token count from output text.\n\n**API Mode**: Calls `/v1/chat/completions` with streaming disabled, reads `usage.completion_tokens` for precise token counts.\n\n## Notes\n\n- Auto mode throughput includes gateway routing overhead (~1-3% lower than direct API)\n- Auto mode token counts are estimates; API mode uses precise values\n- English prompts yield more accurate token estimates in auto mode\n- Anti-cache: random seed suffix appended per iteration\n\n## Prerequisites\n\n- Python 3 (built-in on macOS)\n- `openclaw` CLI (for auto mode)\n\n## File Structure\n\n```\n~/.openclaw/workspace/skills/model-throughput-tester/\n├── SKILL.md           # Agent trigger & execution guide\n├── README.md          # This file\n├── README.zh.md       # Chinese version\n├── throughput.py      # Main script\n└── throughput-report.md  # Generated re"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7dzk85gz4pz5c569zky2pssn875nwe\",\n  \"slug\": \"model-throughput-tester\",\n  \"version\": \"1.0.8\",\n  \"publishedAt\": 1783308456837\n}"},{"path":"README.zh.md","content":"# 模型吞吐率测试器\n\n测试 LLM 模型的吞吐率（tokens/s）。支持两种模式：\n\n- **Auto 模式**：通过 `openclaw infer model run` 测试当前模型，**无需 API Key**\n- **API 模式**：直接调用 OpenAI 兼容 API，需要 URL 和 Key\n\n## 触发规则\n\n**适用场景：** 用户明确要求测试模型吞吐率时使用。\n\n**推荐触发词：**\n- 测一下吞吐率、测速、模型测速、tokens/s\n- 跑个 benchmark、吞吐率测试、模型测试\n\n**不适用：** 宽泛的性能讨论词（如「模型性能」「benchmark」单独出现）不应自动触发执行。\n\n## 核心能力\n\n### 1. Auto 模式（无 Key，推荐）\n\n自动检测当前 session 的模型并测试吞吐率，无需任何配置。\n\n```bash\npython3 throughput.py --auto\n```\n\n指定模型测试：\n```bash\npython3 throughput.py --auto --model \"zai/glm-5-turbo\"\n```\n\n### 2. API 模式（直接调用 API）\n\n```bash\npython3 throughput.py \\\n  --url https://api.example.com/v1 \\\n  --key sk-xxx \\\n  --models gpt-4o-mini,gpt-4o\n```\n\n### 3. 通用参数\n\n| 参数 | 默认值 | 说明 |\n|------|--------|------|\n| `--iterations` | `3` | 每个模型测试次数 |\n| `--max-tokens` | `512` | 最大输出 token 数 |\n| `--test-prompt` | 英文散文（夏天的田野） | 测试提示词 |\n| `--timeout` | `60` | 单次请求超时（秒） |\n| `--output` | `throughput-report.md` | 输出报告文件名 |\n| `--csv` | false | 同时生成 CSV |\n\n## 工作流程\n\n### Auto 模式\n\n```\n1. 从 openclaw.json 读取当前 session 模型（provider/model）\n2. 通过 openclaw infer model run 发送测试 prompt\n3. 计时：命令开始 → 输出完成\n4. 从返回文本估算 token 数（英文 0.75 word/token，中文 1.5 字/token）\n5. 计算 tokens/s\n6. 汇总输出报告\n```\n\n### API 模式\n\n```\n1. 构造 /v1/chat/completions 请求\n2. 计时：请求开始 → 最后一个 token\n3. 从响应中提取 usage.completion_tokens（精确）\n4. 计算 tokens/s、错误率\n5. 汇总输出报告\n```\n\n## 指标说明\n\n| 指标 | 说明 |\n|------|------|\n| **Tokens/s** | 吞吐率 = Output Tokens / Elapsed Time |\n| **Avg Latency** | 平均单次请求延迟 |\n| **Avg Output Tokens** | 平均输出 token 数 |\n| **Error Rate** | 错误请求占比 |\n\n## 使用示例\n\n### 安装后立即测试（Auto 模式）\n\n```bash\n# agent 触发时应传入当前模型\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto --model \"<当前session模型>\"\n\n# 或使用自动检测（可能不是 session 覆盖的模型）\npython3 ~/.openclaw/workspace/skills/model-throughput-tester/throughput.py --auto\n```\n\n### 测试多个模型（API 模式）\n\n```bash\npython3 throughput.py \\\n  --url \"https://api.openai.com/v1\" \\\n  --key \"sk-xxx\" \\\n  --models \"gpt-4o-mini,gpt-4o\" \\\n  --iterations 5\n```\n\n### 自定义提示词\n\n```bash\npython3 throughput.py --auto \\\n  --test-prompt \"Explain quantum computing in detail.\" \\\n  --iterations 5\n```\n\n## 技术实现\n\n- **Auto 模式**：`openclaw infer model run --json`，Python `subprocess` 调用\n- **API 模式**：`urllib`（Python 内置），OpenAI 兼容 `/v1/chat/completions`\n- **计时精度**：`time.perf_counter()` 纳秒级精度\n- **Token 计数**：API 模式优先 `usage.completion_tokens`（精确），Auto 模式按字符估算\n- **URL 拼接**：智能检测 `/v1`、`/v4`、`/chat/completions` 路径\n\n## 注意事项\n\n- Auto 模式的吞吐率包含网关路由开销，会比直接 API 略低（约 1-3%）\n- Auto 模式 Token 数为估算值，API 模式为精确值\n- 建议使用英文 prompt 以获得更准确的 token 估算\n- 防缓存：每次迭代自动附加随机 seed 后缀"},{"path":"skill-card.md","content":"## Description:\n\nBenchmark LLM model throughput by measuring tokens per second, latency, and output speed through auto mode with OpenClaw or direct API mode for OpenAI-compatible endpoints.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[tsag1](https://clawhub.ai/user/tsag1)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and engineers use this skill to benchmark and compare LLM response throughput, latency, token output, and error rate before selecting or switching models.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: API mode can send API keys and benchmark prompts to any configured endpoint, including insecure HTTP endpoints.\n\nMitigation: Prefer auto mode when possible; in API mode, use trusted HTTPS endpoints and avoid placing real keys directly in shell history.\n\nRisk: Generated reports can store the benchmark prompt and API URL locally.\n\nMitigation: Use non-sensitive benchmark prompts and review generated Markdown or CSV reports before sharing them.\n\nRisk: Auto mode estimates token counts and cannot enforce the max token limit, which can make slow models appear to fail by timeout.\n\nMitigation: Use API mode for precise completion-token counts and max-token control, or raise the timeout and use a shorter prompt for slow models in auto mode.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/tsag1/skills/model-throughput-tester)\n- [README](artifact/README.md)\n- [Skill definition](artifact/SKILL.md)\n- [Release evidence](evidence.json)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, shell commands, configuration]\n\n**Output Format:** [Terminal summary text plus a Markdown throughput report, with optional CSV output.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Auto mode estimates token counts and excludes warmup from summary by default; API mode uses reported completion tokens when available.]\n\n## Skill Version(s):\n\n1.0.8 (source: server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":1422,"uniquenessScore":41,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T19:49:18.188Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T19:49:18.188Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T21:51:31.598Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}