{"id":"473d15d7-e7cc-4f6e-a556-df39824d8a7b","entityType":"agent","slug":"clawhub-dataify-server-dataify-github-repository-by-repo-url","name":"Dataify Github Builder","canonicalUrl":"https://www.xpersona.co/agent/clawhub-dataify-server-dataify-github-repository-by-repo-url","canonicalPath":"/agent/clawhub-dataify-server-dataify-github-repository-by-repo-url","generatedAt":"2026-10-11T10:49:28.535Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-11T08:11:10.735Z","emptyReason":null},"description":"Collect Github Builder data and return results Skill: Dataify Github Builder Owner: dataify-server Summary: Collect Github Builder data and return results Tags: latest:1.3.1 Version history: v1.3.1 | 2026-09-08T06:18:16.065Z | user Fix natural-language usage failures: validate required targets and URLs, preserve catalog references, default Amazon region safely, normalize Google News links, and improve UTF-8 error output. v1.3.0 | 2026-09-01T09:13:31.285Z | user 默","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.1K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s17feed8b2qc486skqmjapxmjs86bd4f:dataify-github-repository-by-repo-url","sourceUrl":"https://clawhub.ai/dataify-server/dataify-github-repository-by-repo-url","homepage":"https://clawhub.ai/dataify-server/skills/dataify-github-repository-by-repo-url","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/dataify-server/dataify-github-repository-by-repo-url","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/dataify-server/skills/dataify-github-repository-by-repo-url","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":61,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Collect Github Builder data and return results Skill: Dataify Github Builder Owner: dataify-server Summary: Collect Github Builder data and return results Tags:"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T08:11:10.735Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T08:11:10.735Z","emptyReason":null},"stars":null,"forks":null,"downloads":1116,"packageName":null,"latestVersion":"1.3.1","tractionLabel":"1.1K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T08:11:10.675Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T08:11:10.735Z","lastCrawledAt":"2026-10-11T08:11:10.675Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T08:11:10.675Z","lastVerifiedAt":null,"highlights":[{"version":"1.3.1","createdAt":"2026-09-08T06:18:16.065Z","changelog":"Fix natural-language usage failures: validate required targets and URLs, preserve catalog references, default Amazon region safely, normalize Google News links, and improve UTF-8 error output.","fileCount":14,"zipByteSize":30725},{"version":"1.3.0","createdAt":"2026-09-01T09:13:31.285Z","changelog":"默认返回最终采集结果，完善异步等待下载与安全恢复；修复空目标误执行、Quick Start、Token 配置和触发路由冲突，并补齐发布前自动化测试","fileCount":8,"zipByteSize":9889},{"version":"1.2.0","createdAt":"2026-07-16T08:37:08.505Z","changelog":"新增能力路由与异步任务闭环，修复失效引用、Token 泄漏和 Skill 元数据兼容性","fileCount":8,"zipByteSize":7974},{"version":"1.1.0","createdAt":"2026-06-08T02:06:20.040Z","changelog":"补全中文文档，更新目录结构","fileCount":6,"zipByteSize":7503},{"version":"1.0.0","createdAt":"2026-05-28T08:12:33.354Z","changelog":"Initial release of Dataify builder skill for GitHub repository scraping: - Guides users through preparing Dataify builder requests for github.com scraper tools by repository URL. - Supports environment-based API token checks and detailed setup instructions for Windows, macOS, and Linux. - Allows users to select specific scraping tools and enter or select required tool parameters with Chinese labels. - Dynamically builds JSON arrays of spider parameters, supporting single and multi-value workflows. - Outputs ready-to-use curl commands targeting https://scraperapi.dataify.com/builder with correct parameters. - Provides script usage examples and references for tool parameter configuration and request generation.","fileCount":6,"zipByteSize":7170}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17feed8b2qc486skqmjapxmjs86bd4f:dataify-github-repository-by-repo-url","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-dataify-server-dataify-github-repository-by-repo-url/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-dataify-server-dataify-github-repository-by-repo-url/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-dataify-server-dataify-github-repository-by-repo-url/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-dataify-server-dataify-github-repository-by-repo-url/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-dataify-server-dataify-github-repository-by-repo-url/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-dataify-server-dataify-github-repository-by-repo-url/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T10:49:28.533Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-dataify-server-dataify-github-repository-by-repo-url/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-dataify-server-dataify-github-repository-by-repo-url/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-dataify-server-dataify-github-repository-by-repo-url/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-dataify-server-dataify-github-repository-by-repo-url/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-11T08:11:10.735Z","emptyReason":null},"readme":"Skill: Dataify Github Builder\n\nOwner: dataify-server\n\nSummary: Collect Github Builder data and return results\n\nTags: latest:1.3.1\n\nVersion history:\n\nv1.3.1 | 2026-09-08T06:18:16.065Z | user\n\nFix natural-language usage failures: validate required targets and URLs, preserve catalog references, default Amazon region safely, normalize Google News links, and improve UTF-8 error output.\n\nv1.3.0 | 2026-09-01T09:13:31.285Z | user\n\n默认返回最终采集结果，完善异步等待下载与安全恢复；修复空目标误执行、Quick Start、Token 配置和触发路由冲突，并补齐发布前自动化测试\n\nv1.2.0 | 2026-07-16T08:37:08.505Z | user\n\n新增能力路由与异步任务闭环，修复失效引用、Token 泄漏和 Skill 元数据兼容性\n\nv1.1.0 | 2026-06-08T02:06:20.040Z | user\n\n补全中文文档，更新目录结构\n\nv1.0.0 | 2026-05-28T08:12:33.354Z | auto\n\nInitial release of Dataify builder skill for GitHub repository scraping:\n\n- Guides users through preparing Dataify builder requests for github.com scraper tools by repository URL.\n- Supports environment-based API token checks and detailed setup instructions for Windows, macOS, and Linux.\n- Allows users to select specific scraping tools and enter or select required tool parameters with Chinese labels.\n- Dynamically builds JSON arrays of spider parameters, supporting single and multi-value workflows.\n- Outputs ready-to-use curl commands targeting https://scraperapi.dataify.com/builder with correct parameters.\n- Provides script usage examples and references for tool parameter configuration and request generation.\n\nArchive index:\n\nArchive v1.3.1: 14 files, 30725 bytes\n\nFiles: agents/openai.yaml (378b), references/tool-params.json (472b), scripts/build-dataify-request.ps1 (294b), scripts/build-dataify-request.py (730b), scripts/business_workflow.py (34775b), scripts/catalog_builder.py (7013b), scripts/dataify_client.py (6994b), scripts/task_runtime.py (2259b), scripts/token_setup.py (2585b), scripts/wait_for_task.py (8879b), skill-card.md (2456b), SKILL.md (8407b), SKILL.zh-CN.md (6609b), _meta.json (156b)\n\nFile v1.3.1:SKILL.md\n\n---\nname: \"dataify-github-repository-by-repo-url\"\ndescription: \"Collect structured GitHub repository information from one or more known repository URLs. Do not use for GitHub code search or arbitrary webpages.\"\n---\n\n# Dataify Builder Skill\n\nUse this skill to prepare Dataify builder requests for the scraper family rooted at `github_repository_by-repo-url` on `github.com`.\n\n\n## Quick Start\n\n**Input:** a GitHub repository URL.\n\n```bash\npython3 scripts/build-dataify-request.py --tool-sign github_repository_by-repo-url --params-json '[{\"repo_url\":\"https://github.com/dataify-server/skills\"}]'\n```\n\nThis submits the task, waits for completion, downloads the final result, and returns it. Add `--no-wait` only when submission-only behavior is requested.\n## Workflow\n\n1. Check whether `DATAIFY_API_TOKEN` exists in the environment.\n2. If the token is missing, stop and tell the user to sign in at [Dataify Dashboard](https://dashboard.dataify.com?utm_source=skill)  to obtain it.\n3. Ask the user to choose exactly one tool from the following Chinese list:\n- 通过仓库URL采集 (github_repository_by-repo-url)\n- 通过搜索URL采集 (github_repository_by-search-url)\n- 通过URL采集 (github_repository_by-url)\n4. Read `references/tool-params.json` and find the chosen tool by `tool_sign` or Chinese tool name.\n5. For each parameter in the chosen tool:\n   - If `input_mode` is `user_input`, ask the user for the value.\n   - If `input_mode` is `select`, present the saved options to the user.\n6. Use `scripts/build-dataify-request.py` as the default cross-platform helper.\n7. Use `scripts/build-dataify-request.ps1` as the Windows PowerShell helper when needed.\n8. When a selectable parameter has a human-readable Chinese label, keep that label in `spider_parameters`. Do not replace it with a code such as `HK` unless the user explicitly asks for the coded value.\n9. Build `spider_parameters` as a JSON array.\n10. If every parameter has only one final value, build one object such as `[{\"searchurl\":\"...\",\"country\":\"Hong Kong\"}]`.\n11. If one or more parameters have multiple aligned values, zip them by index and build one object per row. Example: `[{\"search_url\":\"url1\",\"page_turning\":\"1\",\"max_num\":\"15\"},{\"search_url\":\"url2\",\"page_turning\":\"1\",\"max_num\":\"15\"}]`.\n12. If a parameter has one value while another parameter has multiple values, reuse the single value across every generated row.\n13. Set `spider_name` to `github.com`.\n14. Set `spider_id` to the selected tool's `tool_sign`.\n15. Always include `spider_errors=true` and `file_name={{TasksID}}`.\n16. Return a curl command for `https://scraperapi.dataify.com/builder`.\n\n## Set DATAIFY_API_TOKEN\n\nPrefer a permanent environment-variable setup instead of setting the token only for the current terminal session.\n\nWindows PowerShell, permanent for the current user:\n```powershell\n[Environment]::SetEnvironmentVariable(\"DATAIFY_API_TOKEN\", \"your_token_here\", \"User\")\n```\n\nThen reopen PowerShell. If the current session also needs the token immediately, run:\n```powershell\n$env:DATAIFY_API_TOKEN = \"your_token_here\"\n```\n\nmacOS or Linux, permanent for bash:\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.bashrc\nsource ~/.bashrc\n```\n\nmacOS or Linux, permanent for zsh:\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.zshrc\nsource ~/.zshrc\n```\n\n## Script usage\n\nPython:\n```bash\npython scripts/build-dataify-request.py --tool-sign <selected_tool_sign> --values-file values.json\n```\n\nPowerShell:\n```powershell\n& \".\\scripts\\build-dataify-request.ps1\" -ToolSign \"<selected_tool_sign>\" -ValuesFile \".\\values.json\"\n```\n\nThe `values.json` file should contain either one object or an array of objects. Example:\n```json\n[{\"searchurl\":\"https://www.airbnb.com/s/Greece/homes?...\",\"country\":\"Hong Kong\"}]\n```\n\n## Required output shape\n\nGenerate a curl command in this form:\n\n```bash\ncurl -X POST 'https://scraperapi.dataify.com/builder' \\\n  -H \"Authorization: Bearer $DATAIFY_API_TOKEN\" \\\n  -H 'Content-Type: application/x-www-form-urlencoded' \\\n  -d 'spider_name=github.com' \\\n  -d 'spider_id=<selected_tool_sign>' \\\n  -d 'spider_parameters=[{\"param\":\"value\"}]' \\\n  -d 'spider_errors=true' \\\n  -d 'file_name={{TasksID}}'\n```\n\n## Reference usage\n\n- `references/tool-params.json` stores the full saved parameter catalog for every available tool in this scraper family.\n- `scripts/build-dataify-request.py` is the portable implementation and should be preferred.\n- `scripts/build-dataify-request.ps1` mirrors the same behavior for Windows users.\n- If a parameter has no options, the user must provide the value.\n- Do not assume `spider_parameters` always contains exactly one object. Multi-value tools may require multiple objects zipped by index.\n- Use the saved `url_example` only as a reference example. Do not assume the user wants the example values unless they explicitly confirm them.\n\n## Default completion behavior\n\nThe default deliverable is the collected result, not only a `task_id`.\n\n1. Submit the Builder task once and capture its `task_id`.\n2. Immediately continue with `$dataify-task-operations` and monitor the same task ID.\n   - Use the default 600-second wait for ordinary collections.\n   - Use `--timeout 1800` for media downloads or clearly high-volume, multi-page, or multi-input collections.\n3. When the task succeeds, download and return the final JSON result. Summarize large payloads while preserving access to the raw result.\n4. If monitoring times out or is interrupted, return the task ID and a resume command. Do not resubmit the paid task.\n5. Stop after submission only when the user explicitly asks for submission only, a task ID, or `--no-wait` behavior.\n\n## Parameter interaction policy\n\n- For a clear, low-risk, read-only, and low-cost request, apply safe defaults and execute immediately. A short execution summary is optional; do not pause for confirmation.\n- Ask only for a missing required input, a material ambiguity, a high-volume or multi-page scope, a media download, a choice that materially changes credit usage, an irreversible action, or an explicit user request to review parameters.\n- When confirmation is required, show only user-facing values that affect the target, scope, output, or cost. Prefer one concise sentence; use a compact table only when three or more consequential values are easier to compare.\n- Never show fixed fields, empty optional fields, unchanged defaults, credentials, or internal implementation parameters such as engine selectors, response-format flags, offsets, spider IDs, and file-name templates.\n- Keep advanced filters hidden unless the user asks for them or they are needed to resolve ambiguity. Never substitute documentation example values for missing required user input.\n- After returning results, offer relevant refinements instead of forcing all optional decisions before the first result.\n\n## Account CTA policy\n\n- Show a prominent Dataify account CTA only when the API token is missing, rejected/invalid, or the account has insufficient credits.\n- For a missing token, offer https://dashboard.dataify.com/login?utm_source=skill and state: New accounts get 50 free credits, enough for about 6,000 trial results, valid for 7 days, and only successful requests are billed. Never ask the user to paste the token into chat.\n- Detect the current operating system and shell. Show only the matching session-scoped setup command first (`export` for macOS/Linux shells, `$env:` for Windows PowerShell, or `set` for Windows Command Prompt). Show other platforms or persistent setup only when detection is ambiguous or the user asks.\n- After the user says the token is configured, verify only whether `DATAIFY_API_TOKEN` is present; never print its value. If verification succeeds, continue the original task without asking the user to repeat it.\n- Explain that persistent shell changes may require a new terminal or restarting the agent application. Do not recommend a project `.env` unless the execution path explicitly loads it, and ensure `.env` is ignored by version control.\n- For an invalid token, direct the user to API-key management without implying that a new registration is required. For insufficient credits, direct the user to balance or recharge management.\n- During normal submission, processing, and successful completion, do not promote registration or the Dashboard. Never expose the token or include it in CTA attribution parameters.\n\nFile v1.3.1:_meta.json\n\n{\n  \"ownerId\": \"kn74z5hmmwk21kw8tpphd9w21x86bkdf\",\n  \"slug\": \"dataify-github-repository-by-repo-url\",\n  \"version\": \"1.3.1\",\n  \"publishedAt\": 1788848296065\n}\n\nFile v1.3.1:references/tool-params.json\n\n[{\"tool_name_cn\":\"通过仓库URL采集\",\"tool_sign\":\"github_repository_by-repo-url\",\"spider_name\":\"github.com\",\"params\":[{\"param\":\"repo_url\",\"required\":true,\"input_mode\":\"user_input\",\"description\":\"Public GitHub repository URL\"}]},{\"tool_name_cn\":\"通过搜索URL采集\",\"tool_sign\":\"github_repository_by-search-url\",\"spider_name\":\"github.com\",\"params\":[]},{\"tool_name_cn\":\"通过URL采集\",\"tool_sign\":\"github_repository_by-url\",\"spider_name\":\"github.com\",\"params\":[]}]\n\nFile v1.3.1:skill-card.md\n\n## Description:\n\nCollect structured GitHub repository information from one or more known repository URLs. Do not use for GitHub code search or arbitrary webpages.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[dataify-server](https://clawhub.ai/user/dataify-server)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and agents use this skill to collect structured data for known public GitHub repository URLs through Dataify, monitor the asynchronous task, and return the final JSON result.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Repository targets and task results are sent through Dataify using the user's API token.\n\nMitigation: Install only when this data flow is acceptable; keep the token in the environment and do not paste it into chat or generated outputs.\n\nRisk: The package contains broader business, search, and web unlocker workflows beyond the narrow GitHub repository collection purpose.\n\nMitigation: Review the installed artifact and prefer a narrowed release that removes unrelated workflows before using it in sensitive environments.\n\nRisk: The security guidance flags scoping and credential-handling issues, including GitHub URL validation, import path behavior, curl quoting, and reusable tokens in URLs.\n\nMitigation: Validate target URLs as GitHub repository URLs and resolve the flagged implementation issues before production deployment.\n\n## Reference(s):\n\n- [ClawHub Skill Page](https://clawhub.ai/dataify-server/skills/dataify-github-repository-by-repo-url)\n- [Publisher Profile](https://clawhub.ai/user/dataify-server)\n- [Tool Parameter Catalog](references/tool-params.json)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown guidance with shell commands and JSON results from completed Dataify tasks]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Requires a DATAIFY_API_TOKEN environment variable and a public GitHub repository URL; default behavior waits for completion and returns the final collected result.]\n\n## Skill Version(s):\n\n1.3.1 (source: server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.3.1:SKILL.zh-CN.md\n\n---\nname: \"dataify-github-repository-by-repo-url\"\ndescription: \"为 github.com 上以 github_repository_by-repo-url 为根的 scraper 系列准备 Dataify builder 请求。当需要处理成功的 Dataify scraper detail 条目 github_repository_by-repo-url、让用户选择可用工具、读取已保存的 getToolParams 选项，并使用 DATAIFY_API_TOKEN 生成 scraperapi.dataify.com/builder curl 请求时，使用此 skill。\"\n---\n\n# Dataify Builder Skill 中文版\n\n这个 skill 用于为 `github.com` 下、以 `github_repository_by-repo-url` 为入口的 Dataify scraper 工具族生成 builder 请求。\n\n## 工作流程\n\n1. 先检查环境变量中是否存在 `DATAIFY_API_TOKEN`。\n2. 如果 token 缺失，告诉用户：`Dataify 需要 API Token。新账号注册即得 50 免费积分，约可获得 6000 条试用结果，7 天有效，仅成功请求计费。注册完成后告诉我，我会继续当前任务。`。\n3. 先让用户从下面的中文工具列表中明确选择一个工具：\n- 通过仓库URL采集 (github_repository_by-repo-url)\n- 通过搜索URL采集 (github_repository_by-search-url)\n- 通过URL采集 (github_repository_by-url)\n4. 再读取 `references/tool-params.json`，根据 `tool_sign` 或中文工具名找到对应工具。\n5. 对所选工具的每个参数分别处理：\n   - 如果 `input_mode` 是 `user_input`，让用户提供值。\n   - 如果 `input_mode` 是 `select`，把已保存的可选项展示给用户，让用户选择。\n6. 默认优先使用 `scripts/build-dataify-request.py`，因为它是跨平台版本。\n7. Windows 下也可以使用 `scripts/build-dataify-request.ps1`。\n8. 对于可选型参数，如果存在人类可读标签，优先把该标签写入 `spider_parameters`。\n9. `spider_parameters` 必须是一个 JSON 数组。\n10. 像搜索 URL 这类多值工具，需要按索引生成多个对象。\n11. `spider_name` 固定取 `github.com`。\n12. `spider_id` 固定取用户所选工具的 `tool_sign`。\n13. 始终包含 `spider_errors=true` 和 `file_name={{TasksID}}`。\n\n## 设置 DATAIFY_API_TOKEN\n\n推荐使用永久环境变量，而不是只在当前终端临时设置。\n\nWindows PowerShell，当前用户永久设置：\n\n```powershell\n[Environment]::SetEnvironmentVariable(\"DATAIFY_API_TOKEN\", \"your_token_here\", \"User\")\n```\n\n然后重新打开 PowerShell。如果当前会话也要立即生效，再执行：\n\n```powershell\n$env:DATAIFY_API_TOKEN = \"your_token_here\"\n```\n\nmacOS 或 Linux，bash 永久设置：\n\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.bashrc\nsource ~/.bashrc\n```\n\nmacOS 或 Linux，zsh 永久设置：\n\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.zshrc\nsource ~/.zshrc\n```\n\n## 脚本用法\n\nPython：\n\n```bash\npython scripts/build-dataify-request.py --tool-sign <selected_tool_sign> --values-file values.json\n```\n\nPowerShell：\n\n```powershell\n& \".\\scripts\\build-dataify-request.ps1\" -ToolSign \"<selected_tool_sign>\" -ValuesFile \".\\values.json\"\n```\n\n`values.json` 可以是单个对象，也可以是对象数组。\n\n## 输出格式\n\n最终 `curl` 命令应为：\n\n```bash\ncurl -X POST 'https://scraperapi.dataify.com/builder' \\\n  -H \"Authorization: Bearer $DATAIFY_API_TOKEN\" \\\n  -H 'Content-Type: application/x-www-form-urlencoded' \\\n  -d 'spider_name=github.com' \\\n  -d 'spider_id=<selected_tool_sign>' \\\n  -d 'spider_parameters=[{\"param\":\"value\"}]' \\\n  -d 'spider_errors=true' \\\n  -d 'file_name={{TasksID}}'\n```\n\n## 参考文件\n\n- `references/tool-params.json` 保存了这个 skill 下所有工具及参数选项。\n- `scripts/build-dataify-request.py` 是首选的跨平台实现。\n- `scripts/build-dataify-request.ps1` 是 Windows PowerShell 版本。\n- 如果参数没有预设选项，必须向用户要值。\n- 不要假设 `spider_parameters` 永远只有一个对象；多值工具可能需要按索引生成多个对象。\n- `url_example` 仅作为参考，不要默认用户就要用示例值，除非用户明确确认。\n\n## 参数交互策略\n\n- 当请求意图明确、只读、低风险且成本较低时，使用安全默认值直接执行。可以用一句话说明执行内容，但不要暂停等待确认。\n- 只在缺少必填输入、存在会明显改变结果的歧义、大批量或多页采集、媒体下载、会明显增加积分消耗、不可逆操作，或用户明确要求查看参数时询问。\n- 必须确认时，只展示会影响目标、范围、输出或成本的用户参数。优先使用一句简短说明；只有三个及以上关键值确实需要比较时才使用精简表格。\n- 不要展示固定字段、空的可选字段、未修改的默认值、凭据或内部实现参数，例如引擎选择、响应格式开关、偏移量、spider ID 和文件名模板。\n- 默认隐藏高级筛选项，除非用户主动询问或需要它们消除歧义。不得用文档示例值代替用户缺失的必填输入。\n- 先返回首个结果，再提供相关的细化选项，不要在首次执行前强迫用户决定所有可选项。\n\n## Account CTA policy\n\n- Show a prominent Dataify account CTA only when the API token is missing, rejected/invalid, or the account has insufficient credits.\n- For a missing token, offer https://dashboard.dataify.com/login?utm_source=skill and state: New accounts get 50 free credits, enough for about 6,000 trial results, valid for 7 days, and only successful requests are billed. Never ask the user to paste the token into chat.\n- Detect the current operating system and shell. Show only the matching session-scoped setup command first (`export` for macOS/Linux shells, `$env:` for Windows PowerShell, or `set` for Windows Command Prompt). Show other platforms or persistent setup only when detection is ambiguous or the user asks.\n- After the user says the token is configured, verify only whether `DATAIFY_API_TOKEN` is present; never print its value. If verification succeeds, continue the original task without asking the user to repeat it.\n- Explain that persistent shell changes may require a new terminal or restarting the agent application. Do not recommend a project `.env` unless the execution path explicitly loads it, and ensure `.env` is ignored by version control.\n- For an invalid token, direct the user to API-key management without implying that a new registration is required. For insufficient credits, direct the user to balance or recharge management.\n- During normal submission, processing, and successful completion, do not promote registration or the Dashboard. Never expose the token or include it in CTA attribution parameters.\n\nFile v1.3.1:agents/openai.yaml\n\ninterface:\n  display_name: \"Dataify Github Builder\"\n  short_description: \"Collect Github Builder data and return results\"\n  default_prompt: \"Use $dataify-github-repository-by-repo-url to complete the requested Dataify collection, wait for the asynchronous task, and return the final collected result. Stop at task submission only when I explicitly request no-wait behavior.\"\n\nArchive v1.3.0: 8 files, 9889 bytes\n\nFiles: agents/openai.yaml (378b), references/tool-params.json (365b), scripts/build-dataify-request.ps1 (294b), scripts/build-dataify-request.py (408b), skill-card.md (2353b), SKILL.md (8309b), SKILL.zh-CN.md (6414b), _meta.json (156b)\n\nFile v1.3.0:SKILL.md\n\n---\nname: \"dataify-github-repository-by-repo-url\"\ndescription: \"Collect structured GitHub repository information from one or more known repository URLs. Do not use for GitHub code search or arbitrary webpages.\"\n---\n\n# Dataify Builder Skill\n\nUse this skill to prepare Dataify builder requests for the scraper family rooted at `github_repository_by-repo-url` on `github.com`.\n\n\n## Quick Start\n\n**Input:** a GitHub repository URL.\n\n```bash\npython3 scripts/build-dataify-request.py --tool-sign github_repository_by-repo-url --params-json '[{\"url\":\"https://github.com/dataify-server/skills\"}]'\n```\n\nThis submits the task, waits for completion, downloads the final result, and returns it. Add `--no-wait` only when submission-only behavior is requested.\n## Workflow\n\n1. Check whether `DATAIFY_API_TOKEN` exists in the environment.\n2. If the token is missing, stop and tell the user to sign in at [Dataify Dashboard](https://dashboard.dataify.com?utm_source=skill)  to obtain it.\n3. Ask the user to choose exactly one tool from the following Chinese list:\n- 通过仓库URL采集 (github_repository_by-repo-url)\n- 通过搜索URL采集 (github_repository_by-search-url)\n- 通过URL采集 (github_repository_by-url)\n4. Read `references/tool-params.json` and find the chosen tool by `tool_sign` or Chinese tool name.\n5. For each parameter in the chosen tool:\n   - If `input_mode` is `user_input`, ask the user for the value.\n   - If `input_mode` is `select`, present the saved options to the user.\n6. Use `scripts/build-dataify-request.py` as the default cross-platform helper.\n7. Use `scripts/build-dataify-request.ps1` as the Windows PowerShell helper when needed.\n8. When a selectable parameter has a human-readable Chinese label, keep that label in `spider_parameters`. Do not replace it with a code such as `HK` unless the user explicitly asks for the coded value.\n9. Build `spider_parameters` as a JSON array.\n10. If every parameter has only one final value, build one object such as `[{\"searchurl\":\"...\",\"country\":\"Hong Kong\"}]`.\n11. If one or more parameters have multiple aligned values, zip them by index and build one object per row. Example: `[{\"search_url\":\"url1\",\"page_turning\":\"1\",\"max_num\":\"15\"},{\"search_url\":\"url2\",\"page_turning\":\"1\",\"max_num\":\"15\"}]`.\n12. If a parameter has one value while another parameter has multiple values, reuse the single value across every generated row.\n13. Set `spider_name` to `github.com`.\n14. Set `spider_id` to the selected tool's `tool_sign`.\n15. Always include `spider_errors=true` and `file_name={{TasksID}}`.\n16. Return a curl command for `https://scraperapi.dataify.com/builder`.\n\n## Set DATAIFY_API_TOKEN\n\nPrefer a permanent environment-variable setup instead of setting the token only for the current terminal session.\n\nWindows PowerShell, permanent for the current user:\n```powershell\n[Environment]::SetEnvironmentVariable(\"DATAIFY_API_TOKEN\", \"your_token_here\", \"User\")\n```\n\nThen reopen PowerShell. If the current session also needs the token immediately, run:\n```powershell\n$env:DATAIFY_API_TOKEN = \"your_token_here\"\n```\n\nmacOS or Linux, permanent for bash:\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.bashrc\nsource ~/.bashrc\n```\n\nmacOS or Linux, permanent for zsh:\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.zshrc\nsource ~/.zshrc\n```\n\n## Script usage\n\nPython:\n```bash\npython scripts/build-dataify-request.py --tool-sign <selected_tool_sign> --values-file values.json\n```\n\nPowerShell:\n```powershell\n& \".\\scripts\\build-dataify-request.ps1\" -ToolSign \"<selected_tool_sign>\" -ValuesFile \".\\values.json\"\n```\n\nThe `values.json` file should contain either one object or an array of objects. Example:\n```json\n[{\"searchurl\":\"https://www.airbnb.com/s/Greece/homes?...\",\"country\":\"Hong Kong\"}]\n```\n\n## Required output shape\n\nGenerate a curl command in this form:\n\n```bash\ncurl -X POST 'https://scraperapi.dataify.com/builder' \\\n  -H \"Authorization: Bearer $DATAIFY_API_TOKEN\" \\\n  -H 'Content-Type: application/x-www-form-urlencoded' \\\n  -d 'spider_name=github.com' \\\n  -d 'spider_id=<selected_tool_sign>' \\\n  -d 'spider_parameters=[{\"param\":\"value\"}]' \\\n  -d 'spider_errors=true' \\\n  -d 'file_name={{TasksID}}'\n```\n\n## Reference usage\n\n- `references/tool-params.json` stores the full saved parameter catalog for every available tool in this scraper family.\n- `scripts/build-dataify-request.py` is the portable implementation and should be preferred.\n- `scripts/build-dataify-request.ps1` mirrors the same behavior for Windows users.\n- If a parameter has no options, the user must provide the value.\n- Do not assume `spider_parameters` always contains exactly one object. Multi-value tools may require multiple objects zipped by index.\n- Use the saved `url_example` only as a reference example. Do not assume the user wants the example values unless they explicitly confirm them.\n\n## Default completion behavior\n\nThe default deliverable is the collected result, not only a `task_id`.\n\n1. Submit the Builder task once and capture its `task_id`.\n2. Immediately continue with `$dataify-task-operations` and monitor the same task ID.\n   - Use the default 600-second wait for ordinary collections.\n   - Use `--timeout 1800` for media downloads or clearly high-volume, multi-page, or multi-input collections.\n3. When the task succeeds, download and return the final JSON result. Summarize large payloads while preserving access to the raw result.\n4. If monitoring times out or is interrupted, return the task ID and a resume command. Do not resubmit the paid task.\n5. Stop after submission only when the user explicitly asks for submission only, a task ID, or `--no-wait` behavior.\n\n## Parameter interaction policy\n\n- For a clear, low-risk, read-only, and low-cost request, apply safe defaults and execute immediately. A short execution summary is optional; do not pause for confirmation.\n- Ask only for a missing required input, a material ambiguity, a high-volume or multi-page scope, a media download, a choice that materially changes credit usage, an irreversible action, or an explicit user request to review parameters.\n- When confirmation is required, show only user-facing values that affect the target, scope, output, or cost. Prefer one concise sentence; use a compact table only when three or more consequential values are easier to compare.\n- Never show fixed fields, empty optional fields, unchanged defaults, credentials, or internal implementation parameters such as engine selectors, response-format flags, offsets, spider IDs, and file-name templates.\n- Keep advanced filters hidden unless the user asks for them or they are needed to resolve ambiguity. Never substitute documentation example values for missing required user input.\n- After returning results, offer relevant refinements instead of forcing all optional decisions before the first result.\n\n## Account CTA policy\n\n- Show a prominent Dataify account CTA only when the API token is missing, rejected/invalid, or the account has insufficient credits.\n- For a missing token, offer https://dashboard.dataify.com/login?utm_source=skill and state: New accounts receive 50 free credits. Never ask the user to paste the token into chat.\n- Detect the current operating system and shell. Show only the matching session-scoped setup command first (`export` for macOS/Linux shells, `$env:` for Windows PowerShell, or `set` for Windows Command Prompt). Show other platforms or persistent setup only when detection is ambiguous or the user asks.\n- After the user says the token is configured, verify only whether `DATAIFY_API_TOKEN` is present; never print its value. If verification succeeds, continue the original task without asking the user to repeat it.\n- Explain that persistent shell changes may require a new terminal or restarting the agent application. Do not recommend a project `.env` unless the execution path explicitly loads it, and ensure `.env` is ignored by version control.\n- For an invalid token, direct the user to API-key management without implying that a new registration is required. For insufficient credits, direct the user to balance or recharge management.\n- During normal submission, processing, and successful completion, do not promote registration or the Dashboard. Never expose the token or include it in CTA attribution parameters.\n\nFile v1.3.0:_meta.json\n\n{\n  \"ownerId\": \"kn74z5hmmwk21kw8tpphd9w21x86bkdf\",\n  \"slug\": \"dataify-github-repository-by-repo-url\",\n  \"version\": \"1.3.0\",\n  \"publishedAt\": 1788254011285\n}\n\nFile v1.3.0:references/tool-params.json\n\n[{\"tool_name_cn\":\"通过仓库URL采集\",\"tool_sign\":\"github_repository_by-repo-url\",\"spider_name\":\"github.com\",\"params\":[]},{\"tool_name_cn\":\"通过搜索URL采集\",\"tool_sign\":\"github_repository_by-search-url\",\"spider_name\":\"github.com\",\"params\":[]},{\"tool_name_cn\":\"通过URL采集\",\"tool_sign\":\"github_repository_by-url\",\"spider_name\":\"github.com\",\"params\":[]}]\n\nFile v1.3.0:skill-card.md\n\n## Description:\n\nCollect structured GitHub repository information from one or more known repository URLs; do not use for GitHub code search or arbitrary webpages.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[dataify-server](https://clawhub.ai/user/dataify-server)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and external users use this skill to prepare and run Dataify Builder requests for collecting structured data from known GitHub repository URLs, then wait for and return the collected result.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The security summary says the workflow broadens a repository-URL skill into GitHub URL/search collection using an external API token and paid task submission.\n\nMitigation: Use repository URL collection by default and require explicit confirmation before running search URL or generic URL modes or any scope that materially changes credit use.\n\nRisk: The skill sends GitHub targets to Dataify using DATAIFY_API_TOKEN.\n\nMitigation: Verify only that DATAIFY_API_TOKEN is present, never print or ask users to paste the token, and run only user-approved collection targets.\n\nRisk: Interrupted monitoring can lead to accidental duplicate paid submissions if the original task is resubmitted.\n\nMitigation: Capture and return the task ID with a resume command whenever monitoring times out or is interrupted.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/dataify-server/skills/dataify-github-repository-by-repo-url)\n- [Tool parameter catalog](artifact/references/tool-params.json)\n- [Dataify Builder API endpoint](https://scraperapi.dataify.com/builder)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, shell commands, configuration, guidance, JSON]\n\n**Output Format:** [Markdown with shell commands and JSON result summaries]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May return a task ID and resume command if monitoring times out or is interrupted.]\n\n## Skill Version(s):\n\n1.3.0 (source: evidence.release.version)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.3.0:SKILL.zh-CN.md\n\n---\nname: \"dataify-github-repository-by-repo-url\"\ndescription: \"为 github.com 上以 github_repository_by-repo-url 为根的 scraper 系列准备 Dataify builder 请求。当需要处理成功的 Dataify scraper detail 条目 github_repository_by-repo-url、让用户选择可用工具、读取已保存的 getToolParams 选项，并使用 DATAIFY_API_TOKEN 生成 scraperapi.dataify.com/builder curl 请求时，使用此 skill。\"\n---\n\n# Dataify Builder Skill 中文版\n\n这个 skill 用于为 `github.com` 下、以 `github_repository_by-repo-url` 为入口的 Dataify scraper 工具族生成 builder 请求。\n\n## 工作流程\n\n1. 先检查环境变量中是否存在 `DATAIFY_API_TOKEN`。\n2. 如果 token 缺失，提示用户前往 <a href=\"https://dashboard.dataify.com?utm_source=skill\">dataify&#23448;&#32593;</a> 获取。\n3. 先让用户从下面的中文工具列表中明确选择一个工具：\n- 通过仓库URL采集 (github_repository_by-repo-url)\n- 通过搜索URL采集 (github_repository_by-search-url)\n- 通过URL采集 (github_repository_by-url)\n4. 再读取 `references/tool-params.json`，根据 `tool_sign` 或中文工具名找到对应工具。\n5. 对所选工具的每个参数分别处理：\n   - 如果 `input_mode` 是 `user_input`，让用户提供值。\n   - 如果 `input_mode` 是 `select`，把已保存的可选项展示给用户，让用户选择。\n6. 默认优先使用 `scripts/build-dataify-request.py`，因为它是跨平台版本。\n7. Windows 下也可以使用 `scripts/build-dataify-request.ps1`。\n8. 对于可选型参数，如果存在人类可读标签，优先把该标签写入 `spider_parameters`。\n9. `spider_parameters` 必须是一个 JSON 数组。\n10. 像搜索 URL 这类多值工具，需要按索引生成多个对象。\n11. `spider_name` 固定取 `github.com`。\n12. `spider_id` 固定取用户所选工具的 `tool_sign`。\n13. 始终包含 `spider_errors=true` 和 `file_name={{TasksID}}`。\n\n## 设置 DATAIFY_API_TOKEN\n\n推荐使用永久环境变量，而不是只在当前终端临时设置。\n\nWindows PowerShell，当前用户永久设置：\n\n```powershell\n[Environment]::SetEnvironmentVariable(\"DATAIFY_API_TOKEN\", \"your_token_here\", \"User\")\n```\n\n然后重新打开 PowerShell。如果当前会话也要立即生效，再执行：\n\n```powershell\n$env:DATAIFY_API_TOKEN = \"your_token_here\"\n```\n\nmacOS 或 Linux，bash 永久设置：\n\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.bashrc\nsource ~/.bashrc\n```\n\nmacOS 或 Linux，zsh 永久设置：\n\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.zshrc\nsource ~/.zshrc\n```\n\n## 脚本用法\n\nPython：\n\n```bash\npython scripts/build-dataify-request.py --tool-sign <selected_tool_sign> --values-file values.json\n```\n\nPowerShell：\n\n```powershell\n& \".\\scripts\\build-dataify-request.ps1\" -ToolSign \"<selected_tool_sign>\" -ValuesFile \".\\values.json\"\n```\n\n`values.json` 可以是单个对象，也可以是对象数组。\n\n## 输出格式\n\n最终 `curl` 命令应为：\n\n```bash\ncurl -X POST 'https://scraperapi.dataify.com/builder' \\\n  -H \"Authorization: Bearer $DATAIFY_API_TOKEN\" \\\n  -H 'Content-Type: application/x-www-form-urlencoded' \\\n  -d 'spider_name=github.com' \\\n  -d 'spider_id=<selected_tool_sign>' \\\n  -d 'spider_parameters=[{\"param\":\"value\"}]' \\\n  -d 'spider_errors=true' \\\n  -d 'file_name={{TasksID}}'\n```\n\n## 参考文件\n\n- `references/tool-params.json` 保存了这个 skill 下所有工具及参数选项。\n- `scripts/build-dataify-request.py` 是首选的跨平台实现。\n- `scripts/build-dataify-request.ps1` 是 Windows PowerShell 版本。\n- 如果参数没有预设选项，必须向用户要值。\n- 不要假设 `spider_parameters` 永远只有一个对象；多值工具可能需要按索引生成多个对象。\n- `url_example` 仅作为参考，不要默认用户就要用示例值，除非用户明确确认。\n\n## 参数交互策略\n\n- 当请求意图明确、只读、低风险且成本较低时，使用安全默认值直接执行。可以用一句话说明执行内容，但不要暂停等待确认。\n- 只在缺少必填输入、存在会明显改变结果的歧义、大批量或多页采集、媒体下载、会明显增加积分消耗、不可逆操作，或用户明确要求查看参数时询问。\n- 必须确认时，只展示会影响目标、范围、输出或成本的用户参数。优先使用一句简短说明；只有三个及以上关键值确实需要比较时才使用精简表格。\n- 不要展示固定字段、空的可选字段、未修改的默认值、凭据或内部实现参数，例如引擎选择、响应格式开关、偏移量、spider ID 和文件名模板。\n- 默认隐藏高级筛选项，除非用户主动询问或需要它们消除歧义。不得用文档示例值代替用户缺失的必填输入。\n- 先返回首个结果，再提供相关的细化选项，不要在首次执行前强迫用户决定所有可选项。\n\n## Account CTA policy\n\n- Show a prominent Dataify account CTA only when the API token is missing, rejected/invalid, or the account has insufficient credits.\n- For a missing token, offer https://dashboard.dataify.com/login?utm_source=skill and state: New accounts receive 50 free credits. Never ask the user to paste the token into chat.\n- Detect the current operating system and shell. Show only the matching session-scoped setup command first (`export` for macOS/Linux shells, `$env:` for Windows PowerShell, or `set` for Windows Command Prompt). Show other platforms or persistent setup only when detection is ambiguous or the user asks.\n- After the user says the token is configured, verify only whether `DATAIFY_API_TOKEN` is present; never print its value. If verification succeeds, continue the original task without asking the user to repeat it.\n- Explain that persistent shell changes may require a new terminal or restarting the agent application. Do not recommend a project `.env` unless the execution path explicitly loads it, and ensure `.env` is ignored by version control.\n- For an invalid token, direct the user to API-key management without implying that a new registration is required. For insufficient credits, direct the user to balance or recharge management.\n- During normal submission, processing, and successful completion, do not promote registration or the Dashboard. Never expose the token or include it in CTA attribution parameters.\n\nFile v1.3.0:agents/openai.yaml\n\ninterface:\n  display_name: \"Dataify Github Builder\"\n  short_description: \"Collect Github Builder data and return results\"\n  default_prompt: \"Use $dataify-github-repository-by-repo-url to complete the requested Dataify collection, wait for the asynchronous task, and return the final collected result. Stop at task submission only when I explicitly request no-wait behavior.\"\n\nArchive v1.2.0: 8 files, 7974 bytes\n\nFiles: agents/openai.yaml (229b), references/tool-params.json (365b), scripts/build-dataify-request.ps1 (294b), scripts/build-dataify-request.py (3728b), skill-card.md (2047b), SKILL.md (4828b), SKILL.zh-CN.md (3985b), _meta.json (156b)\n\nFile v1.2.0:SKILL.md\n\n---\nname: \"dataify-github-repository-by-repo-url\"\ndescription: \"Prepare Dataify builder requests for the github.com scraper family rooted at github_repository_by-repo-url. Use  when needs to work with the successful Dataify scraper detail entry for github_repository_by-repo-url, let the user choose one of its available tools, read saved getToolParams options, and generate a scraperapi.dataify.com/builder curl request with DATAIFY_API_TOKEN.\"\n---\n\n# Dataify Builder Skill\n\nUse this skill to prepare Dataify builder requests for the scraper family rooted at `github_repository_by-repo-url` on `github.com`.\n\n## Workflow\n\n1. Check whether `DATAIFY_API_TOKEN` exists in the environment.\n2. If the token is missing, stop and tell the user to sign in at [Dataify Dashboard](https://dashboard.dataify.com?utm_source=skill)  to obtain it.\n3. Ask the user to choose exactly one tool from the following Chinese list:\n- 通过仓库URL采集 (github_repository_by-repo-url)\n- 通过搜索URL采集 (github_repository_by-search-url)\n- 通过URL采集 (github_repository_by-url)\n4. Read `references/tool-params.json` and find the chosen tool by `tool_sign` or Chinese tool name.\n5. For each parameter in the chosen tool:\n   - If `input_mode` is `user_input`, ask the user for the value.\n   - If `input_mode` is `select`, present the saved options to the user.\n6. Use `scripts/build-dataify-request.py` as the default cross-platform helper.\n7. Use `scripts/build-dataify-request.ps1` as the Windows PowerShell helper when needed.\n8. When a selectable parameter has a human-readable Chinese label, keep that label in `spider_parameters`. Do not replace it with a code such as `HK` unless the user explicitly asks for the coded value.\n9. Build `spider_parameters` as a JSON array.\n10. If every parameter has only one final value, build one object such as `[{\"searchurl\":\"...\",\"country\":\"Hong Kong\"}]`.\n11. If one or more parameters have multiple aligned values, zip them by index and build one object per row. Example: `[{\"search_url\":\"url1\",\"page_turning\":\"1\",\"max_num\":\"15\"},{\"search_url\":\"url2\",\"page_turning\":\"1\",\"max_num\":\"15\"}]`.\n12. If a parameter has one value while another parameter has multiple values, reuse the single value across every generated row.\n13. Set `spider_name` to `github.com`.\n14. Set `spider_id` to the selected tool's `tool_sign`.\n15. Always include `spider_errors=true` and `file_name={{TasksID}}`.\n16. Return a curl command for `https://scraperapi.dataify.com/builder`.\n\n## Set DATAIFY_API_TOKEN\n\nPrefer a permanent environment-variable setup instead of setting the token only for the current terminal session.\n\nWindows PowerShell, permanent for the current user:\n```powershell\n[Environment]::SetEnvironmentVariable(\"DATAIFY_API_TOKEN\", \"your_token_here\", \"User\")\n```\n\nThen reopen PowerShell. If the current session also needs the token immediately, run:\n```powershell\n$env:DATAIFY_API_TOKEN = \"your_token_here\"\n```\n\nmacOS or Linux, permanent for bash:\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.bashrc\nsource ~/.bashrc\n```\n\nmacOS or Linux, permanent for zsh:\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.zshrc\nsource ~/.zshrc\n```\n\n## Script usage\n\nPython:\n```bash\npython scripts/build-dataify-request.py --tool-sign <selected_tool_sign> --values-file values.json\n```\n\nPowerShell:\n```powershell\n& \".\\scripts\\build-dataify-request.ps1\" -ToolSign \"<selected_tool_sign>\" -ValuesFile \".\\values.json\"\n```\n\nThe `values.json` file should contain either one object or an array of objects. Example:\n```json\n[{\"searchurl\":\"https://www.airbnb.com/s/Greece/homes?...\",\"country\":\"Hong Kong\"}]\n```\n\n## Required output shape\n\nGenerate a curl command in this form:\n\n```bash\ncurl -X POST 'https://scraperapi.dataify.com/builder' \\\n  -H \"Authorization: Bearer $DATAIFY_API_TOKEN\" \\\n  -H 'Content-Type: application/x-www-form-urlencoded' \\\n  -d 'spider_name=github.com' \\\n  -d 'spider_id=<selected_tool_sign>' \\\n  -d 'spider_parameters=[{\"param\":\"value\"}]' \\\n  -d 'spider_errors=true' \\\n  -d 'file_name={{TasksID}}'\n```\n\n## Reference usage\n\n- `references/tool-params.json` stores the full saved parameter catalog for every available tool in this scraper family.\n- `scripts/build-dataify-request.py` is the portable implementation and should be preferred.\n- `scripts/build-dataify-request.ps1` mirrors the same behavior for Windows users.\n- If a parameter has no options, the user must provide the value.\n- If a parameter has options, present those options back to the user before building the final request.\n- Do not assume `spider_parameters` always contains exactly one object. Multi-value tools may require multiple objects zipped by index.\n- Use the saved `url_example` only as a reference example. Do not assume the user wants the example values unless they explicitly confirm them.\n\nFile v1.2.0:_meta.json\n\n{\n  \"ownerId\": \"kn74z5hmmwk21kw8tpphd9w21x86bkdf\",\n  \"slug\": \"dataify-github-repository-by-repo-url\",\n  \"version\": \"1.2.0\",\n  \"publishedAt\": 1784191028505\n}\n\nFile v1.2.0:references/tool-params.json\n\n[{\"tool_name_cn\":\"通过仓库URL采集\",\"tool_sign\":\"github_repository_by-repo-url\",\"spider_name\":\"github.com\",\"params\":[]},{\"tool_name_cn\":\"通过搜索URL采集\",\"tool_sign\":\"github_repository_by-search-url\",\"spider_name\":\"github.com\",\"params\":[]},{\"tool_name_cn\":\"通过URL采集\",\"tool_sign\":\"github_repository_by-url\",\"spider_name\":\"github.com\",\"params\":[]}]\n\nFile v1.2.0:skill-card.md\n\n## Description: <br>\nPrepares Dataify builder requests for the github.com scraper family rooted at github_repository_by-repo-url, including tool selection, saved parameter lookup, and a curl request that uses DATAIFY_API_TOKEN. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[dataify-server](https://clawhub.ai/user/dataify-server) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and operators use this skill to prepare Dataify builder curl requests for GitHub repository scraping workflows after selecting one supported scraper tool and supplying any required values. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The generated curl command sends selected scraper parameters to Dataify and uses a bearer token from the local environment. <br>\nMitigation: Keep DATAIFY_API_TOKEN private, review values.json before generating requests, and run the curl command only when the selected parameters are intended for Dataify. <br>\n\n\n## Reference(s): <br>\n- [Dataify skill page](https://clawhub.ai/dataify-server/skills/dataify-github-repository-by-repo-url) <br>\n- [Saved Dataify tool parameter catalog](references/tool-params.json) <br>\n- [Dataify builder endpoint](https://scraperapi.dataify.com/builder) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [Shell commands, Configuration, Guidance] <br>\n**Output Format:** [Markdown with a curl command and setup instructions] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Produces user-directed requests for Dataify's builder API and relies on DATAIFY_API_TOKEN from the user's environment.] <br>\n\n## Skill Version(s): <br>\n1.2.0 (source: server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.2.0:SKILL.zh-CN.md\n\n---\nname: \"dataify-github-repository-by-repo-url\"\ndescription: \"为 github.com 上以 github_repository_by-repo-url 为根的 scraper 系列准备 Dataify builder 请求。当需要处理成功的 Dataify scraper detail 条目 github_repository_by-repo-url、让用户选择可用工具、读取已保存的 getToolParams 选项，并使用 DATAIFY_API_TOKEN 生成 scraperapi.dataify.com/builder curl 请求时，使用此 skill。\"\n---\n\n# Dataify Builder Skill 中文版\n\n这个 skill 用于为 `github.com` 下、以 `github_repository_by-repo-url` 为入口的 Dataify scraper 工具族生成 builder 请求。\n\n## 工作流程\n\n1. 先检查环境变量中是否存在 `DATAIFY_API_TOKEN`。\n2. 如果 token 缺失，提示用户前往 <a href=\"https://dashboard.dataify.com?utm_source=skill\">dataify&#23448;&#32593;</a> 获取。\n3. 先让用户从下面的中文工具列表中明确选择一个工具：\n- 通过仓库URL采集 (github_repository_by-repo-url)\n- 通过搜索URL采集 (github_repository_by-search-url)\n- 通过URL采集 (github_repository_by-url)\n4. 再读取 `references/tool-params.json`，根据 `tool_sign` 或中文工具名找到对应工具。\n5. 对所选工具的每个参数分别处理：\n   - 如果 `input_mode` 是 `user_input`，让用户提供值。\n   - 如果 `input_mode` 是 `select`，把已保存的可选项展示给用户，让用户选择。\n6. 默认优先使用 `scripts/build-dataify-request.py`，因为它是跨平台版本。\n7. Windows 下也可以使用 `scripts/build-dataify-request.ps1`。\n8. 对于可选型参数，如果存在人类可读标签，优先把该标签写入 `spider_parameters`。\n9. `spider_parameters` 必须是一个 JSON 数组。\n10. 像搜索 URL 这类多值工具，需要按索引生成多个对象。\n11. `spider_name` 固定取 `github.com`。\n12. `spider_id` 固定取用户所选工具的 `tool_sign`。\n13. 始终包含 `spider_errors=true` 和 `file_name={{TasksID}}`。\n\n## 设置 DATAIFY_API_TOKEN\n\n推荐使用永久环境变量，而不是只在当前终端临时设置。\n\nWindows PowerShell，当前用户永久设置：\n\n```powershell\n[Environment]::SetEnvironmentVariable(\"DATAIFY_API_TOKEN\", \"your_token_here\", \"User\")\n```\n\n然后重新打开 PowerShell。如果当前会话也要立即生效，再执行：\n\n```powershell\n$env:DATAIFY_API_TOKEN = \"your_token_here\"\n```\n\nmacOS 或 Linux，bash 永久设置：\n\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.bashrc\nsource ~/.bashrc\n```\n\nmacOS 或 Linux，zsh 永久设置：\n\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.zshrc\nsource ~/.zshrc\n```\n\n## 脚本用法\n\nPython：\n\n```bash\npython scripts/build-dataify-request.py --tool-sign <selected_tool_sign> --values-file values.json\n```\n\nPowerShell：\n\n```powershell\n& \".\\scripts\\build-dataify-request.ps1\" -ToolSign \"<selected_tool_sign>\" -ValuesFile \".\\values.json\"\n```\n\n`values.json` 可以是单个对象，也可以是对象数组。\n\n## 输出格式\n\n最终 `curl` 命令应为：\n\n```bash\ncurl -X POST 'https://scraperapi.dataify.com/builder' \\\n  -H \"Authorization: Bearer $DATAIFY_API_TOKEN\" \\\n  -H 'Content-Type: application/x-www-form-urlencoded' \\\n  -d 'spider_name=github.com' \\\n  -d 'spider_id=<selected_tool_sign>' \\\n  -d 'spider_parameters=[{\"param\":\"value\"}]' \\\n  -d 'spider_errors=true' \\\n  -d 'file_name={{TasksID}}'\n```\n\n## 参考文件\n\n- `references/tool-params.json` 保存了这个 skill 下所有工具及参数选项。\n- `scripts/build-dataify-request.py` 是首选的跨平台实现。\n- `scripts/build-dataify-request.ps1` 是 Windows PowerShell 版本。\n- 如果参数没有预设选项，必须向用户要值。\n- 如果参数有预设选项，先把选项展示给用户，再生成最终请求。\n- 不要假设 `spider_parameters` 永远只有一个对象；多值工具可能需要按索引生成多个对象。\n- `url_example` 仅作为参考，不要默认用户就要用示例值，除非用户明确确认。\n\nFile v1.2.0:agents/openai.yaml\n\ninterface:\n  display_name: \"Dataify Github Builder\"\n  short_description: \"Prepare Dataify builder calls for github.com\"\n  default_prompt: \"Use $dataify-github-repository-by-repo-url to prepare a Dataify builder curl request.\"\n\nArchive v1.1.0: 6 files, 7503 bytes\n\nFiles: agents/openai.yaml (229b), scripts/build-dataify-request.py (3721b), skill-card.md (2433b), SKILL.md (4831b), SKILL.zh-CN.md (3985b), _meta.json (156b)\n\nFile v1.1.0:SKILL.md\n\n---\nname: \"dataify-github-repository-by-repo-url\"\ndescription: \"Prepare Dataify builder requests for the github.com scraper family rooted at github_repository_by-repo-url. Use  when needs to work with the successful Dataify scraper detail entry for github_repository_by-repo-url, let the user choose one of its available tools, read saved getToolParams options, and generate a scraperapi.dataify.com/builder curl request with DATAIFY_API_TOKEN.\"\n---\n\n# Dataify Builder Skill\n\nUse this skill to prepare Dataify builder requests for the scraper family rooted at `github_repository_by-repo-url` on `github.com`.\n\n## Workflow\n\n1. Check whether `DATAIFY_API_TOKEN` exists in the environment.\n2. If the token is missing, stop and tell the user to sign in at [Dataify Dashboard](https://dashboard.dataify.com?utm_source=skill)  to obtain it.\n3. Ask the user to choose exactly one tool from the following Chinese list:\n- 通过仓库URL采集 (github_repository_by-repo-url)\n- 通过搜索URL采集 (github_repository_by-search-url)\n- 通过URL采集 (github_repository_by-url)\n4. Read `references/tool-params.json` and find the chosen tool by `tool_sign` or Chinese tool name.\n5. For each parameter in the chosen tool:\n   - If `input_mode` is `user_input`, ask the user for the value.\n   - If `input_mode` is `select`, present the saved options to the user.\n6. Use `scripts/build-dataify-request.py` as the default cross-platform helper.\n7. Use `scripts/build-dataify-request.ps1` as the Windows PowerShell helper when needed.\n8. When a selectable parameter has a human-readable Chinese label, keep that label in `spider_parameters`. Do not replace it with a code such as `HK` unless the user explicitly asks for the coded value.\n9. Build `spider_parameters` as a JSON array.\n10. If every parameter has only one final value, build one object such as `[{\"searchurl\":\"...\",\"country\":\"Hong Kong\"}]`.\n11. If one or more parameters have multiple aligned values, zip them by index and build one object per row. Example: `[{\"search_url\":\"url1\",\"page_turning\":\"1\",\"max_num\":\"15\"},{\"search_url\":\"url2\",\"page_turning\":\"1\",\"max_num\":\"15\"}]`.\n12. If a parameter has one value while another parameter has multiple values, reuse the single value across every generated row.\n13. Set `spider_name` to `github.com`.\n14. Set `spider_id` to the selected tool's `tool_sign`.\n15. Always include `spider_errors=true` and `file_name={{TasksID}}`.\n16. Return a curl command for `https://scraperapi.dataify.com/builder`.\n\n## Set DATAIFY_API_TOKEN\n\nPrefer a permanent environment-variable setup instead of setting the token only for the current terminal session.\n\nWindows PowerShell, permanent for the current user:\n```powershell\n[Environment]::SetEnvironmentVariable(\"DATAIFY_API_TOKEN\", \"your_token_here\", \"User\")\n```\n\nThen reopen PowerShell. If the current session also needs the token immediately, run:\n```powershell\n$env:DATAIFY_API_TOKEN = \"your_token_here\"\n```\n\nmacOS or Linux, permanent for bash:\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.bashrc\nsource ~/.bashrc\n```\n\nmacOS or Linux, permanent for zsh:\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.zshrc\nsource ~/.zshrc\n```\n\n## Script usage\n\nPython:\n```bash\npython scripts/build-dataify-request.py --tool-sign <selected_tool_sign> --values-file values.json\n```\n\nPowerShell:\n```powershell\n& \".\\scripts\\build-dataify-request.ps1\" -ToolSign \"<selected_tool_sign>\" -ValuesFile \".\\values.json\"\n```\n\nThe `values.json` file should contain either one object or an array of objects. Example:\n```json\n[{\"searchurl\":\"https://www.airbnb.com/s/Greece/homes?...\",\"country\":\"Hong Kong\"}]\n```\n\n## Required output shape\n\nGenerate a curl command in this form:\n\n```bash\ncurl -X POST 'https://scraperapi.dataify.com/builder' \\\n  -H \"Authorization: Bearer $DATAIFY_API_TOKEN\" \\\n  -H 'Content-Type: application/x-www-form-urlencoded' \\\n  -d 'spider_name=github.com' \\\n  -d 'spider_id=<selected_tool_sign>' \\\n  -d 'spider_parameters=[{\"param\":\"value\"}]' \\\n  -d 'spider_errors=true' \\\n  -d 'file_name={{TasksID}}'\n```\n\n## Reference usage\n\n- `references/tool-params.json` stores the full saved parameter catalog for every available tool in this scraper family.\n- `scripts/build-dataify-request.py` is the portable implementation and should be preferred.\n- `scripts/build-dataify-request.ps1` mirrors the same behavior for Windows users.\n- If a parameter has no options, the user must provide the value.\n- If a parameter has options, present those options back to the user before building the final request.\n- Do not assume `spider_parameters` always contains exactly one object. Multi-value tools may require multiple objects zipped by index.\n- Use the saved `url_example` only as a reference example. Do not assume the user wants the example values unless they explicitly confirm them.\n\nFile v1.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn74z5hmmwk21kw8tpphd9w21x86bkdf\",\n  \"slug\": \"dataify-github-repository-by-repo-url\",\n  \"version\": \"1.1.0\",\n  \"publishedAt\": 1780884380040\n}\n\nFile v1.1.0:skill-card.md\n\n## Description: <br>\nPrepares Dataify builder requests for the github.com scraper family rooted at github_repository_by-repo-url, including tool selection, saved parameter options, and generation of a Dataify builder curl request using DATAIFY_API_TOKEN. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[dataify-server](https://clawhub.ai/user/dataify-server) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and operators use this skill to assemble Dataify builder requests for GitHub repository scraping tools, normalize spider_parameters from saved options or user inputs, and produce a curl command for the Dataify builder API. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: DATAIFY_API_TOKEN can be exposed if generated curl commands containing a literal bearer token are shared, logged, or committed. <br>\nMitigation: Keep the token in an environment variable, avoid sharing generated commands that include the token value, and rotate the token if it is exposed. <br>\nRisk: The artifact references a tool-params catalog and a PowerShell helper that are not included in the released files. <br>\nMitigation: Confirm required helper files and parameter catalogs are available before relying on saved options; otherwise collect needed values from the user and prefer the included Python helper. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/dataify-server/dataify-github-repository-by-repo-url) <br>\n- [Dataify Dashboard](https://dashboard.dataify.com?utm_source=skill) <br>\n- [Dataify builder API endpoint](https://scraperapi.dataify.com/builder) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, shell commands, configuration, guidance] <br>\n**Output Format:** [Markdown with inline shell commands and JSON snippets] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Generated curl commands may include a Dataify bearer token when the helper script is run with DATAIFY_API_TOKEN set.] <br>\n\n## Skill Version(s): <br>\n1.1.0 (source: server-resolved release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.1.0:SKILL.zh-CN.md\n\n---\nname: \"dataify-github-repository-by-repo-url\"\ndescription: \"为 github.com 上以 github_repository_by-repo-url 为根的 scraper 系列准备 Dataify builder 请求。当需要处理成功的 Dataify scraper detail 条目 github_repository_by-repo-url、让用户选择可用工具、读取已保存的 getToolParams 选项，并使用 DATAIFY_API_TOKEN 生成 scraperapi.dataify.com/builder curl 请求时，使用此 skill。\"\n---\n\n# Dataify Builder Skill 中文版\n\n这个 skill 用于为 `github.com` 下、以 `github_repository_by-repo-url` 为入口的 Dataify scraper 工具族生成 builder 请求。\n\n## 工作流程\n\n1. 先检查环境变量中是否存在 `DATAIFY_API_TOKEN`。\n2. 如果 token 缺失，提示用户前往 <a href=\"https://dashboard.dataify.com?utm_source=skill\">dataify&#23448;&#32593;</a> 获取。\n3. 先让用户从下面的中文工具列表中明确选择一个工具：\n- 通过仓库URL采集 (github_repository_by-repo-url)\n- 通过搜索URL采集 (github_repository_by-search-url)\n- 通过URL采集 (github_repository_by-url)\n4. 再读取 `references/tool-params.json`，根据 `tool_sign` 或中文工具名找到对应工具。\n5. 对所选工具的每个参数分别处理：\n   - 如果 `input_mode` 是 `user_input`，让用户提供值。\n   - 如果 `input_mode` 是 `select`，把已保存的可选项展示给用户，让用户选择。\n6. 默认优先使用 `scripts/build-dataify-request.py`，因为它是跨平台版本。\n7. Windows 下也可以使用 `scripts/build-dataify-request.ps1`。\n8. 对于可选型参数，如果存在人类可读标签，优先把该标签写入 `spider_parameters`。\n9. `spider_parameters` 必须是一个 JSON 数组。\n10. 像搜索 URL 这类多值工具，需要按索引生成多个对象。\n11. `spider_name` 固定取 `github.com`。\n12. `spider_id` 固定取用户所选工具的 `tool_sign`。\n13. 始终包含 `spider_errors=true` 和 `file_name={{TasksID}}`。\n\n## 设置 DATAIFY_API_TOKEN\n\n推荐使用永久环境变量，而不是只在当前终端临时设置。\n\nWindows PowerShell，当前用户永久设置：\n\n```powershell\n[Environment]::SetEnvironmentVariable(\"DATAIFY_API_TOKEN\", \"your_token_here\", \"User\")\n```\n\n然后重新打开 PowerShell。如果当前会话也要立即生效，再执行：\n\n```powershell\n$env:DATAIFY_API_TOKEN = \"your_token_here\"\n```\n\nmacOS 或 Linux，bash 永久设置：\n\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.bashrc\nsource ~/.bashrc\n```\n\nmacOS 或 Linux，zsh 永久设置：\n\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.zshrc\nsource ~/.zshrc\n```\n\n## 脚本用法\n\nPython：\n\n```bash\npython scripts/build-dataify-request.py --tool-sign <selected_tool_sign> --values-file values.json\n```\n\nPowerShell：\n\n```powershell\n& \".\\scripts\\build-dataify-request.ps1\" -ToolSign \"<selected_tool_sign>\" -ValuesFile \".\\values.json\"\n```\n\n`values.json` 可以是单个对象，也可以是对象数组。\n\n## 输出格式\n\n最终 `curl` 命令应为：\n\n```bash\ncurl -X POST 'https://scraperapi.dataify.com/builder' \\\n  -H \"Authorization: Bearer $DATAIFY_API_TOKEN\" \\\n  -H 'Content-Type: application/x-www-form-urlencoded' \\\n  -d 'spider_name=github.com' \\\n  -d 'spider_id=<selected_tool_sign>' \\\n  -d 'spider_parameters=[{\"param\":\"value\"}]' \\\n  -d 'spider_errors=true' \\\n  -d 'file_name={{TasksID}}'\n```\n\n## 参考文件\n\n- `references/tool-params.json` 保存了这个 skill 下所有工具及参数选项。\n- `scripts/build-dataify-request.py` 是首选的跨平台实现。\n- `scripts/build-dataify-request.ps1` 是 Windows PowerShell 版本。\n- 如果参数没有预设选项，必须向用户要值。\n- 如果参数有预设选项，先把选项展示给用户，再生成最终请求。\n- 不要假设 `spider_parameters` 永远只有一个对象；多值工具可能需要按索引生成多个对象。\n- `url_example` 仅作为参考，不要默认用户就要用示例值，除非用户明确确认。\n\nFile v1.1.0:agents/openai.yaml\n\ninterface:\n  display_name: \"Dataify Github Builder\"\n  short_description: \"Prepare Dataify builder calls for github.com\"\n  default_prompt: \"Use $dataify-github-repository-by-repo-url to prepare a Dataify builder curl request.\"\n\nArchive v1.0.0: 6 files, 7170 bytes\n\nFiles: agents/openai.yaml (229b), scripts/build-dataify-request.py (3699b), skill-card.md (2100b), SKILL.md (4813b), SKILL.zh-CN.md (3524b), _meta.json (156b)\n\nFile v1.0.0:SKILL.md\n\n---\nname: \"dataify-github-repository-by-repo-url\"\ndescription: \"Prepare Dataify builder requests for the github.com scraper family rooted at github_repository_by-repo-url. Use  when needs to work with the successful Dataify scraper detail entry for github_repository_by-repo-url, let the user choose one of its available tools, read saved getToolParams options, and generate a scraperapi.dataify.com/builder curl request with DATAIFY_API_TOKEN.\"\n---\n\n# Dataify Builder Skill\n\nUse this skill to prepare Dataify builder requests for the scraper family rooted at `github_repository_by-repo-url` on `github.com`.\n\n## Workflow\n\n1. Check whether `DATAIFY_API_TOKEN` exists in the environment.\n2. If the token is missing, stop and tell the user to sign in at Dataify Dashboard](https://dataify.com/dashboard)  to obtain it.\n3. Ask the user to choose exactly one tool from the following Chinese list:\n- 通过仓库URL采集 (github_repository_by-repo-url)\n- 通过搜索URL采集 (github_repository_by-search-url)\n- 通过URL采集 (github_repository_by-url)\n4. Read `references/tool-params.json` and find the chosen tool by `tool_sign` or Chinese tool name.\n5. For each parameter in the chosen tool:\n   - If `input_mode` is `user_input`, ask the user for the value.\n   - If `input_mode` is `select`, present the saved options to the user.\n6. Use `scripts/build-dataify-request.py` as the default cross-platform helper.\n7. Use `scripts/build-dataify-request.ps1` as the Windows PowerShell helper when needed.\n8. When a selectable parameter has a human-readable Chinese label, keep that label in `spider_parameters`. Do not replace it with a code such as `HK` unless the user explicitly asks for the coded value.\n9. Build `spider_parameters` as a JSON array.\n10. If every parameter has only one final value, build one object such as `[{\"searchurl\":\"...\",\"country\":\"Hong Kong\"}]`.\n11. If one or more parameters have multiple aligned values, zip them by index and build one object per row. Example: `[{\"search_url\":\"url1\",\"page_turning\":\"1\",\"max_num\":\"15\"},{\"search_url\":\"url2\",\"page_turning\":\"1\",\"max_num\":\"15\"}]`.\n12. If a parameter has one value while another parameter has multiple values, reuse the single value across every generated row.\n13. Set `spider_name` to `github.com`.\n14. Set `spider_id` to the selected tool's `tool_sign`.\n15. Always include `spider_errors=true` and `file_name={{TasksID}}`.\n16. Return a curl command for `https://scraperapi.dataify.com/builder`.\n\n## Set DATAIFY_API_TOKEN\n\nPrefer a permanent environment-variable setup instead of setting the token only for the current terminal session.\n\nWindows PowerShell, permanent for the current user:\n```powershell\n[Environment]::SetEnvironmentVariable(\"DATAIFY_API_TOKEN\", \"your_token_here\", \"User\")\n```\n\nThen reopen PowerShell. If the current session also needs the token immediately, run:\n```powershell\n$env:DATAIFY_API_TOKEN = \"your_token_here\"\n```\n\nmacOS or Linux, permanent for bash:\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.bashrc\nsource ~/.bashrc\n```\n\nmacOS or Linux, permanent for zsh:\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.zshrc\nsource ~/.zshrc\n```\n\n## Script usage\n\nPython:\n```bash\npython scripts/build-dataify-request.py --tool-sign <selected_tool_sign> --values-file values.json\n```\n\nPowerShell:\n```powershell\n& \".\\scripts\\build-dataify-request.ps1\" -ToolSign \"<selected_tool_sign>\" -ValuesFile \".\\values.json\"\n```\n\nThe `values.json` file should contain either one object or an array of objects. Example:\n```json\n[{\"searchurl\":\"https://www.airbnb.com/s/Greece/homes?...\",\"country\":\"Hong Kong\"}]\n```\n\n## Required output shape\n\nGenerate a curl command in this form:\n\n```bash\ncurl -X POST 'https://scraperapi.dataify.com/builder' \\\n  -H \"Authorization: Bearer $DATAIFY_API_TOKEN\" \\\n  -H 'Content-Type: application/x-www-form-urlencoded' \\\n  -d 'spider_name=github.com' \\\n  -d 'spider_id=<selected_tool_sign>' \\\n  -d 'spider_parameters=[{\"param\":\"value\"}]' \\\n  -d 'spider_errors=true' \\\n  -d 'file_name={{TasksID}}'\n```\n\n## Reference usage\n\n- `references/tool-params.json` stores the full saved parameter catalog for every available tool in this scraper family.\n- `scripts/build-dataify-request.py` is the portable implementation and should be preferred.\n- `scripts/build-dataify-request.ps1` mirrors the same behavior for Windows users.\n- If a parameter has no options, the user must provide the value.\n- If a parameter has options, present those options back to the user before building the final request.\n- Do not assume `spider_parameters` always contains exactly one object. Multi-value tools may require multiple objects zipped by index.\n- Use the saved `url_example` only as a reference example. Do not assume the user wants the example values unless they explicitly confirm them.\n\nFile v1.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn74z5hmmwk21kw8tpphd9w21x86bkdf\",\n  \"slug\": \"dataify-github-repository-by-repo-url\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1779955953354\n}\n\nFile v1.0.0:skill-card.md\n\n## Description: <br>\nPrepare Dataify builder requests for the github.com scraper family rooted at github_repository_by-repo-url. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[dataify-server](https://clawhub.ai/user/dataify-server) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and operators use this skill to collect Dataify scraper parameters for GitHub repository tools and generate ready-to-run builder curl commands. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Generated curl commands can expose a raw Dataify API token in shell history, logs, chats, tickets, or CI output. <br>\nMitigation: Use a scoped Dataify token, keep it in an environment variable, and avoid pasting generated commands that contain the raw token into shared systems. <br>\nRisk: The spider_parameters payload is sent to Dataify and may include private repository information or other sensitive values. <br>\nMitigation: Review parameter values before execution and include secrets or private repository details only when that transfer is intended. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/dataify-server/dataify-github-repository-by-repo-url) <br>\n- [Dataify Dashboard](https://dataify.com/dashboard) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, shell commands, configuration, guidance] <br>\n**Output Format:** [Markdown with inline shell commands and JSON request parameters] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Generates Dataify builder curl commands that include spider_name, spider_id, spider_parameters, spider_errors, and file_name fields.] <br>\n\n## Skill Version(s): <br>\n1.0.0 (source: server release metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v1.0.0:SKILL.zh-CN.md\n\n# Dataify Builder Skill 中文版\n\n这个 skill 用于为 `github.com` 下、以 `github_repository_by-repo-url` 为入口的 Dataify scraper 工具族生成 builder 请求。\n\n## 工作流程\n\n1. 先检查环境变量中是否存在 `DATAIFY_API_TOKEN`。\n2. 如果 token 缺失，提示用户前往 <a href=\"https://www.dataify.com/\">dataify&#23448;&#32593;</a> 获取。\n3. 先让用户从下面的中文工具列表中明确选择一个工具：\n- 通过仓库URL采集 (github_repository_by-repo-url)\n- 通过搜索URL采集 (github_repository_by-search-url)\n- 通过URL采集 (github_repository_by-url)\n4. 再读取 `references/tool-params.json`，根据 `tool_sign` 或中文工具名找到对应工具。\n5. 对所选工具的每个参数分别处理：\n   - 如果 `input_mode` 是 `user_input`，让用户提供值。\n   - 如果 `input_mode` 是 `select`，把已保存的可选项展示给用户，让用户选择。\n6. 默认优先使用 `scripts/build-dataify-request.py`，因为它是跨平台版本。\n7. Windows 下也可以使用 `scripts/build-dataify-request.ps1`。\n8. 对于可选型参数，如果存在人类可读标签，优先把该标签写入 `spider_parameters`。\n9. `spider_parameters` 必须是一个 JSON 数组。\n10. 像搜索 URL 这类多值工具，需要按索引生成多个对象。\n11. `spider_name` 固定取 `github.com`。\n12. `spider_id` 固定取用户所选工具的 `tool_sign`。\n13. 始终包含 `spider_errors=true` 和 `file_name={{TasksID}}`。\n\n## 设置 DATAIFY_API_TOKEN\n\n推荐使用永久环境变量，而不是只在当前终端临时设置。\n\nWindows PowerShell，当前用户永久设置：\n\n```powershell\n[Environment]::SetEnvironmentVariable(\"DATAIFY_API_TOKEN\", \"your_token_here\", \"User\")\n```\n\n然后重新打开 PowerShell。如果当前会话也要立即生效，再执行：\n\n```powershell\n$env:DATAIFY_API_TOKEN = \"your_token_here\"\n```\n\nmacOS 或 Linux，bash 永久设置：\n\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.bashrc\nsource ~/.bashrc\n```\n\nmacOS 或 Linux，zsh 永久设置：\n\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.zshrc\nsource ~/.zshrc\n```\n\n## 脚本用法\n\nPython：\n\n```bash\npython scripts/build-dataify-request.py --tool-sign <selected_tool_sign> --values-file values.json\n```\n\nPowerShell：\n\n```powershell\n& \".\\scripts\\build-dataify-request.ps1\" -ToolSign \"<selected_tool_sign>\" -ValuesFile \".\\values.json\"\n```\n\n`values.json` 可以是单个对象，也可以是对象数组。\n\n## 输出格式\n\n最终 `curl` 命令应为：\n\n```bash\ncurl -X POST 'https://scraperapi.dataify.com/builder' \\\n  -H \"Authorization: Bearer $DATAIFY_API_TOKEN\" \\\n  -H 'Content-Type: application/x-www-form-urlencoded' \\\n  -d 'spider_name=github.com' \\\n  -d 'spider_id=<selected_tool_sign>' \\\n  -d 'spider_parameters=[{\"param\":\"value\"}]' \\\n  -d 'spider_errors=true' \\\n  -d 'file_name={{TasksID}}'\n```\n\n## 参考文件\n\n- `references/tool-params.json` 保存了这个 skill 下所有工具及参数选项。\n- `scripts/build-dataify-request.py` 是首选的跨平台实现。\n- `scripts/build-dataify-request.ps1` 是 Windows PowerShell 版本。\n- 如果参数没有预设选项，必须向用户要值。\n- 如果参数有预设选项，先把选项展示给用户，再生成最终请求。\n- 不要假设 `spider_parameters` 永远只有一个对象；多值工具可能需要按索引生成多个对象。\n- `url_example` 仅作为参考，不要默认用户就要用示例值，除非用户明确确认。\n\nFile v1.0.0:agents/openai.yaml\n\ninterface:\n  display_name: \"Dataify Github Builder\"\n  short_description: \"Prepare Dataify builder calls for github.com\"\n  default_prompt: \"Use $dataify-github-repository-by-repo-url to prepare a Dataify builder curl request.\"","readmeExcerpt":"Skill: Dataify Github Builder Owner: dataify-server Summary: Collect Github Builder data and return results Tags: latest:1.3.1 Version history: v1.3.1 | 2026-09-08T06:18:16.065Z | user Fix natural-language usage failures: validate required targets and URLs, preserve catalog references, default Amazon region safely, normalize Google News links, and improve UTF-8 error output. v1.3.0 | 2026-09-01T09:13:31.285Z | user 默","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"python3 scripts/build-dataify-request.py --tool-sign github_repository_by-repo-url --params-json '[{\"repo_url\":\"https://github.com/dataify-server/skills\"}]'"},{"language":"powershell","snippet":"[Environment]::SetEnvironmentVariable(\"DATAIFY_API_TOKEN\", \"your_token_here\", \"User\")"},{"language":"powershell","snippet":"$env:DATAIFY_API_TOKEN = \"your_token_here\""},{"language":"bash","snippet":"echo 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.bashrc\nsource ~/.bashrc"},{"language":"bash","snippet":"echo 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.zshrc\nsource ~/.zshrc"},{"language":"bash","snippet":"python scripts/build-dataify-request.py --tool-sign <selected_tool_sign> --values-file values.json"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: \"dataify-github-repository-by-repo-url\"\ndescription: \"Collect structured GitHub repository information from one or more known repository URLs. Do not use for GitHub code search or arbitrary webpages.\"\n---\n\n# Dataify Builder Skill\n\nUse this skill to prepare Dataify builder requests for the scraper family rooted at `github_repository_by-repo-url` on `github.com`.\n\n\n## Quick Start\n\n**Input:** a GitHub repository URL.\n\n```bash\npython3 scripts/build-dataify-request.py --tool-sign github_repository_by-repo-url --params-json '[{\"repo_url\":\"https://github.com/dataify-server/skills\"}]'\n```\n\nThis submits the task, waits for completion, downloads the final result, and returns it. Add `--no-wait` only when submission-only behavior is requested.\n## Workflow\n\n1. Check whether `DATAIFY_API_TOKEN` exists in the environment.\n2. If the token is missing, stop and tell the user to sign in at [Dataify Dashboard](https://dashboard.dataify.com?utm_source=skill)  to obtain it.\n3. Ask the user to choose exactly one tool from the following Chinese list:\n- 通过仓库URL采集 (github_repository_by-repo-url)\n- 通过搜索URL采集 (github_repository_by-search-url)\n- 通过URL采集 (github_repository_by-url)\n4. Read `references/tool-params.json` and find the chosen tool by `tool_sign` or Chinese tool name.\n5. For each parameter in the chosen tool:\n   - If `input_mode` is `user_input`, ask the user for the value.\n   - If `input_mode` is `select`, present the saved options to the user.\n6. Use `scripts/build-dataify-request.py` as the default cross-platform helper.\n7. Use `scripts/build-dataify-request.ps1` as the Windows PowerShell helper when needed.\n8. When a selectable parameter has a human-readable Chinese label, keep that label in `spider_parameters`. Do not replace it with a code such as `HK` unless the user explicitly asks for the coded value.\n9. Build `spider_parameters` as a JSON array.\n10. If every parameter has only one final value, build one object such as `[{\"searchurl\":\"...\",\"country\":\"Hong Kong\"}]`.\n11. If one or more parameters have multiple aligned values, zip them by index and build one object per row. Example: `[{\"search_url\":\"url1\",\"page_turning\":\"1\",\"max_num\":\"15\"},{\"search_url\":\"url2\",\"page_turning\":\"1\",\"max_num\":\"15\"}]`.\n12. If a parameter has one value while another parameter has multiple values, reuse the single value across every generated row.\n13. Set `spider_name` to `github.com`.\n14. Set `spider_id` to the selected tool's `tool_sign`.\n15. Always include `spider_errors=true` and `file_name={{TasksID}}`.\n16. Return a curl command for `https://scraperapi.dataify.com/builder`.\n\n## Set DATAIFY_API_TOKEN\n\nPrefer a permanent environment-variable setup instead of setting the token only for the current terminal session.\n\nWindows PowerShell, permanent for the current user:\n```powershell\n[Environment]::SetEnvironmentVariable(\"DATAIFY_API_TOKEN\", \"your_token_here\", \"User\")\n```\n\nThen reopen PowerShell. If the current session also needs the token immediately, run:\n```powershell\n$"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn74z5hmmwk21kw8tpphd9w21x86bkdf\",\n  \"slug\": \"dataify-github-repository-by-repo-url\",\n  \"version\": \"1.3.1\",\n  \"publishedAt\": 1788848296065\n}"},{"path":"references/tool-params.json","content":"[{\"tool_name_cn\":\"通过仓库URL采集\",\"tool_sign\":\"github_repository_by-repo-url\",\"spider_name\":\"github.com\",\"params\":[{\"param\":\"repo_url\",\"required\":true,\"input_mode\":\"user_input\",\"description\":\"Public GitHub repository URL\"}]},{\"tool_name_cn\":\"通过搜索URL采集\",\"tool_sign\":\"github_repository_by-search-url\",\"spider_name\":\"github.com\",\"params\":[]},{\"tool_name_cn\":\"通过URL采集\",\"tool_sign\":\"github_repository_by-url\",\"spider_name\":\"github.com\",\"params\":[]}]"},{"path":"skill-card.md","content":"## Description:\n\nCollect structured GitHub repository information from one or more known repository URLs. Do not use for GitHub code search or arbitrary webpages.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[dataify-server](https://clawhub.ai/user/dataify-server)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and agents use this skill to collect structured data for known public GitHub repository URLs through Dataify, monitor the asynchronous task, and return the final JSON result.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Repository targets and task results are sent through Dataify using the user's API token.\n\nMitigation: Install only when this data flow is acceptable; keep the token in the environment and do not paste it into chat or generated outputs.\n\nRisk: The package contains broader business, search, and web unlocker workflows beyond the narrow GitHub repository collection purpose.\n\nMitigation: Review the installed artifact and prefer a narrowed release that removes unrelated workflows before using it in sensitive environments.\n\nRisk: The security guidance flags scoping and credential-handling issues, including GitHub URL validation, import path behavior, curl quoting, and reusable tokens in URLs.\n\nMitigation: Validate target URLs as GitHub repository URLs and resolve the flagged implementation issues before production deployment.\n\n## Reference(s):\n\n- [ClawHub Skill Page](https://clawhub.ai/dataify-server/skills/dataify-github-repository-by-repo-url)\n- [Publisher Profile](https://clawhub.ai/user/dataify-server)\n- [Tool Parameter Catalog](references/tool-params.json)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown guidance with shell commands and JSON results from completed Dataify tasks]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Requires a DATAIFY_API_TOKEN environment variable and a public GitHub repository URL; default behavior waits for completion and returns the final collected result.]\n\n## Skill Version(s):\n\n1.3.1 (source: server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."},{"path":"SKILL.zh-CN.md","content":"---\nname: \"dataify-github-repository-by-repo-url\"\ndescription: \"为 github.com 上以 github_repository_by-repo-url 为根的 scraper 系列准备 Dataify builder 请求。当需要处理成功的 Dataify scraper detail 条目 github_repository_by-repo-url、让用户选择可用工具、读取已保存的 getToolParams 选项，并使用 DATAIFY_API_TOKEN 生成 scraperapi.dataify.com/builder curl 请求时，使用此 skill。\"\n---\n\n# Dataify Builder Skill 中文版\n\n这个 skill 用于为 `github.com` 下、以 `github_repository_by-repo-url` 为入口的 Dataify scraper 工具族生成 builder 请求。\n\n## 工作流程\n\n1. 先检查环境变量中是否存在 `DATAIFY_API_TOKEN`。\n2. 如果 token 缺失，告诉用户：`Dataify 需要 API Token。新账号注册即得 50 免费积分，约可获得 6000 条试用结果，7 天有效，仅成功请求计费。注册完成后告诉我，我会继续当前任务。`。\n3. 先让用户从下面的中文工具列表中明确选择一个工具：\n- 通过仓库URL采集 (github_repository_by-repo-url)\n- 通过搜索URL采集 (github_repository_by-search-url)\n- 通过URL采集 (github_repository_by-url)\n4. 再读取 `references/tool-params.json`，根据 `tool_sign` 或中文工具名找到对应工具。\n5. 对所选工具的每个参数分别处理：\n   - 如果 `input_mode` 是 `user_input`，让用户提供值。\n   - 如果 `input_mode` 是 `select`，把已保存的可选项展示给用户，让用户选择。\n6. 默认优先使用 `scripts/build-dataify-request.py`，因为它是跨平台版本。\n7. Windows 下也可以使用 `scripts/build-dataify-request.ps1`。\n8. 对于可选型参数，如果存在人类可读标签，优先把该标签写入 `spider_parameters`。\n9. `spider_parameters` 必须是一个 JSON 数组。\n10. 像搜索 URL 这类多值工具，需要按索引生成多个对象。\n11. `spider_name` 固定取 `github.com`。\n12. `spider_id` 固定取用户所选工具的 `tool_sign`。\n13. 始终包含 `spider_errors=true` 和 `file_name={{TasksID}}`。\n\n## 设置 DATAIFY_API_TOKEN\n\n推荐使用永久环境变量，而不是只在当前终端临时设置。\n\nWindows PowerShell，当前用户永久设置：\n\n```powershell\n[Environment]::SetEnvironmentVariable(\"DATAIFY_API_TOKEN\", \"your_token_here\", \"User\")\n```\n\n然后重新打开 PowerShell。如果当前会话也要立即生效，再执行：\n\n```powershell\n$env:DATAIFY_API_TOKEN = \"your_token_here\"\n```\n\nmacOS 或 Linux，bash 永久设置：\n\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.bashrc\nsource ~/.bashrc\n```\n\nmacOS 或 Linux，zsh 永久设置：\n\n```bash\necho 'export DATAIFY_API_TOKEN=\"your_token_here\"' >> ~/.zshrc\nsource ~/.zshrc\n```\n\n## 脚本用法\n\nPython：\n\n```bash\npython scripts/build-dataify-request.py --tool-sign <selected_tool_sign> --values-file values.json\n```\n\nPowerShell：\n\n```powershell\n& \".\\scripts\\build-dataify-request.ps1\" -ToolSign \"<selected_tool_sign>\" -ValuesFile \".\\values.json\"\n```\n\n`values.json` 可以是单个对象，也可以是对象数组。\n\n## 输出格式\n\n最终 `curl` 命令应为：\n\n```bash\ncurl -X POST 'https://scraperapi.dataify.com/builder' \\\n  -H \"Authorization: Bearer $DATAIFY_API_TOKEN\" \\\n  -H 'Content-Type: application/x-www-form-urlencoded' \\\n  -d 'spider_name=github.com' \\\n  -d 'spider_id=<selected_tool_sign>' \\\n  -d 'spider_parameters=[{\"param\":\"value\"}]' \\\n  -d 'spider_errors=true' \\\n  -d 'file_name={{TasksID}}'\n```\n\n## 参考文件\n\n- `references/tool-params.json` 保存了这个 skill 下所有工具及参数选项。\n- `scripts/build-dataify-request.py` 是首选的跨平台实现。\n- `scripts/build-dataify-request.ps1` 是 Windows PowerShell 版本。\n- 如果参数没有预设选项，必须向用户要值。\n- 不要假设 `spider_parameters` 永远只有一个对象；多值工具可能需要按索引生成多个对象。\n- `url_example` 仅作为参考，不要默认用户就要用示例值，除非用户明确确认。\n\n## 参数交互策略\n\n- 当请求意图明确、只读、低风险且成本较低时，使用安全默认值直接执行。可以用一句话说明执行内容，但不要暂停等待确认。\n- 只在缺少必填输入、存在会明显改变结果的歧义、大批量或多页采集、媒体下载、会明显增加积分消耗、不可逆操作，或用户明确要求查看参数时询问。\n- 必须确认时，只展示会影响目标、范围、输出或成本的用户参数。优先使用一句简短说明；只有三个及以上"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Collect Github Builder data and return results Skill: Dataify Github Builder Owner: dataify-server Summary: Collect Github Builder data and return results Tags: latest:1.3.1 Version history: v1.3.1 | 2026-09-08T06:18:16.065Z | user Fix natural-language usage failures: validate required targets and URLs, preserve catalog references, default Amazon region safely, normalize Google News links, and improve UTF-8 error output. v1.3.0 | 2026-09-01T09:13:31.285Z | user 默","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1049,"uniquenessScore":48,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T08:11:10.735Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T08:11:10.735Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T10:49:28.535Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}