{"id":"67bb5914-7d91-423a-8a7a-b92f5aaceed1","entityType":"agent","slug":"clawhub-crawlora-org-crawlora-datasets","name":"crawlora-datasets","canonicalUrl":"https://www.xpersona.co/agent/clawhub-crawlora-org-crawlora-datasets","canonicalPath":"/agent/clawhub-crawlora-org-crawlora-datasets","generatedAt":"2026-10-11T07:39:51.579Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-11T04:01:55.945Z","emptyReason":null},"description":"Queries Crawlora's pre-built hosted datasets — Airbnb markets, App Store/Google Play apps, GitHub/Instagram/X users, job postings, US housing markets, Google Maps businesses, Goodreads, PitchBook, Steam, TrustMRR, Product Hunt, SEC companies, tech-stack, and more — via search/facets/item/nearby endpoints, returning clean JSON without live-crawling each platform. Use when the user wants bulk or aggregate analysis, to search a pre-indexed corpus, to facet/filter a large population, or to look up one record by its dataset id, instead of scraping pages one at a time. Skill: crawlora-datasets Owner: crawlora-org Summary: Queries Crawlora's pre-built hosted datasets — Airbnb markets, App Store/Google Play apps, GitHub/Instagram/X users, job postings, US housing markets, Google Maps businesses, Goodreads, PitchBook, Steam, TrustMRR, Product Hunt, SEC companies, tech-stack, and more — via search/facets/item/nearby endpoints, returning clean JSON without live-crawling each platform. U","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.2K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s17d53nb8nd03gyyfdy32rgde58e574f:crawlora-datasets","sourceUrl":"https://clawhub.ai/crawlora-org/crawlora-datasets","homepage":"https://clawhub.ai/crawlora-org/skills/crawlora-datasets","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/crawlora-org/crawlora-datasets","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/crawlora-org/skills/crawlora-datasets","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":61,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Queries Crawlora's pre-built hosted datasets — Airbnb markets, App Store/Google Play apps, GitHub/Instagram/X users, job postings, US housing markets, Google Ma"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T04:01:55.945Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T04:01:55.945Z","emptyReason":null},"stars":null,"forks":null,"downloads":1166,"packageName":null,"latestVersion":"1.0.20","tractionLabel":"1.2K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T04:01:55.832Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T04:01:55.945Z","lastCrawledAt":"2026-10-11T04:01:55.832Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T04:01:55.832Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.20","createdAt":"2026-10-05T01:12:25.933Z","changelog":"Sync skill instructions, references, and helper from GitHub 83bb98ef1362f25cecb5ddd4bc1ea0e564555d97","fileCount":5,"zipByteSize":40825},{"version":"1.0.19","createdAt":"2026-09-26T12:39:43.669Z","changelog":"Sync skill instructions, references, and helper from GitHub 8ac7f99e79c37939913706006fca839ef1108182","fileCount":5,"zipByteSize":39190},{"version":"1.0.18","createdAt":"2026-09-21T01:31:39.617Z","changelog":"Sync skill instructions, references, and helper from GitHub 0cfbceba40b050ba434a0a3f4945ca97b668c805","fileCount":5,"zipByteSize":39480},{"version":"1.0.17","createdAt":"2026-09-17T10:22:14.070Z","changelog":"Security hardening: generated helpers now enforce exact routes, methods, and credential-safe curl behavior.","fileCount":5,"zipByteSize":38570},{"version":"1.0.16","createdAt":"2026-09-14T01:38:44.503Z","changelog":"Sync skill instructions, references, and helper from GitHub 902f58316c643ffbcabc57fc6f15f59d27ec063d","fileCount":5,"zipByteSize":39016},{"version":"1.0.15","createdAt":"2026-09-10T12:31:00.547Z","changelog":"Validate API keys before curl config","fileCount":5,"zipByteSize":38750},{"version":"1.0.14","createdAt":"2026-09-10T12:14:42.273Z","changelog":"Keep API keys out of process arguments","fileCount":5,"zipByteSize":38692},{"version":"1.0.13","createdAt":"2026-09-10T12:05:33.263Z","changelog":"Reject curl local-file query syntax","fileCount":5,"zipByteSize":38658}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17d53nb8nd03gyyfdy32rgde58e574f:crawlora-datasets","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-crawlora-org-crawlora-datasets/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-crawlora-org-crawlora-datasets/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-crawlora-org-crawlora-datasets/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-crawlora-org-crawlora-datasets/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-crawlora-org-crawlora-datasets/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-crawlora-org-crawlora-datasets/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T07:39:51.572Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-crawlora-org-crawlora-datasets/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-crawlora-org-crawlora-datasets/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-crawlora-org-crawlora-datasets/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-crawlora-org-crawlora-datasets/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-11T04:01:55.945Z","emptyReason":null},"readme":"Skill: crawlora-datasets\n\nOwner: crawlora-org\n\nSummary: Queries Crawlora's pre-built hosted datasets — Airbnb markets, App Store/Google Play apps, GitHub/Instagram/X users, job postings, US housing markets, Google Maps businesses, Goodreads, PitchBook, Steam, TrustMRR, Product Hunt, SEC companies, tech-stack, and more — via search/facets/item/nearby endpoints, returning clean JSON without live-crawling each platform. Use when the user wants bulk or aggregate analysis, to search a pre-indexed corpus, to facet/filter a large population, or to look up one record by its dataset id, instead of scraping pages one at a time.\n\nTags: latest:1.0.20\n\nVersion history:\n\nv1.0.20 | 2026-10-05T01:12:25.933Z | user\n\nSync skill instructions, references, and helper from GitHub 83bb98ef1362f25cecb5ddd4bc1ea0e564555d97\n\nv1.0.19 | 2026-09-26T12:39:43.669Z | user\n\nSync skill instructions, references, and helper from GitHub 8ac7f99e79c37939913706006fca839ef1108182\n\nv1.0.18 | 2026-09-21T01:31:39.617Z | user\n\nSync skill instructions, references, and helper from GitHub 0cfbceba40b050ba434a0a3f4945ca97b668c805\n\nv1.0.17 | 2026-09-17T10:22:14.070Z | user\n\nSecurity hardening: generated helpers now enforce exact routes, methods, and credential-safe curl behavior.\n\nv1.0.16 | 2026-09-14T01:38:44.503Z | user\n\nSync skill instructions, references, and helper from GitHub 902f58316c643ffbcabc57fc6f15f59d27ec063d\n\nv1.0.15 | 2026-09-10T12:31:00.547Z | user\n\nValidate API keys before curl config\n\nv1.0.14 | 2026-09-10T12:14:42.273Z | user\n\nKeep API keys out of process arguments\n\nv1.0.13 | 2026-09-10T12:05:33.263Z | user\n\nReject curl local-file query syntax\n\nv1.0.12 | 2026-09-10T11:52:39.370Z | user\n\nStream helper request bodies through curl stdin\n\nv1.0.11 | 2026-09-10T11:36:44.282Z | user\n\nScope helper routes and remove secret-shaped key examples\n\nv1.0.10 | 2026-09-10T06:35:54.226Z | user\n\nMigrate publisher from tonywangcn to crawlora-org for brand consistency with the plugins\n\nv1.0.9 | 2026-09-08T04:29:58.278Z | user\n\nRefresh stale REST examples, endpoint references, and Bash helper from crawlora-skills 1.17.1.\n\nv1.0.8 | 2026-09-07T13:28:29.108Z | user\n\nSync via scripts/sync-directories.sh\n\nv1.0.7 | 2026-09-07T08:14:03.379Z | user\n\nSync via scripts/sync-directories.sh\n\nv1.0.6 | 2026-09-07T06:37:45.829Z | user\n\nSync via scripts/sync-directories.sh\n\nv1.0.5 | 2026-08-24T07:14:57.999Z | user\n\nSync via scripts/sync-directories.sh\n\nv1.0.4 | 2026-08-24T06:27:22.905Z | user\n\nSync via scripts/sync-directories.sh\n\nv1.0.3 | 2026-08-24T05:09:33.452Z | user\n\nSync via scripts/sync-directories.sh\n\nv1.0.2 | 2026-08-14T18:26:54.624Z | user\n\nSync via scripts/sync-directories.sh\n\nv1.0.1 | 2026-08-10T18:25:22.986Z | user\n\nSet categories\n\nv1.0.0 | 2026-08-10T18:00:31.483Z | auto\n\nInitial release: Query pre-built, hosted datasets from Crawlora for bulk or aggregate research, without live crawling.\n\n- Supports full-text search, faceting, record lookup, and geo/nearby queries across multiple public datasets (Airbnb, app stores, GitHub, jobs, housing, Google Maps, and more).\n- Clean, consistent JSON results suitable for bulk analysis or large-scale filtering.\n- Use for aggregate questions, finding all matches to criteria, or retrieving lists, rather than scraping individual records.\n- Setup via a free API key; 2,000 credits/month free tier.\n- Results reflect pre-indexed data periodically refreshed, not real-time.\n\nArchive index:\n\nArchive v1.0.20: 5 files, 40825 bytes\n\nFiles: reference/endpoints.md (179855b), scripts/crawlora.sh (10269b), skill-card.md (1643b), SKILL.md (6177b), _meta.json (137b)\n\nFile v1.0.20:SKILL.md\n\n---\nname: crawlora-datasets\ndescription: Queries Crawlora's pre-built hosted datasets — Airbnb markets, App Store/Google Play apps, GitHub/Instagram/X users, job postings, US housing markets, Google Maps businesses, Goodreads, PitchBook, Steam, TrustMRR, Product Hunt, SEC companies, tech-stack, and more — via search/facets/item/nearby endpoints, returning clean JSON without live-crawling each platform. Use when the user wants bulk or aggregate analysis, to search a pre-indexed corpus, to facet/filter a large population, or to look up one record by its dataset id, instead of scraping pages one at a time.\n---\n\n# Crawlora hosted datasets\n\nQuery Crawlora's own **pre-crawled, pre-indexed datasets** — search, facet, and\nfetch-by-id over corpora Crawlora already built and refreshes on a schedule.\nThis is different from the other skills in this repo: those hit a live\nper-platform endpoint (one request, one page); this skill hits a **search\nindex** over millions of already-collected records, so it's the right tool\nfor population-level questions (\"how many\", \"top N by X\", \"everything\nmatching Y\") rather than one-off lookups.\n\n## When to use this skill\n\n- \"How many / what share of X match Y?\" — facet/aggregate questions.\n- \"Find all X with property Y\" (e.g. jobs paying > $150k, apps with 4.5+\n  rating, GitHub users near a city, houses in a metro).\n- \"Give me the full list of Z\" instead of one record — bulk/list research.\n- Any of: Airbnb markets, app-store apps/reviews/charts, GitHub/Instagram/X\n  users, job postings + which companies are hiring, US housing markets\n  (Redfin-sourced), Google Maps businesses, Goodreads authors/books, Apple\n  Podcasts shows, Chrome Web Store extensions, PitchBook companies/funds/\n  investors/advisors/LPs, PlayStation games, Product Hunt makers/products/\n  trends, Reddit trending, SEC companies + institutional positions, Steam\n  games/prices/playercounts/reviews/news/achievements/charts, TrustMRR\n  startups, journalists, Numbeo cost-of-living cities/countries, website\n  tech-stack.\n- Prefer the platform-specific skill instead when the job is \"look up this\n  one profile/listing right now\" (e.g. `youtube-research`, `movie-tv-research`)\n  — datasets are refreshed periodically, not real-time.\n\n## Setup (one-time)\n\n- Get a free Crawlora API key (2,000 credits/mo, no card) at [https://crawlora.net](https://crawlora.net?utm_source=github&utm_medium=referral&utm_campaign=crawlora-skills).\n- Set `CRAWLORA_API_KEY` in the environment before running the helper.\n- The helper reads `CRAWLORA_API_KEY` from the environment and sends requests to `https://api.crawlora.net/api/v1`. Missing/invalid key → `401`.\n\n## How it works\n\nEvery dataset follows the same shape under `/datasets/<dataset-id>/...`:\n\n1. **Discover** — `GET /datasets` lists every available dataset id and its\n   capabilities (search / facets / item / nearby).\n2. **Search** — `GET /datasets/<id>/search` full-text + filtered search;\n   paginate with `page`/`size` (see `reference/endpoints.md` per dataset).\n3. **Facet** — `GET /datasets/<id>/facets` returns aggregate breakdowns\n   across a dataset's facetable fields at once (e.g. the jobs dataset\n   returns top companies, department, location, seniority, remote share,\n   and more in one call) — use for \"how many / breakdown by X\" questions.\n4. **Item** — `GET /datasets/<id>/items/{id}` fetches one record by its\n   dataset key (varies per dataset: `login`, `username`, `slug`, `cik`,\n   `appid`, `domain`, `region_type/table_id`, …).\n5. **Nearby** — where supported (`airbnb-markets`, `github-users`,\n   `google-map-businesses`, `jobs`) — `GET /datasets/<id>/nearby` finds\n   records near a `lat`/`lon` within `radius_km`.\n\nFull endpoint list, per-dataset ids, and params: [`reference/endpoints.md`](reference/endpoints.md).\n\n## Calling the API\n\n```sh\n# List every dataset id and what it supports:\nscripts/crawlora.sh /datasets | jq '.'\n\n# Search the jobs dataset (all companies' live postings):\nscripts/crawlora.sh /datasets/jobs/search q=\"staff engineer\" location=\"remote\" | jq '.'\n\n# Facet: hiring-market breakdown (top companies, seniority, remote share, ...):\nscripts/crawlora.sh /datasets/jobs/facets | jq '.'\n\n# Item: one GitHub user by login:\nscripts/crawlora.sh /datasets/github-users/items/torvalds | jq '.'\n\n# Nearby: GitHub users within 50km of a coordinate (radius in meters):\nscripts/crawlora.sh /datasets/github-users/nearby lat=37.7749 lon=-122.4194 radius_m=50000 | jq '.'\n```\n\nUse `scripts/crawlora.sh` for all requests; it keeps the API key out of command-line arguments.\n\n\n## Endpoint reference\n\nSee [`reference/endpoints.md`](reference/endpoints.md) for every dataset id,\nits search/facets/item/nearby endpoints, and params.\n\n## Examples\n\n- **Hiring-market pulse:** `/datasets/jobs/facets` for the aggregate\n  breakdown (top companies, seniority, remote share), then\n  `/datasets/jobs/companies` to see which employers are actively posting.\n- **App-store landscape scan:** `/datasets/apps/search` filtered by category\n  and rating, then `/datasets/apps-reviews/search` for the sentiment behind\n  the top results.\n- **Startup revenue leaderboard:** `/datasets/trustmrr/search` sorted by\n  MRR, then `/datasets/trustmrr/history/{slug}` for one company's trend line.\n- **Housing-market snapshot:** `/datasets/housing-markets/search` for a\n  metro, then `/datasets/housing-markets/items/{region_type}/{table_id}` for\n  the full monthly series.\n\n## Notes & limits\n\n- **Credits / pay-on-success:** billed only on `2xx`; free tier 2,000 credits/mo.\n  Key at [https://crawlora.net](https://crawlora.net?utm_source=github&utm_medium=referral&utm_campaign=crawlora-skills).\n- **Public data only** — every dataset is built from public sources.\n- **Security:** key lives in `CRAWLORA_API_KEY` only — never hardcode, query-param, or commit it.\n- Datasets refresh on a schedule (daily/weekly depending on source) — not\n  real-time. For a live single-record lookup, prefer the matching\n  platform-specific skill in this repo (e.g. `job-market-research`,\n  `movie-tv-research`) instead.\n- Results are paginated (`page`/`size`) — walk pages for full coverage.\n\nFile v1.0.20:_meta.json\n\n{\n  \"ownerId\": \"kn70shhkf6qpfwgfrbgtep2wkd8c6b4t\",\n  \"slug\": \"crawlora-datasets\",\n  \"version\": \"1.0.20\",\n  \"publishedAt\": 1791162745933\n}\n\nFile v1.0.20:reference/endpoints.md\n\n# crawlora-datasets — endpoint reference\n\n> Generated from `scripts/tools.json` by `scripts/generate.mjs` — do not edit by hand.\n\nEndpoints this skill uses, grouped by platform. Call them via `scripts/crawlora.sh` (see SKILL.md).\n\nAll paths are relative to the API base `https://api.crawlora.net/api/v1` and require the header `x-api-key: $CRAWLORA_API_KEY`. Path params like `{id}` are substituted into the URL; `GET` params go in the query string; `POST` params go in a JSON body.\n\n**130 endpoints across 1 platform group(s).**\n\n## Datasets (130)\n\n### `datasets_airbnb_facets`\n\n- **HTTP:** `GET /datasets/airbnb-markets/facets`\n- **What:** Facet the Airbnb markets dataset. Returns suppressed distribution counts over the Airbnb markets dataset, honoring the same filters as search. Facet enum: `country`, `market`, `currency`, `superhost`, `guest_favorite`, `rating_band`, `review_band`, `admin1` (top subdivision), `locality` (settlement), `room_type` (`entire_place`/`private_room`/`hotel`/`shared_room`), `property_type` (Airbnb's canonical listing type from the detail page), `amenities` (each amenity with the count of listings offering it). The `admin1`, `locality`, `room_type`, `property_type` and `amenities` facets stay empty until their enrichment coverage is high enough to be reliable. group_by enum: `country`, `market`, `admin1`, `locality`, `room_type`, `property_type`.\n- **Params:** `active_since` (string, optional) — Freshness filter, an ISO-8601 date (YYYY-MM-DD); `country` (string, optional) — Exact ISO-3166-1 alpha-2 country filter, e.g. FR; `facet` (string, **required**) — Facet enum: country, market, currency, superhost, guest_favorite, rating_band, review_band, admin1, locality, room_type, property_type, amenities; `group_by` (string, optional) — Aggregate cell dimension enum: country, market, admin1, locality, room_type, property_type. Defaults to country; `guest_favorite` (boolean, optional) — Count only Guest Favorite listings (an observed lower bound; the badge under-counts); `market` (string, optional) — Exact metro-market filter, max 128 characters; `min_listings` (integer, optional) — Minimum listings per bucket; raises the small-cell suppression floor; `min_rating` (number, optional) — Minimum listing rating, from 0 through 5; `min_review_count` (integer, optional) — Minimum listing review count, 0 or greater; `superhost` (boolean, optional) — Count only Superhost listings\n\n### `datasets_airbnb_item`\n\n- **HTTP:** `GET /datasets/airbnb-markets/items/{country}`\n- **What:** Get an Airbnb market from the dataset. Returns one country's full aggregate Airbnb market profile from dataset id enum value `airbnb-markets` — headline supply, Superhost share, Guest Favorite share (`guest_favorite_pct`, an observed lower bound), `avg_person_capacity` (average guests a listing sleeps over the detail-page-enriched sample), ratings, its top metros, bounding box, per-currency nightly-price percentiles, and a USD-normalized `price_usd` percentile block (converted via an approximate dated FX snapshot) for cross-country comparison. Aggregate-only. Returns 404 for a country below the suppression floor.\n- **Params:** `country` (string, **required**) — ISO-3166-1 alpha-2 country code, e.g. FR\n\n### `datasets_airbnb_nearby`\n\n- **HTTP:** `GET /datasets/airbnb-markets/nearby`\n- **What:** Airbnb market density near a coordinate. Returns an aggregate geohash-grid density map of Airbnb listings within a radius of a coordinate, from dataset id enum value `airbnb-markets`. Each cell reports a centroid, listing count and Superhost share; thin cells are suppressed. Aggregate-only.\n- **Params:** `active_since` (string, optional) — Freshness filter, an ISO-8601 date (YYYY-MM-DD); `country` (string, optional) — Exact ISO-3166-1 alpha-2 country filter, e.g. US; `lat` (number, **required**) — Center latitude, from -90 through 90; `lon` (number, **required**) — Center longitude, from -180 through 180; `min_listings` (integer, optional) — Minimum listings per cell; raises the small-cell suppression floor; `min_rating` (number, optional) — Minimum listing rating, from 0 through 5; `precision` (integer, optional) — Geohash precision, from 1 through 12; defaults to a value derived from the radius; `radius_m` (integer, **required**) — Search radius in meters, from 1 through 50000; `superhost` (boolean, optional) — Count only Superhost listings\n\n### `datasets_airbnb_search`\n\n- **HTTP:** `GET /datasets/airbnb-markets/search`\n- **What:** Search the Airbnb markets dataset. Returns aggregate Airbnb short-term-rental market rollups from the dataset id enum value `airbnb-markets`. Aggregate-only: each row is a market cell, never an individual listing. Thin cells are suppressed. group_by enum: `country`, `market`, `admin1` (top subdivision), `locality` (settlement), `room_type` (`entire_place`/`private_room`/`hotel`/`shared_room`), `property_type` (Airbnb's canonical listing type from the detail page). `admin1`, `locality`, `room_type` and `property_type` are enrichment-derived and stay empty until their coverage is high enough to be reliable. Each cell also carries `median_price_usd`, the median nightly price converted to USD via an approximate dated FX snapshot, for cross-country comparison (combine with `group_by=room_type` for median price by room type); `guest_favorite_pct`, the share of listings carrying the Guest Favorite badge (an observed lower bound, like `superhost_pct`); and `avg_person_capacity`, the average guests a listing sleeps over the detail-page-enriched sample. Sort enum: `listings_desc`, `superhost_pct_desc`, `rating_desc`, `key_asc`.\n- **Params:** `active_since` (string, optional) — Freshness filter, an ISO-8601 date (YYYY-MM-DD); only listings last seen on or after it are counted; `country` (string, optional) — Exact ISO-3166-1 alpha-2 country filter, e.g. FR; `group_by` (string, optional) — Aggregate cell dimension enum: country, market, admin1, locality, room_type, property_type. Defaults to country; `guest_favorite` (boolean, optional) — Count only Guest Favorite listings (an observed lower bound; the badge under-counts); `market` (string, optional) — Exact metro-market filter, e.g. Paris, max 128 characters; `min_listings` (integer, optional) — Minimum listings per cell; raises the small-cell suppression floor (never lowered below the built-in minimum); `min_rating` (number, optional) — Minimum listing rating, from 0 through 5; `min_review_count` (integer, optional) — Minimum listing review count, 0 or greater; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `sort` (string, optional) — Sort enum: listings_desc, superhost_pct_desc, rating_desc, key_asc; `superhost` (boolean, optional) — Count only Superhost listings\n\n### `datasets_apple_podcasts_shows_facets`\n\n- **HTTP:** `GET /datasets/apple-podcasts-shows/facets`\n- **What:** Facet Apple Podcasts shows dataset. Returns terms aggregation counts for the Apple Podcasts shows dataset. Facet enum: `genre`, `genre_id`, `country`, `content_advisory_rating`, `run_id`.\n- **Params:** `country` (string, optional) — Exact storefront country filter, max 128 characters; `explicitness` (string, optional) — Exact explicitness filter, max 128 characters; `facet` (string, **required**) — Facet enum: genre, genre_id, country, content_advisory_rating, run_id; `genre` (string, optional) — Exact primary-genre filter, max 128 characters; `genre_id` (string, optional) — Exact Apple Podcasts genre id filter, max 128 characters; `min_track_count` (integer, optional) — Minimum episode count (track_count), 0 or greater; `q` (string, optional) — Full-text query over show title and artist name, max 256 characters; `run_id` (string, optional) — Exact crawl run-id filter, max 128 characters\n\n### `datasets_apple_podcasts_shows_item`\n\n- **HTTP:** `GET /datasets/apple-podcasts-shows/items/{id}`\n- **What:** Get an Apple Podcasts show from dataset. Returns one crawled Apple Podcasts show record by id from dataset id enum value `apple-podcasts-shows`.\n- **Params:** `id` (string, **required**) — Apple Podcasts numeric show id (e.g. 173001861)\n\n### `datasets_apple_podcasts_shows_search`\n\n- **HTTP:** `GET /datasets/apple-podcasts-shows/search`\n- **What:** Search Apple Podcasts shows dataset. Searches the crawled public Apple Podcasts show catalog stored in a search index. One row per show. Discovered from a country x genre x collection chart grid and a search-term sweep — not a full catalog of every Apple Podcasts show. Sort enum: `relevance`, `popularity`, `track_count_desc`, `release_desc`, `title_asc`.\n- **Params:** `country` (string, optional) — Exact storefront country filter (the crawl's discovery storefront, e.g. us, gb), max 128 characters; `explicitness` (string, optional) — Exact explicitness filter as reported by Apple (e.g. explicit, cleaned), max 128 characters; `genre` (string, optional) — Exact primary-genre filter (e.g. Comedy, True Crime), max 128 characters; `genre_id` (string, optional) — Exact Apple Podcasts genre id filter (e.g. 1303 for Comedy), max 128 characters; `min_track_count` (integer, optional) — Minimum episode count (track_count), 0 or greater; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `q` (string, optional) — Full-text query over show title and artist name, max 256 characters; `run_id` (string, optional) — Exact crawl run-id filter, max 128 characters; `sort` (string, optional) — Sort enum: relevance, popularity, track_count_desc, release_desc, title_asc\n\n### `datasets_apps_charts_search`\n\n- **HTTP:** `GET /datasets/apps-charts/search`\n- **What:** Search the app-charts dataset. Searches daily top-chart snapshots scraped from the iOS App Store and Google Play, stored in a search index (one document per chart × snapshot × rank). With no `date` the latest snapshot is returned (today's chart); pair `app_id` with `sort=date_desc` for an app's rank over time. Store enum: `ios`, `android`. Chart type enum: `top_free`, `top_paid`, `top_grossing`, `new`. Platform enum (Apple device platforms, ios charts only): `phone`, `pad`, `mac`. Sort enum: `rank`, `rank_desc`, `date_desc`.\n- **Params:** `app_id` (string, optional) — Exact app filter — iOS numeric track id or Android package; pair with sort=date_desc for rank history; `category` (string, optional) — Store category/genre filter, max 128 characters; empty for the overall charts; `chart_type` (string, optional) — Chart enum: top_free, top_paid, top_grossing, new; `collection` (string, optional) — Raw store collection id filter (e.g. topgrossingapplications, GROSSING), max 128 characters; `country` (string, optional) — Exact storefront country filter, max 128 characters; `date` (string, optional) — Snapshot date filter yyyy-MM-dd; defaults to the latest snapshot; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `platform` (string, optional) — Apple device-platform filter, iOS charts only; see platform enum above; `q` (string, optional) — Full-text query over chart-entry title and developer, max 256 characters; `sort` (string, optional) — Sort enum: rank, rank_desc, date_desc; `store` (string, optional) — Store enum: ios, android\n\n### `datasets_apps_reviews_search`\n\n- **HTTP:** `GET /datasets/apps-reviews/search`\n- **What:** Search the app-reviews dataset. Searches user reviews scraped from the iOS App Store and Google Play, stored in a search index (one document per review). Store enum: `ios`, `android`. Sort enum: `recent`, `score_desc`, `score_asc`, `helpful_desc`.\n- **Params:** `app_id` (string, optional) — Exact app filter — iOS numeric track id or Android package, max 128 characters; `country` (string, optional) — Exact storefront country filter, max 128 characters; `min_score` (integer, optional) — Minimum star rating, 1 through 5; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `q` (string, optional) — Full-text query over review text, title and author, max 256 characters; `sort` (string, optional) — Sort enum: recent, score_desc, score_asc, helpful_desc; `store` (string, optional) — Store enum: ios, android\n\n### `datasets_apps_search`\n\n- **HTTP:** `GET /datasets/apps/search`\n- **What:** Search the apps-intelligence dataset. Searches resolved iOS App Store and Google Play apps stored in a search index. Store enum: `ios`, `android`, `both`. Platform enum (Apple device platforms, ios records only): `phone`, `pad`, `mac`, `tv`, `watch`, `vision`. Sort enum: `relevance`, `rating_desc`, `reviews_desc`, `installs_desc`, `updated_at_desc`, `popularity_desc`.\n- **Params:** `category` (string, optional) — Exact app-store category filter, max 128 characters; `country` (string, optional) — Exact storefront country filter, max 128 characters; `developer` (string, optional) — Exact developer/publisher name filter, max 128 characters; `free` (boolean, optional) — Filter by price; true keeps only free apps, false only paid; `min_rating` (number, optional) — Minimum store rating, 0 through 5; `min_reviews` (integer, optional) — Minimum ratings/review count; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `platforms` (array, optional) — Repeatable Apple device-platform filter (OR); see platform enum above; `q` (string, optional) — Full-text query over title, developer and category, max 256 characters; `sort` (string, optional) — Sort enum: relevance, rating_desc, reviews_desc, installs_desc, updated_at_desc, popularity_desc; `store` (string, optional) — Store enum: ios, android, both\n\n### `datasets_bbb_businesses_facets`\n\n- **HTTP:** `GET /datasets/bbb-businesses/facets`\n- **What:** Facet the BBB businesses dataset. Returns distribution counts over the BBB businesses index (dataset id enum value `bbb-businesses`), honoring the same filters as search. Facet enum: `category`, `state`, `city`, `rating`, `accredited`, `entity_type`, `run_id`.\n- **Params:** `accredited` (boolean, optional) — Accreditation filter; `category` (string, optional) — Exact category filter; `city` (string, optional) — Exact city filter; `entity_type` (string, optional) — Exact entity-type filter; `facet` (string, **required**) — Facet enum: category, state, city, rating, accredited, entity_type, run_id; `q` (string, optional) — Full-text match on the business name/category, max 256 characters; `rating` (string, optional) — Exact letter-grade rating filter. Enum: A+, A, A-, B+, B, B-, C+, C, C-, D+, D, D-, F; `run_id` (string, optional) — Exact crawl run id filter; `state` (string, optional) — Exact 2-letter state/province filter\n\n### `datasets_bbb_businesses_item`\n\n- **HTTP:** `GET /datasets/bbb-businesses/items/{id}`\n- **What:** Get a business from the BBB businesses dataset. Returns one business by id from dataset id enum value `bbb-businesses`. Returns 404 when the business is not in the index.\n- **Params:** `id` (string, **required**) — Business id (the <bbbLocalId>-<businessId> slug from the profile URL), e.g. 0825-1000223803\n\n### `datasets_bbb_businesses_search`\n\n- **HTTP:** `GET /datasets/bbb-businesses/search`\n- **What:** Search the BBB businesses dataset. Searches the BBB (Better Business Bureau) businesses index (dataset id enum value `bbb-businesses`) — business profiles crawled from bbb.org's own search/category-browse pages: computed A+-F letter-grade rating, paid-accreditation status, category, contact info, business details, operating hours, and products/services. Complaints, full reviews, and the full \"reasons for rating\"/service-area detail are NOT embedded here; each record instead carries complaints_url/reviews_url/more_info_url pointing at the live bbb-business-complaints/bbb-business-reviews/bbb-business-more-info endpoints for on-demand lookup. rating enum: `A+`, `A`, `A-`, `B+`, `B`, `B-`, `C+`, `C`, `C-`, `D+`, `D`, `D-`, `F`. sort enum: `relevance`, `rating_desc`, `rating_asc`, `accredited_first`, `name_asc`, `years_in_business_desc`.\n- **Params:** `accredited` (boolean, optional) — Accreditation filter; true keeps only accredited businesses; `category` (string, optional) — Exact category filter, e.g. Plumber. Use the values returned by facets?facet=category; `city` (string, optional) — Exact city filter, parsed from the profile URL; `entity_type` (string, optional) — Exact entity-type filter, e.g. Limited Liability Company (LLC); `min_rating_rank` (integer, optional) — Numeric floor against the denormalized rating rank (A+=12 down to F=0), e.g. 10 for 'A- and above'; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `q` (string, optional) — Full-text match on the business name/category, max 256 characters; `rating` (string, optional) — Exact letter-grade rating filter. Enum: A+, A, A-, B+, B, B-, C+, C, C-, D+, D, D-, F; `run_id` (string, optional) — Exact crawl run id filter; `sort` (string, optional) — Sort enum: relevance, rating_desc, rating_asc, accredited_first, name_asc, years_in_business_desc; `state` (string, optional) — Exact 2-letter state/province filter, parsed from the profile URL, e.g. tx\n\n### `datasets_boxofficemojo_facets`\n\n- **HTTP:** `GET /datasets/boxofficemojo/facets`\n- **What:** Facet the Box Office Mojo dataset. Returns terms-aggregation counts for one facet of the Box Office Mojo dataset, scoped to the same filters as search. Facet enum: `gross_band`, `years_active`, `lifetime_year`, `franchise_names`, `brand_names`, `genre_names`, `hydrated`, `is_billion_dollar`, `in_lifetime_top_1000_ww`. gross_band enum: `under_50m`, `50_100m`, `100_250m`, `250_500m`, `500m_1b`, `over_1b`.\n- **Params:** `brand` (string, optional) — Brand name filter, max 128 characters; `facet` (string, **required**) — Facet enum: gross_band, years_active, lifetime_year, franchise_names, brand_names, genre_names, hydrated, is_billion_dollar, in_lifetime_top_1000_ww; `franchise` (string, optional) — Franchise name filter, max 128 characters; `genre` (string, optional) — Genre name filter, max 128 characters; `gross_band` (string, optional) — Gross band filter; `hydrated` (boolean, optional) — Hydrated filter; `in_lifetime_top_1000` (boolean, optional) — Only titles in the lifetime worldwide top 1000 chart; `is_billion_dollar` (boolean, optional) — Only titles with worldwide gross of at least $1B; `lifetime_year` (integer, optional) — Primary lifetime chart year; `max_domestic_share` (number, optional) — Maximum domestic share of worldwide gross, 0 through 1; `max_worldwide` (integer, optional) — Maximum lifetime worldwide gross; `min_domestic` (integer, optional) — Minimum lifetime domestic gross; `min_foreign_share` (number, optional) — Minimum foreign share of worldwide gross, 0 through 1; `min_worldwide` (integer, optional) — Minimum lifetime worldwide gross; `q` (string, optional) — Full-text query, max 256 characters; `title_id` (string, optional) — Exact title id (IMDb tt… id used by Box Office Mojo), max 32 characters; `year` (integer, optional) — Year in years_active\n\n### `datasets_boxofficemojo_item`\n\n- **HTTP:** `GET /datasets/boxofficemojo/items/{title_id}`\n- **What:** Get a Box Office Mojo title from the dataset. Returns one Box Office Mojo dataset record by title id (IMDb `tt…` id used on Box Office Mojo title pages), including lifetime grosses, year history, release groups and market grosses when hydrated.\n- **Params:** `title_id` (string, **required**) — Title id (IMDb tt… id), e.g. tt0499549\n\n### `datasets_boxofficemojo_search`\n\n- **HTTP:** `GET /datasets/boxofficemojo/search`\n- **What:** Search the Box Office Mojo dataset. Searches theatrical box-office records from public Box Office Mojo charts and title pages, stored in a search index. Filter by title id, year, franchise/brand/genre, gross band, lifetime top-1000 membership, hydration status, and worldwide/domestic gross ranges. Sort enum: `relevance`, `worldwide_desc`, `domestic_desc`, `peak_worldwide_desc`, `lifetime_rank_asc`, `year_desc`, `year_asc`. gross_band enum: `under_50m`, `50_100m`, `100_250m`, `250_500m`, `500m_1b`, `over_1b`.\n- **Params:** `brand` (string, optional) — Brand name filter, max 128 characters; `franchise` (string, optional) — Franchise name filter, max 128 characters; `genre` (string, optional) — Genre name filter, max 128 characters; `gross_band` (string, optional) — Gross band enum: under_50m, 50_100m, 100_250m, 250_500m, 500m_1b, over_1b; `hydrated` (boolean, optional) — Only titles with hydrated release groups and market grosses; `in_lifetime_top_1000` (boolean, optional) — Only titles in the lifetime worldwide top 1000 chart; `is_billion_dollar` (boolean, optional) — Only titles with worldwide gross of at least $1B; `lifetime_year` (integer, optional) — Primary lifetime chart year; `max_domestic_share` (number, optional) — Maximum domestic share of worldwide gross, 0 through 1; `max_worldwide` (integer, optional) — Maximum lifetime worldwide gross in whole USD dollars; `min_domestic` (integer, optional) — Minimum lifetime domestic gross in whole USD dollars; `min_foreign_share` (number, optional) — Minimum foreign share of worldwide gross, 0 through 1; `min_worldwide` (integer, optional) — Minimum lifetime worldwide gross in whole USD dollars; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `q` (string, optional) — Full-text query over title and taxonomy names, max 256 characters; `sort` (string, optional) — Sort enum: relevance, worldwide_desc, domestic_desc, peak_worldwide_desc, lifetime_rank_asc, year_desc, year_asc; `title_id` (string, optional) — Exact title id (IMDb tt… id used by Box Office Mojo), max 32 characters; `year` (integer, optional) — Year that must appear in years_active\n\n### `datasets_chrome_extensions_changes`\n\n- **HTTP:** `GET /datasets/chrome-extensions/changes`\n- **What:** Get recent Chrome Web Store item changes. Returns recent change observations. Change type enum: `users`, `rating`, `rating_count`, `version`, `developer`, `permissions`, `privacy`, `status`.\n- **Params:** `change_type` (string, optional) — Change type enum: users, rating, rating_count, version, developer, permissions, privacy, status; `limit` (integer, optional) — Maximum observations, default 100, max 500\n\n### `datasets_chrome_extensions_facets`\n\n- **HTTP:** `GET /datasets/chrome-extensions/facets`\n- **What:** Facet the Chrome Web Store dataset. Returns aggregation buckets. Facet enum: `item_type`, `category`, `developer`, `developer_email`, `manifest_version`, `permission`, `status`, `collects_data`, `has_broad_host_access`. Item type enum: `extension`, `theme`, `app`, `unknown`. Search sort, status and manifest-version enums match the search endpoint.\n- **Params:** `category` (string, optional) — Exact category; `collects_data` (boolean, optional) — Data-collection filter; `developer` (string, optional) — Exact developer; `developer_email` (string, optional) — Exact developer email; `facet` (string, **required**) — Facet enum: item_type, category, developer, developer_email, manifest_version, permission, status, collects_data, has_broad_host_access; `has_broad_host_access` (boolean, optional) — Broad-host-access filter; `item_type` (string, optional) — Item type enum: extension, theme, app, unknown; `manifest_version` (integer, optional) — Manifest version enum: 2, 3; `min_rating` (number, optional) — Minimum rating; `min_rating_count` (integer, optional) — Minimum rating count; `min_users` (integer, optional) — Minimum users; `permission` (string, optional) — Exact permission; `q` (string, optional) — Full-text query; `sort` (string, optional) — Sort enum: relevance, users_desc, rating_desc, reviews_desc, updated_desc, trending_desc; `status` (string, optional) — Status enum: active, removed\n\n### `datasets_chrome_extensions_history`\n\n- **HTTP:** `GET /datasets/chrome-extensions/history/{id}`\n- **What:** Get Chrome Web Store item history. Returns chronological change-only observations for a Chrome Web Store item.\n- **Params:** `from` (string, optional) — Inclusive start date, YYYY-MM-DD; `id` (string, **required**) — Chrome Web Store item id; `limit` (integer, optional) — Maximum points, default 365, max 1000; `to` (string, optional) — Inclusive end date, YYYY-MM-DD\n\n### `datasets_chrome_extensions_item`\n\n- **HTTP:** `GET /datasets/chrome-extensions/items/{id}`\n- **What:** Get a Chrome Web Store dataset item. Returns one stored extension, theme or legacy app snapshot by its 32-character Chrome Web Store id.\n- **Params:** `id` (string, **required**) — Chrome Web Store item id\n\n### `datasets_chrome_extensions_metrics`\n\n- **HTTP:** `GET /datasets/chrome-extensions/metrics`\n- **What:** Get Chrome Web Store dataset metrics. Returns chart-ready coverage, adoption, rating, permission, privacy and recent-change aggregates for the stored Chrome Web Store dataset. Days enum: `7`, `30`, `90`.\n- **Params:** `days` (integer, optional) — Recent-change window enum: 7, 30, 90; default 30; `limit` (integer, optional) — Top category and permission buckets, default 10, min 5, max 25\n\n### `datasets_chrome_extensions_search`\n\n- **HTTP:** `GET /datasets/chrome-extensions/search`\n- **What:** Search the Chrome Web Store dataset. Searches stored Chrome Web Store item snapshots. Item type enum: `extension`, `theme`, `app`, `unknown`. Sort enum: `relevance`, `users_desc`, `rating_desc`, `reviews_desc`, `updated_desc`, `trending_desc`. Status enum: `active`, `removed`. Manifest version enum: `2`, `3`.\n- **Params:** `category` (string, optional) — Exact Chrome Web Store category; `collects_data` (boolean, optional) — Filter by public data-collection disclosure; `developer` (string, optional) — Exact displayed developer name; `developer_email` (string, optional) — Exact disclosed developer email; `has_broad_host_access` (boolean, optional) — Filter by broad host access; `item_type` (string, optional) — Item type enum: extension, theme, app, unknown; `manifest_version` (integer, optional) — Manifest version enum: 2, 3; `min_rating` (number, optional) — Minimum rating, 0 through 5; `min_rating_count` (integer, optional) — Minimum rating count; `min_users` (integer, optional) — Minimum displayed user count; `page` (integer, optional) — Page number, default 1; `page_size` (integer, optional) — Page size, default 20, max 100; `permission` (string, optional) — Exact declared permission; `q` (string, optional) — Full-text query, max 256 characters; `sort` (string, optional) — Sort enum: relevance, users_desc, rating_desc, reviews_desc, updated_desc, trending_desc; `status` (string, optional) — Status enum: active, removed\n\n### `datasets_chrome_extensions_trending`\n\n- **HTTP:** `GET /datasets/chrome-extensions/trending`\n- **What:** Get trending Chrome Web Store items. Returns stored Chrome Web Store items ranked by the latest observed user and rating-count movement. Filters match the search endpoint; sort is fixed to `trending_desc`.\n- **Params:** `category` (string, optional) — Exact category; `collects_data` (boolean, optional) — Data-collection filter; `developer` (string, optional) — Exact developer; `developer_email` (string, optional) — Exact developer email; `has_broad_host_access` (boolean, optional) — Broad-host-access filter; `item_type` (string, optional) — Item type enum: extension, theme, app, unknown; `manifest_version` (integer, optional) — Manifest version enum: 2, 3; `min_rating` (number, optional) — Minimum rating; `min_rating_count` (integer, optional) — Minimum rating count; `min_users` (integer, optional) — Minimum users; `page` (integer, optional) — Page number; `page_size` (integer, optional) — Page size, max 100; `permission` (string, optional) — Exact permission; `q` (string, optional) — Full-text query; `status` (string, optional) — Status enum: active, removed\n\n### `datasets_creators_search`\n\n- **HTTP:** `GET /datasets/creators/search`\n- **What:** Search the TikTok creators dataset. Searches TikTok creators stored in a search index (one document per creator), with follower counts, verified status, niche, and engagement. Deleted and private accounts are excluded by default; set `include_inactive=true` to include them for historical lookups. Sort enum: `followers_desc`, `engagement_desc`, `engagement_qualified_desc`, `likes_desc`, `relevance`. Coverage note: `followers_desc`, `likes_desc`, and `relevance` are backed by profile fields present across the full dataset; the post-level engagement metrics (`engagement_rate`, `avg_views`, and the nested `post_stats` object) and the `engagement_desc`/`engagement_qualified_desc` sorts are currently populated for a growing subset of creators, prioritizing the highest-reach accounts. Creators without these metrics are still returned but sort last under `engagement_desc` and omit those fields; `engagement_qualified_desc` excludes them outright (they cannot clear its floors). `engagement_desc` ranks by raw `engagement_rate` with no eligibility floor — it surfaces a real stale-record + ratio-by-design trap: an account whose last real post was years ago can still carry an unrealistic rate computed from a handful of old posts. `engagement_qualified_desc` is the same metric restricted to creators with a recent post (`last_post_at` within 90 days), a minimum reach (`avg_views >= 10000`) and sample size (`post_stats.sampled_posts >= 10`), and a sanity ceiling (`engagement_rate <= 50%`) — use this, not the raw sort, for a \"best engagement\" leaderboard. Sound fields: `post_stats.top_sounds` holds only a creator's FIVE most-used sounds from the sampled posts, ranked by use count with ties broken by lowest `music_id`, so it is a top-5 view and not the creator's full sound list; `post_stats.distinct_sounds` gives the true number of different sounds the sample used. Use each sound's `original` boolean to tell TikTok-generated original audio from catalogue tracks - do NOT infer it from the title, because TikTok localizes the original-audio label (`sonido original`, `som original`, `оригинальный звук`, and at least fifteen more), so a title match silently reclassifies original audio as named tracks.\n- **Params:** `country` (string, optional) — Exact creator country/region filter, max 128 characters; `handle` (string, optional) — Exact handle lookup (case-insensitive), e.g. khaby.lame; returns the single creator with that exact @handle; `has_email` (boolean, optional) — Filter by contact-email presence; true keeps only creators with an email; `include_email` (boolean, optional) — Return the stored contact email instead of a blanked value. Off by default for everyone, and honoured only for entitled (non-Free) API keys; `include_inactive` (boolean, optional) — Include deleted/private accounts; defaults to false (only live accounts returned); `min_followers` (integer, optional) — Minimum follower count; `niche` (string, optional) — Exact content-niche filter, max 128 characters; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `q` (string, optional) — Full-text query over handle, nickname and bio, max 256 characters; `sort` (string, optional) — Sort enum: followers_desc, engagement_desc, engagement_qualified_desc, likes_desc, relevance. engagement_desc ranks by raw post-level engagement rate, currently populated for a subset of creators (highest-reach first); creators without it sort last. engagement_qualified_desc is the same metric restricted to creators with a recent post (<=90d), a minimum reach (avg_views>=10000) and sample size (>=10 posts), and a sanity ceiling (<=50%) -- use this, not the raw sort, for a 'best engagement' leaderboard; `verified` (boolean, optional) — Filter by verified badge; true keeps only verified creators\n\n### `datasets_doordash_stores_facets`\n\n- **HTTP:** `GET /datasets/doordash-stores/facets`\n- **What:** Facet stored DoorDash stores. Returns terms aggregation counts for the DoorDash store directory. Facet enum: `country`, `state`, `city`, `tags`, `display_status`, `price_range`, `dash_pass_eligible`. Accepts the same filter parameters as search to scope the aggregation. Use `facet=tags` to discover the marketplace tag values accepted by the `tag` filter.\n- **Params:** `city` (string, optional) — Exact city filter; `country` (string, optional) — ISO-3166-1 alpha-2 country filter; `dash_pass_only` (boolean, optional) — Keep only DashPass-eligible stores; `facet` (string, **required**) — Facet enum: country, state, city, tags, display_status, price_range, dash_pass_eligible; `q` (string, optional) — Full-text search over store name, address, city, and marketplace tags; `state` (string, optional) — State/region code filter; `tag` (string, optional) — Single marketplace tag filter\n\n### `datasets_doordash_stores_item`\n\n- **HTTP:** `GET /datasets/doordash-stores/items/{store_id}`\n- **What:** Get a stored DoorDash store. Returns one stored DoorDash store by its DoorDash store id (digits only, e.g. `297068`) from dataset id `doordash-stores`. A store discovered but not yet hydrated may have an empty `address`/`phone` and no `location`. `display_status`, `asap_available`, and `pickup_available` are a point-in-time pickup observation at crawl time, not a durable capability.\n- **Params:** `store_id` (string, **required**) — DoorDash store id (digits only), e.g. 297068\n\n### `datasets_doordash_stores_nearby`\n\n- **HTTP:** `GET /datasets/doordash-stores/nearby`\n- **What:** Find nearby stored DoorDash stores. Returns stored DoorDash stores within a radius of a point, nearest first, from dataset id `doordash-stores`. lat, lon, and radius_m are required. Unlike the live /doordash/search endpoint (which is a proximity search capped at roughly five stores), this queries the stored directory, so it can return every discovered store in the radius. Coverage is best-effort rather than provably exhaustive: a store DoorDash never surfaced, or one not yet hydrated with coordinates, will not appear.\n- **Params:** `country` (string, optional) — ISO-3166-1 alpha-2 country filter; `cursor` (string, optional) — Opaque continuation token returned in next_cursor; use with pagination=cursor and the same location and filters; `dash_pass_only` (boolean, optional) — Keep only DashPass-eligible stores; `lat` (number, **required**) — Center latitude, from -90 through 90; `lon` (number, **required**) — Center longitude, from -180 through 180; `page` (integer, optional) — Offset page number, defaults to 1; ignored by cursor pagination except page=1 to start; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; the 10,000-result cap applies only to offset pagination; `pagination` (string, optional) — Pagination mode: offset or cursor. Cursor mode supports full enumeration beyond the offset result window.; `radius_m` (integer, **required**) — Search radius in meters, 1 through 50000; `tag` (string, optional) — Single marketplace tag filter\n\n### `datasets_doordash_stores_search`\n\n- **HTTP:** `GET /datasets/doordash-stores/search`\n- **What:** Search the DoorDash store directory. Searches the DoorDash merchant-location directory (dataset id `doordash-stores`), built by grid-tiling the anonymous guest Explore feed and hydrating every discovered store. Identity and marketplace fields (name, marketplace tags, price range, ratings, DashPass eligibility) come from Explore; contact, address, and coordinates come from the hydration pass, so a store that has not been hydrated yet can appear without them. `display_status`, `display_asap_time`, `asap_available`, and `pickup_available` are a point-in-time observation of the PICKUP surface at crawl time, not a durable capability. Geographic coverage is best-effort: Explore caps at 40 stores per call with no pagination, so the census saturates a tile rather than draining it. Supports full-text `q`, `country`/`state`/`city`/`tag` filters, `dash_pass_only`, `min_price`/`max_price`/`min_rating`, `lat`/`lon`/`radius_m` radius filtering, and `sort` (relevance, rating, distance, distance_asc). `sort=relevance` ranks by text-match score when `q` is supplied; without `q` it sorts by store name. Use `pagination=cursor` and the returned `next_cursor` to enumerate beyond the 10,000-result offset window against a point-in-time snapshot.\n- **Params:** `city` (string, optional) — Exact city filter; `country` (string, optional) — ISO-3166-1 alpha-2 country filter, e.g. US, CA; `cursor` (string, optional) — Opaque continuation token returned in next_cursor; use with pagination=cursor and the same filters and sort; `dash_pass_only` (boolean, optional) — Keep only DashPass-eligible stores; `lat` (number, optional) — Latitude for radius filtering or distance sort, requires lon; `lon` (number, optional) — Longitude for radius filtering or distance sort, requires lat; `max_price` (integer, optional) — Maximum price range tier, inclusive; `min_price` (integer, optional) — Minimum price range tier, inclusive; `min_rating` (number, optional) — Minimum aggregate rating, 0 through 5; excludes unrated stores; `page` (integer, optional) — Offset page number, defaults to 1; ignored by cursor pagination except page=1 to start; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; the 10,000-result cap applies only to offset pagination; `pagination` (string, optional) — Pagination mode: offset or cursor. Cursor mode supports full enumeration beyond the offset result window.; `q` (string, optional) — Full-text search over store name, address, city, and marketplace tags, max 256 characters; `radius_m` (integer, optional) — Radius in meters, 1 through 50000; requires lat and lon; `sort` (string, optional) — Sort enum: relevance, rating, distance, distance_asc; `state` (string, optional) — State/region code filter, e.g. CA; `tag` (string, optional) — Single marketplace tag filter, e.g. Pizza\n\n### `datasets_facebook_pages_facets`\n\n- **HTTP:** `GET /datasets/facebook-pages/facets`\n- **What:** Facet the Facebook Pages dataset. Returns terms aggregation counts for the Facebook Pages dataset. Facet enum: `category`, `discovery_source`.\n- **Params:** `after` (string, optional) — Continuation from next_after; requires order=value_asc and unchanged filters; `category` (string, optional) — Exact Page category filter (case-insensitive), max 128 characters; `discovery_source` (string, optional) — Exact filter for how the Page was discovered (e.g. business_search, warc_domain_scan, wikidata), max 128 characters; `facet` (string, **required**) — Facet enum: category, discovery_source; `has_email` (boolean, optional) — Filter by a public contact email; `has_phone` (boolean, optional) — Filter by at least one public phone number; `has_website` (boolean, optional) — Filter by a linked website; `has_whatsapp` (boolean, optional) — Filter by a public WhatsApp contact; `hydrated_after` (string, optional) — Records last refreshed on or after this date (RFC3339 or YYYY-MM-DD); `hydrated_before` (string, optional) — Records last refreshed on or before this date (RFC3339 or YYYY-MM-DD); `identifier` (string, optional) — Exact Page username/identifier filter (case-insensitive), max 128 characters; `limit` (integer, optional) — Facet values per response, default 50, maximum 200; `max_likes` (integer, optional) — Maximum Page like count; `min_likes` (integer, optional) — Minimum Page like count; `order` (string, optional) — count_desc returns most common values; value_asc pages through all values; `page_id` (string, optional) — Exact Facebook Page id filter, max 128 characters; `q` (string, optional) — Full-text query over title and address, max 256 characters; `sort` (string, optional) — Sort enum: relevance, likes_desc, likes_asc, hydrated_at_desc, hydrated_at_asc\n\n### `datasets_facebook_pages_item`\n\n- **HTTP:** `GET /datasets/facebook-pages/items/{page_id}`\n- **What:** Get a Facebook Page from the dataset. Returns one Facebook Page record by page id from dataset id enum value `facebook-pages`.\n- **Params:** `page_id` (string, **required**) — Facebook Page id, max 128 characters\n\n### `datasets_facebook_pages_search`\n\n- **HTTP:** `GET /datasets/facebook-pages/search`\n- **What:** Search the Facebook Pages dataset. Searches public Facebook Page contact records stored in a search index — website, email, phone, WhatsApp, category and like count, discovered through business-search enumeration, domain-scoped scans, and Wikidata seeding, then hydrated from each Page's public About tab. Sort enum: `relevance`, `likes_desc`, `likes_asc`, `hydrated_at_desc`, `hydrated_at_asc`.\n- **Params:** `category` (string, optional) — Exact Page category filter (case-insensitive), max 128 characters; `discovery_source` (string, optional) — Exact filter for how the Page was discovered (e.g. business_search, warc_domain_scan, wikidata), max 128 characters; `has_email` (boolean, optional) — Filter by a public contact email; `has_phone` (boolean, optional) — Filter by at least one public phone number; `has_website` (boolean, optional) — Filter by a linked website; `has_whatsapp` (boolean, optional) — Filter by a public WhatsApp contact; `hydrated_after` (string, optional) — Records last refreshed on or after this date (RFC3339 or YYYY-MM-DD); `hydrated_before` (string, optional) — Records last refreshed on or before this date (RFC3339 or YYYY-MM-DD); `identifier` (string, optional) — Exact Page username/identifier filter (case-insensitive), max 128 characters; `max_likes` (integer, optional) — Maximum Page like count; `min_likes` (integer, optional) — Minimum Page like count; `page` (integer, optional) — Page number, defaults to 1; `page_id` (string, optional) — Exact Facebook Page id filter, max 128 characters; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `q` (string, optional) — Full-text query over title and address, max 256 characters; `sort` (string, optional) — Sort enum: relevance, likes_desc, likes_asc, hydrated_at_desc, hydrated_at_asc\n\n### `datasets_github_users_facets`\n\n- **HTTP:** `GET /datasets/github-users/facets`\n- **What:** Facet the GitHub users dataset. Returns terms aggregation counts for the GitHub users dataset. Facet enum: `influence_tier`, `type`, `country`, `country_code`, `state`, `city`, `domains`, `company`, `reachable`, `has_email`, `has_twitter`, `has_blog`, `active_90d`, `hireable`, `is_org`, `is_bot`, `is_suspected_automation`. influence_tier enum: `nano`, `micro`, `mid`, `macro`, `mega`. Suspected-automation records are excluded by default unless is_suspected_automation is set.\n- **Params:** `active_90d` (boolean, optional) — Filter by activity within the last 90 days; `city` (string, optional) — Exact geocoded city filter, max 128 characters; `company` (string, optional) — Exact normalized-company filter, max 128 characters; `country` (string, optional) — Exact geocoded country filter, max 128 characters; `country_code` (string, optional) — Exact ISO country-code filter, max 128 characters; `domain` (string, optional) — Interest-domain tag filter, max 128 characters; `facet` (string, **required**) — Facet enum: influence_tier, type, country, country_code, state, city, domains, company, reachable, has_email, has_twitter, has_blog, active_90d, hireable, is_org, is_bot, is_suspected_automation; `has_blog` (boolean, optional) — Filter by public blog/website presence; `has_email` (boolean, optional) — Filter by public email presence; `has_twitter` (boolean, optional) — Filter by public Twitter/X handle presence; `hireable` (boolean, optional) — Filter by the GitHub available-for-hire flag; `influence_tier` (string, optional) — Follower-tier enum: nano, micro, mid, macro, mega; `is_bot` (boolean, optional) — Bot filter; `is_org` (boolean, optional) — Organization filter; `is_suspected_automation` (boolean, optional) — Suspected automation filter; omitted these are hidden by default; `lat` (number, optional) — Latitude for radius filtering; `login` (string, optional) — Exact login filter, max 128 characters; `lon` (number, optional) — Longitude for radius filtering; `max_account_age_years` (number, optional) — Maximum account age in years; `max_followers` (integer, optional) — Maximum follower count; `min_account_age_years` (number, optional) — Minimum account age in years; `min_followers` (integer, optional) — Minimum follower count; `min_rank_score` (integer, optional) — Minimum composite rank score; `min_repos` (integer, optional) — Minimum public repository count; `q` (string, optional) — Full-text query over login, name, company, bio and location, max 256 characters; `radius_m` (integer, optional) — Radius in meters, 1 through 50000; requires lat and lon when supplied; `reachable` (boolean, optional) — Filter by any public contact channel; `sort` (string, optional) — Sort enum: relevance, rank_score_desc, followers_desc, account_age_desc, account_age_asc, distance_asc; `state` (string, optional) — Exact geocoded state filter, max 128 characters\n\n### `datasets_github_users_item`\n\n- **HTTP:** `GET /datasets/github-users/items/{login}`\n- **What:** Get a GitHub user from the dataset. Returns one enriched GitHub user record by login from dataset id enum value `github-users`.\n- **Params:** `login` (string, **required**) — GitHub login, max 128 characters\n\n### `datasets_github_users_nearby`\n\n- **HTTP:** `GET /datasets/github-users/nearby`\n- **What:** Search nearby GitHub users. Searches enriched GitHub users near a coordinate, sorted by distance, in dataset id enum value `github-users`. influence_tier enum: `nano`, `micro`, `mid`, `macro`, `mega`.\n- **Params:** `influence_tier` (string, optional) — Follower-tier enum: nano, micro, mid, macro, mega; `lat` (number, **required**) — Latitude; `lon` (number, **required**) — Longitude; `min_followers` (integer, optional) — Minimum follower count; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `radius_m` (integer, **required**) — Radius in meters, max 50000; `reachable` (boolean, optional) — Filter by any public contact channel\n\n### `datasets_github_users_search`\n\n- **HTTP:** `GET /datasets/github-users/search`\n- **What:** Search the GitHub users dataset. Searches enriched public GitHub user profiles stored in a search index. influence_tier enum: `nano`, `micro`, `mid`, `macro`, `mega`. Sort enum: `relevance`, `rank_score_desc`, `followers_desc`, `account_age_desc`, `account_age_asc`, `distance_asc`.\n- **Params:** `active_90d` (boolean, optional) — Filter by activity within the last 90 days; `city` (string, optional) — Exact geocoded city filter, max 128 characters; `company` (string, optional) — Exact normalized-company filter, max 128 characters; `country` (string, optional) — Exact geocoded country filter, max 128 characters; `country_code` (string, optional) — Exact ISO country-code filter, max 128 characters; `domain` (string, optional) — Interest-domain tag filter (e.g. ml-ai, web, devops), max 128 characters; `has_blog` (boolean, optional) — Filter by public blog/website presence; `has_email` (boolean, optional) — Filter by public email presence; `has_twitter` (boolean, optional) — Filter by public Twitter/X handle presence; `hireable` (boolean, optional) — Filter by the GitHub available-for-hire flag; `influence_tier` (string, optional) — Follower-tier enum: nano, micro, mid, macro, mega; `is_bot` (boolean, optional) — Bot filter (normally false; the crawl skips bots); `is_org` (boolean, optional) — Organization filter (normally false; the crawl indexes individuals); `is_suspected_automation` (boolean, optional) — Suspected automation (commit-farm/mass-repo bots); omitted these are hidden by default, pass true to isolate them; `lat` (number, optional) — Latitude for radius filtering or distance sort; `login` (string, optional) — Exact login filter, max 128 characters; `lon` (number, optional) — Longitude for radius filtering or distance sort; `max_account_age_years` (number, optional) — Maximum account age in years; `max_followers` (integer, optional) — Maximum follower count; `min_account_age_years` (number, optional) — Minimum account age in years; `min_followers` (integer, optional) — Minimum follower count; `min_rank_score` (integer, optional) — Minimum composite rank score; `min_repos` (integer, optional) — Minimum public repository count; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `q` (string, optional) — Full-text query over login, name, company, bio and location, max 256 characters; `radius_m` (integer, optional) — Radius in meters, 1 through 50000; requires lat and lon when supplied; `reachable` (boolean, optional) — Filter by any public contact channel; `sort` (string, optional) — Sort enum: relevance, rank_score_desc, followers_desc, account_age_desc, account_age_asc, distance_asc; `state` (string, optional) — Exact geocoded state filter, max 128 characters\n\n### `datasets_goodreads_authors_facets`\n\n- **HTTP:** `GET /datasets/goodreads-authors/facets`\n- **What:** Facet Goodreads authors dataset. Returns terms aggregation counts for the Goodreads authors dataset. Facet enum: `genres`, `run_id`.\n- **Params:** `facet` (string, **required**)\n\nFile v1.0.20:skill-card.md\n\n## Description:\n\nQueries Crawlora's hosted public datasets for searches, aggregate breakdowns, nearby records, and individual records, returning JSON without crawling each source live.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[crawlora-org](https://clawhub.ai/user/crawlora-org)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and researchers use this skill to search and analyze Crawlora's pre-indexed public datasets, compare populations through facets, and retrieve records by dataset ID.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Queries and the API key are sent to Crawlora's hosted API.\n\nMitigation: Use a dedicated Crawlora key and keep secrets out of query text.\n\nRisk: Public-contact and business records may require careful handling.\n\nMitigation: Use returned data in accordance with applicable policy and law.\n\n## Reference(s):\n\n- [Crawlora dataset endpoint reference](reference/endpoints.md)\n- [Crawlora](https://crawlora.net)\n\n## Skill Output:\n\n**Output Type(s):** [JSON, Text]\n\n**Output Format:** [JSON API responses and concise text summaries]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Paginated results; hosted datasets are refreshed periodically rather than in real time.]\n\n## Skill Version(s):\n\n1.0.20 (source: server-resolved release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.0.19: 5 files, 39190 bytes\n\nFiles: reference/endpoints.md (173673b), scripts/crawlora.sh (10051b), skill-card.md (1679b), SKILL.md (6177b), _meta.json (137b)\n\nFile v1.0.19:SKILL.md\n\n---\nname: crawlora-datasets\ndescription: Queries Crawlora's pre-built hosted datasets — Airbnb markets, App Store/Google Play apps, GitHub/Instagram/X users, job postings, US housing markets, Google Maps businesses, Goodreads, PitchBook, Steam, TrustMRR, Product Hunt, SEC companies, tech-stack, and more — via search/facets/item/nearby endpoints, returning clean JSON without live-crawling each platform. Use when the user wants bulk or aggregate analysis, to search a pre-indexed corpus, to facet/filter a large population, or to look up one record by its dataset id, instead of scraping pages one at a time.\n---\n\n# Crawlora hosted datasets\n\nQuery Crawlora's own **pre-crawled, pre-indexed datasets** — search, facet, and\nfetch-by-id over corpora Crawlora already built and refreshes on a schedule.\nThis is different from the other skills in this repo: those hit a live\nper-platform endpoint (one request, one page); this skill hits a **search\nindex** over millions of already-collected records, so it's the right tool\nfor population-level questions (\"how many\", \"top N by X\", \"everything\nmatching Y\") rather than one-off lookups.\n\n## When to use this skill\n\n- \"How many / what share of X match Y?\" — facet/aggregate questions.\n- \"Find all X with property Y\" (e.g. jobs paying > $150k, apps with 4.5+\n  rating, GitHub users near a city, houses in a metro).\n- \"Give me the full list of Z\" instead of one record — bulk/list research.\n- Any of: Airbnb markets, app-store apps/reviews/charts, GitHub/Instagram/X\n  users, job postings + which companies are hiring, US housing markets\n  (Redfin-sourced), Google Maps businesses, Goodreads authors/books, Apple\n  Podcasts shows, Chrome Web Store extensions, PitchBook companies/funds/\n  investors/advisors/LPs, PlayStation games, Product Hunt makers/products/\n  trends, Reddit trending, SEC companies + institutional positions, Steam\n  games/prices/playercounts/reviews/news/achievements/charts, TrustMRR\n  startups, journalists, Numbeo cost-of-living cities/countries, website\n  tech-stack.\n- Prefer the platform-specific skill instead when the job is \"look up this\n  one profile/listing right now\" (e.g. `youtube-research`, `movie-tv-research`)\n  — datasets are refreshed periodically, not real-time.\n\n## Setup (one-time)\n\n- Get a free Crawlora API key (2,000 credits/mo, no card) at [https://crawlora.net](https://crawlora.net?utm_source=github&utm_medium=referral&utm_campaign=crawlora-skills).\n- Set `CRAWLORA_API_KEY` in the environment before running the helper.\n- The helper reads `CRAWLORA_API_KEY` from the environment and sends requests to `https://api.crawlora.net/api/v1`. Missing/invalid key → `401`.\n\n## How it works\n\nEvery dataset follows the same shape under `/datasets/<dataset-id>/...`:\n\n1. **Discover** — `GET /datasets` lists every available dataset id and its\n   capabilities (search / facets / item / nearby).\n2. **Search** — `GET /datasets/<id>/search` full-text + filtered search;\n   paginate with `page`/`size` (see `reference/endpoints.md` per dataset).\n3. **Facet** — `GET /datasets/<id>/facets` returns aggregate breakdowns\n   across a dataset's facetable fields at once (e.g. the jobs dataset\n   returns top companies, department, location, seniority, remote share,\n   and more in one call) — use for \"how many / breakdown by X\" questions.\n4. **Item** — `GET /datasets/<id>/items/{id}` fetches one record by its\n   dataset key (varies per dataset: `login`, `username`, `slug`, `cik`,\n   `appid`, `domain`, `region_type/table_id`, …).\n5. **Nearby** — where supported (`airbnb-markets`, `github-users`,\n   `google-map-businesses`, `jobs`) — `GET /datasets/<id>/nearby` finds\n   records near a `lat`/`lon` within `radius_km`.\n\nFull endpoint list, per-dataset ids, and params: [`reference/endpoints.md`](reference/endpoints.md).\n\n## Calling the API\n\n```sh\n# List every dataset id and what it supports:\nscripts/crawlora.sh /datasets | jq '.'\n\n# Search the jobs dataset (all companies' live postings):\nscripts/crawlora.sh /datasets/jobs/search q=\"staff engineer\" location=\"remote\" | jq '.'\n\n# Facet: hiring-market breakdown (top companies, seniority, remote share, ...):\nscripts/crawlora.sh /datasets/jobs/facets | jq '.'\n\n# Item: one GitHub user by login:\nscripts/crawlora.sh /datasets/github-users/items/torvalds | jq '.'\n\n# Nearby: GitHub users within 50km of a coordinate (radius in meters):\nscripts/crawlora.sh /datasets/github-users/nearby lat=37.7749 lon=-122.4194 radius_m=50000 | jq '.'\n```\n\nUse `scripts/crawlora.sh` for all requests; it keeps the API key out of command-line arguments.\n\n\n## Endpoint reference\n\nSee [`reference/endpoints.md`](reference/endpoints.md) for every dataset id,\nits search/facets/item/nearby endpoints, and params.\n\n## Examples\n\n- **Hiring-market pulse:** `/datasets/jobs/facets` for the aggregate\n  breakdown (top companies, seniority, remote share), then\n  `/datasets/jobs/companies` to see which employers are actively posting.\n- **App-store landscape scan:** `/datasets/apps/search` filtered by category\n  and rating, then `/datasets/apps-reviews/search` for the sentiment behind\n  the top results.\n- **Startup revenue leaderboard:** `/datasets/trustmrr/search` sorted by\n  MRR, then `/datasets/trustmrr/history/{slug}` for one company's trend line.\n- **Housing-market snapshot:** `/datasets/housing-markets/search` for a\n  metro, then `/datasets/housing-markets/items/{region_type}/{table_id}` for\n  the full monthly series.\n\n## Notes & limits\n\n- **Credits / pay-on-success:** billed only on `2xx`; free tier 2,000 credits/mo.\n  Key at [https://crawlora.net](https://crawlora.net?utm_source=github&utm_medium=referral&utm_campaign=crawlora-skills).\n- **Public data only** — every dataset is built from public sources.\n- **Security:** key lives in `CRAWLORA_API_KEY` only — never hardcode, query-param, or commit it.\n- Datasets refresh on a schedule (daily/weekly depending on source) — not\n  real-time. For a live single-record lookup, prefer the matching\n  platform-specific skill in this repo (e.g. `job-market-research`,\n  `movie-tv-research`) instead.\n- Results are paginated (`page`/`size`) — walk pages for full coverage.\n\nFile v1.0.19:_meta.json\n\n{\n  \"ownerId\": \"kn70shhkf6qpfwgfrbgtep2wkd8c6b4t\",\n  \"slug\": \"crawlora-datasets\",\n  \"version\": \"1.0.19\",\n  \"publishedAt\": 1790426383669\n}\n\nFile v1.0.19:reference/endpoints.md\n\n# crawlora-datasets — endpoint reference\n\n> Generated from `scripts/tools.json` by `scripts/generate.mjs` — do not edit by hand.\n\nEndpoints this skill uses, grouped by platform. Call them via `scripts/crawlora.sh` (see SKILL.md).\n\nAll paths are relative to the API base `https://api.crawlora.net/api/v1` and require the header `x-api-key: $CRAWLORA_API_KEY`. Path params like `{id}` are substituted into the URL; `GET` params go in the query string; `POST` params go in a JSON body.\n\n**126 endpoints across 1 platform group(s).**\n\n## Datasets (126)\n\n### `datasets_airbnb_facets`\n\n- **HTTP:** `GET /datasets/airbnb-markets/facets`\n- **What:** Facet the Airbnb markets dataset. Returns suppressed distribution counts over the Airbnb markets dataset, honoring the same filters as search. Facet enum: `country`, `market`, `currency`, `superhost`, `guest_favorite`, `rating_band`, `review_band`, `admin1` (top subdivision), `locality` (settlement), `room_type` (`entire_place`/`private_room`/`hotel`/`shared_room`), `property_type` (Airbnb's canonical listing type from the detail page), `amenities` (each amenity with the count of listings offering it). The `admin1`, `locality`, `room_type`, `property_type` and `amenities` facets stay empty until their enrichment coverage is high enough to be reliable. group_by enum: `country`, `market`, `admin1`, `locality`, `room_type`, `property_type`.\n- **Params:** `active_since` (string, optional) — Freshness filter, an ISO-8601 date (YYYY-MM-DD); `country` (string, optional) — Exact ISO-3166-1 alpha-2 country filter, e.g. FR; `facet` (string, **required**) — Facet enum: country, market, currency, superhost, guest_favorite, rating_band, review_band, admin1, locality, room_type, property_type, amenities; `group_by` (string, optional) — Aggregate cell dimension enum: country, market, admin1, locality, room_type, property_type. Defaults to country; `guest_favorite` (boolean, optional) — Count only Guest Favorite listings (an observed lower bound; the badge under-counts); `market` (string, optional) — Exact metro-market filter, max 128 characters; `min_listings` (integer, optional) — Minimum listings per bucket; raises the small-cell suppression floor; `min_rating` (number, optional) — Minimum listing rating, from 0 through 5; `min_review_count` (integer, optional) — Minimum listing review count, 0 or greater; `superhost` (boolean, optional) — Count only Superhost listings\n\n### `datasets_airbnb_item`\n\n- **HTTP:** `GET /datasets/airbnb-markets/items/{country}`\n- **What:** Get an Airbnb market from the dataset. Returns one country's full aggregate Airbnb market profile from dataset id enum value `airbnb-markets` — headline supply, Superhost share, Guest Favorite share (`guest_favorite_pct`, an observed lower bound), `avg_person_capacity` (average guests a listing sleeps over the detail-page-enriched sample), ratings, its top metros, bounding box, per-currency nightly-price percentiles, and a USD-normalized `price_usd` percentile block (converted via an approximate dated FX snapshot) for cross-country comparison. Aggregate-only. Returns 404 for a country below the suppression floor.\n- **Params:** `country` (string, **required**) — ISO-3166-1 alpha-2 country code, e.g. FR\n\n### `datasets_airbnb_nearby`\n\n- **HTTP:** `GET /datasets/airbnb-markets/nearby`\n- **What:** Airbnb market density near a coordinate. Returns an aggregate geohash-grid density map of Airbnb listings within a radius of a coordinate, from dataset id enum value `airbnb-markets`. Each cell reports a centroid, listing count and Superhost share; thin cells are suppressed. Aggregate-only.\n- **Params:** `active_since` (string, optional) — Freshness filter, an ISO-8601 date (YYYY-MM-DD); `country` (string, optional) — Exact ISO-3166-1 alpha-2 country filter, e.g. US; `lat` (number, **required**) — Center latitude, from -90 through 90; `lon` (number, **required**) — Center longitude, from -180 through 180; `min_listings` (integer, optional) — Minimum listings per cell; raises the small-cell suppression floor; `min_rating` (number, optional) — Minimum listing rating, from 0 through 5; `precision` (integer, optional) — Geohash precision, from 1 through 12; defaults to a value derived from the radius; `radius_m` (integer, **required**) — Search radius in meters, from 1 through 50000; `superhost` (boolean, optional) — Count only Superhost listings\n\n### `datasets_airbnb_search`\n\n- **HTTP:** `GET /datasets/airbnb-markets/search`\n- **What:** Search the Airbnb markets dataset. Returns aggregate Airbnb short-term-rental market rollups from the dataset id enum value `airbnb-markets`. Aggregate-only: each row is a market cell, never an individual listing. Thin cells are suppressed. group_by enum: `country`, `market`, `admin1` (top subdivision), `locality` (settlement), `room_type` (`entire_place`/`private_room`/`hotel`/`shared_room`), `property_type` (Airbnb's canonical listing type from the detail page). `admin1`, `locality`, `room_type` and `property_type` are enrichment-derived and stay empty until their coverage is high enough to be reliable. Each cell also carries `median_price_usd`, the median nightly price converted to USD via an approximate dated FX snapshot, for cross-country comparison (combine with `group_by=room_type` for median price by room type); `guest_favorite_pct`, the share of listings carrying the Guest Favorite badge (an observed lower bound, like `superhost_pct`); and `avg_person_capacity`, the average guests a listing sleeps over the detail-page-enriched sample. Sort enum: `listings_desc`, `superhost_pct_desc`, `rating_desc`, `key_asc`.\n- **Params:** `active_since` (string, optional) — Freshness filter, an ISO-8601 date (YYYY-MM-DD); only listings last seen on or after it are counted; `country` (string, optional) — Exact ISO-3166-1 alpha-2 country filter, e.g. FR; `group_by` (string, optional) — Aggregate cell dimension enum: country, market, admin1, locality, room_type, property_type. Defaults to country; `guest_favorite` (boolean, optional) — Count only Guest Favorite listings (an observed lower bound; the badge under-counts); `market` (string, optional) — Exact metro-market filter, e.g. Paris, max 128 characters; `min_listings` (integer, optional) — Minimum listings per cell; raises the small-cell suppression floor (never lowered below the built-in minimum); `min_rating` (number, optional) — Minimum listing rating, from 0 through 5; `min_review_count` (integer, optional) — Minimum listing review count, 0 or greater; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `sort` (string, optional) — Sort enum: listings_desc, superhost_pct_desc, rating_desc, key_asc; `superhost` (boolean, optional) — Count only Superhost listings\n\n### `datasets_apple_podcasts_shows_facets`\n\n- **HTTP:** `GET /datasets/apple-podcasts-shows/facets`\n- **What:** Facet Apple Podcasts shows dataset. Returns terms aggregation counts for the Apple Podcasts shows dataset. Facet enum: `genre`, `genre_id`, `country`, `content_advisory_rating`, `run_id`.\n- **Params:** `country` (string, optional) — Exact storefront country filter, max 128 characters; `explicitness` (string, optional) — Exact explicitness filter, max 128 characters; `facet` (string, **required**) — Facet enum: genre, genre_id, country, content_advisory_rating, run_id; `genre` (string, optional) — Exact primary-genre filter, max 128 characters; `genre_id` (string, optional) — Exact Apple Podcasts genre id filter, max 128 characters; `min_track_count` (integer, optional) — Minimum episode count (track_count), 0 or greater; `q` (string, optional) — Full-text query over show title and artist name, max 256 characters; `run_id` (string, optional) — Exact crawl run-id filter, max 128 characters\n\n### `datasets_apple_podcasts_shows_item`\n\n- **HTTP:** `GET /datasets/apple-podcasts-shows/items/{id}`\n- **What:** Get an Apple Podcasts show from dataset. Returns one crawled Apple Podcasts show record by id from dataset id enum value `apple-podcasts-shows`.\n- **Params:** `id` (string, **required**) — Apple Podcasts numeric show id (e.g. 173001861)\n\n### `datasets_apple_podcasts_shows_search`\n\n- **HTTP:** `GET /datasets/apple-podcasts-shows/search`\n- **What:** Search Apple Podcasts shows dataset. Searches the crawled public Apple Podcasts show catalog stored in a search index. One row per show. Discovered from a country x genre x collection chart grid and a search-term sweep — not a full catalog of every Apple Podcasts show. Sort enum: `relevance`, `popularity`, `track_count_desc`, `release_desc`, `title_asc`.\n- **Params:** `country` (string, optional) — Exact storefront country filter (the crawl's discovery storefront, e.g. us, gb), max 128 characters; `explicitness` (string, optional) — Exact explicitness filter as reported by Apple (e.g. explicit, cleaned), max 128 characters; `genre` (string, optional) — Exact primary-genre filter (e.g. Comedy, True Crime), max 128 characters; `genre_id` (string, optional) — Exact Apple Podcasts genre id filter (e.g. 1303 for Comedy), max 128 characters; `min_track_count` (integer, optional) — Minimum episode count (track_count), 0 or greater; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `q` (string, optional) — Full-text query over show title and artist name, max 256 characters; `run_id` (string, optional) — Exact crawl run-id filter, max 128 characters; `sort` (string, optional) — Sort enum: relevance, popularity, track_count_desc, release_desc, title_asc\n\n### `datasets_apps_charts_search`\n\n- **HTTP:** `GET /datasets/apps-charts/search`\n- **What:** Search the app-charts dataset. Searches daily top-chart snapshots scraped from the iOS App Store and Google Play, stored in a search index (one document per chart × snapshot × rank). With no `date` the latest snapshot is returned (today's chart); pair `app_id` with `sort=date_desc` for an app's rank over time. Store enum: `ios`, `android`. Chart type enum: `top_free`, `top_paid`, `top_grossing`, `new`. Platform enum (Apple device platforms, ios charts only): `phone`, `pad`, `mac`. Sort enum: `rank`, `rank_desc`, `date_desc`.\n- **Params:** `app_id` (string, optional) — Exact app filter — iOS numeric track id or Android package; pair with sort=date_desc for rank history; `category` (string, optional) — Store category/genre filter, max 128 characters; empty for the overall charts; `chart_type` (string, optional) — Chart enum: top_free, top_paid, top_grossing, new; `collection` (string, optional) — Raw store collection id filter (e.g. topgrossingapplications, GROSSING), max 128 characters; `country` (string, optional) — Exact storefront country filter, max 128 characters; `date` (string, optional) — Snapshot date filter yyyy-MM-dd; defaults to the latest snapshot; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `platform` (string, optional) — Apple device-platform filter, iOS charts only; see platform enum above; `q` (string, optional) — Full-text query over chart-entry title and developer, max 256 characters; `sort` (string, optional) — Sort enum: rank, rank_desc, date_desc; `store` (string, optional) — Store enum: ios, android\n\n### `datasets_apps_reviews_search`\n\n- **HTTP:** `GET /datasets/apps-reviews/search`\n- **What:** Search the app-reviews dataset. Searches user reviews scraped from the iOS App Store and Google Play, stored in a search index (one document per review). Store enum: `ios`, `android`. Sort enum: `recent`, `score_desc`, `score_asc`, `helpful_desc`.\n- **Params:** `app_id` (string, optional) — Exact app filter — iOS numeric track id or Android package, max 128 characters; `country` (string, optional) — Exact storefront country filter, max 128 characters; `min_score` (integer, optional) — Minimum star rating, 1 through 5; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `q` (string, optional) — Full-text query over review text, title and author, max 256 characters; `sort` (string, optional) — Sort enum: recent, score_desc, score_asc, helpful_desc; `store` (string, optional) — Store enum: ios, android\n\n### `datasets_apps_search`\n\n- **HTTP:** `GET /datasets/apps/search`\n- **What:** Search the apps-intelligence dataset. Searches resolved iOS App Store and Google Play apps stored in a search index. Store enum: `ios`, `android`, `both`. Platform enum (Apple device platforms, ios records only): `phone`, `pad`, `mac`, `tv`, `watch`, `vision`. Sort enum: `relevance`, `rating_desc`, `reviews_desc`, `installs_desc`, `updated_at_desc`, `popularity_desc`.\n- **Params:** `category` (string, optional) — Exact app-store category filter, max 128 characters; `country` (string, optional) — Exact storefront country filter, max 128 characters; `developer` (string, optional) — Exact developer/publisher name filter, max 128 characters; `free` (boolean, optional) — Filter by price; true keeps only free apps, false only paid; `min_rating` (number, optional) — Minimum store rating, 0 through 5; `min_reviews` (integer, optional) — Minimum ratings/review count; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `platforms` (array, optional) — Repeatable Apple device-platform filter (OR); see platform enum above; `q` (string, optional) — Full-text query over title, developer and category, max 256 characters; `sort` (string, optional) — Sort enum: relevance, rating_desc, reviews_desc, installs_desc, updated_at_desc, popularity_desc; `store` (string, optional) — Store enum: ios, android, both\n\n### `datasets_bbb_businesses_facets`\n\n- **HTTP:** `GET /datasets/bbb-businesses/facets`\n- **What:** Facet the BBB businesses dataset. Returns distribution counts over the BBB businesses index (dataset id enum value `bbb-businesses`), honoring the same filters as search. Facet enum: `category`, `state`, `city`, `rating`, `accredited`, `entity_type`, `run_id`.\n- **Params:** `accredited` (boolean, optional) — Accreditation filter; `category` (string, optional) — Exact category filter; `city` (string, optional) — Exact city filter; `entity_type` (string, optional) — Exact entity-type filter; `facet` (string, **required**) — Facet enum: category, state, city, rating, accredited, entity_type, run_id; `q` (string, optional) — Full-text match on the business name/category, max 256 characters; `rating` (string, optional) — Exact letter-grade rating filter. Enum: A+, A, A-, B+, B, B-, C+, C, C-, D+, D, D-, F; `run_id` (string, optional) — Exact crawl run id filter; `state` (string, optional) — Exact 2-letter state/province filter\n\n### `datasets_bbb_businesses_item`\n\n- **HTTP:** `GET /datasets/bbb-businesses/items/{id}`\n- **What:** Get a business from the BBB businesses dataset. Returns one business by id from dataset id enum value `bbb-businesses`. Returns 404 when the business is not in the index.\n- **Params:** `id` (string, **required**) — Business id (the <bbbLocalId>-<businessId> slug from the profile URL), e.g. 0825-1000223803\n\n### `datasets_bbb_businesses_search`\n\n- **HTTP:** `GET /datasets/bbb-businesses/search`\n- **What:** Search the BBB businesses dataset. Searches the BBB (Better Business Bureau) businesses index (dataset id enum value `bbb-businesses`) — business profiles crawled from bbb.org's own search/category-browse pages: computed A+-F letter-grade rating, paid-accreditation status, category, contact info, business details, operating hours, and products/services. Complaints, full reviews, and the full \"reasons for rating\"/service-area detail are NOT embedded here; each record instead carries complaints_url/reviews_url/more_info_url pointing at the live bbb-business-complaints/bbb-business-reviews/bbb-business-more-info endpoints for on-demand lookup. rating enum: `A+`, `A`, `A-`, `B+`, `B`, `B-`, `C+`, `C`, `C-`, `D+`, `D`, `D-`, `F`. sort enum: `relevance`, `rating_desc`, `rating_asc`, `accredited_first`, `name_asc`, `years_in_business_desc`.\n- **Params:** `accredited` (boolean, optional) — Accreditation filter; true keeps only accredited businesses; `category` (string, optional) — Exact category filter, e.g. Plumber. Use the values returned by facets?facet=category; `city` (string, optional) — Exact city filter, parsed from the profile URL; `entity_type` (string, optional) — Exact entity-type filter, e.g. Limited Liability Company (LLC); `min_rating_rank` (integer, optional) — Numeric floor against the denormalized rating rank (A+=12 down to F=0), e.g. 10 for 'A- and above'; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `q` (string, optional) — Full-text match on the business name/category, max 256 characters; `rating` (string, optional) — Exact letter-grade rating filter. Enum: A+, A, A-, B+, B, B-, C+, C, C-, D+, D, D-, F; `run_id` (string, optional) — Exact crawl run id filter; `sort` (string, optional) — Sort enum: relevance, rating_desc, rating_asc, accredited_first, name_asc, years_in_business_desc; `state` (string, optional) — Exact 2-letter state/province filter, parsed from the profile URL, e.g. tx\n\n### `datasets_boxofficemojo_facets`\n\n- **HTTP:** `GET /datasets/boxofficemojo/facets`\n- **What:** Facet the Box Office Mojo dataset. Returns terms-aggregation counts for one facet of the Box Office Mojo dataset, scoped to the same filters as search. Facet enum: `gross_band`, `years_active`, `lifetime_year`, `franchise_names`, `brand_names`, `genre_names`, `hydrated`, `is_billion_dollar`, `in_lifetime_top_1000_ww`. gross_band enum: `under_50m`, `50_100m`, `100_250m`, `250_500m`, `500m_1b`, `over_1b`.\n- **Params:** `brand` (string, optional) — Brand name filter, max 128 characters; `facet` (string, **required**) — Facet enum: gross_band, years_active, lifetime_year, franchise_names, brand_names, genre_names, hydrated, is_billion_dollar, in_lifetime_top_1000_ww; `franchise` (string, optional) — Franchise name filter, max 128 characters; `genre` (string, optional) — Genre name filter, max 128 characters; `gross_band` (string, optional) — Gross band filter; `hydrated` (boolean, optional) — Hydrated filter; `in_lifetime_top_1000` (boolean, optional) — Only titles in the lifetime worldwide top 1000 chart; `is_billion_dollar` (boolean, optional) — Only titles with worldwide gross of at least $1B; `lifetime_year` (integer, optional) — Primary lifetime chart year; `max_domestic_share` (number, optional) — Maximum domestic share of worldwide gross, 0 through 1; `max_worldwide` (integer, optional) — Maximum lifetime worldwide gross; `min_domestic` (integer, optional) — Minimum lifetime domestic gross; `min_foreign_share` (number, optional) — Minimum foreign share of worldwide gross, 0 through 1; `min_worldwide` (integer, optional) — Minimum lifetime worldwide gross; `q` (string, optional) — Full-text query, max 256 characters; `title_id` (string, optional) — Exact title id (IMDb tt… id used by Box Office Mojo), max 32 characters; `year` (integer, optional) — Year in years_active\n\n### `datasets_boxofficemojo_item`\n\n- **HTTP:** `GET /datasets/boxofficemojo/items/{title_id}`\n- **What:** Get a Box Office Mojo title from the dataset. Returns one Box Office Mojo dataset record by title id (IMDb `tt…` id used on Box Office Mojo title pages), including lifetime grosses, year history, release groups and market grosses when hydrated.\n- **Params:** `title_id` (string, **required**) — Title id (IMDb tt… id), e.g. tt0499549\n\n### `datasets_boxofficemojo_search`\n\n- **HTTP:** `GET /datasets/boxofficemojo/search`\n- **What:** Search the Box Office Mojo dataset. Searches theatrical box-office records from public Box Office Mojo charts and title pages, stored in a search index. Filter by title id, year, franchise/brand/genre, gross band, lifetime top-1000 membership, hydration status, and worldwide/domestic gross ranges. Sort enum: `relevance`, `worldwide_desc`, `domestic_desc`, `peak_worldwide_desc`, `lifetime_rank_asc`, `year_desc`, `year_asc`. gross_band enum: `under_50m`, `50_100m`, `100_250m`, `250_500m`, `500m_1b`, `over_1b`.\n- **Params:** `brand` (string, optional) — Brand name filter, max 128 characters; `franchise` (string, optional) — Franchise name filter, max 128 characters; `genre` (string, optional) — Genre name filter, max 128 characters; `gross_band` (string, optional) — Gross band enum: under_50m, 50_100m, 100_250m, 250_500m, 500m_1b, over_1b; `hydrated` (boolean, optional) — Only titles with hydrated release groups and market grosses; `in_lifetime_top_1000` (boolean, optional) — Only titles in the lifetime worldwide top 1000 chart; `is_billion_dollar` (boolean, optional) — Only titles with worldwide gross of at least $1B; `lifetime_year` (integer, optional) — Primary lifetime chart year; `max_domestic_share` (number, optional) — Maximum domestic share of worldwide gross, 0 through 1; `max_worldwide` (integer, optional) — Maximum lifetime worldwide gross in whole USD dollars; `min_domestic` (integer, optional) — Minimum lifetime domestic gross in whole USD dollars; `min_foreign_share` (number, optional) — Minimum foreign share of worldwide gross, 0 through 1; `min_worldwide` (integer, optional) — Minimum lifetime worldwide gross in whole USD dollars; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `q` (string, optional) — Full-text query over title and taxonomy names, max 256 characters; `sort` (string, optional) — Sort enum: relevance, worldwide_desc, domestic_desc, peak_worldwide_desc, lifetime_rank_asc, year_desc, year_asc; `title_id` (string, optional) — Exact title id (IMDb tt… id used by Box Office Mojo), max 32 characters; `year` (integer, optional) — Year that must appear in years_active\n\n### `datasets_chrome_extensions_changes`\n\n- **HTTP:** `GET /datasets/chrome-extensions/changes`\n- **What:** Get recent Chrome Web Store item changes. Returns recent change observations. Change type enum: `users`, `rating`, `rating_count`, `version`, `developer`, `permissions`, `privacy`, `status`.\n- **Params:** `change_type` (string, optional) — Change type enum: users, rating, rating_count, version, developer, permissions, privacy, status; `limit` (integer, optional) — Maximum observations, default 100, max 500\n\n### `datasets_chrome_extensions_facets`\n\n- **HTTP:** `GET /datasets/chrome-extensions/facets`\n- **What:** Facet the Chrome Web Store dataset. Returns aggregation buckets. Facet enum: `item_type`, `category`, `developer`, `developer_email`, `manifest_version`, `permission`, `status`, `collects_data`, `has_broad_host_access`. Item type enum: `extension`, `theme`, `app`, `unknown`. Search sort, status and manifest-version enums match the search endpoint.\n- **Params:** `category` (string, optional) — Exact category; `collects_data` (boolean, optional) — Data-collection filter; `developer` (string, optional) — Exact developer; `developer_email` (string, optional) — Exact developer email; `facet` (string, **required**) — Facet enum: item_type, category, developer, developer_email, manifest_version, permission, status, collects_data, has_broad_host_access; `has_broad_host_access` (boolean, optional) — Broad-host-access filter; `item_type` (string, optional) — Item type enum: extension, theme, app, unknown; `manifest_version` (integer, optional) — Manifest version enum: 2, 3; `min_rating` (number, optional) — Minimum rating; `min_rating_count` (integer, optional) — Minimum rating count; `min_users` (integer, optional) — Minimum users; `permission` (string, optional) — Exact permission; `q` (string, optional) — Full-text query; `sort` (string, optional) — Sort enum: relevance, users_desc, rating_desc, reviews_desc, updated_desc, trending_desc; `status` (string, optional) — Status enum: active, removed\n\n### `datasets_chrome_extensions_history`\n\n- **HTTP:** `GET /datasets/chrome-extensions/history/{id}`\n- **What:** Get Chrome Web Store item history. Returns chronological change-only observations for a Chrome Web Store item.\n- **Params:** `from` (string, optional) — Inclusive start date, YYYY-MM-DD; `id` (string, **required**) — Chrome Web Store item id; `limit` (integer, optional) — Maximum points, default 365, max 1000; `to` (string, optional) — Inclusive end date, YYYY-MM-DD\n\n### `datasets_chrome_extensions_item`\n\n- **HTTP:** `GET /datasets/chrome-extensions/items/{id}`\n- **What:** Get a Chrome Web Store dataset item. Returns one stored extension, theme or legacy app snapshot by its 32-character Chrome Web Store id.\n- **Params:** `id` (string, **required**) — Chrome Web Store item id\n\n### `datasets_chrome_extensions_metrics`\n\n- **HTTP:** `GET /datasets/chrome-extensions/metrics`\n- **What:** Get Chrome Web Store dataset metrics. Returns chart-ready coverage, adoption, rating, permission, privacy and recent-change aggregates for the stored Chrome Web Store dataset. Days enum: `7`, `30`, `90`.\n- **Params:** `days` (integer, optional) — Recent-change window enum: 7, 30, 90; default 30; `limit` (integer, optional) — Top category and permission buckets, default 10, min 5, max 25\n\n### `datasets_chrome_extensions_search`\n\n- **HTTP:** `GET /datasets/chrome-extensions/search`\n- **What:** Search the Chrome Web Store dataset. Searches stored Chrome Web Store item snapshots. Item type enum: `extension`, `theme`, `app`, `unknown`. Sort enum: `relevance`, `users_desc`, `rating_desc`, `reviews_desc`, `updated_desc`, `trending_desc`. Status enum: `active`, `removed`. Manifest version enum: `2`, `3`.\n- **Params:** `category` (string, optional) — Exact Chrome Web Store category; `collects_data` (boolean, optional) — Filter by public data-collection disclosure; `developer` (string, optional) — Exact displayed developer name; `developer_email` (string, optional) — Exact disclosed developer email; `has_broad_host_access` (boolean, optional) — Filter by broad host access; `item_type` (string, optional) — Item type enum: extension, theme, app, unknown; `manifest_version` (integer, optional) — Manifest version enum: 2, 3; `min_rating` (number, optional) — Minimum rating, 0 through 5; `min_rating_count` (integer, optional) — Minimum rating count; `min_users` (integer, optional) — Minimum displayed user count; `page` (integer, optional) — Page number, default 1; `page_size` (integer, optional) — Page size, default 20, max 100; `permission` (string, optional) — Exact declared permission; `q` (string, optional) — Full-text query, max 256 characters; `sort` (string, optional) — Sort enum: relevance, users_desc, rating_desc, reviews_desc, updated_desc, trending_desc; `status` (string, optional) — Status enum: active, removed\n\n### `datasets_chrome_extensions_trending`\n\n- **HTTP:** `GET /datasets/chrome-extensions/trending`\n- **What:** Get trending Chrome Web Store items. Returns stored Chrome Web Store items ranked by the latest observed user and rating-count movement. Filters match the search endpoint; sort is fixed to `trending_desc`.\n- **Params:** `category` (string, optional) — Exact category; `collects_data` (boolean, optional) — Data-collection filter; `developer` (string, optional) — Exact developer; `developer_email` (string, optional) — Exact developer email; `has_broad_host_access` (boolean, optional) — Broad-host-access filter; `item_type` (string, optional) — Item type enum: extension, theme, app, unknown; `manifest_version` (integer, optional) — Manifest version enum: 2, 3; `min_rating` (number, optional) — Minimum rating; `min_rating_count` (integer, optional) — Minimum rating count; `min_users` (integer, optional) — Minimum users; `page` (integer, optional) — Page number; `page_size` (integer, optional) — Page size, max 100; `permission` (string, optional) — Exact permission; `q` (string, optional) — Full-text query; `status` (string, optional) — Status enum: active, removed\n\n### `datasets_creators_search`\n\n- **HTTP:** `GET /datasets/creators/search`\n- **What:** Search the TikTok creators dataset. Searches TikTok creators stored in a search index (one document per creator), with follower counts, verified status, niche, and engagement. Deleted and private accounts are excluded by default; set `include_inactive=true` to include them for historical lookups. Sort enum: `followers_desc`, `engagement_desc`, `engagement_qualified_desc`, `likes_desc`, `relevance`. Coverage note: `followers_desc`, `likes_desc`, and `relevance` are backed by profile fields present across the full dataset; the post-level engagement metrics (`engagement_rate`, `avg_views`, and the nested `post_stats` object) and the `engagement_desc`/`engagement_qualified_desc` sorts are currently populated for a growing subset of creators, prioritizing the highest-reach accounts. Creators without these metrics are still returned but sort last under `engagement_desc` and omit those fields; `engagement_qualified_desc` excludes them outright (they cannot clear its floors). `engagement_desc` ranks by raw `engagement_rate` with no eligibility floor — it surfaces a real stale-record + ratio-by-design trap: an account whose last real post was years ago can still carry an unrealistic rate computed from a handful of old posts. `engagement_qualified_desc` is the same metric restricted to creators with a recent post (`last_post_at` within 90 days), a minimum reach (`avg_views >= 10000`) and sample size (`post_stats.sampled_posts >= 10`), and a sanity ceiling (`engagement_rate <= 50%`) — use this, not the raw sort, for a \"best engagement\" leaderboard. Sound fields: `post_stats.top_sounds` holds only a creator's FIVE most-used sounds from the sampled posts, ranked by use count with ties broken by lowest `music_id`, so it is a top-5 view and not the creator's full sound list; `post_stats.distinct_sounds` gives the true number of different sounds the sample used. Use each sound's `original` boolean to tell TikTok-generated original audio from catalogue tracks - do NOT infer it from the title, because TikTok localizes the original-audio label (`sonido original`, `som original`, `оригинальный звук`, and at least fifteen more), so a title match silently reclassifies original audio as named tracks.\n- **Params:** `country` (string, optional) — Exact creator country/region filter, max 128 characters; `handle` (string, optional) — Exact handle lookup (case-insensitive), e.g. khaby.lame; returns the single creator with that exact @handle; `has_email` (boolean, optional) — Filter by contact-email presence; true keeps only creators with an email; `include_email` (boolean, optional) — Return the stored contact email instead of a blanked value. Off by default for everyone, and honoured only for entitled (non-Free) API keys; `include_inactive` (boolean, optional) — Include deleted/private accounts; defaults to false (only live accounts returned); `min_followers` (integer, optional) — Minimum follower count; `niche` (string, optional) — Exact content-niche filter, max 128 characters; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `q` (string, optional) — Full-text query over handle, nickname and bio, max 256 characters; `sort` (string, optional) — Sort enum: followers_desc, engagement_desc, engagement_qualified_desc, likes_desc, relevance. engagement_desc ranks by raw post-level engagement rate, currently populated for a subset of creators (highest-reach first); creators without it sort last. engagement_qualified_desc is the same metric restricted to creators with a recent post (<=90d), a minimum reach (avg_views>=10000) and sample size (>=10 posts), and a sanity ceiling (<=50%) -- use this, not the raw sort, for a 'best engagement' leaderboard; `verified` (boolean, optional) — Filter by verified badge; true keeps only verified creators\n\n### `datasets_facebook_pages_facets`\n\n- **HTTP:** `GET /datasets/facebook-pages/facets`\n- **What:** Facet the Facebook Pages dataset. Returns terms aggregation counts for the Facebook Pages dataset. Facet enum: `category`, `discovery_source`.\n- **Params:** `after` (string, optional) — Continuation from next_after; requires order=value_asc and unchanged filters; `category` (string, optional) — Exact Page category filter (case-insensitive), max 128 characters; `discovery_source` (string, optional) — Exact filter for how the Page was discovered (e.g. business_search, warc_domain_scan, wikidata), max 128 characters; `facet` (string, **required**) — Facet enum: category, discovery_source; `has_email` (boolean, optional) — Filter by a public contact email; `has_phone` (boolean, optional) — Filter by at least one public phone number; `has_website` (boolean, optional) — Filter by a linked website; `has_whatsapp` (boolean, optional) — Filter by a public WhatsApp contact; `hydrated_after` (string, optional) — Records last refreshed on or after this date (RFC3339 or YYYY-MM-DD); `hydrated_before` (string, optional) — Records last refreshed on or before this date (RFC3339 or YYYY-MM-DD); `identifier` (string, optional) — Exact Page username/identifier filter (case-insensitive), max 128 characters; `limit` (integer, optional) — Facet values per response, default 50, maximum 200; `max_likes` (integer, optional) — Maximum Page like count; `min_likes` (integer, optional) — Minimum Page like count; `order` (string, optional) — count_desc returns most common values; value_asc pages through all values; `page_id` (string, optional) — Exact Facebook Page id filter, max 128 characters; `q` (string, optional) — Full-text query over title and address, max 256 characters; `sort` (string, optional) — Sort enum: relevance, likes_desc, likes_asc, hydrated_at_desc, hydrated_at_asc\n\n### `datasets_facebook_pages_item`\n\n- **HTTP:** `GET /datasets/facebook-pages/items/{page_id}`\n- **What:** Get a Facebook Page from the dataset. Returns one Facebook Page record by page id from dataset id enum value `facebook-pages`.\n- **Params:** `page_id` (string, **required**) — Facebook Page id, max 128 characters\n\n### `datasets_facebook_pages_search`\n\n- **HTTP:** `GET /datasets/facebook-pages/search`\n- **What:** Search the Facebook Pages dataset. Searches public Facebook Page contact records stored in a search index — website, email, phone, WhatsApp, category and like count, discovered through business-search enumeration, domain-scoped scans, and Wikidata seeding, then hydrated from each Page's public About tab. Sort enum: `relevance`, `likes_desc`, `likes_asc`, `hydrated_at_desc`, `hydrated_at_asc`.\n- **Params:** `category` (string, optional) — Exact Page category filter (case-insensitive), max 128 characters; `discovery_source` (string, optional) — Exact filter for how the Page was discovered (e.g. business_search, warc_domain_scan, wikidata), max 128 characters; `has_email` (boolean, optional) — Filter by a public contact email; `has_phone` (boolean, optional) — Filter by at least one public phone number; `has_website` (boolean, optional) — Filter by a linked website; `has_whatsapp` (boolean, optional) — Filter by a public WhatsApp contact; `hydrated_after` (string, optional) — Records last refreshed on or after this date (RFC3339 or YYYY-MM-DD); `hydrated_before` (string, optional) — Records last refreshed on or before this date (RFC3339 or YYYY-MM-DD); `identifier` (string, optional) — Exact Page username/identifier filter (case-insensitive), max 128 characters; `max_likes` (integer, optional) — Maximum Page like count; `min_likes` (integer, optional) — Minimum Page like count; `page` (integer, optional) — Page number, defaults to 1; `page_id` (string, optional) — Exact Facebook Page id filter, max 128 characters; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `q` (string, optional) — Full-text query over title and address, max 256 characters; `sort` (string, optional) — Sort enum: relevance, likes_desc, likes_asc, hydrated_at_desc, hydrated_at_asc\n\n### `datasets_github_users_facets`\n\n- **HTTP:** `GET /datasets/github-users/facets`\n- **What:** Facet the GitHub users dataset. Returns terms aggregation counts for the GitHub users dataset. Facet enum: `influence_tier`, `type`, `country`, `country_code`, `state`, `city`, `domains`, `company`, `reachable`, `has_email`, `has_twitter`, `has_blog`, `active_90d`, `hireable`, `is_org`, `is_bot`, `is_suspected_automation`. influence_tier enum: `nano`, `micro`, `mid`, `macro`, `mega`. Suspected-automation records are excluded by default unless is_suspected_automation is set.\n- **Params:** `active_90d` (boolean, optional) — Filter by activity within the last 90 days; `city` (string, optional) — Exact geocoded city filter, max 128 characters; `company` (string, optional) — Exact normalized-company filter, max 128 characters; `country` (string, optional) — Exact geocoded country filter, max 128 characters; `country_code` (string, optional) — Exact ISO country-code filter, max 128 characters; `domain` (string, optional) — Interest-domain tag filter, max 128 characters; `facet` (string, **required**) — Facet enum: influence_tier, type, country, country_code, state, city, domains, company, reachable, has_email, has_twitter, has_blog, active_90d, hireable, is_org, is_bot, is_suspected_automation; `has_blog` (boolean, optional) — Filter by public blog/website presence; `has_email` (boolean, optional) — Filter by public email presence; `has_twitter` (boolean, optional) — Filter by public Twitter/X handle presence; `hireable` (boolean, optional) — Filter by the GitHub available-for-hire flag; `influence_tier` (string, optional) — Follower-tier enum: nano, micro, mid, macro, mega; `is_bot` (boolean, optional) — Bot filter; `is_org` (boolean, optional) — Organization filter; `is_suspected_automation` (boolean, optional) — Suspected automation filter; omitted these are hidden by default; `lat` (number, optional) — Latitude for radius filtering; `login` (string, optional) — Exact login filter, max 128 characters; `lon` (number, optional) — Longitude for radius filtering; `max_account_age_years` (number, optional) — Maximum account age in years; `max_followers` (integer, optional) — Maximum follower count; `min_account_age_years` (number, optional) — Minimum account age in years; `min_followers` (integer, optional) — Minimum follower count; `min_rank_score` (integer, optional) — Minimum composite rank score; `min_repos` (integer, optional) — Minimum public repository count; `q` (string, optional) — Full-text query over login, name, company, bio and location, max 256 characters; `radius_m` (integer, optional) — Radius in meters, 1 through 50000; requires lat and lon when supplied; `reachable` (boolean, optional) — Filter by any public contact channel; `sort` (string, optional) — Sort enum: relevance, rank_score_desc, followers_desc, account_age_desc, account_age_asc, distance_asc; `state` (string, optional) — Exact geocoded state filter, max 128 characters\n\n### `datasets_github_users_item`\n\n- **HTTP:** `GET /datasets/github-users/items/{login}`\n- **What:** Get a GitHub user from the dataset. Returns one enriched GitHub user record by login from dataset id enum value `github-users`.\n- **Params:** `login` (string, **required**) — GitHub login, max 128 characters\n\n### `datasets_github_users_nearby`\n\n- **HTTP:** `GET /datasets/github-users/nearby`\n- **What:** Search nearby GitHub users. Searches enriched GitHub users near a coordinate, sorted by distance, in dataset id enum value `github-users`. influence_tier enum: `nano`, `micro`, `mid`, `macro`, `mega`.\n- **Params:** `influence_tier` (string, optional) — Follower-tier enum: nano, micro, mid, macro, mega; `lat` (number, **required**) — Latitude; `lon` (number, **required**) — Longitude; `min_followers` (integer, optional) — Minimum follower count; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `radius_m` (integer, **required**) — Radius in meters, max 50000; `reachable` (boolean, optional) — Filter by any public contact channel\n\n### `datasets_github_users_search`\n\n- **HTTP:** `GET /datasets/github-users/search`\n- **What:** Search the GitHub users dataset. Searches enriched public GitHub user profiles stored in a search index. influence_tier enum: `nano`, `micro`, `mid`, `macro`, `mega`. Sort enum: `relevance`, `rank_score_desc`, `followers_desc`, `account_age_desc`, `account_age_asc`, `distance_asc`.\n- **Params:** `active_90d` (boolean, optional) — Filter by activity within the last 90 days; `city` (string, optional) — Exact geocoded city filter, max 128 characters; `company` (string, optional) — Exact normalized-company filter, max 128 characters; `country` (string, optional) — Exact geocoded country filter, max 128 characters; `country_code` (string, optional) — Exact ISO country-code filter, max 128 characters; `domain` (string, optional) — Interest-domain tag filter (e.g. ml-ai, web, devops), max 128 characters; `has_blog` (boolean, optional) — Filter by public blog/website presence; `has_email` (boolean, optional) — Filter by public email presence; `has_twitter` (boolean, optional) — Filter by public Twitter/X handle presence; `hireable` (boolean, optional) — Filter by the GitHub available-for-hire flag; `influence_tier` (string, optional) — Follower-tier enum: nano, micro, mid, macro, mega; `is_bot` (boolean, optional) — Bot filter (normally false; the crawl skips bots); `is_org` (boolean, optional) — Organization filter (normally false; the crawl indexes individuals); `is_suspected_automation` (boolean, optional) — Suspected automation (commit-farm/mass-repo bots); omitted these are hidden by default, pass true to isolate them; `lat` (number, optional) — Latitude for radius filtering or distance sort; `login` (string, optional) — Exact login filter, max 128 characters; `lon` (number, optional) — Longitude for radius filtering or distance sort; `max_account_age_years` (number, optional) — Maximum account age in years; `max_followers` (integer, optional) — Maximum follower count; `min_account_age_years` (number, optional) — Minimum account age in years; `min_followers` (integer, optional) — Minimum follower count; `min_rank_score` (integer, optional) — Minimum composite rank score; `min_repos` (integer, optional) — Minimum public repository count; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `q` (string, optional) — Full-text query over login, name, company, bio and location, max 256 characters; `radius_m` (integer, optional) — Radius in meters, 1 through 50000; requires lat and lon when supplied; `reachable` (boolean, optional) — Filter by any public contact channel; `sort` (string, optional) — Sort enum: relevance, rank_score_desc, followers_desc, account_age_desc, account_age_asc, distance_asc; `state` (string, optional) — Exact geocoded state filter, max 128 characters\n\n### `datasets_goodreads_authors_facets`\n\n- **HTTP:** `GET /datasets/goodreads-authors/facets`\n- **What:** Facet Goodreads authors dataset. Returns terms aggregation counts for the Goodreads authors dataset. Facet enum: `genres`, `run_id`.\n- **Params:** `facet` (string, **required**) — Facet enum: genres, run_id; `genre` (string, optional) — Exact genre filter, max 128 characters; `min_rating` (number, optional) — Minimum average rating, 0 through 5; `min_ratings_count` (integer, optional) — Minimum number of ratings; `name` (string, optional) — Exact author name filter, max 128 characters; `q` (string, optional) — Full-text query over name, about and genres, max 256 characters; `run_id` (string, optional) — Exact crawl run-id filter, max 128 characters\n\n### `datasets_goodreads_authors_item`\n\n- **HTTP:** `GET /datasets/goodreads-authors/items/{id}`\n- **What:** Get a Goodreads author from dataset. Returns one crawled Goodreads author profile record by id from dataset id enum value `goodreads-authors`.\n- **Params:** `id` (string, **required**) — Goodreads author id, e.g. 153394\n\n### `datasets_goodreads_authors_search`\n\n- **HTTP:** `GET /datasets/goodreads-authors/search`\n- **What:** Search Goodreads authors dataset. Searches the crawled public Goodreads author profile index. Authors are discovered as a byproduct of the books crawl (every credited book contributor, plus the genre/search/list seed sources) — not a full catalog. Sort enum: `relevance`, `rating_desc`, `reviews_desc`, `name_asc`.\n- **Params:** `genre` (string, optional) — Exact genre filter (e.g. Fantasy, Romance, Nonfiction), max 128 characters; `min_rating` (number, optional) — Minimum average rating, 0 through 5; `min_ratings_count` (integer, optional) — Minimum number of ratings; `name` (string, optional) — Exact author name filter, max 128 characters; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `q` (string, optional) — Full-text query over name, about and genres, max 256 characters; `run_id` (string, optional) — Exact crawl run-id filter, max 128 characters; `sort` (string, optional) — Sort enum: relevance, rating_desc, reviews_desc, name_asc\n\n### `datasets_goodreads_books_facets`\n\n- **HTTP:** `GET /datasets/goodreads-books/facets`\n- **What:** Facet Goodreads books dataset. Returns terms aggregation counts for the Goodreads books dataset. Facet enum: `genres`, `format`, `language`, `publisher`, `primary_author`, `primary_author_id`, `series_name`, `publication_year`, `run_id`.\n- **Params:** `author` (string, optional) — Exact author name filter, max 128 characters; `author_id` (string, optional) — Exact Goodreads author id filter, max 128 characters; `facet` (string, **required**) — Facet enum: genres, format, language, publisher, primary_author, primary_author_id, series_name, publication_year, run_id; `format` (string, optional) — Exact format filter, max 128 characters; `genre` (string, optional) — Exact genre filter, max 128 characters; `isbn` (string, optional) — Exact ISBN-10 filter, max 128 characters; `isbn13` (string, optional) — Exact ISBN-13 filter, max 128 characters; `language` (string, optional) — Exact language filter, max 128 characters; `max_pages` (integer, optional) — Maximum page count; `max_publication_year` (integer, optional) — Maximum publication year; `min_pages` (integer, optional) — Minimum page count; `min_publication_year` (integer, optional) — Minimum publication year; `min_rating` (number, optional) — Minimum average rating, 0 through 5; `min_ratings_count` (integer, optional) — Minimum number of ratings; `publisher` (string, optional) — Exact publisher filter, max 128 characters; `q` (string, optional) — Full-text query over title, author and description, max 256 characters; `run_id` (string, optional) — Exact crawl run-id filter, max 128 characters; `series` (string, optional) — Exact series name filter, max 128 characters\n\n### `datasets_goodreads_books_item`\n\n- **HTTP:** `GET /datasets/goodreads-books/items/{id}`\n- **What:** Get a Goodreads book from dataset. Returns one crawled Goodreads book record by id from dataset id enum value `goodreads-books`.\n- **Params:** `id` (string, **required**) — Goodreads book id, e.g. 2767052\n\n### `datasets_goodreads_books_search`\n\n- **HTTP:** `GET /datasets/goodreads-books/search`\n- **What:** Search Goodreads books dataset. Searches the crawled public Goodreads book catalog stored in a search index. Discovered from curated Listopia \"best of\" lists, a search-term sweep, and author bibliography expansion — not a full catalog. Sort enum: `relevance`, `rating_desc`, `reviews_desc`, `publication_desc`, `publication_asc`, `pages_desc`, `pages_asc`, `title_asc`.\n- **Params:** `author` (string, optional) — Exact author name filter (matches any credited contributor), max 128 characters; `author_id` (string, optional) — Exact Goodreads author id filter, max 128 characters; `format` (string, optional) — Exact format filter (e.g. Hardcover, Paperback, Kindle Edition), max 128 characters; `genre` (string, optional) — Exact genre filter (e.g. Fantasy, Romance, Nonfiction), max 128 characters; `isbn` (string, optional) — Exact ISBN-10 filter, max 128 characters; `isbn13` (string, optional) — Exact ISBN-13 filter, max 128 characters; `language` (string, optional) — Exact language filter (e.g. English, Spanish), max 128 characters; `max_pages` (integer, optional) — Maximum page count; `max_publication_year` (integer, optional) — Maximum publication year; `min_pages` (integer, optional) — Minimum page count; `min_publication_year` (integer, optional) — Minimum publication year; `min_rating` (number, optional) — Minimum average rating, 0 through 5; `min_ratings_count` (integer, optional) — Minimum number of ratings; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `publisher` (string, optional) — Exact publisher filter, max 128 characters; `q` (string, optional) — Full-text query over title, author and description, max 256 characters; `run_id` (string, optional) — Exact crawl run-id filter, max 128 characters; `series` (string, optional) — Exact series name filter, max 128 characters; `sort` (string, optional) — Sort en\n\nFile v1.0.19:skill-card.md\n\n## Description:\n\nQueries Crawlora's hosted datasets for filtered records, aggregate breakdowns, nearby results, and individual items without live-crawling source platforms.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[crawlora-org](https://clawhub.ai/user/crawlora-org)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and analysts use this skill to search public, pre-indexed datasets, compare populations with facets, and retrieve records for research or reporting.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Dataset requests send the API key and requested routes and query parameters to Crawlora.\n\nMitigation: Use a key stored in CRAWLORA_API_KEY and avoid putting secrets, private personal data, or confidential business details in queries.\n\n## Reference(s):\n\n- [Crawlora Datasets skill release](https://clawhub.ai/crawlora-org/skills/crawlora-datasets)\n- [Crawlora dataset endpoint reference](artifact/reference/endpoints.md)\n- [Crawlora](https://crawlora.net)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, JSON]\n\n**Output Format:** [JSON dataset responses or concise Markdown summaries]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Results are paginated and drawn from periodically refreshed datasets, not live source pages.]\n\n## Skill Version(s):\n\n1.0.19 (source: ClawHub release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.0.18: 5 files, 39480 bytes\n\nFiles: reference/endpoints.md (173558b), scripts/crawlora.sh (10051b), skill-card.md (2370b), SKILL.md (6177b), _meta.json (137b)\n\nFile v1.0.18:SKILL.md\n\n---\nname: crawlora-datasets\ndescription: Queries Crawlora's pre-built hosted datasets — Airbnb markets, App Store/Google Play apps, GitHub/Instagram/X users, job postings, US housing markets, Google Maps businesses, Goodreads, PitchBook, Steam, TrustMRR, Product Hunt, SEC companies, tech-stack, and more — via search/facets/item/nearby endpoints, returning clean JSON without live-crawling each platform. Use when the user wants bulk or aggregate analysis, to search a pre-indexed corpus, to facet/filter a large population, or to look up one record by its dataset id, instead of scraping pages one at a time.\n---\n\n# Crawlora hosted datasets\n\nQuery Crawlora's own **pre-crawled, pre-indexed datasets** — search, facet, and\nfetch-by-id over corpora Crawlora already built and refreshes on a schedule.\nThis is different from the other skills in this repo: those hit a live\nper-platform endpoint (one request, one page); this skill hits a **search\nindex** over millions of already-collected records, so it's the right tool\nfor population-level questions (\"how many\", \"top N by X\", \"everything\nmatching Y\") rather than one-off lookups.\n\n## When to use this skill\n\n- \"How many / what share of X match Y?\" — facet/aggregate questions.\n- \"Find all X with property Y\" (e.g. jobs paying > $150k, apps with 4.5+\n  rating, GitHub users near a city, houses in a metro).\n- \"Give me the full list of Z\" instead of one record — bulk/list research.\n- Any of: Airbnb markets, app-store apps/reviews/charts, GitHub/Instagram/X\n  users, job postings + which companies are hiring, US housing markets\n  (Redfin-sourced), Google Maps businesses, Goodreads authors/books, Apple\n  Podcasts shows, Chrome Web Store extensions, PitchBook companies/funds/\n  investors/advisors/LPs, PlayStation games, Product Hunt makers/products/\n  trends, Reddit trending, SEC companies + institutional positions, Steam\n  games/prices/playercounts/reviews/news/achievements/charts, TrustMRR\n  startups, journalists, Numbeo cost-of-living cities/countries, website\n  tech-stack.\n- Prefer the platform-specific skill instead when the job is \"look up this\n  one profile/listing right now\" (e.g. `youtube-research`, `movie-tv-research`)\n  — datasets are refreshed periodically, not real-time.\n\n## Setup (one-time)\n\n- Get a free Crawlora API key (2,000 credits/mo, no card) at [https://crawlora.net](https://crawlora.net?utm_source=github&utm_medium=referral&utm_campaign=crawlora-skills).\n- Set `CRAWLORA_API_KEY` in the environment before running the helper.\n- The helper reads `CRAWLORA_API_KEY` from the environment and sends requests to `https://api.crawlora.net/api/v1`. Missing/invalid key → `401`.\n\n## How it works\n\nEvery dataset follows the same shape under `/datasets/<dataset-id>/...`:\n\n1. **Discover** — `GET /datasets` lists every available dataset id and its\n   capabilities (search / facets / item / nearby).\n2. **Search** — `GET /datasets/<id>/search` full-text + filtered search;\n   paginate with `page`/`size` (see `reference/endpoints.md` per dataset).\n3. **Facet** — `GET /datasets/<id>/facets` returns aggregate breakdowns\n   across a dataset's facetable fields at once (e.g. the jobs dataset\n   returns top companies, department, location, seniority, remote share,\n   and more in one call) — use for \"how many / breakdown by X\" questions.\n4. **Item** — `GET /datasets/<id>/items/{id}` fetches one record by its\n   dataset key (varies per dataset: `login`, `username`, `slug`, `cik`,\n   `appid`, `domain`, `region_type/table_id`, …).\n5. **Nearby** — where supported (`airbnb-markets`, `github-users`,\n   `google-map-businesses`, `jobs`) — `GET /datasets/<id>/nearby` finds\n   records near a `lat`/`lon` within `radius_km`.\n\nFull endpoint list, per-dataset ids, and params: [`reference/endpoints.md`](reference/endpoints.md).\n\n## Calling the API\n\n```sh\n# List every dataset id and what it supports:\nscripts/crawlora.sh /datasets | jq '.'\n\n# Search the jobs dataset (all companies' live postings):\nscripts/crawlora.sh /datasets/jobs/search q=\"staff engineer\" location=\"remote\" | jq '.'\n\n# Facet: hiring-market breakdown (top companies, seniority, remote share, ...):\nscripts/crawlora.sh /datasets/jobs/facets | jq '.'\n\n# Item: one GitHub user by login:\nscripts/crawlora.sh /datasets/github-users/items/torvalds | jq '.'\n\n# Nearby: GitHub users within 50km of a coordinate (radius in meters):\nscripts/crawlora.sh /datasets/github-users/nearby lat=37.7749 lon=-122.4194 radius_m=50000 | jq '.'\n```\n\nUse `scripts/crawlora.sh` for all requests; it keeps the API key out of command-line arguments.\n\n\n## Endpoint reference\n\nSee [`reference/endpoints.md`](reference/endpoints.md) for every dataset id,\nits search/facets/item/nearby endpoints, and params.\n\n## Examples\n\n- **Hiring-market pulse:** `/datasets/jobs/facets` for the aggregate\n  breakdown (top companies, seniority, remote share), then\n  `/datasets/jobs/companies` to see which employers are actively posting.\n- **App-store landscape scan:** `/datasets/apps/search` filtered by category\n  and rating, then `/datasets/apps-reviews/search` for the sentiment behind\n  the top results.\n- **Startup revenue leaderboard:** `/datasets/trustmrr/search` sorted by\n  MRR, then `/datasets/trustmrr/history/{slug}` for one company's trend line.\n- **Housing-market snapshot:** `/datasets/housing-markets/search` for a\n  metro, then `/datasets/housing-markets/items/{region_type}/{table_id}` for\n  the full monthly series.\n\n## Notes & limits\n\n- **Credits / pay-on-success:** billed only on `2xx`; free tier 2,000 credits/mo.\n  Key at [https://crawlora.net](https://crawlora.net?utm_source=github&utm_medium=referral&utm_campaign=crawlora-skills).\n- **Public data only** — every dataset is built from public sources.\n- **Security:** key lives in `CRAWLORA_API_KEY` only — never hardcode, query-param, or commit it.\n- Datasets refresh on a schedule (daily/weekly depending on source) — not\n  real-time. For a live single-record lookup, prefer the matching\n  platform-specific skill in this repo (e.g. `job-market-research`,\n  `movie-tv-research`) instead.\n- Results are paginated (`page`/`size`) — walk pages for full coverage.\n\nFile v1.0.18:_meta.json\n\n{\n  \"ownerId\": \"kn70shhkf6qpfwgfrbgtep2wkd8c6b4t\",\n  \"slug\": \"crawlora-datasets\",\n  \"version\": \"1.0.18\",\n  \"publishedAt\": 1789954299617\n}\n\nFile v1.0.18:reference/endpoints.md\n\n# crawlora-datasets — endpoint reference\n\n> Generated from `scripts/tools.json` by `scripts/generate.mjs` — do not edit by hand.\n\nEndpoints this skill uses, grouped by platform. Call them via `scripts/crawlora.sh` (see SKILL.md).\n\nAll paths are relative to the API base `https://api.crawlora.net/api/v1` and require the header `x-api-key: $CRAWLORA_API_KEY`. Path params like `{id}` are substituted into the URL; `GET` params go in the query string; `POST` params go in a JSON body.\n\n**126 endpoints across 1 platform group(s).**\n\n## Datasets (126)\n\n### `datasets_airbnb_facets`\n\n- **HTTP:** `GET /datasets/airbnb-markets/facets`\n- **What:** Facet the Airbnb markets dataset. Returns suppressed distribution counts over the Airbnb markets dataset, honoring the same filters as search. Facet enum: `country`, `market`, `currency`, `superhost`, `guest_favorite`, `rating_band`, `review_band`, `admin1` (top subdivision), `locality` (settlement), `room_type` (`entire_place`/`private_room`/`hotel`/`shared_room`), `property_type` (Airbnb's canonical listing type from the detail page), `amenities` (each amenity with the count of listings offering it). The `admin1`, `locality`, `room_type`, `property_type` and `amenities` facets stay empty until their enrichment coverage is high enough to be reliable. group_by enum: `country`, `market`, `admin1`, `locality`, `room_type`, `property_type`.\n- **Params:** `active_since` (string, optional) — Freshness filter, an ISO-8601 date (YYYY-MM-DD); `country` (string, optional) — Exact ISO-3166-1 alpha-2 country filter, e.g. FR; `facet` (string, **required**) — Facet enum: country, market, currency, superhost, guest_favorite, rating_band, review_band, admin1, locality, room_type, property_type, amenities; `group_by` (string, optional) — Aggregate cell dimension enum: country, market, admin1, locality, room_type, property_type. Defaults to country; `guest_favorite` (boolean, optional) — Count only Guest Favorite listings (an observed lower bound; the badge under-counts); `market` (string, optional) — Exact metro-market filter, max 128 characters; `min_listings` (integer, optional) — Minimum listings per bucket; raises the small-cell suppression floor; `min_rating` (number, optional) — Minimum listing rating, from 0 through 5; `min_review_count` (integer, optional) — Minimum listing review count, 0 or greater; `superhost` (boolean, optional) — Count only Superhost listings\n\n### `datasets_airbnb_item`\n\n- **HTTP:** `GET /datasets/airbnb-markets/items/{country}`\n- **What:** Get an Airbnb market from the dataset. Returns one country's full aggregate Airbnb market profile from dataset id enum value `airbnb-markets` — headline supply, Superhost share, Guest Favorite share (`guest_favorite_pct`, an observed lower bound), `avg_person_capacity` (average guests a listing sleeps over the detail-page-enriched sample), ratings, its top metros, bounding box, per-currency nightly-price percentiles, and a USD-normalized `price_usd` percentile block (converted via an approximate dated FX snapshot) for cross-country comparison. Aggregate-only. Returns 404 for a country below the suppression floor.\n- **Params:** `country` (string, **required**) — ISO-3166-1 alpha-2 country code, e.g. FR\n\n### `datasets_airbnb_nearby`\n\n- **HTTP:** `GET /datasets/airbnb-markets/nearby`\n- **What:** Airbnb market density near a coordinate. Returns an aggregate geohash-grid density map of Airbnb listings within a radius of a coordinate, from dataset id enum value `airbnb-markets`. Each cell reports a centroid, listing count and Superhost share; thin cells are suppressed. Aggregate-only.\n- **Params:** `active_since` (string, optional) — Freshness filter, an ISO-8601 date (YYYY-MM-DD); `country` (string, optional) — Exact ISO-3166-1 alpha-2 country filter, e.g. US; `lat` (number, **required**) — Center latitude, from -90 through 90; `lon` (number, **required**) — Center longitude, from -180 through 180; `min_listings` (integer, optional) — Minimum listings per cell; raises the small-cell suppression floor; `min_rating` (number, optional) — Minimum listing rating, from 0 through 5; `precision` (integer, optional) — Geohash precision, from 1 through 12; defaults to a value derived from the radius; `radius_m` (integer, **required**) — Search radius in meters, from 1 through 50000; `superhost` (boolean, optional) — Count only Superhost listings\n\n### `datasets_airbnb_search`\n\n- **HTTP:** `GET /datasets/airbnb-markets/search`\n- **What:** Search the Airbnb markets dataset. Returns aggregate Airbnb short-term-rental market rollups from the dataset id enum value `airbnb-markets`. Aggregate-only: each row is a market cell, never an individual listing. Thin cells are suppressed. group_by enum: `country`, `market`, `admin1` (top subdivision), `locality` (settlement), `room_type` (`entire_place`/`private_room`/`hotel`/`shared_room`), `property_type` (Airbnb's canonical listing type from the detail page). `admin1`, `locality`, `room_type` and `property_type` are enrichment-derived and stay empty until their coverage is high enough to be reliable. Each cell also carries `median_price_usd`, the median nightly price converted to USD via an approximate dated FX snapshot, for cross-country comparison (combine with `group_by=room_type` for median price by room type); `guest_favorite_pct`, the share of listings carrying the Guest Favorite badge (an observed lower bound, like `superhost_pct`); and `avg_person_capacity`, the average guests a listing sleeps over the detail-page-enriched sample. Sort enum: `listings_desc`, `superhost_pct_desc`, `rating_desc`, `key_asc`.\n- **Params:** `active_since` (string, optional) — Freshness filter, an ISO-8601 date (YYYY-MM-DD); only listings last seen on or after it are counted; `country` (string, optional) — Exact ISO-3166-1 alpha-2 country filter, e.g. FR; `group_by` (string, optional) — Aggregate cell dimension enum: country, market, admin1, locality, room_type, property_type. Defaults to country; `guest_favorite` (boolean, optional) — Count only Guest Favorite listings (an observed lower bound; the badge under-counts); `market` (string, optional) — Exact metro-market filter, e.g. Paris, max 128 characters; `min_listings` (integer, optional) — Minimum listings per cell; raises the small-cell suppression floor (never lowered below the built-in minimum); `min_rating` (number, optional) — Minimum listing rating, from 0 through 5; `min_review_count` (integer, optional) — Minimum listing review count, 0 or greater; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `sort` (string, optional) — Sort enum: listings_desc, superhost_pct_desc, rating_desc, key_asc; `superhost` (boolean, optional) — Count only Superhost listings\n\n### `datasets_apple_podcasts_shows_facets`\n\n- **HTTP:** `GET /datasets/apple-podcasts-shows/facets`\n- **What:** Facet Apple Podcasts shows dataset. Returns terms aggregation counts for the Apple Podcasts shows dataset. Facet enum: `genre`, `genre_id`, `country`, `content_advisory_rating`, `run_id`.\n- **Params:** `country` (string, optional) — Exact storefront country filter, max 128 characters; `explicitness` (string, optional) — Exact explicitness filter, max 128 characters; `facet` (string, **required**) — Facet enum: genre, genre_id, country, content_advisory_rating, run_id; `genre` (string, optional) — Exact primary-genre filter, max 128 characters; `genre_id` (string, optional) — Exact Apple Podcasts genre id filter, max 128 characters; `min_track_count` (integer, optional) — Minimum episode count (track_count), 0 or greater; `q` (string, optional) — Full-text query over show title and artist name, max 256 characters; `run_id` (string, optional) — Exact crawl run-id filter, max 128 characters\n\n### `datasets_apple_podcasts_shows_item`\n\n- **HTTP:** `GET /datasets/apple-podcasts-shows/items/{id}`\n- **What:** Get an Apple Podcasts show from dataset. Returns one crawled Apple Podcasts show record by id from dataset id enum value `apple-podcasts-shows`.\n- **Params:** `id` (string, **required**) — Apple Podcasts numeric show id (e.g. 173001861)\n\n### `datasets_apple_podcasts_shows_search`\n\n- **HTTP:** `GET /datasets/apple-podcasts-shows/search`\n- **What:** Search Apple Podcasts shows dataset. Searches the crawled public Apple Podcasts show catalog stored in a search index. One row per show. Discovered from a country x genre x collection chart grid and a search-term sweep — not a full catalog of every Apple Podcasts show. Sort enum: `relevance`, `popularity`, `track_count_desc`, `release_desc`, `title_asc`.\n- **Params:** `country` (string, optional) — Exact storefront country filter (the crawl's discovery storefront, e.g. us, gb), max 128 characters; `explicitness` (string, optional) — Exact explicitness filter as reported by Apple (e.g. explicit, cleaned), max 128 characters; `genre` (string, optional) — Exact primary-genre filter (e.g. Comedy, True Crime), max 128 characters; `genre_id` (string, optional) — Exact Apple Podcasts genre id filter (e.g. 1303 for Comedy), max 128 characters; `min_track_count` (integer, optional) — Minimum episode count (track_count), 0 or greater; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `q` (string, optional) — Full-text query over show title and artist name, max 256 characters; `run_id` (string, optional) — Exact crawl run-id filter, max 128 characters; `sort` (string, optional) — Sort enum: relevance, popularity, track_count_desc, release_desc, title_asc\n\n### `datasets_apps_charts_search`\n\n- **HTTP:** `GET /datasets/apps-charts/search`\n- **What:** Search the app-charts dataset. Searches daily top-chart snapshots scraped from the iOS App Store and Google Play, stored in a search index (one document per chart × snapshot × rank). With no `date` the latest snapshot is returned (today's chart); pair `app_id` with `sort=date_desc` for an app's rank over time. Store enum: `ios`, `android`. Chart type enum: `top_free`, `top_paid`, `top_grossing`, `new`. Platform enum (Apple device platforms, ios charts only): `phone`, `pad`, `mac`. Sort enum: `rank`, `rank_desc`, `date_desc`.\n- **Params:** `app_id` (string, optional) — Exact app filter — iOS numeric track id or Android package; pair with sort=date_desc for rank history; `category` (string, optional) — Store category/genre filter, max 128 characters; empty for the overall charts; `chart_type` (string, optional) — Chart enum: top_free, top_paid, top_grossing, new; `collection` (string, optional) — Raw store collection id filter (e.g. topgrossingapplications, GROSSING), max 128 characters; `country` (string, optional) — Exact storefront country filter, max 128 characters; `date` (string, optional) — Snapshot date filter yyyy-MM-dd; defaults to the latest snapshot; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `platform` (string, optional) — Apple device-platform filter, iOS charts only; see platform enum above; `q` (string, optional) — Full-text query over chart-entry title and developer, max 256 characters; `sort` (string, optional) — Sort enum: rank, rank_desc, date_desc; `store` (string, optional) — Store enum: ios, android\n\n### `datasets_apps_reviews_search`\n\n- **HTTP:** `GET /datasets/apps-reviews/search`\n- **What:** Search the app-reviews dataset. Searches user reviews scraped from the iOS App Store and Google Play, stored in a search index (one document per review). Store enum: `ios`, `android`. Sort enum: `recent`, `score_desc`, `score_asc`, `helpful_desc`.\n- **Params:** `app_id` (string, optional) — Exact app filter — iOS numeric track id or Android package, max 128 characters; `country` (string, optional) — Exact storefront country filter, max 128 characters; `min_score` (integer, optional) — Minimum star rating, 1 through 5; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `q` (string, optional) — Full-text query over review text, title and author, max 256 characters; `sort` (string, optional) — Sort enum: recent, score_desc, score_asc, helpful_desc; `store` (string, optional) — Store enum: ios, android\n\n### `datasets_apps_search`\n\n- **HTTP:** `GET /datasets/apps/search`\n- **What:** Search the apps-intelligence dataset. Searches resolved iOS App Store and Google Play apps stored in a search index. Store enum: `ios`, `android`, `both`. Platform enum (Apple device platforms, ios records only): `phone`, `pad`, `mac`, `tv`, `watch`, `vision`. Sort enum: `relevance`, `rating_desc`, `reviews_desc`, `installs_desc`, `updated_at_desc`, `popularity_desc`.\n- **Params:** `category` (string, optional) — Exact app-store category filter, max 128 characters; `country` (string, optional) — Exact storefront country filter, max 128 characters; `developer` (string, optional) — Exact developer/publisher name filter, max 128 characters; `free` (boolean, optional) — Filter by price; true keeps only free apps, false only paid; `min_rating` (number, optional) — Minimum store rating, 0 through 5; `min_reviews` (integer, optional) — Minimum ratings/review count; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `platforms` (array, optional) — Repeatable Apple device-platform filter (OR); see platform enum above; `q` (string, optional) — Full-text query over title, developer and category, max 256 characters; `sort` (string, optional) — Sort enum: relevance, rating_desc, reviews_desc, installs_desc, updated_at_desc, popularity_desc; `store` (string, optional) — Store enum: ios, android, both\n\n### `datasets_bbb_businesses_facets`\n\n- **HTTP:** `GET /datasets/bbb-businesses/facets`\n- **What:** Facet the BBB businesses dataset. Returns distribution counts over the BBB businesses index (dataset id enum value `bbb-businesses`), honoring the same filters as search. Facet enum: `category`, `state`, `city`, `rating`, `accredited`, `entity_type`, `run_id`.\n- **Params:** `accredited` (boolean, optional) — Accreditation filter; `category` (string, optional) — Exact category filter; `city` (string, optional) — Exact city filter; `entity_type` (string, optional) — Exact entity-type filter; `facet` (string, **required**) — Facet enum: category, state, city, rating, accredited, entity_type, run_id; `q` (string, optional) — Full-text match on the business name/category, max 256 characters; `rating` (string, optional) — Exact letter-grade rating filter. Enum: A+, A, A-, B+, B, B-, C+, C, C-, D+, D, D-, F; `run_id` (string, optional) — Exact crawl run id filter; `state` (string, optional) — Exact 2-letter state/province filter\n\n### `datasets_bbb_businesses_item`\n\n- **HTTP:** `GET /datasets/bbb-businesses/items/{id}`\n- **What:** Get a business from the BBB businesses dataset. Returns one business by id from dataset id enum value `bbb-businesses`. Returns 404 when the business is not in the index.\n- **Params:** `id` (string, **required**) — Business id (the <bbbLocalId>-<businessId> slug from the profile URL), e.g. 0825-1000223803\n\n### `datasets_bbb_businesses_search`\n\n- **HTTP:** `GET /datasets/bbb-businesses/search`\n- **What:** Search the BBB businesses dataset. Searches the BBB (Better Business Bureau) businesses index (dataset id enum value `bbb-businesses`) — business profiles crawled from bbb.org's own search/category-browse pages: computed A+-F letter-grade rating, paid-accreditation status, category, contact info, business details, operating hours, and products/services. Complaints, full reviews, and the full \"reasons for rating\"/service-area detail are NOT embedded here; each record instead carries complaints_url/reviews_url/more_info_url pointing at the live bbb-business-complaints/bbb-business-reviews/bbb-business-more-info endpoints for on-demand lookup. rating enum: `A+`, `A`, `A-`, `B+`, `B`, `B-`, `C+`, `C`, `C-`, `D+`, `D`, `D-`, `F`. sort enum: `relevance`, `rating_desc`, `rating_asc`, `accredited_first`, `name_asc`, `years_in_business_desc`.\n- **Params:** `accredited` (boolean, optional) — Accreditation filter; true keeps only accredited businesses; `category` (string, optional) — Exact category filter, e.g. Plumber. Use the values returned by facets?facet=category; `city` (string, optional) — Exact city filter, parsed from the profile URL; `entity_type` (string, optional) — Exact entity-type filter, e.g. Limited Liability Company (LLC); `min_rating_rank` (integer, optional) — Numeric floor against the denormalized rating rank (A+=12 down to F=0), e.g. 10 for 'A- and above'; `page` (integer, optional) — Page number, defaults to 1; `page_size` (integer, optional) — Page size, defaults to 20 and maxes at 100; page * page_size must be <= 10000; `q` (string, optional) — Full-text match on the business name/category, max 256 characters; `rating` (string, optional) — Exact letter-grade rating filter. Enum: A+, A, A-, B+, B, B-, C+, C, C-, D+, D, D-, F; `run_id` (string, optional) — Exact crawl run id filter; `sort` (string, optional) — Sort enum: relevance, rating_desc, rating_asc, accredited_first, name_asc, years_in_business_desc; `state` (string, optional) — Exact 2-letter state/province filter, parsed from the profile URL, e.g. tx\n\n### `datasets_boxofficemojo_facets`\n\n- **HTTP:** `GET /datasets/boxofficemojo/facets`\n- **What:** Facet the Box Office Mojo dataset. Returns terms-aggregation counts for one facet of the Box Office Mojo dataset, scoped to the same fi\n\nArchive v1.0.17: 5 files, 38570 bytes\n\nFiles: reference/endpoints.md (170509b), scripts/crawlora.sh (10051b), skill-card.md (2272b), SKILL.md (6177b), _meta.json (137b)\n\nArchive v1.0.16: 5 files, 39016 bytes\n\nFiles: reference/endpoints.md (170509b), scripts/crawlora.sh (10969b), skill-card.md (2288b), SKILL.md (6177b), _meta.json (137b)\n\nArchive v1.0.15: 5 files, 38750 bytes\n\nFiles: reference/endpoints.md (170509b), scripts/crawlora.sh (8646b), skill-card.md (2332b), SKILL.md (6177b), _meta.json (137b)\n\nArchive v1.0.14: 5 files, 38692 bytes\n\nFiles: reference/endpoints.md (170509b), scripts/crawlora.sh (8389b), skill-card.md (2357b), SKILL.md (6177b), _meta.json (137b)\n\nArchive v1.0.13: 5 files, 38658 bytes\n\nFiles: reference/endpoints.md (170509b), scripts/crawlora.sh (8064b), skill-card.md (2688b), SKILL.md (6219b), _meta.json (137b)\n\nArchive v1.0.12: 5 files, 38342 bytes\n\nFiles: reference/endpoints.md (170509b), scripts/crawlora.sh (7764b), skill-card.md (2301b), SKILL.md (6219b), _meta.json (137b)\n\nArchive v1.0.11: 5 files, 38521 bytes\n\nFiles: reference/endpoints.md (170509b), scripts/crawlora.sh (7716b), skill-card.md (2692b), SKILL.md (6219b), _meta.json (137b)","readmeExcerpt":"Skill: crawlora-datasets Owner: crawlora-org Summary: Queries Crawlora's pre-built hosted datasets — Airbnb markets, App Store/Google Play apps, GitHub/Instagram/X users, job postings, US housing markets, Google Maps businesses, Goodreads, PitchBook, Steam, TrustMRR, Product Hunt, SEC companies, tech-stack, and more — via search/facets/item/nearby endpoints, returning clean JSON without live-crawling each platform. U","codeSnippets":[],"executableExamples":[{"language":"sh","snippet":"# List every dataset id and what it supports:\nscripts/crawlora.sh /datasets | jq '.'\n\n# Search the jobs dataset (all companies' live postings):\nscripts/crawlora.sh /datasets/jobs/search q=\"staff engineer\" location=\"remote\" | jq '.'\n\n# Facet: hiring-market breakdown (top companies, seniority, remote share, ...):\nscripts/crawlora.sh /datasets/jobs/facets | jq '.'\n\n# Item: one GitHub user by login:\nscripts/crawlora.sh /datasets/github-users/items/torvalds | jq '.'\n\n# Nearby: GitHub users within 50km of a coordinate (radius in meters):\nscripts/crawlora.sh /datasets/github-users/nearby lat=37.7749 lon=-122.4194 radius_m=50000 | jq '.'"},{"language":"sh","snippet":"# List every dataset id and what it supports:\nscripts/crawlora.sh /datasets | jq '.'\n\n# Search the jobs dataset (all companies' live postings):\nscripts/crawlora.sh /datasets/jobs/search q=\"staff engineer\" location=\"remote\" | jq '.'\n\n# Facet: hiring-market breakdown (top companies, seniority, remote share, ...):\nscripts/crawlora.sh /datasets/jobs/facets | jq '.'\n\n# Item: one GitHub user by login:\nscripts/crawlora.sh /datasets/github-users/items/torvalds | jq '.'\n\n# Nearby: GitHub users within 50km of a coordinate (radius in meters):\nscripts/crawlora.sh /datasets/github-users/nearby lat=37.7749 lon=-122.4194 radius_m=50000 | jq '.'"},{"language":"sh","snippet":"# List every dataset id and what it supports:\nscripts/crawlora.sh /datasets | jq '.'\n\n# Search the jobs dataset (all companies' live postings):\nscripts/crawlora.sh /datasets/jobs/search q=\"staff engineer\" location=\"remote\" | jq '.'\n\n# Facet: hiring-market breakdown (top companies, seniority, remote share, ...):\nscripts/crawlora.sh /datasets/jobs/facets | jq '.'\n\n# Item: one GitHub user by login:\nscripts/crawlora.sh /datasets/github-users/items/torvalds | jq '.'\n\n# Nearby: GitHub users within 50km of a coordinate (radius in meters):\nscripts/crawlora.sh /datasets/github-users/nearby lat=37.7749 lon=-122.4194 radius_m=50000 | jq '.'"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: crawlora-datasets\ndescription: Queries Crawlora's pre-built hosted datasets — Airbnb markets, App Store/Google Play apps, GitHub/Instagram/X users, job postings, US housing markets, Google Maps businesses, Goodreads, PitchBook, Steam, TrustMRR, Product Hunt, SEC companies, tech-stack, and more — via search/facets/item/nearby endpoints, returning clean JSON without live-crawling each platform. Use when the user wants bulk or aggregate analysis, to search a pre-indexed corpus, to facet/filter a large population, or to look up one record by its dataset id, instead of scraping pages one at a time.\n---\n\n# Crawlora hosted datasets\n\nQuery Crawlora's own **pre-crawled, pre-indexed datasets** — search, facet, and\nfetch-by-id over corpora Crawlora already built and refreshes on a schedule.\nThis is different from the other skills in this repo: those hit a live\nper-platform endpoint (one request, one page); this skill hits a **search\nindex** over millions of already-collected records, so it's the right tool\nfor population-level questions (\"how many\", \"top N by X\", \"everything\nmatching Y\") rather than one-off lookups.\n\n## When to use this skill\n\n- \"How many / what share of X match Y?\" — facet/aggregate questions.\n- \"Find all X with property Y\" (e.g. jobs paying > $150k, apps with 4.5+\n  rating, GitHub users near a city, houses in a metro).\n- \"Give me the full list of Z\" instead of one record — bulk/list research.\n- Any of: Airbnb markets, app-store apps/reviews/charts, GitHub/Instagram/X\n  users, job postings + which companies are hiring, US housing markets\n  (Redfin-sourced), Google Maps businesses, Goodreads authors/books, Apple\n  Podcasts shows, Chrome Web Store extensions, PitchBook companies/funds/\n  investors/advisors/LPs, PlayStation games, Product Hunt makers/products/\n  trends, Reddit trending, SEC companies + institutional positions, Steam\n  games/prices/playercounts/reviews/news/achievements/charts, TrustMRR\n  startups, journalists, Numbeo cost-of-living cities/countries, website\n  tech-stack.\n- Prefer the platform-specific skill instead when the job is \"look up this\n  one profile/listing right now\" (e.g. `youtube-research`, `movie-tv-research`)\n  — datasets are refreshed periodically, not real-time.\n\n## Setup (one-time)\n\n- Get a free Crawlora API key (2,000 credits/mo, no card) at [https://crawlora.net](https://crawlora.net?utm_source=github&utm_medium=referral&utm_campaign=crawlora-skills).\n- Set `CRAWLORA_API_KEY` in the environment before running the helper.\n- The helper reads `CRAWLORA_API_KEY` from the environment and sends requests to `https://api.crawlora.net/api/v1`. Missing/invalid key → `401`.\n\n## How it works\n\nEvery dataset follows the same shape under `/datasets/<dataset-id>/...`:\n\n1. **Discover** — `GET /datasets` lists every available dataset id and its\n   capabilities (search / facets / item / nearby).\n2. **Search** — `GET /datasets/<id>/search` full-text + filtered search;\n   paginate with `page`/`size` (see `reference/en"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn70shhkf6qpfwgfrbgtep2wkd8c6b4t\",\n  \"slug\": \"crawlora-datasets\",\n  \"version\": \"1.0.20\",\n  \"publishedAt\": 1791162745933\n}"},{"path":"reference/endpoints.md","content":"# crawlora-datasets — endpoint reference\n\n> Generated from `scripts/tools.json` by `scripts/generate.mjs` — do not edit by hand.\n\nEndpoints this skill uses, grouped by platform. Call them via `scripts/crawlora.sh` (see SKILL.md).\n\nAll paths are relative to the API base `https://api.crawlora.net/api/v1` and require the header `x-api-key: $CRAWLORA_API_KEY`. Path params like `{id}` are substituted into the URL; `GET` params go in the query string; `POST` params go in a JSON body.\n\n**130 endpoints across 1 platform group(s).**\n\n## Datasets (130)\n\n### `datasets_airbnb_facets`\n\n- **HTTP:** `GET /datasets/airbnb-markets/facets`\n- **What:** Facet the Airbnb markets dataset. Returns suppressed distribution counts over the Airbnb markets dataset, honoring the same filters as search. Facet enum: `country`, `market`, `currency`, `superhost`, `guest_favorite`, `rating_band`, `review_band`, `admin1` (top subdivision), `locality` (settlement), `room_type` (`entire_place`/`private_room`/`hotel`/`shared_room`), `property_type` (Airbnb's canonical listing type from the detail page), `amenities` (each amenity with the count of listings offering it). The `admin1`, `locality`, `room_type`, `property_type` and `amenities` facets stay empty until their enrichment coverage is high enough to be reliable. group_by enum: `country`, `market`, `admin1`, `locality`, `room_type`, `property_type`.\n- **Params:** `active_since` (string, optional) — Freshness filter, an ISO-8601 date (YYYY-MM-DD); `country` (string, optional) — Exact ISO-3166-1 alpha-2 country filter, e.g. FR; `facet` (string, **required**) — Facet enum: country, market, currency, superhost, guest_favorite, rating_band, review_band, admin1, locality, room_type, property_type, amenities; `group_by` (string, optional) — Aggregate cell dimension enum: country, market, admin1, locality, room_type, property_type. Defaults to country; `guest_favorite` (boolean, optional) — Count only Guest Favorite listings (an observed lower bound; the badge under-counts); `market` (string, optional) — Exact metro-market filter, max 128 characters; `min_listings` (integer, optional) — Minimum listings per bucket; raises the small-cell suppression floor; `min_rating` (number, optional) — Minimum listing rating, from 0 through 5; `min_review_count` (integer, optional) — Minimum listing review count, 0 or greater; `superhost` (boolean, optional) — Count only Superhost listings\n\n### `datasets_airbnb_item`\n\n- **HTTP:** `GET /datasets/airbnb-markets/items/{country}`\n- **What:** Get an Airbnb market from the dataset. Returns one country's full aggregate Airbnb market profile from dataset id enum value `airbnb-markets` — headline supply, Superhost share, Guest Favorite share (`guest_favorite_pct`, an observed lower bound), `avg_person_capacity` (average guests a listing sleeps over the detail-page-enriched sample), ratings, its top metros, bounding box, per-currency nightly-price percentiles, and a USD-normalized `price_usd` percentile block "},{"path":"skill-card.md","content":"## Description:\n\nQueries Crawlora's hosted public datasets for searches, aggregate breakdowns, nearby records, and individual records, returning JSON without crawling each source live.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[crawlora-org](https://clawhub.ai/user/crawlora-org)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and researchers use this skill to search and analyze Crawlora's pre-indexed public datasets, compare populations through facets, and retrieve records by dataset ID.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Queries and the API key are sent to Crawlora's hosted API.\n\nMitigation: Use a dedicated Crawlora key and keep secrets out of query text.\n\nRisk: Public-contact and business records may require careful handling.\n\nMitigation: Use returned data in accordance with applicable policy and law.\n\n## Reference(s):\n\n- [Crawlora dataset endpoint reference](reference/endpoints.md)\n- [Crawlora](https://crawlora.net)\n\n## Skill Output:\n\n**Output Type(s):** [JSON, Text]\n\n**Output Format:** [JSON API responses and concise text summaries]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Paginated results; hosted datasets are refreshed periodically rather than in real time.]\n\n## Skill Version(s):\n\n1.0.20 (source: server-resolved release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Queries Crawlora's pre-built hosted datasets — Airbnb markets, App Store/Google Play apps, GitHub/Instagram/X users, job postings, US housing markets, Google Maps businesses, Goodreads, PitchBook, Steam, TrustMRR, Product Hunt, SEC companies, tech-stack, and more — via search/facets/item/nearby endpoints, returning clean JSON without live-crawling each platform. Use when the user wants bulk or aggregate analysis, to search a pre-indexed corpus, to facet/filter a large population, or to look up one record by its dataset id, instead of scraping pages one at a time. Skill: crawlora-datasets Owner: crawlora-org Summary: Queries Crawlora's pre-built hosted datasets — Airbnb markets, App Store/Google Play apps, GitHub/Instagram/X users, job postings, US housing markets, Google Maps businesses, Goodreads, PitchBook, Steam, TrustMRR, Product Hunt, SEC companies, tech-stack, and more — via search/facets/item/nearby endpoints, returning clean JSON without live-crawling each platform. U","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1561,"uniquenessScore":47,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T04:01:55.945Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T04:01:55.945Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T07:39:51.579Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}