{"id":"a8688e7d-acbf-4b27-be9e-2a4c362e051b","entityType":"agent","slug":"clawhub-sciverse-academic-retrieval","name":"sciverse academic retrieval","canonicalUrl":"https://www.xpersona.co/agent/clawhub-sciverse-academic-retrieval","canonicalPath":"/agent/clawhub-sciverse-academic-retrieval","generatedAt":"2026-10-09T16:07:25.420Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T11:27:22.149Z","emptyReason":null},"description":"Retrieve academic papers by structured metadata, perform semantic chunk search for RAG, and read byte-range content for citation-grade scientific literature.","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 2.8K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s17cs0hvsmy3xp1z9jffmdc8rn86j73f:academic-retrieval","sourceUrl":"https://clawhub.ai/sciverse/academic-retrieval","homepage":"https://clawhub.ai/sciverse/skills/academic-retrieval","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/sciverse/academic-retrieval","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/sciverse/skills/academic-retrieval","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":69,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"sciverse academic retrieval technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T11:27:22.149Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T11:27:22.149Z","emptyReason":null},"stars":null,"forks":null,"downloads":2801,"packageName":null,"latestVersion":"0.14.3","tractionLabel":"2.8K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T11:27:22.117Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T11:27:22.149Z","lastCrawledAt":"2026-10-09T11:27:22.117Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T11:27:22.117Z","lastVerifiedAt":null,"highlights":[{"version":"0.14.3","createdAt":"2026-09-20T04:01:30.789Z","changelog":"- Version bump to 0.14.3. - Documentation updates in SKILL.md. - Removed skill-card.md file.","fileCount":12,"zipByteSize":21104},{"version":"0.14.2","createdAt":"2026-09-11T03:04:28.277Z","changelog":"- Updated description and documentation to clarify that read_content uses Unicode code point offsets for character-range content reading. - Improved and expanded skill documentation in SKILL.md and README.md for better clarity. - Removed the skill-card.md file as part of documentation cleanup.","fileCount":12,"zipByteSize":20213},{"version":"0.14.1","createdAt":"2026-09-10T05:26:42.253Z","changelog":"- The read_content tool now reads paper content by Unicode code-point offsets (character-based, not byte-based), improving support for multilingual and special-character text. - Updated documentation to clarify read_content input/output fields and paging behavior, including defaults and maximum values. - Minor edits to clarify tool descriptions and usage in SKILL.md and supporting files. - Removed skill-card.md as part of documentation cleanup.","fileCount":12,"zipByteSize":20131},{"version":"0.14.0","createdAt":"2026-08-14T11:30:40.557Z","changelog":"- Updated documentation to clarify that sort_by_year now defaults to auto (relevance with query, newest-first for pure filters). - Added guidance discouraging the use of sort order \"desc\" together with query, explaining it disables boosts and degrades to filter-match sorting. - Removed redundant/deprecated documentation file (skill-card.md). - Minor wording and formatting improvements for clarity in usage recipes and boost explanations.","fileCount":12,"zipByteSize":19340},{"version":"0.13.1","createdAt":"2026-08-07T04:24:55.819Z","changelog":"- Bumped version to 0.13.1. - Minor documentation or metadata updates in SKILL.md. - No functional changes to core features.","fileCount":12,"zipByteSize":19112},{"version":"0.13.0","createdAt":"2026-08-07T04:18:15.964Z","changelog":"- Added new ranking boost options for fuzzy search: impact_boost and language_affinity (stackable with freshness_boost). - Updated documentation to describe all three boosts (freshness, impact, language) and their usage. - Clarified soft/hard exclusion distinctions in language and impact ranking options. - Removed outdated skill-card.md file.","fileCount":12,"zipByteSize":19111},{"version":"0.12.0","createdAt":"2026-08-06T08:46:52.575Z","changelog":"- Added support for server-side filtering in semantic_search via a new filters argument, enabling constrained corpus search (e.g., by author or year). - Documented new semantic_search filtering, including both soft and hard (doc_id-based) scoping patterns in the Recipes section. - Removed the deprecated skill-card.md file. - Minor documentation improvements and clarifications throughout SKILL.md.","fileCount":12,"zipByteSize":18377},{"version":"0.11.2","createdAt":"2026-08-05T12:39:31.408Z","changelog":"- Updated skill version to 0.11.2. - Improved semantic search implementation. - Updated documentation in SKILL.md. - Removed deprecated skill-card.md file.","fileCount":12,"zipByteSize":16831}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17cs0hvsmy3xp1z9jffmdc8rn86j73f:academic-retrieval","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s17cs0hvsmy3xp1z9jffmdc8rn86j73f:academic-retrieval` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/sciverse/academic-retrieval before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sciverse-academic-retrieval/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sciverse-academic-retrieval/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sciverse-academic-retrieval/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-sciverse-academic-retrieval/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-sciverse-academic-retrieval/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-sciverse-academic-retrieval/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-09T16:07:25.416Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sciverse-academic-retrieval/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sciverse-academic-retrieval/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sciverse-academic-retrieval/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-sciverse-academic-retrieval/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T11:27:22.149Z","emptyReason":null},"readme":"Skill: sciverse academic retrieval\n\nOwner: sciverse\n\nSummary: Retrieve academic papers by structured metadata, perform semantic chunk search for RAG, and read byte-range content for citation-grade scientific literature.\n\nTags: latest:0.14.3\n\nVersion history:\n\nv0.14.3 | 2026-09-20T04:01:30.789Z | auto\n\n- Version bump to 0.14.3.\n- Documentation updates in SKILL.md.\n- Removed skill-card.md file.\n\nv0.14.2 | 2026-09-11T03:04:28.277Z | auto\n\n- Updated description and documentation to clarify that read_content uses Unicode code point offsets for character-range content reading.\n- Improved and expanded skill documentation in SKILL.md and README.md for better clarity.\n- Removed the skill-card.md file as part of documentation cleanup.\n\nv0.14.1 | 2026-09-10T05:26:42.253Z | auto\n\n- The read_content tool now reads paper content by Unicode code-point offsets (character-based, not byte-based), improving support for multilingual and special-character text.\n- Updated documentation to clarify read_content input/output fields and paging behavior, including defaults and maximum values.\n- Minor edits to clarify tool descriptions and usage in SKILL.md and supporting files.\n- Removed skill-card.md as part of documentation cleanup.\n\nv0.14.0 | 2026-08-14T11:30:40.557Z | auto\n\n- Updated documentation to clarify that sort_by_year now defaults to auto (relevance with query, newest-first for pure filters).\n- Added guidance discouraging the use of sort order \"desc\" together with query, explaining it disables boosts and degrades to filter-match sorting.\n- Removed redundant/deprecated documentation file (skill-card.md).\n- Minor wording and formatting improvements for clarity in usage recipes and boost explanations.\n\nv0.13.1 | 2026-08-07T04:24:55.819Z | auto\n\n- Bumped version to 0.13.1.\n- Minor documentation or metadata updates in SKILL.md.\n- No functional changes to core features.\n\nv0.13.0 | 2026-08-07T04:18:15.964Z | auto\n\n- Added new ranking boost options for fuzzy search: impact_boost and language_affinity (stackable with freshness_boost).\n- Updated documentation to describe all three boosts (freshness, impact, language) and their usage.\n- Clarified soft/hard exclusion distinctions in language and impact ranking options.\n- Removed outdated skill-card.md file.\n\nv0.12.0 | 2026-08-06T08:46:52.575Z | auto\n\n- Added support for server-side filtering in semantic_search via a new filters argument, enabling constrained corpus search (e.g., by author or year).\n- Documented new semantic_search filtering, including both soft and hard (doc_id-based) scoping patterns in the Recipes section.\n- Removed the deprecated skill-card.md file.\n- Minor documentation improvements and clarifications throughout SKILL.md.\n\nv0.11.2 | 2026-08-05T12:39:31.408Z | auto\n\n- Updated skill version to 0.11.2.\n- Improved semantic search implementation.\n- Updated documentation in SKILL.md.\n- Removed deprecated skill-card.md file.\n\nv0.11.1 | 2026-08-05T07:25:15.930Z | auto\n\n- skill-card.md file was removed.\n- Documentation and manifest were updated to reflect version 0.11.1.\n- No changes to user-facing features or APIs.\n\nv0.11.0 | 2026-07-31T11:16:07.851Z | auto\n\n- Bumped version to 0.11.0.\n- Removed the redundant skill-card.md file.\n- No user-facing functionality or API changes; documentation and metadata updates only.\n\nv0.10.0 | 2026-07-28T08:44:33.127Z | auto\n\n- Added details and usage limits for large citation/reference sets in list_paper_relations, clarifying behavior for highly cited papers.\n- Updated documentation to reflect that citations/references/related_works are not projectable in search_papers and can exceed 100,000 entries.\n- Noted result limits and recommended using search_papers with filters_advanced for deep paging of large CITATIONS sets.\n- Removed redundant documentation file (skill-card.md).\n\nv0.9.0 | 2026-07-03T07:11:28.701Z | auto\n\nVersion 0.9.0\n\n- Added documentation for the is_content_accessible field to clarify when fulltext reading via read_content is available.\n- Removed the skill-card.md file.\n- Minor documentation updates in SKILL.md for clearer instructions on checking content access before reading fulltext.\n\nv0.8.1 | 2026-06-24T08:16:17.325Z | auto\n\n- Added new tool: list_paper_relations for paginating full citation/reference/related works lists of a paper.\n- Introduced scripts/list_paper_relations.mjs.\n- Removed deprecated skill-card.md file.\n- Updated documentation to reflect new paper relation features and tool usage.\n\nv0.8.0 | 2026-06-24T06:21:07.353Z | auto\n\n- Added list_paper_relations endpoint for paginating full paper relation lists, such as citations and references.\n- Updated documentation and recipes to include the new relations endpoint and clarify differences between CITATIONS (incoming) and REFERENCES (outgoing).\n- Expanded search_papers functionality to support searching by collection (authors, sources), with documentation and usage examples.\n- Removed skill-card.md file.\n\nv0.7.2 | 2026-06-17T06:27:03.898Z | auto\n\n- Removed deprecated skill-card.md file.\n- Minor internal script updates for get_resource and shared logic.\n- No user-facing changes to tool interface or usage.\n\nv0.7.1 | 2026-05-28T08:49:22.854Z | auto\n\nacademic-retrieval v0.7.1\n\n- Internal documentation update: minor adjustments to SKILL.md.\n- Removed obsolete file: skill-card.md.\n- No user-facing behavior or API changes in this release.\n\nv0.7.0 | 2026-05-28T03:10:38.741Z | auto\n\nacademic-retrieval 0.7.0 introduces a new freshness_boost option for search ranking.\n\n- Added the freshness_boost parameter to search_papers to weight results by recency, with customizable decay rates.\n- Updated SKILL.md documentation with usage examples and explanations for freshness_boost.\n- Removed the deprecated skill-card.md documentation file.\n\nv0.6.3 | 2026-05-25T12:03:15.347Z | auto\n\n- Improved field naming in search_papers results: now returns unique_id (always present), doc_id (only when full text available), and publication_venue_name_unified.\n- Updated documentation to reflect the revised result structure and clarify field presence.\n- No API changes—only changes to returned field names and documentation.\n\nv0.5.3 | 2026-05-19T05:05:17.674Z | auto\n\n- Added stricter file path validation to get_resource for enhanced security.\n- Improved internal documentation and usage messages for scripts.\n- Minor adjustments in README, manifest, and metadata for clarity and consistency.\n\nv0.5.2 | 2026-05-19T04:21:39.207Z | auto\n\n- Bumped version from 0.5.1 to 0.5.2.\n- Documentation and metadata updates in SKILL.md, README.md, and manifest.json.\n- No functional or API changes; internal and docs-only update.\n\nv0.5.1 | 2026-05-18T12:18:06.880Z | auto\n\n- Version bump from 0.5.0 to 0.5.1.\n- No functional or documentation changes except for updating the version in SKILL.md and manifest.json.\n\nv0.5.0 | 2026-05-18T09:21:54.201Z | auto\n\nacademic-retrieval 0.5.0\n\n- Version bump to 0.5.0.\n- No user-facing changes described; documentation and metadata updated.\n\nv0.4.4 | 2026-05-15T09:53:13.203Z | auto\n\nacademic-retrieval v0.4.4\n\n- Version bump from 0.4.3 to 0.4.4 in documentation and manifest files\n- No functional or API changes—documentation and metadata updates only\n\nv0.4.3 | 2026-05-15T08:16:32.509Z | auto\n\n- Bump version to 0.4.3.\n- No functional or user-facing changes; documentation and metadata only.\n\nv0.1.8 | 2026-05-14T10:51:59.023Z | auto\n\nVersion 0.1.8 summary: Adds listing of schema fields and image retrieval for academic papers.\n\n- Added list_catalog tool to return the full schema catalog, including field types, filterability, and sample values.\n- Added get_resource tool to fetch binary images (figures/tables) referenced in paper content.\n- Expanded documentation with usage patterns and recipes for the new features.\n- No changes to existing APIs; fully backwards compatible.\n\nv0.1.7 | 2026-05-13T03:13:52.441Z | auto\n\n- Internal code updates in scripts/_common.mjs.\n- No user-facing feature or documentation changes.\n\nv0.1.6 | 2026-05-12T11:18:44.940Z | auto\n\n- Updated skill name and slug to \"academic-retrieval\" for consistency.\n- Incremented version and metadata info.\n- Improved documentation formatting and naming in SKILL.md and README.md.\n\nv0.1.5 | 2026-05-12T10:53:07.831Z | auto\n\n- Updated documentation and usage instructions in SKILL.md, clarifying when and how to use each tool.\n- Expanded descriptions for search_papers, semantic_search, and read_content tools.\n- Added typical invocation patterns and exit code explanations for better usability.\n- Improved examples and composition patterns for agent workflows requiring citation-grade scientific literature.\n\nArchive index:\n\nArchive v0.14.3: 12 files, 21104 bytes\n\nFiles: manifest.json (25045b), README.md (2969b), scripts/_common.mjs (3137b), scripts/get_resource.mjs (766b), scripts/list_catalog.mjs (465b), scripts/list_paper_relations.mjs (284b), scripts/read_content.mjs (355b), scripts/search_papers.mjs (259b), scripts/semantic_search.mjs (1192b), skill-card.md (2206b), SKILL.md (9354b), _meta.json (138b)\n\nFile v0.14.3:SKILL.md\n\n---\nname: sciverse-academic-retrieval\nslug: academic-retrieval\nversion: 0.14.3\ndescription: Sciverse academic paper retrieval: structured metadata search, semantic chunk retrieval for RAG, and character-range content reading (offsets in Unicode code points). For agent workflows that need citation-grade scientific literature.\nlicense: Apache-2.0\nhomepage: https://sciverse.space\n---\n\n# academic-retrieval\n\nSciverse academic paper retrieval: structured metadata search, semantic chunk retrieval for RAG, and character-range content reading (offsets in Unicode code points). For agent workflows that need citation-grade scientific literature.\n\n## When to use\n\nTrigger this skill when the user's request involves any of:\n\n- Locating academic papers by structured criteria (authors, year, journal, subjects)\n- Grounding answers in paper excerpts (RAG / citations)\n- Expanding the original text around a known doc_id (more text before/after a chunk)\n\n## Authentication\n\nThis skill requires the `SCIVERSE_API_TOKEN` environment variable\n(obtain from https://sciverse.space). Optionally set `SCIVERSE_BASE_URL`\nto override the default API base URL.\n\n## Tools\n\n### search_papers\n\nSearch academic papers by structured filters (title, authors, journal,\nyear, subjects, etc.).\nUse when: \"find Hinton's papers from 2020-2023\", \"Nature papers on\nCRISPR\".\nNot for: natural-language Q&A retrieval (use semantic_search) or\nfull-text snippets (use read_content).\nReturns: list of papers; each entry has unique_id (always present),\ndoc_id (only when full text exists), title, author, abstract,\npublication_venue_name_unified, publication_published_year.\n\n**Invoke**: `node scripts/search_papers.mjs '<JSON args>'`\n\n### semantic_search\n\nNatural-language semantic search returning relevant paper chunks for\nRAG-style answering.\nUse when: \"How does Transformer attention work?\", \"What are recent\nmethods for protein structure prediction?\".\nNot for: precise field filtering (use search_papers) or fetching full\noriginal text (use read_content).\nReturns: list of chunks; each entry has chunk_id, doc_id, abstract,\nchunk, score, title, offset.\nTypical chain: semantic_search → pick chunk → read_content(doc_id,\noffset).\n\n**Invoke**: `node scripts/semantic_search.mjs '<JSON args>'`\n\n### list_catalog\n\nReturns the schema catalog for search_papers: every field name, type,\nwhether it's filterable / sortable, default-return status, human\ndescription, and applicable FilterOperators.\nUse when: \"Which field do I filter by DOI?\", \"What values can\naccess_oa_status take?\", \"What's the right enum for metadata_type?\".\nNot for: actually searching papers (use search_papers / semantic_search).\nTypical pattern: call once when first encountering Sciverse or facing\nan ambiguous field need, then construct precise search_papers filters\nfrom the returned schema.\nPass include_sample_values=true to also fetch top-20 values for\nenum-like fields (OpenSearch terms aggregation, 24h cached).\n\n**Invoke**: `node scripts/list_catalog.mjs '<JSON args>'`\n\n### list_paper_relations\n\nPaginate the full relation list of a paper. citations/references/related_works\nare unbounded arrays (up to 340k entries for a single paper) and are NOT\nprojectable in search_papers, so this endpoint is the only way to read them.\nUse when: \"What does paper X cite?\" (relation=REFERENCES), \"Which papers cite\npaper X?\" (relation=CITATIONS), \"Works related to paper X\" (relation=RELATED_WORKS).\nNote: CITATIONS (incoming: who cites me) and REFERENCES (outgoing: who I cite)\nare opposite directions.\nTypical chain: get unique_id from search_papers / semantic_search, then paginate\nhere by relation.\nTwo limits (CITATIONS only; REFERENCES/RELATED_WORKS max out at 11833/20 in practice):\nmore than 10000 relations returns 429; page*page_size above 10000 returns 400.\nIn both cases switch to search_papers with filters_advanced on\nreferences_unique_id — it supports deep paging and arbitrary sorting.\ntotal_count counts in-corpus matches only, so it can differ from the paper's own\ncitation_count by about 1%.\n\n**Invoke**: `node scripts/list_paper_relations.mjs '<JSON args>'`\n\n### read_content\n\nRead a range of a paper's original text addressed in Unicode code\npoints (offset/limit count characters like Python len(), not bytes).\nTypically used with a doc_id/offset returned by semantic_search to\nexpand context (read more text before or after a chunk).\nReturns: text fragment, bytes_returned (UTF-8 byte length of text, for\nreference only), next_offset (code-point offset of the next fragment —\npage with it, never with bytes_returned), more (boolean).\nServer behaviour: limit above 524288 is silently clamped; omitting\noffset returns the whole document ignoring limit — the SDKs / MCP\nserver send offset=0 and limit=4096 by default, so pass offset\nexplicitly when calling the HTTP API directly.\n\n**Invoke**: `node scripts/read_content.mjs '<JSON args>'`\n\n### get_resource\n\nReturns the binary bytes of a paper figure / table image referenced\ninside read_content's Markdown via `![alt](file_name)` placeholders.\nUse when the user asks to see / display / describe a figure and\nread_content output contains an image reference.\nInput file_name comes from the Markdown URL part (relative path,\nno `\\\\` or `..`).\nReturns: raw image stream + image/* Content-Type. The SDK / MCP\nserver wraps the bytes as base64 + mimeType so Claude (multimodal)\ncan read the image directly.\n\n**Invoke**: `node scripts/get_resource.mjs '<JSON args>'`\n\n## Bootstrap: learn the schema first\n\nIf you're unsure which fields exist or what values an enum takes\n(e.g. `metadata_type`, `language`, `access_oa_status`), call\n`list_catalog` once at the start. Sample values are returned for\nlow-cardinality fields. Use it instead of guessing field names —\nguessing wastes turns.\n\n```\nlist_catalog(include_sample_values=true)\n    └─▶ fields[].name + sample_values  →  precise filter construction\n```\n\n## Recipes\n\n**RAG flow (natural-language Q&A):**\n\n```\nsemantic_search(query=...) → hits[i].doc_id, hits[i].offset\n    └─▶ read_content(doc_id, offset)\n```\n\n**Lookup by DOI:**\n\n```\nsearch_papers(filters_advanced=[{field: \"doi\", value: \"10.1038/...\"}])\n```\n\n**OA + year filter:**\n\n```\nsearch_papers(\n    year_from=2024,\n    filters_advanced=[{field: \"access_is_oa\", value: \"true\"}]\n)\n```\n\n**Scoped semantic search (constrained corpus):**\n\n```\nsemantic_search(\n    query=\"...\",\n    filters={\"author\": [\"Hinton\"],\n             \"publication_published_year\": {\"gte\": 2020}}\n)   # applied at recall time, server-side; AND across fields\n```\n\nSoft semantics: chunks missing that metadata are NOT excluded.\nFor a hard guarantee, or meta-only constraints (fwci, citation graph,\ncomplex hit-sets), scope by doc_id — a HARD recall-time filter:\n\n```\nsearch_papers(..., fields=[\"doc_id\",\"title\"]) → collect doc_id\nsemantic_search(query=..., filters={\"doc_id\": [...]})\n    # hits never leave the set; empty list → empty hits (never global);\n    # up to 1000 deduped ids (400 SCOPE_TOO_LARGE beyond)\n```\n\n**Bias fuzzy search ranking (soft boosts — stackable):**\n\nThree multiplicative boosts (`freshness_boost` / `impact_boost` /\n`language_affinity`, each NONE/MILD/STRONG) reorder fuzzy-search\nresults while keeping relevance. Only effective when `query` is\nnon-empty; ignored when any sort is set; shallow paging while active.\n`sort_by_year` defaults to `auto` (relevance with `query`, newest-first\nfor pure filters); `query`+`desc` is an anti-pattern — it degrades the\nquery to a match filter and disables all boosts; use `freshness_boost`.\n\n```\nsearch_papers(query=\"large language model\", freshness_boost=\"STRONG\")\n    # recent first: STRONG=3-year decay, MILD=10-year\nsearch_papers(query=\"protein folding\", impact_boost=\"MILD\")\n    # highly-cited float up (bounded; zero-citation stays neutral)\nsearch_papers(query=\"深度学习\", language_affinity=\"MILD\")\n    # demote (never exclude) results not in the query's language;\n    # unknown-language papers stay neutral; hard-exclude via\n    # filters_advanced=[{\"field\":\"language\",\"value\":\"zh\"}]\n```\n\n**Search authors or journals (collection):**\n\nSet `collection` to `authors` or `sources` (default `papers`) to search\nthose entities. Each has its own fields — call\nlist_catalog(collection=\"authors\") first; use filters_advanced +\nsort_advanced (papers convenience fields apply to papers only).\n\n```\nsearch_papers(collection=\"authors\",\n    filters_advanced=[{field: \"summary_stats.h_index\", operator: \"FILTER_OP_GTE\", value: 50}],\n    sort_advanced=[{field: \"cited_by_count\", order: \"SORT_ORDER_DESC\"}])\n```\n\n**Fetch a paper figure / image:**\n\nWhen read_content Markdown contains `![alt](file_name)`, call\n`get_resource` with the file_name to fetch image binary.\n\n```\nread_content(doc_id, offset) → markdown ![Figure 3](dt=xxx/p/f3.png)\n    └─▶ get_resource(file_name=\"dt=xxx/p/f3.png\")\n```\n\n**Reading fulltext (check first):**\n\nEach search_papers hit carries `is_content_accessible` (bool): `true` only when\nthe paper has fulltext AND the caller is authorized. Check it before\n`read_content(doc_id, ...)` — `false` means no fulltext or no read permission.\n\n## Exit codes\n\n- `0` — success; stdout is the JSON response\n- `1` — HTTP 4xx/5xx; stderr contains status code and response body\n- `2` — argument error (missing token, malformed JSON, required field absent)\n\nFile v0.14.3:README.md\n\n# academic-retrieval — ClawHub skill bundle\n\n[![ClawHub](https://img.shields.io/badge/clawhub-academic--retrieval-brightgreen)](https://clawhub.ai/sciverse/skills/academic-retrieval)\n\nClawHub skill that gives any OpenClaw agent Sciverse academic-paper retrieval\ncapabilities (English | [中文](#中文说明)).\n\nPublished by **@sciverse** (slug `academic-retrieval`).\n\n## Install\n\n```bash\nopenclaw skills install academic-retrieval\n```\n\n## Configure\n\n```bash\nexport SCIVERSE_API_TOKEN=sv-xxx       # obtain from https://sciverse.space\n```\n\n## Tools at a glance\n\n| Tool | Purpose |\n|---|---|\n| `list_catalog` | Field introspection (call once to learn available fields + enum values) |\n| `search_papers` | Structured metadata search over papers / authors / sources (set `collection`) |\n| `semantic_search` | Natural-language semantic chunk retrieval (for RAG) |\n| `read_content` | Character-range read of a paper's original text (offset/limit in Unicode code points) |\n| `get_resource` | Fetch figure / table image bytes referenced inside `read_content` Markdown |\n\nSee `SKILL.md` for full agent-facing documentation.\n\n## Direct invocation (bypass OpenClaw)\n\n```bash\nnode scripts/semantic_search.mjs '{\"query\":\"Transformer attention mechanism\",\"top_k\":3}'\n```\n\n## Relationship to the SDK\n\nThis skill is **complementary** to the `sciverse` packages on PyPI / npm:\n\n- **This skill** — OpenClaw users only. Zero external deps (Node 18+ native fetch).\n- **PyPI / npm SDK** — Any LLM agent framework (OpenAI, Anthropic, LangChain, LlamaIndex…).\n\n## License\n\nApache-2.0\n\n---\n\n## 中文说明\n\nOpenClaw 用户专用：通过 ClawHub 一键给 agent 加上 Sciverse 学术文献检索能力。\n\n发布者 **@sciverse**，slug `academic-retrieval`。\n\n### 安装\n\n```bash\nopenclaw skills install academic-retrieval\n```\n\n### 配置\n\n```bash\nexport SCIVERSE_API_TOKEN=sv-xxx   # 从 https://sciverse.space 控制台申请\n# 可选：export SCIVERSE_BASE_URL=https://api-custom.sciverse.space\n```\n\n### 工具速览\n\n| Tool | 用途 |\n|---|---|\n| `list_catalog` | 字段 introspection（首次接入调一次，学习可用字段和 enum 取值） |\n| `search_papers` | 按结构化条件查 papers / authors / sources（用 `collection` 切换实体集合） |\n| `semantic_search` | 自然语言语义检索文献片段（RAG 用） |\n| `read_content` | 按 Unicode 码点区间读取文献原文片段 |\n| `get_resource` | 取 `read_content` Markdown 中引用的图片字节流（多模态 RAG） |\n\nagent 视角的完整文档见 `SKILL.md`（英文）。\n\n### 直接调用（不通过 OpenClaw）\n\n```bash\nnode scripts/semantic_search.mjs '{\"query\":\"Transformer 注意力机制\",\"top_k\":3}'\n```\n\n### 与 SDK 的关系\n\n本 skill 与 PyPI/npm 上的 `sciverse` 包是**互补**的：\n\n- **本 skill**：OpenClaw 用户专用，零外部依赖（仅 Node 18+ native fetch）\n- **PyPI/npm SDK**：任意 LLM Agent 框架（OpenAI / Anthropic / LangChain / LlamaIndex...）\n\nFile v0.14.3:_meta.json\n\n{\n  \"ownerId\": \"kn74way11x0gjn6wpa8hcvyhvs85vmkj\",\n  \"slug\": \"academic-retrieval\",\n  \"version\": \"0.14.3\",\n  \"publishedAt\": 1789876890789\n}\n\nFile v0.14.3:skill-card.md\n\n## Description:\n\nSciverse academic paper retrieval supports structured metadata search, semantic chunk retrieval for RAG, citation relation lookup, and character-range content reading for scientific literature workflows.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[sciverse](https://clawhub.ai/user/sciverse)\n\n### License/Terms of Use:\n\nApache-2.0\n\n## Use Case:\n\nDevelopers and research-focused agents use this skill to locate academic papers, retrieve grounded text snippets, inspect citation relationships, and fetch referenced figures or tables for literature review and RAG workflows.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Search queries, document IDs, and resource names are sent to Sciverse endpoints with the user's API token.\n\nMitigation: Use a token approved for the intended workspace and avoid submitting confidential research prompts or identifiers unless Sciverse terms and internal policy allow it.\n\nRisk: Retrieved chunks, full text, citation relations, and images can be permission-limited, partial, or approximate for citation-grade use.\n\nMitigation: Check content accessibility and verify important claims or citations against the source publication before relying on downstream answers.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/sciverse/skills/academic-retrieval)\n- [Sciverse homepage](https://sciverse.space)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, JSON, images, configuration]\n\n**Output Format:** [JSON responses containing paper metadata, semantic chunks, Markdown text fragments, citation relation lists, or base64-encoded image resources]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Requires a Sciverse API token; Node 18+ scripts call Sciverse endpoints and print machine-readable JSON.]\n\n## Skill Version(s):\n\n0.14.3 (source: server release metadata and SKILL.md frontmatter)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v0.14.3:manifest.json\n\n{\n  \"name\": \"sciverse-academic-retrieval\",\n  \"version\": \"0.14.3\",\n  \"slug\": \"academic-retrieval\",\n  \"description\": \"Sciverse academic paper retrieval: structured metadata search, semantic chunk retrieval for RAG, and character-range content reading (offsets in Unicode code points). For agent workflows that need citation-grade scientific literature.\",\n  \"runtime\": \"node>=18\",\n  \"license\": \"Apache-2.0\",\n  \"homepage\": \"https://sciverse.space\",\n  \"env\": [\n    {\n      \"name\": \"SCIVERSE_API_TOKEN\",\n      \"required\": true,\n      \"description\": \"Sciverse API Token (obtain from https://sciverse.space).\"\n    },\n    {\n      \"name\": \"SCIVERSE_BASE_URL\",\n      \"required\": false,\n      \"default\": \"https://api.sciverse.space\",\n      \"description\": \"Override the default API base URL (for dev / self-hosted gateways).\"\n    }\n  ],\n  \"tools\": [\n    {\n      \"name\": \"search_papers\",\n      \"description\": \"Search academic papers by structured filters (title, authors, journal,\\nyear, subjects, etc.).\\nUse when: \\\"find Hinton's papers from 2020-2023\\\", \\\"Nature papers on\\nCRISPR\\\".\\nNot for: natural-language Q&A retrieval (use semantic_search) or\\nfull-text snippets (use read_content).\\nReturns: list of papers; each entry has unique_id (always present),\\ndoc_id (only when full text exists), title, author, abstract,\\npublication_venue_name_unified, publication_published_year.\",\n      \"script\": \"scripts/search_papers.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"properties\": {\n          \"collection\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"papers\",\n              \"authors\",\n              \"sources\"\n            ],\n            \"default\": \"papers\",\n            \"description\": \"检索的实体集合。papers（默认，论文）/ authors（作者）/ sources（来源期刊）。 各 collection 字段集不同，用 list_catalog（collection=<name>）学习对应 schema。 注意：本工具的便捷字段（authors/journals/year_from/subjects 等）只对 papers 有意义； 查 authors/sources 时改用 filters_advanced + 该 collection 的字段名（如 authors 的 summary_stats.h_index / orcid，sources 的 issn / is_oa）。authors 用 orcid、 sources 用 issn 与论文检索结果关联。\",\n            \"x-en-description\": \"Entity collection to search. papers (default) / authors / sources. Each collection has its own field schema — call list_catalog(collection=<name>). The convenience fields (authors/journals/year_from/subjects) apply to papers only; for authors/sources use filters_advanced with that collection's field names.\"\n          },\n          \"query\": {\n            \"type\": \"string\",\n            \"description\": \"BM25 全文关键词，匹配标题/摘要/期刊名/关键词字段。留空则纯靠结构化过滤。\\n普通关键词是宽松匹配（任一词命中、按相关性排序）。\\n\\n也支持布尔检索式：全大写 AND / OR / NOT、括号分组、引号短语；优先级\\nNOT > AND > OR，相邻词隐式 AND。例如\\n  (histopathology OR pathology) AND (\\\"deep learning\\\" OR \\\"machine learning\\\") AND (prognosis OR survival)\\n布尔式里每个检索词都是硬条件、不做放宽——0 命中就是 0。\\n- query 只放检索词：「检索式1（预后预测）」这类标签/说明也会变成必须命中的词。\\n- 小写 and/or 是普通词；要检索字面量 OR（如比值比）加引号 \\\"OR\\\"。\\n- 引号需与运算符同时出现才生效：只写 \\\"spread through air spaces\\\" 不带运算符时\\n  按普通关键词处理；写成 \\\"spread through air spaces\\\" AND lung 才是短语精确匹配。\\n- 最多 64 个检索词、括号嵌套 10 层，超限返回 400；复杂检索请拆成多次调用。\\n\",\n            \"x-en-description\": \"BM25 full-text keywords over title/abstract/venue/keywords; leave empty for\\npure structured filtering. Plain keywords match loosely (any word, ranked by\\nrelevance).\\n\\nBoolean syntax is also supported: UPPERCASE AND / OR / NOT, ( ) grouping and\\n\\\"quoted phrases\\\"; precedence NOT > AND > OR, adjacent words are implicitly AND.\\nExample:\\n  (histopathology OR pathology) AND (\\\"deep learning\\\" OR \\\"machine learning\\\") AND (prognosis OR survival)\\nIn boolean mode every term is a hard requirement and nothing is relaxed —\\n0 hits means 0.\\n- Put only search terms in query: labels or instructions such as\\n  \\\"检索式1（预后预测）\\\" become required terms too.\\n- Lowercase and/or are ordinary words; quote \\\"OR\\\" to search the literal\\n  (e.g. odds ratio).\\n- Quotes only take effect alongside an operator: a query that is just\\n  \\\"spread through air spaces\\\" is treated as plain keywords, whereas\\n  \\\"spread through air spaces\\\" AND lung is an exact phrase match.\\n- Max 64 terms and 10 nesting levels (400 otherwise); split complex searches\\n  into several calls.\\n\"\n          },\n          \"title_contains\": {\n            \"type\": \"string\",\n            \"description\": \"标题中必须包含的词（仅匹配 title 字段）。\"\n          },\n          \"abstract_contains\": {\n            \"type\": \"string\",\n            \"description\": \"摘要中必须包含的词（仅匹配 abstract 字段）。\"\n          },\n          \"authors\": {\n            \"type\": \"array\",\n            \"items\": {\n              \"type\": \"string\"\n            },\n            \"description\": \"作者名（任一命中即可）。SDK 内部映射到后端 `author` 字段（FILTER_OP_IN）。\"\n          },\n          \"year_from\": {\n            \"type\": \"integer\",\n            \"description\": \"起始发表年（含）。\"\n          },\n          \"year_to\": {\n            \"type\": \"integer\",\n            \"description\": \"结束发表年（含）。\"\n          },\n          \"journals\": {\n            \"type\": \"array\",\n            \"items\": {\n              \"type\": \"string\"\n            },\n            \"description\": \"期刊名（任一命中即可）。SDK 内部映射到后端 `publication_venue_name_unified` 字段（FILTER_OP_IN，规范化后的载体名）。\"\n          },\n          \"subjects\": {\n            \"type\": \"array\",\n            \"items\": {\n              \"type\": \"string\"\n            },\n            \"description\": \"学科分类，如 \\\"computer science\\\"、\\\"biology\\\"。\"\n          },\n          \"filters_advanced\": {\n            \"type\": \"array\",\n            \"description\": \"高级过滤逃生舱（仅当上述字段不够用时使用）。可用字段见 get_field_catalog。\\n\\n引文反查（常用）：field=\\\"references_unique_id\\\" 查「谁引用了某篇论文」，\\nvalue 填目标论文的 unique_id。相比 list_paper_relations 的 CITATIONS，\\n它支持深翻页与任意排序，适合超高被引论文。可叠加条件，\\n例如「引用了 ResNet 且 2023 年后发表」：\\n  [{\\\"field\\\":\\\"references_unique_id\\\",\\\"value\\\":\\\"paper:10.1109/cvpr.2016.90\\\"},\\n   {\\\"field\\\":\\\"publication_published_year\\\",\\\"operator\\\":\\\"FILTER_OP_GTE\\\",\\\"value\\\":2023}]\\n该字段仅支持过滤，不能排序/聚合，也不能放进 fields 返回。\\n\",\n            \"items\": {\n              \"type\": \"object\",\n              \"required\": [\n                \"field\",\n                \"value\"\n              ],\n              \"properties\": {\n                \"field\": {\n                  \"type\": \"string\"\n                },\n                \"operator\": {\n                  \"type\": \"string\",\n                  \"description\": \"过滤操作符。MATCH（分词模糊）适用于 author、keywords（输入 \\\"Hinton\\\" 命中 \\\"Geoffrey Hinton\\\"）； MATCH_PHRASE（短语模糊）适用于 publication_venue_name_unified，整词连续匹配（\\\"Nature\\\" 命中 \\\"Nature Communications\\\"；非前缀匹配，\\\"Nature Comm\\\" 不会命中）； doi 用 EQ，服务端归一化（去 doi.org 前缀+转小写）后精确匹配。MATCH/MATCH_PHRASE 仅对配了 text 子字段的字段有效。\",\n                  \"enum\": [\n                    \"FILTER_OP_EQ\",\n                    \"FILTER_OP_NE\",\n                    \"FILTER_OP_GT\",\n                    \"FILTER_OP_GTE\",\n                    \"FILTER_OP_LT\",\n                    \"FILTER_OP_LTE\",\n                    \"FILTER_OP_IN\",\n                    \"FILTER_OP_NIN\",\n                    \"FILTER_OP_CONTAINS\",\n                    \"FILTER_OP_MATCH\",\n                    \"FILTER_OP_MATCH_PHRASE\"\n                  ],\n                  \"default\": \"FILTER_OP_EQ\"\n                },\n                \"value\": {}\n              }\n            }\n          },\n          \"sort_advanced\": {\n            \"type\": \"array\",\n            \"description\": \"高级排序逃生舱（按任意可排序字段）。papers 用 sort_by_year 即可； authors/sources 想按 h-index / 被引 / works_count 排序时用本字段。 与 query 互斥（query 走相关性排序）。\",\n            \"items\": {\n              \"type\": \"object\",\n              \"required\": [\n                \"field\",\n                \"order\"\n              ],\n              \"properties\": {\n                \"field\": {\n                  \"type\": \"string\"\n                },\n                \"order\": {\n                  \"type\": \"string\",\n                  \"enum\": [\n                    \"SORT_ORDER_DESC\",\n                    \"SORT_ORDER_ASC\"\n                  ],\n                  \"default\": \"SORT_ORDER_DESC\"\n                }\n              }\n            }\n          },\n          \"sort_by_year\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"auto\",\n              \"desc\",\n              \"asc\",\n              \"none\"\n            ],\n            \"default\": \"auto\",\n            \"description\": \"按发表年份排序。默认 auto：传了 query（或 sort_advanced）时不加年份排序\\n——保留 BM25 相关性排序，且 freshness/impact/language_affinity 软加权可用；\\n纯结构化筛选（无 query）时按年份降序（否则后端默认序是 unique_id，实质乱序）。\\n⚠️ 不要用 query + desc 求「最相关且最新」：显式排序会让 query 退化为命中\\n过滤（OR 语义、无相关性排序）、三个软加权全部失效——返回的是「含任一关键词\\n的最新文档」。要「相关且偏新」请用 freshness_boost。\\n\"\n          },\n          \"freshness_boost\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"NONE\",\n              \"MILD\",\n              \"STRONG\"\n            ],\n            \"default\": \"NONE\",\n            \"description\": \"模糊搜索新鲜度加权：结果偏向新文献（仅 query 非空时生效；传排序\\n（sort_by_year 非 none / sort_advanced）时被忽略，硬排优先）。\\nMILD: 近 10 年加权，适合日常查文献；STRONG: 近 3 年加权，适合跟踪\\n研究方向 / 追最新进展。与 impact_boost / language_affinity 可叠加\\n（均为乘法因子）。boost 生效时为浅翻页：不产 next_cursor、不支持\\ncursor 深翻页。\\n\"\n          },\n          \"impact_boost\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"NONE\",\n              \"MILD\",\n              \"STRONG\"\n            ],\n            \"default\": \"NONE\",\n            \"description\": \"模糊搜索影响力加权：高被引文献在保留相关性的前提下上浮（仅 query\\n非空时生效；传排序时被忽略）。MILD: 轻度上浮，相关性仍主导；\\nSTRONG: 明显偏向高被引。引用因子有界、零被引中性（不会归零）。\\n与 freshness_boost / language_affinity 可叠加；boost 生效时为浅翻页。\\n\"\n          },\n          \"language_affinity\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"NONE\",\n              \"MILD\",\n              \"STRONG\"\n            ],\n            \"default\": \"NONE\",\n            \"description\": \"模糊搜索语言亲和加权：非 query 语言的结果降序、但不排除（仅 query\\n非空时生效；传排序时被忽略）。目标语言由服务端从 query 文本判定\\n（假名→ja / 谚文→ko / 汉字→zh / 拉丁→en，其他书写系统不生效）；\\n语言未知的文献保持中性不降权。MILD: 非目标语言 ×0.5，跨语言强相关\\n结果仍可上浮；STRONG: ×0.2，几乎只看目标语言。与 freshness_boost /\\nimpact_boost 可叠加；boost 生效时为浅翻页。要硬排除某语言请改用\\nfilters_advanced 的 language 字段（如 {\\\"field\\\":\\\"language\\\",\\\"value\\\":\\\"en\\\"}，\\n软硬两层语义不同：本参数只调序，filter 直接排除）。\\n\"\n          },\n          \"page\": {\n            \"type\": \"integer\",\n            \"default\": 1,\n            \"minimum\": 1\n          },\n          \"page_size\": {\n            \"type\": \"integer\",\n            \"default\": 25,\n            \"minimum\": 1,\n            \"maximum\": 50,\n            \"description\": \"每页条数。省略时为服务端默认 25（SDK / MCP 不注入默认值）；工具层上限 50。\"\n          }\n        }\n      }\n    },\n    {\n      \"name\": \"semantic_search\",\n      \"description\": \"Natural-language semantic search returning relevant paper chunks for\\nRAG-style answering.\\nUse when: \\\"How does Transformer attention work?\\\", \\\"What are recent\\nmethods for protein structure prediction?\\\".\\nNot for: precise field filtering (use search_papers) or fetching full\\noriginal text (use read_content).\\nReturns: list of chunks; each entry has chunk_id, doc_id, abstract,\\nchunk, score, title, offset.\\nTypical chain: semantic_search → pick chunk → read_content(doc_id,\\noffset).\",\n      \"script\": \"scripts/semantic_search.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"required\": [\n          \"query\"\n        ],\n        \"properties\": {\n          \"query\": {\n            \"type\": \"string\",\n            \"minLength\": 1,\n            \"maxLength\": 4096,\n            \"description\": \"自然语言查询，1-200 字最佳。\"\n          },\n          \"top_k\": {\n            \"type\": \"integer\",\n            \"default\": 10,\n            \"minimum\": 1,\n            \"maximum\": 100,\n            \"description\": \"返回命中条数上限，合法 1-100（服务端校验，超出报 400）。\\n实际条数还受 mode 影响：balanced 单路混合召回在服务端固定截到约 50 条，\\ntop_k 超过 50 时多出的部分不会返回；fast 与 quality 可取到 top_k。\\n另外同一篇论文最多返回约 3 个 chunk，因此高 top_k 需要命中足够多的不同论文。\\n\"\n          },\n          \"source_types\": {\n            \"type\": \"array\",\n            \"items\": {\n              \"type\": \"string\",\n              \"enum\": [\n                \"web\",\n                \"pdf\"\n              ]\n            }\n          },\n          \"filters\": {\n            \"type\": \"object\",\n            \"description\": \"结构化过滤（可选）。在召回阶段与语义检索同时生效（ES+Milvus 双引擎下推，\\n不是结果后过滤）；多个字段之间 AND，同一字段传数组时数组内 OR。\\n⚠️ 宽松（软）语义：chunk 侧元数据缺失的文档不会被排除——例如按年份过滤时，\\n缺年份信息的 chunk 仍可能返回。需要严格范围保证时勿当硬约束使用，\\n表述结论时注明范围为\\\"近似过滤\\\"。\\n数值/日期字段支持区间：{\\\"gte\\\":2020,\\\"lte\\\":2025} 或 [min,max]（null 表示一侧不限）；\\n日期接受 YYYY / YYYY-MM / YYYY-MM-DD。\\n实际可用字段受账号字段权限约束；未知字段服务端返回 400。\\n例：{\\\"author\\\":[\\\"Hinton\\\"],\\\"publication_published_year\\\":{\\\"gte\\\":2023},\\n    \\\"topics\\\":{\\\"dimensions\\\":{\\\"primary_topic_domain\\\":\\\"Health Sciences\\\"}}}\\n\",\n            \"properties\": {\n              \"lang\": {\n                \"description\": \"语言代码，如 \\\"en\\\"、\\\"zh\\\"；也接受别名 language。\"\n              },\n              \"metadata_type\": {\n                \"description\": \"资源类型，仅单值：\\\"paper\\\" 或 \\\"ebook\\\"。\"\n              },\n              \"author\": {\n                \"description\": \"作者名，string 或 string[]（数组=任一命中）。\"\n              },\n              \"publication_venue_name_unified\": {\n                \"description\": \"发表载体名称（期刊/会议，规范化名，适合精确匹配）。\"\n              },\n              \"publication_venue_type\": {\n                \"description\": \"载体类型：\\\"journal\\\"、\\\"conference\\\"、\\\"repository\\\"、\\\"book series\\\"、\\\"ebook platform\\\"、\\\"metadata\\\"、\\\"raidRegistry\\\"、\\\"igsnCatalog\\\"、\\\"other\\\"（不区分大小写）。\"\n              },\n              \"publication_published_year\": {\n                \"description\": \"发表年份，单值或区间（{\\\"gte\\\":..,\\\"lte\\\":..} / [min,max]）。\"\n              },\n              \"publication_published_date\": {\n                \"description\": \"发表日期，\\\"YYYY[-MM[-DD]]\\\" 单值或区间。\"\n              },\n              \"citation_count\": {\n                \"description\": \"被引次数，单值或区间。\"\n              },\n              \"influential_citation_count\": {\n                \"description\": \"高影响力被引次数，单值或区间。\"\n              },\n              \"title\": {\n                \"description\": \"标题精确匹配（标题检索一般更适合 search_papers）。\"\n              },\n              \"topics\": {\n                \"description\": \"主题组合过滤：{\\\"logic\\\":\\\"and|or\\\",\\\"dimensions\\\":{\\\"primary_topic\\\":\\\"...\\\",\\\"primary_topic_domain\\\":\\\"Physical Sciences|Social Sciences|Health Sciences|Life Sciences\\\"}}；logic 省略默认 or。\"\n              },\n              \"doc_id\": {\n                \"description\": \"唯一的硬约束字段（其余字段均为软语义）：命中绝不越出给定集合。\\n值为 64 位小写 hex sha256（即 search_papers 返回的 doc_id；仅有全文的论文才有），\\nstring 或 string[]，仅 eq/in。去重后上限默认 1000，超限返回 400 SCOPE_TOO_LARGE；\\n显式传空数组返回 200 空 hits（候选集为空，不退化为全局检索）。\\n典型用法：先 search_papers 圈定候选集合，再在集合内做受限语义检索。\\n\"\n              }\n            }\n          },\n          \"mode\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"fast\",\n              \"balanced\",\n              \"quality\"\n            ],\n            \"default\": \"balanced\",\n            \"description\": \"fast = 仅关键词召回 (~200ms)；balanced = 混合检索 (~600ms)；quality = LLM 改写 + 混合 (~2-4s)。\\n\"\n          }\n        }\n      }\n    },\n    {\n      \"name\": \"list_catalog\",\n      \"description\": \"Returns the schema catalog for search_papers: every field name, type,\\nwhether it's filterable / sortable, default-return status, human\\ndescription, and applicable FilterOperators.\\nUse when: \\\"Which field do I filter by DOI?\\\", \\\"What values can\\naccess_oa_status take?\\\", \\\"What's the right enum for metadata_type?\\\".\\nNot for: actually searching papers (use search_papers / semantic_search).\\nTypical pattern: call once when first encountering Sciverse or facing\\nan ambiguous field need, then construct precise search_papers filters\\nfrom the returned schema.\\nPass include_sample_values=true to also fetch top-20 values for\\nenum-like fields (OpenSearch terms aggregation, 24h cached).\",\n      \"script\": \"scripts/list_catalog.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"properties\": {\n          \"collection\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"papers\",\n              \"authors\",\n              \"sources\"\n            ],\n            \"default\": \"papers\",\n            \"description\": \"字段 catalog 所属实体集合。papers（默认）/ authors / sources，各 collection 字段不同。\"\n          },\n          \"include_sample_values\": {\n            \"type\": \"boolean\",\n            \"default\": false,\n            \"description\": \"是否拉取 enum-like 字段的取值样本。false 仅返回静态 schema（毫秒级）；true 触发 OpenSearch terms agg（首次几百毫秒，之后 24h 走缓存）。\"\n          },\n          \"include_field_stats\": {\n            \"type\": \"boolean\",\n            \"default\": false,\n            \"description\": \"是否返回字段统计（keyword 字段基数 + 数值字段 min/max/avg/p50/p95）。触发 OpenSearch 聚合，缓存 24h。\"\n          }\n        },\n        \"required\": []\n      }\n    },\n    {\n      \"name\": \"list_paper_relations\",\n      \"description\": \"Paginate the full relation list of a paper. citations/references/related_works\\nare unbounded arrays (up to 340k entries for a single paper) and are NOT\\nprojectable in search_papers, so this endpoint is the only way to read them.\\nUse when: \\\"What does paper X cite?\\\" (relation=REFERENCES), \\\"Which papers cite\\npaper X?\\\" (relation=CITATIONS), \\\"Works related to paper X\\\" (relation=RELATED_WORKS).\\nNote: CITATIONS (incoming: who cites me) and REFERENCES (outgoing: who I cite)\\nare opposite directions.\\nTypical chain: get unique_id from search_papers / semantic_search, then paginate\\nhere by relation.\\nTwo limits (CITATIONS only; REFERENCES/RELATED_WORKS max out at 11833/20 in practice):\\nmore than 10000 relations returns 429; page*page_size above 10000 returns 400.\\nIn both cases switch to search_papers with filters_advanced on\\nreferences_unique_id — it supports deep paging and arbitrary sorting.\\ntotal_count counts in-corpus matches only, so it can differ from the paper's own\\ncitation_count by about 1%.\",\n      \"script\": \"scripts/list_paper_relations.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"required\": [\n          \"unique_id\",\n          \"relation\"\n        ],\n        \"properties\": {\n          \"unique_id\": {\n            \"type\": \"string\",\n            \"description\": \"目标论文 unique_id（如 paper:10.1038/xxx），来自 search_papers / semantic_search；勿传 doc_id。\"\n          },\n          \"relation\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"CITATIONS\",\n              \"REFERENCES\",\n              \"RELATED_WORKS\"\n            ],\n            \"description\": \"关系类型。CITATIONS=被引（谁引用了我）；REFERENCES=参考文献（我引用了谁）；RELATED_WORKS=相关工作。\"\n          },\n          \"page\": {\n            \"type\": \"integer\",\n            \"default\": 1,\n            \"minimum\": 1\n          },\n          \"page_size\": {\n            \"type\": \"integer\",\n            \"default\": 25,\n            \"minimum\": 1,\n            \"maximum\": 200\n          }\n        }\n      }\n    },\n    {\n      \"name\": \"read_content\",\n      \"description\": \"Read a range of a paper's original text addressed in Unicode code\\npoints (offset/limit count characters like Python len(), not bytes).\\nTypically used with a doc_id/offset returned by semantic_search to\\nexpand context (read more text before or after a chunk).\\nReturns: text fragment, bytes_returned (UTF-8 byte length of text, for\\nreference only), next_offset (code-point offset of the next fragment —\\npage with it, never with bytes_returned), more (boolean).\\nServer behaviour: limit above 524288 is silently clamped; omitting\\noffset returns the whole document ignoring limit — the SDKs / MCP\\nserver send offset=0 and limit=4096 by default, so pass offset\\nexplicitly when calling the HTTP API directly.\",\n      \"script\": \"scripts/read_content.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"properties\": {\n          \"doc_id\": {\n            \"type\": \"string\",\n            \"description\": \"文献 ID（来自 search_papers / semantic_search）。\"\n          },\n          \"offset\": {\n            \"type\": \"integer\",\n            \"format\": \"int64\",\n            \"default\": 0,\n            \"minimum\": 0,\n            \"description\": \"起始 Unicode 码点偏移，直接使用 semantic_search 返回的 offset。默认 0。\"\n          },\n          \"limit\": {\n            \"type\": \"integer\",\n            \"format\": \"int64\",\n            \"default\": 4096,\n            \"minimum\": 1,\n            \"maximum\": 524288,\n            \"description\": \"最多返回的 Unicode 码点数。默认 4096；服务端上限 524288（超出静默钳制）。LLM 场景建议 ≤ 16384，避免撑爆上下文。\"\n          }\n        },\n        \"required\": [\n          \"doc_id\"\n        ]\n      }\n    },\n    {\n      \"name\": \"get_resource\",\n      \"description\": \"Returns the binary bytes of a paper figure / table image referenced\\ninside read_content's Markdown via `![alt](file_name)` placeholders.\\nUse when the user asks to see / display / describe a figure and\\nread_content output contains an image reference.\\nInput file_name comes from the Markdown URL part (relative path,\\nno `\\\\\\\\` or `..`).\\nReturns: raw image stream + image/* Content-Type. The SDK / MCP\\nserver wraps the bytes as base64 + mimeType so Claude (multimodal)\\ncan read the image directly.\",\n      \"script\": \"scripts/get_resource.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"properties\": {\n          \"file_name\": {\n            \"type\": \"string\",\n            \"description\": \"图片相对路径，来自 read_content Markdown 中的 `![alt](file_name)` 占位。禁止 `\\\\\\\\` 与 `..`，不能以 `/` 开头。\"\n          }\n        },\n        \"required\": [\n          \"file_name\"\n        ]\n      }\n    }\n  ]\n}\n\nArchive v0.14.2: 12 files, 20213 bytes\n\nFiles: manifest.json (22993b), README.md (2969b), scripts/_common.mjs (3137b), scripts/get_resource.mjs (766b), scripts/list_catalog.mjs (465b), scripts/list_paper_relations.mjs (284b), scripts/read_content.mjs (355b), scripts/search_papers.mjs (259b), scripts/semantic_search.mjs (1192b), skill-card.md (2406b), SKILL.md (9354b), _meta.json (138b)\n\nFile v0.14.2:SKILL.md\n\n---\nname: sciverse-academic-retrieval\nslug: academic-retrieval\nversion: 0.14.2\ndescription: Sciverse academic paper retrieval: structured metadata search, semantic chunk retrieval for RAG, and character-range content reading (offsets in Unicode code points). For agent workflows that need citation-grade scientific literature.\nlicense: Apache-2.0\nhomepage: https://sciverse.space\n---\n\n# academic-retrieval\n\nSciverse academic paper retrieval: structured metadata search, semantic chunk retrieval for RAG, and character-range content reading (offsets in Unicode code points). For agent workflows that need citation-grade scientific literature.\n\n## When to use\n\nTrigger this skill when the user's request involves any of:\n\n- Locating academic papers by structured criteria (authors, year, journal, subjects)\n- Grounding answers in paper excerpts (RAG / citations)\n- Expanding the original text around a known doc_id (more text before/after a chunk)\n\n## Authentication\n\nThis skill requires the `SCIVERSE_API_TOKEN` environment variable\n(obtain from https://sciverse.space). Optionally set `SCIVERSE_BASE_URL`\nto override the default API base URL.\n\n## Tools\n\n### search_papers\n\nSearch academic papers by structured filters (title, authors, journal,\nyear, subjects, etc.).\nUse when: \"find Hinton's papers from 2020-2023\", \"Nature papers on\nCRISPR\".\nNot for: natural-language Q&A retrieval (use semantic_search) or\nfull-text snippets (use read_content).\nReturns: list of papers; each entry has unique_id (always present),\ndoc_id (only when full text exists), title, author, abstract,\npublication_venue_name_unified, publication_published_year.\n\n**Invoke**: `node scripts/search_papers.mjs '<JSON args>'`\n\n### semantic_search\n\nNatural-language semantic search returning relevant paper chunks for\nRAG-style answering.\nUse when: \"How does Transformer attention work?\", \"What are recent\nmethods for protein structure prediction?\".\nNot for: precise field filtering (use search_papers) or fetching full\noriginal text (use read_content).\nReturns: list of chunks; each entry has chunk_id, doc_id, abstract,\nchunk, score, title, offset.\nTypical chain: semantic_search → pick chunk → read_content(doc_id,\noffset).\n\n**Invoke**: `node scripts/semantic_search.mjs '<JSON args>'`\n\n### list_catalog\n\nReturns the schema catalog for search_papers: every field name, type,\nwhether it's filterable / sortable, default-return status, human\ndescription, and applicable FilterOperators.\nUse when: \"Which field do I filter by DOI?\", \"What values can\naccess_oa_status take?\", \"What's the right enum for metadata_type?\".\nNot for: actually searching papers (use search_papers / semantic_search).\nTypical pattern: call once when first encountering Sciverse or facing\nan ambiguous field need, then construct precise search_papers filters\nfrom the returned schema.\nPass include_sample_values=true to also fetch top-20 values for\nenum-like fields (OpenSearch terms aggregation, 24h cached).\n\n**Invoke**: `node scripts/list_catalog.mjs '<JSON args>'`\n\n### list_paper_relations\n\nPaginate the full relation list of a paper. citations/references/related_works\nare unbounded arrays (up to 340k entries for a single paper) and are NOT\nprojectable in search_papers, so this endpoint is the only way to read them.\nUse when: \"What does paper X cite?\" (relation=REFERENCES), \"Which papers cite\npaper X?\" (relation=CITATIONS), \"Works related to paper X\" (relation=RELATED_WORKS).\nNote: CITATIONS (incoming: who cites me) and REFERENCES (outgoing: who I cite)\nare opposite directions.\nTypical chain: get unique_id from search_papers / semantic_search, then paginate\nhere by relation.\nTwo limits (CITATIONS only; REFERENCES/RELATED_WORKS max out at 11833/20 in practice):\nmore than 10000 relations returns 429; page*page_size above 10000 returns 400.\nIn both cases switch to search_papers with filters_advanced on\nreferences_unique_id — it supports deep paging and arbitrary sorting.\ntotal_count counts in-corpus matches only, so it can differ from the paper's own\ncitation_count by about 1%.\n\n**Invoke**: `node scripts/list_paper_relations.mjs '<JSON args>'`\n\n### read_content\n\nRead a range of a paper's original text addressed in Unicode code\npoints (offset/limit count characters like Python len(), not bytes).\nTypically used with a doc_id/offset returned by semantic_search to\nexpand context (read more text before or after a chunk).\nReturns: text fragment, bytes_returned (UTF-8 byte length of text, for\nreference only), next_offset (code-point offset of the next fragment —\npage with it, never with bytes_returned), more (boolean).\nServer behaviour: limit above 524288 is silently clamped; omitting\noffset returns the whole document ignoring limit — the SDKs / MCP\nserver send offset=0 and limit=4096 by default, so pass offset\nexplicitly when calling the HTTP API directly.\n\n**Invoke**: `node scripts/read_content.mjs '<JSON args>'`\n\n### get_resource\n\nReturns the binary bytes of a paper figure / table image referenced\ninside read_content's Markdown via `![alt](file_name)` placeholders.\nUse when the user asks to see / display / describe a figure and\nread_content output contains an image reference.\nInput file_name comes from the Markdown URL part (relative path,\nno `\\\\` or `..`).\nReturns: raw image stream + image/* Content-Type. The SDK / MCP\nserver wraps the bytes as base64 + mimeType so Claude (multimodal)\ncan read the image directly.\n\n**Invoke**: `node scripts/get_resource.mjs '<JSON args>'`\n\n## Bootstrap: learn the schema first\n\nIf you're unsure which fields exist or what values an enum takes\n(e.g. `metadata_type`, `language`, `access_oa_status`), call\n`list_catalog` once at the start. Sample values are returned for\nlow-cardinality fields. Use it instead of guessing field names —\nguessing wastes turns.\n\n```\nlist_catalog(include_sample_values=true)\n    └─▶ fields[].name + sample_values  →  precise filter construction\n```\n\n## Recipes\n\n**RAG flow (natural-language Q&A):**\n\n```\nsemantic_search(query=...) → hits[i].doc_id, hits[i].offset\n    └─▶ read_content(doc_id, offset)\n```\n\n**Lookup by DOI:**\n\n```\nsearch_papers(filters_advanced=[{field: \"doi\", value: \"10.1038/...\"}])\n```\n\n**OA + year filter:**\n\n```\nsearch_papers(\n    year_from=2024,\n    filters_advanced=[{field: \"access_is_oa\", value: \"true\"}]\n)\n```\n\n**Scoped semantic search (constrained corpus):**\n\n```\nsemantic_search(\n    query=\"...\",\n    filters={\"author\": [\"Hinton\"],\n             \"publication_published_year\": {\"gte\": 2020}}\n)   # applied at recall time, server-side; AND across fields\n```\n\nSoft semantics: chunks missing that metadata are NOT excluded.\nFor a hard guarantee, or meta-only constraints (fwci, citation graph,\ncomplex hit-sets), scope by doc_id — a HARD recall-time filter:\n\n```\nsearch_papers(..., fields=[\"doc_id\",\"title\"]) → collect doc_id\nsemantic_search(query=..., filters={\"doc_id\": [...]})\n    # hits never leave the set; empty list → empty hits (never global);\n    # up to 1000 deduped ids (400 SCOPE_TOO_LARGE beyond)\n```\n\n**Bias fuzzy search ranking (soft boosts — stackable):**\n\nThree multiplicative boosts (`freshness_boost` / `impact_boost` /\n`language_affinity`, each NONE/MILD/STRONG) reorder fuzzy-search\nresults while keeping relevance. Only effective when `query` is\nnon-empty; ignored when any sort is set; shallow paging while active.\n`sort_by_year` defaults to `auto` (relevance with `query`, newest-first\nfor pure filters); `query`+`desc` is an anti-pattern — it degrades the\nquery to a match filter and disables all boosts; use `freshness_boost`.\n\n```\nsearch_papers(query=\"large language model\", freshness_boost=\"STRONG\")\n    # recent first: STRONG=3-year decay, MILD=10-year\nsearch_papers(query=\"protein folding\", impact_boost=\"MILD\")\n    # highly-cited float up (bounded; zero-citation stays neutral)\nsearch_papers(query=\"深度学习\", language_affinity=\"MILD\")\n    # demote (never exclude) results not in the query's language;\n    # unknown-language papers stay neutral; hard-exclude via\n    # filters_advanced=[{\"field\":\"language\",\"value\":\"zh\"}]\n```\n\n**Search authors or journals (collection):**\n\nSet `collection` to `authors` or `sources` (default `papers`) to search\nthose entities. Each has its own fields — call\nlist_catalog(collection=\"authors\") first; use filters_advanced +\nsort_advanced (papers convenience fields apply to papers only).\n\n```\nsearch_papers(collection=\"authors\",\n    filters_advanced=[{field: \"summary_stats.h_index\", operator: \"FILTER_OP_GTE\", value: 50}],\n    sort_advanced=[{field: \"cited_by_count\", order: \"SORT_ORDER_DESC\"}])\n```\n\n**Fetch a paper figure / image:**\n\nWhen read_content Markdown contains `![alt](file_name)`, call\n`get_resource` with the file_name to fetch image binary.\n\n```\nread_content(doc_id, offset) → markdown ![Figure 3](dt=xxx/p/f3.png)\n    └─▶ get_resource(file_name=\"dt=xxx/p/f3.png\")\n```\n\n**Reading fulltext (check first):**\n\nEach search_papers hit carries `is_content_accessible` (bool): `true` only when\nthe paper has fulltext AND the caller is authorized. Check it before\n`read_content(doc_id, ...)` — `false` means no fulltext or no read permission.\n\n## Exit codes\n\n- `0` — success; stdout is the JSON response\n- `1` — HTTP 4xx/5xx; stderr contains status code and response body\n- `2` — argument error (missing token, malformed JSON, required field absent)\n\nFile v0.14.2:README.md\n\n# academic-retrieval — ClawHub skill bundle\n\n[![ClawHub](https://img.shields.io/badge/clawhub-academic--retrieval-brightgreen)](https://clawhub.ai/sciverse/skills/academic-retrieval)\n\nClawHub skill that gives any OpenClaw agent Sciverse academic-paper retrieval\ncapabilities (English | [中文](#中文说明)).\n\nPublished by **@sciverse** (slug `academic-retrieval`).\n\n## Install\n\n```bash\nopenclaw skills install academic-retrieval\n```\n\n## Configure\n\n```bash\nexport SCIVERSE_API_TOKEN=sv-xxx       # obtain from https://sciverse.space\n```\n\n## Tools at a glance\n\n| Tool | Purpose |\n|---|---|\n| `list_catalog` | Field introspection (call once to learn available fields + enum values) |\n| `search_papers` | Structured metadata search over papers / authors / sources (set `collection`) |\n| `semantic_search` | Natural-language semantic chunk retrieval (for RAG) |\n| `read_content` | Character-range read of a paper's original text (offset/limit in Unicode code points) |\n| `get_resource` | Fetch figure / table image bytes referenced inside `read_content` Markdown |\n\nSee `SKILL.md` for full agent-facing documentation.\n\n## Direct invocation (bypass OpenClaw)\n\n```bash\nnode scripts/semantic_search.mjs '{\"query\":\"Transformer attention mechanism\",\"top_k\":3}'\n```\n\n## Relationship to the SDK\n\nThis skill is **complementary** to the `sciverse` packages on PyPI / npm:\n\n- **This skill** — OpenClaw users only. Zero external deps (Node 18+ native fetch).\n- **PyPI / npm SDK** — Any LLM agent framework (OpenAI, Anthropic, LangChain, LlamaIndex…).\n\n## License\n\nApache-2.0\n\n---\n\n## 中文说明\n\nOpenClaw 用户专用：通过 ClawHub 一键给 agent 加上 Sciverse 学术文献检索能力。\n\n发布者 **@sciverse**，slug `academic-retrieval`。\n\n### 安装\n\n```bash\nopenclaw skills install academic-retrieval\n```\n\n### 配置\n\n```bash\nexport SCIVERSE_API_TOKEN=sv-xxx   # 从 https://sciverse.space 控制台申请\n# 可选：export SCIVERSE_BASE_URL=https://api-custom.sciverse.space\n```\n\n### 工具速览\n\n| Tool | 用途 |\n|---|---|\n| `list_catalog` | 字段 introspection（首次接入调一次，学习可用字段和 enum 取值） |\n| `search_papers` | 按结构化条件查 papers / authors / sources（用 `collection` 切换实体集合） |\n| `semantic_search` | 自然语言语义检索文献片段（RAG 用） |\n| `read_content` | 按 Unicode 码点区间读取文献原文片段 |\n| `get_resource` | 取 `read_content` Markdown 中引用的图片字节流（多模态 RAG） |\n\nagent 视角的完整文档见 `SKILL.md`（英文）。\n\n### 直接调用（不通过 OpenClaw）\n\n```bash\nnode scripts/semantic_search.mjs '{\"query\":\"Transformer 注意力机制\",\"top_k\":3}'\n```\n\n### 与 SDK 的关系\n\n本 skill 与 PyPI/npm 上的 `sciverse` 包是**互补**的：\n\n- **本 skill**：OpenClaw 用户专用，零外部依赖（仅 Node 18+ native fetch）\n- **PyPI/npm SDK**：任意 LLM Agent 框架（OpenAI / Anthropic / LangChain / LlamaIndex...）\n\nFile v0.14.2:_meta.json\n\n{\n  \"ownerId\": \"kn74way11x0gjn6wpa8hcvyhvs85vmkj\",\n  \"slug\": \"academic-retrieval\",\n  \"version\": \"0.14.2\",\n  \"publishedAt\": 1789095868277\n}\n\nFile v0.14.2:skill-card.md\n\n## Description:\n\nRetrieves academic papers by structured metadata, performs semantic chunk retrieval for RAG, and reads character-range content for citation-grade scientific literature.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[sciverse](https://clawhub.ai/user/sciverse)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and external agent users use this skill to find scholarly papers, retrieve semantically relevant excerpts, expand context around known document offsets, inspect paper relationships, and fetch referenced paper images for citation-grounded workflows.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: A custom SCIVERSE_BASE_URL using http:// can send the Sciverse API token without TLS.\n\nMitigation: Use the default https://api.sciverse.space endpoint or an HTTPS-only sciverse.space host, and rotate the token if it may have been sent over plaintext HTTP.\n\nRisk: Semantic search filters are documented as soft constraints when chunk metadata is missing, so filtered results may include papers outside the intended scope.\n\nMitigation: Use search_papers for hard metadata filtering or scope semantic_search by an explicit doc_id set when strict filtering is required.\n\nRisk: Full-text access depends on document availability and authorization.\n\nMitigation: Check is_content_accessible before calling read_content and ground citations only in returned metadata or text fragments.\n\n## Reference(s):\n\n- [Sciverse](https://sciverse.space)\n- [ClawHub Skill Page](https://clawhub.ai/sciverse/skills/academic-retrieval)\n\n## Skill Output:\n\n**Output Type(s):** [API Calls, JSON, Text, Markdown, Guidance]\n\n**Output Format:** [JSON responses containing paper metadata, text or Markdown content fragments, relation lists, and optional base64-encoded image resources.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Requires SCIVERSE_API_TOKEN. The optional SCIVERSE_BASE_URL should remain HTTPS and on a sciverse.space host.]\n\n## Skill Version(s):\n\n0.14.2 (source: evidence.json release.version, SKILL.md frontmatter, manifest.json)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v0.14.2:manifest.json\n\n{\n  \"name\": \"sciverse-academic-retrieval\",\n  \"version\": \"0.14.2\",\n  \"slug\": \"academic-retrieval\",\n  \"description\": \"Sciverse academic paper retrieval: structured metadata search, semantic chunk retrieval for RAG, and character-range content reading (offsets in Unicode code points). For agent workflows that need citation-grade scientific literature.\",\n  \"runtime\": \"node>=18\",\n  \"license\": \"Apache-2.0\",\n  \"homepage\": \"https://sciverse.space\",\n  \"env\": [\n    {\n      \"name\": \"SCIVERSE_API_TOKEN\",\n      \"required\": true,\n      \"description\": \"Sciverse API Token (obtain from https://sciverse.space).\"\n    },\n    {\n      \"name\": \"SCIVERSE_BASE_URL\",\n      \"required\": false,\n      \"default\": \"https://api.sciverse.space\",\n      \"description\": \"Override the default API base URL (for dev / self-hosted gateways).\"\n    }\n  ],\n  \"tools\": [\n    {\n      \"name\": \"search_papers\",\n      \"description\": \"Search academic papers by structured filters (title, authors, journal,\\nyear, subjects, etc.).\\nUse when: \\\"find Hinton's papers from 2020-2023\\\", \\\"Nature papers on\\nCRISPR\\\".\\nNot for: natural-language Q&A retrieval (use semantic_search) or\\nfull-text snippets (use read_content).\\nReturns: list of papers; each entry has unique_id (always present),\\ndoc_id (only when full text exists), title, author, abstract,\\npublication_venue_name_unified, publication_published_year.\",\n      \"script\": \"scripts/search_papers.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"properties\": {\n          \"collection\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"papers\",\n              \"authors\",\n              \"sources\"\n            ],\n            \"default\": \"papers\",\n            \"description\": \"检索的实体集合。papers（默认，论文）/ authors（作者）/ sources（来源期刊）。 各 collection 字段集不同，用 list_catalog（collection=<name>）学习对应 schema。 注意：本工具的便捷字段（authors/journals/year_from/subjects 等）只对 papers 有意义； 查 authors/sources 时改用 filters_advanced + 该 collection 的字段名（如 authors 的 summary_stats.h_index / orcid，sources 的 issn / is_oa）。authors 用 orcid、 sources 用 issn 与论文检索结果关联。\",\n            \"x-en-description\": \"Entity collection to search. papers (default) / authors / sources. Each collection has its own field schema — call list_catalog(collection=<name>). The convenience fields (authors/journals/year_from/subjects) apply to papers only; for authors/sources use filters_advanced with that collection's field names.\"\n          },\n          \"query\": {\n            \"type\": \"string\",\n            \"description\": \"BM25 全文关键词，匹配标题/摘要/期刊名/关键词字段。留空则纯靠结构化过滤。\"\n          },\n          \"title_contains\": {\n            \"type\": \"string\",\n            \"description\": \"标题中必须包含的词（仅匹配 title 字段）。\"\n          },\n          \"abstract_contains\": {\n            \"type\": \"string\",\n            \"description\": \"摘要中必须包含的词（仅匹配 abstract 字段）。\"\n          },\n          \"authors\": {\n            \"type\": \"array\",\n            \"items\": {\n              \"type\": \"string\"\n            },\n            \"description\": \"作者名（任一命中即可）。SDK 内部映射到后端 `author` 字段（FILTER_OP_IN）。\"\n          },\n          \"year_from\": {\n            \"type\": \"integer\",\n            \"description\": \"起始发表年（含）。\"\n          },\n          \"year_to\": {\n            \"type\": \"integer\",\n            \"description\": \"结束发表年（含）。\"\n          },\n          \"journals\": {\n            \"type\": \"array\",\n            \"items\": {\n              \"type\": \"string\"\n            },\n            \"description\": \"期刊名（任一命中即可）。SDK 内部映射到后端 `publication_venue_name_unified` 字段（FILTER_OP_IN，规范化后的载体名）。\"\n          },\n          \"subjects\": {\n            \"type\": \"array\",\n            \"items\": {\n              \"type\": \"string\"\n            },\n            \"description\": \"学科分类，如 \\\"computer science\\\"、\\\"biology\\\"。\"\n          },\n          \"filters_advanced\": {\n            \"type\": \"array\",\n            \"description\": \"高级过滤逃生舱（仅当上述字段不够用时使用）。可用字段见 get_field_catalog。\\n\\n引文反查（常用）：field=\\\"references_unique_id\\\" 查「谁引用了某篇论文」，\\nvalue 填目标论文的 unique_id。相比 list_paper_relations 的 CITATIONS，\\n它支持深翻页与任意排序，适合超高被引论文。可叠加条件，\\n例如「引用了 ResNet 且 2023 年后发表」：\\n  [{\\\"field\\\":\\\"references_unique_id\\\",\\\"value\\\":\\\"paper:10.1109/cvpr.2016.90\\\"},\\n   {\\\"field\\\":\\\"publication_published_year\\\",\\\"operator\\\":\\\"FILTER_OP_GTE\\\",\\\"value\\\":2023}]\\n该字段仅支持过滤，不能排序/聚合，也不能放进 fields 返回。\\n\",\n            \"items\": {\n              \"type\": \"object\",\n              \"required\": [\n                \"field\",\n                \"value\"\n              ],\n              \"properties\": {\n                \"field\": {\n                  \"type\": \"string\"\n                },\n                \"operator\": {\n                  \"type\": \"string\",\n                  \"description\": \"过滤操作符。MATCH（分词模糊）适用于 author、keywords（输入 \\\"Hinton\\\" 命中 \\\"Geoffrey Hinton\\\"）； MATCH_PHRASE（短语模糊）适用于 publication_venue_name_unified，整词连续匹配（\\\"Nature\\\" 命中 \\\"Nature Communications\\\"；非前缀匹配，\\\"Nature Comm\\\" 不会命中）； doi 用 EQ，服务端归一化（去 doi.org 前缀+转小写）后精确匹配。MATCH/MATCH_PHRASE 仅对配了 text 子字段的字段有效。\",\n                  \"enum\": [\n                    \"FILTER_OP_EQ\",\n                    \"FILTER_OP_NE\",\n                    \"FILTER_OP_GT\",\n                    \"FILTER_OP_GTE\",\n                    \"FILTER_OP_LT\",\n                    \"FILTER_OP_LTE\",\n                    \"FILTER_OP_IN\",\n                    \"FILTER_OP_NIN\",\n                    \"FILTER_OP_CONTAINS\",\n                    \"FILTER_OP_MATCH\",\n                    \"FILTER_OP_MATCH_PHRASE\"\n                  ],\n                  \"default\": \"FILTER_OP_EQ\"\n                },\n                \"value\": {}\n              }\n            }\n          },\n          \"sort_advanced\": {\n            \"type\": \"array\",\n            \"description\": \"高级排序逃生舱（按任意可排序字段）。papers 用 sort_by_year 即可； authors/sources 想按 h-index / 被引 / works_count 排序时用本字段。 与 query 互斥（query 走相关性排序）。\",\n            \"items\": {\n              \"type\": \"object\",\n              \"required\": [\n                \"field\",\n                \"order\"\n              ],\n              \"properties\": {\n                \"field\": {\n                  \"type\": \"string\"\n                },\n                \"order\": {\n                  \"type\": \"string\",\n                  \"enum\": [\n                    \"SORT_ORDER_DESC\",\n                    \"SORT_ORDER_ASC\"\n                  ],\n                  \"default\": \"SORT_ORDER_DESC\"\n                }\n              }\n            }\n          },\n          \"sort_by_year\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"auto\",\n              \"desc\",\n              \"asc\",\n              \"none\"\n            ],\n            \"default\": \"auto\",\n            \"description\": \"按发表年份排序。默认 auto：传了 query（或 sort_advanced）时不加年份排序\\n——保留 BM25 相关性排序，且 freshness/impact/language_affinity 软加权可用；\\n纯结构化筛选（无 query）时按年份降序（否则后端默认序是 unique_id，实质乱序）。\\n⚠️ 不要用 query + desc 求「最相关且最新」：显式排序会让 query 退化为命中\\n过滤（OR 语义、无相关性排序）、三个软加权全部失效——返回的是「含任一关键词\\n的最新文档」。要「相关且偏新」请用 freshness_boost。\\n\"\n          },\n          \"freshness_boost\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"NONE\",\n              \"MILD\",\n              \"STRONG\"\n            ],\n            \"default\": \"NONE\",\n            \"description\": \"模糊搜索新鲜度加权：结果偏向新文献（仅 query 非空时生效；传排序\\n（sort_by_year 非 none / sort_advanced）时被忽略，硬排优先）。\\nMILD: 近 10 年加权，适合日常查文献；STRONG: 近 3 年加权，适合跟踪\\n研究方向 / 追最新进展。与 impact_boost / language_affinity 可叠加\\n（均为乘法因子）。boost 生效时为浅翻页：不产 next_cursor、不支持\\ncursor 深翻页。\\n\"\n          },\n          \"impact_boost\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"NONE\",\n              \"MILD\",\n              \"STRONG\"\n            ],\n            \"default\": \"NONE\",\n            \"description\": \"模糊搜索影响力加权：高被引文献在保留相关性的前提下上浮（仅 query\\n非空时生效；传排序时被忽略）。MILD: 轻度上浮，相关性仍主导；\\nSTRONG: 明显偏向高被引。引用因子有界、零被引中性（不会归零）。\\n与 freshness_boost / language_affinity 可叠加；boost 生效时为浅翻页。\\n\"\n          },\n          \"language_affinity\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"NONE\",\n              \"MILD\",\n              \"STRONG\"\n            ],\n            \"default\": \"NONE\",\n            \"description\": \"模糊搜索语言亲和加权：非 query 语言的结果降序、但不排除（仅 query\\n非空时生效；传排序时被忽略）。目标语言由服务端从 query 文本判定\\n（假名→ja / 谚文→ko / 汉字→zh / 拉丁→en，其他书写系统不生效）；\\n语言未知的文献保持中性不降权。MILD: 非目标语言 ×0.5，跨语言强相关\\n结果仍可上浮；STRONG: ×0.2，几乎只看目标语言。与 freshness_boost /\\nimpact_boost 可叠加；boost 生效时为浅翻页。要硬排除某语言请改用\\nfilters_advanced 的 language 字段（如 {\\\"field\\\":\\\"language\\\",\\\"value\\\":\\\"en\\\"}，\\n软硬两层语义不同：本参数只调序，filter 直接排除）。\\n\"\n          },\n          \"page\": {\n            \"type\": \"integer\",\n            \"default\": 1,\n            \"minimum\": 1\n          },\n          \"page_size\": {\n            \"type\": \"integer\",\n            \"default\": 25,\n            \"minimum\": 1,\n            \"maximum\": 50,\n            \"description\": \"每页条数。省略时为服务端默认 25（SDK / MCP 不注入默认值）；工具层上限 50。\"\n          }\n        }\n      }\n    },\n    {\n      \"name\": \"semantic_search\",\n      \"description\": \"Natural-language semantic search returning relevant paper chunks for\\nRAG-style answering.\\nUse when: \\\"How does Transformer attention work?\\\", \\\"What are recent\\nmethods for protein structure prediction?\\\".\\nNot for: precise field filtering (use search_papers) or fetching full\\noriginal text (use read_content).\\nReturns: list of chunks; each entry has chunk_id, doc_id, abstract,\\nchunk, score, title, offset.\\nTypical chain: semantic_search → pick chunk → read_content(doc_id,\\noffset).\",\n      \"script\": \"scripts/semantic_search.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"required\": [\n          \"query\"\n        ],\n        \"properties\": {\n          \"query\": {\n            \"type\": \"string\",\n            \"minLength\": 1,\n            \"maxLength\": 4096,\n            \"description\": \"自然语言查询，1-200 字最佳。\"\n          },\n          \"top_k\": {\n            \"type\": \"integer\",\n            \"default\": 10,\n            \"minimum\": 1,\n            \"maximum\": 100,\n            \"description\": \"返回命中条数上限，合法 1-100（服务端校验，超出报 400）。\\n实际条数还受 mode 影响：balanced 单路混合召回在服务端固定截到约 50 条，\\ntop_k 超过 50 时多出的部分不会返回；fast 与 quality 可取到 top_k。\\n另外同一篇论文最多返回约 3 个 chunk，因此高 top_k 需要命中足够多的不同论文。\\n\"\n          },\n          \"source_types\": {\n            \"type\": \"array\",\n            \"items\": {\n              \"type\": \"string\",\n              \"enum\": [\n                \"web\",\n                \"pdf\"\n              ]\n            }\n          },\n          \"filters\": {\n            \"type\": \"object\",\n            \"description\": \"结构化过滤（可选）。在召回阶段与语义检索同时生效（ES+Milvus 双引擎下推，\\n不是结果后过滤）；多个字段之间 AND，同一字段传数组时数组内 OR。\\n⚠️ 宽松（软）语义：chunk 侧元数据缺失的文档不会被排除——例如按年份过滤时，\\n缺年份信息的 chunk 仍可能返回。需要严格范围保证时勿当硬约束使用，\\n表述结论时注明范围为\\\"近似过滤\\\"。\\n数值/日期字段支持区间：{\\\"gte\\\":2020,\\\"lte\\\":2025} 或 [min,max]（null 表示一侧不限）；\\n日期接受 YYYY / YYYY-MM / YYYY-MM-DD。\\n实际可用字段受账号字段权限约束；未知字段服务端返回 400。\\n例：{\\\"author\\\":[\\\"Hinton\\\"],\\\"publication_published_year\\\":{\\\"gte\\\":2023},\\n    \\\"topics\\\":{\\\"dimensions\\\":{\\\"primary_topic_domain\\\":\\\"Health Sciences\\\"}}}\\n\",\n            \"properties\": {\n              \"lang\": {\n                \"description\": \"语言代码，如 \\\"en\\\"、\\\"zh\\\"；也接受别名 language。\"\n              },\n              \"metadata_type\": {\n                \"description\": \"资源类型，仅单值：\\\"paper\\\" 或 \\\"ebook\\\"。\"\n              },\n              \"author\": {\n                \"description\": \"作者名，string 或 string[]（数组=任一命中）。\"\n              },\n              \"publication_venue_name_unified\": {\n                \"description\": \"发表载体名称（期刊/会议，规范化名，适合精确匹配）。\"\n              },\n              \"publication_venue_type\": {\n                \"description\": \"载体类型：\\\"journal\\\"、\\\"conference\\\"、\\\"repository\\\"、\\\"book series\\\"、\\\"ebook platform\\\"、\\\"metadata\\\"、\\\"raidRegistry\\\"、\\\"igsnCatalog\\\"、\\\"other\\\"（不区分大小写）。\"\n              },\n              \"publication_published_year\": {\n                \"description\": \"发表年份，单值或区间（{\\\"gte\\\":..,\\\"lte\\\":..} / [min,max]）。\"\n              },\n              \"publication_published_date\": {\n                \"description\": \"发表日期，\\\"YYYY[-MM[-DD]]\\\" 单值或区间。\"\n              },\n              \"citation_count\": {\n                \"description\": \"被引次数，单值或区间。\"\n              },\n              \"influential_citation_count\": {\n                \"description\": \"高影响力被引次数，单值或区间。\"\n              },\n              \"title\": {\n                \"description\": \"标题精确匹配（标题检索一般更适合 search_papers）。\"\n              },\n              \"topics\": {\n                \"description\": \"主题组合过滤：{\\\"logic\\\":\\\"and|or\\\",\\\"dimensions\\\":{\\\"primary_topic\\\":\\\"...\\\",\\\"primary_topic_domain\\\":\\\"Physical Sciences|Social Sciences|Health Sciences|Life Sciences\\\"}}；logic 省略默认 or。\"\n              },\n              \"doc_id\": {\n                \"description\": \"唯一的硬约束字段（其余字段均为软语义）：命中绝不越出给定集合。\\n值为 64 位小写 hex sha256（即 search_papers 返回的 doc_id；仅有全文的论文才有），\\nstring 或 string[]，仅 eq/in。去重后上限默认 1000，超限返回 400 SCOPE_TOO_LARGE；\\n显式传空数组返回 200 空 hits（候选集为空，不退化为全局检索）。\\n典型用法：先 search_papers 圈定候选集合，再在集合内做受限语义检索。\\n\"\n              }\n            }\n          },\n          \"mode\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"fast\",\n              \"balanced\",\n              \"quality\"\n            ],\n            \"default\": \"balanced\",\n            \"description\": \"fast = 仅关键词召回 (~200ms)；balanced = 混合检索 (~600ms)；quality = LLM 改写 + 混合 (~2-4s)。\\n\"\n          }\n        }\n      }\n    },\n    {\n      \"name\": \"list_catalog\",\n      \"description\": \"Returns the schema catalog for search_papers: every field name, type,\\nwhether it's filterable / sortable, default-return status, human\\ndescription, and applicable FilterOperators.\\nUse when: \\\"Which field do I filter by DOI?\\\", \\\"What values can\\naccess_oa_status take?\\\", \\\"What's the right enum for metadata_type?\\\".\\nNot for: actually searching papers (use search_papers / semantic_search).\\nTypical pattern: call once when first encountering Sciverse or facing\\nan ambiguous field need, then construct precise search_papers filters\\nfrom the returned schema.\\nPass include_sample_values=true to also fetch top-20 values for\\nenum-like fields (OpenSearch terms aggregation, 24h cached).\",\n      \"script\": \"scripts/list_catalog.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"properties\": {\n          \"collection\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"papers\",\n              \"authors\",\n              \"sources\"\n            ],\n            \"default\": \"papers\",\n            \"description\": \"字段 catalog 所属实体集合。papers（默认）/ authors / sources，各 collection 字段不同。\"\n          },\n          \"include_sample_values\": {\n            \"type\": \"boolean\",\n            \"default\": false,\n            \"description\": \"是否拉取 enum-like 字段的取值样本。false 仅返回静态 schema（毫秒级）；true 触发 OpenSearch terms agg（首次几百毫秒，之后 24h 走缓存）。\"\n          },\n          \"include_field_stats\": {\n            \"type\": \"boolean\",\n            \"default\": false,\n            \"description\": \"是否返回字段统计（keyword 字段基数 + 数值字段 min/max/avg/p50/p95）。触发 OpenSearch 聚合，缓存 24h。\"\n          }\n        },\n        \"required\": []\n      }\n    },\n    {\n      \"name\": \"list_paper_relations\",\n      \"description\": \"Paginate the full relation list of a paper. citations/references/related_works\\nare unbounded arrays (up to 340k entries for a single paper) and are NOT\\nprojectable in search_papers, so this endpoint is the only way to read them.\\nUse when: \\\"What does paper X cite?\\\" (relation=REFERENCES), \\\"Which papers cite\\npaper X?\\\" (relation=CITATIONS), \\\"Works related to paper X\\\" (relation=RELATED_WORKS).\\nNote: CITATIONS (incoming: who cites me) and REFERENCES (outgoing: who I cite)\\nare opposite directions.\\nTypical chain: get unique_id from search_papers / semantic_search, then paginate\\nhere by relation.\\nTwo limits (CITATIONS only; REFERENCES/RELATED_WORKS max out at 11833/20 in practice):\\nmore than 10000 relations returns 429; page*page_size above 10000 returns 400.\\nIn both cases switch to search_papers with filters_advanced on\\nreferences_unique_id — it supports deep paging and arbitrary sorting.\\ntotal_count counts in-corpus matches only, so it can differ from the paper's own\\ncitation_count by about 1%.\",\n      \"script\": \"scripts/list_paper_relations.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"required\": [\n          \"unique_id\",\n          \"relation\"\n        ],\n        \"properties\": {\n          \"unique_id\": {\n            \"type\": \"string\",\n            \"description\": \"目标论文 unique_id（如 paper:10.1038/xxx），来自 search_papers / semantic_search；勿传 doc_id。\"\n          },\n          \"relation\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"CITATIONS\",\n              \"REFERENCES\",\n              \"RELATED_WORKS\"\n            ],\n            \"description\": \"关系类型。CITATIONS=被引（谁引用了我）；REFERENCES=参考文献（我引用了谁）；RELATED_WORKS=相关工作。\"\n          },\n          \"page\": {\n            \"type\": \"integer\",\n            \"default\": 1,\n            \"minimum\": 1\n          },\n          \"page_size\": {\n            \"type\": \"integer\",\n            \"default\": 25,\n            \"minimum\": 1,\n            \"maximum\": 200\n          }\n        }\n      }\n    },\n    {\n      \"name\": \"read_content\",\n      \"description\": \"Read a range of a paper's original text addressed in Unicode code\\npoints (offset/limit count characters like Python len(), not bytes).\\nTypically used with a doc_id/offset returned by semantic_search to\\nexpand context (read more text before or after a chunk).\\nReturns: text fragment, bytes_returned (UTF-8 byte length of text, for\\nreference only), next_offset (code-point offset of the next fragment —\\npage with it, never with bytes_returned), more (boolean).\\nServer behaviour: limit above 524288 is silently clamped; omitting\\noffset returns the whole document ignoring limit — the SDKs / MCP\\nserver send offset=0 and limit=4096 by default, so pass offset\\nexplicitly when calling the HTTP API directly.\",\n      \"script\": \"scripts/read_content.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"properties\": {\n          \"doc_id\": {\n            \"type\": \"string\",\n            \"description\": \"文献 ID（来自 search_papers / semantic_search）。\"\n          },\n          \"offset\": {\n            \"type\": \"integer\",\n            \"format\": \"int64\",\n            \"default\": 0,\n            \"minimum\": 0,\n            \"description\": \"起始 Unicode 码点偏移，直接使用 semantic_search 返回的 offset。默认 0。\"\n          },\n          \"limit\": {\n            \"type\": \"integer\",\n            \"format\": \"int64\",\n            \"default\": 4096,\n            \"minimum\": 1,\n            \"maximum\": 524288,\n            \"description\": \"最多返回的 Unicode 码点数。默认 4096；服务端上限 524288（超出静默钳制）。LLM 场景建议 ≤ 16384，避免撑爆上下文。\"\n          }\n        },\n        \"required\": [\n          \"doc_id\"\n        ]\n      }\n    },\n    {\n      \"name\": \"get_resource\",\n      \"description\": \"Returns the binary bytes of a paper figure / table image referenced\\ninside read_content's Markdown via `![alt](file_name)` placeholders.\\nUse when the user asks to see / display / describe a figure and\\nread_content output contains an image reference.\\nInput file_name comes from the Markdown URL part (relative path,\\nno `\\\\\\\\` or `..`).\\nReturns: raw image stream + image/* Content-Type. The SDK / MCP\\nserver wraps the bytes as base64 + mimeType so Claude (multimodal)\\ncan read the image directly.\",\n      \"script\": \"scripts/get_resource.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"properties\": {\n          \"file_name\": {\n            \"type\": \"string\",\n            \"description\": \"图片相对路径，来自 read_content Markdown 中的 `![alt](file_name)` 占位。禁止 `\\\\\\\\` 与 `..`，不能以 `/` 开头。\"\n          }\n        },\n        \"required\": [\n          \"file_name\"\n        ]\n      }\n    }\n  ]\n}\n\nArchive v0.14.1: 12 files, 20131 bytes\n\nFiles: manifest.json (22955b), README.md (2926b), scripts/_common.mjs (3137b), scripts/get_resource.mjs (766b), scripts/list_catalog.mjs (465b), scripts/list_paper_relations.mjs (284b), scripts/read_content.mjs (355b), scripts/search_papers.mjs (259b), scripts/semantic_search.mjs (1192b), skill-card.md (2405b), SKILL.md (9278b), _meta.json (138b)\n\nFile v0.14.1:SKILL.md\n\n---\nname: sciverse-academic-retrieval\nslug: academic-retrieval\nversion: 0.14.1\ndescription: Sciverse academic paper retrieval: structured metadata search, semantic chunk retrieval for RAG, and byte-range content reading. For agent workflows that need citation-grade scientific literature.\nlicense: Apache-2.0\nhomepage: https://sciverse.space\n---\n\n# academic-retrieval\n\nSciverse academic paper retrieval: structured metadata search, semantic chunk retrieval for RAG, and byte-range content reading. For agent workflows that need citation-grade scientific literature.\n\n## When to use\n\nTrigger this skill when the user's request involves any of:\n\n- Locating academic papers by structured criteria (authors, year, journal, subjects)\n- Grounding answers in paper excerpts (RAG / citations)\n- Expanding the original text around a known doc_id (more text before/after a chunk)\n\n## Authentication\n\nThis skill requires the `SCIVERSE_API_TOKEN` environment variable\n(obtain from https://sciverse.space). Optionally set `SCIVERSE_BASE_URL`\nto override the default API base URL.\n\n## Tools\n\n### search_papers\n\nSearch academic papers by structured filters (title, authors, journal,\nyear, subjects, etc.).\nUse when: \"find Hinton's papers from 2020-2023\", \"Nature papers on\nCRISPR\".\nNot for: natural-language Q&A retrieval (use semantic_search) or\nfull-text snippets (use read_content).\nReturns: list of papers; each entry has unique_id (always present),\ndoc_id (only when full text exists), title, author, abstract,\npublication_venue_name_unified, publication_published_year.\n\n**Invoke**: `node scripts/search_papers.mjs '<JSON args>'`\n\n### semantic_search\n\nNatural-language semantic search returning relevant paper chunks for\nRAG-style answering.\nUse when: \"How does Transformer attention work?\", \"What are recent\nmethods for protein structure prediction?\".\nNot for: precise field filtering (use search_papers) or fetching full\noriginal text (use read_content).\nReturns: list of chunks; each entry has chunk_id, doc_id, abstract,\nchunk, score, title, offset.\nTypical chain: semantic_search → pick chunk → read_content(doc_id,\noffset).\n\n**Invoke**: `node scripts/semantic_search.mjs '<JSON args>'`\n\n### list_catalog\n\nReturns the schema catalog for search_papers: every field name, type,\nwhether it's filterable / sortable, default-return status, human\ndescription, and applicable FilterOperators.\nUse when: \"Which field do I filter by DOI?\", \"What values can\naccess_oa_status take?\", \"What's the right enum for metadata_type?\".\nNot for: actually searching papers (use search_papers / semantic_search).\nTypical pattern: call once when first encountering Sciverse or facing\nan ambiguous field need, then construct precise search_papers filters\nfrom the returned schema.\nPass include_sample_values=true to also fetch top-20 values for\nenum-like fields (OpenSearch terms aggregation, 24h cached).\n\n**Invoke**: `node scripts/list_catalog.mjs '<JSON args>'`\n\n### list_paper_relations\n\nPaginate the full relation list of a paper. citations/references/related_works\nare unbounded arrays (up to 340k entries for a single paper) and are NOT\nprojectable in search_papers, so this endpoint is the only way to read them.\nUse when: \"What does paper X cite?\" (relation=REFERENCES), \"Which papers cite\npaper X?\" (relation=CITATIONS), \"Works related to paper X\" (relation=RELATED_WORKS).\nNote: CITATIONS (incoming: who cites me) and REFERENCES (outgoing: who I cite)\nare opposite directions.\nTypical chain: get unique_id from search_papers / semantic_search, then paginate\nhere by relation.\nTwo limits (CITATIONS only; REFERENCES/RELATED_WORKS max out at 11833/20 in practice):\nmore than 10000 relations returns 429; page*page_size above 10000 returns 400.\nIn both cases switch to search_papers with filters_advanced on\nreferences_unique_id — it supports deep paging and arbitrary sorting.\ntotal_count counts in-corpus matches only, so it can differ from the paper's own\ncitation_count by about 1%.\n\n**Invoke**: `node scripts/list_paper_relations.mjs '<JSON args>'`\n\n### read_content\n\nRead a range of a paper's original text addressed in Unicode code\npoints (offset/limit count characters like Python len(), not bytes).\nTypically used with a doc_id/offset returned by semantic_search to\nexpand context (read more text before or after a chunk).\nReturns: text fragment, bytes_returned (UTF-8 byte length of text, for\nreference only), next_offset (code-point offset of the next fragment —\npage with it, never with bytes_returned), more (boolean).\nServer behaviour: limit above 524288 is silently clamped; omitting\noffset returns the whole document ignoring limit — the SDKs / MCP\nserver send offset=0 and limit=4096 by default, so pass offset\nexplicitly when calling the HTTP API directly.\n\n**Invoke**: `node scripts/read_content.mjs '<JSON args>'`\n\n### get_resource\n\nReturns the binary bytes of a paper figure / table image referenced\ninside read_content's Markdown via `![alt](file_name)` placeholders.\nUse when the user asks to see / display / describe a figure and\nread_content output contains an image reference.\nInput file_name comes from the Markdown URL part (relative path,\nno `\\\\` or `..`).\nReturns: raw image stream + image/* Content-Type. The SDK / MCP\nserver wraps the bytes as base64 + mimeType so Claude (multimodal)\ncan read the image directly.\n\n**Invoke**: `node scripts/get_resource.mjs '<JSON args>'`\n\n## Bootstrap: learn the schema first\n\nIf you're unsure which fields exist or what values an enum takes\n(e.g. `metadata_type`, `language`, `access_oa_status`), call\n`list_catalog` once at the start. Sample values are returned for\nlow-cardinality fields. Use it instead of guessing field names —\nguessing wastes turns.\n\n```\nlist_catalog(include_sample_values=true)\n    └─▶ fields[].name + sample_values  →  precise filter construction\n```\n\n## Recipes\n\n**RAG flow (natural-language Q&A):**\n\n```\nsemantic_search(query=...) → hits[i].doc_id, hits[i].offset\n    └─▶ read_content(doc_id, offset)\n```\n\n**Lookup by DOI:**\n\n```\nsearch_papers(filters_advanced=[{field: \"doi\", value: \"10.1038/...\"}])\n```\n\n**OA + year filter:**\n\n```\nsearch_papers(\n    year_from=2024,\n    filters_advanced=[{field: \"access_is_oa\", value: \"true\"}]\n)\n```\n\n**Scoped semantic search (constrained corpus):**\n\n```\nsemantic_search(\n    query=\"...\",\n    filters={\"author\": [\"Hinton\"],\n             \"publication_published_year\": {\"gte\": 2020}}\n)   # applied at recall time, server-side; AND across fields\n```\n\nSoft semantics: chunks missing that metadata are NOT excluded.\nFor a hard guarantee, or meta-only constraints (fwci, citation graph,\ncomplex hit-sets), scope by doc_id — a HARD recall-time filter:\n\n```\nsearch_papers(..., fields=[\"doc_id\",\"title\"]) → collect doc_id\nsemantic_search(query=..., filters={\"doc_id\": [...]})\n    # hits never leave the set; empty list → empty hits (never global);\n    # up to 1000 deduped ids (400 SCOPE_TOO_LARGE beyond)\n```\n\n**Bias fuzzy search ranking (soft boosts — stackable):**\n\nThree multiplicative boosts (`freshness_boost` / `impact_boost` /\n`language_affinity`, each NONE/MILD/STRONG) reorder fuzzy-search\nresults while keeping relevance. Only effective when `query` is\nnon-empty; ignored when any sort is set; shallow paging while active.\n`sort_by_year` defaults to `auto` (relevance with `query`, newest-first\nfor pure filters); `query`+`desc` is an anti-pattern — it degrades the\nquery to a match filter and disables all boosts; use `freshness_boost`.\n\n```\nsearch_papers(query=\"large language model\", freshness_boost=\"STRONG\")\n    # recent first: STRONG=3-year decay, MILD=10-year\nsearch_papers(query=\"protein folding\", impact_boost=\"MILD\")\n    # highly-cited float up (bounded; zero-citation stays neutral)\nsearch_papers(query=\"深度学习\", language_affinity=\"MILD\")\n    # demote (never exclude) results not in the query's language;\n    # unknown-language papers stay neutral; hard-exclude via\n    # filters_advanced=[{\"field\":\"language\",\"value\":\"zh\"}]\n```\n\n**Search authors or journals (collection):**\n\nSet `collection` to `authors` or `sources` (default `papers`) to search\nthose entities. Each has its own fields — call\nlist_catalog(collection=\"authors\") first; use filters_advanced +\nsort_advanced (papers convenience fields apply to papers only).\n\n```\nsearch_papers(collection=\"authors\",\n    filters_advanced=[{field: \"summary_stats.h_index\", operator: \"FILTER_OP_GTE\", value: 50}],\n    sort_advanced=[{field: \"cited_by_count\", order: \"SORT_ORDER_DESC\"}])\n```\n\n**Fetch a paper figure / image:**\n\nWhen read_content Markdown contains `![alt](file_name)`, call\n`get_resource` with the file_name to fetch image binary.\n\n```\nread_content(doc_id, offset) → markdown ![Figure 3](dt=xxx/p/f3.png)\n    └─▶ get_resource(file_name=\"dt=xxx/p/f3.png\")\n```\n\n**Reading fulltext (check first):**\n\nEach search_papers hit carries `is_content_accessible` (bool): `true` only when\nthe paper has fulltext AND the caller is authorized. Check it before\n`read_content(doc_id, ...)` — `false` means no fulltext or no read permission.\n\n## Exit codes\n\n- `0` — success; stdout is the JSON response\n- `1` — HTTP 4xx/5xx; stderr contains status code and response body\n- `2` — argument error (missing token, malformed JSON, required field absent)\n\nFile v0.14.1:README.md\n\n# academic-retrieval — ClawHub skill bundle\n\n[![ClawHub](https://img.shields.io/badge/clawhub-academic--retrieval-brightgreen)](https://clawhub.ai/sciverse/skills/academic-retrieval)\n\nClawHub skill that gives any OpenClaw agent Sciverse academic-paper retrieval\ncapabilities (English | [中文](#中文说明)).\n\nPublished by **@sciverse** (slug `academic-retrieval`).\n\n## Install\n\n```bash\nopenclaw skills install academic-retrieval\n```\n\n## Configure\n\n```bash\nexport SCIVERSE_API_TOKEN=sv-xxx       # obtain from https://sciverse.space\n```\n\n## Tools at a glance\n\n| Tool | Purpose |\n|---|---|\n| `list_catalog` | Field introspection (call once to learn available fields + enum values) |\n| `search_papers` | Structured metadata search over papers / authors / sources (set `collection`) |\n| `semantic_search` | Natural-language semantic chunk retrieval (for RAG) |\n| `read_content` | Byte-range read of a paper's original text |\n| `get_resource` | Fetch figure / table image bytes referenced inside `read_content` Markdown |\n\nSee `SKILL.md` for full agent-facing documentation.\n\n## Direct invocation (bypass OpenClaw)\n\n```bash\nnode scripts/semantic_search.mjs '{\"query\":\"Transformer attention mechanism\",\"top_k\":3}'\n```\n\n## Relationship to the SDK\n\nThis skill is **complementary** to the `sciverse` packages on PyPI / npm:\n\n- **This skill** — OpenClaw users only. Zero external deps (Node 18+ native fetch).\n- **PyPI / npm SDK** — Any LLM agent framework (OpenAI, Anthropic, LangChain, LlamaIndex…).\n\n## License\n\nApache-2.0\n\n---\n\n## 中文说明\n\nOpenClaw 用户专用：通过 ClawHub 一键给 agent 加上 Sciverse 学术文献检索能力。\n\n发布者 **@sciverse**，slug `academic-retrieval`。\n\n### 安装\n\n```bash\nopenclaw skills install academic-retrieval\n```\n\n### 配置\n\n```bash\nexport SCIVERSE_API_TOKEN=sv-xxx   # 从 https://sciverse.space 控制台申请\n# 可选：export SCIVERSE_BASE_URL=https://api-custom.sciverse.space\n```\n\n### 工具速览\n\n| Tool | 用途 |\n|---|---|\n| `list_catalog` | 字段 introspection（首次接入调一次，学习可用字段和 enum 取值） |\n| `search_papers` | 按结构化条件查 papers / authors / sources（用 `collection` 切换实体集合） |\n| `semantic_search` | 自然语言语义检索文献片段（RAG 用） |\n| `read_content` | 按 Unicode 码点区间读取文献原文片段 |\n| `get_resource` | 取 `read_content` Markdown 中引用的图片字节流（多模态 RAG） |\n\nagent 视角的完整文档见 `SKILL.md`（英文）。\n\n### 直接调用（不通过 OpenClaw）\n\n```bash\nnode scripts/semantic_search.mjs '{\"query\":\"Transformer 注意力机制\",\"top_k\":3}'\n```\n\n### 与 SDK 的关系\n\n本 skill 与 PyPI/npm 上的 `sciverse` 包是**互补**的：\n\n- **本 skill**：OpenClaw 用户专用，零外部依赖（仅 Node 18+ native fetch）\n- **PyPI/npm SDK**：任意 LLM Agent 框架（OpenAI / Anthropic / LangChain / LlamaIndex...）\n\nFile v0.14.1:_meta.json\n\n{\n  \"ownerId\": \"kn74way11x0gjn6wpa8hcvyhvs85vmkj\",\n  \"slug\": \"academic-retrieval\",\n  \"version\": \"0.14.1\",\n  \"publishedAt\": 1789018002253\n}\n\nFile v0.14.1:skill-card.md\n\n## Description:\n\nSciverse academic paper retrieval: structured metadata search, semantic chunk retrieval for RAG, and byte-range content reading. For agent workflows that need citation-grade scientific literature.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[sciverse](https://clawhub.ai/user/sciverse)\n\n### License/Terms of Use:\n\nApache-2.0\n\n## Use Case:\n\nDevelopers and external agent users use this skill to locate academic papers, retrieve citation-ready paper chunks, inspect citation relationships, and expand selected paper text for research-grounded answers.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Academic search queries, document IDs, and retrieval requests are sent to Sciverse as an external service.\n\nMitigation: Use the skill only when that data sharing is acceptable for the workflow and avoid sending sensitive or confidential research prompts.\n\nRisk: A misconfigured or untrusted API endpoint could expose the Sciverse API token.\n\nMitigation: Keep SCIVERSE_BASE_URL at the default HTTPS Sciverse endpoint or a trusted HTTPS gateway, and rotate the token if it was ever used with an HTTP endpoint.\n\nRisk: Retrieved paper excerpts may be partial or unavailable when full text is not accessible.\n\nMitigation: Check the returned accessibility fields and use read_content pagination around selected offsets before relying on excerpts for citations.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/sciverse/skills/academic-retrieval)\n- [Sciverse homepage](https://sciverse.space)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, JSON, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown guidance with JSON tool responses and base64-encoded image resources when figures or tables are fetched.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Requires SCIVERSE_API_TOKEN. Retrieval calls send academic search queries, document IDs, relation requests, and content-range requests to Sciverse.]\n\n## Skill Version(s):\n\n0.14.1 (source: SKILL.md frontmatter, manifest.json, evidence release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v0.14.1:manifest.json\n\n{\n  \"name\": \"sciverse-academic-retrieval\",\n  \"version\": \"0.14.1\",\n  \"slug\": \"academic-retrieval\",\n  \"description\": \"Sciverse academic paper retrieval: structured metadata search, semantic chunk retrieval for RAG, and byte-range content reading. For agent workflows that need citation-grade scientific literature.\",\n  \"runtime\": \"node>=18\",\n  \"license\": \"Apache-2.0\",\n  \"homepage\": \"https://sciverse.space\",\n  \"env\": [\n    {\n      \"name\": \"SCIVERSE_API_TOKEN\",\n      \"required\": true,\n      \"description\": \"Sciverse API Token (obtain from https://sciverse.space).\"\n    },\n    {\n      \"name\": \"SCIVERSE_BASE_URL\",\n      \"required\": false,\n      \"default\": \"https://api.sciverse.space\",\n      \"description\": \"Override the default API base URL (for dev / self-hosted gateways).\"\n    }\n  ],\n  \"tools\": [\n    {\n      \"name\": \"search_papers\",\n      \"description\": \"Search academic papers by structured filters (title, authors, journal,\\nyear, subjects, etc.).\\nUse when: \\\"find Hinton's papers from 2020-2023\\\", \\\"Nature papers on\\nCRISPR\\\".\\nNot for: natural-language Q&A retrieval (use semantic_search) or\\nfull-text snippets (use read_content).\\nReturns: list of papers; each entry has unique_id (always present),\\ndoc_id (only when full text exists), title, author, abstract,\\npublication_venue_name_unified, publication_published_year.\",\n      \"script\": \"scripts/search_papers.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"properties\": {\n          \"collection\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"papers\",\n              \"authors\",\n              \"sources\"\n            ],\n            \"default\": \"papers\",\n            \"description\": \"检索的实体集合。papers（默认，论文）/ authors（作者）/ sources（来源期刊）。 各 collection 字段集不同，用 list_catalog（collection=<name>）学习对应 schema。 注意：本工具的便捷字段（authors/journals/year_from/subjects 等）只对 papers 有意义； 查 authors/sources 时改用 filters_advanced + 该 collection 的字段名（如 authors 的 summary_stats.h_index / orcid，sources 的 issn / is_oa）。authors 用 orcid、 sources 用 issn 与论文检索结果关联。\",\n            \"x-en-description\": \"Entity collection to search. papers (default) / authors / sources. Each collection has its own field schema — call list_catalog(collection=<name>). The convenience fields (authors/journals/year_from/subjects) apply to papers only; for authors/sources use filters_advanced with that collection's field names.\"\n          },\n          \"query\": {\n            \"type\": \"string\",\n            \"description\": \"BM25 全文关键词，匹配标题/摘要/期刊名/关键词字段。留空则纯靠结构化过滤。\"\n          },\n          \"title_contains\": {\n            \"type\": \"string\",\n            \"description\": \"标题中必须包含的词（仅匹配 title 字段）。\"\n          },\n          \"abstract_contains\": {\n            \"type\": \"string\",\n            \"description\": \"摘要中必须包含的词（仅匹配 abstract 字段）。\"\n          },\n          \"authors\": {\n            \"type\": \"array\",\n            \"items\": {\n              \"type\": \"string\"\n            },\n            \"description\": \"作者名（任一命中即可）。SDK 内部映射到后端 `author` 字段（FILTER_OP_IN）。\"\n          },\n          \"year_from\": {\n            \"type\": \"integer\",\n            \"description\": \"起始发表年（含）。\"\n          },\n          \"year_to\": {\n            \"type\": \"integer\",\n            \"description\": \"结束发表年（含）。\"\n          },\n          \"journals\": {\n            \"type\": \"array\",\n            \"items\": {\n              \"type\": \"string\"\n            },\n            \"description\": \"期刊名（任一命中即可）。SDK 内部映射到后端 `publication_venue_name_unified` 字段（FILTER_OP_IN，规范化后的载体名）。\"\n          },\n          \"subjects\": {\n            \"type\": \"array\",\n            \"items\": {\n              \"type\": \"string\"\n            },\n            \"description\": \"学科分类，如 \\\"computer science\\\"、\\\"biology\\\"。\"\n          },\n          \"filters_advanced\": {\n            \"type\": \"array\",\n            \"description\": \"高级过滤逃生舱（仅当上述字段不够用时使用）。可用字段见 get_field_catalog。\\n\\n引文反查（常用）：field=\\\"references_unique_id\\\" 查「谁引用了某篇论文」，\\nvalue 填目标论文的 unique_id。相比 list_paper_relations 的 CITATIONS，\\n它支持深翻页与任意排序，适合超高被引论文。可叠加条件，\\n例如「引用了 ResNet 且 2023 年后发表」：\\n  [{\\\"field\\\":\\\"references_unique_id\\\",\\\"value\\\":\\\"paper:10.1109/cvpr.2016.90\\\"},\\n   {\\\"field\\\":\\\"publication_published_year\\\",\\\"operator\\\":\\\"FILTER_OP_GTE\\\",\\\"value\\\":2023}]\\n该字段仅支持过滤，不能排序/聚合，也不能放进 fields 返回。\\n\",\n            \"items\": {\n              \"type\": \"object\",\n              \"required\": [\n                \"field\",\n                \"value\"\n              ],\n              \"properties\": {\n                \"field\": {\n                  \"type\": \"string\"\n                },\n                \"operator\": {\n                  \"type\": \"string\",\n                  \"description\": \"过滤操作符。MATCH（分词模糊）适用于 author、keywords（输入 \\\"Hinton\\\" 命中 \\\"Geoffrey Hinton\\\"）； MATCH_PHRASE（短语模糊）适用于 publication_venue_name_unified，整词连续匹配（\\\"Nature\\\" 命中 \\\"Nature Communications\\\"；非前缀匹配，\\\"Nature Comm\\\" 不会命中）； doi 用 EQ，服务端归一化（去 doi.org 前缀+转小写）后精确匹配。MATCH/MATCH_PHRASE 仅对配了 text 子字段的字段有效。\",\n                  \"enum\": [\n                    \"FILTER_OP_EQ\",\n                    \"FILTER_OP_NE\",\n                    \"FILTER_OP_GT\",\n                    \"FILTER_OP_GTE\",\n                    \"FILTER_OP_LT\",\n                    \"FILTER_OP_LTE\",\n                    \"FILTER_OP_IN\",\n                    \"FILTER_OP_NIN\",\n                    \"FILTER_OP_CONTAINS\",\n                    \"FILTER_OP_MATCH\",\n                    \"FILTER_OP_MATCH_PHRASE\"\n                  ],\n                  \"default\": \"FILTER_OP_EQ\"\n                },\n                \"value\": {}\n              }\n            }\n          },\n          \"sort_advanced\": {\n            \"type\": \"array\",\n            \"description\": \"高级排序逃生舱（按任意可排序字段）。papers 用 sort_by_year 即可； authors/sources 想按 h-index / 被引 / works_count 排序时用本字段。 与 query 互斥（query 走相关性排序）。\",\n            \"items\": {\n              \"type\": \"object\",\n              \"required\": [\n                \"field\",\n                \"order\"\n              ],\n              \"properties\": {\n                \"field\": {\n                  \"type\": \"string\"\n                },\n                \"order\": {\n                  \"type\": \"string\",\n                  \"enum\": [\n                    \"SORT_ORDER_DESC\",\n                    \"SORT_ORDER_ASC\"\n                  ],\n                  \"default\": \"SORT_ORDER_DESC\"\n                }\n              }\n            }\n          },\n          \"sort_by_year\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"auto\",\n              \"desc\",\n              \"asc\",\n              \"none\"\n            ],\n            \"default\": \"auto\",\n            \"description\": \"按发表年份排序。默认 auto：传了 query（或 sort_advanced）时不加年份排序\\n——保留 BM25 相关性排序，且 freshness/impact/language_affinity 软加权可用；\\n纯结构化筛选（无 query）时按年份降序（否则后端默认序是 unique_id，实质乱序）。\\n⚠️ 不要用 query + desc 求「最相关且最新」：显式排序会让 query 退化为命中\\n过滤（OR 语义、无相关性排序）、三个软加权全部失效——返回的是「含任一关键词\\n的最新文档」。要「相关且偏新」请用 freshness_boost。\\n\"\n          },\n          \"freshness_boost\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"NONE\",\n              \"MILD\",\n              \"STRONG\"\n            ],\n            \"default\": \"NONE\",\n            \"description\": \"模糊搜索新鲜度加权：结果偏向新文献（仅 query 非空时生效；传排序\\n（sort_by_year 非 none / sort_advanced）时被忽略，硬排优先）。\\nMILD: 近 10 年加权，适合日常查文献；STRONG: 近 3 年加权，适合跟踪\\n研究方向 / 追最新进展。与 impact_boost / language_affinity 可叠加\\n（均为乘法因子）。boost 生效时为浅翻页：不产 next_cursor、不支持\\ncursor 深翻页。\\n\"\n          },\n          \"impact_boost\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"NONE\",\n              \"MILD\",\n              \"STRONG\"\n            ],\n            \"default\": \"NONE\",\n            \"description\": \"模糊搜索影响力加权：高被引文献在保留相关性的前提下上浮（仅 query\\n非空时生效；传排序时被忽略）。MILD: 轻度上浮，相关性仍主导；\\nSTRONG: 明显偏向高被引。引用因子有界、零被引中性（不会归零）。\\n与 freshness_boost / language_affinity 可叠加；boost 生效时为浅翻页。\\n\"\n          },\n          \"language_affinity\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"NONE\",\n              \"MILD\",\n              \"STRONG\"\n            ],\n            \"default\": \"NONE\",\n            \"description\": \"模糊搜索语言亲和加权：非 query 语言的结果降序、但不排除（仅 query\\n非空时生效；传排序时被忽略）。目标语言由服务端从 query 文本判定\\n（假名→ja / 谚文→ko / 汉字→zh / 拉丁→en，其他书写系统不生效）；\\n语言未知的文献保持中性不降权。MILD: 非目标语言 ×0.5，跨语言强相关\\n结果仍可上浮；STRONG: ×0.2，几乎只看目标语言。与 freshness_boost /\\nimpact_boost 可叠加；boost 生效时为浅翻页。要硬排除某语言请改用\\nfilters_advanced 的 language 字段（如 {\\\"field\\\":\\\"language\\\",\\\"value\\\":\\\"en\\\"}，\\n软硬两层语义不同：本参数只调序，filter 直接排除）。\\n\"\n          },\n          \"page\": {\n            \"type\": \"integer\",\n            \"default\": 1,\n            \"minimum\": 1\n          },\n          \"page_size\": {\n            \"type\": \"integer\",\n            \"default\": 25,\n            \"minimum\": 1,\n            \"maximum\": 50,\n            \"description\": \"每页条数。省略时为服务端默认 25（SDK / MCP 不注入默认值）；工具层上限 50。\"\n          }\n        }\n      }\n    },\n    {\n      \"name\": \"semantic_search\",\n      \"description\": \"Natural-language semantic search returning relevant paper chunks for\\nRAG-style answering.\\nUse when: \\\"How does Transformer attention work?\\\", \\\"What are recent\\nmethods for protein structure prediction?\\\".\\nNot for: precise field filtering (use search_papers) or fetching full\\noriginal text (use read_content).\\nReturns: list of chunks; each entry has chunk_id, doc_id, abstract,\\nchunk, score, title, offset.\\nTypical chain: semantic_search → pick chunk → read_content(doc_id,\\noffset).\",\n      \"script\": \"scripts/semantic_search.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"required\": [\n          \"query\"\n        ],\n        \"properties\": {\n          \"query\": {\n            \"type\": \"string\",\n            \"minLength\": 1,\n            \"maxLength\": 4096,\n            \"description\": \"自然语言查询，1-200 字最佳。\"\n          },\n          \"top_k\": {\n            \"type\": \"integer\",\n            \"default\": 10,\n            \"minimum\": 1,\n            \"maximum\": 100,\n            \"description\": \"返回命中条数上限，合法 1-100（服务端校验，超出报 400）。\\n实际条数还受 mode 影响：balanced 单路混合召回在服务端固定截到约 50 条，\\ntop_k 超过 50 时多出的部分不会返回；fast 与 quality 可取到 top_k。\\n另外同一篇论文最多返回约 3 个 chunk，因此高 top_k 需要命中足够多的不同论文。\\n\"\n          },\n          \"source_types\": {\n            \"type\": \"array\",\n            \"items\": {\n              \"type\": \"string\",\n              \"enum\": [\n                \"web\",\n                \"pdf\"\n              ]\n            }\n          },\n          \"filters\": {\n            \"type\": \"object\",\n            \"description\": \"结构化过滤（可选）。在召回阶段与语义检索同时生效（ES+Milvus 双引擎下推，\\n不是结果后过滤）；多个字段之间 AND，同一字段传数组时数组内 OR。\\n⚠️ 宽松（软）语义：chunk 侧元数据缺失的文档不会被排除——例如按年份过滤时，\\n缺年份信息的 chunk 仍可能返回。需要严格范围保证时勿当硬约束使用，\\n表述结论时注明范围为\\\"近似过滤\\\"。\\n数值/日期字段支持区间：{\\\"gte\\\":2020,\\\"lte\\\":2025} 或 [min,max]（null 表示一侧不限）；\\n日期接受 YYYY / YYYY-MM / YYYY-MM-DD。\\n实际可用字段受账号字段权限约束；未知字段服务端返回 400。\\n例：{\\\"author\\\":[\\\"Hinton\\\"],\\\"publication_published_year\\\":{\\\"gte\\\":2023},\\n    \\\"topics\\\":{\\\"dimensions\\\":{\\\"primary_topic_domain\\\":\\\"Health Sciences\\\"}}}\\n\",\n            \"properties\": {\n              \"lang\": {\n                \"description\": \"语言代码，如 \\\"en\\\"、\\\"zh\\\"；也接受别名 language。\"\n              },\n              \"metadata_type\": {\n                \"description\": \"资源类型，仅单值：\\\"paper\\\" 或 \\\"ebook\\\"。\"\n              },\n              \"author\": {\n                \"description\": \"作者名，string 或 string[]（数组=任一命中）。\"\n              },\n              \"publication_venue_name_unified\": {\n                \"description\": \"发表载体名称（期刊/会议，规范化名，适合精确匹配）。\"\n              },\n              \"publication_venue_type\": {\n                \"description\": \"载体类型：\\\"journal\\\"、\\\"conference\\\"、\\\"repository\\\"、\\\"book series\\\"、\\\"ebook platform\\\"、\\\"metadata\\\"、\\\"raidRegistry\\\"、\\\"igsnCatalog\\\"、\\\"other\\\"（不区分大小写）。\"\n              },\n              \"publication_published_year\": {\n                \"description\": \"发表年份，单值或区间（{\\\"gte\\\":..,\\\"lte\\\":..} / [min,max]）。\"\n              },\n              \"publication_published_date\": {\n                \"description\": \"发表日期，\\\"YYYY[-MM[-DD]]\\\" 单值或区间。\"\n              },\n              \"citation_count\": {\n                \"description\": \"被引次数，单值或区间。\"\n              },\n              \"influential_citation_count\": {\n                \"description\": \"高影响力被引次数，单值或区间。\"\n              },\n              \"title\": {\n                \"description\": \"标题精确匹配（标题检索一般更适合 search_papers）。\"\n              },\n              \"topics\": {\n                \"description\": \"主题组合过滤：{\\\"logic\\\":\\\"and|or\\\",\\\"dimensions\\\":{\\\"primary_topic\\\":\\\"...\\\",\\\"primary_topic_domain\\\":\\\"Physical Sciences|Social Sciences|Health Sciences|Life Sciences\\\"}}；logic 省略默认 or。\"\n              },\n              \"doc_id\": {\n                \"description\": \"唯一的硬约束字段（其余字段均为软语义）：命中绝不越出给定集合。\\n值为 64 位小写 hex sha256（即 search_papers 返回的 doc_id；仅有全文的论文才有），\\nstring 或 string[]，仅 eq/in。去重后上限默认 1000，超限返回 400 SCOPE_TOO_LARGE；\\n显式传空数组返回 200 空 hits（候选集为空，不退化为全局检索）。\\n典型用法：先 search_papers 圈定候选集合，再在集合内做受限语义检索。\\n\"\n              }\n            }\n          },\n          \"mode\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"fast\",\n              \"balanced\",\n              \"quality\"\n            ],\n            \"default\": \"balanced\",\n            \"description\": \"fast = 仅关键词召回 (~200ms)；balanced = 混合检索 (~600ms)；quality = LLM 改写 + 混合 (~2-4s)。\\n\"\n          }\n        }\n      }\n    },\n    {\n      \"name\": \"list_catalog\",\n      \"description\": \"Returns the schema catalog for search_papers: every field name, type,\\nwhether it's filterable / sortable, default-return status, human\\ndescription, and applicable FilterOperators.\\nUse when: \\\"Which field do I filter by DOI?\\\", \\\"What values can\\naccess_oa_status take?\\\", \\\"What's the right enum for metadata_type?\\\".\\nNot for: actually searching papers (use search_papers / semantic_search).\\nTypical pattern: call once when first encountering Sciverse or facing\\nan ambiguous field need, then construct precise search_papers filters\\nfrom the returned schema.\\nPass include_sample_values=true to also fetch top-20 values for\\nenum-like fields (OpenSearch terms aggregation, 24h cached).\",\n      \"script\": \"scripts/list_catalog.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"properties\": {\n          \"collection\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"papers\",\n              \"authors\",\n              \"sources\"\n            ],\n            \"default\": \"papers\",\n            \"description\": \"字段 catalog 所属实体集合。papers（默认）/ authors / sources，各 collection 字段不同。\"\n          },\n          \"include_sample_values\": {\n            \"type\": \"boolean\",\n            \"default\": false,\n            \"description\": \"是否拉取 enum-like 字段的取值样本。false 仅返回静态 schema（毫秒级）；true 触发 OpenSearch terms agg（首次几百毫秒，之后 24h 走缓存）。\"\n          },\n          \"include_field_stats\": {\n            \"type\": \"boolean\",\n            \"default\": false,\n            \"description\": \"是否返回字段统计（keyword 字段基数 + 数值字段 min/max/avg/p50/p95）。触发 OpenSearch 聚合，缓存 24h。\"\n          }\n        },\n        \"required\": []\n      }\n    },\n    {\n      \"name\": \"list_paper_relations\",\n      \"description\": \"Paginate the full relation list of a paper. citations/references/related_works\\nare unbounded arrays (up to 340k entries for a single paper) and are NOT\\nprojectable in search_papers, so this endpoint is the only way to read them.\\nUse when: \\\"What does paper X cite?\\\" (relation=REFERENCES), \\\"Which papers cite\\npaper X?\\\" (relation=CITATIONS), \\\"Works related to paper X\\\" (relation=RELATED_WORKS).\\nNote: CITATIONS (incoming: who cites me) and REFERENCES (outgoing: who I cite)\\nare opposite directions.\\nTypical chain: get unique_id from search_papers / semantic_search, then paginate\\nhere by relation.\\nTwo limits (CITATIONS only; REFERENCES/RELATED_WORKS max out at 11833/20 in practice):\\nmore than 10000 relations returns 429; page*page_size above 10000 returns 400.\\nIn both cases switch to search_papers with filters_advanced on\\nreferences_unique_id — it supports deep paging and arbitrary sorting.\\ntotal_count counts in-corpus matches only, so it can differ from the paper's own\\ncitation_count by about 1%.\",\n      \"script\": \"scripts/list_paper_relations.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"required\": [\n          \"unique_id\",\n          \"relation\"\n        ],\n        \"properties\": {\n          \"unique_id\": {\n            \"type\": \"string\",\n            \"description\": \"目标论文 unique_id（如 paper:10.1038/xxx），来自 search_papers / semantic_search；勿传 doc_id。\"\n          },\n          \"relation\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"CITATIONS\",\n              \"REFERENCES\",\n              \"RELATED_WORKS\"\n            ],\n            \"description\": \"关系类型。CITATIONS=被引（谁引用了我）；REFERENCES=参考文献（我引用了谁）；RELATED_WORKS=相关工作。\"\n          },\n          \"page\": {\n            \"type\": \"integer\",\n            \"default\": 1,\n            \"minimum\": 1\n          },\n          \"page_size\": {\n            \"type\": \"integer\",\n            \"default\": 25,\n            \"minimum\": 1,\n            \"maximum\": 200\n          }\n        }\n      }\n    },\n    {\n      \"name\": \"read_content\",\n      \"description\": \"Read a range of a paper's original text addressed in Unicode code\\npoints (offset/limit count characters like Python len(), not bytes).\\nTypically used with a doc_id/offset returned by semantic_search to\\nexpand context (read more text before or after a chunk).\\nReturns: text fragment, bytes_returned (UTF-8 byte length of text, for\\nreference only), next_offset (code-point offset of the next fragment —\\npage with it, never with bytes_returned), more (boolean).\\nServer behaviour: limit above 524288 is silently clamped; omitting\\noffset returns the whole document ignoring limit — the SDKs / MCP\\nserver send offset=0 and limit=4096 by default, so pass offset\\nexplicitly when calling the HTTP API directly.\",\n      \"script\": \"scripts/read_content.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"properties\": {\n          \"doc_id\": {\n            \"type\": \"string\",\n            \"description\": \"文献 ID（来自 search_papers / semantic_search）。\"\n          },\n          \"offset\": {\n            \"type\": \"integer\",\n            \"format\": \"int64\",\n            \"default\": 0,\n            \"minimum\": 0,\n            \"description\": \"起始 Unicode 码点偏移，直接使用 semantic_search 返回的 offset。默认 0。\"\n          },\n          \"limit\": {\n            \"type\": \"integer\",\n            \"format\": \"int64\",\n            \"default\": 4096,\n            \"minimum\": 1,\n            \"maximum\": 524288,\n            \"description\": \"最多返回的 Unicode 码点数。默认 4096；服务端上限 524288（超出静默钳制）。LLM 场景建议 ≤ 16384，避免撑爆上下文。\"\n          }\n        },\n        \"required\": [\n          \"doc_id\"\n        ]\n      }\n    },\n    {\n      \"name\": \"get_resource\",\n      \"description\": \"Returns the binary bytes of a paper figure / table image referenced\\ninside read_content's Markdown via `![alt](file_name)` placeholders.\\nUse when the user asks to see / display / describe a figure and\\nread_content output contains an image reference.\\nInput file_name comes from the Markdown URL part (relative path,\\nno `\\\\\\\\` or `..`).\\nReturns: raw image stream + image/* Content-Type. The SDK / MCP\\nserver wraps the bytes as base64 + mimeType so Claude (multimodal)\\ncan read the image directly.\",\n      \"script\": \"scripts/get_resource.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"properties\": {\n          \"file_name\": {\n            \"type\": \"string\",\n            \"description\": \"图片相对路径，来自 read_content Markdown 中的 `![alt](file_name)` 占位。禁止 `\\\\\\\\` 与 `..`，不能以 `/` 开头。\"\n          }\n        },\n        \"required\": [\n          \"file_name\"\n        ]\n      }\n    }\n  ]\n}\n\nArchive v0.14.0: 12 files, 19340 bytes\n\nFiles: manifest.json (21998b), README.md (2917b), scripts/_common.mjs (3137b), scripts/get_resource.mjs (766b), scripts/list_catalog.mjs (465b), scripts/list_paper_relations.mjs (284b), scripts/read_content.mjs (355b), scripts/search_papers.mjs (259b), scripts/semantic_search.mjs (1192b), skill-card.md (2003b), SKILL.md (8819b), _meta.json (138b)\n\nFile v0.14.0:SKILL.md\n\n---\nname: sciverse-academic-retrieval\nslug: academic-retrieval\nversion: 0.14.0\ndescription: Sciverse academic paper retrieval: structured metadata search, semantic chunk retrieval for RAG, and byte-range content reading. For agent workflows that need citation-grade scientific literature.\nlicense: Apache-2.0\nhomepage: https://sciverse.space\n---\n\n# academic-retrieval\n\nSciverse academic paper retrieval: structured metadata search, semantic chunk retrieval for RAG, and byte-range content reading. For agent workflows that need citation-grade scientific literature.\n\n## When to use\n\nTrigger this skill when the user's request involves any of:\n\n- Locating academic papers by structured criteria (authors, year, journal, subjects)\n- Grounding answers in paper excerpts (RAG / citations)\n- Expanding the original text around a known doc_id (more bytes before/after a chunk)\n\n## Authentication\n\nThis skill requires the `SCIVERSE_API_TOKEN` environment variable\n(obtain from https://sciverse.space). Optionally set `SCIVERSE_BASE_URL`\nto override the default API base URL.\n\n## Tools\n\n### search_papers\n\nSearch academic papers by structured filters (title, authors, journal,\nyear, subjects, etc.).\nUse when: \"find Hinton's papers from 2020-2023\", \"Nature papers on\nCRISPR\".\nNot for: natural-language Q&A retrieval (use semantic_search) or\nfull-text snippets (use read_content).\nReturns: list of papers; each entry has unique_id (always present),\ndoc_id (only when full text exists), title, author, abstract,\npublication_venue_name_unified, publication_published_year.\n\n**Invoke**: `node scripts/search_papers.mjs '<JSON args>'`\n\n### semantic_search\n\nNatural-language semantic search returning relevant paper chunks for\nRAG-style answering.\nUse when: \"How does Transformer attention work?\", \"What are recent\nmethods for protein structure prediction?\".\nNot for: precise field filtering (use search_papers) or fetching full\noriginal text (use read_content).\nReturns: list of chunks; each entry has chunk_id, doc_id, abstract,\nchunk, score, title, offset.\nTypical chain: semantic_search → pick chunk → read_content(doc_id,\noffset).\n\n**Invoke**: `node scripts/semantic_search.mjs '<JSON args>'`\n\n### list_catalog\n\nReturns the schema catalog for search_papers: every field name, type,\nwhether it's filterable / sortable, default-return status, human\ndescription, and applicable FilterOperators.\nUse when: \"Which field do I filter by DOI?\", \"What values can\naccess_oa_status take?\", \"What's the right enum for metadata_type?\".\nNot for: actually searching papers (use search_papers / semantic_search).\nTypical pattern: call once when first encountering Sciverse or facing\nan ambiguous field need, then construct precise search_papers filters\nfrom the returned schema.\nPass include_sample_values=true to also fetch top-20 values for\nenum-like fields (OpenSearch terms aggregation, 24h cached).\n\n**Invoke**: `node scripts/list_catalog.mjs '<JSON args>'`\n\n### list_paper_relations\n\nPaginate the full relation list of a paper. citations/references/related_works\nare unbounded arrays (up to 340k entries for a single paper) and are NOT\nprojectable in search_papers, so this endpoint is the only way to read them.\nUse when: \"What does paper X cite?\" (relation=REFERENCES), \"Which papers cite\npaper X?\" (relation=CITATIONS), \"Works related to paper X\" (relation=RELATED_WORKS).\nNote: CITATIONS (incoming: who cites me) and REFERENCES (outgoing: who I cite)\nare opposite directions.\nTypical chain: get unique_id from search_papers / semantic_search, then paginate\nhere by relation.\nTwo limits (CITATIONS only; REFERENCES/RELATED_WORKS max out at 11833/20 in practice):\nmore than 10000 relations returns 429; page*page_size above 10000 returns 400.\nIn both cases switch to search_papers with filters_advanced on\nreferences_unique_id — it supports deep paging and arbitrary sorting.\ntotal_count counts in-corpus matches only, so it can differ from the paper's own\ncitation_count by about 1%.\n\n**Invoke**: `node scripts/list_paper_relations.mjs '<JSON args>'`\n\n### read_content\n\nRead a UTF-8 byte range of a paper's original text. Typically used with\na doc_id/offset returned by semantic_search to expand context (read\nmore bytes before or after a chunk).\nReturns: text fragment, bytes_returned, next_offset, more (boolean).\n\n**Invoke**: `node scripts/read_content.mjs '<JSON args>'`\n\n### get_resource\n\nReturns the binary bytes of a paper figure / table image referenced\ninside read_content's Markdown via `![alt](file_name)` placeholders.\nUse when the user asks to see / display / describe a figure and\nread_content output contains an image reference.\nInput file_name comes from the Markdown URL part (relative path,\nno `\\\\` or `..`).\nReturns: raw image stream + image/* Content-Type. The SDK / MCP\nserver wraps the bytes as base64 + mimeType so Claude (multimodal)\ncan read the image directly.\n\n**Invoke**: `node scripts/get_resource.mjs '<JSON args>'`\n\n## Bootstrap: learn the schema first\n\nIf you're unsure which fields exist or what values an enum takes\n(e.g. `metadata_type`, `language`, `access_oa_status`), call\n`list_catalog` once at the start. Sample values are returned for\nlow-cardinality fields. Use it instead of guessing field names —\nguessing wastes turns.\n\n```\nlist_catalog(include_sample_values=true)\n    └─▶ fields[].name + sample_values  →  precise filter construction\n```\n\n## Recipes\n\n**RAG flow (natural-language Q&A):**\n\n```\nsemantic_search(query=...) → hits[i].doc_id, hits[i].offset\n    └─▶ read_content(doc_id, offset)\n```\n\n**Lookup by DOI:**\n\n```\nsearch_papers(filters_advanced=[{field: \"doi\", value: \"10.1038/...\"}])\n```\n\n**OA + year filter:**\n\n```\nsearch_papers(\n    year_from=2024,\n    filters_advanced=[{field: \"access_is_oa\", value: \"true\"}]\n)\n```\n\n**Scoped semantic search (constrained corpus):**\n\n```\nsemantic_search(\n    query=\"...\",\n    filters={\"author\": [\"Hinton\"],\n             \"publication_published_year\": {\"gte\": 2020}}\n)   # applied at recall time, server-side; AND across fields\n```\n\nSoft semantics: chunks missing that metadata are NOT excluded.\nFor a hard guarantee, or meta-only constraints (fwci, citation graph,\ncomplex hit-sets), scope by doc_id — a HARD recall-time filter:\n\n```\nsearch_papers(..., fields=[\"doc_id\",\"title\"]) → collect doc_id\nsemantic_search(query=..., filters={\"doc_id\": [...]})\n    # hits never leave the set; empty list → empty hits (never global);\n    # up to 1000 deduped ids (400 SCOPE_TOO_LARGE beyond)\n```\n\n**Bias fuzzy search ranking (soft boosts — stackable):**\n\nThree multiplicative boosts (`freshness_boost` / `impact_boost` /\n`language_affinity`, each NONE/MILD/STRONG) reorder fuzzy-search\nresults while keeping relevance. Only effective when `query` is\nnon-empty; ignored when any sort is set; shallow paging while active.\n`sort_by_year` defaults to `auto` (relevance with `query`, newest-first\nfor pure filters); `query`+`desc` is an anti-pattern — it degrades the\nquery to a match filter and disables all boosts; use `freshness_boost`.\n\n```\nsearch_papers(query=\"large language model\", freshness_boost=\"STRONG\")\n    # recent first: STRONG=3-year decay, MILD=10-year\nsearch_papers(query=\"protein folding\", impact_boost=\"MILD\")\n    # highly-cited float up (bounded; zero-citation stays neutral)\nsearch_papers(query=\"深度学习\", language_affinity=\"MILD\")\n    # demote (never exclude) results not in the query's language;\n    # unknown-language papers stay neutral; hard-exclude via\n    # filters_advanced=[{\"field\":\"language\",\"value\":\"zh\"}]\n```\n\n**Search authors or journals (collection):**\n\nSet `collection` to `authors` or `sources` (default `papers`) to search\nthose entities. Each has its own fields — call\nlist_catalog(collection=\"authors\") first; use filters_advanced +\nsort_advanced (papers convenience fields apply to papers only).\n\n```\nsearch_papers(collection=\"authors\",\n    filters_advanced=[{field: \"summary_stats.h_index\", operator: \"FILTER_OP_GTE\", value: 50}],\n    sort_advanced=[{field: \"cited_by_count\", order: \"SORT_ORDER_DESC\"}])\n```\n\n**Fetch a paper figure / image:**\n\nWhen read_content Markdown contains `![alt](file_name)`, call\n`get_resource` with the file_name to fetch image binary.\n\n```\nread_content(doc_id, offset) → markdown ![Figure 3](dt=xxx/p/f3.png)\n    └─▶ get_resource(file_name=\"dt=xxx/p/f3.png\")\n```\n\n**Reading fulltext (check first):**\n\nEach search_papers hit carries `is_content_accessible` (bool): `true` only when\nthe paper has fulltext AND the caller is authorized. Check it before\n`read_content(doc_id, ...)` — `false` means no fulltext or no read permission.\n\n## Exit codes\n\n- `0` — success; stdout is the JSON response\n- `1` — HTTP 4xx/5xx; stderr contains status code and response body\n- `2` — argument error (missing token, malformed JSON, required field absent)\n\nFile v0.14.0:README.md\n\n# academic-retrieval — ClawHub skill bundle\n\n[![ClawHub](https://img.shields.io/badge/clawhub-academic--retrieval-brightgreen)](https://clawhub.ai/sciverse/skills/academic-retrieval)\n\nClawHub skill that gives any OpenClaw agent Sciverse academic-paper retrieval\ncapabilities (English | [中文](#中文说明)).\n\nPublished by **@sciverse** (slug `academic-retrieval`).\n\n## Install\n\n```bash\nopenclaw skills install academic-retrieval\n```\n\n## Configure\n\n```bash\nexport SCIVERSE_API_TOKEN=sv-xxx       # obtain from https://sciverse.space\n```\n\n## Tools at a glance\n\n| Tool | Purpose |\n|---|---|\n| `list_catalog` | Field introspection (call once to learn available fields + enum values) |\n| `search_papers` | Structured metadata search over papers / authors / sources (set `collection`) |\n| `semantic_search` | Natural-language semantic chunk retrieval (for RAG) |\n| `read_content` | Byte-range read of a paper's original text |\n| `get_resource` | Fetch figure / table image bytes referenced inside `read_content` Markdown |\n\nSee `SKILL.md` for full agent-facing documentation.\n\n## Direct invocation (bypass OpenClaw)\n\n```bash\nnode scripts/semantic_search.mjs '{\"query\":\"Transformer attention mechanism\",\"top_k\":3}'\n```\n\n## Relationship to the SDK\n\nThis skill is **complementary** to the `sciverse` packages on PyPI / npm:\n\n- **This skill** — OpenClaw users only. Zero external deps (Node 18+ native fetch).\n- **PyPI / npm SDK** — Any LLM agent framework (OpenAI, Anthropic, LangChain, LlamaIndex…).\n\n## License\n\nApache-2.0\n\n---\n\n## 中文说明\n\nOpenClaw 用户专用：通过 ClawHub 一键给 agent 加上 Sciverse 学术文献检索能力。\n\n发布者 **@sciverse**，slug `academic-retrieval`。\n\n### 安装\n\n```bash\nopenclaw skills install academic-retrieval\n```\n\n### 配置\n\n```bash\nexport SCIVERSE_API_TOKEN=sv-xxx   # 从 https://sciverse.space 控制台申请\n# 可选：export SCIVERSE_BASE_URL=https://api-custom.sciverse.space\n```\n\n### 工具速览\n\n| Tool | 用途 |\n|---|---|\n| `list_catalog` | 字段 introspection（首次接入调一次，学习可用字段和 enum 取值） |\n| `search_papers` | 按结构化条件查 papers / authors / sources（用 `collection` 切换实体集合） |\n| `semantic_search` | 自然语言语义检索文献片段（RAG 用） |\n| `read_content` | 按字节区间读取文献原文片段 |\n| `get_resource` | 取 `read_content` Markdown 中引用的图片字节流（多模态 RAG） |\n\nagent 视角的完整文档见 `SKILL.md`（英文）。\n\n### 直接调用（不通过 OpenClaw）\n\n```bash\nnode scripts/semantic_search.mjs '{\"query\":\"Transformer 注意力机制\",\"top_k\":3}'\n```\n\n### 与 SDK 的关系\n\n本 skill 与 PyPI/npm 上的 `sciverse` 包是**互补**的：\n\n- **本 skill**：OpenClaw 用户专用，零外部依赖（仅 Node 18+ native fetch）\n- **PyPI/npm SDK**：任意 LLM Agent 框架（OpenAI / Anthropic / LangChain / LlamaIndex...）\n\nFile v0.14.0:_meta.json\n\n{\n  \"ownerId\": \"kn74way11x0gjn6wpa8hcvyhvs85vmkj\",\n  \"slug\": \"academic-retrieval\",\n  \"version\": \"0.14.0\",\n  \"publishedAt\": 1786707040557\n}\n\nFile v0.14.0:skill-card.md\n\n## Description:\n\nSciverse academic paper retrieval: structured metadata search, semantic chunk retrieval for RAG, and byte-range content reading.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[sciverse](https://clawhub.ai/user/sciverse)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nExternal developers and OpenClaw users use this skill to search academic paper metadata, retrieve semantically relevant paper chunks for RAG, inspect citation/reference relationships, and read authorized paper content for citation-grade scientific workflows.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Research queries, paper identifiers, and retrieval requests are sent to Sciverse under the user's API token.\n\nMitigation: Use only where the organization approves Sciverse as a provider and avoid confidential, regulated, or proprietary research prompts unless that data flow is approved.\n\nRisk: The skill depends on an API token and network access for retrieval.\n\nMitigation: Configure the token through the documented environment variable and review access controls before deployment.\n\n## Reference(s):\n\n- [Sciverse homepage](https://sciverse.space)\n- [ClawHub skill page](https://clawhub.ai/sciverse/skills/academic-retrieval)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown guidance with JSON tool responses and inline shell commands]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Requires SCIVERSE_API_TOKEN; resource retrieval can return base64-encoded image bytes with a MIME type.]\n\n## Skill Version(s):\n\n0.14.0 (source: server evidence, frontmatter, and manifest)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v0.14.0:manifest.json\n\n{\n  \"name\": \"sciverse-academic-retrieval\",\n  \"version\": \"0.14.0\",\n  \"slug\": \"academic-retrieval\",\n  \"description\": \"Sciverse academic paper retrieval: structured metadata search, semantic chunk retrieval for RAG, and byte-range content reading. For agent workflows that need citation-grade scientific literature.\",\n  \"runtime\": \"node>=18\",\n  \"license\": \"Apache-2.0\",\n  \"homepage\": \"https://sciverse.space\",\n  \"env\": [\n    {\n      \"name\": \"SCIVERSE_API_TOKEN\",\n      \"required\": true,\n      \"description\": \"Sciverse API Token (obtain from https://sciverse.space).\"\n    },\n    {\n      \"name\": \"SCIVERSE_BASE_URL\",\n      \"required\": false,\n      \"default\": \"https://api.sciverse.space\",\n      \"description\": \"Override the default API base URL (for dev / self-hosted gateways).\"\n    }\n  ],\n  \"tools\": [\n    {\n      \"name\": \"search_papers\",\n      \"description\": \"Search academic papers by structured filters (title, authors, journal,\\nyear, subjects, etc.).\\nUse when: \\\"find Hinton's papers from 2020-2023\\\", \\\"Nature papers on\\nCRISPR\\\".\\nNot for: natural-language Q&A retrieval (use semantic_search) or\\nfull-text snippets (use read_content).\\nReturns: list of papers; each entry has unique_id (always present),\\ndoc_id (only when full text exists), title, author, abstract,\\npublication_venue_name_unified, publication_published_year.\",\n      \"script\": \"scripts/search_papers.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"properties\": {\n          \"collection\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"papers\",\n              \"authors\",\n              \"sources\"\n            ],\n            \"default\": \"papers\",\n            \"description\": \"检索的实体集合。papers（默认，论文）/ authors（作者）/ sources（来源期刊）。 各 collection 字段集不同，用 list_catalog（collection=<name>）学习对应 schema。 注意：本工具的便捷字段（authors/journals/year_from/subjects 等）只对 papers 有意义； 查 authors/sources 时改用 filters_advanced + 该 collection 的字段名（如 authors 的 summary_stats.h_index / orcid，sources 的 issn / is_oa）。authors 用 orcid、 sources 用 issn 与论文检索结果关联。\",\n            \"x-en-description\": \"Entity collection to search. papers (default) / authors / sources. Each collection has its own field schema — call list_catalog(collection=<name>). The convenience fields (authors/journals/year_from/subjects) apply to papers only; for authors/sources use filters_advanced with that collection's field names.\"\n          },\n          \"query\": {\n            \"type\": \"string\",\n            \"description\": \"BM25 全文关键词，匹配标题/摘要/期刊名/关键词字段。留空则纯靠结构化过滤。\"\n          },\n          \"title_contains\": {\n            \"type\": \"string\",\n            \"description\": \"标题中必须包含的词（仅匹配 title 字段）。\"\n          },\n          \"abstract_contains\": {\n            \"type\": \"string\",\n            \"description\": \"摘要中必须包含的词（仅匹配 abstract 字段）。\"\n          },\n          \"authors\": {\n            \"type\": \"array\",\n            \"items\": {\n              \"type\": \"string\"\n            },\n            \"description\": \"作者名（任一命中即可）。SDK 内部映射到后端 `author` 字段（FILTER_OP_IN）。\"\n          },\n          \"year_from\": {\n            \"type\": \"integer\",\n            \"description\": \"起始发表年（含）。\"\n          },\n          \"year_to\": {\n            \"type\": \"integer\",\n            \"description\": \"结束发表年（含）。\"\n          },\n          \"journals\": {\n            \"type\": \"array\",\n            \"items\": {\n              \"type\": \"string\"\n            },\n            \"description\": \"期刊名（任一命中即可）。SDK 内部映射到后端 `publication_venue_name_unified` 字段（FILTER_OP_IN，规范化后的载体名）。\"\n          },\n          \"subjects\": {\n            \"type\": \"array\",\n            \"items\": {\n              \"type\": \"string\"\n            },\n            \"description\": \"学科分类，如 \\\"computer science\\\"、\\\"biology\\\"。\"\n          },\n          \"filters_advanced\": {\n            \"type\": \"array\",\n            \"description\": \"高级过滤逃生舱（仅当上述字段不够用时使用）。可用字段见 get_field_catalog。\\n\\n引文反查（常用）：field=\\\"references_unique_id\\\" 查「谁引用了某篇论文」，\\nvalue 填目标论文的 unique_id。相比 list_paper_relations 的 CITATIONS，\\n它支持深翻页与任意排序，适合超高被引论文。可叠加条件，\\n例如「引用了 ResNet 且 2023 年后发表」：\\n  [{\\\"field\\\":\\\"references_unique_id\\\",\\\"value\\\":\\\"paper:10.1109/cvpr.2016.90\\\"},\\n   {\\\"field\\\":\\\"publication_published_year\\\",\\\"operator\\\":\\\"FILTER_OP_GTE\\\",\\\"value\\\":2023}]\\n该字段仅支持过滤，不能排序/聚合，也不能放进 fields 返回。\\n\",\n            \"items\": {\n              \"type\": \"object\",\n              \"required\": [\n                \"field\",\n                \"value\"\n              ],\n              \"properties\": {\n                \"field\": {\n                  \"type\": \"string\"\n                },\n                \"operator\": {\n                  \"type\": \"string\",\n                  \"description\": \"过滤操作符。MATCH（分词模糊）适用于 author、keywords（输入 \\\"Hinton\\\" 命中 \\\"Geoffrey Hinton\\\"）； MATCH_PHRASE（短语模糊）适用于 publication_venue_name_unified，整词连续匹配（\\\"Nature\\\" 命中 \\\"Nature Communications\\\"；非前缀匹配，\\\"Nature Comm\\\" 不会命中）； doi 用 EQ，服务端归一化（去 doi.org 前缀+转小写）后精确匹配。MATCH/MATCH_PHRASE 仅对配了 text 子字段的字段有效。\",\n                  \"enum\": [\n                    \"FILTER_OP_EQ\",\n                    \"FILTER_OP_NE\",\n                    \"FILTER_OP_GT\",\n                    \"FILTER_OP_GTE\",\n                    \"FILTER_OP_LT\",\n                    \"FILTER_OP_LTE\",\n                    \"FILTER_OP_IN\",\n                    \"FILTER_OP_NIN\",\n                    \"FILTER_OP_CONTAINS\",\n                    \"FILTER_OP_MATCH\",\n                    \"FILTER_OP_MATCH_PHRASE\"\n                  ],\n                  \"default\": \"FILTER_OP_EQ\"\n                },\n                \"value\": {}\n              }\n            }\n          },\n          \"sort_advanced\": {\n            \"type\": \"array\",\n            \"description\": \"高级排序逃生舱（按任意可排序字段）。papers 用 sort_by_year 即可； authors/sources 想按 h-index / 被引 / works_count 排序时用本字段。 与 query 互斥（query 走相关性排序）。\",\n            \"items\": {\n              \"type\": \"object\",\n              \"required\": [\n                \"field\",\n                \"order\"\n              ],\n              \"properties\": {\n                \"field\": {\n                  \"type\": \"string\"\n                },\n                \"order\": {\n                  \"type\": \"string\",\n                  \"enum\": [\n                    \"SORT_ORDER_DESC\",\n                    \"SORT_ORDER_ASC\"\n                  ],\n                  \"default\": \"SORT_ORDER_DESC\"\n                }\n              }\n            }\n          },\n          \"sort_by_year\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"auto\",\n              \"desc\",\n              \"asc\",\n              \"none\"\n            ],\n            \"default\": \"auto\",\n            \"description\": \"按发表年份排序。默认 auto：传了 query（或 sort_advanced）时不加年份排序\\n——保留 BM25 相关性排序，且 freshness/impact/language_affinity 软加权可用；\\n纯结构化筛选（无 query）时按年份降序（否则后端默认序是 unique_id，实质乱序）。\\n⚠️ 不要用 query + desc 求「最相关且最新」：显式排序会让 query 退化为命中\\n过滤（OR 语义、无相关性排序）、三个软加权全部失效——返回的是「含任一关键词\\n的最新文档」。要「相关且偏新」请用 freshness_boost。\\n\"\n          },\n          \"freshness_boost\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"NONE\",\n              \"MILD\",\n              \"STRONG\"\n            ],\n            \"default\": \"NONE\",\n            \"description\": \"模糊搜索新鲜度加权：结果偏向新文献（仅 query 非空时生效；传排序\\n（sort_by_year 非 none / sort_advanced）时被忽略，硬排优先）。\\nMILD: 近 10 年加权，适合日常查文献；STRONG: 近 3 年加权，适合跟踪\\n研究方向 / 追最新进展。与 impact_boost / language_affinity 可叠加\\n（均为乘法因子）。boost 生效时为浅翻页：不产 next_cursor、不支持\\ncursor 深翻页。\\n\"\n          },\n          \"impact_boost\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"NONE\",\n              \"MILD\",\n              \"STRONG\"\n            ],\n            \"default\": \"NONE\",\n            \"description\": \"模糊搜索影响力加权：高被引文献在保留相关性的前提下上浮（仅 query\\n非空时生效；传排序时被忽略）。MILD: 轻度上浮，相关性仍主导；\\nSTRONG: 明显偏向高被引。引用因子有界、零被引中性（不会归零）。\\n与 freshness_boost / language_affinity 可叠加；boost 生效时为浅翻页。\\n\"\n          },\n          \"language_affinity\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"NONE\",\n              \"MILD\",\n              \"STRONG\"\n            ],\n            \"default\": \"NONE\",\n            \"description\": \"模糊搜索语言亲和加权：非 query 语言的结果降序、但不排除（仅 query\\n非空时生效；传排序时被忽略）。目标语言由服务端从 query 文本判定\\n（假名→ja / 谚文→ko / 汉字→zh / 拉丁→en，其他书写系统不生效）；\\n语言未知的文献保持中性不降权。MILD: 非目标语言 ×0.5，跨语言强相关\\n结果仍可上浮；STRONG: ×0.2，几乎只看目标语言。与 freshness_boost /\\nimpact_boost 可叠加；boost 生效时为浅翻页。要硬排除某语言请改用\\nfilters_advanced 的 language 字段（如 {\\\"field\\\":\\\"language\\\",\\\"value\\\":\\\"en\\\"}，\\n软硬两层语义不同：本参数只调序，filter 直接排除）。\\n\"\n          },\n          \"page\": {\n            \"type\": \"integer\",\n            \"default\": 1,\n            \"minimum\": 1\n          },\n          \"page_size\": {\n            \"type\": \"integer\",\n            \"default\": 10,\n            \"minimum\": 1,\n            \"maximum\": 50\n          }\n        }\n      }\n    },\n    {\n      \"name\": \"semantic_search\",\n      \"description\": \"Natural-language semantic search returning relevant paper chunks for\\nRAG-style answering.\\nUse when: \\\"How does Transformer attention work?\\\", \\\"What are recent\\nmethods for protein structure prediction?\\\".\\nNot for: precise field filtering (use search_papers) or fetching full\\noriginal text (use read_content).\\nReturns: list of chunks; each entry has chunk_id, doc_id, abstract,\\nchunk, score, title, offset.\\nTypical chain: semantic_search → pick chunk → read_content(doc_id,\\noffset).\",\n      \"script\": \"scripts/semantic_search.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"required\": [\n          \"query\"\n        ],\n        \"properties\": {\n          \"query\": {\n            \"type\": \"string\",\n            \"minLength\": 1,\n            \"maxLength\": 4096,\n            \"description\": \"自然语言查询，1-200 字最佳。\"\n          },\n          \"top_k\": {\n            \"type\": \"integer\",\n            \"default\": 10,\n            \"minimum\": 1,\n            \"maximum\": 100,\n            \"description\": \"返回命中条数上限，合法 1-100（服务端校验，超出报 400）。\\n实际条数还受 mode 影响：balanced 单路混合召回在服务端固定截到约 50 条，\\ntop_k 超过 50 时多出的部分不会返回；fast 与 quality 可取到 top_k。\\n另外同一篇论文最多返回约 3 个 chunk，因此高 top_k 需要命中足够多的不同论文。\\n\"\n          },\n          \"source_types\": {\n            \"type\": \"array\",\n            \"items\": {\n              \"type\": \"string\",\n              \"enum\": [\n                \"web\",\n                \"pdf\"\n              ]\n            }\n          },\n          \"filters\": {\n            \"type\": \"object\",\n            \"description\": \"结构化过滤（可选）。在召回阶段与语义检索同时生效（ES+Milvus 双引擎下推，\\n不是结果后过滤）；多个字段之间 AND，同一字段传数组时数组内 OR。\\n⚠️ 宽松（软）语义：chunk 侧元数据缺失的文档不会被排除——例如按年份过滤时，\\n缺年份信息的 chunk 仍可能返回。需要严格范围保证时勿当硬约束使用，\\n表述结论时注明范围为\\\"近似过滤\\\"。\\n数值/日期字段支持区间：{\\\"gte\\\":2020,\\\"lte\\\":2025} 或 [min,max]（null 表示一侧不限）；\\n日期接受 YYYY / YYYY-MM / YYYY-MM-DD。\\n实际可用字段受账号字段权限约束；未知字段服务端返回 400。\\n例：{\\\"author\\\":[\\\"Hinton\\\"],\\\"publication_published_year\\\":{\\\"gte\\\":2023},\\n    \\\"topics\\\":{\\\"dimensions\\\":{\\\"primary_topic_domain\\\":\\\"Health Sciences\\\"}}}\\n\",\n            \"properties\": {\n              \"lang\": {\n                \"description\": \"语言代码，如 \\\"en\\\"、\\\"zh\\\"；也接受别名 language。\"\n              },\n              \"metadata_type\": {\n                \"description\": \"资源类型，仅单值：\\\"paper\\\" 或 \\\"ebook\\\"。\"\n              },\n              \"author\": {\n                \"description\": \"作者名，string 或 string[]（数组=任一命中）。\"\n              },\n              \"publication_venue_name_unified\": {\n                \"description\": \"发表载体名称（期刊/会议，规范化名，适合精确匹配）。\"\n              },\n              \"publication_venue_type\": {\n                \"description\": \"载体类型：\\\"journal\\\"、\\\"conference\\\"、\\\"repository\\\"、\\\"book series\\\"、\\\"ebook platform\\\"、\\\"metadata\\\"、\\\"raidRegistry\\\"、\\\"igsnCatalog\\\"、\\\"other\\\"（不区分大小写）。\"\n              },\n              \"publication_published_year\": {\n                \"description\": \"发表年份，单值或区间（{\\\"gte\\\":..,\\\"lte\\\":..} / [min,max]）。\"\n              },\n              \"publication_published_date\": {\n                \"description\": \"发表日期，\\\"YYYY[-MM[-DD]]\\\" 单值或区间。\"\n              },\n              \"citation_count\": {\n                \"description\": \"被引次数，单值或区间。\"\n              },\n              \"influential_citation_count\": {\n                \"description\": \"高影响力被引次数，单值或区间。\"\n              },\n              \"title\": {\n                \"description\": \"标题精确匹配（标题检索一般更适合 search_papers）。\"\n              },\n              \"topics\": {\n                \"description\": \"主题组合过滤：{\\\"logic\\\":\\\"and|or\\\",\\\"dimensions\\\":{\\\"primary_topic\\\":\\\"...\\\",\\\"primary_topic_domain\\\":\\\"Physical Sciences|Social Sciences|Health Sciences|Life Sciences\\\"}}；logic 省略默认 or。\"\n              },\n              \"doc_id\": {\n                \"description\": \"唯一的硬约束字段（其余字段均为软语义）：命中绝不越出给定集合。\\n值为 64 位小写 hex sha256（即 search_papers 返回的 doc_id；仅有全文的论文才有），\\nstring 或 string[]，仅 eq/in。去重后上限默认 1000，超限返回 400 SCOPE_TOO_LARGE；\\n显式传空数组返回 200 空 hits（候选集为空，不退化为全局检索）。\\n典型用法：先 search_papers 圈定候选集合，再在集合内做受限语义检索。\\n\"\n              }\n            }\n          },\n          \"mode\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"fast\",\n              \"balanced\",\n              \"quality\"\n            ],\n            \"default\": \"balanced\",\n            \"description\": \"fast = 仅关键词召回 (~200ms)；balanced = 混合检索 (~600ms)；quality = LLM 改写 + 混合 (~2-4s)。\\n\"\n          }\n        }\n      }\n    },\n    {\n      \"name\": \"list_catalog\",\n      \"description\": \"Returns the schema catalog for search_papers: every field name, type,\\nwhether it's filterable / sortable, default-return status, human\\ndescription, and applicable FilterOperators.\\nUse when: \\\"Which field do I filter by DOI?\\\", \\\"What values can\\naccess_oa_status take?\\\", \\\"What's the right enum for metadata_type?\\\".\\nNot for: actually searching papers (use search_papers / semantic_search).\\nTypical pattern: call once when first encountering Sciverse or facing\\nan ambiguous field need, then construct precise search_papers filters\\nfrom the returned schema.\\nPass include_sample_values=true to also fetch top-20 values for\\nenum-like fields (OpenSearch terms aggregation, 24h cached).\",\n      \"script\": \"scripts/list_catalog.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"properties\": {\n          \"collection\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"papers\",\n              \"authors\",\n              \"sources\"\n            ],\n            \"default\": \"papers\",\n            \"description\": \"字段 catalog 所属实体集合。papers（默认）/ authors / sources，各 collection 字段不同。\"\n          },\n          \"include_sample_values\": {\n            \"type\": \"boolean\",\n            \"default\": false,\n            \"description\": \"是否拉取 enum-like 字段的取值样本。false 仅返回静态 schema（毫秒级）；true 触发 OpenSearch terms agg（首次几百毫秒，之后 24h 走缓存）。\"\n          },\n          \"include_field_stats\": {\n            \"type\": \"boolean\",\n            \"default\": false,\n            \"description\": \"是否返回字段统计（keyword 字段基数 + 数值字段 min/max/avg/p50/p95）。触发 OpenSearch 聚合，缓存 24h。\"\n          }\n        },\n        \"required\": []\n      }\n    },\n    {\n      \"name\": \"list_paper_relations\",\n      \"description\": \"Paginate the full relation list of a paper. citations/references/related_works\\nare unbounded arrays (up to 340k entries for a single paper) and are NOT\\nprojectable in search_papers, so this endpoint is the only way to read them.\\nUse when: \\\"What does paper X cite?\\\" (relation=REFERENCES), \\\"Which papers cite\\npaper X?\\\" (relation=CITATIONS), \\\"Works related to paper X\\\" (relation=RELATED_WORKS).\\nNote: CITATIONS (incoming: who cites me) and REFERENCES (outgoing: who I cite)\\nare opposite directions.\\nTypical chain: get unique_id from search_papers / semantic_search, then paginate\\nhere by relation.\\nTwo limits (CITATIONS only; REFERENCES/RELATED_WORKS max out at 11833/20 in practice):\\nmore than 10000 relations returns 429; page*page_size above 10000 returns 400.\\nIn both cases switch to search_papers with filters_advanced on\\nreferences_unique_id — it supports deep paging and arbitrary sorting.\\ntotal_count counts in-corpus matches only, so it can differ from the paper's own\\ncitation_count by about 1%.\",\n      \"script\": \"scripts/list_paper_relations.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"required\": [\n          \"unique_id\",\n          \"relation\"\n        ],\n        \"properties\": {\n          \"unique_id\": {\n            \"type\": \"string\",\n            \"description\": \"目标论文 unique_id（如 paper:10.1038/xxx），来自 search_papers / semantic_search；勿传 doc_id。\"\n          },\n          \"relation\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"CITATIONS\",\n              \"REFERENCES\",\n              \"RELATED_WORKS\"\n            ],\n            \"description\": \"关系类型。CITATIONS=被引（谁引用了我）；REFERENCES=参考文献（我引用了谁）；RELATED_WORKS=相关工作。\"\n          },\n          \"page\": {\n            \"type\": \"integer\",\n            \"default\": 1,\n            \"minimum\": 1\n          },\n          \"page_size\": {\n            \"type\": \"integer\",\n            \"default\": 25,\n            \"minimum\": 1,\n            \"maximum\": 200\n          }\n        }\n      }\n    },\n    {\n      \"name\": \"read_content\",\n      \"description\": \"Read a UTF-8 byte range of a paper's original text. Typically used with\\na doc_id/offset returned by semantic_search to expand context (read\\nmore bytes before or after a chunk).\\nReturns: text fragment, bytes_returned, next_offset, more (boolean).\",\n      \"script\": \"scripts/read_content.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"properties\": {\n          \"doc_id\": {\n            \"type\": \"string\",\n            \"description\": \"文献 ID（来自 search_papers / semantic_search）。\"\n          },\n          \"offset\": {\n            \"type\": \"integer\",\n            \"format\": \"int64\",\n            \"default\": 0\n          },\n          \"limit\": {\n            \"type\": \"integer\",\n            \"format\": \"int64\",\n            \"default\": 4096,\n            \"maximum\": 16384\n          }\n        },\n        \"required\": [\n          \"doc_id\"\n        ]\n      }\n    },\n    {\n      \"name\": \"get_resource\",\n      \"description\": \"Returns the binary bytes of a paper figure / table image referenced\\ninside read_content's Markdown via `![alt](file_name)` placeholders.\\nUse when the user asks to see / display / describe a figure and\\nread_content output contains an image reference.\\nInput file_name comes from the Markdown URL part (relative path,\\nno `\\\\\\\\` or `..`).\\nReturns: raw image stream + image/* Content-Type. The SDK / MCP\\nserver wraps the bytes as base64 + mimeType so Claude (multimodal)\\ncan read the image directly.\",\n      \"script\": \"scripts/get_resource.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"properties\": {\n          \"file_name\": {\n            \"type\": \"string\",\n            \"description\": \"图片相对路径，来自 read_content Markdown 中的 `![alt](file_name)` 占位。禁止 `\\\\\\\\` 与 `..`，不能以 `/` 开头。\"\n          }\n        },\n        \"required\": [\n          \"file_name\"\n        ]\n      }\n    }\n  ]\n}\n\nArchive v0.13.1: 12 files, 19112 bytes\n\nFiles: manifest.json (21356b), README.md (2917b), scripts/_common.mjs (3137b), scripts/get_resource.mjs (766b), scripts/list_catalog.mjs (465b), scripts/list_paper_relations.mjs (284b), scripts/read_content.mjs (355b), scripts/search_papers.mjs (259b), scripts/semantic_search.mjs (1192b), skill-card.md (2269b), SKILL.md (8602b), _meta.json (138b)\n\nFile v0.13.1:SKILL.md\n\n---\nname: sciverse-academic-retrieval\nslug: academic-retrieval\nversion: 0.13.1\ndescription: Sciverse academic paper retrieval: structured metadata search, semantic chunk retrieval for RAG, and byte-range content reading. For agent workflows that need citation-grade scientific literature.\nlicense: Apache-2.0\nhomepage: https://sciverse.space\n---\n\n# academic-retrieval\n\nSciverse academic paper retrieval: structured metadata search, semantic chunk retrieval for RAG, and byte-range content reading. For agent workflows that need citation-grade scientific literature.\n\n## When to use\n\nTrigger this skill when the user's request involves any of:\n\n- Locating academic papers by structured criteria (authors, year, journal, subjects)\n- Grounding answers in paper excerpts (RAG / citations)\n- Expanding the original text around a known doc_id (more bytes before/after a chunk)\n\n## Authentication\n\nThis skill requires the `SCIVERSE_API_TOKEN` environment variable\n(obtain from https://sciverse.space). Optionally set `SCIVERSE_BASE_URL`\nto override the default API base URL.\n\n## Tools\n\n### search_papers\n\nSearch academic papers by structured filters (title, authors, journal,\nyear, subjects, etc.).\nUse when: \"find Hinton's papers from 2020-2023\", \"Nature papers on\nCRISPR\".\nNot for: natural-language Q&A retrieval (use semantic_search) or\nfull-text snippets (use read_content).\nReturns: list of papers; each entry has unique_id (always present),\ndoc_id (only when full text exists), title, author, abstract,\npublication_venue_name_unified, publication_published_year.\n\n**Invoke**: `node scripts/search_papers.mjs '<JSON args>'`\n\n### semantic_search\n\nNatural-language semantic search returning relevant paper chunks for\nRAG-style answering.\nUse when: \"How does Transformer attention work?\", \"What are recent\nmethods for protein structure prediction?\".\nNot for: precise field filtering (use search_papers) or fetching full\noriginal text (use read_content).\nReturns: list of chunks; each entry has chunk_id, doc_id, abstract,\nchunk, score, title, offset.\nTypical chain: semantic_search → pick chunk → read_content(doc_id,\noffset).\n\n**Invoke**: `node scripts/semantic_search.mjs '<JSON args>'`\n\n### list_catalog\n\nReturns the schema catalog for search_papers: every field name, type,\nwhether it's filterable / sortable, default-return status, human\ndescription, and applicable FilterOperators.\nUse when: \"Which field do I filter by DOI?\", \"What values can\naccess_oa_status take?\", \"What's the right enum for metadata_type?\".\nNot for: actually searching papers (use search_papers / semantic_search).\nTypical pattern: call once when first encountering Sciverse or facing\nan ambiguous field need, then construct precise search_papers filters\nfrom the returned schema.\nPass include_sample_values=true to also fetch top-20 values for\nenum-like fields (OpenSearch terms aggregation, 24h cached).\n\n**Invoke**: `node scripts/list_catalog.mjs '<JSON args>'`\n\n### list_paper_relations\n\nPaginate the full relation list of a paper. citations/references/related_works\nare unbounded arrays (up to 340k entries for a single paper) and are NOT\nprojectable in search_papers, so this endpoint is the only way to read them.\nUse when: \"What does paper X cite?\" (relation=REFERENCES), \"Which papers cite\npaper X?\" (relation=CITATIONS), \"Works related to paper X\" (relation=RELATED_WORKS).\nNote: CITATIONS (incoming: who cites me) and REFERENCES (outgoing: who I cite)\nare opposite directions.\nTypical chain: get unique_id from search_papers / semantic_search, then paginate\nhere by relation.\nTwo limits (CITATIONS only; REFERENCES/RELATED_WORKS max out at 11833/20 in practice):\nmore than 10000 relations returns 429; page*page_size above 10000 returns 400.\nIn both cases switch to search_papers with filters_advanced on\nreferences_unique_id — it supports deep paging and arbitrary sorting.\ntotal_count counts in-corpus matches only, so it can differ from the paper's own\ncitation_count by about 1%.\n\n**Invoke**: `node scripts/list_paper_relations.mjs '<JSON args>'`\n\n### read_content\n\nRead a UTF-8 byte range of a paper's original text. Typically used with\na doc_id/offset returned by semantic_search to expand context (read\nmore bytes before or after a chunk).\nReturns: text fragment, bytes_returned, next_offset, more (boolean).\n\n**Invoke**: `node scripts/read_content.mjs '<JSON args>'`\n\n### get_resource\n\nReturns the binary bytes of a paper figure / table image referenced\ninside read_content's Markdown via `![alt](file_name)` placeholders.\nUse when the user asks to see / display / describe a figure and\nread_content output contains an image reference.\nInput file_name comes from the Markdown URL part (relative path,\nno `\\\\` or `..`).\nReturns: raw image stream + image/* Content-Type. The SDK / MCP\nserver wraps the bytes as base64 + mimeType so Claude (multimodal)\ncan read the image directly.\n\n**Invoke**: `node scripts/get_resource.mjs '<JSON args>'`\n\n## Bootstrap: learn the schema first\n\nIf you're unsure which fields exist or what values an enum takes\n(e.g. `metadata_type`, `language`, `access_oa_status`), call\n`list_catalog` once at the start. Sample values are returned for\nlow-cardinality fields. Use it instead of guessing field names —\nguessing wastes turns.\n\n```\nlist_catalog(include_sample_values=true)\n    └─▶ fields[].name + sample_values  →  precise filter construction\n```\n\n## Recipes\n\n**RAG flow (natural-language Q&A):**\n\n```\nsemantic_search(query=...) → hits[i].doc_id, hits[i].offset\n    └─▶ read_content(doc_id, offset)\n```\n\n**Lookup by DOI:**\n\n```\nsearch_papers(filters_advanced=[{field: \"doi\", value: \"10.1038/...\"}])\n```\n\n**OA + year filter:**\n\n```\nsearch_papers(\n    year_from=2024,\n    filters_advanced=[{field: \"access_is_oa\", value: \"true\"}]\n)\n```\n\n**Scoped semantic search (constrained corpus):**\n\n```\nsemantic_search(\n    query=\"...\",\n    filters={\"author\": [\"Hinton\"],\n             \"publication_published_year\": {\"gte\": 2020}}\n)   # applied at recall time, server-side; AND across fields\n```\n\nSoft semantics: chunks missing that metadata are NOT excluded.\nFor a hard guarantee, or meta-only constraints (fwci, c\n\nArchive v0.13.0: 12 files, 19111 bytes\n\nFiles: manifest.json (21356b), README.md (2917b), scripts/_common.mjs (3137b), scripts/get_resource.mjs (766b), scripts/list_catalog.mjs (465b), scripts/list_paper_relations.mjs (284b), scripts/read_content.mjs (355b), scripts/search_papers.mjs (259b), scripts/semantic_search.mjs (1192b), skill-card.md (2280b), SKILL.md (8602b), _meta.json (138b)\n\nArchive v0.12.0: 12 files, 18377 bytes\n\nFiles: manifest.json (19643b), README.md (2917b), scripts/_common.mjs (3137b), scripts/get_resource.mjs (766b), scripts/list_catalog.mjs (465b), scripts/list_paper_relations.mjs (284b), scripts/read_content.mjs (355b), scripts/search_papers.mjs (259b), scripts/semantic_search.mjs (1192b), skill-card.md (2412b), SKILL.md (8253b), _meta.json (138b)\n\nArchive v0.11.2: 12 files, 16831 bytes\n\nFiles: manifest.json (16207b), README.md (2917b), scripts/_common.mjs (3137b), scripts/get_resource.mjs (766b), scripts/list_catalog.mjs (465b), scripts/list_paper_relations.mjs (284b), scripts/read_content.mjs (355b), scripts/search_papers.mjs (259b), scripts/semantic_search.mjs (1192b), skill-card.md (2344b), SKILL.md (7713b), _meta.json (138b)\n\nArchive v0.11.1: 12 files, 16088 bytes\n\nFiles: manifest.json (15801b), README.md (2917b), scripts/_common.mjs (3137b), scripts/get_resource.mjs (766b), scripts/list_catalog.mjs (465b), scripts/list_paper_relations.mjs (284b), scripts/read_content.mjs (355b), scripts/search_papers.mjs (259b), scripts/semantic_search.mjs (374b), skill-card.md (2094b), SKILL.md (7713b), _meta.json (138b)\n\nArchive v0.11.0: 12 files, 16156 bytes\n\nFiles: manifest.json (15801b), README.md (2917b), scripts/_common.mjs (3137b), scripts/get_resource.mjs (766b), scripts/list_catalog.mjs (465b), scripts/list_paper_relations.mjs (284b), scripts/read_content.mjs (355b), scripts/search_papers.mjs (259b), scripts/semantic_search.mjs (374b), skill-card.md (2508b), SKILL.md (7713b), _meta.json (138b)","readmeExcerpt":"Skill: sciverse academic retrieval Owner: sciverse Summary: Retrieve academic papers by structured metadata, perform semantic chunk search for RAG, and read byte-range content for citation-grade scientific literature. Tags: latest:0.14.3 Version history: v0.14.3 | 2026-09-20T04:01:30.789Z | auto - Version bump to 0.14.3. - Documentation updates in SKILL.md. - Removed skill-card.md file. v0.14.2 | 2026-09-11T03:04:28.","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"list_catalog(include_sample_values=true)\n    └─▶ fields[].name + sample_values  →  precise filter construction"},{"language":"text","snippet":"semantic_search(query=...) → hits[i].doc_id, hits[i].offset\n    └─▶ read_content(doc_id, offset)"},{"language":"text","snippet":"search_papers(filters_advanced=[{field: \"doi\", value: \"10.1038/...\"}])"},{"language":"text","snippet":"search_papers(\n    year_from=2024,\n    filters_advanced=[{field: \"access_is_oa\", value: \"true\"}]\n)"},{"language":"text","snippet":"semantic_search(\n    query=\"...\",\n    filters={\"author\": [\"Hinton\"],\n             \"publication_published_year\": {\"gte\": 2020}}\n)   # applied at recall time, server-side; AND across fields"},{"language":"text","snippet":"search_papers(..., fields=[\"doc_id\",\"title\"]) → collect doc_id\nsemantic_search(query=..., filters={\"doc_id\": [...]})\n    # hits never leave the set; empty list → empty hits (never global);\n    # up to 1000 deduped ids (400 SCOPE_TOO_LARGE beyond)"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: sciverse-academic-retrieval\nslug: academic-retrieval\nversion: 0.14.3\ndescription: Sciverse academic paper retrieval: structured metadata search, semantic chunk retrieval for RAG, and character-range content reading (offsets in Unicode code points). For agent workflows that need citation-grade scientific literature.\nlicense: Apache-2.0\nhomepage: https://sciverse.space\n---\n\n# academic-retrieval\n\nSciverse academic paper retrieval: structured metadata search, semantic chunk retrieval for RAG, and character-range content reading (offsets in Unicode code points). For agent workflows that need citation-grade scientific literature.\n\n## When to use\n\nTrigger this skill when the user's request involves any of:\n\n- Locating academic papers by structured criteria (authors, year, journal, subjects)\n- Grounding answers in paper excerpts (RAG / citations)\n- Expanding the original text around a known doc_id (more text before/after a chunk)\n\n## Authentication\n\nThis skill requires the `SCIVERSE_API_TOKEN` environment variable\n(obtain from https://sciverse.space). Optionally set `SCIVERSE_BASE_URL`\nto override the default API base URL.\n\n## Tools\n\n### search_papers\n\nSearch academic papers by structured filters (title, authors, journal,\nyear, subjects, etc.).\nUse when: \"find Hinton's papers from 2020-2023\", \"Nature papers on\nCRISPR\".\nNot for: natural-language Q&A retrieval (use semantic_search) or\nfull-text snippets (use read_content).\nReturns: list of papers; each entry has unique_id (always present),\ndoc_id (only when full text exists), title, author, abstract,\npublication_venue_name_unified, publication_published_year.\n\n**Invoke**: `node scripts/search_papers.mjs '<JSON args>'`\n\n### semantic_search\n\nNatural-language semantic search returning relevant paper chunks for\nRAG-style answering.\nUse when: \"How does Transformer attention work?\", \"What are recent\nmethods for protein structure prediction?\".\nNot for: precise field filtering (use search_papers) or fetching full\noriginal text (use read_content).\nReturns: list of chunks; each entry has chunk_id, doc_id, abstract,\nchunk, score, title, offset.\nTypical chain: semantic_search → pick chunk → read_content(doc_id,\noffset).\n\n**Invoke**: `node scripts/semantic_search.mjs '<JSON args>'`\n\n### list_catalog\n\nReturns the schema catalog for search_papers: every field name, type,\nwhether it's filterable / sortable, default-return status, human\ndescription, and applicable FilterOperators.\nUse when: \"Which field do I filter by DOI?\", \"What values can\naccess_oa_status take?\", \"What's the right enum for metadata_type?\".\nNot for: actually searching papers (use search_papers / semantic_search).\nTypical pattern: call once when first encountering Sciverse or facing\nan ambiguous field need, then construct precise search_papers filters\nfrom the returned schema.\nPass include_sample_values=true to also fetch top-20 values for\nenum-like fields (OpenSearch terms aggregation, 24h cached).\n\n**Invoke**: `node scripts/list_catalog.mjs '<"},{"path":"README.md","content":"# academic-retrieval — ClawHub skill bundle\n\n[![ClawHub](https://img.shields.io/badge/clawhub-academic--retrieval-brightgreen)](https://clawhub.ai/sciverse/skills/academic-retrieval)\n\nClawHub skill that gives any OpenClaw agent Sciverse academic-paper retrieval\ncapabilities (English | [中文](#中文说明)).\n\nPublished by **@sciverse** (slug `academic-retrieval`).\n\n## Install\n\n```bash\nopenclaw skills install academic-retrieval\n```\n\n## Configure\n\n```bash\nexport SCIVERSE_API_TOKEN=sv-xxx       # obtain from https://sciverse.space\n```\n\n## Tools at a glance\n\n| Tool | Purpose |\n|---|---|\n| `list_catalog` | Field introspection (call once to learn available fields + enum values) |\n| `search_papers` | Structured metadata search over papers / authors / sources (set `collection`) |\n| `semantic_search` | Natural-language semantic chunk retrieval (for RAG) |\n| `read_content` | Character-range read of a paper's original text (offset/limit in Unicode code points) |\n| `get_resource` | Fetch figure / table image bytes referenced inside `read_content` Markdown |\n\nSee `SKILL.md` for full agent-facing documentation.\n\n## Direct invocation (bypass OpenClaw)\n\n```bash\nnode scripts/semantic_search.mjs '{\"query\":\"Transformer attention mechanism\",\"top_k\":3}'\n```\n\n## Relationship to the SDK\n\nThis skill is **complementary** to the `sciverse` packages on PyPI / npm:\n\n- **This skill** — OpenClaw users only. Zero external deps (Node 18+ native fetch).\n- **PyPI / npm SDK** — Any LLM agent framework (OpenAI, Anthropic, LangChain, LlamaIndex…).\n\n## License\n\nApache-2.0\n\n---\n\n## 中文说明\n\nOpenClaw 用户专用：通过 ClawHub 一键给 agent 加上 Sciverse 学术文献检索能力。\n\n发布者 **@sciverse**，slug `academic-retrieval`。\n\n### 安装\n\n```bash\nopenclaw skills install academic-retrieval\n```\n\n### 配置\n\n```bash\nexport SCIVERSE_API_TOKEN=sv-xxx   # 从 https://sciverse.space 控制台申请\n# 可选：export SCIVERSE_BASE_URL=https://api-custom.sciverse.space\n```\n\n### 工具速览\n\n| Tool | 用途 |\n|---|---|\n| `list_catalog` | 字段 introspection（首次接入调一次，学习可用字段和 enum 取值） |\n| `search_papers` | 按结构化条件查 papers / authors / sources（用 `collection` 切换实体集合） |\n| `semantic_search` | 自然语言语义检索文献片段（RAG 用） |\n| `read_content` | 按 Unicode 码点区间读取文献原文片段 |\n| `get_resource` | 取 `read_content` Markdown 中引用的图片字节流（多模态 RAG） |\n\nagent 视角的完整文档见 `SKILL.md`（英文）。\n\n### 直接调用（不通过 OpenClaw）\n\n```bash\nnode scripts/semantic_search.mjs '{\"query\":\"Transformer 注意力机制\",\"top_k\":3}'\n```\n\n### 与 SDK 的关系\n\n本 skill 与 PyPI/npm 上的 `sciverse` 包是**互补**的：\n\n- **本 skill**：OpenClaw 用户专用，零外部依赖（仅 Node 18+ native fetch）\n- **PyPI/npm SDK**：任意 LLM Agent 框架（OpenAI / Anthropic / LangChain / LlamaIndex...）"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn74way11x0gjn6wpa8hcvyhvs85vmkj\",\n  \"slug\": \"academic-retrieval\",\n  \"version\": \"0.14.3\",\n  \"publishedAt\": 1789876890789\n}"},{"path":"skill-card.md","content":"## Description:\n\nSciverse academic paper retrieval supports structured metadata search, semantic chunk retrieval for RAG, citation relation lookup, and character-range content reading for scientific literature workflows.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[sciverse](https://clawhub.ai/user/sciverse)\n\n### License/Terms of Use:\n\nApache-2.0\n\n## Use Case:\n\nDevelopers and research-focused agents use this skill to locate academic papers, retrieve grounded text snippets, inspect citation relationships, and fetch referenced figures or tables for literature review and RAG workflows.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Search queries, document IDs, and resource names are sent to Sciverse endpoints with the user's API token.\n\nMitigation: Use a token approved for the intended workspace and avoid submitting confidential research prompts or identifiers unless Sciverse terms and internal policy allow it.\n\nRisk: Retrieved chunks, full text, citation relations, and images can be permission-limited, partial, or approximate for citation-grade use.\n\nMitigation: Check content accessibility and verify important claims or citations against the source publication before relying on downstream answers.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/sciverse/skills/academic-retrieval)\n- [Sciverse homepage](https://sciverse.space)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, JSON, images, configuration]\n\n**Output Format:** [JSON responses containing paper metadata, semantic chunks, Markdown text fragments, citation relation lists, or base64-encoded image resources]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Requires a Sciverse API token; Node 18+ scripts call Sciverse endpoints and print machine-readable JSON.]\n\n## Skill Version(s):\n\n0.14.3 (source: server release metadata and SKILL.md frontmatter)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."},{"path":"manifest.json","content":"{\n  \"name\": \"sciverse-academic-retrieval\",\n  \"version\": \"0.14.3\",\n  \"slug\": \"academic-retrieval\",\n  \"description\": \"Sciverse academic paper retrieval: structured metadata search, semantic chunk retrieval for RAG, and character-range content reading (offsets in Unicode code points). For agent workflows that need citation-grade scientific literature.\",\n  \"runtime\": \"node>=18\",\n  \"license\": \"Apache-2.0\",\n  \"homepage\": \"https://sciverse.space\",\n  \"env\": [\n    {\n      \"name\": \"SCIVERSE_API_TOKEN\",\n      \"required\": true,\n      \"description\": \"Sciverse API Token (obtain from https://sciverse.space).\"\n    },\n    {\n      \"name\": \"SCIVERSE_BASE_URL\",\n      \"required\": false,\n      \"default\": \"https://api.sciverse.space\",\n      \"description\": \"Override the default API base URL (for dev / self-hosted gateways).\"\n    }\n  ],\n  \"tools\": [\n    {\n      \"name\": \"search_papers\",\n      \"description\": \"Search academic papers by structured filters (title, authors, journal,\\nyear, subjects, etc.).\\nUse when: \\\"find Hinton's papers from 2020-2023\\\", \\\"Nature papers on\\nCRISPR\\\".\\nNot for: natural-language Q&A retrieval (use semantic_search) or\\nfull-text snippets (use read_content).\\nReturns: list of papers; each entry has unique_id (always present),\\ndoc_id (only when full text exists), title, author, abstract,\\npublication_venue_name_unified, publication_published_year.\",\n      \"script\": \"scripts/search_papers.mjs\",\n      \"input_schema\": {\n        \"type\": \"object\",\n        \"properties\": {\n          \"collection\": {\n            \"type\": \"string\",\n            \"enum\": [\n              \"papers\",\n              \"authors\",\n              \"sources\"\n            ],\n            \"default\": \"papers\",\n            \"description\": \"检索的实体集合。papers（默认，论文）/ authors（作者）/ sources（来源期刊）。 各 collection 字段集不同，用 list_catalog（collection=<name>）学习对应 schema。 注意：本工具的便捷字段（authors/journals/year_from/subjects 等）只对 papers 有意义； 查 authors/sources 时改用 filters_advanced + 该 collection 的字段名（如 authors 的 summary_stats.h_index / orcid，sources 的 issn / is_oa）。authors 用 orcid、 sources 用 issn 与论文检索结果关联。\",\n            \"x-en-description\": \"Entity collection to search. papers (default) / authors / sources. Each collection has its own field schema — call list_catalog(collection=<name>). The convenience fields (authors/journals/year_from/subjects) apply to papers only; for authors/sources use filters_advanced with that collection's field names.\"\n          },\n          \"query\": {\n            \"type\": \"string\",\n            \"description\": \"BM25 全文关键词，匹配标题/摘要/期刊名/关键词字段。留空则纯靠结构化过滤。\\n普通关键词是宽松匹配（任一词命中、按相关性排序）。\\n\\n也支持布尔检索式：全大写 AND / OR / NOT、括号分组、引号短语；优先级\\nNOT > AND > OR，相邻词隐式 AND。例如\\n  (histopathology OR pathology) AND (\\\"deep learning\\\" OR \\\"machine learning\\\") AND (prognosis OR survival)\\n布尔式里每个检索词都是硬条件、不做放宽——0 命中就是 0。\\n- query 只放检索词：「检索式1（预后预测）」这类标签/说明也会变成必须命中的词。\\n- 小写 and/or 是普通词；要检索字面量 OR（如比值比）加引号 \\\"OR\\\"。\\n- 引号需与运算符同时出现才生效：只写 \\\"spread through air spaces\\\" 不带运算符时\\n  按普通关键词处理；写成 \\\"spread through air spaces\\\" AND lung 才是短语精确匹配。"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":1932,"uniquenessScore":40,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T11:27:22.149Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T11:27:22.149Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T16:07:25.420Z","emptyReason":null},"items":[{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-10T18:48:31.762Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}