{"id":"3a563949-2c2a-4d6d-bf61-f270549b10c1","entityType":"agent","slug":"clawhub-tenequm-lance-format","name":"lance-format","canonicalUrl":"https://www.xpersona.co/agent/clawhub-tenequm-lance-format","canonicalPath":"/agent/clawhub-tenequm-lance-format","generatedAt":"2026-10-09T22:08:15.761Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T19:57:48.230Z","emptyReason":null},"description":"Deep reference for Lance v13 columnar format, its Rust crates, and pylance - file encodings, table format, indexes, schema evolution, time travel. Use when building on the Lance crates or reading .lance datasets, not the LanceDB product.","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 2.6K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s17bp3v1hm1dnkzey0c9tfh02183j0y5:lance-format","sourceUrl":"https://clawhub.ai/tenequm/lance-format","homepage":"https://clawhub.ai/tenequm/skills/lance-format","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/tenequm/lance-format","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/tenequm/skills/lance-format","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":68,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"lance-format technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T19:57:48.230Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T19:57:48.230Z","emptyReason":null},"stars":null,"forks":null,"downloads":2617,"packageName":null,"latestVersion":"0.20.0","tractionLabel":"2.6K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T12:51:59.743Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T19:57:48.230Z","lastCrawledAt":"2026-10-09T12:51:59.743Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T12:51:59.743Z","lastVerifiedAt":null,"highlights":[{"version":"0.20.0","createdAt":"2026-09-17T20:33:42.469Z","changelog":"Updated lance-format from 0.19.0 to 0.20.0. Changes: - modified `CHANGELOG.md` - modified `SKILL.md` - renamed `references/changelog-v7-v12.md` -> `references/changelog-v7-v13.md` - modified `references/docs/format/file/versioning.md` - modified `references/docs/format/index/index.md` - modified `references/docs/format/index/system/frag_reuse.md` - modified `references/docs/format/index/vector/index.md` - modified `references/docs/format/table/row_id_lineage.md` - modified `references/docs/format/table/transaction.md` - modified `references/docs/format/table/versioning.md` - modified `references/docs/guide/object_store.md` - modified `references/docs/guide/performance.md`","fileCount":62,"zipByteSize":404016},{"version":"0.19.0","createdAt":"2026-09-09T18:02:03.291Z","changelog":"Updated lance-format from 0.18.1 to 0.19.0. Changes: - modified `CHANGELOG.md` - modified `SKILL.md` - modified `references/changelog-v7-v12.md` - modified `references/docs/format/table/mem_wal.md` - modified `references/docs/guide/blob.md` - modified `references/docs/guide/object_store.md` - modified `references/docs/guide/performance.md` - modified `references/format-file.md` - modified `references/format-table.md` - modified `references/indexes.md` - modified `references/lance-reference.md` - modified `references/maintenance.md`","fileCount":62,"zipByteSize":376724},{"version":"0.18.1","createdAt":"2026-09-09T09:59:56.666Z","changelog":"Updated lance-format from 0.18.0 to 0.18.1. Changes: - modified `CHANGELOG.md` - modified `SKILL.md`","fileCount":62,"zipByteSize":358637},{"version":"0.18.0","createdAt":"2026-09-01T12:52:22.890Z","changelog":"Updated lance-format from 0.17.1 to 0.18.0. Changes: - modified `CHANGELOG.md` - modified `SKILL.md` - renamed `references/changelog-v7-v11.md` -> `references/changelog-v7-v12.md` - modified `references/docs/format/index/index.md` - modified `references/docs/format/table/data_overlay_file.md` - modified `references/docs/format/table/mem_wal.md` - modified `references/docs/format/table/versioning.md` - modified `references/docs/guide/object_store.md` - modified `references/docs/quickstart/vector-search.md` - modified `references/format-file.md` - modified `references/format-table.md` - modified `references/indexes.md`","fileCount":62,"zipByteSize":358639},{"version":"0.17.1","createdAt":"2026-08-21T12:07:09.979Z","changelog":"Updated lance-format from 0.17.0 to 0.17.1. Changes: - modified `CHANGELOG.md` - modified `SKILL.md` - deleted `skill-card.md`","fileCount":62,"zipByteSize":341584},{"version":"0.17.0","createdAt":"2026-08-21T10:17:30.448Z","changelog":"Updated lance-format from 0.16.0 to 0.17.0. Changes: - modified `CHANGELOG.md` - modified `SKILL.md` - modified `references/format-table.md` - modified `skill-card.md`","fileCount":62,"zipByteSize":341258},{"version":"0.16.0","createdAt":"2026-08-21T09:39:46.104Z","changelog":"Updated lance-format from 0.15.0 to 0.16.0. Changes: - modified `CHANGELOG.md` - modified `SKILL.md` - modified `references/changelog-v7-v11.md` - modified `references/docs/format/table/transaction.md` - modified `references/docs/guide/distributed_write.md` - modified `references/docs/guide/performance.md` - modified `references/docs/quickstart/versioning.md` - modified `references/format-file.md` - modified `references/format-table.md` - modified `references/indexes.md` - modified `references/lance-reference.md` - modified `references/maintenance.md`","fileCount":62,"zipByteSize":342017},{"version":"0.15.0","createdAt":"2026-08-12T11:39:10.271Z","changelog":"Updated lance-format from 0.14.1 to 0.15.0. Changes: - modified `CHANGELOG.md` - modified `SKILL.md` - modified `references/changelog-v7-v11.md` - modified `references/docs/format/index/vector/index.md` - modified `references/docs/format/table/mem_wal.md` - modified `references/docs/format/table/versioning.md` - modified `references/docs/guide/blob.md` - modified `references/docs/guide/data_types.md` - modified `references/docs/guide/distributed_write.md` - modified `references/docs/guide/object_store.md` - modified `references/docs/guide/observability.md` - modified `references/docs/guide/performance.md`","fileCount":62,"zipByteSize":328153}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17bp3v1hm1dnkzey0c9tfh02183j0y5:lance-format","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s17bp3v1hm1dnkzey0c9tfh02183j0y5:lance-format` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/tenequm/lance-format before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tenequm-lance-format/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tenequm-lance-format/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tenequm-lance-format/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-tenequm-lance-format/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-tenequm-lance-format/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-tenequm-lance-format/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-09T22:08:15.756Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tenequm-lance-format/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tenequm-lance-format/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tenequm-lance-format/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-tenequm-lance-format/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T19:57:48.230Z","emptyReason":null},"readme":"Skill: lance-format\n\nOwner: tenequm\n\nSummary: Deep reference for Lance v13 columnar format, its Rust crates, and pylance - file encodings, table format, indexes, schema evolution, time travel. Use when building on the Lance crates or reading .lance datasets, not the LanceDB product.\n\nTags: latest:0.20.0\n\nVersion history:\n\nv0.20.0 | 2026-09-17T20:33:42.469Z | user\n\nUpdated lance-format from 0.19.0 to 0.20.0.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- renamed `references/changelog-v7-v12.md` -> `references/changelog-v7-v13.md`\n- modified `references/docs/format/file/versioning.md`\n- modified `references/docs/format/index/index.md`\n- modified `references/docs/format/index/system/frag_reuse.md`\n- modified `references/docs/format/index/vector/index.md`\n- modified `references/docs/format/table/row_id_lineage.md`\n- modified `references/docs/format/table/transaction.md`\n- modified `references/docs/format/table/versioning.md`\n- modified `references/docs/guide/object_store.md`\n- modified `references/docs/guide/performance.md`\n\nv0.19.0 | 2026-09-09T18:02:03.291Z | user\n\nUpdated lance-format from 0.18.1 to 0.19.0.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- modified `references/changelog-v7-v12.md`\n- modified `references/docs/format/table/mem_wal.md`\n- modified `references/docs/guide/blob.md`\n- modified `references/docs/guide/object_store.md`\n- modified `references/docs/guide/performance.md`\n- modified `references/format-file.md`\n- modified `references/format-table.md`\n- modified `references/indexes.md`\n- modified `references/lance-reference.md`\n- modified `references/maintenance.md`\n\nv0.18.1 | 2026-09-09T09:59:56.666Z | user\n\nUpdated lance-format from 0.18.0 to 0.18.1.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n\nv0.18.0 | 2026-09-01T12:52:22.890Z | user\n\nUpdated lance-format from 0.17.1 to 0.18.0.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- renamed `references/changelog-v7-v11.md` -> `references/changelog-v7-v12.md`\n- modified `references/docs/format/index/index.md`\n- modified `references/docs/format/table/data_overlay_file.md`\n- modified `references/docs/format/table/mem_wal.md`\n- modified `references/docs/format/table/versioning.md`\n- modified `references/docs/guide/object_store.md`\n- modified `references/docs/quickstart/vector-search.md`\n- modified `references/format-file.md`\n- modified `references/format-table.md`\n- modified `references/indexes.md`\n\nv0.17.1 | 2026-08-21T12:07:09.979Z | user\n\nUpdated lance-format from 0.17.0 to 0.17.1.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- deleted `skill-card.md`\n\nv0.17.0 | 2026-08-21T10:17:30.448Z | user\n\nUpdated lance-format from 0.16.0 to 0.17.0.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- modified `references/format-table.md`\n- modified `skill-card.md`\n\nv0.16.0 | 2026-08-21T09:39:46.104Z | user\n\nUpdated lance-format from 0.15.0 to 0.16.0.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- modified `references/changelog-v7-v11.md`\n- modified `references/docs/format/table/transaction.md`\n- modified `references/docs/guide/distributed_write.md`\n- modified `references/docs/guide/performance.md`\n- modified `references/docs/quickstart/versioning.md`\n- modified `references/format-file.md`\n- modified `references/format-table.md`\n- modified `references/indexes.md`\n- modified `references/lance-reference.md`\n- modified `references/maintenance.md`\n\nv0.15.0 | 2026-08-12T11:39:10.271Z | user\n\nUpdated lance-format from 0.14.1 to 0.15.0.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- modified `references/changelog-v7-v11.md`\n- modified `references/docs/format/index/vector/index.md`\n- modified `references/docs/format/table/mem_wal.md`\n- modified `references/docs/format/table/versioning.md`\n- modified `references/docs/guide/blob.md`\n- modified `references/docs/guide/data_types.md`\n- modified `references/docs/guide/distributed_write.md`\n- modified `references/docs/guide/object_store.md`\n- modified `references/docs/guide/observability.md`\n- modified `references/docs/guide/performance.md`\n\nv0.14.1 | 2026-08-07T14:44:14.771Z | user\n\nUpdated lance-format from 0.14.0 to 0.14.1.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- modified `references/lance-reference.md`\n- modified `skill-card.md`\n\nv0.14.0 | 2026-08-07T13:43:08.461Z | user\n\nUpdated lance-format from 0.13.0 to 0.14.0.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- added `references/changelog-v7-v11.md`\n- added `references/format-file.md`\n- added `references/format-table.md`\n- added `references/indexes.md`\n- modified `references/lance-reference.md`\n- added `references/maintenance.md`\n- added `references/ops.md`\n- modified `references/performance.md`\n- modified `skill-card.md`\n\nv0.13.0 | 2026-08-07T12:26:21.014Z | user\n\nUpdated lance-format from 0.12.0 to 0.13.0.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- modified `references/docs/guide/migration.md`\n- modified `references/docs/guide/object_store.md`\n- modified `references/docs/guide/observability.md`\n- modified `references/docs/quickstart/full-text-search.md`\n- modified `references/lance-reference.md`\n- modified `references/performance.md`\n- modified `skill-card.md`\n\nv0.12.0 | 2026-07-30T14:46:37.453Z | user\n\nUpdated lance-format from 0.11.1 to 0.12.0.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- modified `references/docs/format/file/encoding.md`\n- modified `references/docs/format/index/scalar/fts.md`\n- modified `references/docs/format/index/system/mem_wal.md`\n- modified `references/docs/format/table/mem_wal.md`\n- modified `references/docs/guide/blob.md`\n- modified `references/lance-reference.md`\n- modified `references/performance.md`\n- modified `skill-card.md`\n\nv0.11.1 | 2026-07-22T18:45:00.126Z | user\n\nUpdated lance-format from 0.11.0 to 0.11.1.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- added `skill-card.md`\n\nv0.11.0 | 2026-07-22T08:31:42.558Z | user\n\nUpdated lance-format from 0.10.1 to 0.11.0.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- modified `references/docs/format/file/encoding.md`\n- modified `references/docs/format/file/versioning.md`\n- modified `references/docs/format/index/index.md`\n- modified `references/docs/format/index/scalar/bloom_filter.md`\n- modified `references/docs/format/index/scalar/fts.md`\n- modified `references/docs/format/index/scalar/zonemap.md`\n- added `references/docs/format/table/data_overlay_file.md`\n- modified `references/docs/format/table/index.md`\n- modified `references/docs/format/table/transaction.md`\n- modified `references/docs/guide/arrays.md`\n\nv0.10.1 | 2026-07-10T13:49:51.192Z | user\n\nUpdated lance-format from 0.10.0 to 0.10.1.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n\nv0.10.0 | 2026-07-08T08:55:12.898Z | user\n\nUpdated lance-format from 0.9.0 to 0.10.0.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- added `references/docs/format/file/encoding.md`\n- added `references/docs/format/file/index.md`\n- added `references/docs/format/file/versioning.md`\n- added `references/docs/format/index.md`\n- added `references/docs/format/index/index.md`\n- added `references/docs/format/index/indices-compaction.drawio.svg`\n- added `references/docs/format/index/indices-fragment handling.drawio.svg`\n- added `references/docs/format/index/scalar/bitmap.md`\n- added `references/docs/format/index/scalar/bloom_filter.md`\n- added `references/docs/format/index/scalar/btree.md`\n\nv0.9.0 | 2026-07-06T16:09:56.169Z | user\n\nUpdated lance-format from 0.8.0 to 0.9.0.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- modified `references/lance-reference.md`\n\nv0.8.0 | 2026-07-01T11:06:59.917Z | user\n\nUpdated lance-format from 0.7.0 to 0.8.0.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- modified `references/lance-reference.md`\n\nv0.7.0 | 2026-06-16T11:38:05.938Z | user\n\nUpdated lance-format from 0.6.0 to 0.7.0.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- modified `references/lance-reference.md`\n\nv0.6.0 | 2026-06-10T13:12:56.631Z | user\n\nUpdated lance-format from 0.5.0 to 0.6.0.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- modified `references/lance-reference.md`\n\nv0.5.0 | 2026-06-05T10:56:36.149Z | user\n\nUpdated lance-format from 0.4.0 to 0.5.0.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- modified `references/lance-reference.md`\n\nv0.4.0 | 2026-05-25T21:59:20.350Z | user\n\nUpdated lance-format from 0.3.0 to 0.4.0.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- modified `references/lance-reference.md`\n\nv0.3.0 | 2026-05-21T22:46:09.786Z | user\n\nUpdated lance-format from 0.2.0 to 0.3.0.\nChanges:\n- modified `CHANGELOG.md`\n- modified `SKILL.md`\n- modified `references/lance-reference.md`\n\nv0.2.0 | 2026-05-21T22:33:00.592Z | user\n\nUpdated lance-format from 0.1.0 to 0.2.0.\nChanges:\n- added `CHANGELOG.md`\n- modified `SKILL.md`\n- modified `references/lance-reference.md`\n\nv0.1.0 | 2026-05-20T16:07:48.874Z | user\n\nInitial publish of lance-format 0.1.0.\nChanges:\n- added `LICENSE.txt`\n- added `SKILL.md`\n- added `references/lance-reference.md`\n\nArchive index:\n\nArchive v0.20.0: 62 files, 404016 bytes\n\nFiles: CHANGELOG.md (57150b), LICENSE.txt (9157b), references/changelog-v7-v13.md (94958b), references/docs/format/file/encoding.md (47128b), references/docs/format/file/index.md (9612b), references/docs/format/file/versioning.md (3943b), references/docs/format/index.md (3514b), references/docs/format/index/index.md (16087b), references/docs/format/index/indices-compaction.drawio.svg (51382b), references/docs/format/index/indices-fragment handling.drawio.svg (21857b), references/docs/format/index/scalar_index.drawio.svg (8211b), references/docs/format/index/scalar/bitmap.md (1405b), references/docs/format/index/scalar/bloom_filter.md (5393b), references/docs/format/index/scalar/btree.md (3070b), references/docs/format/index/scalar/fmindex.md (4646b), references/docs/format/index/scalar/fts.md (19092b), references/docs/format/index/scalar/label_list.md (1729b), references/docs/format/index/scalar/ngram.md (1894b), references/docs/format/index/scalar/rtree.md (8350b), references/docs/format/index/scalar/zonemap.md (2986b), references/docs/format/index/starter-example.drawio.svg (14623b), references/docs/format/index/system/frag_reuse.md (11303b), references/docs/format/index/system/mem_wal.md (840b), references/docs/format/index/vector/index.md (21720b), references/docs/format/table/branch_tag.md (4730b), references/docs/format/table/data_overlay_file.md (17920b), references/docs/format/table/index.md (9772b), references/docs/format/table/layout.md (9479b), references/docs/format/table/mem_wal.md (35391b), references/docs/format/table/row_id_lineage.md (14498b), references/docs/format/table/schema.md (15517b), references/docs/format/table/transaction.md (31514b), references/docs/format/table/versioning.md (3842b), references/docs/guide/arrays.md (6630b), references/docs/guide/blob.md (15695b), references/docs/guide/data_evolution.md (8933b), references/docs/guide/data_types.md (14311b), references/docs/guide/distributed_indexing.md (7190b), references/docs/guide/distributed_write.md (11694b), references/docs/guide/json.md (12339b), references/docs/guide/migration.md (5426b), references/docs/guide/object_store.md (31197b), references/docs/guide/observability.md (3646b), references/docs/guide/performance.md (32076b), references/docs/guide/read_and_write.md (22785b), references/docs/guide/tags_and_branches.md (4492b), references/docs/guide/tokenizer.md (4001b), references/docs/integrations/datafusion.md (4071b), references/docs/quickstart/full-text-search.md (15573b), references/docs/quickstart/index.md (3366b), references/docs/quickstart/vector-search.md (9993b), references/docs/quickstart/versioning.md (3658b), references/format-file.md (38269b), references/format-table.md (66603b), references/indexes.md (59707b), references/lance-reference.md (1136b), references/maintenance.md (5922b), references/ops.md (26076b), references/performance.md (66208b), skill-card.md (2758b), SKILL.md (35903b), _meta.json (132b)\n\nFile v0.20.0:SKILL.md\n\n---\nname: lance-format\ndescription: Deep reference for Lance v13 columnar format, its Rust crates, and pylance - file encodings, table format, indexes, schema evolution, time travel. Use when building on the Lance crates or reading .lance datasets, not the LanceDB product.\nmetadata:\n  version: \"0.20.0\"\n  categories: \"development, integrations\"\n  topics: \"lance, columnar-format, vector-search, rust, lakehouse\"\n  upstream: \"lance-format/lance@v13.0.0-beta.4\"\n  openclaw:\n    homepage: https://github.com/tenequm/skills/tree/main/skills/lance-format\n    emoji: \"🗄️\"\n---\n\n# Lance v13 reference\n\nLance is an open columnar format for multimodal AI - \"a columnar data format that is 100x\nfaster than Parquet for random access.\" It is not one format but a stack of interoperating\nspecs: a **file format**, a **table format**, **index formats**, **catalog specs**, and a\n**namespace client spec**. The Rust workspace at `lance-format/lance` implements all of them\nplus Python (`pylance`) and Java bindings.\n\nThis skill tracks **`v13.0.0-beta.4`** (the `lance-format/lance` git tag), the current\ndevelopment frontier; **`v12.0.0`** is the stable pin, released 2026-09-17. Pin against tags, not\n`main` - Lance ships beta tags every few days and `next`-format encodings can change. Version\nlandscape below.\n\nThree layers of reference, load what the task needs:\n\n- **The deep reference** - any concrete schema, parameter, proto, or constraint. Split by topic:\n\n  | File in `references/` | Covers | Sections |\n  |------|--------|----------|\n  | `format-file.md` | What Lance is, the 26 crates, file format, data types | 1-4 |\n  | `format-table.md` | Dataset layout, manifests, fragments, schema evolution, versioning/tags/branches, row IDs, transactions + OCC, MemWAL | 5-10 |\n  | `indexes.md` | Vector / scalar / FTS / geo indexes, distributed builds | 11-12 |\n  | `ops.md` | Object store, capability matrix, source map | 13, 15, 16 |\n  | `changelog-v7-v13.md` | The full v7 -> v13 delta | 14 |\n\n  Cross-references written as \"section N\" resolve through `references/lance-reference.md`.\n- `references/performance.md` - ALL performance guidance. Part A routes to the official text and\n  adds the source-derived changes upstream has not documented; Part B is field-verified\n  remote-storage practice. Load for any performance, tuning, maintenance-cost, or \"why is this\n  slow\" question.\n- `references/docs/` - a **verbatim mirror of the official docs** (`docs/src` at the tracked\n  tag): every guide, quickstart, and format spec, unedited. Load when you need the full official\n  text. Directory map below.\n\n`references/maintenance.md` covers refreshing this skill against a new upstream tag.\n\n## Lance vs LanceDB\n\nThese are two different things and conflating them produces wrong answers.\n\n- **Lance** - the format and engine. The `lance-format/lance` repo; the `lance` /`lance-*`\n  Rust crates; `pylance`. It gives you datasets, the file/table format, indexes, commits,\n  scans. Consumed directly by DuckDB, Polars, Ray, Spark, PyTorch, DataFusion, or your own\n  Rust/Python code. **This skill is about Lance.**\n- **LanceDB** - a separate database *product* (`lancedb/lancedb`) built on top of Lance. It\n  adds a query-builder API, an embedding registry, rerankers-as-API, multi-language SDK\n  parity, and managed Cloud / Enterprise tiers. Not covered here.\n\n**The wider ecosystem** (separate repos, own version lines, none covered here): Flink streaming\nwrites (`lance-flink`), PostgreSQL reads via `pglance`, a Cypher graph engine (`lance-graph`), a\ndataset browser (`lance-data-viewer`), agentic context management (`lance-context`), and\nnamespace catalogs for Hive, Polaris, Gravitino, Unity Catalog, and AWS Glue.\nThe canonical docs site is **`lance.org`**. Generated per-language SDK docs live at\n`lance-format.github.io/lance-python-doc` for Python and\n[javadoc.io](https://www.javadoc.io/doc/org.lance/lance-core/latest/index.html) for Java - the\nmatching `lance-format.github.io/lance-java-doc` path 404s.\n\nLinking the `lance` crate in `Cargo.toml` means you are using Lance directly - use this skill.\nFor LanceDB internals, the storage layer underneath is still Lance, so this skill remains the\nauthority for the format itself.\n\n**The wrapper can hide format features.** LanceDB's `create_table` cannot enable stable row IDs;\nonly pylance's `write_dataset(enable_stable_row_ids=True)` can. If a format-level capability\nmatters to your design, check whether the wrapper exposes it before assuming the underlying\nformat settles the question - and reach for `pylance` directly when it does not.\n\n## The crate workspace\n\n26 crate directories under `rust/`. **`lance` is the public entry point** - `Dataset`, scanner,\nindexes, commits; everything else (`lance-table`, `lance-file`, `lance-encoding`, `lance-index`,\n`lance-io`, `lance-core`, `lance-datafusion`, `lance-linalg`, `lance-namespace*`, ...) is a layer\nbeneath it. Edition 2024, MSRV 1.91.0, arrow 58, datafusion 54; Python bindings need 3.10+. Full\ntable with roles, versions, and every workspace dep in `references/format-file.md` section 2.\n\n**If you depend on anything below `lance`, v11 will break you** - PRs #8020-#8026 deleted\n`lance-encoding::version` with no re-export (`LanceFileVersion` and `ConcreteFileVersion` both\nlive in `lance-file::version` now), removed `lance_io::encodings` and the `previous` namespaces,\nand gave each current format its own `versions/v2_{0,1,2,3}` module. Section 2.1.\n\nThe transaction code moved too (#8053/#8054/#8056): `rust/lance/src/dataset/transaction.rs` is\n**deleted**, replaced by a `rust/lance-table/src/transaction/` module tree (`builder`,\n`conflicts`, `operation`, `proto`, `manifest_build`, `validate`, `index_maintenance`,\n`row_version`, `update_map`). A `lance::dataset::transaction` shim still re-exports `Operation`,\n`Transaction`, `TransactionBuilder`, `RewriteGroup`, `UpdateMap` and friends, so the common\nsurface is unbroken - but a symbol the shim omits, or a citation of the old path, needs\nretargeting.\n\n## File format versions\n\nThe file format carries a single major.minor version. `data_storage_version` is set per dataset\nat creation - but as of `v12.0.0` it is **no longer fixed once the dataset exists**. It is the\n*default* for writes that omit a target, not a summary of what the dataset holds: \"Create and\noverwrite establish this default; append, update, merge-insert, and compaction do not change it.\"\nAn existing V2 dataset can take `\"2.0\"`, `\"2.1\"`, `\"2.2\"` or `\"2.3\"` per operation without\nrewriting its other files (#8582-#8585), so one dataset can hold data files at several exact V2\nversions. **V1 and V2 still cannot be mixed.** Section 3.\n\n| Version | Status | Notes |\n|---------|--------|-------|\n| `0.1` (`legacy`) | read-only | Original format; no longer writable |\n| `2.0` | stable | Removed row groups; null support for lists/FSL/primitives |\n| `2.1` | previous default | Adaptive structural encodings; better integer/string compression; nulls in struct fields; better nested random access. Was the default from Lance 5.0.0 until `v12.0.0-beta.15` |\n| `2.2` | **current default** (`stable`) | Map type, Blob v2, `VariablePackedStruct`, larger mini-blocks. Required for Map and Blob v2 |\n| `2.3` | unstable (`next`) | The current `next` alias target (`V2_3` in the enum). Ships **sparse structural pages**, which the 2.3 writer now auto-selects under a rep/def budget heuristic |\n\n**`stable` now resolves to 2.2, not 2.1** (#8657, beta.15), and `2.2` is the enum `#[default]`,\nso a dataset created without an explicit `data_storage_version` is written as 2.2. The change\nreaches new-dataset creation through `DataStorageFormat::default() -> stable_file_version()`,\nand Python's `write_dataset` inherits it because its default routes through `stable`. **The docs\nwere not updated with it** - `format/file/versioning.md` still only says `stable` is an \"alias\nfor the default version\", so the code is the authority here. `next` resolves to 2.3. Pin an\nexplicit number for deterministic behavior across builds.\n\n2.3 is the only version the code flags unstable; 2.2 never was, and is now what you get by\ndefault. The release *selectors* (`LanceFileVersion`) are a type distinct from the persisted\nidentity (`ConcreteFileVersion`). Details, plus the sparse auto-selection rules, in\n`references/format-file.md` sections 3.1 and 3.6.\n\n## Version landscape\n\nThe major is bumped by a bot, not a human: `ci/publish_beta.sh` re-roots at `MAJOR+1` whenever\nany PR since the release root carries the GitHub `breaking-change` label - the marker is the\n**label**, not a conventional-commit `!`. A major bump therefore means \"some labeled breaking\nchange landed\", not a redesign, and a `!` without the label bumps nothing. It has now fired on\n**four consecutive lines**, which is why **none of `v9.1.0`, `v10.1.0`, `v11.1.0`, or `v12.1.0`\nwas ever released**. The 12.1 line is the clearest case: `main` took a\n`chore: bump main to 12.1.0-beta.0` commit, and four commits later the bot re-rooted to 13, so\n`release-root/12.1.0-beta.N` and `release-root/13.0.0-beta.N` are the **same base commit**\n(`c3c9632a2`) and no `v12.1.0-beta.*` tag exists.\n\nThree recent lines **did** ship a final: `v10.0.0` (2026-08-08), `v11.0.0` (2026-08-30) and\n`v12.0.0` (2026-09-17). Each sits on a stabilization branch that is **not an ancestor of `main`**\n- normal for a Lance final, not a sign the release is unofficial.\n\n| Major | Its breaking theme |\n|-------|--------------------|\n| **v13** (current, `v13.0.0-beta.4`) | `WriteParams` gained `file_writer_options` (#9192 - the one labeled PR that re-rooted the major); lazy page-metadata init changed the `StructuralFieldScheduler` signature and the metadata **cache key shape** (#7465); `json_extract`/`json_get` no longer route to JSON indices (#9101). Delta below |\n| **v12** (`v12.0.0`, 2026-09-17) | `WrappingObjectStore` implementors must add `wrap_paginated` (no default); MemWAL `ShardManifestStore` renamed and narrowed; `lance-namespace` returns response objects; external stores gained predecessor-conditioned publication; namespace merge-insert keys became a list; the caller-provided Writer / `open_part` flow was removed (#9072). Unlabeled but bigger: `stable` -> 2.2 and the IVF_RQ 5-bit default. Net-new format capability: mixed data-file versions. Delta below |\n| **v11** (`v11.0.0`, 2026-08-30) | Fragment ids became a dataset-lifetime high-water mark; large internal reorganization of `lance-file` / `lance-encoding`; the first new manifest feature flag since v7 - which was then **reallocated before the final**. Net-new: covering indexes, `merge_insert` `write_mode`, row-address prefilter. Delta below |\n| **v10** | Blob APIs preserve null selections; cache keys became opaque BLAKE3 digests (every warm or persisted cache cold-misses, no legacy fallback); async `create_remapper`; MemWAL renamed generation -> SSTable, merge -> compaction (wire-compatible, symbol-breaking) |\n| **v9.1** (never released; renamed into v10) | FTS/inverted creation took a `block_size` param. Net-new: Data Overlay Files (cell-level updates without base-file rewrite, unstable + env-gated), sparse structural pages, `lance-index-core` |\n| **v9** | Python 3.9 dropped; `alter_columns` fails fast when casting an indexed column; FM-Index proto rename made existing FM indexes unreadable; FTS/inverted defaults to on-disk format v2 |\n| **v8** | All index builds unified onto one segment-based lifecycle. Net-new: `lance-derive`, FM-Index, multi-bit IVF_RQ, public `approx_mode`, TOS + GooseFS object stores |\n| **v7** | MemWAL, branches, the geo/RTree index, the `lance-select` crate, ICU FTS |\n\n**`v12.0.0` is the stable pin** and what GitHub Releases marks `Latest`. crates.io carries\n**finals only** (newest `lance 12.0.0`, no 13.x); PyPI `pylance` is likewise at `12.0.0`. So a\nbeta pin means a git dependency - beta wheels publish to fury.io instead, under the renamed org\n(`https://pypi.fury.io/lance-format`), which currently carries `pylance-13.0.0b1` through `b4`.\n\nFull per-tag deltas with every PR citation: `references/changelog-v7-v13.md`.\n\n## The v11 delta\n\n357 commits from `v10.0.0-beta.7` to the `v11.0.0` final, with **16 `breaking-change`-labeled\nPRs** (14 through `beta.16`, plus #8407 and #8535 in the final). Most structural invariants held:\n**26 crates**, **16 transaction ops**, `CommitConfig.num_retries` **20**, arrow 58 / datafusion 54,\nMSRV 1.91.0, Edition 2024, Python 3.10+ - and all of them still hold at `v13.0.0-beta.4`.\n\n**`references/changelog-v7-v13.md` has the full delta** - every PR citation, the per-tag\nbreakdown from v7 forward, the Python/Java surface, and each correctness fix with its trigger\ncondition. Load it for any \"what changed / will this break me\" question. What follows is only\nwhat bites hardest.\n\n**Five things that break you at v11:**\n\n- **Fragment ids are a dataset-lifetime high-water mark** (#8206) - a *format* invariant, not\n  just an API. Overwrite no longer restarts ids at 0, an overwrite fragment carrying a deletion\n  file is rejected, and any commit producing duplicate ids is rejected - so datasets written by\n  Lance 0.16 and earlier may still read but no longer commit. `dataset.get_fragment(0)` after an\n  overwrite must read ids from the manifest. Section 5 - which also covers a resolution hazard on\n  pre-0.10 unsorted manifests that can make a fragment-filtered index cover the wrong fragments.\n- **The file-version types and reader/writer composition moved** (#8020-#8026) -\n  `lance-encoding::version` deleted with no re-export; `LanceFileVersion` lost `PartialOrd`/`Ord`\n  (#8027, #8028), so `v >= LanceFileVersion::Next` no longer compiles. `FileWriter` is now an\n  enum with all constructors removed. Most of these break silently at compile time. Section 3.6.\n- **Transaction code moved to `lance-table`** (#8053/#8054/#8056) - see the crate-workspace note\n  above; the `lance::dataset::transaction` shim covers the common surface.\n- **`Operation::Project` / `Merge` gained `preserves_nullability`** (#8347) - a nullability\n  *tightening* must not set it, and such a projection now conflicts with any concurrent\n  value-write. This closed a real hole where `alter_columns` could let a racing write land nulls\n  unreadable under the tightened schema. Section 9.2.\n- **The external-manifest protocol changed** (#8499) - object storage is authoritative, the\n  external store's put-if-not-exists is a *reservation*, and a stored ETag must be **ignored**;\n  a retained one makes readers reject a good manifest with `Manifest e_tag mismatch`. Section 9.\n\n**The manifest feature flags changed - and bit 128 was reallocated before the final.** v11 added\nthe first new bit since v7 and moved `FLAG_UNKNOWN` 128 -> 256. But the bit it added,\n`FLAG_MEM_WAL_INDEX_CATCHUP`, was **retired again** (#8680) and the reclaimed bit handed to\n`FLAG_COVERED_INDEX_METADATA = 128` (#8535) before `v11.0.0` shipped. At the final and at v12\nthere is no index-catchup flag and no `require_index_catchup` proto field; a shard absent from\n`index_catchup` now unconditionally means *unknown*. Both reader and writer must hold bit 128 or\nrefuse the table. Section 7.\n\n**Do not pin anywhere in `v11.0.0-beta.4` through `beta.17`.** Those builds treat bit 128 as a\nMemWAL flag they support, so they *open* a covering-index dataset instead of refusing it - wrong\nneighbours, no error. The exposure is inherited by whichever flag takes the bit.\n\n**Covering indexes are the v11 net-new format feature** (#8535), **redefined at v13** (#8856).\n`IndexMetadata.covering_fields` (proto field 11) names the columns an index *carries* values for,\nso a query projecting only those columns is answered without a base-table take. It is **no longer\na trailing suffix of `fields`**: it \"must be a subset of `fields`, in the order the index emits\nthem. A column is carried if and only if it is named here\", including a column the index is also\nkeyed on - and `fields[0]` remains a keyed column. Index invalidation stays wide: **any** index\nwhose `fields` include the updated column, \"whether the index is keyed on it or merely carries\nit\".\n\nThe old \"no index builder writes carried values yet\" no longer holds. V3 IVF auxiliary files can\nphysically carry columns, and \"a reader discovers carried columns by exclusion, not by position:\nany column in the auxiliary file's schema that is not one of the quantizer's internal columns is\na carried column\", bound to source fields by a new `covering_field_ids` metadata key. Coverage is\nnow per-segment, not per-index: \"one logical index may hold values for some of its segments and\nnot others\". `VectorQueryProto.covering_projection` (field 15) reserves the query-side tag, where\nabsent / present-and-empty / present-and-non-empty are three distinct meanings. Section 11.\n\n**Bit 8 was spent in the `v12.0.0` final.** `FLAG_MIXED_DATA_FILE_VERSIONS = 1 << 8` (256) is no\nlonger a reservation pinned equal to `FLAG_UNKNOWN`: the assert relaxed to\n`FLAG_MIXED_DATA_FILE_VERSIONS < FLAG_UNKNOWN`, `FLAG_UNKNOWN` moved `1 << 8` -> `1 << 9` (512),\nand the build now both reads and writes mixed-version datasets. It is still carried by\n`STICKY_PAIRED_FLAGS`, and a **half-set** manifest is now a hard error: \"Manifest has only one of\nthe mixed data-file-version reader and writer feature bits set, so its semantics are undefined\".\nSection 7.\n\n**Bit 1024 is where the docs and the code disagree - trust the code.**\n`FLAG_FRAGMENT_REUSE_INDEX = 1 << 10` is declared at `rust/lance-table/src/feature_flags.rs:69`\nand, at `v13.0.0-beta.4`, **that declaration is its only occurrence in the entire tree**. It sits\n*above* `FLAG_UNKNOWN` (512), and the supported set is computed as `FLAG_UNKNOWN - 1`, so a\nmanifest setting it is **refused**. The spec page meanwhile lists it as reader `Yes` / writer\n`Yes` and puts the unknown boundary at 2048. The docs describe the intended end state; the code\nhas only reserved the constant. Anything you build against tagged FRI today is building against\nprose, not behavior.\n\n**Two `LANCE_*` env vars landed** (from the AMX work, #8540): `LANCE_DISABLE_AMX` (runtime kill\nswitch) and `LANCE_AMX_FP16_CC` (build-time compiler override). Grep trap: `LANCE_AMX_CFG_*` and\n`LANCE_AMX_TILE_COUNT` are **C macros in `amx_fp16.c`, not env vars**, and `LANCE_FACTOR` is a\nsubstring of `BALANCE_FACTOR` - a plain `LANCE_*` grep reports all four as if they were real.\n\n**Worth knowing without reading the full delta:** FTS gained a document-boundary axis\n(`DocumentGranularity`, #7788) whose `list_element` mode is a third trigger requiring FTS on-disk\nformat v3; transactions above **20 MiB** spill out of the manifest entirely (#7881); MemWAL\ncatch-up became derived rather than declared (#8481); transaction proto field 9 is deprecated for\nfield 10 (#7432); compaction gained row/byte budgets plus fragment exclusion (#8235, #8532);\n`merge_insert` gained `write_mode` (#8423); and Python commit conflicts became\n`lance.commit.CommitConflictError`, a subclass of `OSError`, so existing handlers keep working\n(#8563). Full list with citations in `references/changelog-v7-v13.md`.\n\n**Address-domain indexes stopped falsely claiming compacted fragments** (v11, `beta.16` or\nearlier). A rewrite used to advance *every* index's `fragment_bitmap` onto the new fragment ids -\nincluding ZoneMap, whose stored addresses point into the fragments the rewrite dropped. The\nRewrite path now branches on `results_are_row_addrs()`: an address-domain index gets\n`drop_rewritten_fragments` and a full-scan fallback, correct-but-slower instead of stale\naddresses. **Heals only for new compactions**: an index already damaged under v10 or earlier must\nbe recreated, and the damage does not self-heal through routine maintenance, because the\nrefreshed `fragment_bitmap` also makes incremental folds a no-op. Section 11.\n\n**Correctness fixes split by whether upgrading is enough.** Most are read-path only and heal on\nupgrade. These do **not** - they need data rewritten or repaired: #8382, #8669, #8509, #7703,\n#8539, #8459, #8378, #8482, #8834 (rebuild HNSW - a persisted graph can hold edges to ids it does\nnot contain; lost recall stays lost), #8101 (**nullable primary keys silently duplicated rows** on\nevery repeat `merge_insert`; existing duplicates must be removed by hand), #8511, #8427, #8513,\n#8839, #8904. Conditions for each in `references/changelog-v7-v13.md`.\n\n## The v12 delta\n\n**225 commits** from `release-root/12.0.0-beta.N` to the `v12.0.0` final, with **7\n`breaking-change`-labeled PRs** - the five visible at beta.15 plus **#9072** and **#9101** in the\nrun-up to the final. No new index types and no new crates; every structural invariant above still\nholds. **The label is a floor, not a ceiling** - the two biggest behavior changes in the line\ncarry a conventional-commit `!` but no label, so the bot never counted them: the `stable` -> 2.2\nmove (#8657, above) and the IVF_RQ 5-bit default (below).\n\n- **`WrappingObjectStore` implementors must add `wrap_paginated`** (#8606) - \"There is\n  deliberately no default: getting this wrong is either a silent loss of speed or a silent loss\n  of the wrapper, and neither announces itself.\" Return `Some` to keep listing pushdown through\n  the wrapper, `None` to give it up and fall back through `inner`. One wrapper giving it up gives\n  it up for the whole chain. Anything wrapping the object store fails to compile until updated.\n- **New paged listing: `ObjectStore::read_dir_page`** (#8606) - one page of a prefix's immediate\n  children plus an opaque resume token. The trap: \"One page is one request, so a page can hold\n  fewer children than `limit` asked for and still be followed by more\" - walk until the token is\n  `None`, never until a page comes back short.\n- **MemWAL `ShardManifestStore` renamed and narrowed** (#8640) - `read_latest` -> `latest`,\n  `read_latest_uncached` -> `refresh_latest`, and `write` is now crate-private (reach it through\n  `commit_update`, `claim_epoch`, or `initialize_shard`). Existing `commit_update` closures need\n  no change. Section 10.\n- **`lance-namespace` 0.8.5 -> 0.11.1** (#8903) - four `LanceNamespace` methods now return\n  response objects instead of bare values: `count_table_rows` -> `CountTableRowsResponse`,\n  `query_table` -> `QueryTableResponse`, `namespace_exists` / `table_exists` -> their own\n  response types. Callers unwrap; anyone implementing the trait needs the same signature updates.\n- **External manifest stores gained predecessor-conditioned publication** (#8800) -\n  `put_if_predecessor` reserves a version only while the predecessor still carries the identity\n  the writer observed, and `commit_after` refuses with `PrerequisiteFailed`, \"never a conflict\".\n  The hard compile break is the new `ManifestLocation.identity` field, not the trait methods\n  (all default-implemented). No built-in store implements it. Section 9.\n- **Namespace merge-insert keys became a list** (#8915) - `on` moves from `Option<String>` to\n  `Option<Vec<String>>`, with **arity-dependent NULL semantics**: a single-column key treats NULL\n  as equal to NULL, a composite key uses SQL equality, \"under which a NULL key matches nothing -\n  not even a byte-identical NULL\".\n\n**The `lance-namespace` pin is no longer one number.** #8915 moved the Rust client to **0.12.0**\nand #8979 moved **Java** to 0.12.0 as well; only **Python** still holds `>=0.11.1,<0.12`, because\nits generated models still send `on` as a bare string. Quote a language-specific pin, never one\nnumber for all three - and note this split moved once already, so re-check it rather than\ncarrying the pairing forward.\n\n**IVF_RQ now defaults to 5 bits per dimension, not 1** (#8936) - roughly a **4.4x index-size\nincrease** at the default (upstream's 100M x 768d example: ~10.8 GiB -> ~47.3 GiB). `Fast` search\nmode \"uses only the 1-bit sign code even when the index stores additional bits\", so it pays the\nstorage without using it; set `num_bits=1` explicitly to opt out, at the cost of the multi-bit\ndistance estimate and some recall. Sizing formulas in `references/indexes.md`.\n\n**Column slice stitching (#8660) was reverted** at beta.9 (#8926) - it \"should not ship while the\ncaller-managed replacement in #8923 is being developed\". `rust/lance-file/src/concat.rs` exists\nagain at beta.15, but holds #8923's caller-managed data file parts, not the reverted stitching.\n\n**Two proto additions.** MemWAL `SsTable` gained `in_memory_bytes`, `physical_rows` and\n`primary_key_bytes` (fields 3-5, #8981); all optional, and **absence must not be read as zero**.\n`FilteredReadOptions` gained `materialization_readahead_bytes` and `batch_size_bytes` (13, 14) -\nnot a format change under a new rule in `protos/AGENTS.md`: execution-plan schemas \"are wire\ncontracts, not persisted Lance formats\". `transaction.proto` / `ann.proto` / `index.proto` are\nuntouched.\n\n**Net-new, non-breaking:** provider-native bulk copy and a deep-clone concurrency bound (section\n13); Python `ObjectStoreProvider` registration (#8522); `BinaryView` in the packed blob writer\n(#8700); caller-managed data file parts (#8923); cleanup of specific versions (#8617);\n`LanceDataset.slice()` (#8059) and six more `LanceFragment.scanner` options (#8429); restored\nPython index retraining (#8786); namespace-managed clone deprecated to a shim (#8964). Namespace\nlatest-version resolution no longer lists the whole `_versions/` prefix (#8679) - on a\n~340k-version table that was ~344 list pages, \"~25s of pure I/O wait\", paid by every open.\n\n**Fixes needing a rebuild or rewrite, not just an upgrade:** #8779 (rebuild NGRAM indexes), #8510\n(rewrite data compacted from uniformly reordered fragments), #8984 (re-drop a resurrected index),\n#8837 (repair a MemWAL shard below ~2.7KB/row - it cannot be reopened). Full per-PR conditions,\nplus the much longer list that *does* heal on upgrade, in `references/changelog-v7-v13.md`.\n\n**Mixed data-file versions LANDED** - it is no longer \"1 of 6\". #8581-#8584 shipped in `v12.0.0`\n(validation, per-operation V2 write targets, propagation across dataset operations, compaction\ntargeting) and #8585 exposed it in the bindings in the v13 line. The proto changed with it:\n`DataStorageFormat.version` is now \"the default format version used when writing data files\",\nand \"each DataFile's version is authoritative for decoding\" once the capability is set.\n\n**In flight, not landed - do not treat as shipped:** generic block v5 compression is **still 1 of\n10** PRs merged (#8324; #8325-#8333 all remain open). Next big dependency break in the queue:\n#8997, \"upgrade to arrow 59, DataFusion 55, and pyo3 0.29\" - **still open** at `v13.0.0-beta.4`,\nso arrow 58 / datafusion 54 still hold. It also gates two outstanding PyO3 advisories\n(RUSTSEC-2026-0176/0177); rustls was separately patched to 0.23.45 for RUSTSEC-2026-0285 (#9212).\n\n## The v13 delta\n\n**66 commits** from `release-root/13.0.0-beta.N` to `v13.0.0-beta.4`, with **2\n`breaking-change`-labeled PRs**. No new crates and no new index types; 26 crates, 16 transaction\nops, `CommitConfig.num_retries` 20, arrow 58 / datafusion 54, MSRV 1.91.0, Edition 2024 and\nPython 3.10+ all still hold.\n\n**The `!`-vs-label rule inverted in this window.** All three conventional-commit `!` commits\n(#7465, #9192, #9101) *do* carry the `breaking-change` label. Keep treating the label as a floor\nrather than a ceiling - but this window is the counter-example, not more evidence for the gap.\n\n- **`WriteParams` gained `file_writer_options`** (#9192) - the single labeled PR that re-rooted\n  the major. `FileWriterOptions { data_cache_bytes, max_page_bytes, keep_original_array }` is now\n  reachable from the dataset write APIs in Rust, Python and Java. A zero `max_page_bytes` is\n  rejected before encoder construction rather than misbehaving later. Anything constructing\n  `WriteParams` by struct literal fails to compile.\n- **Page metadata is initialized lazily, and the metadata cache key changed shape** (#7465).\n  `StructuralFieldScheduler::initialize` now takes `requested_ranges`, and the page-scheduler\n  `initialize` splits into `init_ranges()` and `init_from_buffers(buffers, io)` - any external\n  implementor fails to compile. The public `DecodeBatchScheduler::try_new` kept its signature;\n  the range-aware entry point is the crate-private `try_new_with_ranges`. The part that bites\n  without a compile error: caching moved from a per-column `FieldDataCacheKey` to a per-page\n  `PageDataCacheKey { column_index, page_index, view_tag }`, so **every warm or persisted\n  metadata cache cold-misses** across this upgrade. The payoff is real - \"a cold point/range\n  read's metadata IO is invariant to the column's total page count\".\n- **`json_extract` and `json_get` no longer route to JSON indices** (#9101). Only the four typed\n  accessors (`json_get_int` / `_float` / `_bool` / `_string`) reach the index; everything else\n  falls back to a full scan. This fixes three real wrong-answer bugs - a quoted-key mismatch that\n  \"searched for a quoted key and matched nothing\", a `Utf8` literal driving an `Int64` btree into\n  a panic, and an unsound range because \"quoting is not order-preserving (`ab` < `ab!` but\n  `\"ab\"` > `\"ab!\"`)\". **The cost is silent**: a `json_extract` workload that used to hit an index\n  now scans, with no error and no plan warning. Rewrite those predicates onto the typed\n  accessors.\n\n**The Fragment Reuse Index gained a versioned on-disk contract** (#9136). `InlineContent` field 1\nwas renamed `versions` -> `legacy_versions` and a tagged `transitions` list added at field 2,\ngated on `index_version >= 1`; mappings are now a oneof of `OrderedCompaction` or\n`StablePartition`. A stable partition \"assigns source rows to destination fragments while\npreserving their relative source order within each destination\", which lets FRI reuse existing\nindices after **reclustering** - a second use case the v0 model had no concept of. Its physical\nform is an immutable row-map Lance file with `uint16` labels and an `LSPC`-magic counts matrix.\nTwo hard rules: **stable row IDs and tagged FRI are mutually exclusive** (\"writers must not\npublish `index_version >= 1` on them\"), and cleanup \"must retain intermediate transitions still\nneeded to translate old addresses\". Upstream also softened the old claim - \"FRI does not remove\nconflicts between overlapping rewrites\". Section 11.\n\n**Net-new, non-breaking:** `Dataset::frag_reuse_index()` is public (#9112) and documented in the\nperformance guide; `FileFragment::write_overlay` returns a real `OverlayWriter` (#8761, still\nenv-gated); an `hf://` object store with `hf_enable_resolve_cache` (#9236); Python\n`lance.bitmap.Bitmap`, `deep_clone()` (#9181), `base_paths()` (#9191) and\n`update_columns(with_offsets=True)` (#8891); Java `DataStorageVersion`, `FileWriteOptions` and\n`ScanOptions.indexSegments`; and namespace table listing finally bounded by `read_dir_page`\n(#9165). `inline_optimization_enabled` **flipped `true` -> `false`** (#9180), which upstream\njustifies with a -49% write-p50 measurement at 1M entries.\n\n## Performance questions\n\nFor anything performance-shaped - slow scans or searches, remote/object-storage cost, index\nmaintenance cost, memory sizing, version bloat, benchmarking - load\n`references/performance.md` first. Part A routes to the official guidance plus the undocumented\nsource-derived changes; Part B is field-verified practice against S3-compatible storage. The\ngoverning rule stays **minimize remote calls** - fewer commits, fewer scans, fewer round trips -\nbecause that is where the order-of-magnitude wins are. The official **\"Tuning remote scans\"**\nsection (v11, unchanged at v12) gives a starting point for cross-region or public-internet\naccess, where the cloud default of 64 concurrent requests is too aggressive: `LANCE_IO_THREADS=8`,\n`fragment_readahead=1`, `batch_readahead=2`, `io_buffer_size=64MB`. It is a legitimate second\nmove once call volume is already minimized.\n\n**AMX-FP16** (#8540, beta.16) is the one v11 performance change that alters *results*, not just\nspeed: where it engages, IVF partition assignment becomes **exact instead of approximate**, so\nrecall improves *and* assignments differ from an older build. It is shape-gated (`float16` +\n`dot`, `dimension >= 32`, `num_centroids >= 32`); everything else keeps the previous path.\n`LANCE_DISABLE_AMX=1` disables it, but reverts assignment to the approximate path too - so an\nindex built with it set is not equivalent to one built without it.\n\nTwo cache facts to know before tuning anything remote: Lance has **no resident data cache** (a\n`Session` holds only index and metadata caches, never decoded values, so repeated point reads\nre-pay object-store IO), and one `Arc<Session>` shared via `DatasetBuilder::with_session` lets\ndatasets share it. Cold first search is dominated by paging indexes in - `prewarm_index` is the\nremedy. Note that **#7465 changes the metadata cache key shape**, so the first run after a v13\nupgrade re-pays that paging even against a warm or persisted cache. Details and build-time\nrequirements in `references/performance.md`.\n\n**Time travel is not an archive mechanism.** Versions look like free history, but the default\ncleanup reclaims anything older than 7 days and cleanup is part of routine optimize - so a design\nthat treats old versions as the durable record loses it on the first maintenance pass. Keep an\nexplicit archive if you need one.\n\n## Official docs mirror\n\n`references/docs/` mirrors `docs/src` of `lance-format/lance` at the tracked tag, verbatim -\n45 markdown files plus 4 diagrams, all directly readable.\n\n| Directory | Files | Covers |\n|-----------|-------|--------|\n| `guide/` | 14 | CRUD, performance, object store, distributed write + indexing, JSON, tokenizers, data types, data evolution, blob, arrays, tags/branches, migration, observability |\n| `quickstart/` | 4 | First dataset, vector search, full-text search, versioning |\n| `format/` | 1 | Spec-stack overview |\n| `format/file/` | 3 | Container spec, structural encodings + compression, format versions |\n| `format/table/` | 9 | Layout, schema, transactions (**conflict-resolution matrix**), versioning, row-id lineage, branch/tag, MemWAL, data overlay files |\n| `format/index/` | 1 + 4 svg | Index lifecycle, fragment coverage, compaction interplay |\n| `format/index/scalar/` | 9 | fts, fmindex, ngram, btree, bitmap, bloom_filter, label_list (`array_has_any/all`), zonemap, rtree |\n| `format/index/vector/` | 1 | IVF / PQ / SQ / RQ / HNSW concepts and storage layout |\n| `format/index/system/` | 2 | Fragment reuse index, MemWAL system index |\n| `integrations/` | 1 | DataFusion SQL over Lance, incl. JSON functions |\n\n**Not mirrored:** `docs/src/images/` (PNG/GIF assets), so image links in the mirrored pages do\nnot resolve - the prose is self-contained, and the four `.drawio.svg` diagrams *are* mirrored.\nAlso out by design: `community/`, `examples/`, `integrations/{index,pytorch,tensorflow}.md`; and\nthe landing stubs and contributor files (`format/AGENTS.md`, `format/CLAUDE.md`).\n\n**A whole tier of docs is not in this repo at all**, so it cannot be mirrored and cannot be\nenumerated from a clone. `docs/make-full-website.sh` assembles `format/catalog`,\n`format/namespace`, and the `integrations/{duckdb,huggingface,spark,ray,trino,context}` sections\nat build time from six sibling repos with their own version lines - the checked-in\n`integrations/index.md` links `spark/`, `duckdb` and `trino` as if they were local, but those\npaths do not exist in the tree. **Lance Context** and the **HuggingFace** integration docs are\nwhole nav sections that exist only on the built site. For any of those, read `lance.org` rather\nthan this mirror. Protobuf message bodies are likewise expanded at build time from `protos/` by\n`mkdocs_protobuf`, so the mirrored spec pages show `%%% proto.message.X %%%` placeholders where\nthe site shows a rendered schema.\n\nFile v0.20.0:_meta.json\n\n{\n  \"ownerId\": \"kn76gpsgjw5chv0xvzbzcb8cxn81x46r\",\n  \"slug\": \"lance-format\",\n  \"version\": \"0.20.0\",\n  \"publishedAt\": 1789677222469\n}\n\nFile v0.20.0:references/changelog-v7-v13.md\n\n# Lance changelog - v7 -> v12 (section 14)\n\nPart of the Lance v13 reference (`lance-format/lance@v13.0.0-beta.4`). Citations are `path:line`\nrelative to the repo root; build a permalink as\n`https://github.com/lance-format/lance/blob/v13.0.0-beta.4/<path>`. Line numbers drift between\ntags - treat them as approximate. Cross-references written as \"section N\" use the original\n16-section numbering; `lance-reference.md` maps every number to its file.\n\n**Release-line shape.** The major is bumped by a bot, not a human: `ci/publish_beta.sh:65,87`\nre-roots at `MAJOR+1` whenever any PR since the release root carries the GitHub\n`breaking-change` label (`ci/check_breaking_changes.py:31`). The marker is the **label**, not a\nconventional-commit `!` - of the 13 labeled PRs in the v11 window (#8024, #8025, #8026, #8027,\n#8028, #8051, #8095, #8159, #8172, #8188, #8206, #8347, #8360) only two carry `!` in the\nsubject. It has now fired on two\nconsecutive lines: `9.1.0-beta.*` -> `10.0.0-beta.*` (2026-07-23), then `10.1.0-beta.*` ->\n`11.0.0-beta.*` (2026-08-05, `649076df1 chore: bump to 11.0.0-beta.1 based on breaking change\ndetection`). So **neither `v9.1.0` nor `v10.1.0` was ever released**, and `v10.1.0-beta.2` is\nthe direct ancestor of `v11.0.0-beta.1`, one bump commit apart. The re-root renumbers in place:\n`release-root/10.1.0-beta.N` and `release-root/11.0.0-beta.N` point at the same commit\n(`ee0a60d0c`), both recording `Base: 10.0.0-rc.1`.\n\n**`v10.0.0` final was tagged on 2026-08-08** - an annotated, PGP-signed tag (\"Release version\n10.0.0\") on `release/v10.0`, one commit past `v10.0.0-rc.3` (2026-08-02). That branch forked at\n`v10.0.0-beta.7` and took one substantive backport (`10d0c9f2e fix: backport encoding and FTS\nfixes to release/v10.0`, #8146). It is **not** an ancestor of `main` - finals are cut on\n`release/vX.Y` branches, so that is normal. `v10.0.0-beta.7` **is** an ancestor of\n`v11.0.0-beta.16`, but `v10.0.0-rc.3` and `v10.0.0` are not.\n\n**`v10.0.0` is the stable pin** (2026-08-08, superseding `v9.0.1`), and it is what GitHub\nReleases marks `Latest`. `v9.0.1` (2026-08-06, superseding `v9.0.0`, 2026-07-24) shipped with\nfive sibling patch finals that day - `v8.0.1`, `v7.1.0`, `v6.1.0`, `v4.0.2`, `v3.0.2` - each on\nits own `release/vX.Y` branch. `v5.0.0` still has no final despite `v5.0.0-rc.2`. crates.io\npublishes **finals only** (`max_stable_version` = `10.0.0`; no 11.x, and the only pre-release\namong ~186 versions is the ancient `0.0.1-alpha0`); PyPI `pylance` is likewise at `10.0.0`. So\nany beta pin is a git dependency; beta artifacts publish to fury.io\n(`.github/workflows/publish-beta.yml:114`) under the renamed org,\n`https://pypi.fury.io/lance-format`.\n\n## Contents\n\n- [The v7.1.0-beta.1 delta](#the-v710-beta1-delta)\n- [The v7.1.0-beta.2 delta](#the-v710-beta2-delta)\n- [The v7.1.0-beta.2 -> v7.2.0-beta.5 delta](#the-v710-beta2---v720-beta5-delta)\n- [The v7.2.0-beta.5 -> v8.0.0-beta.9 delta (major-version boundary)](#the-v720-beta5---v800-beta9-delta-major-version-boundary)\n- [The v8.0.0-beta.9 -> v8.0.0-beta.14 delta](#the-v800-beta9---v800-beta14-delta)\n- [The v8.0.0-beta.14 -> v9.0.0-beta.10 delta (v8 -> v9 major boundary)](#the-v800-beta14---v900-beta10-delta-v8---v9-major-boundary)\n- [The v9.0.0-beta.10 -> v9.0.0-beta.16 delta](#the-v900-beta10---v900-beta16-delta)\n- [The v9.0.0-beta.16 -> v9.0.0-beta.18 delta](#the-v900-beta16---v900-beta18-delta)\n- [The v9.0.0-beta.18 -> v9.1.0-beta.8 delta](#the-v900-beta18---v910-beta8-delta)\n- [The v9.1.0-beta.8 -> v10.0.0-beta.7 delta](#the-v910-beta8---v1000-beta7-delta)\n- [The v10.0.0-beta.7 -> v11.0.0-beta.2 delta](#the-v1000-beta7---v1100-beta2-delta)\n- [The v11.0.0-beta.2 -> v11.0.0-beta.6 delta](#the-v1100-beta2---v1100-beta6-delta)\n- [The v11.0.0-beta.6 -> v11.0.0-beta.16 delta (the v11 beta frontier)](#the-v1100-beta6---v1100-beta16-delta-the-v11-beta-frontier)\n- [v11 silent-corruption and wrong-results fixes](#v11-silent-corruption-and-wrong-results-fixes)\n- [v11.0.0 final (the beta.16 -> final delta)](#v1100-final-the-beta16---final-delta)\n- [v12 (release-root/12.0.0-beta.N -> v12.0.0-beta.6)](#v12-release-root1200-betan---v1200-beta6)\n- [The v12.0.0-beta.6 -> v12.0.0-beta.15 delta](#the-v1200-beta6---v1200-beta15-delta)\n- [The v12.0.0-beta.15 -> v12.0.0 final delta](#the-v1200-beta15---v1200-final-delta)\n- [The v13 line (release-root/13.0.0-beta.N -> v13.0.0-beta.4)](#the-v13-line-release-root1300-betan---v1300-beta4)\n\nOther files: `format-file.md` (1-4), `format-table.md` (5-10), `indexes.md` (11-12),\n`ops.md` (13, 15, 16).\n\n---\n\n## 14. What changed (v7 -> v13)\n\nThe v7 tag line ran `v7.0.0-beta.1` through `v7.0.0-beta.17`, then `v7.0.0-rc.1` and\n`v7.0.0`. The v7.1 line opened at `v7.1.0-beta.1`, continued through `v7.1.0-beta.4` and\n`v7.1.0-rc.1`; the v7.2 line ran through `v7.2.0-beta.5`; the **v8 line** ran through\n`v8.0.0-beta.19` to `v8.0.0` final (2026-07-01); the **v9 line opened** (auto-bumped from\na `breaking-change`-labeled PR) and ran through `v9.0.0-rc.2` to **`v9.0.0` final** (2026-07-24,\non the `release/v9.0` branch, later `v9.0.1` on 2026-08-06 - **the current stable pin**); the\n**v9.1 line opened** at `9.1.0-beta.0` when `v9.0.0-rc.1` was cut and ran to `9.1.0-beta.8`; on\n2026-07-23 that same dev line was **mechanically re-rooted as `10.0.0-beta.*`** by the\nbreaking-change detector, so `v9.1.0` was never tagged; the **v10 line** ran to\n`v10.0.0-beta.7`, then stabilization forked to `release/v10.0`, reached `v10.0.0-rc.3` and\nshipped **`v10.0.0` final on 2026-08-08 - the current stable pin** - while `main` opened\n`10.1.0-beta.1/2`; and on 2026-08-05 **that** line was re-rooted in place as `11.0.0-beta.*`, so\n`v10.1.0` was never tagged. This section keeps\nthe full v7 history below (still useful context), the **v7.2.0-beta.5 -> v8.0.0-beta.9 delta**\n(the v7->v8 major boundary), the **v8.0.0-beta.9 -> v8.0.0-beta.14 delta**, the\n**v8.0.0-beta.14 -> v9.0.0-beta.10 delta** (the v8->v9 major boundary), the\n**v9.0.0-beta.10 -> v9.0.0-beta.16 delta**, the **v9.0.0-beta.16 -> v9.0.0-beta.18 delta**,\nthe **v9.0.0-beta.18 -> v9.1.0-beta.8 delta**, the **v9.1.0-beta.8 -> v10.0.0-beta.7 delta**,\nthe **v10.0.0-beta.7 -> v11.0.0-beta.2 delta**, the **v11.0.0-beta.2 -> v11.0.0-beta.6 delta**,\nand finally the **v11.0.0-beta.6 -> v11.0.0-beta.16 delta** (most important\nfor a v11 reader) plus a consolidated list of v11's silent-corruption and wrong-results fixes\nat the very end.\n\n**The v6 -> v7 breaking change.** `feat!: make dataset object store access base-aware`\n(PR #6647, commit `456198cd`), immediately followed by the automated bump to `7.0.0-beta.1`.\nObject-store access is now scoped to a dataset *base* instead of a flat global path -\ngroundwork for multi-base storage (hot/cold tiering, multi-region, shallow clones). The\nrelated `refactor!: vendor the tokenizer stack into lance` (PR #6512) is what created the\n`lance-tokenizer` crate.\n\n**MemWAL / LSM** is the dominant v7 theme: the WAL appender/tailer primitives (PR #6669), the\n`shared-memory://` object-store scheme, `ShardWriter` manual-compaction APIs (#6766), a\nbuilder-style MemWAL init API (#6815), append-only tables without primary keys (#6848), and\n`ShardSpec` renamed to `ShardingSpec` (#6813). See section 10 - MemWAL is experimental.\n\n**Lance-native in-memory HNSW** for the MemWAL shard writer (PR #6795).\n\n**Indexes** - segmented btree indexes (#6605), zonemap index segments (#6593), incremental /\nsegmented FTS index merging (#6737, #6790), distributed bitmap index build (#6598), segmented\ninverted index build and search (#6305), FTS exec internals exposed for distributed planning\n(#6648). The geo / RTree index and the `lance-geo` crate; an RTree index-type parsing fix\n(#6568).\n\n**Branches and tags** - branch/tag metadata maps and tag timestamps (PR #6364); the `tree/`\nand `_refs/branches/` layout (section 7).\n\n**Commits** - manifest version hint for fast latest-version lookup (PR #6752); uncommitted\ndelete transactions exposed (#6781); the `Clone` transaction (shallow / deep).\n\n**Spec restructuring** - the lakehouse spec was formally split into separate catalog /\nnamespace / table / index specifications (PR #6750x), reflected in `docs/src/format/`.\n\n### The v7.1.0-beta.1 delta\n\nThe 19 commits in `v7.0.0-beta.16..v7.1.0-beta.1` are mostly bug fixes and internal\nperformance work (serializable BTree/Bitmap/LabelList index caches, deterministic HNSW\ngraph builds, roaring range-iterator speedups). The user-facing additions:\n\n- **Materialized-view namespace API** (PR #6891) - `create_materialized_view` and\n  `refresh_materialized_view` on the `LanceNamespace` trait. A materialized view is a\n  query / UDTF / chunker backed by a stored spec, with an optional initial refresh. The\n  `RestNamespace` implements both (`POST /v1/materialized_view/{id}/create` and\n  `/refresh`); `DirectoryNamespace` and the default trait return `not_supported`.\n- **Typed vector index details** (PR #6099) - the `VectorIndexDetails` and\n  `HnswParameters` messages moved into `protos/index.proto` (section 11.1).\n- **Multi-base `write_fragments`** (PR #6855) - multi-base storage config is now reachable\n  from the Python and Java `write_fragments` API, not just Rust.\n- **Granular tracing targets** (PR #6853) - `pylance` emits trace events under a\n  `lance::events::` prefix so they filter separately from log records; new\n  `lance::dataset_events` and `lance::object_store::throttle` targets. Example:\n  `LANCE_LOG=\"warn,lance::events::object_store::throttle=info\"`\n  (`docs/src/guide/performance.md`).\n- **MemWAL** - a sharding evaluator (PR #6854), L0 flushed-generation dataset caching\n  (PR #6816), and exact primary-key dedup fixes for LSM point lookup and vector search\n  (PR #6881).\n\n### The v7.1.0-beta.2 delta\n\nSeven commits in `v7.1.0-beta.1..v7.1.0-beta.2`, mostly MemWAL correctness work plus one\nworkspace refactor:\n\n- **New `lance-select` crate** (PR #6879, commit `52c6ac34`) - mask code (`RowAddrMask`,\n  `NullableRowAddrMask`, `RowIdMask`, set types, `bitmap_to_ranges`/`ranges_to_bitmap`) and\n  scalar-index expression-result types (`IndexExprResult`, `NullableIndexExprResult` with\n  their `Not`/`BitAnd`/`BitOr` boolean algebra) were extracted from `lance-core` and\n  `lance-index` into `rust/lance-select/`. Downstream filtering code and the new\n  `index_expr_result` / `row_addr_mask` benches can now depend on masks without pulling in\n  either larger crate.\n- **MemWAL: build secondary indexes when flushing the active memtable** (PR #6901, commit\n  `cee7d32f`) - `MemTableFlushHandler` previously called `flush`, persisting the data file\n  and bloom filter but **building no secondary indexes**, and never received the shard's\n  `index_configs` in the first place. Over flushed generations this made flushed vector rows\n  invisible to `fast_search()` (a correctness bug for KNN, not just perf), and point lookups\n  fell back to a full scan instead of routing through a scalar index. The fix threads\n  `index_configs` into the handler and calls `flush_with_indexes` when any index is\n  configured, while keeping plain `flush` when none are so empty-index shards avoid an extra\n  pass.\n- **MemWAL: per-source PK-hash block-list post-filter** (PR #6899, commit `77db998a`)\n  fixes a stale-read in LSM vector search. `LsmGlobalPkDedupExec` (introduced in #6881) is\n  exact only over candidates each source surfaces; if a primary key's fresh row is pushed\n  out of its source's top-k by closer rows, the dedup never sees it and a superseded copy\n  from an older generation can win. The fix makes staleness a per-source PK-hash post-filter\n  (`PkHashFilterExec`) applied to each source's KNN *before* the cross-source union, so a\n  stale row never reaches the merge. Each generation's membership is an\n  `Arc<HashSet<u64>>` of PK hashes (`compute_pk_hash`, the same hash the dedup nodes use).\n- **Docs** - new integrations landing page at `docs/src/integrations/index.md` (PR #6915);\n  Java doc URL updated from `com.lancedb` to `org.lance` (#6467).\n\nNote: there is **no Tantivy-FTS-removal commit in the v7 range**. Lance FTS at this tag is\nalready its own native inverted-index implementation; the tokenizer vendoring (#6512)\npredates `v7.0.0-beta.1`. Do not attribute a Tantivy removal to v7.\n\n### The v7.1.0-beta.2 -> v7.2.0-beta.5 delta\n\n66 commits, **no breaking change** (no `!:` commit, no `BREAKING CHANGE` footer), **no new\ncrate** (still 24), **no new transaction op** (still 15), no proto change. User-facing\nadditions:\n\n- **ICU FTS tokenizer** (PR #6956) - `base_tokenizer=\"icu\"`, ICU4X dictionary segmentation\n  with bundled data, no external model. A PR making ICU the default (#6968) was reverted\n  (#7006); the default base tokenizer stays `simple`. See section 11.3.\n- **Scalar-index fast search** (PR #6784) - `fast_search` routes through scalar/BTREE-indexed\n  fragments and skips unindexed ones (not on legacy file version). See section 11.2.\n- **Batched vector queries** (PR #6828) - `Scanner::nearest` takes a batch of query vectors\n  and exposes a synthetic 0-based `query_index` column. See section 11.1.\n- **Streaming IVF k-means** (PR #6913) - `streaming_sample_rate` / `streaming_coreset_rate` /\n  `streaming_refine_passes` for bounded-memory IVF training. See section 11.1.\n- **Arrow view-type support** (PR #6985) - `Utf8View` / `BinaryView` now encode (fixes an\n  encoder `todo!()` panic) and coerce correctly in filters.\n- **HuggingFace `download_mode`** (PR #7022) - storage-option keys `hf_download_mode` /\n  `download_mode` select the OpenDAL `http` (default) or `xet` backend on the existing\n  `hf://` provider; not a new object-store scheme.\n- **MemWAL LSM local-scoring FTS** (PR #6951) - `LsmScanner::full_text_search(column, query, k)`,\n  contained entirely in the `mem_wal` module.\n- Dependency bumps: pylance `lance-namespace>=0.8.0,<0.9` (PR #7031), `opendal 0.57`\n  (PR #7018), `jieba-rs 0.10` (PR #6955).\n- Doc clarification: RaBitQ (RQ) is documented 1-bit-only with multi-bit as future work, and\n  the RQ metadata schema gained a `code_dim` field (`docs/src/format/index/vector/index.md`).\n\nUnchanged and reverified at this tag: 15 transaction ops, the scalar/vector index-type set,\nall `protos/*.proto`, file-format `version.rs`, `rust-version 1.91.0`, `resolver 3`, edition\n2024, `CommitConfig num_retries=20`, MemWAL still experimental.\n\n### The v7.2.0-beta.5 -> v8.0.0-beta.9 delta (major-version boundary)\n\n86 commits. This is a **major version bump** whose unifying theme is moving *every* index\nbuild onto one segment-based lifecycle. **Six breaking changes** (`!:` commits):\n\n- **`feat!: migrate bitmap to index segment based`** (PR #6869) - the defining v8 change.\n  Bitmap now flows through the segment workflow; the old public Python Bitmap shard path\n  (`create_scalar_index(..., fragment_ids=)` + `merge_index_metadata(..., \"BITMAP\")`) \"is no\n  longer exposed; callers should use the segment workflow instead.\" `execute_uncommitted`\n  writes canonical `bitmap_page_lookup.lance` segment roots\n  (`rust/lance-index/src/scalar/bitmap.rs:59`).\n- **`refactor!: remove index segment builder`** (PR #6997) - the `IndexSegmentBuilder` API\n  was removed from Rust, Python, and Java; staged publishing routes through\n  `create_index_uncommitted` / `execute_uncommitted` + `merge_existing_index_segments` +\n  `commit_existing_index_segments`. `build_all()` and `target_segment_bytes` size-based\n  grouping are gone with no direct replacement (`docs/src/guide/migration.md` \"7.2.0\").\n- **`refactor(index)!: move distributed BTree build to segmented index framework`** (PR #7013)\n  - distributed BTree now uses the same `create_index_uncommitted` / merge / commit path.\n- **`feat!: return write summaries from file writers`** (PR #7096) - `finish()` changed from\n  `Result<u64>` to `Result<FileWriteSummary>` (`{ num_rows: u64, size_bytes: u64 }`,\n  `rust/lance-file/src/writer.rs:54-58,768`). Python `LanceFileWriter.finish` keeps its\n  row-count return.\n- **`fix(python)!: derive index type from details`** (PR #6903) - `describe_indices()` \"now\n  reports nested and special-character field names as full field paths (e.g. `meta.lang`)\n  instead of just the leaf name\"; `list_indices()` is a thin typed `IndexInformation` wrapper\n  that no longer opens each index; the `load_indices()` Python binding was removed.\n- **`perf!: avoid listing index files after writes`** (PR #7129) - `IndexFile` metadata is\n  propagated from writer/builder APIs into manifest metadata instead of listing index\n  directories after writes (a writer/builder trait-level break).\n\nNet-new user-facing features:\n\n- **`lance-derive` crate** (PR #6229) - `#[derive(DeepSizeOf)]` for Arrow-aware memory\n  accounting, replacing the external `deepsize` crate. Crate workspace 24 -> 25. See section 2.\n- **FM-Index scalar index** (`docs/src/format/index/scalar/fmindex.md`,\n  `protos/index.proto` `FMIndexIndexDetails`) - BWT substring/prefix/regex search on raw\n  bytes via the Segmented Index architecture (`num_segments`). See section 11.2.\n- **Multi-bit IVF_RQ** (PR #7038) - RaBitQ `num_bits` 1..=9; ex-code bits in `__ex_codes`\n  (+ `__add_factors_ex` / `__scale_factors_ex`). **Raw-query RQ search** (PR #7078) adds the\n  `query_estimator` field and `__error_factors` lower-bound pruning. See section 11.1.\n- **Independent per-worker vector index models** (PR #7148) for distributed builds; zone-map\n  segments now mergeable via `merge_existing_index_segments` (PR #7128); HNSW segment merge\n  (PR #7178); segmented BTree merge (PR #6889). See section 12.\n- **Volcengine TOS** (`tos://`) and feature-gated **GooseFS** (`goosefs://`, PR #7034) object\n  stores. See section 13.\n- Smaller: `tracked_files` / `all_files` on `LanceDataset` (PR #6011); multi-segment FM-Index\n  build config; `add_columns` UDFs no longer require pandas (PR #7131); FTS flat match now\n  searches all unindexed fragments (PR #7188); AVX-512 distance tables compiled for the\n  target CPU (PR #7121).\n- Dependency facts: arrow 58, datafusion 53, opendal 0.57, jieba-rs 0.10,\n  lance-namespace-reqwest-client 0.8.2; pylance `lance-namespace>=0.8.0,<0.9`.\n\nUnchanged and reverified at `v8.0.0-beta.9`: **15 transaction ops** (`protos/transaction.proto`\ndiff empty); file-format `version.rs` (`Next => 2.3`, `#[default]` still `V2_1`, no 2.4 - so\nsection 3 holds unchanged); `CommitConfig num_retries = 20`\n(`rust/lance-table/src/io/commit.rs:1530`); the feature-flag bits; the\n`ConditionalPutCommitHandler` routing; `rust-version 1.91.0`, `resolver 3`, edition 2024;\nMemWAL docs and system-index docs byte-identical; MemWAL still experimental.\n\n### The v8.0.0-beta.9 -> v8.0.0-beta.14 delta\n\n31 commits, **two breaking changes** - both vector/RaBitQ. No new crate (still 25), no new\ntransaction op (still 15), no file-format change.\n\n- **`feat(vector)!: add approx mode for RaBitQ search`** (PR #7179) - a public\n  `approx_mode` with values `fast` / `normal` / `accurate` for vector search \"when the backing\n  index supports it\" (commit `e25620710`), threaded through the Rust scanner, Python query\n  parsing, and ANN proto serialization. **Breaking proto change**: the ANN query proto now\n  carries `VectorApproxMode approx_mode` (`protos/ann.proto:16,45`) - regenerate any consumer\n  that matches the serialized ANN proto. See section 11.1.\n- **`perf(vector)!: add dedicated SIMD kernels for RaBitQ ex-code reranking`** (PR #7205,\n  `rust/lance-index/src/vector/bq/ex_dot.rs`).\n- **IVF_RQ default `target_partition_size` is now 4096** (was the generic fallback, PR #7273).\n- **Cleanup explain API** (PR #7147) - `Dataset::cleanup(policy)` splits into `explain()`\n  (returns a `CleanupExplanation`, a dry run) and `execute()`. See section 7.\n- **Object-store docs** (PR #7151) - the guide gained full Tencent COS and GooseFS config\n  sections (`docs/src/guide/object_store.md:333,396`); GooseFS is no longer undocumented. See\n  section 13.\n- **Smaller adds**: Python zonemap segment builds exposed (PR #7177); per-query I/O metrics\n  (`bytes_read` / `iops` / `requests`) on `ANNSubIndex` / `ANNIvfPartition` in EXPLAIN ANALYZE\n  (PR #7204); branch-aware version ops in the Directory/REST namespaces (CreateTableBranch /\n  ListTableBranches / DeleteTableBranch, PR #7166); enriched `IndexContent` fields in dir\n  namespace `ListTableIndices` (PR #7109).\n- **Fixes**: resolve Blob v2 external URIs and clean failed writes in `add_columns`\n  (PR #7152); coerce filter literals for dictionary-encoded columns (PR #7003);\n  composite-key `merge_insert` probes every indexed key column (PR #6878).\n- **Removals**: `table_version_storage_enabled` and the `__manifest`-backed table-version path\n  removed - version ops now use `_versions/` exclusively (PR #7222); brotli dropped from the\n  dependency graph (PR #7270).\n- **Dep pins**: `lance-namespace-reqwest-client` 0.8.2 -> 0.8.4; pylance `lance-namespace`\n  `>=0.8.0,<0.9` -> `>=0.8.5,<0.9`. arrow 58 / datafusion 53 / opendal 0.57 / jieba-rs 0.10\n  unchanged.\n\nUnchanged and reverified at `v8.0.0-beta.14`: 25 crates; 15 transaction ops; file-format\n`version.rs` (`Next => 2.3`, `#[default] V2_1`, no 2.4); `CommitConfig num_retries = 20`\n(`rust/lance-table/src/io/commit.rs:1550`); `rust-version 1.91.0`, `resolver 3`, edition 2024;\nfeature-flag bits; `ConditionalPutCommitHandler` routing.\n\n### The v8.0.0-beta.14 -> v9.0.0-beta.10 delta (v8 -> v9 major boundary)\n\n129 commits. A **light major bump** - structurally v9 is nearly identical to v8. The major\nversion was auto-triggered by `ci/check_breaking_changes.py` (GitHub `breaking-change`-label\ndetection), fired by two PRs merged before the 2026-06-22 bump: the Python 3.9 drop (#7345)\nand the `alter_columns` fail-fast cast (#7158). The FMIndex rename (#7397) carries the label\ntoo but merged after the bump, so it rode the already-bumped 9.0.0 series rather than\ntriggering it. **Three breaking changes:**\n\n- **`refactor!: rename FMIndexIndexDetails to FMIndexDetails`** (PR #7397) - proto message\n  `protos/index.proto:251` `FMIndexDetails {}` (was `FMIndexIndexDetails`); Rust type\n  `pb::FmIndexDetails`; the `get_plugin_name_from_details_name` `fmindex`->`fm` special-case\n  was deleted. Author's note: \"This change would be a breaking change to any existing FM\n  indexes!\" - existing FM indexes become unreadable. See section 11.2.\n- **Drop Python 3.9** (PR #7345) - `python/pyproject.toml` `requires-python = \">=3.10\"`; the\n  `Python :: 3.9` classifier removed; PyO3 abi3 floor raised `abi3-py39` -> `abi3-py310`;\n  release wheels no longer built for 3.9. See section 2.\n- **`fix(dataset)!`: `alter_columns` cast fails fast with an attached index** (PR #7158) -\n  previously a cast silently dropped/invalidated the index; now it errors and you must\n  `drop_index()` first. See section 6.\n\n**Removal (Rust API, not conventional-`!`):** `as_vector_index` removed from the public\n`Index` trait (PR #7392) - callers downcast via `as_any()`. See section 11.1.\n\n**Net-new features:**\n\n- **Hamming clustering** (PR #7379) - SIMD near-duplicate detection over 64-bit binary\n  hashes (`pairwise_hamming_distance`, `UnionFind`, `hamming_clustering_for_ivf_partition`).\n  See section 11.1.\n- **COUNT(*) pushdown on stable-row-id datasets** (PR #7360) - the fast path no longer falls\n  back to a full scan when stable row IDs are enabled. See section 11.1.\n- **Per-column blob size thresholds** (PR #7269) - `lance-encoding:blob-inline-size-threshold`\n  / `...-dedicated-size-threshold`; appends with a different threshold are rejected. See\n  section 3.5.\n- **Tunable 32k miniblock chunks** via `LANCE_MINIBLOCK_MAX_VALUES` (PR #7356; default still\n  4096). See section 3.3.\n- **`icu/split` FTS tokenizer** (PR #7474) and **mixed-language stop words** (PR #7324). See\n  section 11.3.\n- **Distributed LabelList index builds** (PR #7223). See section 12.\n- **ngram index accelerates regex + infix LIKE** (PR #7139). See section 11.2.\n- `alter_columns` **Dict <-> value-type casts** (PR #7289, section 6); cleanup-explain\n  exposed to **Python and Java** (PR #7248, section 7); Python **fragment-reuse remap +\n  delete-by-offset** (PR #7438); v2 file writer/reader support **columns of unequal length**\n  (PR #7406); a `SpillStore` trait with local-disk impl (PR #7311); a versioned cache-codec\n  envelope (PR #7163).\n\n**Notable fixes:** compaction rejects `defer_index_remap` with stable row IDs (#7468);\nnested legacy blobs rejected in v2.2 and blob v2 supported in nested structs (#7278, #7281);\n`merge_insert` no longer drops matches when a leading payload column is all-null (#7251); SQ\noffset accounted for in dot distance (#7481); manifests >5 GB via size-aware copy (#7047);\ndouble percent-encoding in object-store paths resolved (#6643/#6695).\n\n**Dependency changes:** `lance-namespace-reqwest-client` 0.8.4 -> 0.8.6 (#7254); `itertools`\n0.13 -> 0.14 (#7424). The pylance runtime pin `lance-namespace>=0.8.5,<0.9` is unchanged.\n\nUnchanged and reverified at `v9.0.0-beta.10`: **25 crates**; **15 transaction ops**\n(`protos/transaction.proto` unchanged); file-format `version.rs` (`Next => 2.3`,\n`#[default] V2_1`, no 2.4); `CommitConfig num_retries = 20`\n(`rust/lance-table/src/io/commit.rs:1550`); `rust-version 1.91.0`, `resolver 3`, edition 2024;\narrow 58 / datafusion 53 / opendal 0.57 / jieba-rs 0.10; feature-flag bits;\n`ConditionalPutCommitHandler` routing; MemWAL still experimental.\n\n### The v9.0.0-beta.10 -> v9.0.0-beta.16 delta\n\n58 commits. **One breaking change**; no new crate (still 25), no new transaction op (still 15),\nno file-format change, no dependency-pin change. **`v8.0.0` final also shipped** in this window\n(tag `v8.0.0`, commit `15f2ff594`, 2026-07-01) - use it as the stable pin (section intro).\n\n**Breaking change:**\n\n- **`feat(fts)!: make v2 the default index format`** (PR #7512) - newly created FTS / inverted\n  indexes default to on-disk **format v2**; `LANCE_FTS_FORMAT_VERSION` no longer controls new\n  indexes; pass `format_version=1` for older-reader compatibility. Existing v1 indexes stay\n  queryable and are maintained as v1. See section 11.3.\n\n**Net-new features:**\n\n- **Blob read-API rework** (PR #7530, #7558) - `read_blobs` is now the primary full-payload\n  API, `take_blobs` is for streaming/seeking, `scanner(blob_handling=\"all_binary\")` reads blobs\n  as Arrow binary; documented auto-tiering defaults (16 KiB inline / 2 MiB dedicated) and a new\n  `lance-encoding:blob-pack-file-size-threshold` field key (PR #7322). `dataset.update()` now\n  works on blob-encoded columns (PR #7579). See section 3.5.\n- **Per-base `storage_options`** via `base_<id>.<key>` keys (PR #7608); **multi-base\n  merge-insert** with target-base routing (`MergeInsertBuilder::target_bases`, round-robin new\n  fragments, `DataFile.base_id` stamped; PR #7610). See sections 13 and 6.\n- **ZoneMap `value_range`** min/max without a scan (PR #7463); **BTREE + ZONEMAP accept\n  `LargeUtf8`** (PR #7525). See section 11.2.\n- **Prefiltered LSM vector + FTS search** across base/flushed/in-memory sources (PR #7138). See\n  section 10.\n- **Schema evolution allows all-null `Map` columns** (PR #7462). See section 6.\n- **DirectoryNamespace** now implements `update_table` / `delete_from_table` (PR #6923) and\n  `alter_transaction` (PR #6974) - previously `not_supported`.\n- Compaction `RowAddrRemap` structure to avoid remap HashMap OOM (PR #7237); single-flight\n  scalar-index opens (PR #7464); session-cached manifest reuse on open (PR #7576).\n\n**Notable fixes:** `DataReplacement` commits preserve `DataFile.base_id` on multi-base datasets\n(#7609); blob descriptor views kept opaque - the reader no longer recurses into `position`/\n`size` child fields (#7618); stable row-id index tolerates sparse/overlapping chunks (#7480);\nngram posting-list writes chunked by byte size to avoid i32 offset overflow (#7607); scheduler\ndeadlock on same-priority chunks fixed (#7588).\n\nUnchanged and reverified at `v9.0.0-beta.18`: **25 crates**; **15 transaction ops**\n(`protos/transaction.proto` unchanged); file-format `version.rs` (`Next => 2.3`,\n`#[default] V2_1`, no 2.4); `CommitConfig num_retries = 20`; `rust-version 1.91.0`,\n`resolver 3`, edition 2024; arrow 58 / datafusion 53 / opendal 0.57 / jieba-rs 0.10 /\nitertools 0.14 / lance-namespace-reqwest-client 0.8.6; pylance `lance-namespace>=0.8.5,<0.9`;\nfeature-flag bits; `ConditionalPutCommitHandler` routing; MemWAL still experimental.\n\n### The v9.0.0-beta.16 -> v9.0.0-beta.18 delta\n\n36 commits. **No breaking changes**; no new crate (still 25), no new transaction op\n(`rust/lance/src/dataset/transaction.rs` untouched), no file-format change. Mostly fixes\nplus three additive features:\n\n- **pylance prewarm gains segment selection** (#7677) - warm only chosen index segments.\n- **Object-store metrics published via the `metrics` crate** (#7533).\n- **RLE v2 run-length widths** (#7376), with width selection by encoded size (#7636).\n\n**Docs:** the performance guide gained a **Fragment Sizing** section (#6606); cleanup and\nautomatic-cleanup documentation added to the read-and-write guide (#6546); a new\n`guide/observability.md` page; MemWAL format spec updated (#7655). All reproduced in this\nskill's `references/docs/` mirror.\n\n**Notable fixes:** FTS list columns indexed as row documents (#7656); fuzzy\n`max_expansions` enforced globally across index partitions instead of per-partition; FTS\ntail-partition merge split by the worker memory budget and a `num_tokens`-only DocSet\ncached on `LazyDocSet` (#7600); MemWAL writer fenced on WAL persistence failure (#7547)\nand slice-aware memtable flush-threshold size estimate; Arrow-JSON -> Lance-JSON\nconversion fixed across the merge/update, single-fragment-create, and merge-insert\nfull-fragment-rewrite paths (`take` now returns Arrow JSON, #7470/#7471);\n`object_store::Error::NotFound` mapped to `Error::NotFound` instead of a generic IO\nerror; PQ `num_bits` respected for numpy codebooks (#7586); hang fixed in\n`train_streaming_coreset_ivf_model` (#7676).\n\n### The v9.0.0-beta.18 -> v9.1.0-beta.8 delta\n\n127 commits straddling the tail of the 9.0.0 beta line (through `rc.2`) and the new 9.1.0 dev\nline. The 9.1.0 minor bump is **automatic release-train cadence**, not a breaking change: when\n`v9.0.0-rc.1` was cut, `main` advanced to `9.1.0-beta.0`. **One breaking-labeled PR** in the\nwindow: FTS `block_size` (#7466, below). Structural changes reverified at `v9.1.0-beta.8`:\n**26 crates** (new `lance-index-core`, #7713); **16 transaction ops** (new `DataOverlay`);\n**datafusion 53 -> 54** (#7793), geodatafusion 0.4 -> 0.5, build toolchain 1.91 -> 1.97 (#7712,\nMSRV `rust-version` unchanged at 1.91.0). Unchanged: arrow 58, opendal 0.57, jieba 0.10,\n`lance-namespace-reqwest-client` 0.8.6, itertools 0.14, edition 2024, resolver 3,\n`lance-arrow-scalar =58.0.0`; `CommitConfig num_retries = 20`; file-format `version.rs`\n(`Next => 2.3`, `#[default] V2_1`); Python min 3.10 (3.14 added, #7728).\n\n**Breaking (labeled):**\n- **FTS configurable posting `block_size`** (#7466) - `InvertedIndexParams` gains `block_size`\n  (128/256, default 128, 512 rejected). `block_size=256` and the code analyzer require FTS\n  on-disk **format v3** (#7866). Sections 11.3.\n\n**Additive features:**\n- **Data Overlay Files** (#7535 write path, #7536 read path, #7540 Python commit op) - the new\n  16th transaction op `DataOverlay`, feature flag 64, spec `data_overlay_file.md`. Cell-level\n  `(offset, field)` updates without rewriting base data files; **unstable**, env-gated by\n  `LANCE_ENABLE_UNSTABLE_DATA_OVERLAY_FILES` (release builds refuse overlay datasets).\n  Compaction folds fragments over an overlay-count limit (#7772). Section 5.5.\n- **Sparse structural pages** (#7889) - first real 2.3 encoding; `structural-encoding=sparse`.\n  Section 3.1.\n- **Exact `IS NULL`** for ZONEMAP and BLOOM_FILTER via a new `null_bitmap`. Section 11.2.\n- **Nested-field FTS** (#7686) index leaf fields like `data.text`; code-analyzer tokenizer\n  (#7681); impact-skip / bulk MAXSCORE top-k / bulk conjunction FTS paths (#7602/#7603/#7624);\n  read inverted-index params without opening the segment (#7816). Section 11.3.\n- **OpenTelemetry metrics for Python** (`instrument_lance_metrics`, `pylance[otel]`, #7537);\n  zone-map seeds written into data-file footers during append (#7427).\n- **AWS creds via `AssumeRoleWithWebIdentity`** to avoid role chaining (#7757); batch/list\n  blob reads (#7864, #7664) and a bulk packed-blob writer (#7743); MemWAL flush-interval\n  ticker (#7894) and Python/Java shard delete (#7649, #7688); wider hamming hashes and\n  multi-segment hamming clustering (#7767, #7758); RLE child-buffer zstd compression (#7663);\n  cached file-metadata APIs on `FileFragment` (#7820); `TableProvider` write inputs for\n  `merge_insert`/`insert` (#7368); runtime SIMD dispatch for pre-Haswell x86_64 from-source\n  builds (#6630).\n\n**Other:** writes now reject system column names (#7797). The TensorFlow integration moved\nfrom built-in to an external `lance-tensorflow` package, and the image array decoder/encoder\nis now Pillow-only (not vendored in this skill's docs mirror - integrations mirror is\n`datafusion.md` only).\n\n### The v9.1.0-beta.8 -> v10.0.0-beta.7 delta\n\n78 commits. The major bump is **mechanical**, not a redesign: `ci/publish_beta.sh` re-roots at\n`MAJOR+1` on any `breaking-change`-labeled PR, and `fb88621f8 chore: bump to 10.0.0-beta.1 based\non breaking change detection` landed immediately after `3a72f8a61 fix(blob)!: preserve null\nselections across blob APIs (#7903)`. Only one bump happens per series, so the two later `!`\ncommits rode the already-bumped line. Structural invariants **all reverified unchanged**:\n26 crates (no crate added or removed), 16 transaction ops, `CommitConfig num_retries = 20`,\nfile-format `version.rs` (`Next => 2.3`, `#[default] V2_1`), feature-flag bits (newest still 64,\ndata overlay), MSRV 1.91.0, toolchain 1.97.0, edition 2024, resolver 3, arrow 58 /\ndatafusion 54 / opendal 0.57 / jieba 0.10 / itertools 0.14 /\n`lance-namespace-reqwest-client` 0.8.6, Python 3.10-3.14.\n\n**Breaking (four):**\n\n- **`fix(blob)!: preserve null selections across blob APIs`** (#7903) - the bump trigger. Every\n  selection API returns one result per request, nulls as `None` instead of omitted. Rust,\n  Python, and Java signatures all change. Section 3.5.\n- **`perf(cache)!: use fixed-size cache keys`** (#7878) - opaque 16-byte BLAKE3 keys\n  (`CACHE_KEY_FORMAT = \"blake3-128-v1\"`); all warm/persisted caches cold-miss, no legacy\n  fallback; prefix-invalidation and key-inventory APIs removed. Section 9.4.\n- **`perf(compaction)!: skip building row-address maps when index remapping is not needed`**\n  (#7778) - `IndexRemapperOptions::create_remapper` becomes async and returns\n  `Result<Option<Box<dyn IndexRemapper>>>`; compaction skips the `_rowid` scan and\n  RoaringTreemap entirely for FRI-only or system-index-only datasets.\n- **MemWAL rename** (#7943, #7957) - flushed generation -> SSTable, merge -> compaction, across\n  spec, Rust, Python, Java, and protos. Wire-compatible, symbol-breaking, no shims. Section 10.\n\n**Additive:**\n\n- **`ConcreteFileVersion`** (#7879) exact file identity, unordered by design; manifests reject\n  selector aliases; `try_from_major_minor` / `to_numbers` removed; byte-exact writer fixtures\n  with SHA-256 locks (#8019). Section 3.6.\n- **Sparse structural pages auto-select** in the 2.3 writer (#7756). Section 3.1.\n- **Segment-native BLOOMFILTER / RTREE / NGRAM / LABEL_LIST** (#7925, #7932, #7244, #7884);\n  `IndexSegment::new` 4 -> 6 params; merged segments inherit the minimum source\n  `dataset_version`; concurrent LIST on segment commit, ~8x faster remote (#7657). Section 12.\n- **ACORN-1 prefiltered HNSW** (#7927), opt-in via `approx_mode=\"fast\"`, with a documented\n  recall regression on uniform-random masks. Section 11.1.\n- **FTS**: `total_tokens` metadata key and `bm25_search` removal (#7863),\n  `LANCE_FTS_SEARCH_CHUNK` (#7950), top-k row-id resolution 26x (#7897), deterministic tie\n  order (#8073), segment-uuid-scoped exec nodes (#7976). Section 11.3.\n- **Data-overlay/index correctness** (#7549, #7926, #7918) - index results exclude\n  overlay-superseded rows. Sections 5.5 and 9.1.\n- **quick_cache** as the default index and metadata cache backend (#7953, #8013), with a\n  per-shard admission ceiling that silently refuses oversized entries. Section 9.4.\n- Cross-store `deep_clone` via `CommitBuilder::with_source_store` (#7545); commit-retry backoff\n  overflow capped at `MAX_SLOTS = 128` (#7883); external-manifest finalization always HEADs\n  (superseded at beta.8 by #8499 - the ETag from that HEAD must not be persisted; see below)\n  (#7964); `memory://` `DatasetNotFound` fix (#8068); tokio-shutdown panic becomes an I/O error\n  (#7478); `LANCE_CPU_THREADS` / `LANCE_IO_CORE_RESERVATION` validated (#7856); dir namespace\n  surfaces throttles instead of `TableNotFound` (#7931) and honors `structured_query` FTS\n  (#7592); Java `CacheStats` + `Session.metadataCacheStats()` (#7885); vector index append\n  across heterogeneous segment models (#8047).\n\n**Fixes worth knowing:** `Dataset::filter_deleted_ids` was wrong on stable-row-id datasets,\nbreaking `optimize_indices` (#7704); filtered scans and `add_columns(AllNulls)` returned a valid\nstruct with null children instead of a null struct on storage 2.1 (#8049); `LIKE ... ESCAPE ''`\nwas treated as no-escape and `ESCAPE 'ab'` silently truncated - both now error (#7810);\n`list_indices` no longer backtick-quotes ordinary column names (#7503).\n\n**Security:** `quinn-proto` 0.11.14 -> 0.11.16 via Dependabot security alert, applied to the\nroot workspace, `/python`, and `/java/lance-jni` (#7983, #7984, #7982) - \"proto: yield error on\ntoo many gaps in assembler\". Plus bulk Dependabot cargo-group bumps (38 root, 28 python,\n27 java-jni).\n\n### The v10.0.0-beta.7 -> v11.0.0-beta.2 delta\n\n128 commits. The bump is again **mechanical** - nine PRs carried the `breaking-change` label\n(#8024, #8025, #8026, #8051, #8095, #8159, #8172, #8188, #8206), of which only two carry `!` in\nthe subject, and the bot re-rooted `10.1.0-beta.2` as `11.0.0-beta.1` in place. Structural\ninvariants **all reverified unchanged**: **26 crates** (the `rust/` `Cargo.toml` inventory is\nbyte-identical across the two tags), **16 transaction ops** (`protos/transaction.proto` is\nbyte-identical), `CommitConfig.num_retries` **20**, file-format enum `next => 2.3` / default 2.1\nwith no 2.4, manifest feature flags 1-128 unchanged, MSRV 1.91.0 / toolchain 1.97.0, Python\n3.10-3.14, arrow 58 / datafusion 54 / `object_store` 0.13.2 / jieba 0.10 / blake3 1.8.5, and the\n`=58.0.0` pins on `lance-arrow-scalar` / `lance-arrow-stats`. The only proto change in the whole\nrange is `protos/index_old.proto` (+17 lines).\n\n**Breaking (labeled):**\n\n- **#8206** - fragment ids became a dataset-lifetime high-water mark; overwrite no longer\n  restarts at 0, overwrite fragments with deletion files are rejected, duplicate ids block all\n  commits. A format invariant, not just an API change. Section 5.3.\n- **#8024 / #8025 / #8026** - the exact-version reader/writer composition: `ReaderProjection`\n  constructors became free functions, `FileReader::version()` and\n  `Dataset::storage_version_or_default()` return `ConcreteFileVersion`,\n  `FileReader::supports_projection` and `open_writer` were removed, and\n  `lance-encoding::version` was deleted with no re-export. Section 2.1.\n- **#8051** - `force_seal_active` returns `SealFence`. **#8095** -\n  `MemIndexConfig::detect_index_type` replaced by `is_maintainable_index_type` + `MemIndexKind`.\n  Section 10.\n- **#8159** - `CacheBackend::deep_size_of_entries`; reported cache sizes shrink. Section 9.4.\n- **#8172** - `DataBlockBuilder::append` is fallible; corrupt variable-width offsets now error\n  instead of panicking. **#8188** - HNSW `try_with_capacity`, `m >= 4` enforced, persisted level\n  layout corrected; different graphs and recall. Sections 2.1 and 11.1.\n\n**Breaking (unlabeled but source-breaking - the #7877 series):** #8020 removed\n`lance_io::encodings` and moved `lance-file::previous` to `versions::v1`; #8021 deleted the\n`lance-encoding::previous` public encoder surface; #8023 turned `FileWriter` into an enum and\nremoved all its constructors plus two `FileWriterOptions` fields; #8038 added a context parameter\nto `MiniBlockCompressor::compress`. Also `#8141` removed `GraphBuilderStats`, and `#7788` added a\nthird parameter to `load_segments`. Section 2.1.\n\n**Net-new:**\n\n- **FTS document granularity** (#7788) - `DocumentGranularity` ROW/LIST_ELEMENT,\n  `posting_format_version`, `_doc_index` column, and a third FTS-v3 trigger. Section 11.3.\n- **Compound FTS scoring core** (#8092, #8093, #8094, #8131, #8299) - Boolean/Phrase/Boost\n  composition, public `CompoundQueryExec`, cost-ordered conjunctions, `AND` as scoring `MUST`.\n  Section 11.3.\n- **Zone maps**: `has_null_bitmap` making `IS NOT NULL` scan-free (#8088) and all-type support\n  including nested (#8017). Section 11.2.\n- **Manifest transaction spilling** above 20 MiB (#7881, ~50% manifest shrink measured).\n  Section 9.1. **Pluggable cache backends** with a `moka://` URI form (#7683). Section 9.4.\n- **`aws_provider_scheme`** token/ecs/irsa (#8103); **`goosefs://` via\n  `ConditionalPutCommitHandler`** (#8134) with a mixed-version overwrite hazard; multipart\n  part-identity retry fix (#8174) and removal of `LANCE_CONN_RESET_RETRIES`. Section 13.\n- **Encoding performance**: exact decode-buffer preallocation via `decoded_size_bytes` (#8091,\n  index-cache weight down up to 74% on IVF_SQ, no on-disk change); a zero-copy typed view for\n  inline bitpacking (#7696, 13-22% faster unchunk); `O(n*m)` fragment compares removed from\n  `build_manifest` for Update/Delete (#8210); parallel doc-length preload on the cold deferred\n  FTS search path (#8119).\n- **Python**: pydantic auto-conversion in `write_dataset` plus\n  `LanceDataset.from_pydantic_model(model_class, data, uri=None, **kwargs)` (#7383);\n  `LanceFileWriteSummary` giving `LanceFileWriter` a `size_bytes` (#7876); `max_source_fragments`\n  on `compact_files` for incremental compaction, also settable via the manifest config key\n  `lance.compaction.max_source_fragments` (#8116); `blob_handling` on the SQL/DataFrame builder\n  (#8087).\n- **Java** got the biggest build-out of any binding: an OpenTelemetry metrics bridge\n  (`org.lance.otel.LanceMetrics`, #8064 - the docs now say metrics are available \"from the Rust,\n  Python, and Java APIs\"); scanner tuning via `ScanOptions` (`batchSizeBytes`, `ioBufferSize`,\n  `fragmentReadahead`, `scanInOrder`) and a typed `MaterializationStyle` (#8288);\n  `FragmentStatistics` (#8072); a typed `LanceException` replacing bare `RuntimeException`\n  (#8184); and `IndexBuildProgress` callbacks (#8090).\n- Minor: `QuantizationType` accepts `\"RQ\"` (#8214); HNSW greedy descent stops at level 1\n  (#8035, +3.7% recall@10); `BlobV2Layout` classification helper (#8266); a\n  `ConditionalPutCommitHandler` test matrix covering every routed scheme.\n\n**Security / supply chain:** `rust-stemmers 1.2.0` -> **`frostem`** (#8183) - the unmaintained\ncrate's \"Greek implementation can retain stale UTF-8 byte offsets after shortening a word, then\npanic while slicing the shortened string.\" `frostem` is generated from current upstream Snowball\nand exposes the same 18 algorithms. `strum` and the direct `goosefs-sdk` dependency were dropped;\n`crc32c` left the lockfile and `opendal-http-transport-reqwest` entered it.\n\n**It is not a drop-in for existing FTS indexes.** The two stemmers disagree on a small but\nnon-trivial slice of ordinary English. Measured over `/usr/share/dict/words`, **484 of 235,976\nwords (0.2%) stem differently** - most are the `-ogist` family (`anthropologist` ->\n`anthropolog`), but the rest are everyday vocabulary: `internal` (old `intern`, new `internal`),\n`added` (`ad` vs `add`), `emergency`, `evening`, `interfering`, `erring`. An index built with\n`.stem(true)` under v10 and queried by a v11 binary therefore **silently misses those forms** -\nthe query stems to `internal` while the index holds `intern`. There is no error and no version\ncheck. Two consequences for a rollout: a mixed v10/v11 fleet appending FTS segments to the same\nstore produces segments stemmed two ways, so the upgrade wants to be fleet-coordinated, and any\nstemmed FTS index built before the swap needs one rebuild to become self-consistent.\n\n---\n\n### The v11.0.0-beta.2 -> v11.0.0-beta.6 delta\n\n94 commits, 90 PRs, **four `breaking-change`-labeled**: #8027, #8028, #8347, #8360. At that tag\nthe full v11 delta from `v10.0.0-beta.7` stood at **222 commits and 13 breaking PRs** (#8024,\n#8025, #8026, #8027, #8028, #8051, #8095, #8159, #8172, #8188, #8206, #8347, #8360).\n\n**Breaking:**\n\n- **`LanceFileVersion` lost its ordering** (#8028). `PartialOrd`/`Ord` are gone, so\n  `v >= LanceFileVersion::Next` no longer compiles, and both `From` conversions between selector\n  and concrete version were deleted. Index readers, writers, shufflers and distributed mergers\n  now take an exact `ConcreteFileVersion`. Per the PR: \"Remaining version decisions are\n  exhaustive matches at declared boundaries rather than `>=`, `max`, or selector round-trips.\"\n- **`LanceFileVersion::resolve` changed signature** (#8027):\n  `pub fn resolve(&self) -> Self` became `pub const fn resolve(self) -> ConcreteFileVersion`.\n  Deleted: `iter_non_legacy()`, `support_add_sub_column()`, `support_remove_sub_column(&Field)`.\n  Added: `stable_file_version() -> ConcreteFileVersion` (V2_1), `next_file_version()` (V2_3),\n  `ConcreteFileVersion::to_selector()` and `::is_unstable()`. #8027 also centralized dataset\n  version policies.\n- **`Operation::Project` / `Merge` gained `preserves_nullability: bool`** (#8347). See section\n  9.2 - a nullability tightening must not set it, and setting it makes the operation conflict\n  with concurrent value-writes in either commit order.\n- **`is_maintainable_index_type(&str)` removed** (#8360), replaced by\n  `validate_maintained_indexes(dataset, index_names) -> Result<()>`. Type-URL filtering was\n  unsound: an IVF-PQ over `FixedSizeList<Float64>` passed the check and then made the table\n  unwritable. The replacement is all-or-nothing - it \"reports the first index it cannot maintain\n  rather than returning a usable subset\". Error text: \"index '{}' has type {}, which the MemWAL\n  cannot maintain. Supported: BTree, Inverted, Vector\".\n\n**Format-level:**\n\n- **New manifest feature flag at bit 128** (#8263); `FLAG_UNKNOWN` moved 128 -> 256. Reader and\n  writer must both hold it. **The flag added here did not survive the major.** It was\n  `FLAG_MEM_WAL_INDEX_CATCHUP` from `beta.4` to `beta.17`, then #8680 retired it and #8535 gave\n  the reclaimed bit to `FLAG_COVERED_INDEX_METADATA`, which is what `v11.0.0` shipped. The proto\n  field `Transaction.UpdateMemWalState.require_index_catchup` was deleted with it, and MemWAL\n  catch-up lost its flag gate: an absent `index_catchup` shard now unconditionally means\n  *unknown*. Builds pinned inside `beta.4`..`beta.17` still treat bit 128 as supported and will\n  open a covering dataset instead of refusing it. Section 7.\n- **Transaction proto field 9 deprecated** (#7432): `updated_fragment_offsets` gives way to\n  field 10 `updated_fragment_offset_bitmaps`, \"Per-fragment matched offsets as portable\n  RoaringBitmap bytes\". Writers emit field 10 only; readers prefer 10, falling back to 9 for\n  manifests written before the change.\n- `MemWalIndexDetails.index_catchup` added as `table.proto` field 10.\n- **`IndexCatchupAdvance` never shipped.** #8263 added the message and\n  `CreateIndex.mem_wal_index_catchup_advances`; #8481 deleted both within the same beta window,\n  replacing them with catch-up derived from the version the transaction read. Present at\n  `v11.0.0-beta.5`, absent at `v11.0.0-beta.6`.\n\n**Net-new:**\n\n- MemWAL backpressure is observable: `MemTableStats.frozen_count` / `frozen_bytes` and\n  `ShardWriter::backpressure_stats()` (#8241) - \"Heap bytes still owed to flush\".\n- MemWAL splits logical from storage schema, widening non-PK top-level fields to nullable, so\n  `ShardWriter::delete` no longer requires nullable base columns (#8352).\n- `write_fragments(session=...)` / Java `WriteFragmentBuilder.session(...)` (#8034); a foreign\n  session against a dataset-backed target is rejected.\n- `analyze_plan` appends `tokenized_query=` to FTS leaves (#8414); `explain_plan` deliberately\n  unchanged. Python `lance.tokenize(...)` / `lance.FtsToken(text, position)` preview tokenization\n  with no dataset or index (#8415).\n- `LanceFragment.validate()` (#8428) validates one fragment rather than the whole dataset;\n  `Dataset::validate()` gained stable-row-id invariant checks (#8258, no-op when unused).\n- `BlobFile.read_ranges(ranges) -> list[bytes]` (#8319) - \"The underlying physical reads may be\n  reordered, coalesced, or split for efficiency.\"\n- `lance.fragment.RowIdSequence` (#8356); duplicate ids now rejected.\n- `LanceOperation.Update` carries `updated_fragment_offsets` in Python (#8447) and\n  `updatedFragmentOffsets` in Java (#6748).\n- Java: `Session.Builder` selects registered native cache backends by URI or\n  `CacheBackendConfig`, e.g. `moka://?capacity=1048576` (#8446); `Index.getSizeBytes()` and\n  `IndexDescription.getSegments()` (#8355).\n- Blob v2 supported in `FileFragment::update_columns` (#8344).\n\n**Performance / build:**\n\n- FTS same-column `MUST + SHOULD` scores optional clauses lazily (#8448); conjunction\n  confirmations ordered by `match_cost`, measured **200 -> 120** two-phase `matches()` calls per\n  query (#8354).\n- Release JNI cdylib stripped: `liblance_jni.so` linux-x86-64 **278.35 MB -> 221.0 MB**\n  (-20.6%), `.dynsym` preserved (#8314).\n- x86_64-linux build baseline dropped `target-cpu=haswell` -> **`x86-64-v2`** (#8377), so\n  binaries no longer trap on import on pre-AVX2 hosts.\n- The `time = \"=0.3.47\"` pin was removed from `lance-namespace-impls` (#8296).\n- `retain_versions=0` now errors instead of panicking (#8467); deleting a branch referenced by a\n  tag is reje\n\nFile v0.20.0:references/docs/format/file/encoding.md\n\n# Lance Encoding Strategy\n\nThe encoding strategy determines how array data is encoded into a disk page. The encoding strategy tends to evolve\nmore quickly than the file format itself.\n\n## Older Encoding Strategies\n\nThe 0.1 and 2.0 encoding strategies are no longer documented. They were significantly different from future encoding\nstrategies and describing them in detail would be a distraction.\n\n## Terminology\n\nAn array is a sequence of values. An array has a data type which describes the semantic interpretation of the values.\nA layout is a way to encode an array into a set of buffers and child arrays. A buffer is a contiguous sequence of\nbytes. An encoding describes how the semantic interpretation of data is mapped to the layout. An encoder converts\ndata from one layout to another.\n\nData types and layouts are orthogonal concepts. An integer array might be encoded into two completely different\nlayouts which represent the same data.\n\n![Multiple Encodings](../../images/encoding_v_array.png)\n\n### Data Types\n\nLance uses a subset of Arrow's type system for data types. An Arrow data type is both a data type and an encoding.\nWhen writing data Lance will often normalize Arrow data types. For example, a string array and a large string array\nmight end up traveling down the same path (variable width data). In fact, most types fall into two general paths. One\nfor fixed-width data and one for variable-width data (where we recognize both 32-bit and 64-bit offsets).\n\nAt read time, the Arrow data type is used to determine the target encoding. For example, a string array and large\nstring array might both be stored in the same layout but, at read time, we will use the Arrow data type to determine\nthe size of the offsets returned to the user. There is no requirement the output Arrow type matches the input Arrow\ntype. For example, it is acceptable to write an array as \"large string\" and then read it back as \"string\".\n\n## Search Cache\n\nThe search cache is a key component of the Lance file reader. Random access requires that we locate the physical\nlocation of the data in the file. To do so we need to know information such as the encoding used for a column,\nthe location of the page, and potentially other information. This information is collectively known as the \"search\ncache\" and is implemented as a basic LRU cache. We define a \"initialization phase\" which is when we load the various indexing information into the search cache. The cost of initialization is assumed to be amortized over the lifetime\nof the reader.\n\nWhen performing full scans (i.e. not random access), we should be able to ignore the search cache and sometimes\ncan avoid loading it entirely. We _do_ want to optimize for cold scans as the initialization phase is often not\namortized over the lifetime of the reader.\n\n## Structural Encoding\n\nThe first step in encoding an array is to determine the structural encoding of the array. A structural encoding\nbreaks the data into smaller units which can be independently decoded. Structural encodings are also responsible\nfor encoding the \"structure\" (struct validity, list validity, list offsets, etc.) typically utilizing repetition\nlevels and definition levels.\n\nStructural encoding is fairly complicated! However, the goal is to suck out all the details related to I/O\nscheduling so that compression libraries can focus on compression. This keeps our compression traits simple\nwithout sacrificing our ability to perform random access.\n\nThere are only a few structural encodings. The structural encoding is described by the `PageLayout` message and\nis the top-level message for the encoding.\n\n```protobuf\n%%% proto.message.PageLayout %%%\n```\n\n### Repetition and Definition Levels\n\nRepetition and definition levels are an alternative to validity bitmaps and offset arrays for expressing struct\nand list information. They have a significant advantage in that they combine all of these buffers into a single\nbuffer which allows us to avoid multiple IOPS.\n\nA more extensive explanation of repetition and definition levels can be found in the code. One particular note\nis that we use 0 to represent the \"inner-most\" item and Parquet uses 0 to represent the \"outer-most\" item. Here\nis an example:\n\n#### Definition Levels\n\nConsider the following array:\n\n```text\n[{\"middle\": {\"inner\": 1]}}, NULL, {\"middle\": NULL}, {\"middle\": {\"inner\": NULL}}]\n```\n\nIn Arrow we would have the following validity arrays:\n\n```text\nOuter validity : 1, 0, 1, 1\nMiddle validity: 1, ?, 0, 1\nInner validity : 1, ?, ?, 0\nValues         : 1, ?, ?, ?\n```\n\nThe ? values are undefined in the Arrow format. We can convert these into definition levels as follows:\n\n| Values | Definition | Notes                |\n| ------ | ---------- | -------------------- |\n| 1      | 0          | Valid at all levels  |\n| ?      | 3          | Null at outer level  |\n| ?      | 2          | Null at middle level |\n| ?      | 1          | Null at inner level  |\n\n#### Repetition Levels\n\nConsider the following list array with 3 rows\n\n```text\n[{<0,1>, <>, <2>}, {<3>}, {}], [], [{<4>}]\n```\n\nWe would have three offsets arrays in Arrow:\n\n```text\nOuter-most ([]): [0, 3, 3, 4]\nMiddle     ({}): [0, 3, 4, 4, 5]\nInner      (<>): [0, 2, 2, 3, 4, 5]\nValues         : [0, 1, 2, 3, 4]\n```\n\nWe can convert these into repetition levels as follows:\n\n| Values | Repetition | Notes                                     |\n| ------ | ---------- | ----------------------------------------- |\n| 0      | 3          | Start of outer-most list                  |\n| 1      | 0          | Continues inner-most list (no new lists)  |\n| ?      | 1          | Start of new inner-most list (empty list) |\n| 2      | 1          | Start of new inner-most list              |\n| 3      | 2          | Start of new middle list                  |\n| ?      | 2          | Start of new inner-most list (empty list) |\n| ?      | 3          | Start of new outer-most list (empty list) |\n| 4      | 3          | Start of new outer-most list              |\n\n### Mini Block Page Layout\n\nThe mini block page layout is the default layout for smallish types. This fits most of the classical data types\n(integers, floats, booleans, small strings, etc.) that Parquet and related formats already handle well. As is no\nsurprise, the approach used is pretty similar to those formats.\n\n![Mini Block Layout](../../images/miniblock.png)\n\nThe data is divided into small mini-blocks. Each mini-block should contain a power-of-two number of values (except\nfor the last mini-block) and should be less than 32KiB of compressed data. We have to read an entire mini-block to\nget a single value so we want to keep the mini-block size small. Mini blocks are padded to 8 byte boundaries. This\nhelps to avoid alignment issues. Each mini-block starts with a small header which helps us figure out how much\npadding has been applied.\n\nThe repetition and definition levels are sliced up and stored in the mini-blocks along with the compressed buffers.\nSince we need to read an entire mini-block there is no need to zip up the various buffers and they are stored\none after the other (repetition, definition, values, ...).\n\n#### Buffer 1 (Mini Blocks)\n\n| Bytes | Meaning                             |\n| ----- | ----------------------------------- |\n| 1     | Number of buffers in the mini-block |\n| 2     | Size of buffer 0                    |\n| 2     | Size of buffer 1                    |\n| ...   | ...                                 |\n| 2     | Size of buffer N                    |\n| 0-7   | Padding to ensure 8 byte alignment  |\n| \\*    | Buffer 0                            |\n| 0-7   | Padding to ensure 8 byte alignment  |\n| \\*    | Buffer 1                            |\n| ...   | ...                                 |\n| 0-7   | Padding to ensure 8 byte alignment  |\n| \\*    | Buffer N                            |\n| 0-7   | Padding to ensure 8 byte alignment  |\n\nNote: It is natural to explain this buffer first but it is actually the second buffer in the page.\n\n#### Buffer 0 (Mini Block Metadata)\n\n![Mini Block Layout](../../images/miniblock_meta.png)\n\nTo enable random access we have a small metadata lookup which contains two bytes per mini-block. This lookup\ntells us how many bytes are in each mini block and how many items are in the mini block. This metadata lookup\nmust be loaded at initialization time and placed in the search cache.\n\n| Bits (not bytes) | Meaning                             |\n| ---------------- | ----------------------------------- |\n| 12               | Number of 8-byte words in block 0   |\n| 4                | Log2 of number of values in block 0 |\n| 12               | Number of 8-byte words in block 1   |\n| 4                | Log2 of number of values in block 1 |\n| ...              | ...                                 |\n| 12               | Number of 8-byte words in block N   |\n| 4                | Log2 of number of values in block N |\n\nFor all chunks except the last, the lower 4 bits store `log2(num_values)` and `num_values` must be a power of two.\nFor the last chunk, these bits are set to `0`. The protobuf stores the total number of values in the page, so readers\ncan derive the final chunk size by subtracting the values from earlier chunks.\n\n#### Buffer 2 (Dictionary, optional)\n\nDictionary encoding is an encoding that can be applied at many different levels throughout a file. For example,\nit could be used as a compressive encoding or it could even be entirely external to the file. We've found the\nmost convenient simple place to apply dictionary encoding is at the structural level. Since dictionary indices are\nsmall we always use the mini block layout for dictionary encoding. When we use dictionary encoding we store the\ndictionary in the buffer at index 2. We require the dictionary to be full loaded and decoded at initialization time.\nThis means we don't have to load the dictionary during random access but it does require the dictionary be placed\nin the search cache.\n\nDictionary values are stored as a single buffer and compressed through the block compression path. The compression\nscheme for dictionary values can be configured separately (see `lance-encoding:dict-values-compression` below).\n\n#### Buffer 2 (or 3) (Repetition Index, optional)\n\nIf there is repetition (list levels) then we need some way to translate row offsets into item offsets. The mini\nblocks always store items. During a full scan the list offsets are restored when we decode the repetition levels.\nHowever, to support random access, we don't have the repetition levels available. Instead we store a repetition\nindex in the next available buffer (index 2 or 3 depending on whether the dictionary is present).\n\nThe repetition index is a flat buffer of u64 values. We have N \\* D values where N is the number of mini blocks\nand D Is the desired depth of random access plus one. For example, to support 1-dimensional lookups (random access\nby rows) then D is 2. To support two-dimensional lookups (e.g. rows\\[50\\]\\[17\\]) then we could set D to 3.\n\nCurrently we only support 1-dimensional random access.\nCurrently we do not compress the repetition index.\n\nThis may change in future versions.\n\n| Bytes | Meaning                            |\n| ----- | ---------------------------------- |\n| 8     | Number of rows in block 0          |\n| 8     | Number of partial items in block 0 |\n| 8     | Number of rows in block 1          |\n| 8     | Number of partial items in block 1 |\n| ...   | ...                                |\n| 8     | Number of rows in block N          |\n| 8     | Number of partial items in block N |\n\nThe last 8 bytes of each block stores the number of \"partial\" items. These are items leftover after the last\ncomplete row. We don't require rows to be bounded by mini-blocks so we need to keep track of this. For example,\nif we have 10,000 items per row then we might have several mini-blocks with only partial items and 0 rows.\n\nAt read time we can use this repetition index to translate row offsets into item offsets.\n\n#### Mini Block Compression\n\nThe mini block layout relies on the compression algorithm to handle the splitting of data into mini-blocks. This\nis because the number of values per block will depend on the compressibility of the data. As a result, there is\na special trait for mini block compression.\n\nThe data compression algorithm is the algorithm that decides chunk boundaries. The repetition and definition levels\nare then sliced appropriately and sent to a block compressor. This means there are no constraints on how the repetition\nand definition levels are compressed.\n\nBeyond splitting the data into mini-blocks, there are no additional constraints. We expect to fully decode mini\nblocks as opaque chunks. This means we can use any compression algorithm that we deem suitable.\n\n#### Protobuf\n\n```protobuf\n%%% proto.message.MiniBlockLayout %%%\n```\n\nThe protobuf for the mini block layout describes the compression of the various buffers. It also tells us\nsome information about the dictionary (if present) and the repetition index (if present).\n\n### Full Zip Page Layout\n\nThe full zip page layout is a layout for larger values (e.g. vector embeddings) which are large but not so large\nthat we can justify a single IOP per value. In this case we are trying to avoid storing a large amount of \"chunk\noverhead\" (both in terms of buffer space and the RAM space in the search cache that we would need to store the\nrepetition index). As a tradeoff, we are introducing a second IOP per-range for random access reads (unless the\ndata is fixed-width such as vector embeddings).\n\nWe currently use 256 bytes as the cutoff for the full zip layout. At this point we would only be fitting 16 values\nin a 4KiB disk sector and so creating a mini-block descriptor for every 16 values would be too much overhead.\n\nAs a further consequence, we must ensure that the compression algorithm is \"transparent\" so that we can index\nindividual values after compression has been applied. This prevents us from using compression algorithms such\nas delta encoding. If we want to apply general compression we have to apply them on a per-value basis. The way\nwe enforce this is by requiring the compression to return either a flat fixed-width or variable-width layout\nso that we know the location of each element.\n\nThe repetition and definition levels, along with all compressed buffers, are all zipped together into a single\nbuffer.\n\n#### Data Buffer (Buffer 0)\n\n![Full Zip Layout](../../images/fullzip.png)\n\nThe data buffer is a single buffer that contains the repetition, definition, and value data, all zipped into a\nsingle buffer. The repetition and definition information are combined and byte packed. This is referred to as\na control word. If the value is null or an empty list, then the control word is all that is serialized. If there\nis no validity or repetition information then control words are not serialized. If the value is variable-width\nthen we encode the size of the value. This is either a 4-byte or 8-byte integer depending on the width used in\nthe offsets returned by the compression (in future versions this will likely be encoded with some kind of\nvariable-width integer encoding). Finally the value buffers themselves are appended.\n\n| Bytes | Meaning        |\n| ----- | -------------- |\n| 0-4   | Control word 0 |\n| 0/4/8 | Value 0 size   |\n| \\*    | Value 0 data   |\n| ...   | ...            |\n| 0-4   | Control word N |\n| 0/4/8 | Value N size   |\n| \\*    | Value N data   |\n\nNote: a fixed-width data type that has no validity information (e.g. non-nullable vector embeddings) is simply a\nflat buffer of data.\n\n#### Repetition Index (Buffer 1)\n\n![Full Zip Layout](../../images/fullzip_rep.png)\n\nIf there is repetition information or the values are variable width then we need additional help to locate values\nin the disk page. The repetition index is an array of u64 values. There is one value per row and the value is an\noffset to the start of that row in the data buffer. To perform random access we require two IOPS. First we issue\nan IOP into the repetition index to determine the location and then a second IOP into the data buffer to load the\ndata. Alternatively, the entire repetition index can be loaded into memory in the initialization phase though this\ncan lead to high RAM usage by the search cache.\n\nThe repetition index must have a fixed width (or else we would need a repetition index to read the repetition\nindex!) and be transparent. As a result the compression options are limited. That being said, there is little\nvalue (in terms of performance) in compressing the repetition index. It is never read in its entirety as it is\nnot needed for full scans. Currently the repetition index is always compressed with simple (non-chunked) byte\npacking into 1,2,4, or 8 byte values.\n\n#### Protobuf\n\n```protobuf\n%%% proto.message.FullZipLayout %%%\n```\n\nThe protobuf for the full zip layout describes the compression of the data buffer. It also tells us the\nsize of the control words and how many bits we have per value (for fixed-width data) or how many bits we\nhave per offset (for variable-width data).\n\n### Sparse Page Layout\n\nSparse pages require Lance 2.3. They represent flat or nested Arrow structure directly as slot-domain mappings instead\nof dense repetition and definition events. Writers emit this layout only in files declared as 2.3 or above. The layout is\nidentified only by `PageLayout`; field metadata does not identify the layout of an existing page.\n\nA domain is a layer-local integer coordinate space `[0, num_slots)`, and a slot is one element in that space. The\nouter-most domain contains the page's top-level rows. Each layer maps its parent domain to the next layer's parent\ndomain, and the terminal child domain contains the leaf value slots stored in value chunks.\n\nStructural layers are ordered from outer-most to inner-most:\n\n- validity maps a nullable item or struct slot to valid or null\n- list maps non-empty parent slots to variable-size child ranges\n- fixed-size-list maps each parent slot to a child range of a fixed dimension\n\nThe layer list may be empty for a flat, non-nullable leaf page. In that case the scheduling domain and\n`num_visible_items` must be equal. The explicit writer currently emits its normalized all-valid layer even when a flat\npage could use this shorter wire representation.\n\nA list slot that is valid and absent from `non_empty_positions` is an empty list. Maps use the same structural contract\nas lists. The terminal child-domain size equals `SparseLayout.num_visible_items`.\n\n`SparseLayout.num_visible_items` is the number of leaf value slots encoded in value chunks. Null leaf slots count\nbecause they still occupy positions in Arrow's leaf value buffer; a nullable primitive with 100 slots, including 30\nnulls, has 100 visible items. `SparseLayout.num_items` is the number of entries in the equivalent dense repetition and\ndefinition stream. It equals `num_visible_items` plus one structural placeholder for every list slot without children.\nThe first layer's `num_slots` is the logical top-level row count used for projection.\n\nPosition sets have four semantic representations: `empty`, `all`, one non-empty `range`, or an `explicit`\ndelta-compressed `u64` buffer. Count sets are `empty`, one positive `constant` value, or an `explicit` compressed `u64`\nbuffer. Every layer has a `SparseValiditySet` whose meaning is explicit:\n\n- `SPARSE_VALIDITY_NULL_POSITIONS`: stored positions are null and all other positions are valid\n- `SPARSE_VALIDITY_VALID_POSITIONS`: stored positions are valid and all other positions are null\n\nThe unspecified validity meaning is invalid. Both polarities are part of the wire contract and have identical Arrow\nsemantics after normalization.\n\n#### Writer Selection\n\nWriters may emit this layout only for Lance 2.3+ fields. A field can request it explicitly with\n`lance-encoding:structural-encoding=sparse`; the same request is an input error for earlier file versions. Without an\nexplicit structural encoding, the Lance 2.3 writer selects sparse only when the dense mini-block repetition/definition\nbudget would split the page or one top-level row exceeds that budget, and only when the value path is supported by the\nsparse writer. Explicit `miniblock`, `fullzip`, and `sparse` requests are not changed by this automatic policy. Lance\n2.2 and earlier writers never select sparse.\n\nUnsupported sparse value paths, including dictionary values and variable-width packed structs, retain their dense\nbehavior. Writers normalize Arrow validity and list structure once. Within-budget dense pages do not build sparse\nposition/count plans. All-valid layers use null positions plus `empty`; all-null layers use valid positions plus\n`empty`. Other layers choose the validity polarity with the lower semantic encoded cost, with ties using null\npositions. Field metadata controls writer selection only: readers always use `PageLayout` to determine the layout of\nan encoded page and must not use field metadata for that decision.\n\nPages without a value payload keep the existing canonical `ConstantLayout`: structural-only types such as an empty\nstruct, and leaf pages whose visible values are all null, do not emit `SparseLayout`. An explicitly sparse page with\nat least one non-null visible value does emit `SparseLayout`, even when all non-null values are equal. This boundary\navoids introducing a second structural-only representation without evidence that it improves the existing constant\nencoding.\n\n#### Buffers and Selective Reads\n\nA sparse page contains the following physical buffers:\n\n| Buffer | Contents |\n| ------ | -------- |\n| 0 | Value chunk metadata, one 8-byte entry per chunk |\n| 1 | Mini-block compressed value chunks without repetition or definition levels |\n| 2+ | One buffer for each explicit position or count set, in structural-layer field order |\n\nEach value chunk metadata entry stores `(chunk_size / 8) - 1` as little-endian `u32`, followed by its visible value\ncount as little-endian `u32`. Chunk sizes must be positive multiples of 8 and fit this representation. The sum of\nchunk sizes must equal buffer 1 exactly and the sum of chunk value counts must equal `num_visible_items`. A value chunk\ncontains at most 32,768 visible values. `num_buffers` describes the number of value buffers inside every chunk and\nexcludes the structural buffers.\n\nGeneral-compressed sparse buffers use the existing length-prefixed LZ4 or Zstd representation and must not contain\nanother general-compression wrapper. SparseLayout does not impose additional size or descriptor-complexity limits on\notherwise representable buffers.\n\nReaders normalize structural metadata once, project requested top-level ranges through each layer, and read only value\nchunks that intersect the resulting leaf ranges. When no leaf range remains, readers rebuild offsets and validity from\nthe structural plan without reading buffer 1.\n\n#### Caching and Point Reads\n\nReader initialization loads buffer 0 and every explicit structural buffer, then validates and normalizes them into a\ncached page plan. The cached state contains parsed value-chunk descriptors and prefix offsets, decoded semantic\nposition/count sets, validity, and the ordered structural layers. It does not contain value payload bytes from buffer\n1. The plan is cached per field and page and reused by later scans, range reads, and takes.\n\nAfter that plan is cached, reading one primitive leaf value reads only the value chunk that contains it. A cold read\nfirst loads the structural metadata and then the intersecting value chunk. Reading one top-level list or\nfixed-size-list value may intersect multiple leaf chunks and reads each intersecting chunk. A selection whose projected\nstructure contains no leaf slots reads no value chunk.\n\n#### Validation\n\nReaders must reject malformed sparse metadata instead of inferring or repairing it. Required checks include:\n\n- physical buffer count, chunk-count bounds, and every checked offset/size range\n- first-layer row domain, adjacent parent/child domain chaining, and terminal visible-value domain\n- semantic set cardinality, explicit position ordering and bounds, and validity meaning\n- exact `num_items`\n- list non-empty positions being valid, count cardinality, positive counts, and child-count sum\n- fixed-size-list dimension and checked child-domain multiplication\n- value chunk byte/value sums, size representation and alignment, general-compression headers, descriptor buffer\n  count, and complete chunk consumption\n\n```protobuf\n%%% proto.message.SparseLayout %%%\n```\n\n```protobuf\n%%% proto.message.SparseStructuralLayer %%%\n```\n\n```protobuf\n%%% proto.message.SparseValidityLayer %%%\n```\n\n```protobuf\n%%% proto.message.SparseListLayer %%%\n```\n\n```protobuf\n%%% proto.message.SparseFixedSizeListLayer %%%\n```\n\n```protobuf\n%%% proto.message.SparseValiditySet %%%\n```\n\n```protobuf\n%%% proto.message.SparsePositionSet %%%\n```\n\n```protobuf\n%%% proto.message.SparseCountSet %%%\n```\n\n### Constant Page Layout\n\nThis layout is used when all (visible) values in the page are the same scalar value.\n\nThe all-null case is represented by a constant page without an inline scalar value. Surprisingly, this does not\nmean there is no data. If there are any levels of struct or list then we need to store the rep/def levels so that\nwe can distinguish between null structs, null lists, empty lists, and null values.\n\n#### Repetition and Definition Levels (Buffers 0 and 1)\n\nNote: We currently store rep levels in the first buffer with a flat layout of 16-bit values and def levels\nin the second buffer with a flat layout of 16-bit values. This will likely change in future versions.\n\n#### Protobuf\n\n```protobuf\n%%% proto.message.ConstantLayout %%%\n```\n\nAll we need to know is the meaning of each rep/def level and (when present) the inline scalar value bytes.\n\n### Blob Page Layout\n\nThe blob page layout is a layout for large binary values where we would only have a few values per disk page.\nThe actual data is stored out-of-line in external buffers. The disk page stores a \"description\" which is a\nstruct array of two fields: `position` and `size`. The `position` is the absolute file offset of the blob and\nthe `size` is the size (in bytes) of the blob. The inner page layout describes how the descriptions are encoded.\n\nThe validity information (definition levels) is smuggled into the descriptions. If the size and position are\nboth zero then the value is empty. Otherwise, if the size is zero and the position is non-zero then the\nvalue is null and the position is the definition level.\n\nThis layout is only recommended when you can justify a single IOP per value. For example, when values are 1MiB\nor larger.\n\nThis layout has no buffers of its own and merely wraps an inner layout.\n\n#### Protobuf\n\n```protobuf\n%%% proto.message.BlobLayout %%%\n```\n\nSince we smuggle the validity into the descriptions we don't need to store it in the inner layout and so the\nrep/def meaning is stored in the blob layout and the rep/def meaning in the inner layout will be 1 all valid item\nlayer.\n\n## Semi-Structural Transformations\n\nThere are some data transformations that are applied to the data before (or during) the structural encoding process.\nThese are described here.\n\n### Dictionary Encoding\n\nDictionary encoding is a technique that can be applied to any kind of array. It is useful when there are not very\nmany unique values in the array. First, a \"dictionary\" of unique values is created. Then we create a second array\nof indices into the dictionary.\n\nDictionary encoding is also known as \"categorical encoding\" in other contexts.\n\nDictionary encoding could be treated as simply another compression technique but, when applied, it would be an\nopaque compression technique which would limit its usability (e.g. in a full zip context). As a result, we apply\nit before any structural encoding takes place. This allows us to place the dictionary in the search cache for\nrandom access.\n\n### Struct Packing\n\nStruct packing is an alternative representation to apply to struct values. Instead of storing that struct in a\ncolumnar fashion it will be stored in a row-major fashion. This will reduce the number of IOPS needed for random\naccess but will prevent the ability to read a single field at a time. This is useful when all fields in the struct\nare always accessed together.\n\nPacked struct is always opt-in (see section on configuration below).\n\nIn Lance 2.1, packed struct is limited to fixed-width children (`PackedStruct`).\nStarting with Lance 2.2, variable-width children are also supported via `VariablePackedStruct`.\n\n### Fixed Size List\n\nFixed size lists are an Arrow data type that needs specialized handling at the structural level. If the underlying\ndata type is primitive then the fixed size list will be primitive (e.g. a tensor). If the underlying data type\nis structural (struct/list) then the fixed size list is structural and should be treated the same as a\nvariable-size list.\n\nWe don't want compression libraries to need to worry about the intricacies of fixed-size lists. As a result we\nflatten the list as part of structural encoding. This complicates random access as we must translate between\nrows (an entire fixed size list) and items (a single item in the list).\n\nIf the items in a fixed size list are nullable then we do not treat that validity array as a repetition or\ndefinition level. Instead, we store the validity as a separate buffer. For example, when encoding nullable fixed\nsize lists with mini-block encoding the validity buffer is another buffer in the mini-block. When encoding\nnullable fixed size lists with full-zip encoding the validity buffer is zipped together with the values.\n\nThe good news is that fixed size lists are entirely a structural encoding concern. Compression techniques are\nfree to pretend that the fixed-size list data type does not exist.\n\n## Compression\n\nOnce a structural encoding is chosen we must determine how to compress the data. There are various buffers that\nmight be compressed (e.g. data, repetition, definition, dictionary, etc.). The available compression algorithms\nare also constrained by the structural encoding chosen. For example, when using the full zip layout we require\ntransparent compression. As a result, each encoding technique may or may not be usable in a given scenario. In\naddition, the same technique may be applied in a different way depending on the encoding chosen.\n\nIn implementation terms we have a trait for each compression constraint. The techniques then implement the traits\nthat they can be applied to. To start with, here is a summary of compression techniques which are implemented in\nat least one scenario and a list of which traits the technique implements. A ❓ is used to indicate that the\ntechnique should be usable in that context but we do not yet do so while a ❌ indicates that the technique is\nnot usable because it is not transparent. Note, even though a technique is not transparent it can still be applied\non a per-value basis. We use ☑️ to mark a technique that is applied on a per-value basis:\n\n| Compression     | Used in Block Context | Used in Full Zip Context | Used in Mini-Block Context |\n| --------------- | --------------------- | ------------------------ | -------------------------- |\n| Flat            | ✅ (2.1)              | ✅ (2.1)                 | ✅ (2.1)                   |\n| Variable        | ✅ (2.1)              | ✅ (2.1)                 | ✅ (2.1)                   |\n| Constant        | ✅ (2.1)              | ❓                       | ❓                         |\n| Bitpacking      | ✅ (2.1)              | ❓                       | ✅ (2.1)                   |\n| Fsst            | ❓                    | ✅ (2.1)                 | ✅ (2.1)                   |\n| Rle             | ✅ (2.2)              | ❌                       | ✅ (2.1)                   |\n| ByteStreamSplit | ❓                    | ❌                       | ✅ (2.1)                   |\n| General         | ✅ (2.2)              | ☑️ (2.1)                 | ✅ (2.1)                   |\n\nIn the following sections we will describe each technique in a bit more detail and explain how it is utilized\nin various contexts.\n\n### Flat\n\nFlat compression is the uncompressed representation of fixed-width data. There is a single buffer of data\nwith a fixed number of bits per value.\n\nWhen applied in a mini-block context we find the largest power of 2 number of values that will be less than\n8,186 bytes and use that as the block size.\n\n### Variable\n\nVariable compression is the uncompressed representation of variable-width data. There is a buffer of values and\na buffer of offsets.\n\nWhen applied in a mini-block context each block may have a different number of values. We walk through the values\nuntil we find the point that would exceed 4,096 bytes and then use the most recent power of 2 number of values that\nwe have passed.\n\n### Constant\n\nConstant compression is currently only utilized in a few specialized scenarios such as all-null arrays.\n\nThis will likely change in future versions.\n\n### Bitpacking\n\nBitpacking is a compression technique that removes the unused bits from a set of values. For example, if we have\na u32 array and the maximum value is 5000 then we only need 13 bits to store each value.\n\nWhen used in a mini-block context we always use 1024 values per block. In addition, we store the compressed bit\nwidth inline in the block itself.\n\nBitpacking is, in theory, usable in a full zip context. However, values in this context are so large that shaving\noff a few bits is unlikely to have any meaningful impact. Also, the full-zip context keeps things byte-aligned and\nso we would have to remove at least 8 bits per value.\n\n### Fsst\n\nFsst is a fast and transparent compression algorithm for variable-width data. It is the primary compression\nalgorithm that we apply to variable-width data.\n\nCurrently we use a single FSST symbol table per disk page and store that symbol table in the protobuf description.\nThis is for historical reasons and is not ideal and will likely change in future versions.\n\nWhen FSST is applied in a mini-block context we simply compress the data and let the underlying compressor (always\n`Variable` at the moment) handle the chunking.\n\n### Run Length Encoding (RLE)\n\nRun length encoding is a compression technique that compresses large runs of identical values into an array\nof values and an array of run lengths. This is currently used in the mini-block context. To determine if we\nshould apply run-length encoding we look at the number of runs divided by the number of values. If the ratio is\nbelow a threshold (by default 0.5) then we apply run-length encoding.\n\n### Byte Stream Split (BSS)\n\nByte stream split is a compression technique that splits multi-byte values by byte position, creating separate streams\nfor each byte position across all values. This is a rudimentary and simple form of translating floating point values\ninto a more compressible format because it tends to cluster the mantissa bits together which are often consistent\nacross a column of floating point values. It does not actually make the data smaller by itself. As a result, BSS is\nonly applied if general compression is also applied on the column.\n\nWe currently determine whether or not to apply BSS by looking at an entropy statistics. There is a configurable\nsensitivity parameter. A sensitivity of 0.0 means never apply BSS and a sensitivity of 1.0 means always apply BSS.\n\n### General\n\nGeneral compression is a catch-all term for classical opaque compression techniques such as LZ4, ZStandard, Snappy,\netc. These techniques are typically back-referencing compressors which replace values with a \"back reference\" to\na spot where we already saw the value.\n\nWhen applied in a mini-block context we run general compression after all other compression and compress the entire\nmini-block.\n\nWhen applied in a full zip context we run general compression on each value.\n\nThe only time general compression is automatically applied is in a full-zip context when we have values that are at\nleast 32KiB large. This is because general compression can be CPU intensive.\n\nHowever, general compression is highly effective and we allow it to be opted into in other contexts via configuration.\n\n## Compression Configuration\n\nThe following section lists the available configuration options. These can be set programmatically through writer\noptions. However, they can also be set in the field metadata in the schema.\n\n| Key                                  | Values                               | Default          | Description                                                                             |\n| ------------------------------------ | ------------------------------------ | ---------------- | --------------------------------------------------------------------------------------- |\n| `lance-encoding:compression`         | `lz4`, `zstd`, `none`, ...           | `none`           | Opt-in to general compression. The value indicates the scheme.                          |\n| `lance-encoding:compression-level`   | Integers (range is scheme dependent) | Varies by scheme | Higher indicates more work should be done to compress the data.                         |\n| `lance-encoding:rle-threshold`       | `0.0-1.0`                            | `0.5`            | See below                                                                               |\n| `lance-encoding:bss`                 | `off`, `on`, `auto`                  | `auto`           | See below                                                                               |\n| `lance-encoding:dict-divisor`        | Integers greater than 1              | `2`              | See below                                                                               |\n| `lance-encoding:dict-size-ratio`     | `0.0-1.0`                            | `0.8`            | See below                                                                               |\n| `lance-encoding:dict-values-compression` | `lz4`, `zstd`, `none`             | `lz4`            | Select general compression scheme for dictionary values                                 |\n| `lance-encoding:dict-values-compression-level` | Integers (scheme dependent) | Varies by scheme | Compression level for dictionary values general compression                             |\n| `lance-encoding:general`             | `off`, `on`                          | `off`            | Whether to apply general compression.                                                   |\n| `lance-encoding:packed`              | Any string                           | Not set          | Whether to apply packed struct encoding (see above).                                    |\n| `lance-encoding:structural-encoding` | `miniblock`, `fullzip`, `sparse`     | Not set          | Force a structural encoding; `sparse` requires Lance 2.3.                                |\n\n### Configuration Details\n\n#### Compression Scheme\n\nThe `lance-encoding:compression` setting enables general-purpose compression algorithms to be applied. Available schemes:\n\n- **`lz4`**: Fast compression with good compression ratios. Default compression level is fast mode.\n- **`zstd`**: High compression ratios with configurable levels (0-22). Better compression than LZ4 but slower.\n- **`none`**: No general compression applied (default).\n- **`fsst`**: Fast Static Symbol Table compression for string data.\n\nGeneral compression is applied on top of other encoding techniques (RLE, BSS, bitpacking, etc.) to further reduce\ndata size. For mini-block layouts, compression is applied to entire mini-blocks. For full-zip layouts with large values\n(≥32KiB), compression is automatically applied per-value.\n\n#### Compression Level\n\nThe compression level is scheme dependent. Currently the following schemes support the following levels:\n\n| Scheme | Crate Used                              | Levels | Default                                                                                                                                                                                                                           |\n| ------ | --------------------------------------- | ------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |\n| `zstd` | [`zstd`](https://crates.io/crates/zstd) | `0-22` | `crate dependent` (3 as of this writing)                                                                                                                                                                                          |\n| `lz4`  | [`lz4`](https://crates.io/crates/lz4)   | N/A    | The LZ4 crate has two modes (fast and high compression) and currently this is not exposed to configuration. The LZ4 crate wraps a C library and the default is dependent on the C library. The default as of this writing is fast |\n\nHigher compression levels generally provide better compression at the cost of slower encoding speed. Decoding speed\nis typically less affected by the compression level.\n\n#### Run Length Encoding (RLE) Threshold\n\nThe RLE threshold is used to determine whether or not to apply run-length encoding. The threshold is a ratio\ncalculated by dividing the number of runs by the number of values. If the ratio is less than the threshold then\nwe apply run-length encoding. The default is 0.5 which means we apply run-length encoding if the number of runs\nis less than half the number of values.\n\n**Key points:**\n- RLE is automatically selected when data has sufficient repetition (run_count / num_values < threshold)\n- Supported types: All fixed-width primitives (u8, i8, u16, i16, u32, i32, f32, u64, i64, f64)\n- Maximum chunk size: 2048 values per mini-block\n- Setting threshold to `0.0` effectively disables RLE\n- Setting threshold to `1.0` makes RLE very aggressive (used whenever any runs exist)\n\nRLE is particularly effective for:\n- Sorted or partially sorted data\n- Columns with many repeated values (status codes, categories, etc.)\n- Low-cardinality columns\n\n#### Byte Stream Split (BSS)\n\nThe configuration variable for BSS is a simple enum. A value of `off` means to never apply BSS, a value of `on`\nmeans to always apply BSS, and a value of `auto` means to apply BSS based on an entropy calculation (see code for\ndetails).\n\n**Important:** BSS is only applied when the `lance-encoding:compression` variable is also set (to a non-`none` value).\nBSS is a data transformation that makes floating-point data more compressible; it does not reduce size on its own.\n\n**Key points:**\n- Supported types: Only 32-bit and 64-bit data (f32, f64, timestamps)\n- Maximum chunk sizes: 1024 values (f32), 512 values (f64)\n- `auto` mode: Uses entropy analysis with 0.5 sensitivity threshold\n- `on` mode: Always applies BSS for supported types\n- `off` mode: Never applies BSS\n\nBSS works by splitting multi-byte values by byte position, creating separate byte streams. This clusters similar\nbits together (especially mantissa bits in floating-point numbers), which general compression algorithms can then\ncompress more effectively.\n\nBSS is particularly effective for:\n- Floating-point measurements with similar ranges\n- Time-series data with consistent precision\n- Scientific data with correlated mantissa patterns\n\n#### Dictionary Encoding Controls\n\nDictionary encoding is gated by a few heuristics.\nThe decision is made on the leaf value page, so nested types can still benefit.\nFor example, `List<u32>` can use dictionary encoding for its `u32` values.\n\nTwo field-level metadata keys control when dictionary encoding is attempted:\n\n- `lance-encoding:dict-divisor` (default `2`): the encoder computes a unique-value budget as `num_values / divisor`\n- `lance-encoding:dict-size-ratio` (default `0.8`): the estimated dictionary-encoded representation must stay below this ratio of the raw page size\n\nThere are additional global guards available as environment variables:\n\n- `LANCE_ENCODING_DICT_TOO_SMALL` (minimum page size before trying dictionary encoding, default `100` values)\n- `LANCE_ENCODING_DICT_DIVISOR` (fallback divisor when field metadata is not set, default `2`)\n- `LANCE_ENCODING_DICT_MAX_CARDINALITY` (upper cap for dictionary entries, default `100000`)\n- `LANCE_ENCODING_DICT_SIZE_RATIO` (fallback ratio when field metadata is not set, default `0.8`)\n\nDictionary encoding is effective when values repeat frequently and the number of distinct values stays low.\n\n#### Dictionary Values Compression\n\nDictionary values are compressed through the block-compression path and have their own configuration:\n\n- `lance-encoding:dict-values-compression`: `lz4`, `zstd`, `none`\n- `lance-encoding:dict-values-compression-level`: optional scheme-specific level\n\nEnvironment-variable fallbacks:\n\n- `LANCE_ENCODING_DICT_VALUES_COMPRESSION`\n- `LANCE_ENCODING_DICT_VALUES_COMPRESSION_LEVEL`\n\nPriority order is:\n\n1. Field metadata (`dict-values-*`)\n2. Environment variables (`LANCE_ENCODING_DICT_VALUES_*`)\n3. Default (`lz4`)\n\n`none` disables general (opaque) compression for dictionary values. For fixed-width dictionary values, structural\nencodings such as RLE or bitpacking may still be selected when beneficial.\n\n#### Packed Struct Encoding\n\nPacked struct encoding is a semi-structural transformation described above. When enabled, struct values are stored\nin row-major format rather than the default columnar format. This reduces the number of I/O operations needed for\nrandom access but prevents reading individual fields independently.\n\nThis is always opt-in and should only be used when all struct fields are typically accessed together.\n\n#### Mini-Block Size Tuning\n\nEach mini-block contains at most 4096 values by default. Because an entire mini-block must be fetched to\nread any value within it, workloads that read only a small contiguous slice of each mini-block may experience\nread amplification.\n\nThe default is appropriate for the vast majority of deployments. Local disks and typical cloud object storage\n(where the client and bucket are in the same region) have more than enough bandwidth that the overhead from\nthe default mini-block size is negligible. You should only consider changing this setting if you have\nconfirmed — through profiling — that mini-block read amplification is saturating your available bandwidth\n(for example, accessing a remote object store over a constrained network link).\n\nThe maximum number of values per mini-block can be tuned via an environment variable:\n\n- `LANCE_MINIBLOCK_MAX_VALUES` (default `4096`, maximum `32768`): upper bound on the number of values in a single mini-block chunk.\n\nReducing this value produces smaller mini-blocks, which reduces the amount of data fetched per read at the\ncost of more mini-blocks and slightly more metadata overhead. Increasing it can reduce metadata overhead and\nimprove throughput for highly compressible data, but it may increase random-read amplification.\n\nFile v0.20.0:references/docs/format/file/index.md\n\n# Lance File Format\n\nThe Lance file format is a columnar container optimized for cloud object stores, random access, and Arrow-native processing. It deliberately focuses on page layout and encoding mechanics, while leaving table semantics and search structures to higher layers.\n\n## Design Goals\n\n### No Row Groups\n\nLance does not use Parquet-style row groups. Each column may have its own number of pages, which keeps column data in large storage-friendly chunks regardless of schema width and avoids coupling scanner partitioning to physical file layout.\n\n### Random-Access-Friendly Encoding\n\nPages are designed so readers can fetch contiguous row ranges with a small and predictable number of I/O operations. This is important for selective filters, point lookups, vector-search follow-up reads, and ML training workloads that sample rows non-sequentially.\n\n### Functional Decomposition\n\nThe file layer does not bundle table-level statistics or query-side indices into the base file structure. Those capabilities are defined as separate index formats so they can evolve independently of the core file container.\n\n## File Structure\n\nA Lance file is a container for tabular data. The data is stored in \"disk pages\". Each disk page contains some rows\nfor a single column. There may be one or more disk pages per column. Different columns may have different numbers of\ndisk pages. Metadata at the end of the file describes where the pages are located and how the data is encoded.\n\n![Format Overview](../../images/file_high_level_overview.png)\n\n!!! Note\n\n    This page describes the container specification. We also have a set of default encodings that are used to encode\n    data into disk pages. See the [Encoding Strategy](encoding.md) page for more details.\n\n### Disk Pages\n\nDisk pages are designed to be large enough to justify a dedicated I/O operation, even on cloud storage, typically several megabytes. Using a larger page size may reduce the number of I/O operations required to read a file, but it also increases the amount of memory required to write the file. In practice, very large page sizes are not useful when high speed reads are required because large contiguous reads need to be broken into smaller reads for performance (particularly on cloud storage). As a result, a default of 8MB is recommended for the page size and should yield ideal performance on all storage systems.\n\nDisk pages should not generally be opaque. It is possible to read a portion of a disk page when a subset of the rows are\nrequired. However, the specifics of this process depend on the column encoding which is described in a later section.\n\n### No Row Groups\n\nUnlike similar formats, there is no \"row group\" concept, only pages. We believe the concept of row groups to be\nfundamentally harmful to performance. If the row group size is too small then columns will be split into \"runt pages\" which yield poor read performance on cloud storage. If the row group size is too large then a file writer will need\na large amount of RAM since an entire row group must be buffered in memory before it can be written. Instead, to split\na file amongst multiple readers we rely on the fact that partial page reads are possible and have minimal read\namplification. As a result, you can split the file at whatever row boundary you want.\n\n### Buffer Alignment\n\nThe file format does not require that buffers be contiguous as buffers are referenced by absolute offsets. In practice,\nwe always align buffers to 64 byte boundaries.\n\n### External Buffers\n\nEvery page in the file is referenced by an absolute offset. This means that non-page data may be inserted amongst the\npages. This can be useful for storing extremely large data types which might only fit a few rows per page otherwise. We\ncan instead store the data out-of-line and store the locations in a page.\n\nIn addition, the file format supports \"global buffers\" which can be used for auxiliary data. This may be used to\nstore a file schema, file indexes, column statistics, or other metadata. References to the global buffers are stored\nin a special spot in the footer.\n\n### Column Descriptors\n\nAt the tail of the file is metadata that describes each page in the file, particularly the encoding strategy used.\nThis metadata consists of a series of \"column descriptors\", which are standalone protobuf messages for each column\nin the file. Since each column has its own message there is no need to read all file metadata if you are only interested\nin a subset of the columns. However, in many cases, the column descriptors are small enough that it is cheaper to read\nthe entire footer in a single read than split it into multiple reads.\n\n### Offsets & Footer\n\nAfter the column descriptors there are offset arrays for the column descriptors and global buffers. These simply\npoint to the locations of each item. Finally, there is a fixed-size footer which describes the position of the\noffset arrays and start of the metadata section.\n\n### Identifiers and Type Systems\n\nThis basic container format has no concept of types. These are added later by the encoding layer. All columns are\nreferenced by an integer \"column index\". All global buffers are referenced by an integer \"global buffer index\".\nThe schema is typically stored in the global buffers, but the file format is unaware of this.\n\n## Reading Strategy\n\nThe file metadata will need to be known before reading the data. A simple approach for loading the footer is to\nread one sector from the end (sector depends on the filesystem, 4KiB for local disk, larger for cloud storage). Then\nparse the footer and read the rest of the metadata (at this point the size will be known). This requires 1-2 IOPS. By\nstoring the metadata size in some other location (e.g. table manifest) it is possible to always read the footer in\na single IOP. If there are _many_ columns in the file and only some are desired then it may be better to read\nindividual columns instead of reading all column metadata, increasing the number of IOPS but decreasing the amount\nof data read.\n\nNext, to read the data, scan through the pages for each column to determine which pages are needed. Each page stores\nthe row offset of the first row in the page. This makes it easy to quickly determine the required pages. The encoding\ninformation for the page can then be used to determine exactly which byte ranges are needed from the page.\n\nDisk pages should be large enough that there should no significant benefit to sequentially reading the file. However,\nif such a use case is desired then the file can be read sequentially once the metadata is known, assuming you want to\nread all columns in the file.\n\n## Detailed Overview\n\n![Format Overview](../../images/file_overview.png)\n\nA detailed description of the file layout follows:\n\n```protobuf\n// Note: the number of buffers (BN) is independent of the number of columns (CN)\n//       and pages.\n//\n//       \n\nArchive v0.19.0: 62 files, 376724 bytes\n\nFiles: CHANGELOG.md (50766b), LICENSE.txt (9157b), references/changelog-v7-v12.md (82549b), references/docs/format/file/encoding.md (47128b), references/docs/format/file/index.md (9612b), references/docs/format/file/versioning.md (2406b), references/docs/format/index.md (3514b), references/docs/format/index/index.md (15455b), references/docs/format/index/indices-compaction.drawio.svg (51382b), references/docs/format/index/indices-fragment handling.drawio.svg (21857b), references/docs/format/index/scalar_index.drawio.svg (8211b), references/docs/format/index/scalar/bitmap.md (1405b), references/docs/format/index/scalar/bloom_filter.md (5393b), references/docs/format/index/scalar/btree.md (3070b), references/docs/format/index/scalar/fmindex.md (4646b), references/docs/format/index/scalar/fts.md (19092b), references/docs/format/index/scalar/label_list.md (1729b), references/docs/format/index/scalar/ngram.md (1894b), references/docs/format/index/scalar/rtree.md (8350b), references/docs/format/index/scalar/zonemap.md (2986b), references/docs/format/index/starter-example.drawio.svg (14623b), references/docs/format/index/system/frag_reuse.md (2747b), references/docs/format/index/system/mem_wal.md (840b), references/docs/format/index/vector/index.md (20287b), references/docs/format/table/branch_tag.md (4730b), references/docs/format/table/data_overlay_file.md (17920b), references/docs/format/table/index.md (9772b), references/docs/format/table/layout.md (9479b), references/docs/format/table/mem_wal.md (35391b), references/docs/format/table/row_id_lineage.md (12963b), references/docs/format/table/schema.md (15517b), references/docs/format/table/transaction.md (31316b), references/docs/format/table/versioning.md (2969b), references/docs/guide/arrays.md (6630b), references/docs/guide/blob.md (15695b), references/docs/guide/data_evolution.md (8933b), references/docs/guide/data_types.md (14311b), references/docs/guide/distributed_indexing.md (7190b), references/docs/guide/distributed_write.md (11694b), references/docs/guide/json.md (12339b), references/docs/guide/migration.md (5426b), references/docs/guide/object_store.md (29677b), references/docs/guide/observability.md (3646b), references/docs/guide/performance.md (31441b), references/docs/guide/read_and_write.md (19670b), references/docs/guide/tags_and_branches.md (4492b), references/docs/guide/tokenizer.md (4001b), references/docs/integrations/datafusion.md (4071b), references/docs/quickstart/full-text-search.md (15573b), references/docs/quickstart/index.md (3366b), references/docs/quickstart/vector-search.md (9993b), references/docs/quickstart/versioning.md (3658b), references/format-file.md (35481b), references/format-table.md (63181b), references/indexes.md (54636b), references/lance-reference.md (1137b), references/maintenance.md (4907b), references/ops.md (22178b), references/performance.md (65064b), skill-card.md (2454b), SKILL.md (27095b), _meta.json (132b)\n\nArchive v0.18.1: 62 files, 358637 bytes\n\nFiles: CHANGELOG.md (45501b), LICENSE.txt (9157b), references/changelog-v7-v12.md (68368b), references/docs/format/file/encoding.md (47128b), references/docs/format/file/index.md (9612b), references/docs/format/file/versioning.md (2406b), references/docs/format/index.md (3514b), references/docs/format/index/index.md (15455b), references/docs/format/index/indices-compaction.drawio.svg (51382b), references/docs/format/index/indices-fragment handling.drawio.svg (21857b), references/docs/format/index/scalar_index.drawio.svg (8211b), references/docs/format/index/scalar/bitmap.md (1405b), references/docs/format/index/scalar/bloom_filter.md (5393b), references/docs/format/index/scalar/btree.md (3070b), references/docs/format/index/scalar/fmindex.md (4646b), references/docs/format/index/scalar/fts.md (19092b), references/docs/format/index/scalar/label_list.md (1729b), references/docs/format/index/scalar/ngram.md (1894b), references/docs/format/index/scalar/rtree.md (8350b), references/docs/format/index/scalar/zonemap.md (2986b), references/docs/format/index/starter-example.drawio.svg (14623b), references/docs/format/index/system/frag_reuse.md (2747b), references/docs/format/index/system/mem_wal.md (840b), references/docs/format/index/vector/index.md (20287b), references/docs/format/table/branch_tag.md (4730b), references/docs/format/table/data_overlay_file.md (17920b), references/docs/format/table/index.md (9772b), references/docs/format/table/layout.md (9479b), references/docs/format/table/mem_wal.md (34282b), references/docs/format/table/row_id_lineage.md (12963b), references/docs/format/table/schema.md (15517b), references/docs/format/table/transaction.md (31316b), references/docs/format/table/versioning.md (2969b), references/docs/guide/arrays.md (6630b), references/docs/guide/blob.md (14668b), references/docs/guide/data_evolution.md (8933b), references/docs/guide/data_types.md (14311b), references/docs/guide/distributed_indexing.md (7190b), references/docs/guide/distributed_write.md (11694b), references/docs/guide/json.md (12339b), references/docs/guide/migration.md (5426b), references/docs/guide/object_store.md (29285b), references/docs/guide/observability.md (3646b), references/docs/guide/performance.md (30722b), references/docs/guide/read_and_write.md (19670b), references/docs/guide/tags_and_branches.md (4492b), references/docs/guide/tokenizer.md (4001b), references/docs/integrations/datafusion.md (4071b), references/docs/quickstart/full-text-search.md (15573b), references/docs/quickstart/index.md (3366b), references/docs/quickstart/vector-search.md (9993b), references/docs/quickstart/versioning.md (3658b), references/format-file.md (32927b), references/format-table.md (58271b), references/indexes.md (50554b), references/lance-reference.md (1136b), references/maintenance.md (4214b), references/ops.md (21508b), references/performance.md (59440b), skill-card.md (3015b), SKILL.md (24740b), _meta.json (132b)\n\nArchive v0.18.0: 62 files, 358639 bytes\n\nFiles: CHANGELOG.md (45399b), LICENSE.txt (9157b), references/changelog-v7-v12.md (68368b), references/docs/format/file/encoding.md (47128b), references/docs/format/file/index.md (9612b), references/docs/format/file/versioning.md (2406b), references/docs/format/index.md (3514b), references/docs/format/index/index.md (15455b), references/docs/format/index/indices-compaction.drawio.svg (51382b), references/docs/format/index/indices-fragment handling.drawio.svg (21857b), references/docs/format/index/scalar_index.drawio.svg (8211b), references/docs/format/index/scalar/bitmap.md (1405b), references/docs/format/index/scalar/bloom_filter.md (5393b), references/docs/format/index/scalar/btree.md (3070b), references/docs/format/index/scalar/fmindex.md (4646b), references/docs/format/index/scalar/fts.md (19092b), references/docs/format/index/scalar/label_list.md (1729b), references/docs/format/index/scalar/ngram.md (1894b), references/docs/format/index/scalar/rtree.md (8350b), references/docs/format/index/scalar/zonemap.md (2986b), references/docs/format/index/starter-example.drawio.svg (14623b), references/docs/format/index/system/frag_reuse.md (2747b), references/docs/format/index/system/mem_wal.md (840b), references/docs/format/index/vector/index.md (20287b), references/docs/format/table/branch_tag.md (4730b), references/docs/format/table/data_overlay_file.md (17920b), references/docs/format/table/index.md (9772b), references/docs/format/table/layout.md (9479b), references/docs/format/table/mem_wal.md (34282b), references/docs/format/table/row_id_lineage.md (12963b), references/docs/format/table/sche...","readmeExcerpt":"Skill: lance-format Owner: tenequm Summary: Deep reference for Lance v13 columnar format, its Rust crates, and pylance - file encodings, table format, indexes, schema evolution, time travel. Use when building on the Lance crates or reading .lance datasets, not the LanceDB product. Tags: latest:0.20.0 Version history: v0.20.0 | 2026-09-17T20:33:42.469Z | user Updated lance-format from 0.19.0 to 0.20.0. Changes: - modi","codeSnippets":[],"executableExamples":[{"language":"protobuf","snippet":"%%% proto.message.PageLayout %%%"},{"language":"text","snippet":"[{\"middle\": {\"inner\": 1]}}, NULL, {\"middle\": NULL}, {\"middle\": {\"inner\": NULL}}]"},{"language":"text","snippet":"Outer validity : 1, 0, 1, 1\nMiddle validity: 1, ?, 0, 1\nInner validity : 1, ?, ?, 0\nValues         : 1, ?, ?, ?"},{"language":"text","snippet":"[{<0,1>, <>, <2>}, {<3>}, {}], [], [{<4>}]"},{"language":"text","snippet":"Outer-most ([]): [0, 3, 3, 4]\nMiddle     ({}): [0, 3, 4, 4, 5]\nInner      (<>): [0, 2, 2, 3, 4, 5]\nValues         : [0, 1, 2, 3, 4]"},{"language":"protobuf","snippet":"%%% proto.message.MiniBlockLayout %%%"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: lance-format\ndescription: Deep reference for Lance v13 columnar format, its Rust crates, and pylance - file encodings, table format, indexes, schema evolution, time travel. Use when building on the Lance crates or reading .lance datasets, not the LanceDB product.\nmetadata:\n  version: \"0.20.0\"\n  categories: \"development, integrations\"\n  topics: \"lance, columnar-format, vector-search, rust, lakehouse\"\n  upstream: \"lance-format/lance@v13.0.0-beta.4\"\n  openclaw:\n    homepage: https://github.com/tenequm/skills/tree/main/skills/lance-format\n    emoji: \"🗄️\"\n---\n\n# Lance v13 reference\n\nLance is an open columnar format for multimodal AI - \"a columnar data format that is 100x\nfaster than Parquet for random access.\" It is not one format but a stack of interoperating\nspecs: a **file format**, a **table format**, **index formats**, **catalog specs**, and a\n**namespace client spec**. The Rust workspace at `lance-format/lance` implements all of them\nplus Python (`pylance`) and Java bindings.\n\nThis skill tracks **`v13.0.0-beta.4`** (the `lance-format/lance` git tag), the current\ndevelopment frontier; **`v12.0.0`** is the stable pin, released 2026-09-17. Pin against tags, not\n`main` - Lance ships beta tags every few days and `next`-format encodings can change. Version\nlandscape below.\n\nThree layers of reference, load what the task needs:\n\n- **The deep reference** - any concrete schema, parameter, proto, or constraint. Split by topic:\n\n  | File in `references/` | Covers | Sections |\n  |------|--------|----------|\n  | `format-file.md` | What Lance is, the 26 crates, file format, data types | 1-4 |\n  | `format-table.md` | Dataset layout, manifests, fragments, schema evolution, versioning/tags/branches, row IDs, transactions + OCC, MemWAL | 5-10 |\n  | `indexes.md` | Vector / scalar / FTS / geo indexes, distributed builds | 11-12 |\n  | `ops.md` | Object store, capability matrix, source map | 13, 15, 16 |\n  | `changelog-v7-v13.md` | The full v7 -> v13 delta | 14 |\n\n  Cross-references written as \"section N\" resolve through `references/lance-reference.md`.\n- `references/performance.md` - ALL performance guidance. Part A routes to the official text and\n  adds the source-derived changes upstream has not documented; Part B is field-verified\n  remote-storage practice. Load for any performance, tuning, maintenance-cost, or \"why is this\n  slow\" question.\n- `references/docs/` - a **verbatim mirror of the official docs** (`docs/src` at the tracked\n  tag): every guide, quickstart, and format spec, unedited. Load when you need the full official\n  text. Directory map below.\n\n`references/maintenance.md` covers refreshing this skill against a new upstream tag.\n\n## Lance vs LanceDB\n\nThese are two different things and conflating them produces wrong answers.\n\n- **Lance** - the format and engine. The `lance-format/lance` repo; the `lance` /`lance-*`\n  Rust crates; `pylance`. It gives you datasets, the file/table format, indexes, commits,\n  scans. Consumed directly by DuckDB, P"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn76gpsgjw5chv0xvzbzcb8cxn81x46r\",\n  \"slug\": \"lance-format\",\n  \"version\": \"0.20.0\",\n  \"publishedAt\": 1789677222469\n}"},{"path":"references/changelog-v7-v13.md","content":"# Lance changelog - v7 -> v12 (section 14)\n\nPart of the Lance v13 reference (`lance-format/lance@v13.0.0-beta.4`). Citations are `path:line`\nrelative to the repo root; build a permalink as\n`https://github.com/lance-format/lance/blob/v13.0.0-beta.4/<path>`. Line numbers drift between\ntags - treat them as approximate. Cross-references written as \"section N\" use the original\n16-section numbering; `lance-reference.md` maps every number to its file.\n\n**Release-line shape.** The major is bumped by a bot, not a human: `ci/publish_beta.sh:65,87`\nre-roots at `MAJOR+1` whenever any PR since the release root carries the GitHub\n`breaking-change` label (`ci/check_breaking_changes.py:31`). The marker is the **label**, not a\nconventional-commit `!` - of the 13 labeled PRs in the v11 window (#8024, #8025, #8026, #8027,\n#8028, #8051, #8095, #8159, #8172, #8188, #8206, #8347, #8360) only two carry `!` in the\nsubject. It has now fired on two\nconsecutive lines: `9.1.0-beta.*` -> `10.0.0-beta.*` (2026-07-23), then `10.1.0-beta.*` ->\n`11.0.0-beta.*` (2026-08-05, `649076df1 chore: bump to 11.0.0-beta.1 based on breaking change\ndetection`). So **neither `v9.1.0` nor `v10.1.0` was ever released**, and `v10.1.0-beta.2` is\nthe direct ancestor of `v11.0.0-beta.1`, one bump commit apart. The re-root renumbers in place:\n`release-root/10.1.0-beta.N` and `release-root/11.0.0-beta.N` point at the same commit\n(`ee0a60d0c`), both recording `Base: 10.0.0-rc.1`.\n\n**`v10.0.0` final was tagged on 2026-08-08** - an annotated, PGP-signed tag (\"Release version\n10.0.0\") on `release/v10.0`, one commit past `v10.0.0-rc.3` (2026-08-02). That branch forked at\n`v10.0.0-beta.7` and took one substantive backport (`10d0c9f2e fix: backport encoding and FTS\nfixes to release/v10.0`, #8146). It is **not** an ancestor of `main` - finals are cut on\n`release/vX.Y` branches, so that is normal. `v10.0.0-beta.7` **is** an ancestor of\n`v11.0.0-beta.16`, but `v10.0.0-rc.3` and `v10.0.0` are not.\n\n**`v10.0.0` is the stable pin** (2026-08-08, superseding `v9.0.1`), and it is what GitHub\nReleases marks `Latest`. `v9.0.1` (2026-08-06, superseding `v9.0.0`, 2026-07-24) shipped with\nfive sibling patch finals that day - `v8.0.1`, `v7.1.0`, `v6.1.0`, `v4.0.2`, `v3.0.2` - each on\nits own `release/vX.Y` branch. `v5.0.0` still has no final despite `v5.0.0-rc.2`. crates.io\npublishes **finals only** (`max_stable_version` = `10.0.0`; no 11.x, and the only pre-release\namong ~186 versions is the ancient `0.0.1-alpha0`); PyPI `pylance` is likewise at `10.0.0`. So\nany beta pin is a git dependency; beta artifacts publish to fury.io\n(`.github/workflows/publish-beta.yml:114`) under the renamed org,\n`https://pypi.fury.io/lance-format`.\n\n## Contents\n\n- [The v7.1.0-beta.1 delta](#the-v710-beta1-delta)\n- [The v7.1.0-beta.2 delta](#the-v710-beta2-delta)\n- [The v7.1.0-beta.2 -> v7.2.0-beta.5 delta](#the-v710-beta2---v720-beta5-delta)\n- [The v7.2.0-beta.5 -> v8.0.0-beta.9 delta (major-version boundary)](#the-v720-beta5---v800-beta9-del"},{"path":"references/docs/format/file/encoding.md","content":"# Lance Encoding Strategy\n\nThe encoding strategy determines how array data is encoded into a disk page. The encoding strategy tends to evolve\nmore quickly than the file format itself.\n\n## Older Encoding Strategies\n\nThe 0.1 and 2.0 encoding strategies are no longer documented. They were significantly different from future encoding\nstrategies and describing them in detail would be a distraction.\n\n## Terminology\n\nAn array is a sequence of values. An array has a data type which describes the semantic interpretation of the values.\nA layout is a way to encode an array into a set of buffers and child arrays. A buffer is a contiguous sequence of\nbytes. An encoding describes how the semantic interpretation of data is mapped to the layout. An encoder converts\ndata from one layout to another.\n\nData types and layouts are orthogonal concepts. An integer array might be encoded into two completely different\nlayouts which represent the same data.\n\n![Multiple Encodings](../../images/encoding_v_array.png)\n\n### Data Types\n\nLance uses a subset of Arrow's type system for data types. An Arrow data type is both a data type and an encoding.\nWhen writing data Lance will often normalize Arrow data types. For example, a string array and a large string array\nmight end up traveling down the same path (variable width data). In fact, most types fall into two general paths. One\nfor fixed-width data and one for variable-width data (where we recognize both 32-bit and 64-bit offsets).\n\nAt read time, the Arrow data type is used to determine the target encoding. For example, a string array and large\nstring array might both be stored in the same layout but, at read time, we will use the Arrow data type to determine\nthe size of the offsets returned to the user. There is no requirement the output Arrow type matches the input Arrow\ntype. For example, it is acceptable to write an array as \"large string\" and then read it back as \"string\".\n\n## Search Cache\n\nThe search cache is a key component of the Lance file reader. Random access requires that we locate the physical\nlocation of the data in the file. To do so we need to know information such as the encoding used for a column,\nthe location of the page, and potentially other information. This information is collectively known as the \"search\ncache\" and is implemented as a basic LRU cache. We define a \"initialization phase\" which is when we load the various indexing information into the search cache. The cost of initialization is assumed to be amortized over the lifetime\nof the reader.\n\nWhen performing full scans (i.e. not random access), we should be able to ignore the search cache and sometimes\ncan avoid loading it entirely. We _do_ want to optimize for cold scans as the initialization phase is often not\namortized over the lifetime of the reader.\n\n## Structural Encoding\n\nThe first step in encoding an array is to determine the structural encoding of the array. A structural encoding\nbreaks the data into smaller units which can be independentl"},{"path":"references/docs/format/file/index.md","content":"# Lance File Format\n\nThe Lance file format is a columnar container optimized for cloud object stores, random access, and Arrow-native processing. It deliberately focuses on page layout and encoding mechanics, while leaving table semantics and search structures to higher layers.\n\n## Design Goals\n\n### No Row Groups\n\nLance does not use Parquet-style row groups. Each column may have its own number of pages, which keeps column data in large storage-friendly chunks regardless of schema width and avoids coupling scanner partitioning to physical file layout.\n\n### Random-Access-Friendly Encoding\n\nPages are designed so readers can fetch contiguous row ranges with a small and predictable number of I/O operations. This is important for selective filters, point lookups, vector-search follow-up reads, and ML training workloads that sample rows non-sequentially.\n\n### Functional Decomposition\n\nThe file layer does not bundle table-level statistics or query-side indices into the base file structure. Those capabilities are defined as separate index formats so they can evolve independently of the core file container.\n\n## File Structure\n\nA Lance file is a container for tabular data. The data is stored in \"disk pages\". Each disk page contains some rows\nfor a single column. There may be one or more disk pages per column. Different columns may have different numbers of\ndisk pages. Metadata at the end of the file describes where the pages are located and how the data is encoded.\n\n![Format Overview](../../images/file_high_level_overview.png)\n\n!!! Note\n\n    This page describes the container specification. We also have a set of default encodings that are used to encode\n    data into disk pages. See the [Encoding Strategy](encoding.md) page for more details.\n\n### Disk Pages\n\nDisk pages are designed to be large enough to justify a dedicated I/O operation, even on cloud storage, typically several megabytes. Using a larger page size may reduce the number of I/O operations required to read a file, but it also increases the amount of memory required to write the file. In practice, very large page sizes are not useful when high speed reads are required because large contiguous reads need to be broken into smaller reads for performance (particularly on cloud storage). As a result, a default of 8MB is recommended for the page size and should yield ideal performance on all storage systems.\n\nDisk pages should not generally be opaque. It is possible to read a portion of a disk page when a subset of the rows are\nrequired. However, the specifics of this process depend on the column encoding which is described in a later section.\n\n### No Row Groups\n\nUnlike similar formats, there is no \"row group\" concept, only pages. We believe the concept of row groups to be\nfundamentally harmful to performance. If the row group size is too small then columns will be split into \"runt pages\" which yield poor read performance on cloud storage. If the row group size is too large then a file writer will need"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":2491,"uniquenessScore":37,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T19:57:48.230Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T19:57:48.230Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T22:08:15.761Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}