{
  "schema": "kbatch-data-lake-manifest-v1",
  "generated": "2026-08-13T06:51:39.762Z",
  "status": "growing",
  "claim": "Open data lake under ugrad.kbatch — μ-speed coordinate field for source (every line + empty cell); word descriptors stay on the spelling plane; inclusive 2Spirit·LGBTQI+; never closed encyclopedias or pirate text. Not HF/blob/shard/sliver/sand.",
  "rights": {
    "allow": [
      "Project Gutenberg PD",
      "Wikidata CC0",
      "Open Library metadata",
      "IA PD",
      "our geometry metrics"
    ],
    "forbid": [
      "Turner/OUP",
      "sacred-texts scrape",
      "commercial lyrics",
      "closed baby-name DBs without license"
    ]
  },
  "corpora": [
    {
      "id": "books-living",
      "path": "data/living-books/",
      "rowsApprox": 286,
      "target": "all Gutenberg stubs + open titles",
      "status": "partial"
    },
    {
      "id": "books-gutenberg-catalog",
      "path": "data/living-books/gutenberg-stubs/",
      "rowsApprox": 67688,
      "target": "~70k ebook stubs",
      "status": "live"
    },
    {
      "id": "deities",
      "path": "data/mythology/deities-index.json",
      "rowsApprox": 5958,
      "target": "world deities + alsoKnownAs + seeAlso graph",
      "status": "partial-enriched"
    },
    {
      "id": "names-human",
      "path": "data/open-names/",
      "rowsApprox": 17470,
      "target": "world given/family/history",
      "status": "growing"
    },
    {
      "id": "doctrine-sources",
      "path": "data/lake/doctrine-sources.json",
      "rowsApprox": 200,
      "target": "sacred/wisdom/open source registry beyond sacred-texts map",
      "status": "seeded-pd-meta"
    },
    {
      "id": "viewer-layouts",
      "path": "data/living-books/viewer-layouts.json",
      "status": "live-schema"
    },
    {
      "id": "world-lang-packs",
      "path": "data/words/lang-index.json",
      "rowsApprox": 54592004,
      "target": "all open frequency packs + honor seeds",
      "status": "live",
      "note": "sum of pack totals not deduped"
    },
    {
      "id": "lang-tree",
      "path": "data/lake/lang-tree.json",
      "status": "live",
      "target": "family/script tree of World packs"
    },
    {
      "id": "lang-cross-analysis",
      "path": "data/lake/derived/lang-cross-analysis.json",
      "status": "live"
    },
    {
      "id": "qbit-codec",
      "path": "data/lake/qbit-codec/index.json",
      "status": "live",
      "target": "Quantum gutter·Prefixes·GlueLam·DAC·IronLine·StenoStrip·Whitespace Glyphs"
    },
    {
      "id": "mu-capsule",
      "path": "data/lake/qbit-codec/capsules/index.json",
      "schema": "data/lake/qbit-codec/mu-capsule.schema.json",
      "status": "live",
      "target": "μ-capsules — one concept × all langs × all planes · μ-scope Stair/Genetic/Planes",
      "scope": "/labs/mu-scope.html"
    },
    {
      "id": "lake-route",
      "path": "data/lake/derived/lake-route.json",
      "status": "live",
      "target": "Lake × zen join — sort / rearrange / narrow / expand on the fly",
      "zen": "https://lang.ugrad.ai/zen.html"
    },
    {
      "id": "books-research-join",
      "path": "data/lake/derived/books-research-join.json",
      "index": "data/lake/derived/books-train-index.jsonl",
      "status": "live",
      "langs": 126,
      "target": "126-seat lattice × living-books × GrokBots/books — RAG key is book id, not spelling atlas",
      "desk": "/labs/books-research",
      "grokbots": "/Volumes/qbitOS/GrokBots/books/kbatch-join.md"
    },
    {
      "id": "mu-coord",
      "path": "data/lake/qbit-codec/mu-coord-index.json",
      "schema": "data/lake/qbit-codec/mu-coord.schema.json",
      "doctrine": "data/lake/qbit-codec/MU-SPEED.md",
      "sample": "data/lake/qbit-codec/mu-coord-sample.jsonl",
      "status": "live",
      "target": "μ-speed JSONL — every line and empty cell packed into a 64-bit μ address",
      "not": [
        "hf",
        "blob",
        "shard",
        "sliver",
        "sand"
      ]
    },
    {
      "id": "inclusive-language",
      "path": "data/lake/inclusive-language.json",
      "status": "live",
      "target": "2Spirit·LGBTQI+ doctrine for etymology/names/ancestory"
    },
    {
      "id": "ancestory-lineage",
      "path": "data/ancestory/",
      "rowsApprox": 6587,
      "status": "live"
    },
    {
      "id": "word-descriptor-lake",
      "path": "data/lake/word-descriptor-coverage.json",
      "schema": "data/lake/word-descriptor-schema-v1.json",
      "rowsApprox": 55400568,
      "langs": 126,
      "status": "ledger-live",
      "target": "every word × every descriptor in all seated langs",
      "note": "54 full orthography packs · 126 with geometry · 113467 glosses in 1 lang(s) · 1 langs with form+geometry+senses"
    }
  ],
  "aiPipeline": {
    "role": "dedupe · entity link · lang-tree walk · qbit gutter classify · μ-coord encode · inclusive gloss",
    "notRole": "invent licensed prose · OCR closed books · out living people · re-shard source as HF blobs",
    "output": "data/lake/derived/",
    "loadOrder": [
      "data/llm/train-pack.json",
      "data/lake/qbit-codec/index.json",
      "data/lake/qbit-codec/mu-coord-index.json",
      "data/lake/derived/lake-route.json",
      "data/lake/derived/books-research-join.json",
      "data/lake/layer-coverage.json",
      "data/lake/derived/language-inventory.json",
      "data/lake/inclusive-language.json",
      "data/lake/lang-tree.json",
      "data/words/lang-index.json",
      "data/lake/derived/lang-cross-analysis.json",
      "data/catalog/index.json",
      "data/ancestory/index.json",
      "data/lake/manifest.json"
    ]
  },
  "next": [
    "grow-multilang fill placeholders (14)",
    "promote thin ready packs",
    "analyze-lang-slivers stamps spelling-plane geometry only — source is μ-coord",
    "npm run lake:mu after codec edits (seek μ / line:col, never shard-walk)",
    "Agent continuous calibrate_check in long sessions"
  ],
  "muCoord": "data/lake/qbit-codec/mu-coord-index.json",
  "llm": "data/llm/train-pack.json",
  "freya": "https://freya.qbitos.ai/",
  "recalibrate": "kbatch_recalibrate",
  "qbitCodec": "data/lake/qbit-codec/index.json",
  "inclusive": "data/lake/inclusive-language.json",
  "langTree": "data/lake/lang-tree.json",
  "wordLake": "data/lake/word-descriptor-coverage.json"
}
