From e0bf45d716e40f1a29eb83e8802bfb7acf3128e2 Mon Sep 17 00:00:00 2001 From: Nolan Nichols Date: Fri, 10 Jul 2026 15:10:52 -0700 Subject: [PATCH] =?UTF-8?q?Federation=20Phase=200:=20export=20the=20regist?= =?UTF-8?q?ry=20producer=20contract=20+=20SPEC=20=C2=A711?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Groundwork for meta-lokf, a registry that aggregates independent lokf bundles into one cross-repo graph without a central database. `lokf export` now additively writes two artifacts a registry harvests: - graph.nt — the whole bundle as N-Triples (the SPARQL harvest source) - concepts.jsonld — every concept's frontmatter + body under one @context/@graph (offline access to a concept's source document) alongside the existing graph.json + datasets.jsonld. The two are isomorphic (86 triples on the acme example); concepts.jsonld additionally carries the raw frontmatter+body so a consumer can read a foreign concept without refetching markdown. Both new outputs are gitignored under web/public/. SPEC gains §11 "Federation: registries of bundles" (Versioning → §12): the producer contract, the lokf-registry.yaml DCAT/VoID manifest, owner(iri) longest-base_iri-prefix resolution as the inverse of §7 IRI minting plus its three guards, the forthcoming lazy named-graph traversal + agent surface, and governance (public/local-only membership, metadata-not-rows, best-effort freshness). Adds the void: prefix to Appendix B. 135 tests pass; docs site builds and renders §11. --- SPEC.md | 118 +++++++++++++++++++++++++++++++++++++++++++++- src/lokf/cli.py | 32 ++++++++++--- tests/test_cli.py | 25 ++++++++++ web/.gitignore | 2 + 4 files changed, 169 insertions(+), 8 deletions(-) diff --git a/SPEC.md b/SPEC.md index 0d1e30f..19f6db1 100644 --- a/SPEC.md +++ b/SPEC.md @@ -388,7 +388,122 @@ graph using the *same* vocabularies the public web already speaks. --- -## 11. Versioning +## 11. Federation: registries of bundles + +A single LOKF bundle is self-contained, but knowledge rarely is: one team's +`Metric` `dependsOn` another team's `GlossaryTerm`, whose canonical definition +lives in a *different* bundle in a *different* repository. A **registry** +(*meta-lokf*) aggregates a set of independent bundles into one navigable graph +**without a central database**, so an agent can follow a relation out of one +bundle into another and read the target concept's source document. A registry is +itself just linked data — a DCAT catalog of catalogs — so it is built with the +same vocabularies and tooling as the bundles it indexes. + +### 11.1 The producer contract + +A bundle becomes federatable by publishing the artifacts `lokf export` writes to +a location a registry can read (GitHub Pages, a w3id-fronted host, or a sibling +checkout). Its **`base_iri`** (from the root `index.md`, §4) is its identity in +the registry. + +| Artifact | Role in federation | +|--------------------|------------------------------------------------------------------------| +| `graph.nt` | The whole bundle as N-Triples — the RDF a registry loads for SPARQL. | +| `concepts.jsonld` | Every concept's frontmatter + body under one `@context`/`@graph` — offline access to a concept's *source document* without re-fetching markdown. | +| `graph.json` | cytoscape.js elements — drives the (multi-bundle) graph explorer. | +| `datasets.jsonld` | schema.org `Dataset` docs — discovery via Google Dataset Search. | + +### 11.2 The registry manifest + +A registry is a git-committed `lokf-registry.yaml` — a `dcat:Catalog` of +`void:Dataset` entries, one per member bundle, keyed by `base_iri` +(== `void:uriSpace`). Each entry records where the member's artifacts live, a +lightweight `void` planning index (triple and per-type counts, so an agent can +pick a bundle *without* dereferencing it), and harvest `status`. It parses with a +plain YAML reader — no LinkML on the read path — and round-trips to a crawlable +`registry.jsonld`. + +```yaml +lokf_registry_version: "0.1" +type: dcat:Catalog +id: https://w3id.org/lokf/registry/example +repos: + - base_iri: https://acme.example/knowledge/ # routing key == void:uriSpace + title: Acme Knowledge Bundle + repo: git+https://github.com/acme/knowledge.git@main + source_base: https://raw.githubusercontent.com/acme/knowledge/main + distribution: + rdf: https://acme.example/knowledge/graph.nt # SPARQL harvest source + concepts: https://acme.example/knowledge/concepts.jsonld # offline document access + void: { triples: 86, class_partition: { Metric: 1, Dataset: 1, GlossaryTerm: 1 } } + id_index: [ ] # explicit `id:` IRIs outside base_iri + status: ok +``` + +### 11.3 Cross-bundle resolution + +The load-bearing operation is **`owner(iri)`**: the registered `base_iri` that is +the **longest string prefix** of the IRI. This is the exact inverse of IRI +minting (§7): a concept's IRI is `base_iri + concept_id`, so given any IRI, +`concept_id = iri[len(base_iri):]` and the owning bundle is its longest-prefix +match. Resolution is therefore **pure string arithmetic — no network, no shared +database** — and a cross-bundle relation whose author wrote a full `https://…` +target is *already* a correct triple. Three rules keep it trustworthy: + +1. **Explicit-id index.** A concept's frontmatter `id:` may diverge from + `base_iri + concept_id` (§5). Such IRIs are harvested into the entry's + `id_index` and checked as an exact-match fallback before an IRI is declared + external, so they still route. +2. **Namespace precedence & non-nesting.** The packaged vocabulary namespace + (`https://w3id.org/lokf/`) always resolves to the built-in schema; registered + `base_iri`s must be strictly longer and may not nest inside one another, so + routing is unambiguous. +3. **Ownership validation.** At registration a member's sampled concept IRIs must + actually start with its declared `base_iri`, so a bundle cannot claim a + namespace it does not own. + +An IRI owned by no entry is returned as a tolerated **dangling link**, not an +error — preserving OKF's permissive stance on broken cross-references. + +### 11.4 Traversal and access *(informative — delivered in phases)* + +The intended runtime model, layered on the primitives above: + +- **Federated graph.** A registry loads each member's `graph.nt` into a + **named graph whose IRI is its `base_iri`**. Union queries make a cross-bundle + edge resolve transparently once both members are loaded, while `GRAPH ?g` + recovers *which* bundle asserted a triple for free (`?g` binds the `base_iri`). + Members load **lazily** — only as a walk reaches into their namespace — so an + agent never pays to materialize the whole federation to answer a local + question. +- **Document access.** A concept's source markdown is fetched by + `source_base + concept_id + ".md"` through an offline-first chain: a local + checkout, else the cached `concepts.jsonld`, else a live fetch — every path + yielding the same `{frontmatter, body}` shape. +- **Agent surface.** A `lokf registry` CLI and a `lokf-registry` MCP server + expose `resolve_iri`, `neighbors`, `subgraph`, `federated_sparql`, and + `read_document`, each depth/breadth-bounded so a cross-bundle hop is a single + token-frugal call rather than a chatty chain. + +### 11.5 Governance + +- **Membership is public-artifact or local-checkout only** in v0.1; credentials + never enter the shared manifest. Federating a private or perimeter-bound bundle + is a future extension, not a v0.1 capability. +- **Metadata only, never row-level data.** A registry aggregates schema- and + concept-level knowledge; it does not move records. A per-entry `sensitivity` + gate lets `harvest` refuse un-cleared members, because metadata each cleared + *individually* can be *jointly* re-identifying and even `void` counts can leak + small cells — so aggregating sensitive-domain bundles requires explicit + clearance from the registry's owner. +- **Freshness is best-effort.** The federated store reflects the last harvest, + guarded by conditional requests and a surfaced per-entry `status`; there is no + live-HEAD guarantee, and an unreachable member degrades to its last-good state + rather than failing a walk. + +--- + +## 12. Versioning LOKF versions are `.`, tracking OKF's scheme. A minor bump adds backward-compatible fields, types, relation predicates, or mappings; a major bump @@ -419,6 +534,7 @@ README.md How the pieces fit and how to regenerate them. | `lokf` | `https://w3id.org/lokf/` | | `schema` | `http://schema.org/` | | `dcat` | `http://www.w3.org/ns/dcat#` | +| `void` | `http://rdfs.org/ns/void#` | | `dcterms` | `http://purl.org/dc/terms/` | | `prov` | `http://www.w3.org/ns/prov#` | | `skos` | `http://www.w3.org/2004/02/skos/core#` | diff --git a/src/lokf/cli.py b/src/lokf/cli.py index 260acab..16206fa 100644 --- a/src/lokf/cli.py +++ b/src/lokf/cli.py @@ -337,20 +337,32 @@ def export( ..., exists=True, file_okay=False, help="A LOKF bundle directory." ), out_dir: Path = typer.Option( - ..., "--out-dir", "-d", help="Directory to write graph.json + datasets.jsonld." + ..., + "--out-dir", + "-d", + help="Directory to write graph.json, datasets.jsonld, graph.nt, concepts.jsonld.", ), source_base: Optional[str] = typer.Option( None, "--source-base", help="URL prefix for a node's source file (graph meta)." ), ) -> None: - """Export a bundle's graph + Dataset JSON-LD for a static site to consume. - - Writes ``graph.json`` (cytoscape.js elements, typed-relation edges, plus - ``meta.source_base``) and ``datasets.jsonld`` (schema.org Dataset docs for - Google Dataset Search). This is the data step behind the docs site. + """Export a bundle's artifacts for a static site — and a registry — to consume. + + Writes four files (the *producer contract* a meta-lokf registry harvests): + + - ``graph.json`` — cytoscape.js elements (typed-relation edges) plus + ``meta.source_base``; drives the docs-site graph explorer. + - ``datasets.jsonld`` — schema.org Dataset docs for Google Dataset Search. + - ``graph.nt`` — the whole bundle as N-Triples; the RDF a registry loads + into its federated store for cross-bundle SPARQL. + - ``concepts.jsonld`` — every concept's frontmatter + body under one + ``@context``/``@graph``; lets a registry read a foreign concept's source + document offline, without re-fetching the markdown. """ + from lokf import rdf from lokf.export import dataset_search_jsonld, to_cytoscape from lokf.model import load_bundle + from lokf.schema import load_context bundle = load_bundle(bundle_dir) graph = to_cytoscape(bundle) @@ -360,9 +372,15 @@ def export( (out_dir / "datasets.jsonld").write_text( json.dumps(dataset_search_jsonld(bundle), indent=2), encoding="utf-8" ) + (out_dir / "graph.nt").write_text(rdf.serialize(bundle_dir, "nt"), encoding="utf-8") + (out_dir / "concepts.jsonld").write_text( + json.dumps({"@context": load_context(), "@graph": bundle.docs()}, indent=2), + encoding="utf-8", + ) typer.echo( f"wrote {out_dir}/graph.json ({len(graph['nodes'])} nodes, " - f"{len(graph['edges'])} edges) and {out_dir}/datasets.jsonld" + f"{len(graph['edges'])} edges), datasets.jsonld, graph.nt, " + f"and concepts.jsonld ({len(bundle.concepts)} concepts)" ) diff --git a/tests/test_cli.py b/tests/test_cli.py index 47907af..8905c93 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -127,3 +127,28 @@ def test_export_writes_graph_and_datasets(tmp_path): assert graph["meta"]["source_base"] == "https://x/" datasets = json.loads((tmp_path / "datasets.jsonld").read_text()) assert len(datasets) == 2 and all(d["@type"] == "Dataset" for d in datasets) + + +def test_export_writes_registry_producer_contract(tmp_path): + """export also writes graph.nt + concepts.jsonld (the registry harvest source).""" + result = runner.invoke(app, ["export", str(BUNDLE), "--out-dir", str(tmp_path)]) + assert result.exit_code == 0 + + # graph.nt is the whole bundle as N-Triples (one statement per line). + nt = (tmp_path / "graph.nt").read_text() + assert nt.strip() and all( + line.endswith(" .") for line in nt.strip().splitlines() + ) + + # concepts.jsonld is one @context/@graph document carrying frontmatter + body, + # each concept keyed by its IRI, and it round-trips to the same triples as nt. + doc = json.loads((tmp_path / "concepts.jsonld").read_text()) + assert set(doc) == {"@context", "@graph"} + assert len(doc["@graph"]) == 6 + assert all(c.get("id", "").startswith("http") and "body" in c for c in doc["@graph"]) + + from rdflib import Graph + + g_nt = Graph().parse(str(tmp_path / "graph.nt"), format="nt") + g_jsonld = Graph().parse(str(tmp_path / "concepts.jsonld"), format="json-ld") + assert len(g_nt) and g_nt.isomorphic(g_jsonld) diff --git a/web/.gitignore b/web/.gitignore index 6657034..2e44f5b 100644 --- a/web/.gitignore +++ b/web/.gitignore @@ -23,6 +23,8 @@ pnpm-debug.log* # generated by `npm run sync` (lokf export) public/graph.json public/datasets.jsonld +public/graph.nt +public/concepts.jsonld src/generated/ src/content/docs/specification.md src/content/docs/reference/api.md