diff --git a/CHANGELOG.md b/CHANGELOG.md index 57e42941..90178209 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -22,7 +22,7 @@ and this project adheres to ### Added -- Native-first provenance backfills (Phase 2): +- Native-first provenance backfills (native SPDX backfill): - `hasConcludedLicense` vs. `hasDeclaredLicense` separation for detected vs. author-asserted licenses. - Native `ExternalIdentifier` (DOI) and `ExternalRef` diff --git a/src/pitloom/assemble/spdx3/ai.py b/src/pitloom/assemble/spdx3/ai.py index 26b2cbe1..5575ee12 100644 --- a/src/pitloom/assemble/spdx3/ai.py +++ b/src/pitloom/assemble/spdx3/ai.py @@ -203,7 +203,7 @@ def _add_base_model_lineage( ai_model: AiModelMetadata, ctx: _LineageContext, ) -> None: - """Add native Relationship (descendantOf) linking ai_pkg to its base model (N5).""" + """Add native descendantOf Relationship linking ai_pkg to its base model.""" base_model_id = ai_model.base_model or ai_model.extra_data.get("hf.base_model") if not base_model_id: return diff --git a/src/pitloom/assemble/spdx3/fragments.py b/src/pitloom/assemble/spdx3/fragments.py index e3c08f04..0d295d2c 100644 --- a/src/pitloom/assemble/spdx3/fragments.py +++ b/src/pitloom/assemble/spdx3/fragments.py @@ -50,7 +50,7 @@ #: genuinely *distinct* id into the survivor ("these two different ids are the #: same bytes"), which SPDX cannot express and which the merge would otherwise #: discard. A same-id registry match (dropped id == survivor id) carries no such -#: fact -- its only content is the fragment origin, which belongs to the Phase-2 +#: fact -- its only content is the fragment origin, which belongs to the native #: ``SpdxDocument.imports`` native anchor, not here. Agent/Tool structural dedup #: (every fragment mints its own "Pitloom") is likewise excluded. _UnificationEvents = dict[str, dict[str, dict[str, set[str]]]] @@ -475,7 +475,7 @@ def _merge_fragment_set( if by_id is not None: # Same-id (registry) match: the fragment reused an existing id, so # nothing distinct is folded. Its only fact is the fragment origin, - # which is Phase-2 SpdxDocument.imports territory -- not annotated + # which is SpdxDocument.imports territory -- not annotated # here (see _UnificationEvents). Properties are still merged. _merge_properties(_as_element(by_id), _as_element(obj)) remap[obj] = require_spdx_id(_as_element(by_id)) @@ -598,7 +598,7 @@ def _add_fragment_imports( fragment_imports: list[spdx3.ExternalMap], ) -> None: """Populate ``main_doc.import_`` with ``ExternalMap`` entries for merged - fragment documents (N1).""" + fragment documents.""" if not fragment_imports: return existing_imports = list(main_doc.import_ or []) diff --git a/src/pitloom/assemble/spdx3/provenance.py b/src/pitloom/assemble/spdx3/provenance.py index 62f554d5..e7d9cbc5 100644 --- a/src/pitloom/assemble/spdx3/provenance.py +++ b/src/pitloom/assemble/spdx3/provenance.py @@ -304,7 +304,7 @@ def build_unification_annotation( the survivor *subject_spdx_id* (A1). SPDX has no native home for this: the merged-away ids vanish from the graph - and ``SpdxDocument.imports`` (a Phase-2 native anchor, see the design doc) + and ``SpdxDocument.imports`` (a native anchor, see the design doc) can only say an element came from a fragment, not the matching *criterion*. Args: diff --git a/tests/test_fragments.py b/tests/test_fragments.py index 61a870a0..a06e0c78 100644 --- a/tests/test_fragments.py +++ b/tests/test_fragments.py @@ -990,7 +990,8 @@ def test_unification_annotation_records_sha256_merge() -> None: def test_merge_fragments_populates_spdx_document_imports(tmp_path: Path) -> None: """merge_fragments() must populate main_doc.import_ with ExternalMap - entries naming each merged fragment's SpdxDocument spdxId and location hint (N1). + entries naming each merged fragment's SpdxDocument spdxId and location hint + (fragment origin). """ frag_namespace = "https://spdx.org/spdxdocs/fragment-import-test" envelope_fragment = { diff --git a/tests/test_generator.py b/tests/test_generator.py index cefd4994..402a720a 100644 --- a/tests/test_generator.py +++ b/tests/test_generator.py @@ -1388,7 +1388,7 @@ def test_build_model_without_license() -> None: def test_build_model_external_identifiers() -> None: """build_model() must emit ExternalIdentifier for DOI and ExternalRef for - arXiv paper IDs and model page URLs (N4). + arXiv paper IDs and model page URLs (external identifiers). """ model = AiModelMetadata( format_info=AiModelFormatInfo(model_format=AiModelFormat.SAFETENSORS), @@ -1440,7 +1440,8 @@ def test_build_model_external_identifiers() -> None: def test_build_model_with_dataset_creator() -> None: """build_model() for a model linked to a dataset with creator metadata - must emit an Agent element and a publishedBy Relationship (N6). + must emit an Agent element and a publishedBy Relationship + (dataset creator attribution). """ ds_meta = DatasetMetadata(name="squad", creator="Stanford NLP") ds_ref = DatasetReference(role="trainedOn", metadata=ds_meta) @@ -1619,7 +1620,8 @@ def test_generate_deployed_sbom_mocked_pipdeptree( def test_build_model_base_model_lineage() -> None: """build_model() must emit a stub base model ai_AIPackage and a descendantOf - Relationship when base_model and base_model_relation are present (N5). + Relationship when base_model and base_model_relation are present + (base-model lineage). """ meta = AiModelMetadata( name="my-finetuned-model", diff --git a/tests/test_native_spdx_integration.py b/tests/test_native_spdx_integration.py new file mode 100644 index 00000000..b1eb614f --- /dev/null +++ b/tests/test_native_spdx_integration.py @@ -0,0 +1,322 @@ +# SPDX-FileContributor: Arthit Suriyawongkul +# SPDX-FileCopyrightText: 2026-present Arthit Suriyawongkul +# SPDX-FileType: SOURCE +# SPDX-License-Identifier: Apache-2.0 + +"""End-to-end integration test for native SPDX 3 constructs. + +Exercises fragment origin / ExternalMap, declared vs. concluded license, +ExternalIdentifier & ExternalRef, descendantOf base-model lineage, and dataset +creator Agent + publishedBy together on a single representative input. +Verifies that: + +- all five native constructs appear correctly on the same document; +- provenance Annotations are trimmed to residual signal and do not duplicate + values now covered natively; +- output is byte-identical across two generation runs with identical inputs; +- the combined document can be round-tripped through the repo's SPDX 3 model. +""" + +from __future__ import annotations + +import io +import json +from pathlib import Path +from typing import Any + +from spdx_python_model.bindings import v3_0_1 as spdx3 + +from pitloom.assemble.spdx3.document import build_model +from pitloom.core.ai_metadata import AiModelFormat, AiModelFormatInfo, AiModelMetadata +from pitloom.core.creation import CreationMetadata +from pitloom.core.dataset_metadata import DatasetMetadata, DatasetReference + + +def _build_integrated_model() -> AiModelMetadata: + """Return metadata exercising native license, identifier, lineage, and + creator data. + """ + ds_meta = DatasetMetadata( + name="ai-training-data", + creator="Example Research Group", + dataset_types=["text"], + dataset_size=100000, + ) + return AiModelMetadata( + format_info=AiModelFormatInfo(model_format=AiModelFormat.SAFETENSORS), + name="native-spdx-integrated-model", + version="1.0.0", + license="mit", + # External identifiers + doi="10.1234/example.doi", + arxiv_ids=["2301.12345"], + url="https://huggingface.co/example/native-spdx-integrated-model", + # Base-model lineage + base_model="openai-community/gpt2", + base_model_relation="finetune", + # Dataset creator attribution + datasets=[DatasetReference(role="trainedOn", metadata=ds_meta)], + # Make the license concluded by using a non-transparent source/method. + provenance={ + "license": ( + "Source: model card scan | Field: cardData.license | " + "Method: spdx-license-detector" + ) + }, + ) + + +def _load_graph(model: AiModelMetadata) -> tuple[list[dict[str, Any]], str]: + """Build a standalone SBOM for *model* and return its @graph plus raw JSON.""" + exporter = build_model(model, CreationMetadata()) + raw_json = exporter.to_json(pretty=True) + graph = json.loads(raw_json)["@graph"] + return graph, raw_json + + +def test_all_native_construct_types_present() -> None: + """All native constructs appear on the same document.""" + graph, _ = _load_graph(_build_integrated_model()) + types = {e.get("type") for e in graph} + + assert "ai_AIPackage" in types + assert "dataset_DatasetPackage" in types + assert "Agent" in types + assert "Relationship" in types + assert "simplelicensing_SimpleLicensingText" in types + + +def test_declared_vs_concluded_license_present() -> None: + """A license produces a hasConcludedLicense relationship because the + provenance carries a detection method.""" + graph, _ = _load_graph(_build_integrated_model()) + ai_pkgs = [e for e in graph if e.get("type") == "ai_AIPackage"] + assert len(ai_pkgs) == 2, "expected derived + base model packages" + derived_pkg = next( + p for p in ai_pkgs if p["name"] == "native-spdx-integrated-model" + ) + + rels = [e for e in graph if e.get("type") == "Relationship"] + concluded = [ + r + for r in rels + if r.get("relationshipType") == "hasConcludedLicense" + and r.get("from") == derived_pkg["spdxId"] + ] + assert len(concluded) == 1, "expected one hasConcludedLicense relationship" + + license_id = concluded[0]["to"][0] + license_elems = [ + e + for e in graph + if e.get("type") == "simplelicensing_SimpleLicensingText" + and e.get("spdxId") == license_id + ] + assert len(license_elems) == 1 + assert license_elems[0]["simplelicensing_licenseText"] == "mit" + + spdx_doc = next(e for e in graph if e.get("type") == "SpdxDocument") + assert "simpleLicensing" in spdx_doc["profileConformance"] + + +def test_external_identifiers_and_refs_present() -> None: + """DOI ExternalIdentifier and arXiv/URL ExternalRefs are present.""" + graph, _ = _load_graph(_build_integrated_model()) + derived_pkg = next( + e + for e in graph + if e.get("type") == "ai_AIPackage" + and e.get("name") == "native-spdx-integrated-model" + ) + + ext_ids = derived_pkg.get("externalIdentifier", []) + doi_ids = [ + i + for i in ext_ids + if i.get("externalIdentifierType") == "other" + and i.get("identifier") == "10.1234/example.doi" + and i.get("comment") == "DOI" + ] + assert len(doi_ids) == 1, "expected one DOI ExternalIdentifier" + + ext_refs = derived_pkg.get("externalRef", []) + arxiv_refs = [ + r + for r in ext_refs + if r.get("externalRefType") == "documentation" + and "https://arxiv.org/abs/2301.12345" in r.get("locator", []) + and r.get("comment") == "arXiv:2301.12345" + ] + assert len(arxiv_refs) == 1, "expected one arXiv ExternalRef" + + url_refs = [ + r + for r in ext_refs + if r.get("externalRefType") == "altWebPage" + and "https://huggingface.co/example/native-spdx-integrated-model" + in r.get("locator", []) + and r.get("comment") == "Model page URL" + ] + assert len(url_refs) == 1, "expected one model-page ExternalRef" + + +def test_base_model_lineage_present() -> None: + """A base-model stub and descendantOf relationship are present.""" + graph, _ = _load_graph(_build_integrated_model()) + ai_pkgs = [e for e in graph if e.get("type") == "ai_AIPackage"] + assert len(ai_pkgs) == 2 + + derived_pkg = next( + p for p in ai_pkgs if p["name"] == "native-spdx-integrated-model" + ) + base_pkg = next(p for p in ai_pkgs if p["name"] == "gpt2") + + base_refs = [ + r + for r in base_pkg.get("externalRef", []) + if r.get("externalRefType") == "altWebPage" + and "https://huggingface.co/openai-community/gpt2" in r.get("locator", []) + ] + assert len(base_refs) == 1, "expected base model altWebPage ExternalRef" + + rels = [e for e in graph if e.get("type") == "Relationship"] + lineage_rels = [ + r + for r in rels + if r.get("relationshipType") == "descendantOf" + and r.get("from") == derived_pkg["spdxId"] + and base_pkg["spdxId"] in r.get("to", []) + and r.get("comment") == "base_model_relation:finetune" + ] + assert len(lineage_rels) == 1, "expected one descendantOf relationship" + + +def test_dataset_creator_agent_and_published_by_present() -> None: + """A dataset creator Agent and publishedBy relationship are present.""" + graph, _ = _load_graph(_build_integrated_model()) + + agents = [e for e in graph if e.get("type") == "Agent"] + creator_agents = [a for a in agents if a.get("name") == "Example Research Group"] + assert len(creator_agents) == 1, "expected one creator Agent" + agent_id = creator_agents[0]["spdxId"] + + ds_pkgs = [e for e in graph if e.get("type") == "dataset_DatasetPackage"] + assert len(ds_pkgs) == 1 + ds_pkg_id = ds_pkgs[0]["spdxId"] + + rels = [e for e in graph if e.get("type") == "Relationship"] + published_by = [ + r + for r in rels + if r.get("relationshipType") == "publishedBy" + and r.get("from") == ds_pkg_id + and agent_id in r.get("to", []) + ] + assert len(published_by) == 1, "expected one publishedBy relationship" + + +def test_fragment_origin_round_trips_when_merged() -> None: + """A fragment-origin ExternalMap can be added and round-tripped. + + The standalone build_model path does not perform fragment merging, so we + add an ExternalMap directly to the SpdxDocument and verify the document + round-trips through spdx-python-model without data loss. + """ + exporter = build_model(_build_integrated_model(), CreationMetadata()) + graph = json.loads(exporter.to_json(pretty=True))["@graph"] + spdx_doc = next(e for e in graph if e.get("type") == "SpdxDocument") + + fragment_ns = "https://spdx.org/spdxdocs/example-fragment" + spdx_doc.setdefault("import", []).append( + { + "type": "ExternalMap", + "externalSpdxId": fragment_ns, + "locationHint": "fragments/example.spdx3.json", + } + ) + + # Round-trip through the model to confirm validity. + object_set = spdx3.SHACLObjectSet() + spdx3.JSONLDDeserializer().read( + io.BytesIO( + json.dumps({"@context": spdx_doc.get("@context"), "@graph": graph}).encode( + "utf-8" + ) + ), + object_set, + ) + docs = [o for o in object_set.objects if isinstance(o, spdx3.SpdxDocument)] + assert len(docs) == 1 + imports = list(docs[0].import_ or []) + assert any(getattr(item, "externalSpdxId", None) == fragment_ns for item in imports) + + +def test_provenance_annotations_do_not_duplicate_native_values() -> None: + """Residual provenance Annotations must not re-state values now native.""" + graph, _ = _load_graph(_build_integrated_model()) + annotations = [e for e in graph if e.get("type") == "Annotation"] + + for ann in annotations: + statement = json.loads(ann["statement"]) + # Fields covered by native SPDX 3 constructs should not appear in + # provenance statements anymore. + fields = statement.get("fields", {}) + assert "doi" not in fields, "DOI should be native, not in an Annotation" + assert "arxiv" not in fields, "arXiv should be native" + assert "url" not in fields, "model URL should be native" + assert "creator" not in fields, "dataset creator should be native" + + # The license detection *method* is residual signal, but the license value + # itself lives in the SimpleLicensingText element. + schema_url = "https://pitloom.dev/provenance/1" + license_statements = [ + stmt + for a in annotations + if (stmt := json.loads(a["statement"])).get("schema") == schema_url + and "license" in stmt.get("fields", {}) + ] + for stmt in license_statements: + license_field = stmt["fields"]["license"] + # Method/source evidence is allowed; the raw SPDX id is not duplicated. + assert "mit" not in str(license_field), ( + "license value 'mit' should not be repeated in Annotation" + ) + + +def test_combined_output_is_deterministic() -> None: + """Two generation runs with identical inputs produce byte-identical JSON.""" + model = _build_integrated_model() + fixed_ci = CreationMetadata(creation_datetime="2026-08-08T00:00:00+00:00") + + run1 = build_model(model, fixed_ci).to_json(pretty=False) + run2 = build_model(model, fixed_ci).to_json(pretty=False) + + assert run1 == run2, "combined SBOM output must be byte-identical" + + +def test_combined_document_round_trips_through_spdx_model() -> None: + """The combined SBOM is accepted by spdx-python-model deserialization.""" + exporter = build_model(_build_integrated_model(), CreationMetadata()) + raw_json = exporter.to_json(pretty=False) + + object_set = spdx3.SHACLObjectSet() + spdx3.JSONLDDeserializer().read(io.BytesIO(raw_json.encode("utf-8")), object_set) + + element_types = {type(o).__name__ for o in object_set.objects} + assert "ai_AIPackage" in element_types + assert "dataset_DatasetPackage" in element_types + assert "Agent" in element_types + assert "Relationship" in element_types + assert "SpdxDocument" in element_types + assert "software_Sbom" in element_types + + +def test_fragment_fixture_files_available() -> None: + """The pre-existing fragment fixtures used by integration tests exist.""" + fragment_dir = Path(__file__).parent / "fixtures" / "fragments" + for name in ( + "ai-model-fragment.spdx3.json", + "dataset-fragment.spdx3.json", + "training-run-fragment.spdx3.json", + ): + assert (fragment_dir / name).is_file(), f"missing fixture {name}" diff --git a/working-docs/design/metadata-provenance.md b/working-docs/design/metadata-provenance.md index a9d1f599..b833ef86 100644 --- a/working-docs/design/metadata-provenance.md +++ b/working-docs/design/metadata-provenance.md @@ -1,6 +1,6 @@ --- Created: 2026-02-07 -Last-Modified: 2026-07-20 +Last-Modified: 2026-08-08 SPDX-FileCopyrightText: 2026-present Arthit Suriyawongkul SPDX-FileType: DOCUMENTATION SPDX-License-Identifier: CC0-1.0 @@ -68,7 +68,7 @@ no native home at all and are always recorded: **fragment-unification** rationale (which criterion merged two elements) and **artifact-metadata preservation** (an AI model's verbatim original metadata, config-gated to when the model is not shipped and can't be re-extracted). See §10 of the -implementation plan for the use-case catalog and the Phase 2 native-backfill +implementation plan for the use-case catalog and the native SPDX backfill checklist. Controlled by `[tool.pitloom.provenance]` in `pyproject.toml`: diff --git a/working-docs/design/sbom-fragments.md b/working-docs/design/sbom-fragments.md index 2e301082..041e9b43 100644 --- a/working-docs/design/sbom-fragments.md +++ b/working-docs/design/sbom-fragments.md @@ -1,6 +1,6 @@ --- Created: 2026-04-13 -Last-Modified: 2026-07-09 +Last-Modified: 2026-08-08 SPDX-FileCopyrightText: 2026-present Arthit Suriyawongkul SPDX-FileType: DOCUMENTATION SPDX-License-Identifier: CC0-1.0 @@ -663,7 +663,7 @@ AIDOC assemblers reference the same fragment metadata without reading The work is ordered by user-impact priority. No item depends on completing all earlier items; each can be delivered independently. -### Phase 1: Structural improvements (high impact, low effort) +### Structural improvements (high impact, low effort) 1. **`FragmentConfig` data class** -- replace `list[str]` in `PitloomConfig`. Loader remains backward-compatible with plain strings. @@ -674,7 +674,7 @@ all earlier items; each can be delivered independently. 4. **SHA-256 verification in merge** -- add `fragment sign` CLI command + hash check on merge. -### Phase 2: SDK improvements (notebook and ML workflow ergonomics) +### SDK improvements (notebook and ML workflow ergonomics) 1. **`log_param`, `log_metric`, `log_tag` on `_ActiveRun`** -- expands the existing `Run` API without breaking changes. diff --git a/working-docs/implementation/annotation-provenance-full-plan.md b/working-docs/implementation/annotation-provenance-full-plan.md index 4c2fd15d..b392d67f 100644 --- a/working-docs/implementation/annotation-provenance-full-plan.md +++ b/working-docs/implementation/annotation-provenance-full-plan.md @@ -1,5 +1,6 @@ --- Created: 2026-08-08 +Last-Modified: 2026-08-08 SPDX-FileCopyrightText: 2026-present Arthit Suriyawongkul SPDX-FileType: DOCUMENTATION SPDX-License-Identifier: CC0-1.0 @@ -10,9 +11,9 @@ SPDX-License-Identifier: CC0-1.0 > Archived copy of the plan originally approved via Claude Code plan mode > (source: `~/.claude/plans/annotation-statement-should-not-stateful-narwhal.md`, > a path outside this repo and not guaranteed to exist in a future -> session). Saved here so the full Phase 1 design record travels with the -> repo. Phase 1 (this plan) is **complete and merged** — see -> [`phase2-native-backfill-handover.md`](phase2-native-backfill-handover.md) +> session). Saved here so the full provenance Annotation work design record travels with the +> repo. provenance Annotation work (this plan) is **complete and merged** — see +> [`native-spdx-backfill-handover.md`](native-spdx-backfill-handover.md) > for current status and what to do next. This file is historical > reference, not a live task list. @@ -53,8 +54,8 @@ squarely inside the boundary principle. Symmetrically, several facts currently living only in a comment/Annotation actually **do** have a native SPDX home that Pitloom does not yet populate (declared-vs-concluded license, fragment `imports`, external identifiers, enrichment `CreationInfo`). Moving those to -native constructs is a **documented Phase 2** (built after this Annotation work — see the -"Phase 2" section) so we don't forget them and can see, per use case, which half is native +native constructs is a **documented native SPDX backfill** (built after this Annotation work — see the +"native SPDX backfill" section) so we don't forget them and can see, per use case, which half is native and which half is genuinely Annotation-only. ## Boundary principle (native-first) @@ -80,15 +81,15 @@ and which half is genuinely Annotation-only. | license *value* | Yes (`hasDeclaredLicense` + `SimpleLicensingText`) | never the value; only "detected vs declared" (G1) | | dependency edge | Yes (`dependsOn`) | never on the edge; see G3 on the package | | hashes / files / relationships / PURL / ExternalRef | Yes | never | -| **inferred / detected / AI-generated** qualifier | Partly — see N2 | **G1 — necessary** (assertedness beyond N2) | -| **declared vs. concluded license** (author-stated vs Pitloom-detected) | **Yes, native, but NOT built** — `hasDeclaredLicense` / `hasConcludedLicense` are currently mirrored (`deps.py:246-252`, "no inference yet") | **Phase 2 → N2**; Annotation keeps only the evidence/why | -| **multi-source disagreement / which source won** | Partly (license → N2) | **G2 — necessary on conflict** | +| **inferred / detected / AI-generated** qualifier | Partly — see declared vs. concluded license | **G1 — necessary** (assertedness beyond declared vs. concluded license) | +| **declared vs. concluded license** (author-stated vs Pitloom-detected) | **Yes, native, but NOT built** — `hasDeclaredLicense` / `hasConcludedLicense` are currently mirrored (`deps.py:246-252`, "no inference yet") | **native SPDX backfill → declared vs. concluded license**; Annotation keeps only the evidence/why | +| **multi-source disagreement / which source won** | Partly (license → declared vs. concluded license) | **G2 — necessary on conflict** | | **declared constraint vs resolved version** | **No** (`software_packageVersion` holds only the resolved value) | **G3 — useful** | | **sub-file extraction location** (safetensors `__metadata__`, GGUF kv, pt2 `extra/*`) | **No** | **G4 — useful** | -| **DOI / arXiv / repo & model-card URLs** (from HF `extra_data`) | **Yes, native, but NOT built** — `ExternalIdentifier` (doi) / `ExternalRef` | **Phase 2 → N4**; today only in provenance/`extra_data` | -| **element came from fragment document F** | **Yes, native, but NOT built** — `SpdxDocument.imports` + `ExternalMap` (doc-level; flagged unbuilt in `sbom-fragments.md:146,698-701`) | **Phase 2 → N1**; Annotation keeps the *criterion* | +| **DOI / arXiv / repo & model-card URLs** (from HF `extra_data`) | **Yes, native, but NOT built** — `ExternalIdentifier` (doi) / `ExternalRef` | **native SPDX backfill → external identifiers**; today only in provenance/`extra_data` | +| **element came from fragment document F** | **Yes, native, but NOT built** — `SpdxDocument.imports` + `ExternalMap` (doc-level; flagged unbuilt in `sbom-fragments.md:146,698-701`) | **native SPDX backfill → fragment origin**; Annotation keeps the *criterion* | | **why two fragments were unified** (criterion: registry-id / sha256 / structural) | **No** (merged-away ids vanish; `imports` can't say *why*) | **A1 — necessary** | -| **who/when enriched** | **Yes, native, but NOT built** — a second `CreationInfo` per enrichment run | **Phase 2 → N3**; Annotation keeps which-field + before/after | +| **who/when enriched** | **Yes, native, but NOT built** — a second `CreationInfo` per enrichment run | **native SPDX backfill → enrichment CreationInfo**; Annotation keeps which-field + before/after | | **field overridden A→B by enrichment source Y** | **No** (`CreationInfo` is element-level, no before/after) | **E1/E2 — necessary (design only now)** | | **verbatim original artifact metadata blob** (GGUF kv-store, safetensors `__metadata__`, HF `config.json` + model card) | **No** (native mapping is a lossy subset; `ExternalRef` is only a *pointer* that can dangle) | **P1 — useful; necessary when the artifact is not shipped** | @@ -195,31 +196,32 @@ relevant subset is still emitted as SPDX classes/properties/relationships — ra `extra_data` maps in `_gguf.py`, `_safetensors.py`, `_huggingface.py` — see step below on retaining the *complete* raw map. -## Phase 2 (documented now, built after the Annotation work): native-first backfill +## native SPDX backfill (documented now, built after the Annotation work): native-first backfill Several facts above have a real SPDX 3 home that Pitloom does not yet populate — it records them only in a comment/Annotation, or mirrors/normalizes them away. The native-first principle says the *fact/value* belongs in the native construct; the Annotation should then shrink to only -the non-native residual. Building these is deferred to a next phase, but documenting the target +the non-native residual. Building these is deferred to later implementation, but documenting the target now keeps the two halves coherent and prevents the Annotation from ossifying as the permanent home for things that ought to be native. **For each item: build the native construct, then trim the corresponding Annotation content.** | # | Fact | Native construct to build | Currently | Residual left to Annotation | | --- | --- | --- | --- | --- | -| **N1** | Element originates from fragment document F | `SpdxDocument.imports` + `ExternalMap` (one per source fragment) | Fragment origin discarded at merge (`fragments.py:461-464`); `imports` unbuilt (`sbom-fragments.md:146,698-701`) | Unification *criterion* only (registry-id/sha256/structural) — A1 | -| **N2** | Declared vs. concluded license | Distinct `hasDeclaredLicense` (author-stated) and `hasConcludedLicense` (Pitloom-detected from LICENSE/licenseid evidence) | Concluded is set equal to declared, "no inference yet" (`deps.py:246-252`) | The *evidence* (which file/heuristic) behind the concluded license — G1/G2 | -| **N3** | Who/when enriched | A second `CreationInfo` (createdBy = enricher agent, createdUsing = enricher tool, created = enrichment time) attached to enriched elements | Enrichers mutate in-place under the original `CreationInfo`; agent path only tags a comment | Which field + before/after value + inferred-marker — E1/E2 | -| **N4** | External identifiers (DOI, arXiv, repo URL, model-card URL) | `ExternalIdentifier` (type `doi`, …) / `ExternalRef` on the AI package | Captured into `extra_data`/provenance only (`_huggingface.py:710-764`) | none once mapped (fully native) | -| **N5** | Base-model lineage (HF `base_model` / `base_model_relation`) | A `Relationship` to a base-model element if a suitable `relationshipType` exists, else `ExternalRef` | In `extra_data` only | none once mapped, or the raw relation string if no native type fits | -| **N6** | Dataset `creator` | `Agent` + a creation/attribution relationship on the dataset package | Extracted (`_croissant.py:208`) but not wired onto the `dataset_DatasetPackage` | none once mapped | - -Relationship to Phase 1 (this plan): every use case splits into a **native part** (Phase 2 above) -and an **Annotation part** (this plan). E.g. G2 license = N2 native relationships + Annotation -evidence; A1 unification = N1 native `imports` + Annotation criterion; E1/E2 enrichment = N3 native -`CreationInfo` + Annotation before/after. Phase 1 deliberately does **not** pre-empt these — where a -native home is coming in Phase 2, the Phase-1 Annotation still carries the whole fact for now and is -trimmed to the residual when N-x lands. Track each N-item as a checklist in +| **fragment origin** | Element originates from fragment document F | `SpdxDocument.imports` + `ExternalMap` (one per source fragment) | Fragment origin discarded at merge (`fragments.py:461-464`); `imports` unbuilt (`sbom-fragments.md:146,698-701`) | Unification *criterion* only (registry-id/sha256/structural) — A1 | +| **declared vs. concluded license** | Declared vs. concluded license | Distinct `hasDeclaredLicense` (author-stated) and `hasConcludedLicense` (Pitloom-detected from LICENSE/licenseid evidence) | Concluded is set equal to declared, "no inference yet" (`deps.py:246-252`) | The *evidence* (which file/heuristic) behind the concluded license — G1/G2 | +| **enrichment CreationInfo** | Who/when enriched | A second `CreationInfo` (createdBy = enricher agent, createdUsing = enricher tool, created = enrichment time) attached to enriched elements | Enrichers mutate in-place under the original `CreationInfo`; agent path only tags a comment | Which field + before/after value + inferred-marker — E1/E2 | +| **external identifiers** | External identifiers (DOI, arXiv, repo URL, model-card URL) | `ExternalIdentifier` (type `doi`, …) / `ExternalRef` on the AI package | Captured into `extra_data`/provenance only (`_huggingface.py:710-764`) | none once mapped (fully native) | +| **base-model lineage** | Base-model lineage (HF `base_model` / `base_model_relation`) | A `Relationship` to a base-model element if a suitable `relationshipType` exists, else `ExternalRef` | In `extra_data` only | none once mapped, or the raw relation string if no native type fits | +| **dataset creator attribution** | Dataset `creator` | `Agent` + a creation/attribution relationship on the dataset package | Extracted (`_croissant.py:208`) but not wired onto the `dataset_DatasetPackage` | none once mapped | + +Relationship to provenance Annotation work (this plan): every use case splits into a **native part** (native SPDX backfill above) +and an **Annotation part** (this plan). E.g. G2 license = declared vs. concluded license native relationships + Annotation +evidence; A1 unification = fragment origin native `imports` + Annotation criterion; E1/E2 enrichment = enrichment CreationInfo native +`CreationInfo` + Annotation before/after. provenance Annotation work deliberately does **not** pre-empt these — where a +native home is coming in native SPDX backfill, the Annotation work still carries +the whole fact for now and is trimmed to the residual when the corresponding +construct lands. Track each native construct as a checklist in `working-docs/implementation/annotation-provenance.md` so the trim is not forgotten. ## Design changes @@ -319,7 +321,7 @@ the fragment-merge caveat: a fragment can only *fill* an empty scalar today (can conflict, `fragments.py`), so structured override needs the enricher to run in-process (the designed `enrich/` subpackage) rather than via fragment. No code now. -## Files to modify (Phase 1 — historical, already applied) +## Files to modify (provenance Annotation work — historical, already applied) - `src/pitloom/assemble/spdx3/provenance.py` — `_is_high_signal`, minimal-mode filter in the `pitloom/1` encoder, new `build_unification_annotation`, `build_source_metadata_annotation`, @@ -335,7 +337,7 @@ conflict, `fragments.py`), so structured override needs the enricher to run in-p `plugins/hatch.py`, `loom.py` — thread `provenance_detail` (+ preserve flag where AI models flow); drop the two relationship annotations. - Docs: `working-docs/implementation/annotation-provenance.md` (boundary + use cases + `detail` + - the **Phase 2 native-backfill checklist N1–N6** with per-item "trim the Annotation to residual" + the **native SPDX backfill checklist** with per-item "trim the Annotation to residual" notes), `working-docs/design/metadata-provenance.md` (native-first principle), `CHANGELOG.md`. - Tests: extend `tests/test_annotation_provenance.py` (high-signal filter, minimal vs full, unification + preservation annotation shapes + determinism); update `tests/test_provenance.py`, @@ -343,7 +345,7 @@ conflict, `fragments.py`), so structured override needs the enricher to run in-p `tests/test_fragments.py` case asserting a unification annotation on a hash-merged survivor; add an AI-model case asserting P1 `auto` embeds the raw blob for a URL/HF model but not a bundled one. -## Verification (Phase 1 — historical) +## Verification (provenance Annotation work — historical) - `python3 -m pytest tests/ -q` (pyenv `pitloom310`); mypy + ruff clean. - Generate an SBOM at default `detail="minimal"`: assert annotations exist ONLY for inferred/detected @@ -361,8 +363,8 @@ conflict, `fragments.py`), so structured override needs the enricher to run in-p ## Outcome -Phase 1 shipped as PR [#102](https://github.com/bact/pitloom/pull/102), merged to `main`. -Phase 2 (the N1-N6 table above) is tracked separately — see -[`phase2-native-backfill-handover.md`](phase2-native-backfill-handover.md) for -current status (N1, N2, N4, N5, N6 merged via PRs #108, #105, #106, #109, #107; -N3 blocked on `enrich/` subpackage) and next steps. +provenance Annotation work shipped as PR [#102](https://github.com/bact/pitloom/pull/102), merged to `main`. +native SPDX backfill (the six native constructs table above) is tracked separately — see +[`native-spdx-backfill-handover.md`](native-spdx-backfill-handover.md) for +current status (fragment origin, license conclusion, external identifiers, base-model lineage, and dataset creator attribution merged via PRs #108, #105, #106, #109, #107; +enrichment CreationInfo blocked on `enrich/` subpackage) and next steps. diff --git a/working-docs/implementation/annotation-provenance.md b/working-docs/implementation/annotation-provenance.md index 482b7a3a..21364bc3 100644 --- a/working-docs/implementation/annotation-provenance.md +++ b/working-docs/implementation/annotation-provenance.md @@ -10,8 +10,8 @@ SPDX-License-Identifier: CC0-1.0 **Status:** implemented (2026-07-20, uncommitted on branch `provenance-annotation`) -- see §9 for what shipped vs. deferred, and **§10 for the 2026-07-20 boundary -refinement** (non-native / high-signal only, config-gated) plus the Phase 2 -native-backfill checklist. +refinement** (non-native / high-signal only, config-gated) plus the native SPDX +backfill checklist. **Planned with:** Opus 4.8. **Implemented by:** Sonnet 5. **Related design docs:** [`working-docs/design/metadata-provenance.md`](../design/metadata-provenance.md), [`working-docs/design/model-metadata-extraction.md`](../design/model-metadata-extraction.md). @@ -661,7 +661,7 @@ entry points exactly as `provenance_format`/`schema` already were. survivor, which SPDX cannot express — and emits a `provenance/unification/1` Annotation on the survivor. A same-id registry match carries no such fact (nothing distinct was folded) and is not annotated; its fragment origin is - Phase-2 `SpdxDocument.imports` territory (see N1 below). + native SPDX `SpdxDocument.imports` territory (see fragment origin below). - **Enrichment** — E1 override lineage, E2 AI-inferred-vs-extracted marker (both necessary; design-only — the `enrich/` subpackage is unbuilt). - **Preservation** — P1 verbatim original AI-model metadata @@ -687,31 +687,31 @@ entry points exactly as `provenance_format`/`schema` already were. re-extractable). A future fix, if needed, should truncate with an explicit, visible marker (e.g. `"_truncated": true`) rather than cutting silently. -### Phase 2 (documented; built after this Annotation work): native-first backfill +### native SPDX backfill (documented; built after this Annotation work): native-first backfill Several facts still live only in an Annotation/comment but have a real SPDX home Pitloom does not yet populate. Build the native construct, then **trim the corresponding Annotation to the residual**. Track here so it is not forgotten: -- [x] **N1 — Fragment origin** → `SpdxDocument.imports` + `ExternalMap` (per +- [x] **fragment origin — Fragment origin** → `SpdxDocument.imports` + `ExternalMap` (per source fragment). Residual in Annotation: the unification *criterion* only. (PR [#108](https://github.com/bact/pitloom/pull/108)) -- [x] **N2 — Declared vs. concluded license** → distinct `hasDeclaredLicense` +- [x] **declared vs. concluded license — Declared vs. concluded license** → distinct `hasDeclaredLicense` (author-stated) / `hasConcludedLicense` (Pitloom-detected). Today they are mirrored (`deps.py`, "no inference yet"). Residual: the detection evidence. (PR [#105](https://github.com/bact/pitloom/pull/105)) -- [ ] **N3 — Who/when enriched** → a second `CreationInfo` per enrichment run. +- [ ] **enrichment CreationInfo — Who/when enriched** → a second `CreationInfo` per enrichment run. Residual: which field + before/after value + inferred marker (E1/E2). (Blocked on `enrich/` subpackage) -- [x] **N4 — External identifiers** (DOI, arXiv, repo / model-card URL) → +- [x] **external identifiers — External identifiers** (DOI, arXiv, repo / model-card URL) → `ExternalIdentifier` / `ExternalRef` on the AI package (today only in `extra_data`/provenance). Residual: none once mapped. (PR [#106](https://github.com/bact/pitloom/pull/106)) -- [x] **N5 — Base-model lineage** (HF `base_model`) → `descendantOf` +- [x] **base-model lineage — Base-model lineage** (HF `base_model`) → `descendantOf` `Relationship`. Residual: raw relation subtype in comment. (PR [#109](https://github.com/bact/pitloom/pull/109)) -- [x] **N6 — Dataset `creator`** → `Agent` + `publishedBy` relationship on the +- [x] **dataset creator attribution — Dataset `creator`** → `Agent` + `publishedBy` relationship on the dataset package (extracted but not wired). Residual: none once mapped. (PR [#107](https://github.com/bact/pitloom/pull/107)) -Every use case splits into a **native part** (Phase 2) and an **Annotation -part** (this phase); e.g. G2 license = N2 relationships + Annotation evidence, -A1 unification = N1 `imports` + Annotation criterion. +Every use case splits into a **native part** (native SPDX backfill) and an **Annotation +part** (the Annotation role); e.g. G2 license = declared vs. concluded license relationships + Annotation evidence, +A1 unification = fragment origin `imports` + Annotation criterion. --- diff --git a/working-docs/implementation/phase2-native-backfill-handover.md b/working-docs/implementation/native-spdx-backfill-handover.md similarity index 59% rename from working-docs/implementation/phase2-native-backfill-handover.md rename to working-docs/implementation/native-spdx-backfill-handover.md index d6022155..3698ebe1 100644 --- a/working-docs/implementation/phase2-native-backfill-handover.md +++ b/working-docs/implementation/native-spdx-backfill-handover.md @@ -6,45 +6,51 @@ SPDX-FileType: DOCUMENTATION SPDX-License-Identifier: CC0-1.0 --- -# Handover: Phase 2 native-first backfill +# Handover: Native SPDX backfill -> **What this is**: Handover note for Phase 2 native-first backfill work following Phase 1 (provenance-as-Annotation). +> **What this is**: Handover note for native SPDX backfill work following +> provenance Annotation work. > -> **Goal**: move six facts (N1-N6) that previously lived only in a free-text comment or Annotation into their proper native SPDX 3 constructs (`hasConcludedLicense`, `ExternalIdentifier`, `imports`, relationships, etc.), then trim each corresponding Annotation down to just the residual that still has no native home. +> **Goal**: move six facts that previously lived only in a free-text comment or +> Annotation into their proper native SPDX 3 constructs +> (`hasConcludedLicense`, `ExternalIdentifier`, `imports`, relationships, etc.), +> then trim each corresponding Annotation down to just the residual that still +> has no native home. > > **Full design record**: [`annotation-provenance-full-plan.md`](annotation-provenance-full-plan.md) -> has the entire original Phase 1 plan (boundary principle, use-case catalog -> G1-G4/A1/A2/E1/E2/P1, the N1-N6 table with rationale, config/schema design) +> has the entire original provenance Annotation work plan (boundary principle, use-case catalog +> G1-G4/A1/A2/E1/E2/P1, the six native constructs table with rationale, +> config/schema design) > archived in-repo. This handover is a status/next-steps summary; read the -> full plan if you need the *why* behind any N-item or Annotation shape. +> full plan if you need the *why* behind any native construct or Annotation shape. ## Status -Phase 2 native-first backfill is **largely complete and merged**: -- ✅ **N2 — Declared vs. Concluded License**: PR [#105](https://github.com/bact/pitloom/pull/105) merged to `main`. -- ✅ **N4 — ExternalIdentifier & ExternalRef (DOI / arXiv / URLs)**: PR [#106](https://github.com/bact/pitloom/pull/106) merged to `main`. -- ✅ **N6 — Dataset Creator Agent & publishedBy Relationship**: PR [#107](https://github.com/bact/pitloom/pull/107) merged to `main`. -- ✅ **N1 — Fragment Origin (`SpdxDocument.imports` + `ExternalMap`)**: PR [#108](https://github.com/bact/pitloom/pull/108) merged to `main`. -- ✅ **N5 — Base-Model Lineage (`descendantOf` Relationship)**: PR [#109](https://github.com/bact/pitloom/pull/109) merged to `main`. -- 🛑 **N3 — Enrichment `CreationInfo`**: Blocked (waiting for `enrich/` subpackage). +Native SPDX backfill is **largely complete and merged**: +- ✅ **Declared vs. Concluded License**: PR [#105](https://github.com/bact/pitloom/pull/105) merged to `main`. +- ✅ **ExternalIdentifier & ExternalRef (DOI / arXiv / URLs)**: PR [#106](https://github.com/bact/pitloom/pull/106) merged to `main`. +- ✅ **Dataset Creator Agent & publishedBy Relationship**: PR [#107](https://github.com/bact/pitloom/pull/107) merged to `main`. +- ✅ **Fragment Origin (`SpdxDocument.imports` + `ExternalMap`)**: PR [#108](https://github.com/bact/pitloom/pull/108) merged to `main`. +- ✅ **Base-Model Lineage (`descendantOf` Relationship)**: PR [#109](https://github.com/bact/pitloom/pull/109) merged to `main`. +- 🛑 **Enrichment `CreationInfo`**: Blocked (waiting for `enrich/` subpackage). -## Principle (carried over from Phase 1) +## Principle (carried over from provenance Annotation work) Never put a value in an Annotation that has a native SPDX home. For each -N-item: **build the native construct, then trim the corresponding +native construct: **build the native construct, then trim the corresponding Annotation content to the residual** (the part that still has no native home — usually the *evidence* or *criterion* behind a value, not the value itself). -## Remaining work: N3 — enrichment `CreationInfo` +## Remaining work: Enrichment `CreationInfo` **Blocked** on the `enrich/` subpackage, which does not exist yet (see -`sbom-enrichment.md:145-169`). This is the only unimplemented N-item. -Do not start N3 itself until `enrich/` exists — instead: +`sbom-enrichment.md:145-169`). This is the only unimplemented native construct. +Do not start it until `enrich/` exists — instead: 1. **Check whether `enrich/` has landed.** Search for a `src/pitloom/enrich/` (or similarly named) subpackage and any related - PRs/branches. If it still doesn't exist, N3 stays blocked — report + PRs/branches. if it still doesn't exist, Enrichment `CreationInfo` stays blocked — report that back rather than building enrichment machinery as a side effect of this task. 2. **If `enrich/` exists**, build a second `CreationInfo` attached to @@ -59,16 +65,19 @@ Do not start N3 itself until `enrich/` exists — instead: the inferred-vs-extracted marker (`Source: AI agent | Method: inference`, today only in `skills/enrich/SKILL.md`'s free-text convention). This is speced as design-only in - `annotation-provenance.md` (E1/E2) — implement it as part of N3, not + `annotation-provenance.md` (E1/E2) — implement it as part of Enrichment + `CreationInfo`, not separately. -4. Mirror the N1/N2/N4/N5/N6 PRs' shape: one focused PR, tests in +4. Mirror the preceding focused PRs' shape: one focused PR, tests in `tests/test_annotation_provenance.py` plus wherever enrichment gets - its own test file, docs update in `annotation-provenance.md` (flip N3 + its own test file, docs update in `annotation-provenance.md` (flip enrichment CreationInfo from "not yet built" to done, same as the other five rows). ## Integration test — recommended, not yet done -N1, N2, N4, N5, N6 each landed as separate PRs, each presumably tested in +Fragment origin, license conclusion, external identifiers, base-model lineage, +and dataset creator attribution each landed as separate PRs, each presumably +tested in isolation (unit/compliance tests scoped to that one native construct). What's missing: a single **end-to-end test that exercises all five together** on one representative input (e.g. an AI-model package with a @@ -77,37 +86,37 @@ dataset with a creator, assembled from ≥2 fragments so unification also fires) and asserts on the *whole* generated SBOM: - All five native constructs appear correctly on the same document at - once (no interaction bugs — e.g. does adding `ExternalIdentifier` (N4) - change spdxId minting in a way that breaks fragment unification (N1)?). + once (no interaction bugs — e.g. does adding `ExternalIdentifier` change + spdxId minting in a way that breaks fragment unification?). - Each Annotation is trimmed to its residual and does **not** duplicate a value now covered natively (the core regression risk: an old Annotation shape lingering after its native counterpart landed). - Determinism holds across the combined output — `sort_keys=True` and - sorted collections were verified per-feature in Phase 1, but a + sorted collections were verified per-feature in provenance Annotation work, but a multi-feature SBOM has more interleaving to get wrong; run generation twice and diff for byte-identical output. - `pyspdxtools`/whatever SPDX 3 validator the repo already uses (see `tests/test_spdx3_compliance.py`) accepts the combined document. -Suggested location: a new `tests/test_phase2_integration.py`, or extend +Suggested location: a new `tests/test_native_spdx_integration.py`, or extend `tests/test_spdx3_compliance.py` if that's already the repo's home for -whole-document assertions. This can be built once N3 lands (to cover all +whole-document assertions. This can be built once enrichment CreationInfo lands (to cover all six), or sooner covering the five that are already done — the user should decide which. -## Workflow notes carried from Phase 1 +## Workflow notes carried from provenance Annotation work - **Never commit/push without explicit user instruction.** The user merges PRs themselves; don't merge or push branches unprompted. - Dev/test env: pyenv `pitloom310` (see project memory `project_dev_environment.md` if available in this session) — use its explicit python path for scratch/out-of-repo builds. -- Verification loop that worked well in Phase 1: implement → run +- Verification loop that worked well in provenance Annotation work: implement → run `python3 -m pytest tests/ -q` + mypy + ruff → self-review or spawn narrow-focus Sonnet review agents (parallel, read-only, each required to produce a concrete repro) → triage findings → fix → re-verify. Two rounds of this caught 5 real issues (delimiter injection, JSON - NaN/Infinity validity, non-deterministic set serialization) in Phase 1. + NaN/Infinity validity, non-deterministic set serialization) in provenance Annotation work. - Determinism requirement: `sort_keys=True` in all JSON serialization, sorted lists/sets before emission — Pitloom SBOMs must be byte-stable across runs. @@ -118,26 +127,26 @@ should decide which. ## Suggested first action for the picking-up session 1. Confirm `main` has PRs #105, #106, #107, #108, #109 merged. -2. Check whether `enrich/` subpackage exists yet (N3's blocker). Report +2. Check whether `enrich/` subpackage exists yet (enrichment CreationInfo's blocker). Report status either way before doing anything else. 3. If still blocked, ask the user whether to prioritize the integration - test (covering the five done items) or wait on N3. + test (covering the five done items) or wait on enrichment CreationInfo. 4. Re-read `annotation-provenance.md` §10 in full before starting, since this handover only summarizes it. ## Prompt to start a new session on this handover ``` -Read working-docs/implementation/phase2-native-backfill-handover.md in +Read working-docs/implementation/native-spdx-backfill-handover.md in full, then working-docs/implementation/annotation-provenance-full-plan.md for the complete original design (boundary principle, use-case catalog, -N1-N6 rationale) if you need background on any item. +the six native constructs rationale) if you need background on any item. -N1, N2, N4, N5, N6 are merged (PRs #108, #105, #106, #109, #107). -N3 (enrichment CreationInfo) is blocked on the enrich/ subpackage not +fragment origin, license conclusion, external identifiers, base-model lineage, and dataset creator attribution are merged (PRs #108, #105, #106, #109, #107). +enrichment CreationInfo (enrichment CreationInfo) is blocked on the enrich/ subpackage not existing yet — check if it has landed since this doc was written. Also evaluate whether to build the integration test described in the "Integration test" section now (covering the five merged items) versus -waiting for N3. Report status and recommended next step before making +waiting for enrichment CreationInfo. Report status and recommended next step before making any code changes. ```