Skip to content

API reference

Auto-generated from docstrings -- full signatures, parameter types, and defaults for the same Python API surface that page introduces narratively. Start there for how to use it; come here for the exact call signature.

Generator functions

pitloom.assemble.generate

generate(
    target: Path | str = ".",
    *,
    offline: bool | None = None,
    output_path: Path | None = None,
    creation_metadata: CreationMetadata | None = None,
    pretty: bool | None = None,
    describe_relationship: bool | None = None,
    id_registry: str | Path | IdRegistry | None = None,
    provenance: ProvenanceConfig | None = None,
    enrich: bool | None = None,
    extract_file_header: bool | None = None,
    scan_model_usage: bool | None = None,
    trust_wheel_model: bool | None = None,
    content_type: bool | None = None,
    content_type_method: str | None = None,
    update_id_registry: bool | None = None,
    use_lockfile: bool | None = None,
    build_options: BuildOptions = BuildOptions(),
    max_source_metadata_bytes: int | None = None,
    pitloom_config: PitloomConfig | None = None,
) -> str

Smart unified entrypoint for generating SPDX 3 SBOMs across all target types.

build_options (see :class:~pitloom.core.build_options.BuildOptions) only takes effect for a project directory target (the generate_project_sbom() dispatch below); any other target logs one WARNING: per given build flag here, immediately, before dispatching. A "project" classification covers both a project directory and an sdist archive (the two aren't told apart until generate_project_sbom() itself checks), so that dispatch settles/ warns about build_options on its own, right before its own metadata read -- not repeated here.

Source code in pitloom/assemble/__init__.py
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
def generate(
    target: Path | str = ".",
    *,
    offline: bool | None = None,
    output_path: Path | None = None,
    creation_metadata: CreationMetadata | None = None,
    pretty: bool | None = None,
    describe_relationship: bool | None = None,
    id_registry: str | Path | IdRegistry | None = None,
    provenance: ProvenanceConfig | None = None,
    enrich: bool | None = None,
    extract_file_header: bool | None = None,
    scan_model_usage: bool | None = None,
    trust_wheel_model: bool | None = None,
    content_type: bool | None = None,
    content_type_method: str | None = None,
    update_id_registry: bool | None = None,
    use_lockfile: bool | None = None,
    build_options: BuildOptions = BuildOptions(),
    max_source_metadata_bytes: int | None = None,
    pitloom_config: PitloomConfig | None = None,
) -> str:
    """Smart unified entrypoint for generating SPDX 3 SBOMs across all target types.

    ``build_options`` (see :class:`~pitloom.core.build_options.BuildOptions`)
    only takes effect for a project directory target (the
    ``generate_project_sbom()`` dispatch below); any other target logs
    one ``WARNING:`` per given build flag here, immediately, before
    dispatching. A "project" classification covers both a project
    directory and an sdist archive (the two aren't told apart until
    ``generate_project_sbom()`` itself checks), so that dispatch settles/
    warns about ``build_options`` on its own, right before its own
    metadata read -- not repeated here.
    """
    # Before the no-effect warnings below, which precede any delegate's own
    # configure_logging() call.
    configure_logging()
    target_str = str(target).strip()
    classification = _classify_target(target_str)

    if classification != "project":
        # Reset to defaults too (not just warn): nothing below reuses
        # build_options for a non-project classification, but this keeps
        # the same "warn once, reset to defaults" contract every other
        # settle_not_applicable() call site follows, so a future caller
        # added here can't accidentally double-warn.
        build_options = build_options.settle_not_applicable(
            target_str, NON_PROJECT_TARGET_REASON
        )

    options: dict[str, Any] = {
        "output_path": output_path,
        "creation_metadata": creation_metadata,
        "pretty": pretty,
        "describe_relationship": describe_relationship,
        "id_registry": id_registry,
        "provenance": provenance,
        "offline": offline,
        "enrich": enrich,
        "extract_file_header": extract_file_header,
        "scan_model_usage": scan_model_usage,
        "trust_wheel_model": trust_wheel_model,
        "content_type": content_type,
        "content_type_method": content_type_method,
        "update_id_registry": update_id_registry,
        "max_source_metadata_bytes": max_source_metadata_bytes,
        "pitloom_config": pitloom_config,
        "use_lockfile": use_lockfile,
    }
    # Each delegate gets what it accepts; an option it does not accept is
    # settled here, once, with the target kind's reason.
    if classification == "env":
        return generate_env_sbom(
            **forward_options(ENV, target_str, generate_env_sbom, options)
        )
    if classification == "wheel":
        return generate_wheel_sbom(
            target_str,
            **forward_options(WHEEL, target_str, generate_wheel_sbom, options),
        )
    if classification == "hf":
        return generate_model_sbom(
            target_str,
            **forward_options(HF, target_str, generate_model_sbom, options),
        )
    if classification == "model_file":
        return generate_model_sbom(
            Path(target_str),
            **forward_options(MODEL_FILE, target_str, generate_model_sbom, options),
        )
    return generate_project_sbom(
        Path(target_str),
        **options,
        build_options=build_options,
    )

pitloom.assemble.generate_project_sbom

generate_project_sbom(
    project_target: Path | str,
    *,
    output_path: Path | None = None,
    creation_metadata: CreationMetadata | None = None,
    pretty: bool | None = None,
    describe_relationship: bool | None = None,
    project_metadata: ProjectMetadata | None = None,
    pitloom_config: PitloomConfig | None = None,
    id_registry: str | Path | IdRegistry | None = None,
    provenance: ProvenanceConfig | None = None,
    enrich: bool | None = None,
    extract_file_header: bool | None = None,
    scan_model_usage: bool | None = None,
    trust_wheel_model: bool | None = None,
    content_type: bool | None = None,
    content_type_method: str | None = None,
    offline: bool | None = None,
    update_id_registry: bool | None = None,
    use_lockfile: bool | None = None,
    build_options: BuildOptions = BuildOptions(),
    max_source_metadata_bytes: int | None = None,
) -> str

Generate a Source SPDX 3 SBOM for a Python project or sdist archive.

build_options (see :class:~pitloom.core.build_options.BuildOptions), unlike every other flag-shaped parameter here, deliberately has no pitloom_config.* fallback to defer to when unset. A caller must pass BuildOptions(allow=True) explicitly every time it wants Pitloom to execute the target project's own PEP 517 build backend; there is no config-cascade layer for it, since the config file lives in the (untrusted) project being scanned and must never be able to silently opt itself into code execution. For an sdist archive target every given build flag is ignored with one WARNING: each.

Settings come from the arguments, then pitloom_config, then the target's own [tool.pitloom], then the built-in defaults. pitloom_config alone replaces the target's config (--config); the project metadata is still read from project_target, and its lock-file cascade follows use_lockfile, else pitloom_config's use-lockfile.

use_lockfile only affects metadata resolved by this call: if the caller pre-supplies BOTH project_metadata and pitloom_config together, this parameter has no effect -- the lock-file cascade decision was already made when that metadata was produced. project_metadata alone is not supported: it is re-read from project_target, with a WARNING: explaining why (see "no silent deviations" in AGENTS.md).

For an sdist archive, extract_file_header, content_type, enrich, scan_model_usage, use_lockfile and trust_wheel_model have no effect, and for a directory trust_wheel_model has none; each warns when given (see :data:pitloom.core.inert_options.INERT).

Source code in pitloom/assemble/_generators.py
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
def generate_project_sbom(
    project_target: Path | str,
    *,
    output_path: Path | None = None,
    creation_metadata: CreationMetadata | None = None,
    pretty: bool | None = None,
    describe_relationship: bool | None = None,
    project_metadata: ProjectMetadata | None = None,
    pitloom_config: PitloomConfig | None = None,
    id_registry: str | Path | IdRegistry | None = None,
    provenance: ProvenanceConfig | None = None,
    enrich: bool | None = None,
    extract_file_header: bool | None = None,
    scan_model_usage: bool | None = None,
    trust_wheel_model: bool | None = None,
    content_type: bool | None = None,
    content_type_method: str | None = None,
    offline: bool | None = None,
    update_id_registry: bool | None = None,
    use_lockfile: bool | None = None,
    build_options: BuildOptions = BuildOptions(),
    max_source_metadata_bytes: int | None = None,
) -> str:
    """Generate a Source SPDX 3 SBOM for a Python project or sdist archive.

    ``build_options`` (see :class:`~pitloom.core.build_options.BuildOptions`),
    unlike every other flag-shaped parameter here, deliberately has no
    ``pitloom_config.*`` fallback to defer to when unset. A caller must
    pass ``BuildOptions(allow=True)`` explicitly every time it wants
    Pitloom to execute the target project's own PEP 517 build backend;
    there is no config-cascade layer for it, since the config file lives
    in the (untrusted) project being scanned and must never be able to
    silently opt itself into code execution. For an sdist archive target
    every given build flag is ignored with one ``WARNING:`` each.

    Settings come from the arguments, then *pitloom_config*, then the
    target's own ``[tool.pitloom]``, then the built-in defaults.
    *pitloom_config* alone replaces the target's config (``--config``); the
    project metadata is still read from *project_target*, and its lock-file
    cascade follows ``use_lockfile``, else *pitloom_config*'s
    ``use-lockfile``.

    ``use_lockfile`` only affects metadata resolved by this call: if the
    caller pre-supplies BOTH ``project_metadata`` and ``pitloom_config``
    together, this parameter has no effect -- the lock-file cascade
    decision was already made when that metadata was produced.
    *project_metadata* alone is not supported: it is re-read from
    *project_target*, with a ``WARNING:`` explaining why (see "no silent
    deviations" in AGENTS.md).

    For an sdist archive, ``extract_file_header``, ``content_type``,
    ``enrich``, ``scan_model_usage``, ``use_lockfile`` and
    ``trust_wheel_model`` have no effect, and for a directory
    ``trust_wheel_model`` has none; each warns when given (see
    :data:`pitloom.core.inert_options.INERT`).
    """
    configure_logging()
    target_path = Path(project_target)

    # A cheap stat, before any project-metadata/lock-file read, so the
    # build-flag WARNING: is the first thing this call logs.
    build_options = build_options.settle_target(target_path)

    if target_path.is_file():
        # Every option, not only today's inert ones, so a new INERT[SDIST]
        # row entry needs no change here.
        settle_inert(
            SDIST,
            target_path,
            {
                "pretty": pretty,
                "describe_relationship": describe_relationship,
                "enrich": enrich,
                "extract_file_header": extract_file_header,
                "scan_model_usage": scan_model_usage,
                "trust_wheel_model": trust_wheel_model,
                "content_type": content_type,
                "content_type_method": content_type_method,
                "max_source_metadata_bytes": max_source_metadata_bytes,
                "offline": offline,
                "id_registry": id_registry,
                "update_id_registry": update_id_registry,
                "creation_metadata": creation_metadata,
                "use_lockfile": use_lockfile,
            },
        )
    else:
        settle_inert(PROJECT, target_path, {"trust_wheel_model": trust_wheel_model})

    if project_metadata is None or pitloom_config is None:
        _warn_if_metadata_without_config(project_metadata, pitloom_config, target_path)
        project_metadata, pitloom_config, _ = resolve_project_with_lockfile(
            target_path, use_lockfile, pitloom_config
        )

    cfg = apply_overrides(
        pitloom_config,
        ConfigOverrides(
            provenance=provenance,
            enrich=enrich,
            extract_file_header=extract_file_header,
            scan_model_usage=scan_model_usage,
            content_type=content_type,
            content_type_method=content_type_method,
            offline=offline,
            pretty=pretty,
            describe_relationship=describe_relationship,
            update_id_registry=update_id_registry,
            max_source_metadata_bytes=max_source_metadata_bytes,
        ),
    )

    resolved_registry = resolve_registry(
        id_registry,
        cfg.id_registry,
        registry_base_dir(target_path),
    )

    # Owns SIGTERM/SIGHUP handling for the whole lifetime of a
    # build-and-read result: once a build ran, a signal until the end of
    # this block removes its extraction directory before the process
    # ends (see pitloom.core.build_signals). Without a build, inert.
    with TerminationGuard():
        if target_path.is_file():
            # Warned above, before metadata resolution -- nothing left to
            # warn about here.
            merkle_root = None
            project_files = project_metadata.files
            cleanup_discovery: Callable[[], None] = _noop_cleanup
        else:
            merkle_root, project_files, cleanup_discovery = get_wheel_files(
                target_path,
                scan_file_headers=cfg.extract_file_header,
                detect_content_type=cfg.content_type.enabled,
                content_type_method=cfg.content_type.method,
                content_type_overrides=cfg.content_type.overrides,
                build_options=build_options,
            )

        # cleanup_discovery (a no-op unless --allow-build's build-and-read
        # sourced project_files) must stay alive -- and this whole block
        # must run inside its try -- through every step below that either
        # re-reads a ProjectFile's bytes from disk via physical_path (AI-
        # model scanning, enrichment) or could itself raise before reaching
        # them (the fresh-containers copy): any of these raising before
        # cleanup_discovery() runs would leak the build-and-read temp
        # directory. Nothing after this block reads file bytes again
        # (document assembly only uses distribution_path/physical_path as
        # string keys, never re-opens the file).
        try:
            if not target_path.is_file():
                # `project_files` becomes the metadata's authoritative file
                # list for a directory target. Every other dict/list field
                # gets its own fresh copy too, so nothing downstream can
                # mutate the caller's own `project_metadata`.
                project_metadata = project_metadata.replace_with_fresh_containers(
                    files=project_files
                )

            ai_models = (
                scan_project_for_ai_models(
                    target_path,
                    project_files,
                    scan_usage=cfg.scan_model_usage is True,
                    usage_hint=lambda: cfg.scan_model_usage is None,
                )
                if target_path.is_dir()
                else []
            )

            enrichment_results_by_model = run_enrichers_for_models(
                ai_models, cfg.enrich, target_path
            )
        finally:
            cleanup_discovery()

    doc = DocumentModel(
        project=project_metadata,
        creation_metadata=creation_metadata or cfg.creation_metadata,
        ai_models=ai_models,
    )
    exporter = build(
        doc,
        merkle_root=merkle_root,
        sbom_type=spdx3_bindings.software_SbomType.source,
        registry=resolved_registry,
        enrichment_results_by_model=enrichment_results_by_model,
        **cfg.assemble_options,
    )

    if target_path.is_dir():
        merge_fragments(target_path, cfg.fragments, exporter)

    _sync_registry(exporter, resolved_registry, cfg.update_id_registry)

    sbom_json = exporter.to_json(
        pretty=cfg.pretty,
        describe_relationship=bool(cfg.describe_relationship),
    )

    write_sbom_output(sbom_json, output_path)

    return sbom_json

pitloom.assemble.generate_wheel_sbom

generate_wheel_sbom(
    wheel_path: Path | str,
    *,
    output_path: Path | None = None,
    creation_metadata: CreationMetadata | None = None,
    pretty: bool | None = None,
    describe_relationship: bool | None = None,
    id_registry: str | Path | IdRegistry | None = None,
    provenance: ProvenanceConfig | None = None,
    offline: bool | None = None,
    content_type_method: str | None = None,
    update_id_registry: bool | None = None,
    max_source_metadata_bytes: int | None = None,
    scan_model_usage: bool | None = None,
    trust_wheel_model: bool | None = None,
    pitloom_config: PitloomConfig | None = None,
) -> str

Generate an Analyzed SPDX 3 SBOM for a built Python wheel.

A wheel has no [tool.pitloom] of its own, and none is borrowed: not from the current directory, not from beside the wheel -- either may belong to an unrelated project. Settings come from the arguments, then pitloom_config when the caller names one explicitly, then the built-in defaults. The same holds for the registry: id_registry, else the explicit config's id-registry; no loom-id-registry.json is searched for.

An explicit pitloom_config applies in full, identity included: its creators, creation datetime and comment fill in when creation_metadata is not given.

AI models inside the wheel are found; scan_model_usage also records which Python files in it reference them. One model file is copied out of the wheel at a time, each up to the config's max-model-extract-bytes (no parameter: it is configuration only) and four times that in all, counting bytes copied and bytes read from archive members; a model beyond either limit is listed without metadata. Models in a format whose reader a hostile file can crash or hang (fastText, GGUF, HDF5, ONNX, PyTorch .pt/.pth) are listed without metadata too, with one INFO:, unless trust_wheel_model: for a wheel you trust only. It has no config key, so a config cannot opt in.

extract_file_header/content_type/enrich have no parameter here: reading a built wheel scans no file headers or content types, and its AI models are not enriched (see :data:pitloom.core.inert_options.INERT). content_type_method does apply, because it also steers whether dependency originator enrichment fetches a remote authors file.

Raises:

Type Description
ValueError

The wheel is refused as a whole (:class:~pitloom.core.wheel_dist_info.WheelRefused): not a ZIP archive, a member cannot be read, two members have one name or one holds a NUL.

OSError

wheel_path cannot be opened (missing, permission denied).

Source code in pitloom/assemble/_generators_wheel.py
 40
 41
 42
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
def generate_wheel_sbom(
    wheel_path: Path | str,
    *,
    output_path: Path | None = None,
    creation_metadata: CreationMetadata | None = None,
    pretty: bool | None = None,
    describe_relationship: bool | None = None,
    id_registry: str | Path | IdRegistry | None = None,
    provenance: ProvenanceConfig | None = None,
    offline: bool | None = None,
    content_type_method: str | None = None,
    update_id_registry: bool | None = None,
    max_source_metadata_bytes: int | None = None,
    scan_model_usage: bool | None = None,
    trust_wheel_model: bool | None = None,
    pitloom_config: PitloomConfig | None = None,
) -> str:
    """Generate an Analyzed SPDX 3 SBOM for a built Python wheel.

    A wheel has no ``[tool.pitloom]`` of its own, and none is borrowed: not
    from the current directory, not from beside the wheel -- either may
    belong to an unrelated project. Settings come from the arguments, then
    *pitloom_config* when the caller names one explicitly, then the built-in
    defaults. The same holds for the registry: *id_registry*, else the explicit
    config's ``id-registry``; no ``loom-id-registry.json`` is searched for.

    An explicit *pitloom_config* applies in full, identity included: its
    creators, creation datetime and comment fill in when *creation_metadata*
    is not given.

    AI models inside the wheel are found; *scan_model_usage* also records
    which Python files in it reference them. One model file is copied out
    of the wheel at a time, each up to the config's ``max-model-extract-bytes``
    (no parameter: it is configuration only) and four times that in all,
    counting bytes copied and bytes read from archive members; a
    model beyond either limit is listed without metadata. Models in a format
    whose reader a hostile file can crash or hang (fastText, GGUF, HDF5,
    ONNX, PyTorch ``.pt``/``.pth``) are listed without metadata too, with one
    ``INFO:``, unless *trust_wheel_model*: for a wheel you trust only. It has no config
    key, so a config cannot opt in.

    ``extract_file_header``/``content_type``/``enrich`` have no parameter
    here: reading a built wheel scans no file headers or content types, and
    its AI models are not enriched (see :data:`pitloom.core.inert_options.INERT`).
    ``content_type_method`` does apply, because it also steers whether
    dependency originator enrichment fetches a remote authors file.

    Raises:
        ValueError: The wheel is refused as a whole
            (:class:`~pitloom.core.wheel_dist_info.WheelRefused`): not a ZIP
            archive, a member cannot be read, two members have one name or
            one holds a NUL.
        OSError: *wheel_path* cannot be opened (missing, permission denied).
    """
    return generate_wheel_sbom_with_metadata(
        wheel_path,
        output_path=output_path,
        creation_metadata=creation_metadata,
        pretty=pretty,
        describe_relationship=describe_relationship,
        id_registry=id_registry,
        provenance=provenance,
        offline=offline,
        content_type_method=content_type_method,
        update_id_registry=update_id_registry,
        max_source_metadata_bytes=max_source_metadata_bytes,
        scan_model_usage=scan_model_usage,
        trust_wheel_model=trust_wheel_model,
        pitloom_config=pitloom_config,
    )[0]

pitloom.assemble.generate_model_sbom

generate_model_sbom(
    source: Path | str,
    *,
    offline: bool | None = None,
    output_path: Path | None = None,
    creation_metadata: CreationMetadata | None = None,
    pretty: bool | None = None,
    describe_relationship: bool | None = None,
    id_registry: str | Path | IdRegistry | None = None,
    provenance: ProvenanceConfig | None = None,
    enrich: bool | None = None,
    max_source_metadata_bytes: int | None = None,
    pitloom_config: PitloomConfig | None = None,
) -> str

Generate an Analyzed SPDX 3 AIBOM for a local model file or HF repository.

Settings resolve as :func:~pitloom.assemble.generate_wheel_sbom describes: arguments, then an explicit pitloom_config, then the built-in defaults. Nothing is read from the current directory or from the model file's own directory.

Three parameters apply to one source kind only, and warn when given for the other (see :data:pitloom.core.inert_options.INERT): offline for a Hugging Face source (a local file never reaches the network), and enrich/id_registry for a local file.

Source code in pitloom/assemble/_model_generator.py
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
def generate_model_sbom(
    source: Path | str,
    *,
    offline: bool | None = None,
    output_path: Path | None = None,
    creation_metadata: CreationMetadata | None = None,
    pretty: bool | None = None,
    describe_relationship: bool | None = None,
    id_registry: str | Path | IdRegistry | None = None,
    provenance: ProvenanceConfig | None = None,
    enrich: bool | None = None,
    max_source_metadata_bytes: int | None = None,
    pitloom_config: PitloomConfig | None = None,
) -> str:
    """Generate an Analyzed SPDX 3 AIBOM for a local model file or HF repository.

    Settings resolve as :func:`~pitloom.assemble.generate_wheel_sbom`
    describes: arguments, then an explicit *pitloom_config*, then the
    built-in defaults. Nothing is read from the current directory or from
    the model file's own directory.

    Three parameters apply to one source kind only, and warn when given for
    the other (see :data:`pitloom.core.inert_options.INERT`): *offline* for
    a Hugging Face source (a local file never reaches the network), and
    *enrich*/*id_registry* for a local file.
    """
    configure_logging()
    source_str = str(source)
    is_hf = is_huggingface_source(source_str)
    settle_inert(
        HF if is_hf else MODEL_FILE,
        source_str,
        {"offline": offline, "enrich": enrich, "id_registry": id_registry},
    )
    cfg = resolve_standalone_config(
        pitloom_config,
        ConfigOverrides(
            provenance=provenance,
            enrich=enrich,
            offline=offline,
            pretty=pretty,
            describe_relationship=describe_relationship,
            max_source_metadata_bytes=max_source_metadata_bytes,
        ),
    )
    enrichment_results: list[EnrichmentResult] = []

    if is_hf:
        if cfg.offline:
            raise ValueError(
                "Offline mode enabled: cannot fetch remote Hugging Face source "
                f"'{source_str}'"
            )
        model = read_huggingface(source_str)
        entity_spdx_id = None
    else:
        model_path = Path(source)
        # Resolved before _read_local_model()/run_enrichers() below -- a
        # declared-but-missing/malformed registry should fail fast, never
        # after paying for a model read and enrichment first.
        resolved_registry = resolve_registry(id_registry, cfg.id_registry, Path.cwd())
        model = _read_local_model(model_path)
        entity_spdx_id = IdRegistrySession(resolved_registry).entity_id(
            model_path.stem, [model_path.stem], "ai_AIPackage"
        )
        enrichment_results = run_enrichers(model, cfg.enrich, model_path.parent)

    exporter = build_model(
        model,
        creation_metadata or cfg.creation_metadata,
        entity_spdx_id=entity_spdx_id,
        provenance=cfg.provenance,
        enrichment_results=enrichment_results,
    )

    sbom_json = exporter.to_json(
        pretty=cfg.pretty,
        describe_relationship=bool(cfg.describe_relationship),
    )

    write_sbom_output(sbom_json, output_path)

    return sbom_json

pitloom.assemble.generate_env_sbom

generate_env_sbom(
    *,
    output_path: Path | None = None,
    creation_metadata: CreationMetadata | None = None,
    pretty: bool | None = None,
    describe_relationship: bool | None = None,
    id_registry: str | Path | IdRegistry | None = None,
    provenance: ProvenanceConfig | None = None,
    offline: bool | None = None,
    content_type_method: str | None = None,
    update_id_registry: bool | None = None,
    max_source_metadata_bytes: int | None = None,
    pitloom_config: PitloomConfig | None = None,
) -> str

Generate a Deployed SPDX 3 SBOM for the current installed environment.

Settings and the registry resolve exactly as :func:~pitloom.assemble.generate_wheel_sbom describes: arguments, then an explicit pitloom_config, then the built-in defaults, with nothing borrowed from the current directory.

content_type_method applies here for one of its two jobs only: it steers whether each installed package's originator enrichment fetches a remote authors file. The per-file contentType half needs file scanning, which reading an installed environment never performs.

Source code in pitloom/assemble/_generators_env.py
 34
 35
 36
 37
 38
 39
 40
 41
 42
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
def generate_env_sbom(
    *,
    output_path: Path | None = None,
    creation_metadata: CreationMetadata | None = None,
    pretty: bool | None = None,
    describe_relationship: bool | None = None,
    id_registry: str | Path | IdRegistry | None = None,
    provenance: ProvenanceConfig | None = None,
    offline: bool | None = None,
    content_type_method: str | None = None,
    update_id_registry: bool | None = None,
    max_source_metadata_bytes: int | None = None,
    pitloom_config: PitloomConfig | None = None,
) -> str:
    """Generate a Deployed SPDX 3 SBOM for the current installed environment.

    Settings and the registry resolve exactly as
    :func:`~pitloom.assemble.generate_wheel_sbom` describes: arguments, then
    an explicit *pitloom_config*, then the built-in defaults, with nothing
    borrowed from the current directory.

    ``content_type_method`` applies here for one of its two jobs only: it
    steers whether each installed package's originator enrichment fetches a
    remote authors file. The per-file ``contentType`` half needs file
    scanning, which reading an installed environment never performs.
    """
    configure_logging()
    cfg = resolve_standalone_config(
        pitloom_config,
        ConfigOverrides(
            provenance=provenance,
            offline=offline,
            content_type_method=content_type_method,
            pretty=pretty,
            describe_relationship=describe_relationship,
            update_id_registry=update_id_registry,
            max_source_metadata_bytes=max_source_metadata_bytes,
        ),
    )
    # Resolved before the expensive read_environment() call (pipdeptree)
    # below -- a declared-but-missing/malformed registry should fail fast,
    # never after paying for a full environment scan first.
    resolved_registry = resolve_registry(id_registry, cfg.id_registry, Path.cwd())
    project_metadata, env_tree = read_environment()

    doc = DocumentModel(
        project=project_metadata,
        creation_metadata=creation_metadata or cfg.creation_metadata,
        ai_models=[],
    )
    exporter = build_deployed(
        doc,
        env_tree=env_tree,
        registry=resolved_registry,
        **cfg.assemble_options,
    )

    _sync_registry(exporter, resolved_registry, cfg.update_id_registry)

    sbom_json = exporter.to_json(
        pretty=cfg.pretty,
        describe_relationship=bool(cfg.describe_relationship),
    )

    write_sbom_output(sbom_json, output_path)

    return sbom_json

pitloom.assemble.enrich_model

enrich_model(
    source: Path | str,
    *,
    output_path: Path | None = None,
    creation_metadata: CreationMetadata | None = None,
    pretty: bool | None = None,
    enrich: bool | None = None,
    project_target: Path | str | None = None,
    id_registry: str | Path | IdRegistry | None = None,
    use_lockfile: bool | None = None,
    pitloom_config: PitloomConfig | None = None,
) -> str

Run enrichment only for a local model file.

Settings come from the arguments, then an explicit pitloom_config, then the built-in defaults; nothing is read from the current directory or from the model file's directory. The registry is id_registry, else the explicit config's id-registry; with project_target, the project's own id-registry key applies instead, since that project is the document the fragment will merge into.

Source code in pitloom/assemble/_model_generator.py
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
def enrich_model(
    source: Path | str,
    *,
    output_path: Path | None = None,
    creation_metadata: CreationMetadata | None = None,
    pretty: bool | None = None,
    enrich: bool | None = None,
    project_target: Path | str | None = None,
    id_registry: str | Path | IdRegistry | None = None,
    use_lockfile: bool | None = None,
    pitloom_config: PitloomConfig | None = None,
) -> str:
    """Run enrichment only for a local model file.

    Settings come from the arguments, then an explicit *pitloom_config*,
    then the built-in defaults; nothing is read from the current directory
    or from the model file's directory. The registry is *id_registry*,
    else the explicit config's ``id-registry``; with *project_target*, the
    project's own ``id-registry`` key applies instead, since that project
    is the document the fragment will merge into.
    """
    configure_logging()
    source_str = str(source)
    if is_huggingface_source(source_str):
        raise ValueError(
            f"'{source_str}' is a Hugging Face source; local enrichment "
            "does not apply there -- Hugging Face model cards are already "
            "parsed natively when generating the SBOM."
        )
    # The CLI passes use_lockfile through, so this is the one layer that
    # settles it, before any work (an sdist project target as a project).
    if project_target is None:
        settle_inert(ENRICH_STANDALONE, source_str, {"use_lockfile": use_lockfile})
    elif is_sdist_archive(Path(project_target)):
        settle_inert(SDIST, project_target, {"use_lockfile": use_lockfile})
    cfg = resolve_standalone_config(pitloom_config, ConfigOverrides(pretty=pretty))

    model_path = Path(source)
    # Registry resolved before _read_local_model()/run_enrichers() below -- a
    # declared-but-missing/malformed registry should fail fast, never
    # after paying for a model read and enrichment first.
    if project_target is None:
        base_doc_identity = None
        resolved_registry = resolve_registry(id_registry, cfg.id_registry, Path.cwd())
    else:
        # The project the fragment merges into, resolved as its base SBOM
        # is: its identity and registry come from the config that SBOM
        # used (an explicit config, else the project's own). Registry
        # resolved right after resolve_project_with_lockfile(), before
        # _doc_identity_of()'s own file walk.
        project_dir = Path(project_target)
        base_metadata, base_config, _ = resolve_project_with_lockfile(
            project_dir, use_lockfile, pitloom_config
        )
        # An sdist's directory is not its project, as in its base SBOM.
        resolved_registry = resolve_registry(
            id_registry,
            base_config.id_registry,
            registry_base_dir(project_dir),
        )
        base_doc_identity = _doc_identity_of(project_dir, base_metadata)

    model = _read_local_model(model_path)
    # Unlike generate_model_sbom()/generate_project_sbom(), the config's
    # enrich setting is NOT an "off by default" gate here: calling
    # enrich_model() at all is itself the opt-in (see
    # test_enrich_model_writes_bare_graph_fragment's docstring). Only an
    # explicit enrich=False turns it back off.
    enrich_config = dataclasses.replace(cfg.enrich, local=enrich is not False)
    results = run_enrichers(model, enrich_config, model_path.parent)

    entity_spdx_id = IdRegistrySession(resolved_registry).entity_id(
        model_path.stem, [model_path.stem], "ai_AIPackage"
    )

    exporter = build_enrichment_fragment(
        model,
        results,
        creation_metadata or cfg.creation_metadata,
        entity_spdx_id=entity_spdx_id,
        base_doc_identity=base_doc_identity,
    )

    fragment_json = exporter.to_json(pretty=cfg.pretty)

    write_sbom_output(fragment_json, output_path)

    return fragment_json

Wheel embedding

pitloom.embed.embed_wheel_sbom

embed_wheel_sbom(
    wheel_path: Path | str,
    *,
    project_dir: Path | str | None = None,
    pitloom_config: PitloomConfig | None = None,
    sbom_path: Path | str | None = None,
    output_path: Path | str | None = None,
    sbom_basename: str | None = None,
    creation_metadata: CreationMetadata | None = None,
    id_registry: str | Path | IdRegistry | None = None,
    overrides: ConfigOverrides | None = None,
    allow_mismatch: bool = False,
    allow_signed_wheel: bool = False,
    file_cache: EmbedFileCache | None = None,
) -> tuple[Path, str, str, tuple[str, ...], bool]

Generate and embed a PEP 770 SBOM into a built Python wheel.

When sbom_path supplies an externally-generated SBOM, its declared subject name/version is cross-checked against the wheel's own .dist-info/METADATA before anything is written -- see :func:_enforce_sbom_name_version. A genuine mismatch raises ValueError unless allow_mismatch. A signed wheel (RECORD.jws/ RECORD.p7s) raises ValueError unless allow_signed_wheel, which removes the signature the embed would invalidate. A Pitloom-generated SBOM (sbom_path unset) is never checked -- it's built from this same wheel_metadata, so it can't diverge.

file_cache: advanced/batch use only -- share one :class:EmbedFileCache across several calls that target the same project_dir/pitloom_config/overrides.build_options (e.g. one wheel per call, in a loop) to resolve project_dir's file list (and run any --allow-build real PEP 517 build) once for the whole batch instead of once per call. A batch logs each ineffective-option warning, the model-usage INFO: hint and each gated-format INFO: once. Left None (the default), this call resolves and cleans up its own file list. When given, this call does NOT clean up -- make every call of the batch inside one with EmbedFileCache() as file_cache: block, whose exit does; a cache used outside its block raises :class:RuntimeError.

Raises:

Type Description
ValueError

The wheel is refused as a whole (:class:~pitloom.core.wheel_dist_info.WheelRefused: not a ZIP archive, a member cannot be read, two members have one name or one holds a NUL, a member of its own .dist-info has a non-conforming name), it has no single own .dist-info, or the SBOM's name/version mismatches (see above). The wheel is left as it was.

OSError

wheel_path cannot be opened (missing, permission denied).

Source code in pitloom/embed.py
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
def embed_wheel_sbom(
    wheel_path: Path | str,
    *,
    project_dir: Path | str | None = None,
    pitloom_config: PitloomConfig | None = None,
    sbom_path: Path | str | None = None,
    output_path: Path | str | None = None,
    sbom_basename: str | None = None,
    creation_metadata: CreationMetadata | None = None,
    id_registry: str | Path | IdRegistry | None = None,
    overrides: ConfigOverrides | None = None,
    allow_mismatch: bool = False,
    allow_signed_wheel: bool = False,
    file_cache: EmbedFileCache | None = None,
) -> tuple[Path, str, str, tuple[str, ...], bool]:
    """Generate and embed a PEP 770 SBOM into a built Python wheel.

    When *sbom_path* supplies an externally-generated SBOM, its declared
    subject name/version is cross-checked against the wheel's own
    ``.dist-info/METADATA`` before anything is written -- see
    :func:`_enforce_sbom_name_version`. A genuine mismatch raises
    ``ValueError`` unless *allow_mismatch*. A signed wheel (``RECORD.jws``/
    ``RECORD.p7s``) raises ``ValueError`` unless *allow_signed_wheel*, which
    removes the signature the embed would invalidate. A Pitloom-generated SBOM
    (*sbom_path* unset) is never checked -- it's built from this same
    *wheel_metadata*, so it can't diverge.

    *file_cache*: advanced/batch use only -- share one
    :class:`EmbedFileCache` across several calls that target the same
    *project_dir*/*pitloom_config*/``overrides.build_options`` (e.g. one
    wheel per call, in a loop) to resolve *project_dir*'s file list (and
    run any ``--allow-build`` real PEP 517 build) once for the whole
    batch instead of once per call. A batch logs each ineffective-option
    warning, the model-usage ``INFO:`` hint and each gated-format ``INFO:``
    once. Left ``None`` (the default), this
    call resolves and cleans up its own file list. When given, *this*
    call does NOT clean up -- make every call of the batch inside one
    ``with EmbedFileCache() as file_cache:`` block, whose exit does; a
    cache used outside its block raises :class:`RuntimeError`.

    Raises:
        ValueError: The wheel is refused as a whole
            (:class:`~pitloom.core.wheel_dist_info.WheelRefused`: not a ZIP
            archive, a member cannot be read, two members have one name or
            one holds a NUL, a member of its own ``.dist-info`` has a
            non-conforming name),
            it has no single own ``.dist-info``, or the SBOM's name/version
            mismatches (see above). The wheel is left as it was.
        OSError: *wheel_path* cannot be opened (missing, permission denied).
    """
    configure_logging()
    if sbom_basename:
        sbom_basename = sbom_base_name(sbom_basename, "--sbom-basename")
    require_wheel_path(wheel_path)
    # Not resolved: a symlink's name is the one judged, and the one the
    # embed reports; ``embed_sbom_in_wheel`` resolves to write.
    wheel_obj = Path(wheel_path).absolute()
    eff_overrides = overrides if overrides is not None else ConfigOverrides()
    if sbom_path is None:
        # Fail on a declared-but-bad registry before the wheel is ever
        # read -- a real ``read_wheel()`` opens/parses the archive, work
        # worth skipping when this run cannot proceed anyway. Only when
        # *sbom_path* is unset: with an external SBOM,
        # ``_generate_embed_sbom_json``'s own early-return branch never
        # touches the registry at all, so there is nothing to resolve here.
        # The resolved ``IdRegistry`` (or ``None``) is fed back in as *this
        # call's own* ``id_registry`` below, so
        # ``_generate_embed_sbom_json``'s own ``resolve_registry()`` call
        # short-circuits on the already-resolved instance rather than
        # loading the file again.
        id_registry = _resolve_embed_registry(project_dir, pitloom_config, id_registry)
    # Before the SBOM is generated: that may run a build.
    refuse_unembeddable_wheel(wheel_obj, allow_signed_wheel)
    wheel_metadata, _ = read_wheel(wheel_obj)

    sbom_json, eff_basename = _generate_embed_sbom_json(
        wheel_metadata,
        wheel_path=wheel_obj,
        project_dir=project_dir,
        pitloom_config=pitloom_config,
        sbom_path=sbom_path,
        sbom_basename=sbom_basename,
        creation_metadata=creation_metadata,
        id_registry=id_registry,
        overrides=eff_overrides,
        file_cache=file_cache,
    )
    wheel_name, wheel_version = wheel_identity(wheel_metadata)
    if sbom_path is not None:
        _enforce_sbom_name_version(
            wheel_obj.name,
            wheel_name,
            wheel_version,
            sbom_json,
            allow_mismatch=allow_mismatch,
        )
    # The identity just read goes down, so the embed does not read (and warn
    # about) the same METADATA again.
    res_path, arcname, removed_arcnames, timestamp_floored = embed_sbom_in_wheel(
        wheel_obj,
        sbom_json,
        sbom_filename=embed_filename(eff_basename),
        identity=(wheel_name, wheel_version),
        allow_signed_wheel=allow_signed_wheel,
    )

    write_sbom_output(sbom_json, output_path)

    return res_path, arcname, sbom_json, removed_arcnames, timestamp_floored

pitloom.embed.embed_sbom_in_wheel

embed_sbom_in_wheel(
    wheel_path: Path | str,
    sbom_content: str | bytes,
    *,
    sbom_filename: str | None = None,
    identity: tuple[str | None, str | None] | None = None,
    allow_signed_wheel: bool = False,
) -> tuple[Path, str, tuple[str, ...], bool]

Embed an SPDX 3 SBOM into a built wheel archive (PEP 770).

A wheel carrying RECORD.jws/RECORD.p7s is refused unless allow_signed_wheel: the embed rewrites RECORD, so the signature would no longer verify, and it is removed (the removed names are in the result, as for a stale SBOM).

identity is the wheel's declared (name, version), where the caller has already read them from its METADATA: the default file name is made from it, and METADATA is not read, or warned about, again.

Raises:

Type Description
FileNotFoundError

wheel_path doesn't exist.

ValueError

sbom_content is empty, or the wheel's content is bad (missing/ambiguous .dist-info), or the wheel is refused (:class:~pitloom.core.wheel_dist_info.WheelRefused): not a ZIP archive, a member cannot be read (damaged, encrypted, unsupported, badly named), two members have one name or one holds a NUL, or a member of its own .dist-info has a non-conforming name, or it is signed and not allow_signed_wheel. The wheel is left as it was.

OSError

An environment problem opening wheel_path (permission denied, a transient I/O error) -- kept as its own exception type, not folded into ValueError.

Source code in pitloom/_embed_wheel.py
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
def embed_sbom_in_wheel(
    wheel_path: Path | str,
    sbom_content: str | bytes,
    *,
    sbom_filename: str | None = None,
    identity: tuple[str | None, str | None] | None = None,
    allow_signed_wheel: bool = False,
) -> tuple[Path, str, tuple[str, ...], bool]:
    """Embed an SPDX 3 SBOM into a built wheel archive (PEP 770).

    A wheel carrying ``RECORD.jws``/``RECORD.p7s`` is refused unless
    *allow_signed_wheel*: the embed rewrites ``RECORD``, so the signature
    would no longer verify, and it is removed (the removed names are in the
    result, as for a stale SBOM).

    *identity* is the wheel's declared (name, version), where the caller has
    already read them from its ``METADATA``: the default file name is made
    from it, and ``METADATA`` is not read, or warned about, again.

    Raises:
        FileNotFoundError: *wheel_path* doesn't exist.
        ValueError: *sbom_content* is empty, or the wheel's content is bad
            (missing/ambiguous ``.dist-info``), or the wheel is refused
            (:class:`~pitloom.core.wheel_dist_info.WheelRefused`): not a ZIP
            archive, a member cannot be read (damaged, encrypted,
            unsupported, badly named), two members have one name or one
            holds a NUL, or a member of its own ``.dist-info`` has a
            non-conforming name, or it is signed and not *allow_signed_wheel*.
            The wheel is left as it was.
        OSError: An environment problem opening *wheel_path* (permission
            denied, a transient I/O error) -- kept as its own exception
            type, not folded into ``ValueError``.
    """
    configure_logging()
    require_wheel_path(wheel_path)
    wheel_obj = Path(wheel_path).resolve()
    if not wheel_obj.exists():
        raise FileNotFoundError(f"Wheel file not found: {wheel_obj}")

    sbom_bytes = (
        sbom_content.encode("utf-8") if isinstance(sbom_content, str) else sbom_content
    )
    if not sbom_bytes.strip():
        raise ValueError("SBOM content cannot be empty")

    orig_mode = wheel_obj.stat().st_mode if wheel_obj.exists() else None

    # Every use of the file name (which ``.dist-info`` is the wheel's own, what
    # a refusal names) is of the name as given; the resolved path, which a
    # symlink points elsewhere, is for the directory and the write only.
    named = Path(wheel_path).absolute()
    with open_wheel_zip(wheel_obj, name=named) as original_zf:
        members = wheel_members(original_zf, named.name)
        dist_info = _find_dist_info_prefix(original_zf, named, members=members)
        _refuse_unembeddable(named.name, dist_info, members, allow_signed_wheel)
        plan = _plan_embed(
            original_zf, dist_info, members, sbom_filename, sbom_bytes, identity
        )
        temp_path = _rewrite_wheel_archive(
            wheel_obj,
            original_zf,
            plan.sbom_arcname,
            sbom_bytes,
            plan.record_arcname,
            plan.new_record_bytes,
            plan.timestamp,
            plan.stale_arcnames,
            name=named.name,
        )

    try:
        os.replace(temp_path, wheel_obj)
        if orig_mode is not None:
            try:
                os.chmod(wheel_obj, orig_mode)
            except OSError:
                pass
    finally:
        if temp_path.exists():
            temp_path.unlink()

    return (
        named,  # as given: later steps judge its name
        plan.sbom_arcname,
        tuple(sorted(plan.stale_arcnames)),
        plan.timestamp_floored,
    )

pitloom.embed.ConfigOverrides dataclass

ConfigOverrides(
    provenance: ProvenanceConfig | None = None,
    enrich: bool | None = None,
    extract_file_header: bool | None = None,
    scan_model_usage: bool | None = None,
    content_type: bool | None = None,
    content_type_method: str | None = None,
    offline: bool | None = None,
    pretty: bool | None = None,
    describe_relationship: bool | None = None,
    update_id_registry: bool | None = None,
    max_source_metadata_bytes: int | None = None,
    trust_wheel_model: bool | None = None,
    build_options: BuildOptions = BuildOptions(),
)

Per-run overrides layered onto a project's [tool.pitloom] config.

Every field defaults to None, meaning "not given, defer to the config"; any other value wins, including one equal to the built-in default. Each maps to the PitloomConfig field of the same name, except enrich -> enrich_local and content_type -> content_type_enabled.

Attributes:

Name Type Description
provenance ProvenanceConfig | None

Replaces the config's whole provenance settings, not field by field: a field left at its default resets the config's value.

max_source_metadata_bytes int | None

Overrides that one provenance field and leaves the others as the config (or provenance) set them. Applied after provenance, so it wins over that object's own value. Checked by :func:~pitloom.core.provenance.require_max_source_metadata_bytes: 0 or at least 8, else ValueError.

pretty bool | None

The embed path (embed_wheel_sbom(overrides=...)) always writes JCS-canonical JSON and warns that a given value has no effect, as it does for describe_relationship and update_id_registry (see :data:pitloom.core.inert_options.INERT).

trust_wheel_model bool | None

--trust-wheel-model: read a wheel's AI model files with every format reader, including those gated in a wheel. Like build_options, no [tool.pitloom] cascade (a config must not opt in) and :func:apply_overrides never touches it; only a standalone wheel embed reads it, and any other embed warns.

build_options BuildOptions

--allow-build and its companion flags (see :class:~pitloom.core.build_options.BuildOptions). Unlike every other field here, deliberately has no [tool.pitloom] cascade to defer to, and is read by embed-wheel alone -- :func:apply_overrides never touches it, and generate_project_sbom() takes its own separate build_options parameter rather than reading this one. Threaded into _build_sbom_from_project_and_wheel()'s own project-dir rescan, whose only use for the resulting file list is layering content-type/file-header extras onto the wheel's already-known files (see that function's own comment on discarding the rescan's merkle_root/digests) -- so on a project whose backend has no static discovery module (or whose static discovery fails), enabling this runs a full, real, potentially slow PEP 517 build purely to compute those extras more accurately, not to learn the file list itself (the wheel's own read_wheel() result already has that). Deliberate: this is the only way embed-wheel avoids silently staying stuck on the Hatchling-heuristic rescan for such a project's content-type/header extras.

Fragment merging

pitloom.assemble.generate_merged_sbom

generate_merged_sbom(
    fragments_dir: Path | str,
    *,
    output_path: Path | str | None = None,
    pretty: bool = True,
) -> str

Merge every *.json SPDX 3 fragment in fragments_dir into one SBOM, as loom merge does, and return it as JSON-LD.

The output is one SpdxDocument rooted at what the fragments' own envelopes rooted; equal elements unify as in a project build with fragments. The same fragments give the same bytes, whatever the directory or the order the files were written in.

Raises:

Type Description
FileNotFoundError

fragments_dir does not exist.

ValueError

it has no *.json file.

FragmentMergeError

the merge left a dangling reference, a root included.

Source code in pitloom/assemble/_generators_merge.py
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
def generate_merged_sbom(
    fragments_dir: Path | str,
    *,
    output_path: Path | str | None = None,
    pretty: bool = True,
) -> str:
    """Merge every ``*.json`` SPDX 3 fragment in *fragments_dir* into one
    SBOM, as ``loom merge`` does, and return it as JSON-LD.

    The output is one ``SpdxDocument`` rooted at what the fragments' own
    envelopes rooted; equal elements unify as in a project build with
    fragments. The same fragments give the same bytes, whatever the
    directory or the order the files were written in.

    Raises:
        FileNotFoundError: *fragments_dir* does not exist.
        ValueError: it has no ``*.json`` file.
        pitloom.assemble.FragmentMergeError: the merge left a dangling
            reference, a root included.
    """
    configure_logging()
    fragments_dir = Path(fragments_dir)
    if not fragments_dir.exists():
        raise FileNotFoundError(f"fragments directory not found: {fragments_dir}")
    files = fragment_files(fragments_dir)
    if not files:
        raise ValueError(f"no JSON fragment files found in {fragments_dir}")
    exporter = new_merge_document(fragments_dir, files)
    merge_fragments(
        fragments_dir,
        [FragmentConfig(path=f) for f in files],
        exporter,
        adopt_fragment_roots=True,
    )
    finish_merge_document(exporter)
    sbom_json = exporter.to_json(pretty=pretty)
    write_sbom_output(sbom_json, output_path)
    return sbom_json

pitloom.assemble.merge_fragments

merge_fragments(
    project_dir: Path,
    fragments: list[FragmentConfig],
    exporter: Spdx3JsonExporter,
    *,
    adopt_fragment_roots: bool = False,
) -> None

Load SPDX 3 JSON-LD fragment files and merge them into the exporter.

The main document's profileConformance gains what the merged fragments' envelopes declared. With adopt_fragment_roots (loom merge), its rootElement also gains what those envelopes rooted, after unification; a project keeps its own roots. Every root goes through the dangling-reference check below.

A fragment whose SpdxDocument id is the exporter's own document (an earlier SBOM of the same project) is skipped with a WARNING:, as a missing one is.

Raises :class:FragmentMergeError if any required=True fragment (see :class:~pitloom.core.config.FragmentConfig) is missing, couldn't be read or is the document itself, or if the merge leaves the graph referentially broken (see :func:_raise_on_dangling_references) -- the latter is skipped when fragments is empty or none of it could be ingested, since there is then nothing new whose references could be dangling.

Source code in pitloom/assemble/spdx3/fragments.py
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
def merge_fragments(
    project_dir: Path,
    fragments: list[FragmentConfig],
    exporter: Spdx3JsonExporter,
    *,
    adopt_fragment_roots: bool = False,
) -> None:
    """Load SPDX 3 JSON-LD fragment files and merge them into the exporter.

    The main document's ``profileConformance`` gains what the merged
    fragments' envelopes declared. With *adopt_fragment_roots*
    (``loom merge``), its ``rootElement`` also gains what those envelopes
    rooted, after unification; a project keeps its own roots. Every root
    goes through the dangling-reference check below.

    A fragment whose ``SpdxDocument`` id is the exporter's own document
    (an earlier SBOM of the same project) is skipped with a ``WARNING:``,
    as a missing one is.

    Raises :class:`FragmentMergeError` if any ``required=True`` fragment
    (see :class:`~pitloom.core.config.FragmentConfig`) is missing, couldn't
    be read or is the document itself, or if the merge leaves the graph
    referentially broken (see :func:`_raise_on_dangling_references`) -- the
    latter is skipped when *fragments* is empty or none of it could be
    ingested, since there is then nothing new whose references could be
    dangling.
    """
    configure_logging()
    index = _MergeIndex(exporter)
    events: _UnificationEvents = {}
    fragment_imports: list[spdx3.ExternalMap] = []
    merged_any = False
    unmet_required: list[str] = []
    roots: list[str] = []
    profiles: set[str] = set()
    main_doc = _find_main_document(exporter.object_set)
    main_doc_id = str(main_doc.spdxId) if main_doc and main_doc.spdxId else None

    for frag in fragments:
        loaded = _load_fragment(
            Path(frag.base_dir or project_dir) / frag.path, frag.required, main_doc_id
        )
        if loaded is None:
            if frag.required:
                unmet_required.append(frag.path)
            continue
        fragment_set, frag_doc_id = loaded
        if frag_doc_id and frag_doc_id not in {
            m.externalSpdxId for m in fragment_imports
        }:
            fragment_imports.append(
                spdx3.ExternalMap(externalSpdxId=frag_doc_id, locationHint=frag.path)
            )
        profiles |= _declared_profiles(fragment_set)
        roots.extend(_merge_fragment_set(fragment_set, index, frag.path, events))
        merged_any = True

    _dedupe_relationships(exporter)

    if main_doc is not None:
        _update_profile_conformance(main_doc, exporter, profiles)
        _add_fragment_imports(main_doc, fragment_imports)
        _emit_unification_annotations(events, main_doc, exporter)
        _add_model_sbom(main_doc, exporter)
        if adopt_fragment_roots:
            main_doc.rootElement = sorted({*roots, *(main_doc.rootElement or [])})

    # Checked unconditionally -- NOT gated by `merged_any`. A required
    # fragment that's missing is exactly the scenario most likely to leave
    # merged_any False (nothing else may have merged either), so gating
    # this on merged_any would silently suppress the one case it exists to
    # catch. Checked before the dangling-references raise below: a missing
    # required fragment is usually the *root cause* of any dangling
    # references the rest of the graph would otherwise show (other
    # fragments' relationships pointing at elements the missing required
    # fragment was supposed to supply) -- report the root cause, not the
    # downstream symptom, when both would otherwise fire in the same run.
    if unmet_required:
        raise FragmentMergeError(
            f"{len(unmet_required)} required fragment(s) could not be "
            f"merged: {', '.join(unmet_required)} -- check the configured "
            "path(s) under [tool.pitloom.fragment]"
        )

    if merged_any:
        _raise_on_dangling_references(exporter)

pitloom.assemble.project_document_id

project_document_id(project_dir: Path) -> str

The SpdxDocument id a loom project build of project_dir gives (without --allow-build), resolved as :func:_doc_identity_of resolves it; loom fragment list compares fragments with it.

Source code in pitloom/assemble/_model_generator.py
113
114
115
116
117
118
119
120
def project_document_id(project_dir: Path) -> str:
    """The ``SpdxDocument`` id a ``loom project`` build of *project_dir*
    gives (without ``--allow-build``), resolved as :func:`_doc_identity_of`
    resolves it; ``loom fragment list`` compares fragments with it."""
    configure_logging()
    metadata, _config, _path = resolve_project_with_lockfile(project_dir, None)
    doc_name, doc_uuid = _doc_identity_of(project_dir, metadata)
    return generate_spdx_id("SpdxDocument", doc_name=doc_name, doc_uuid=doc_uuid)

pitloom.assemble.FragmentMergeError

Raised when merging fragments would produce a referentially-broken SBOM -- a Relationship/Annotation endpoint or a rootElement that resolves to neither an object in the merged graph nor a declared external reference. Merging must not silently succeed in that case; see :func:_raise_on_dangling_references.

Tracking decorator

loom.run is the Run class below (run = Run) -- use it as a decorator or a context manager, as shown on the Python API page.

pitloom.loom.Run

Run(
    output_file: str | Path,
    pretty: bool = False,
    creation_metadata: CreationMetadata | None = None,
    id_registry: str | Path | IdRegistry | None = None,
)

Context manager and decorator for capturing SPDX fragments.

Each Run is a single recording session that weaves metadata about a model and its datasets into an SBOM fragment.

Can be used as a context manager::

with loom.run("fragments/train.spdx3.json") as run:
    run.set_model("my-model")
    run.add_dataset("train.txt")
    run.add_validation_dataset("valid.txt")
    # ... training code ...
    run.set_model_hyperparameters({"lr": "0.1", "epoch": "5"})

Or as a function decorator::

@loom.run("fragments/preprocess.spdx3.json")
def preprocess():
    loom.add_input_dataset("rawdata/neg.txt")
    loom.add_output_dataset("data/train.txt",
                            data_preprocessing=["tokenization"])

The fragment's SPDX CreationInfo is configurable on par with the CLI and Hatchling build hook: pass a CreationMetadata to name a creator (a person, organization, or automated agent), or override the tool, timestamp, and comment. With none given, the fragment records the SoftwareAgent "Pitloom" (createdBy) and Tool "Pitloom" (createdUsing) of an unattended run::

loom.run(
    "fragments/train.spdx3.json",
    creation_metadata=CreationMetadata(
        creators=[Creator(name="Alice", type="person")]
    ),
)

Parameters:

Name Type Description Default
output_file str | Path

Path to write the SBOM fragment to.

required
pretty bool

Indent the JSON output with 2 spaces when True.

False
creation_metadata CreationMetadata | None

Creator, tool, timestamp, and comment overrides for the fragment's CreationInfo. See CreationMetadata for all fields. When None (default), the comment defaults to an auto-generated note identifying the loom SDK and its version, and the creator defaults to the SoftwareAgent "Pitloom".

None
id_registry str | Path | IdRegistry | None

A pitloom.id_registry.IdRegistry, a path to a registry JSON file (relative to the current working directory), or None (default) for no registry -- never searched for. Consulted read-only: datasets, the model, and the generating script all get the registered spdxId when one exists for them, so independently generated fragments can be unified at merge time without name-based matching.

None
Source code in pitloom/loom.py
81
82
83
84
85
86
87
88
89
90
91
92
def __init__(
    self,
    output_file: str | Path,
    pretty: bool = False,
    creation_metadata: CreationMetadata | None = None,
    id_registry: str | Path | IdRegistry | None = None,
):
    self.output_file = str(output_file)
    self.pretty = pretty
    self.creation_metadata = creation_metadata or CreationMetadata()
    self.id_registry = id_registry
    self.previous_run: _ActiveRun | None = None

pitloom.loom.set_model

set_model(
    name: str,
    model_type: str | None = None,
    hyperparameters: dict[str, str] | None = None,
    generated: bool | None = None,
) -> None

Set the name of the AI model being trained in the current run.

Source code in pitloom/loom.py
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
def set_model(
    name: str,
    model_type: str | None = None,
    hyperparameters: dict[str, str] | None = None,
    generated: bool | None = None,
) -> None:
    """Set the name of the AI model being trained in the current run."""
    if _active_run is None:
        raise RuntimeError(
            "No active loom.run() found. Please use `loom.set_model()` inside a "
            "`with pitloom.loom.run():` block or decorated function."
        )
    _active_run.set_model(
        name,
        model_type=model_type,
        hyperparameters=hyperparameters,
        generated=generated,
    )

pitloom.loom.use_model

use_model(
    name: str,
    model_type: str | None = None,
    hyperparameters: dict[str, str] | None = None,
) -> None

Explicitly declare an AI model consumed by the current run (for inference).

Source code in pitloom/loom.py
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
def use_model(
    name: str,
    model_type: str | None = None,
    hyperparameters: dict[str, str] | None = None,
) -> None:
    """Explicitly declare an AI model consumed by the current run (for inference)."""
    if _active_run is None:
        raise RuntimeError(
            "No active loom.run() found. Please use `loom.use_model()` inside a "
            "`with pitloom.loom.run():` block or decorated function."
        )
    _active_run.use_model(
        name,
        model_type=model_type,
        hyperparameters=hyperparameters,
    )

pitloom.loom.set_model_hyperparameters

set_model_hyperparameters(
    hyperparameters: dict[str, str],
) -> None

Update the active model with hyperparameters captured after training.

Source code in pitloom/loom.py
171
172
173
174
175
176
177
178
def set_model_hyperparameters(hyperparameters: dict[str, str]) -> None:
    """Update the active model with hyperparameters captured after training."""
    if _active_run is None:
        raise RuntimeError(
            "No active loom.run() found. Please use `loom.set_model_hyperparameters()`"
            " inside a `with pitloom.loom.run():` block or decorated function."
        )
    _active_run.set_model_hyperparameters(hyperparameters)

pitloom.loom.add_dataset

add_dataset(name: str, dataset_type: str = 'text') -> None

Add a dataset utilized by the AI model in the current run.

Source code in pitloom/loom.py
181
182
183
184
185
186
187
188
def add_dataset(name: str, dataset_type: str = "text") -> None:
    """Add a dataset utilized by the AI model in the current run."""
    if _active_run is None:
        raise RuntimeError(
            "No active loom.run() found. Please use `loom.add_dataset()` inside a "
            "`with pitloom.loom.run():` block or decorated function."
        )
    _active_run.add_dataset(name, dataset_type)

pitloom.loom.add_validation_dataset

add_validation_dataset(
    name: str, dataset_type: str = "text"
) -> None

Add a validation/test dataset in the current run.

Source code in pitloom/loom.py
191
192
193
194
195
196
197
198
199
def add_validation_dataset(name: str, dataset_type: str = "text") -> None:
    """Add a validation/test dataset in the current run."""
    if _active_run is None:
        raise RuntimeError(
            "No active loom.run() found. Please use "
            "`loom.add_validation_dataset()` inside a "
            "`with pitloom.loom.run():` block or decorated function."
        )
    _active_run.add_validation_dataset(name, dataset_type)

pitloom.loom.add_input_dataset

add_input_dataset(
    name: str, dataset_type: str = "text"
) -> None

Declare a raw/source dataset consumed by a preprocessing step.

Source code in pitloom/loom.py
202
203
204
205
206
207
208
209
210
def add_input_dataset(name: str, dataset_type: str = "text") -> None:
    """Declare a raw/source dataset consumed by a preprocessing step."""
    if _active_run is None:
        raise RuntimeError(
            "No active loom.run() found. Please use "
            "`loom.add_input_dataset()` inside a "
            "`with pitloom.loom.run():` block or decorated function."
        )
    _active_run.add_input_dataset(name, dataset_type)

pitloom.loom.add_output_dataset

add_output_dataset(
    name: str,
    dataset_type: str = "text",
    data_preprocessing: list[str] | None = None,
    input_datasets: list[str] | None = None,
) -> None

Declare a derived/processed dataset produced by a preprocessing step.

Source code in pitloom/loom.py
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
def add_output_dataset(
    name: str,
    dataset_type: str = "text",
    data_preprocessing: list[str] | None = None,
    input_datasets: list[str] | None = None,
) -> None:
    """Declare a derived/processed dataset produced by a preprocessing step."""
    if _active_run is None:
        raise RuntimeError(
            "No active loom.run() found. Please use "
            "`loom.add_output_dataset()` inside a "
            "`with pitloom.loom.run():` block or decorated function."
        )
    _active_run.add_output_dataset(
        name,
        dataset_type,
        data_preprocessing=data_preprocessing,
        input_datasets=input_datasets,
    )

Creation metadata

pitloom.core.creation.CreationMetadata dataclass

CreationMetadata(
    creators: list[Creator] = list(),
    tools: list[Tool] | None = None,
    creation_datetime: str | None = None,
    creation_comment: str | None = None,
    build_datetime: str | None = None,
)

Metadata describing who and what generated an SBOM.

Pitloom's own model for creation provenance -- distinct from, but mapping onto, SPDX 3 CreationInfo: each creator becomes an Agent in createdBy (Person, Organization, SoftwareAgent, or the generic Agent -- see Creator.type), and each tool becomes a Tool in createdUsing. When no creator is named, the assembler records the automated SoftwareAgent "Pitloom" in createdBy -- Pitloom acting on its own -- rather than inventing a Person.

Attributes:

Name Type Description
creators list[Creator]

Named creators, in order. When empty (default), no named creator is asserted and the assembler emits the SoftwareAgent "Pitloom" in createdBy instead. When multiple creators are supplied, each becomes its own Agent; the SPDX 3 suppliedBy of the main package (single-valued) is set to the first named creator.

tools list[Tool] | None

Creation tools, in order. None (default) means the default single Tool "Pitloom". An empty list suppresses createdUsing entirely (matches --no-creation-tool). A non-empty list emits one Tool per entry.

creation_datetime str | None

ISO 8601 string for the creation timestamp. Full ISO forms are accepted (e.g. offsets and fractional seconds). Pitloom preserves input precision internally and normalises to SPDX 3 DateTime (YYYY-MM-DDThh:mm:ssZ) only at export time. When None, the assembler falls back to SOURCE_DATE_EPOCH (reproducible-builds.org) if set, else the current UTC time -- see :func:resolve_source_date_epoch.

creation_comment str | None

Optional comment to include on the SPDX CreationInfo element. Callers (CLI, Hatchling build hook) set this to a static description of the invocation channel, e.g. "Generated via Pitloom CLI".

build_datetime str | None

ISO 8601 string for when the artifact was built (e.g. the moment the Hatchling hook fires). When set, the assembler records it as builtTime on the main software_Package element. When None (default), builtTime is omitted from the SBOM.

pitloom.core.creation.Creator dataclass

Creator(
    name: str,
    type: str = "person",
    email: str | None = None,
)

A single named creator, mapping onto an SPDX 3 Agent.

Attributes:

Name Type Description
name str

Display name of the person or organisation that initiated the SBOM generation.

type str

Agent subclass: "person" (default), "organization", "software-agent", or the generic "agent". All four are valid createdBy types per the SPDX 3 spec. Validated (and normalised to lower-case, stripped) in __post_init__.

email str | None

E-mail address of the creator. Recorded as an email external identifier on the creator Agent.

Raises:

Type Description
ValueError

If name is empty or whitespace-only, or if type (after normalisation) is not one of :data:VALID_CREATOR_TYPES.

pitloom.core.creation.Tool dataclass

Tool(name: str)

A single creation tool, mapping onto an SPDX 3 Tool.

Attributes:

Name Type Description
name str

Name of the tool. A tool literally named "Pitloom" gets a version summary appended automatically.

Raises:

Type Description
ValueError

If name is empty or whitespace-only.

pitloom.core.creation.VALID_CREATOR_TYPES module-attribute

VALID_CREATOR_TYPES: frozenset[str] = frozenset(
    {"person", "organization", "software-agent", "agent"}
)

pitloom.core.creation.resolve_source_date_epoch

resolve_source_date_epoch() -> datetime | None

Read SOURCE_DATE_EPOCH as a UTC datetime, per reproducible-builds.org.

https://reproducible-builds.org/specs/source-date-epoch/: a single environment variable the whole build toolchain honours for every embedded timestamp, so an operator opts a build into reproducibility once rather than configuring each tool individually. Shared by every Pitloom timestamp that should respect it (SBOM created, the Hatchling build hook's builtTime, embedded wheel ZIP entries) so the parsing/validation rule -- and what counts as invalid -- stays in one place.

Returns:

Type Description
datetime | None

The resolved UTC datetime, or None when unset or invalid

datetime | None

(logged as a warning) -- callers fall back to their own default.

Source code in pitloom/core/creation.py
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
def resolve_source_date_epoch() -> datetime | None:
    """Read ``SOURCE_DATE_EPOCH`` as a UTC datetime, per reproducible-builds.org.

    <https://reproducible-builds.org/specs/source-date-epoch/>: a single
    environment variable the whole build toolchain honours for every
    embedded timestamp, so an operator opts a build into reproducibility
    once rather than configuring each tool individually. Shared by every
    Pitloom timestamp that should respect it (SBOM ``created``, the
    Hatchling build hook's ``builtTime``, embedded wheel ZIP entries) so
    the parsing/validation rule -- and what counts as invalid -- stays in
    one place.

    Returns:
        The resolved UTC datetime, or ``None`` when unset or invalid
        (logged as a warning) -- callers fall back to their own default.
    """
    raw = os.environ.get("SOURCE_DATE_EPOCH")
    if not raw:
        return None
    try:
        return datetime.fromtimestamp(int(raw), tz=timezone.utc)
    except (ValueError, OverflowError, OSError) as exc:
        log.warning("Ignoring invalid SOURCE_DATE_EPOCH: %s", exc)
        return None

Provenance configuration

pitloom.core.provenance.ProvenanceConfig dataclass

ProvenanceConfig(
    format: str = "both",
    schema: str = DEFAULT_PROVENANCE_SCHEMA,
    detail: str = "minimal",
    preserve_source_metadata: str = "auto",
    max_source_metadata_bytes: int = 0,
)

Configuration settings for SPDX 3 metadata provenance annotations.

Attributes:

Name Type Description
format str

How to record metadata provenance ("annotation", "comment", "both").

schema str

Schema id for provenance Annotations.

detail str

Provenance detail level ("minimal", "full").

preserve_source_metadata str

How to preserve source metadata ("auto", "always", "never").

max_source_metadata_bytes int

Byte budget for the serialized artifact-metadata Annotation.statement; 0 (default) means unlimited, below :data:MIN_SOURCE_METADATA_BYTES is an error. See :func:require_max_source_metadata_bytes.

__post_init__

__post_init__() -> None

Reject an invalid budget, however it got here; keep it as a plain int.

Source code in pitloom/core/provenance.py
133
134
135
136
137
138
139
def __post_init__(self) -> None:
    """Reject an invalid budget, however it got here; keep it as a plain int."""
    object.__setattr__(
        self,
        "max_source_metadata_bytes",
        require_max_source_metadata_bytes(self.max_source_metadata_bytes),
    )

pitloom.core.provenance.require_max_source_metadata_bytes

require_max_source_metadata_bytes(
    value: object, label: str = "max_source_metadata_bytes"
) -> int

value as an int when it is 0 (no cap) or a budget of at least :data:MIN_SOURCE_METADATA_BYTES, the smallest annotation that holds any metadata. A smaller one is an error, never a silent "unlimited".

The one validator for every surface that takes the budget: the config key, --max-source-metadata-bytes, ConfigOverrides and the library kwargs. label names the setting in the error, e.g. [tool.pitloom.provenance] 'max-source-metadata-bytes'; "" leaves the message to a caller that names the setting itself. Any :func:operator.index-able integer (a numpy.int64) passes; a bool or a float does not.

Raises:

Type Description
ValueError

value is not an integer, is negative, or is below the minimum without being 0.

Source code in pitloom/core/provenance.py
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
def require_max_source_metadata_bytes(
    value: object, label: str = "max_source_metadata_bytes"
) -> int:
    """*value* as an ``int`` when it is ``0`` (no cap) or a budget of at least
    :data:`MIN_SOURCE_METADATA_BYTES`, the smallest annotation that holds any
    metadata. A smaller one is an error, never a silent "unlimited".

    The one validator for every surface that takes the budget: the config
    key, ``--max-source-metadata-bytes``, ``ConfigOverrides`` and the library
    kwargs. *label* names the setting in the error, e.g.
    ``[tool.pitloom.provenance] 'max-source-metadata-bytes'``; ``""`` leaves
    the message to a caller that names the setting itself. Any
    :func:`operator.index`-able integer (a ``numpy.int64``) passes; a ``bool``
    or a ``float`` does not.

    Raises:
        ValueError: *value* is not an integer, is negative, or is below the
            minimum without being ``0``.
    """
    try:
        if isinstance(value, bool):
            raise TypeError
        number = operator.index(value)  # type: ignore[arg-type]
    except TypeError:
        raise ValueError(
            f"{label} must be an integer, got {type(value).__name__}: {value!r}".strip()
        ) from None
    if number != 0 and number < MIN_SOURCE_METADATA_BYTES:
        raise ValueError(
            f"{label} must be 0 (unlimited) or at least "
            f"{MIN_SOURCE_METADATA_BYTES} bytes, got {number}".strip()
        )
    return number

ID registry

pitloom.id_registry.IdRegistry

IdRegistry(
    namespace: str,
    files: dict[str, FileEntry] | None = None,
    entities: dict[tuple[str, str], EntityEntry]
    | None = None,
    path: Path | None = None,
)

A Loom ID registry: a stable file/entity -> SPDX ID registry, persisted as JSON.

Source code in pitloom/id_registry/_registry.py
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
def __init__(
    self,
    namespace: str,
    files: dict[str, FileEntry] | None = None,
    entities: dict[tuple[str, str], EntityEntry] | None = None,
    path: Path | None = None,
) -> None:
    self.namespace = namespace
    self.files: dict[str, FileEntry] = files if files is not None else {}
    #: Keyed by ``(type_name, name)`` -- a directory and a package (or
    #: any two entities of different SPDX 3 types) sharing one name are
    #: distinct entries, never overwriting each other.
    self.entities: dict[tuple[str, str], EntityEntry] = (
        entities if entities is not None else {}
    )
    self.path = path

generate

generate(paths: list[Path], project_root: Path) -> None

(Re-)index files under paths into this registry.

Source code in pitloom/id_registry/_registry.py
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
def generate(self, paths: list[Path], project_root: Path) -> None:
    """(Re-)index files under *paths* into this registry."""
    # pylint: disable=import-outside-toplevel,cyclic-import
    from pitloom.extract.ai_model import AiModelFormat, detect_ai_model_format
    from pitloom.extract.scanner import is_model_candidate_name

    for file_path in _iter_files(paths, project_root):
        try:
            sha256 = sha256_file(file_path)
        except OSError as exc:
            log.warning("ID registry: could not read %s: %s", file_path, exc)
            continue
        rel_path = file_path.relative_to(project_root).as_posix()
        self.register_file(rel_path, sha256)

        # The scan's rule (suffix filter, then header), so an entity is
        # registered for the files a scan lists as models, and only those.
        if (
            is_model_candidate_name(file_path.name)
            and detect_ai_model_format(file_path) != AiModelFormat.UNKNOWN
        ):
            self.register_entity(file_path.stem, "ai_AIPackage")

harvest

harvest(
    object_set: SHACLObjectSet,
) -> tuple[int, int, bool]

Harvest every named element in object_set into this registry.

Used both by :meth:import_sbom (after deserializing an existing SBOM from disk) and by SBOM generation itself, directly on a :class:~pitloom.export.spdx3_json.Spdx3JsonExporter's in-memory object set -- no serialize/reparse round trip needed there, since every element already carries its assigned spdxId.

Returns (new_files, new_entities, changed): the first two are net count deltas (for a caller's own log message), the third is a proper "did anything actually change" signal a caller should gate a save() on instead -- :func:~pitloom.id_registry._harvest._release_stale_keys_for_id can drop one stale key in the same pass that adds another, which nets to a zero size delta despite real content changing (the surviving key's id, or the stale key's removal, both need persisting); the net-count deltas alone cannot detect that case.

Source code in pitloom/id_registry/_registry.py
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
def harvest(self, object_set: spdx3.SHACLObjectSet) -> tuple[int, int, bool]:
    """Harvest every named element in *object_set* into this registry.

    Used both by :meth:`import_sbom` (after deserializing an existing
    SBOM from disk) and by SBOM generation itself, directly on a
    :class:`~pitloom.export.spdx3_json.Spdx3JsonExporter`'s in-memory
    object set -- no serialize/reparse round trip needed there, since
    every element already carries its assigned ``spdxId``.

    Returns ``(new_files, new_entities, changed)``: the first two are
    *net* count deltas (for a caller's own log message), the third is
    a proper "did anything actually change" signal a caller should
    gate a ``save()`` on instead --
    :func:`~pitloom.id_registry._harvest._release_stale_keys_for_id`
    can drop one stale key in the same pass that adds another, which
    nets to a zero size delta despite real content changing (the
    surviving key's id, or the stale key's removal, both need
    persisting); the net-count deltas alone cannot detect that case.
    """
    return self._harvest_sorted(_sorted_by_spdx_id(object_set))

has_entity_named

has_entity_named(name: str) -> bool

Return whether name is registered under any type at all, as given or as the type's own key form (:func:_entity_key).

Source code in pitloom/id_registry/_registry.py
147
148
149
150
151
152
def has_entity_named(self, name: str) -> bool:
    """Return whether *name* is registered under any type at all, as
    given or as the type's own key form (:func:`_entity_key`)."""
    return any(
        key[1] == name or key == _entity_key(name, key[0]) for key in self.entities
    )

import_sbom

import_sbom(sbom_path: Path) -> frozenset[tuple[str, str]]

Harvest ids from an existing SPDX 3 JSON-LD SBOM into this registry.

Returns the entities keys not imported because the SBOM holds several elements under one name (see :func:~pitloom.id_registry._ambiguous._ambiguous_entity_keys).

Source code in pitloom/id_registry/_registry.py
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
def import_sbom(self, sbom_path: Path) -> frozenset[tuple[str, str]]:
    """Harvest ids from an existing SPDX 3 JSON-LD SBOM into this registry.

    Returns the ``entities`` keys not imported because the SBOM holds
    several elements under one name (see
    :func:`~pitloom.id_registry._ambiguous._ambiguous_entity_keys`).
    """
    object_set = spdx3.SHACLObjectSet()
    with open(sbom_path, "rb") as f:
        spdx3.JSONLDDeserializer().read(f, object_set)

    sorted_objects = _sorted_by_spdx_id(object_set)

    if not self.files and not self.entities:
        for obj in sorted_objects:
            if isinstance(obj, spdx3.SpdxDocument) and obj.spdxId:
                self.namespace = obj.spdxId
                break

    return _harvest_elements(self, sorted_objects)

load classmethod

load(path: Path) -> IdRegistry

Load a registry from path.

Every failure (missing, unreadable, malformed JSON, wrong shape, wrong version, malformed entry, duplicate) raises one ValueError via :func:~pitloom.id_registry._types.registry_file_error -- never returns None. "Maybe there's a registry" is resolved upstream, by :func:~pitloom.id_registry.resolve.resolve_registry.

Source code in pitloom/id_registry/_registry.py
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
@classmethod
def load(cls, path: Path) -> IdRegistry:
    """Load a registry from *path*.

    Every failure (missing, unreadable, malformed JSON, wrong shape,
    wrong version, malformed entry, duplicate) raises one ``ValueError``
    via :func:`~pitloom.id_registry._types.registry_file_error` --
    never returns ``None``. "Maybe there's a registry" is resolved
    upstream, by :func:`~pitloom.id_registry.resolve.resolve_registry`.
    """
    if not os.path.isfile(path):
        raise registry_file_error(path, "not found or not a file")
    try:
        with open(path, "rb") as f:
            raw = f.read()
    except OSError as exc:
        raise registry_file_error(path, f"cannot be read: {exc}") from exc

    try:
        data: Any = json.loads(raw)
    except (ValueError, RecursionError) as exc:
        raise registry_file_error(path, f"not valid JSON: {exc}") from exc

    if not isinstance(data, dict):
        raise registry_file_error(path, "not a JSON object")

    namespace = data.get("namespace")
    if not isinstance(namespace, str) or not namespace:
        raise registry_file_error(path, "missing a valid 'namespace'")

    version = data.get("version")
    if version != _REGISTRY_VERSION:
        raise registry_file_error(
            path,
            f"version {version!r}, expected {_REGISTRY_VERSION} (no "
            "migration support -- delete it and re-run `pitloom id "
            "generate` or `pitloom id import`)",
        )

    try:
        files = {
            str(rel_path): FileEntry(
                spdx_id=_require_scalar_str(entry["spdxId"], "spdxId"),
                sha256=_require_scalar_str(entry["sha256"], "sha256"),
            )
            for rel_path, entry in data.get("files", {}).items()
        }
        entities: dict[tuple[str, str], EntityEntry] = {}
        for type_name, names in data.get("entities", {}).items():
            for name, entry in names.items():
                key = _entity_key(str(name), str(type_name))
                if key in entities:
                    raise registry_file_error(
                        path, f"two {key[0]} entries for {key[1]!r}"
                    )
                entities[key] = EntityEntry(
                    spdx_id=_require_scalar_str(entry["spdxId"], "spdxId")
                )
    except (KeyError, TypeError, AttributeError) as exc:
        raise registry_file_error(path, f"malformed entry: {exc}") from exc

    return cls(namespace=namespace, files=files, entities=entities, path=path)

lookup_entity

lookup_entity(name: str, type_name: str) -> str | None

Return the registered spdxId for the named entity of type_name.

Source code in pitloom/id_registry/_registry.py
142
143
144
145
def lookup_entity(self, name: str, type_name: str) -> str | None:
    """Return the registered ``spdxId`` for the named entity of *type_name*."""
    entry = self.entities.get(_entity_key(name, type_name))
    return entry.spdx_id if entry is not None else None

lookup_file

lookup_file(path: str, sha256: str) -> str | None

Return the registered spdxId for path.

Source code in pitloom/id_registry/_registry.py
137
138
139
140
def lookup_file(self, path: str, sha256: str) -> str | None:
    """Return the registered ``spdxId`` for *path*."""
    entry = self.files.get(path)
    return entry.spdx_id if entry is not None and entry.sha256 == sha256 else None

new classmethod

new(
    project_name: str, path: Path | None = None
) -> IdRegistry

Create a fresh, empty registry with a freshly minted namespace.

Source code in pitloom/id_registry/_registry.py
68
69
70
71
72
@classmethod
def new(cls, project_name: str, path: Path | None = None) -> IdRegistry:
    """Create a fresh, empty registry with a freshly minted namespace."""
    namespace = doc_namespace(project_name, str(uuid4()))
    return cls(namespace=namespace, path=path)

register_entity

register_entity(name: str, type_name: str) -> str

Register (or reuse) a named entity and return its spdxId.

Keyed by (type_name, name) (see :func:_entity_key for the :data:PACKAGE_ENTITY_TYPE canonicalization applied to name): an entity already registered under name but a different type is a distinct entry, not a conflict -- both are kept, each looked up only by its own type.

Source code in pitloom/id_registry/_registry.py
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
def register_entity(self, name: str, type_name: str) -> str:
    """Register (or reuse) a named entity and return its ``spdxId``.

    Keyed by ``(type_name, name)`` (see :func:`_entity_key` for the
    :data:`PACKAGE_ENTITY_TYPE` canonicalization applied to *name*): an
    entity already registered under *name* but a *different* type is a
    distinct entry, not a conflict -- both are kept, each looked up
    only by its own type.
    """
    key = _entity_key(name, type_name)
    existing = self.entities.get(key)
    if existing is not None:
        return existing.spdx_id
    spdx_id = self._mint_id(_type_id_prefix(type_name))
    self.entities[key] = EntityEntry(spdx_id=spdx_id)
    return spdx_id

register_file

register_file(path: str, sha256: str) -> str

Register (or refresh) a file entry and return its spdxId.

Source code in pitloom/id_registry/_registry.py
173
174
175
176
177
178
179
180
181
182
183
184
185
186
def register_file(self, path: str, sha256: str) -> str:
    """Register (or refresh) a file entry and return its ``spdxId``."""
    existing = self.files.get(path)
    if existing is not None and existing.sha256 == sha256:
        return existing.spdx_id
    if existing is not None:
        log.info(
            "ID registry: content changed for %s; minting a new spdxId (old: %s).",
            path,
            existing.spdx_id,
        )
    spdx_id = self._mint_id("File")
    self.files[path] = FileEntry(spdx_id=spdx_id, sha256=sha256)
    return spdx_id

save

save(path: Path | None = None) -> None

Write this registry as JSON to path.

Source code in pitloom/id_registry/_registry.py
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
def save(self, path: Path | None = None) -> None:
    """Write this registry as JSON to *path*."""
    target = path or self.path
    if target is None:
        raise ValueError("No path given and registry has no default path")
    entities_by_type: dict[str, dict[str, dict[str, str]]] = {}
    for (type_name, name), entry in sorted(self.entities.items()):
        entities_by_type.setdefault(type_name, {})[name] = {"spdxId": entry.spdx_id}
    data = {
        "version": _REGISTRY_VERSION,
        "namespace": self.namespace,
        "files": {
            rel_path: {"spdxId": entry.spdx_id, "sha256": entry.sha256}
            for rel_path, entry in sorted(self.files.items())
        },
        "entities": entities_by_type,
    }
    target.parent.mkdir(parents=True, exist_ok=True)
    with open_text_lf(target) as f:
        f.write(canonical_json_indented(data))
        f.write("\n")
    self.path = target

pitloom.id_registry.IdRegistrySession

IdRegistrySession(registry: IdRegistry | None)

One document's registry lookups, first-claimant-wins.

Every element that would otherwise look registry.lookup_file/ lookup_entity up directly, across an entire document build (files, directories, AI models, the project's own, dependency and phantom packages, deployed packages, loom fragments), instead goes through one shared session so a registry hit reused by two different elements is claimed by only the first -- see :meth:file_id/:meth:entity_id.

Source code in pitloom/id_registry/_session.py
48
49
50
51
def __init__(self, registry: IdRegistry | None) -> None:
    self._registry = registry
    #: spdxId -> the claimant key that first claimed it.
    self._claimed: dict[str, str] = {}

registry property

registry: IdRegistry | None

The underlying :class:IdRegistry, or None when none applies.

claimed_ids

claimed_ids() -> list[str]

Return every id claimed so far in this session, in claim order.

Source code in pitloom/id_registry/_session.py
141
142
143
def claimed_ids(self) -> list[str]:
    """Return every id claimed so far in this session, in claim order."""
    return list(self._claimed)

entity_id

entity_id(
    claimant: str,
    names: Sequence[str],
    type_name: str,
    *,
    on_miss: Callable[[], None] | None = None,
) -> str | None

Look up a registered entity id for the first of names that matches under type_name, claim it, and return it -- or None.

names are tried in order (e.g. an AI model's declared name, then its file stem), each as given, then display-escaped (:func:~pitloom.core.untrusted_text.escape_display_controls): a registry imported from an SBOM holds a model's name as the SBOM shows it, escaped. The first raw hit is used. on_miss is called only when every name in names is a raw miss -- never when a hit exists but this session's claim is rejected. Returns None immediately, without calling on_miss, when no registry is loaded in this session.

Source code in pitloom/id_registry/_session.py
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
def entity_id(
    self,
    claimant: str,
    names: Sequence[str],
    type_name: str,
    *,
    on_miss: Callable[[], None] | None = None,
) -> str | None:
    """Look up a registered entity id for the first of *names* that
    matches under *type_name*, claim it, and return it -- or ``None``.

    *names* are tried in order (e.g. an AI model's declared name,
    then its file stem), each as given, then display-escaped
    (:func:`~pitloom.core.untrusted_text.escape_display_controls`): a
    registry imported from an SBOM holds a model's ``name`` as the SBOM
    shows it, escaped. The first raw hit is used. *on_miss* is
    called only when every name in *names* is a raw miss -- never
    when a hit exists but this session's claim is rejected. Returns
    ``None`` immediately, without calling *on_miss*, when no registry
    is loaded in this session.
    """
    if self._registry is None:
        return None
    for name in names:
        shown = escape_display_controls(name)
        for spelling in (name, shown) if shown != name else (name,):
            spdx_id = self._registry.lookup_entity(spelling, type_name)
            if spdx_id is not None:
                return self._claim(claimant, spdx_id)
    if on_miss is not None:
        on_miss()
    return None

file_id

file_id(
    claimant: str,
    paths: Sequence[str],
    sha256: str,
    *,
    on_miss: Callable[[], None] | None = None,
) -> str | None

Look up a registered file id for the first of paths that matches sha256, claim it, and return it -- or None.

paths are tried in order (e.g. a src/-layout file's physical path, then its distribution path); the first raw hit is used. on_miss is called only when every path in paths is a raw miss (not found, or found with a different hash) -- never when a hit exists but this session's claim is rejected (see :meth:_claim). Returns None immediately, without calling on_miss, when no registry is loaded in this session.

Source code in pitloom/id_registry/_session.py
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
def file_id(
    self,
    claimant: str,
    paths: Sequence[str],
    sha256: str,
    *,
    on_miss: Callable[[], None] | None = None,
) -> str | None:
    """Look up a registered file id for the first of *paths* that
    matches *sha256*, claim it, and return it -- or ``None``.

    *paths* are tried in order (e.g. a src/-layout file's physical
    path, then its distribution path); the first raw hit is used.
    *on_miss* is called only when every path in *paths* is a raw miss
    (not found, or found with a different hash) -- never when a hit
    exists but this session's claim is rejected (see :meth:`_claim`).
    Returns ``None`` immediately, without calling *on_miss*, when no
    registry is loaded in this session.
    """
    if self._registry is None:
        return None
    for path in paths:
        spdx_id = self._registry.lookup_file(path, sha256)
        if spdx_id is not None:
            return self._claim(claimant, spdx_id)
    if on_miss is not None:
        on_miss()
    return None

pitloom.id_registry.resolve_registry

resolve_registry(
    id_registry: str | Path | IdRegistry | None,
    configured: str | None,
    base_dir: Path,
) -> IdRegistry | None

Resolve the registry a build should consult -- an explicit source only, never searched for.

Precedence: id_registry (a flag/kwarg, or an already-loaded :class:IdRegistry), else configured (the applicable config's own id-registry: the project's own [tool.pitloom], or an explicit --config, which replaces it). Neither given: None, silently -- no loom-id-registry.json is searched for, near the target or anywhere else. A relative path resolves against base_dir (the project directory, or the current directory for a target with none of its own), itself resolved first, so the loaded registry's path (and every message naming it) is absolute even for a relative base_dir; an already-absolute path (e.g. a config's own key, made absolute by :func:~pitloom.core.config_cascade.load_config_file, or a CLI flag made absolute against cwd) is used as given.

Raises ValueError (via :meth:IdRegistry.load) when the resolved path does not load -- a declared registry is always meant to load; this function never swallows that failure.

Source code in pitloom/id_registry/resolve.py
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
def resolve_registry(
    id_registry: str | Path | IdRegistry | None,
    configured: str | None,
    base_dir: Path,
) -> IdRegistry | None:
    """Resolve the registry a build should consult -- an explicit source
    only, never searched for.

    Precedence: *id_registry* (a flag/kwarg, or an already-loaded
    :class:`IdRegistry`), else *configured* (the applicable config's own
    ``id-registry``: the project's own ``[tool.pitloom]``, or an explicit
    ``--config``, which replaces it). Neither given: ``None``, silently --
    no ``loom-id-registry.json`` is searched for, near the target or
    anywhere else. A relative path resolves against *base_dir* (the
    project directory, or the current directory for a target with none of
    its own), itself resolved first, so the loaded registry's path (and
    every message naming it) is absolute even for a relative *base_dir*;
    an already-absolute path (e.g. a config's own key, made absolute by
    :func:`~pitloom.core.config_cascade.load_config_file`, or a CLI flag
    made absolute against cwd) is used as given.

    Raises ``ValueError`` (via :meth:`IdRegistry.load`) when the resolved
    path does not load -- a declared registry is always meant to load;
    this function never swallows that failure.
    """
    source = id_registry if id_registry is not None else configured
    if source is None:
        return None
    if isinstance(source, IdRegistry):
        return source
    path = Path(source)
    return IdRegistry.load(path if path.is_absolute() else base_dir.resolve() / path)

pitloom.id_registry.registry_base_dir

registry_base_dir(target: Path) -> Path

The base directory a relative declared id-registry path resolves against, for a target that may be either a project directory or a single file (e.g. an sdist archive).

A file target has no directory of its own to resolve a relative registry path against -- :func:~pitloom.core.project.read_project resolves an sdist archive's own config keys relative to the archive itself, but a registry path is not read from inside the archive, so that convention doesn't apply here. Falls back to the current directory instead, same as the no-project_dir case.

Shared by every resolve_registry() call site that resolves a target which might be an sdist (or, in :mod:pitloom.embed, any single-file project_dir) -- callers that already know they have a directory (or already know they have none, using Path.cwd() directly) don't need this helper.

Source code in pitloom/id_registry/resolve.py
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
def registry_base_dir(target: Path) -> Path:
    """The base directory a relative declared ``id-registry`` path
    resolves against, for a *target* that may be either a project
    directory or a single file (e.g. an sdist archive).

    A file target has no directory of its own to resolve a relative
    registry path against -- :func:`~pitloom.core.project.read_project`
    resolves an sdist archive's own config keys relative to the archive
    itself, but a registry path is not read from inside the archive, so
    that convention doesn't apply here. Falls back to the current
    directory instead, same as the no-``project_dir`` case.

    Shared by every ``resolve_registry()`` call site that resolves a
    *target* which might be an sdist (or, in :mod:`pitloom.embed`, any
    single-file ``project_dir``) -- callers that already know they have a
    directory (or already know they have none, using ``Path.cwd()``
    directly) don't need this helper.
    """
    return Path.cwd() if os.path.isfile(target) else target

pitloom.id_registry.warn_claim_collision

warn_claim_collision(
    spdx_id: str, first: str, second: str
) -> None

Log the one WARNING: for spdx_id wanted by second after first already took it; second mints its own id.

Source code in pitloom/id_registry/_session.py
24
25
26
27
28
29
30
31
32
33
def warn_claim_collision(spdx_id: str, first: str, second: str) -> None:
    """Log the one ``WARNING:`` for *spdx_id* wanted by *second* after *first*
    already took it; *second* mints its own id."""
    log.warning(
        "ID registry: %s is registered for both %s and %s; %s gets a new id",
        spdx_id,
        first,
        second,
        second,
    )

pitloom.id_registry.EntityEntry dataclass

EntityEntry(spdx_id: str)

A single registered named entity (e.g. an AI model) and its spdxId.

No type field: :class:~pitloom.id_registry.IdRegistry keys its entities dict by (type, name), so the type already lives in the key -- storing it again here would be the same fact in two places, free to drift out of sync on a hand-edited registry file.

pitloom.id_registry.FileEntry dataclass

FileEntry(spdx_id: str, sha256: str)

A single registered file: its stable spdxId and content hash.

pitloom.id_registry.DEFAULT_ID_REGISTRY_FILENAME module-attribute

DEFAULT_ID_REGISTRY_FILENAME = 'loom-id-registry.json'