Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -835,6 +835,9 @@ The top-level shape is (this example shows a full LLM-backed scan; with `--no-ll
- `risk_assessment.severity` ∈ `LOW | MEDIUM | HIGH | CRITICAL`.
- `risk_assessment.recommendation` ∈ `SAFE | CAUTION | DO_NOT_INSTALL`, mapped from severity: `LOW → SAFE`, `MEDIUM → CAUTION`, `HIGH`/`CRITICAL → DO_NOT_INSTALL`.
- `metadata.llm_error` appears only when LLM analysis was requested but unavailable.
- Archive-member components retain their virtual `path` and include `outer_path`, `nested_path`,
and container provenance. Virtual paths identify in-memory members, not on-disk files; see
[JSON component locations](docs/NESTED_ARTIFACT_INSPECTION.md#json-component-locations).
- AE1 findings use **Incomplete referenced artifact analysis**. Their source
location identifies the reference; `evidence` identifies the affected target,
analyzer reasons, and available bounds. Review the target's completeness
Expand Down
31 changes: 31 additions & 0 deletions docs/NESTED_ARTIFACT_INSPECTION.md
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,37 @@ Nested members use a stable virtual path that retains their full provenance:
outer-file!/nested.zip!/scripts/setup.sh
```

## JSON component locations

JSON component rows preserve this virtual path in `path`, matching finding locations and
inspection-ledger entries. Archive members also expose their existing inventory provenance:

```json
{
"path": "bundle.whl!/package/module.py",
"type": "python",
"executable": true,
"outer_path": "bundle.whl",
"nested_path": "package/module.py",
"container_type": "zip",
"container_ancestry": ["zip"],
"container_depth": 1
}
```

`outer_path` names the physical container relative to that component's source root. `nested_path`
retains the full member chain inside it, including any nested `!/` boundaries. Each member keeps
its own component row, executable flag, type, line count, and byte size. Ordinary files do not gain
archive fields merely because their names contain `!/`.

Consumers must distinguish a virtual member from an on-disk file: `executable: true` describes
the member's content, not the existence of its virtual path on disk. Use the source identity and
full `path` to join components with findings; do not collapse members by `outer_path`. When
validating a report, check archive provenance and containment within the corresponding source
root, and retain incomplete-inspection failures. These additive fields do not change existing
consumer validators automatically, prove that every member was inspected, or justify extracting
or executing archive content.

## Security invariants

- Members are read in memory. SkillSpector never extracts, renders, imports, installs, or executes
Expand Down
51 changes: 38 additions & 13 deletions src/skillspector/nodes/report.py
Original file line number Diff line number Diff line change
Expand Up @@ -1370,6 +1370,43 @@ def _build_metadata(
return meta


def _json_component(component: Mapping[str, object]) -> dict[str, object]:
"""Preserve component identity and expose structured archive provenance."""
result = {
key: component.get(key)
for key in (
"path",
"type",
"lines",
"executable",
"size_bytes",
"source_url",
"source_identity",
"source_digest",
)
}
outer_path = component.get("outer_path")
nested_path = component.get("nested_path")
if (
isinstance(outer_path, str)
and outer_path
and isinstance(nested_path, str)
and nested_path
and component.get("path") == f"{outer_path}!/{nested_path}"
):
result.update(outer_path=outer_path, nested_path=nested_path)
container_type = component.get("container_type")
if isinstance(container_type, str) and container_type:
result["container_type"] = container_type
depth = component.get("container_depth")
if isinstance(depth, int) and not isinstance(depth, bool) and depth > 0:
result["container_depth"] = depth
ancestry = component.get("container_ancestry")
if isinstance(ancestry, list) and all(isinstance(item, str) and item for item in ancestry):
result["container_ancestry"] = list(ancestry)
return result


def _format_json(
findings: list[Finding],
component_metadata: list[dict[str, object]],
Expand Down Expand Up @@ -1410,19 +1447,7 @@ def _format_json(
"recommendation": risk_recommendation,
"max_issue_severity": _max_issue_severity(findings),
},
"components": [
{
"path": c.get("path"),
"type": c.get("type"),
"lines": c.get("lines"),
"executable": c.get("executable"),
"size_bytes": c.get("size_bytes"),
"source_url": c.get("source_url"),
"source_identity": c.get("source_identity"),
"source_digest": c.get("source_digest"),
}
for c in component_metadata
],
"components": [_json_component(component) for component in component_metadata],
"structured_summaries": structured_summaries or [],
"issues": [f.to_dict() for f in findings],
"suppressed_count": len(suppressed),
Expand Down
176 changes: 176 additions & 0 deletions tests/nodes/test_report_archive_provenance.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,176 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0

"""Archive provenance remains available without changing reported member identity."""

from __future__ import annotations

import io
import json
import zipfile
from pathlib import Path
from typing import cast

import pytest

from skillspector.models import Finding
from skillspector.nodes.analyzers.static_patterns_supply_chain import (
_analyze_concealed_executables,
)
from skillspector.nodes.build_context import build_context
from skillspector.nodes.finalize_inspection_ledger import finalize_inspection_ledger
from skillspector.nodes.report import report
from skillspector.state import SkillspectorState

PROVENANCE_FIELDS = (
"outer_path",
"nested_path",
"container_type",
"container_ancestry",
"container_depth",
)


@pytest.fixture(autouse=True)
def disable_provider_probe(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setattr(
"skillspector.nodes.report.is_llm_available", lambda **_: (False, "Disabled in test")
)


def _zip_bytes(members: dict[str, bytes]) -> bytes:
buffer = io.BytesIO()
with zipfile.ZipFile(buffer, "w") as archive:
for name, content in members.items():
archive.writestr(name, content)
return buffer.getvalue()


def _json_report(context: dict[str, object], findings: list[Finding] | None = None) -> dict:
state = cast(
SkillspectorState,
{**context, "findings": findings or [], "output_format": "json", "use_llm": False},
)
return json.loads(report(state)["report_body"])


def test_wheel_and_recursive_zip_keep_complete_member_identity(tmp_path: Path) -> None:
(tmp_path / "package.whl").write_bytes(
_zip_bytes({"package/__init__.py": b"VALUE = 1\n", "package/data.txt": b"data\n"})
)
(tmp_path / "outer.zip").write_bytes(
_zip_bytes({"inner.zip": _zip_bytes({"run.sh": b"#!/bin/sh\necho hello\n"})})
)
context = build_context({"skill_path": str(tmp_path)})
metadata = context["component_metadata"]
source = {
"source_url": "https://example.com/package.zip",
"source_identity": "source-sha256:" + "a" * 64,
"source_digest": "b" * 64,
}
for component in metadata:
component.update(source)

payload = _json_report(context)
rows = {row["path"]: row for row in payload["components"]}

assert len(payload["components"]) == len(metadata) == len(rows)
assert rows.keys() == {component["path"] for component in metadata}
for component in metadata:
row = rows[component["path"]]
for field in ("path", "type", "lines", "executable", "size_bytes"):
assert row[field] == component[field]
assert {field: row[field] for field in source} == source
for path, outer, nested, depth in (
("package.whl!/package/__init__.py", "package.whl", "package/__init__.py", 1),
("package.whl!/package/data.txt", "package.whl", "package/data.txt", 1),
("outer.zip!/inner.zip", "outer.zip", "inner.zip", 1),
("outer.zip!/inner.zip!/run.sh", "outer.zip", "inner.zip!/run.sh", 2),
):
row = rows[path]
assert row["outer_path"] == outer
assert row["nested_path"] == nested
assert row["container_type"] == "zip"
assert row["container_ancestry"] == ["zip"] * depth
assert row["container_depth"] == depth
assert rows["package.whl!/package/__init__.py"]["executable"] is True
assert rows["outer.zip!/inner.zip!/run.sh"]["executable"] is True
assert rows["package.whl!/package/data.txt"]["executable"] is False
assert not any(field in rows["package.whl"] for field in PROVENANCE_FIELDS)


def test_concealed_member_keeps_security_finding_and_risk(tmp_path: Path) -> None:
(tmp_path / ".hidden.zip").write_bytes(_zip_bytes({"run.sh": b"#!/bin/sh\necho hello\n"}))
context = build_context({"skill_path": str(tmp_path)})
findings = _analyze_concealed_executables(context["component_metadata"])
finding = next(item for item in findings if item.rule_id == "SC9")
original_fingerprint = finding.fingerprint()

payload = _json_report(context, findings)
issue = next(item for item in payload["issues"] if item["id"] == "SC9")
member = next(row for row in payload["components"] if row["path"] == finding.file)

assert issue["location"]["file"] == member["path"] == ".hidden.zip!/run.sh"
assert member["outer_path"] == issue["evidence"]["outer_path"] == ".hidden.zip"
assert member["nested_path"] == issue["evidence"]["nested_path"] == "run.sh"
assert member["executable"] is True
assert issue["severity"] == "HIGH"
assert issue["match_fingerprint"] == original_fingerprint
assert payload["risk_assessment"]["score"] > 0
assert payload["risk_assessment"]["recommendation"] != "SAFE"


def test_truncated_nested_archive_preserves_incomplete_verdict(tmp_path: Path) -> None:
(tmp_path / "outer.zip").write_bytes(_zip_bytes({"broken.zip": b"PK\x03\x04truncated"}))
context = build_context({"skill_path": str(tmp_path), "use_llm": False})
finalized = finalize_inspection_ledger(cast(SkillspectorState, {**context, "use_llm": False}))

payload = _json_report({**context, **finalized})
completeness = payload["analysis_completeness"]
member = next(row for row in payload["components"] if row["path"] == "outer.zip!/broken.zip")

assert member["outer_path"] == "outer.zip"
assert member["nested_path"] == "broken.zip"
assert completeness == finalized["analysis_completeness"]
assert completeness["is_complete"] is False
assert any(
row["path"] == member["path"] and row["reason_code"] == "archive_truncated"
for row in completeness["ledger_exceptions"]
)
assert payload["execution_successful"] == completeness["execution_successful"]
assert payload["risk_assessment"]["recommendation"] != "SAFE"


def test_literal_archive_separator_in_physical_path_has_no_provenance(tmp_path: Path) -> None:
directory = tmp_path / "notes!"
directory.mkdir()
(directory / "run.py").write_text("VALUE = 1\n", encoding="utf-8")
context = build_context({"skill_path": str(tmp_path)})

payload = _json_report(context)
row = next(item for item in payload["components"] if item["path"] == "notes!/run.py")

assert row["executable"] is True
assert not any(field in row for field in PROVENANCE_FIELDS)


@pytest.mark.parametrize(
("outer_path", "nested_path"),
[("other.zip", "run.py"), ("bundle.zip", "other.py"), ("", "run.py"), (None, "run.py")],
)
def test_inconsistent_provenance_is_not_exported(outer_path: str | None, nested_path: str) -> None:
component = {
"path": "bundle.zip!/run.py",
"executable": True,
"outer_path": outer_path,
"nested_path": nested_path,
"container_type": "zip",
"container_ancestry": ["zip"],
"container_depth": 1,
}

row = _json_report({"component_metadata": [component]})["components"][0]

assert row["path"] == component["path"]
assert row["executable"] is True
assert not any(field in row for field in PROVENANCE_FIELDS)
Loading