Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
58 commits
Select commit Hold shift + click to select a range
0239f80
fix(analyzer): stabilize TM1 window identity
chrisknvidia Sep 23, 2026
193bb90
Merge main into fix/christopherk/tm1-window-identity
chrisknvidia Sep 23, 2026
7fee7af
Merge remote-tracking branch 'origin/main' into fix/christopherk/tm1-…
chrisknvidia Sep 28, 2026
912263f
Merge remote-tracking branch 'origin/main' into fix/christopherk/tm1-…
chrisknvidia Sep 28, 2026
68e22d0
fix(analyzer): track eager receiver side effects
chrisknvidia Sep 28, 2026
6d877de
feat(python): refresh execution surfaces on current main
chrisknvidia Oct 5, 2026
2eefaa0
fix(analyzer): preserve lexical TM1 when dataflow abstains
chrisknvidia Oct 5, 2026
60a88ce
fix(analyzer): require a retained companion owner
chrisknvidia Oct 5, 2026
24924e8
fix(analyzer): preserve execution scope in binding evidence
chrisknvidia Oct 5, 2026
388acb8
fix(analyzer): preserve lexical TM1 across conservative Python analysis
chrisknvidia Oct 5, 2026
e3c3bfd
style(analyzer): simplify binding event deduplication
chrisknvidia Oct 5, 2026
ebc69d7
fix: preserve runtime and retained shell evidence
chrisknvidia Oct 5, 2026
8bdb341
fix: preserve runtime and retained shell evidence
chrisknvidia Oct 5, 2026
03e862e
fix(analyzer): preserve called-slot identity and cached method evidence
chrisknvidia Oct 5, 2026
87c67a0
fix(analyzer): preserve called-slot identity and cached method evidence
chrisknvidia Oct 5, 2026
637c2fb
fix: preserve cached subprocess slot state across scoped imports
chrisknvidia Oct 5, 2026
6842348
fix: preserve cached subprocess slot state across scoped imports
chrisknvidia Oct 5, 2026
cf7805f
fix(analyzer): reconcile explicit cached-slot evidence across scopes
chrisknvidia Oct 5, 2026
e96c5b1
fix(analyzer): reconcile explicit cached-slot evidence across scopes
chrisknvidia Oct 5, 2026
0fb9acd
Integrate current main with reviewed Python analyzer changes
chrisknvidia Oct 5, 2026
c11f26a
Integrate current main with reviewed Python analyzer changes
chrisknvidia Oct 5, 2026
2433033
fix: abstain from cached slot proof after eager effects
chrisknvidia Oct 5, 2026
e564c1c
fix: abstain from cached slot proof after eager effects
chrisknvidia Oct 5, 2026
d8cef86
test: retain lexical findings after unknown cached-slot effects
chrisknvidia Oct 5, 2026
afd5a7e
test: retain lexical findings after unknown cached-slot effects
chrisknvidia Oct 5, 2026
53ac4d1
fix: bound direct called-slot evidence to known execution order
chrisknvidia Oct 5, 2026
6c1431d
fix: bound direct called-slot evidence to known execution order
chrisknvidia Oct 5, 2026
4ed5340
fix: limit direct slot proof to single assignment targets
chrisknvidia Oct 5, 2026
1f22395
fix: limit direct slot proof to single assignment targets
chrisknvidia Oct 5, 2026
46dd4ca
fix: invalidate slot evidence on protocols and unsafe releases
chrisknvidia Oct 5, 2026
4378003
fix: invalidate slot evidence on protocols and unsafe releases
chrisknvidia Oct 5, 2026
d6f7de3
fix: keep legacy shadow proof isolated from later effects
chrisknvidia Oct 5, 2026
cc687fe
fix: keep legacy shadow proof isolated from later effects
chrisknvidia Oct 5, 2026
8fe6c77
fix: abstain from cached replacement proof after unknown effects
chrisknvidia Oct 5, 2026
b175297
fix: abstain from cached replacement proof after unknown effects
chrisknvidia Oct 5, 2026
f2d34c8
fix: retain native callable detection across cached slot stores
chrisknvidia Oct 5, 2026
5abb725
fix: retain native callable detection across cached slot stores
chrisknvidia Oct 5, 2026
5e51caa
fix: retain shell findings for native receiver assignments
chrisknvidia Oct 5, 2026
2d3ff10
fix: retain shell findings for native receiver assignments
chrisknvidia Oct 5, 2026
481a8fb
fix: preserve native Popen replacement warnings
chrisknvidia Oct 5, 2026
7b94996
fix: preserve native Popen replacement warnings
chrisknvidia Oct 5, 2026
3477217
test: retain occurrence metadata in prose shell reports
chrisknvidia Oct 5, 2026
9162964
Merge current main into fix/christopherk/tm1-window-identity
rng1995 Oct 8, 2026
15b4e58
fix(analyzer): require proven replacement before revoking lexical TM1
rng1995 Oct 8, 2026
eb74b76
Merge current main into feat/christopherk/python-execution-surface-cl…
rng1995 Oct 8, 2026
bbdf081
fix(analyzer): require proven replacement before revoking lexical TM1
rng1995 Oct 8, 2026
8ed05b6
test: account for TM1 Python parsing in marker ownership ledger
rng1995 Oct 8, 2026
548b17c
test: account for TM1 Python parsing in marker ownership ledger
rng1995 Oct 8, 2026
2ba149d
fix(analyzer): keep one TM1 owner for true-prefixed names without AST…
rng1995 Oct 8, 2026
7b643ff
fix(analyzer): keep one TM1 owner for true-prefixed names without AST…
rng1995 Oct 8, 2026
327591b
fix(analyzer): report each true-prefixed shell call that #577 reports
rng1995 Oct 8, 2026
7ef4bf1
fix(analyzer): report each true-prefixed shell call that #577 reports
rng1995 Oct 8, 2026
efecb7a
fix(supply-chain): opt into Python source-type propagation
rng1995 Oct 8, 2026
4b0043f
fix(build-context): decode PEP 263 primary Python before UTF-8 rejection
rng1995 Oct 8, 2026
4031711
Merge current main (with #577) into fix/christopherk/tm1-window-identity
rng1995 Oct 8, 2026
a4131c9
Merge current main (with #577) into feat/christopherk/python-executio…
rng1995 Oct 8, 2026
8d4b7d9
Merge #578 head 4031711 into feat/christopherk/python-execution-surfa…
rng1995 Oct 8, 2026
e89f08d
Merge current main (with #578) into feat/christopherk/python-executio…
rng1995 Oct 9, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
15 changes: 15 additions & 0 deletions src/skillspector/artifacts.py
Original file line number Diff line number Diff line change
Expand Up @@ -230,6 +230,7 @@ class _ObfuscatedIgnoreState:
".markdown",
".txt",
".py",
".pyw",
".sh",
".json",
".yaml",
Expand Down Expand Up @@ -403,6 +404,20 @@ def classify_artifact(path: str, data: bytes, *, referenced: bool = False) -> Ar
}


def promote_artifact_to_decoded_text(artifact: ArtifactRecord) -> None:
"""Apply a successful format-aware text decode without erasing prior limits."""
generic_binary_scope = artifact["content_kind"] is ContentKind.BINARY and (
artifact["disposition"] is ArtifactDisposition.OUT_OF_SCOPE
or artifact["disposition"] is ArtifactDisposition.PARTIAL
and "reason" not in artifact
)
artifact["content_kind"] = ContentKind.TEXT
artifact["decodable"] = True
artifact["misleading_extension"] = _suffix(artifact["path"]) in _BINARY_EXTENSIONS
if generic_binary_scope:
artifact["disposition"] = ArtifactDisposition.ANALYZED


def decode_text(data: bytes) -> str:
"""Return the loss-tolerant local text projection for static analyzers."""
return data.decode("utf-8", errors="replace")
Expand Down
16 changes: 15 additions & 1 deletion src/skillspector/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -59,7 +59,7 @@
)
from skillspector.logging_config import get_logger, set_level
from skillspector.mcp_registry import scan_registry
from skillspector.models import Finding
from skillspector.models import OCCURRENCE_FINDING_ID_KEY, Finding
from skillspector.multi_skill import (
MultiSkillDetectionResult,
SkillDirectory,
Expand Down Expand Up @@ -1314,18 +1314,32 @@ def _cache_transitive_result(
child_filtered = _coerce_findings_list(child_result.get("filtered_findings"))
child_findings = _coerce_findings_list(child_result.get("findings"))
all_ids = {finding.finding_id for finding in [*child_filtered, *child_findings]}
all_ids.update(
occurrence_id
for finding in [*child_filtered, *child_findings]
for occurrence in finding.occurrences
if isinstance((occurrence_id := occurrence.get(OCCURRENCE_FINDING_ID_KEY)), str)
)
all_ids.update(_effective_finding_ids(child_result))
finding_id_map = {
finding_id: _scoped_finding_id(source_identity, finding_id) for finding_id in all_ids
}

def _scope_finding(finding: Finding) -> Finding:
occurrences = []
for raw in finding.occurrences:
occurrence = dict(raw)
occurrence_id = occurrence.get(OCCURRENCE_FINDING_ID_KEY)
if isinstance(occurrence_id, str):
occurrence[OCCURRENCE_FINDING_ID_KEY] = finding_id_map[occurrence_id]
occurrences.append(occurrence)
return replace(
finding,
finding_id=finding_id_map[finding.finding_id],
source_url=target,
source_identity=source_identity,
source_digest=source_digest,
occurrences=occurrences,
)

scoped_filtered = [_scope_finding(item) for item in child_filtered[:_TRANSITIVE_MAX_FINDINGS]]
Expand Down
1 change: 1 addition & 0 deletions src/skillspector/input_handler.py
Original file line number Diff line number Diff line change
Expand Up @@ -82,6 +82,7 @@
_DIRECT_FILE_URL_SUFFIXES = (
".md",
".py",
".pyw",
".sh",
)

Expand Down
8 changes: 8 additions & 0 deletions src/skillspector/inspection_ledger.py
Original file line number Diff line number Diff line change
Expand Up @@ -108,6 +108,8 @@ class LedgerReason(StrEnum):
OUTPUT_LIMIT = "output_limit"
TRANSITIVE_CHILD_SCAN_FAILED = "transitive_child_scan_failed"
STATIC_PARSE_LIMIT = "static_parse_limit"
PYTHON_SOURCE_AMBIGUOUS = "python_source_ambiguous"
PYTHON_SOURCE_DECODE_ERROR = "python_source_decode_error"
OBFUSCATED_INSTRUCTION_TEXT = "obfuscated_instruction_text"


Expand Down Expand Up @@ -221,6 +223,12 @@ class LedgerReason(StrEnum):
LedgerReason.STATIC_PARSE_LIMIT: (
"A security-relevant expression exceeded a bounded static parser's span limit."
),
LedgerReason.PYTHON_SOURCE_AMBIGUOUS: (
"Python execution intent depends on runtime or platform-specific shebang semantics."
),
LedgerReason.PYTHON_SOURCE_DECODE_ERROR: (
"Python source bytes could not be decoded under their declared encoding."
),
LedgerReason.OBFUSCATED_INSTRUCTION_TEXT: (
"Obfuscated instruction text could not be fully evaluated by the deterministic layer."
),
Expand Down
9 changes: 8 additions & 1 deletion src/skillspector/models.py
Original file line number Diff line number Diff line change
Expand Up @@ -120,6 +120,11 @@ def _new_finding_id() -> str:
return f"finding-{uuid4().hex}"


OCCURRENCE_FINDING_ID_KEY = "_skillspector_report_finding_id"
OCCURRENCE_CODE_SNIPPET_KEY = "_skillspector_report_code_snippet"
_PRIVATE_OCCURRENCE_KEYS = frozenset({OCCURRENCE_FINDING_ID_KEY, OCCURRENCE_CODE_SNIPPET_KEY})


@dataclass
class Finding:
"""Finding model for graph state and report output (shape aligned with to_dict)."""
Expand Down Expand Up @@ -218,7 +223,9 @@ def _serialized_occurrences(self) -> list[dict[str, object]]:
]
serialized: list[dict[str, object]] = []
for raw in occurrences:
occurrence = dict(raw)
occurrence = {
key: value for key, value in raw.items() if key not in _PRIVATE_OCCURRENCE_KEYS
}
if self.source_identity:
occurrence.setdefault("source_identity", self.source_identity)
if self.source_digest:
Expand Down
26 changes: 24 additions & 2 deletions src/skillspector/nested_artifacts.py
Original file line number Diff line number Diff line change
Expand Up @@ -41,6 +41,7 @@
LedgerRecordType,
ledger_event,
)
from skillspector.python_ast import PythonSourceClassification, classify_python_source

ARCHIVE_MAX_DEPTH = 3
ARCHIVE_MAX_MEMBERS = 1_000
Expand Down Expand Up @@ -80,6 +81,7 @@
".phtml",
".ps1",
".py",
".pyw",
".pyc",
".pyo",
".rb",
Expand Down Expand Up @@ -114,6 +116,11 @@ class NestedInspectionResult:
components: list[str] = field(default_factory=list)
file_cache: dict[str, str] = field(default_factory=dict)
raw_file_cache: dict[str, bytes] = field(default_factory=dict)
# Classify with the archive member's execution path, while retaining the
# virtual path as the stable cache/report key used by downstream analyzers.
python_source_classifications: dict[str, PythonSourceClassification] = field(
default_factory=dict
)
artifact_inventory: list[ArtifactRecord] = field(default_factory=list)
metadata: list[dict[str, object]] = field(default_factory=list)
outer_metadata: dict[str, dict[str, object]] = field(default_factory=dict)
Expand Down Expand Up @@ -539,14 +546,22 @@ def _record_outer_metadata(
}


def _virtual_type(path: str, data: bytes, nested_type: str | None) -> str:
def _virtual_type(
path: str,
data: bytes,
nested_type: str | None,
source_classification: PythonSourceClassification,
) -> str:
if nested_type is not None:
return nested_type
if source_classification is PythonSourceClassification.PYTHON:
return "python"
suffix = Path(path).suffix.lower()
return {
".md": "markdown",
".markdown": "markdown",
".py": "python",
".pyw": "python",
".sh": "shell",
".bash": "shell",
".zsh": "shell",
Expand Down Expand Up @@ -1098,10 +1113,17 @@ def _inspect_zip_bytes(
executable = _member_executable(info, safe_name, member_data)
member_hidden = _is_hidden_path(safe_name)
concealed = executable and bool(concealment_reasons)
virtual_type = _virtual_type(safe_name, member_data, nested_type)
source_classification = classify_python_source(safe_name, member_data)
virtual_type = _virtual_type(
safe_name,
member_data,
nested_type,
source_classification,
)
result.components.append(virtual_path)
result.file_cache[virtual_path] = member_data.decode("utf-8", errors="replace")
result.raw_file_cache[virtual_path] = member_data
result.python_source_classifications[virtual_path] = source_classification
artifact = classify_artifact(virtual_path, member_data)
result.artifact_inventory.append(artifact)
result.metadata.append(
Expand Down
80 changes: 76 additions & 4 deletions src/skillspector/nodes/analyzers/behavioral_ast.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,7 +32,12 @@
)
from skillspector.logging_config import get_logger
from skillspector.models import AnalyzerFinding, Finding, Location, Severity
from skillspector.python_ast import ParsedPythonFile, get_python_ast
from skillspector.python_ast import (
ParsedPythonFile,
PythonSourceClassification,
get_python_ast,
resolve_python_source_classification,
)
from skillspector.state import (
AnalyzerNodeResponse,
SkillspectorState,
Expand Down Expand Up @@ -648,14 +653,42 @@ def node(state: SkillspectorState) -> AnalyzerNodeResponse:
"""Parse Python files via AST and detect dangerous execution patterns."""
components: list[str] = state.get("components") or []
file_cache: dict[str, str] = state.get("local_file_cache") or state.get("file_cache") or {}
raw_file_cache = state.get("raw_file_cache")
source_classifications = (
state.get("python_source_classifications")
if "python_source_classifications" in state
else None
)
source_classification_limitations = state.get("python_source_classification_limitations") or {}
source_decode_failures = state.get("python_source_decode_failures") or {}
python_ast_cache_key = state.get("python_ast_cache_key")
all_findings: list[Finding] = []
ledger_events: list[InspectionLedgerEvent] = []
budget = _BehavioralBudget(state)
terminal_limit: _BehavioralResourceLimitError | None = None

for path in components:
if not path.endswith(".py"):
content = file_cache.get(path)
source_classification: PythonSourceClassification | None = None
if source_classifications is not None and path in source_classifications:
source_classification = resolve_python_source_classification(
path,
content,
source_classifications=source_classifications,
raw_file_cache=raw_file_cache,
)
if source_classification is PythonSourceClassification.NON_PYTHON:
continue
if path in source_classification_limitations:
ledger_events.append(
ledger_event(
outcome=LedgerOutcome.PARTIAL,
phase="behavioral",
analyzer_id=ANALYZER_ID,
path=path,
reason=LedgerReason.RUNTIME_LIMIT,
)
)
continue
if terminal_limit is None and budget.analyzer_exhausted():
terminal_limit = _BehavioralResourceLimitError(
Expand All @@ -665,11 +698,41 @@ def node(state: SkillspectorState) -> AnalyzerNodeResponse:
"limit_findings": MAX_FINDINGS_PER_ANALYZER,
},
)
if terminal_limit is None:
try:
budget.check_runtime()
except _BehavioralResourceLimitError as exc:
terminal_limit = exc
if terminal_limit is not None:
event = _partial_limit_event(path, terminal_limit)
ledger_events.append(event)
continue
content = file_cache.get(path)
if path in source_decode_failures:
ledger_events.append(
ledger_event(
outcome=LedgerOutcome.PARTIAL,
phase="behavioral",
analyzer_id=ANALYZER_ID,
path=path,
reason=LedgerReason.PYTHON_SOURCE_DECODE_ERROR,
)
)
continue
if source_classification is None:
source_classification = resolve_python_source_classification(
path,
content,
source_classifications=source_classifications,
raw_file_cache=raw_file_cache,
)
try:
budget.check_runtime()
except _BehavioralResourceLimitError as exc:
terminal_limit = exc
ledger_events.append(_partial_limit_event(path, exc))
continue
if source_classification is PythonSourceClassification.NON_PYTHON:
continue
if content is None:
event = ledger_event(
outcome=LedgerOutcome.FAILED,
Expand Down Expand Up @@ -725,10 +788,19 @@ def node(state: SkillspectorState) -> AnalyzerNodeResponse:
)
else:
event = ledger_event(
outcome=LedgerOutcome.COMPLETED,
outcome=(
LedgerOutcome.PARTIAL
if source_classification is PythonSourceClassification.AMBIGUOUS
else LedgerOutcome.COMPLETED
),
phase="behavioral",
analyzer_id=ANALYZER_ID,
path=path,
reason=(
LedgerReason.PYTHON_SOURCE_AMBIGUOUS
if source_classification is PythonSourceClassification.AMBIGUOUS
else None
),
emitted_finding_ids=[finding.finding_id for finding in path_findings],
)
ledger_events.append(event)
Expand Down
Loading
Loading