Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
182 changes: 181 additions & 1 deletion src/skillspector/nodes/analyzers/static_patterns_anti_refusal.py
Original file line number Diff line number Diff line change
Expand Up @@ -31,9 +31,12 @@

import re
import sys
from bisect import bisect_right
from collections.abc import Callable

from skillspector.logging_config import get_logger
from skillspector.models import AnalyzerFinding, Severity
from skillspector.python_tokens import python_literal_spans
Comment thread
rng1995 marked this conversation as resolved.
from skillspector.state import AnalyzerNodeResponse, SkillspectorState

from . import static_runner
Expand All @@ -43,6 +46,9 @@
logger = get_logger(__name__)

ANALYZER_ID = "static_patterns_anti_refusal"
# Proving Python comment ownership parses the module, so it runs under the
# runner's per-artifact deadline.
USES_RUNTIME_CHECK = True

# AR1: Refusal suppression - instruct the agent to never refuse / always comply.
AR1_PATTERNS = [
Expand Down Expand Up @@ -249,6 +255,71 @@
),
)

# AR2 in descriptive Python comments.
#
# A bare "no warning(s)/disclaimer(s)/caveat(s)" is the AR2 signal in "respond with
# no warnings". In "# the server emits no warning either way" it is the object of a
# program's reported behavior: a note to developers, not an instruction to suppress
# warnings. Such a match is dropped only when ALL of the following hold:
#
# 1. The match is the bare determiner form. The "do not/don't/never" forms and every
# other AR2 pattern keep their findings.
# 2. The match lies wholly inside one proven Python comment token. The whole analyzed
# text must parse as a module, so SKILL.md, markdown, prompts, docstrings, string
# literals, other languages, fragments, and malformed source never qualify.
# 3. The words right before the match are a finite report verb from a closed allowlist,
# in third-person "-s" or past-tense form. Base forms ("give no warnings") read as
# imperatives and are not on it. The verb's subject is either a program noun from a
# closed allowlist ("the server emits"), or it is omitted because a code-behavior
# verb opens the comment ("# Returns no warning when ..."). An unlisted subject or
# verb keeps the finding, so "the assistant emits no warnings" stays active.
# 4. The contiguous comment block around the match never addresses an agent or the
# reader ("# Assistant: ...", "you", "model", "prompt", ...). The walk over that block
# is bounded, and a block longer than the bound keeps the finding.
_AR2_BARE_NO_WARNING_PATTERN = re.compile(
r"no\s+(?:any\s+)?(?:warnings?|disclaimers?|caveats?)",
re.IGNORECASE,
)
_AR2_REPORT_VERBS = (
r"contains|contained|displays|displayed|emits|emitted|generates|generated|gives|gave|"
r"has|had|includes|included|issues|issued|logs|logged|outputs|prints|printed|"
r"produces|produced|raises|raised|reports|reported|returns|returned|sends|sent|"
r"shows|showed|surfaces|surfaced|throws|threw|writes|wrote|yields|yielded"
)
_AR2_COMMENT_OPENING_REPORT_VERBS = (
r"emits|emitted|logs|logged|prints|printed|raises|raised|returns|returned|"
r"throws|threw|yields|yielded"
)
_AR2_PROGRAM_SUBJECTS = (
r"apis?|backends?|binary|binaries|builds?|calls?|checks?|cli|clients?|commands?|"
r"compilers?|daemons?|databases?|db|drivers?|endpoints?|functions?|handlers?|helpers?|"
r"interpreters?|jobs?|library|libraries|linters?|methods?|modules?|packages?|"
r"parsers?|pipelines?|process|processes|programs?|query|queries|requests?|runtimes?|"
r"scripts?|sdks?|servers?|services?|subprocess|subprocesses|tests?|tools?|upstream|"
r"validators?|wrappers?"
)
_AR2_REPORT_ADVERBS = r"(?:(?:[a-z]+ly|also|already|always|even|just|now|still|then)\s+){0,2}"
_AR2_PROGRAM_SUBJECT_REPORT_PATTERN = re.compile(
rf"(?<![\w-])(?:{_AR2_PROGRAM_SUBJECTS})\s+{_AR2_REPORT_ADVERBS}"
rf"(?:{_AR2_REPORT_VERBS})\s+\Z",
re.IGNORECASE,
)
_AR2_COMMENT_OPENING_REPORT_PATTERN = re.compile(
rf"#+[ \t]*{_AR2_REPORT_ADVERBS}(?:{_AR2_COMMENT_OPENING_REPORT_VERBS})[ \t]+",
re.IGNORECASE,
)
_AR2_AGENT_ADDRESS_PATTERN = re.compile(
r"\b(?:you|your|yours|yourself|yourselves|assistants?|agents?|models?|ai|llms?|"
r"chatbots?|bots?|claude|chatgpt|gpt|copilot|gemini|codex|personas?|prompts?|"
r"instructions?)\b",
re.IGNORECASE,
)
# A subject or comment opening, two adverbs, and a verb fit well inside this many
# characters. Bounding both grammar checks keeps a long comment line linear.
_AR2_REPORT_PREFIX_CHARS = 160
# A comment block reaching this many lines on either side of the match keeps it.
_AR2_COMMENT_BLOCK_MAX_LINES = 64


def _is_directly_instructive(context: str, matched_text: str) -> bool:
"""Return True when the match still looks like an active adversarial instruction."""
Expand Down Expand Up @@ -393,17 +464,126 @@ def _is_benign_ar_context(
)


def analyze(content: str, file_path: str, file_type: str) -> list[AnalyzerFinding]:
def _no_runtime_check() -> None:
"""Stand in for the runner deadline when analyze() is called directly."""


class _PythonComments:
"""Lazily proven comment tokens of one analyzed Python text."""

def __init__(self, content: str, check_runtime: Callable[[], None]) -> None:
self._content = content
self._check_runtime = check_runtime
self._starts: tuple[int, ...] = ()
self._ends: tuple[int, ...] = ()
self._computed = False
self._block_unaddressed: dict[int, bool] = {}

def comment_index(self, start: int, end: int) -> int | None:
"""Return the comment token that wholly contains ``[start, end)``, if proven."""
if not self._computed:
spans = python_literal_spans(self._content, self._check_runtime)
if spans is not None:
self._starts, self._ends = spans
self._computed = True
index = bisect_right(self._starts, start) - 1
if index < 0 or end > self._ends[index] or not self._is_comment(index):
return None
return index

def comment_start(self, index: int) -> int:
return self._starts[index]

def block_is_unaddressed(self, index: int) -> bool:
"""Return True when no agent or reader is addressed in the comment block."""
bounds = self._block_bounds(index)
if bounds is None:
return False
first, last = bounds
cached = self._block_unaddressed.get(first)
if cached is None:
cached = (
_AR2_AGENT_ADDRESS_PATTERN.search(
self._content, self._starts[first], self._ends[last]
)
is None
)
self._block_unaddressed[first] = cached
return cached

def _is_comment(self, index: int) -> bool:
return self._content[self._starts[index]] == "#"

def _on_adjacent_lines(self, earlier: int, later: int) -> bool:
if not (self._is_comment(earlier) and self._is_comment(later)):
return False
gap = self._content[self._ends[earlier] : self._starts[later]]
return gap.count("\n") == 1 and not gap.strip()

def _block_bounds(self, index: int) -> tuple[int, int] | None:
first = last = index
for _ in range(_AR2_COMMENT_BLOCK_MAX_LINES):
if first == 0 or not self._on_adjacent_lines(first - 1, first):
break
first -= 1
else:
return None
for _ in range(_AR2_COMMENT_BLOCK_MAX_LINES):
if last + 1 == len(self._starts) or not self._on_adjacent_lines(last, last + 1):
break
last += 1
else:
return None
return first, last


def _is_descriptive_python_comment(comments: _PythonComments, match: re.Match[str]) -> bool:
"""Return True when an AR2 match reports a program's behavior in a Python comment."""
if not _AR2_BARE_NO_WARNING_PATTERN.fullmatch(match.group(0)):
return False
index = comments.comment_index(match.start(), match.end())
if index is None:
return False
content = match.string
comment_start = comments.comment_start(index)
prefix_start = max(comment_start, match.start() - _AR2_REPORT_PREFIX_CHARS)
reports_program_behavior = bool(
_AR2_PROGRAM_SUBJECT_REPORT_PATTERN.search(content, prefix_start, match.start())
or (
prefix_start == comment_start
and _AR2_COMMENT_OPENING_REPORT_PATTERN.fullmatch(content, comment_start, match.start())
)
)
return reports_program_behavior and comments.block_is_unaddressed(index)


def analyze(
content: str,
file_path: str,
file_type: str,
check_runtime: Callable[[], None] | None = None,
) -> list[AnalyzerFinding]:
"""Analyze content for anti-refusal statements (AR1-AR3)."""
findings: list[AnalyzerFinding] = []
locations = SourceLocationIndex(content, file_path)
tag = [PatternCategory.ANTI_REFUSAL.value]
python_comments = (
_PythonComments(content, check_runtime or _no_runtime_check)
if file_type == "python"
else None
)

for rule_id, patterns in _RULES:
for pattern, base_confidence in patterns:
for match in static_runner.iter_paragraph_matches(
pattern, content, re.IGNORECASE | re.MULTILINE
):
if (
rule_id == "AR2"
and python_comments is not None
and _is_descriptive_python_comment(python_comments, match)
):
continue
lines = content.splitlines()
line_num = get_line_number(content, match.start())
match_line = lines[line_num - 1] if lines else content
Expand Down
31 changes: 21 additions & 10 deletions src/skillspector/python_tokens.py
Original file line number Diff line number Diff line change
Expand Up @@ -49,16 +49,27 @@
# comment tokens, in source order. Results are shared, so they are immutable.
PythonLiteralSpans = tuple[tuple[int, ...], tuple[int, ...]]

# One static scan of a module asks for its spans twice: the declared-marker
# pass proves string closers, and the tool-misuse shell parser proves token
# ownership. The most recent results are remembered so both share one parse.
# Entries are keyed by length and hash and confirmed by string equality, which
# is an identity check when both callers hold the same file-cache string. A
# second entry absorbs a scan of another file on a concurrent analyzer thread.
# Oversized source is never stored, and eviction bounds how long a module's
# text outlives its scan. This lock guards only the entries, so a lookup never
# waits for another thread's parse.
_PYTHON_LITERAL_SPANS_CACHE_SIZE = 2
# Three consumers ask for a module's spans during one scan: the declared-marker
# pass proves string closers, the tool-misuse shell parser proves token
# ownership, and AR2 proves comment ownership. Recent results are remembered so
# they share one parse. Entries are keyed by length and hash and confirmed by
# string equality, which is an identity check when the callers hold the same
# file-cache string.
#
# The consumers run in different analyzer nodes (the marker pass in every
# static-pattern node). The graph starts all analyzer nodes together on worker
# threads; the CLI and MCP server set no ``max_concurrency``, so the pool has
# Python's default size, min(32, CPUs + 4). Every node walks the components in
# the same order but at its own pace, so requests for other modules arrive
# between a module's first and last consumer. An entry holds one module however
# many consumers ask, so eight entries keep a result while up to seven other
# modules are requested in that gap. A wider gap only repeats the parse.
# Oversized source is never stored, so eight entries hold at most eight times
# MAX_PYTHON_AST_SOURCE_CHARS, which equals one scan's AST cache budget
# (MAX_PYTHON_AST_CACHE_SOURCE_CHARS), and eviction bounds how long that text
# outlives its scan. This lock guards only the entries, so a lookup never waits
# for another thread's parse.
_PYTHON_LITERAL_SPANS_CACHE_SIZE = 8
_PYTHON_LITERAL_SPANS_CACHE_LOCK = Lock()
_PYTHON_LITERAL_SPANS_CACHE: OrderedDict[tuple[int, int], tuple[str, PythonLiteralSpans | None]] = (
OrderedDict()
Expand Down
52 changes: 35 additions & 17 deletions tests/nodes/analyzers/test_python_string_marker_ownership.py
Original file line number Diff line number Diff line change
Expand Up @@ -26,6 +26,7 @@
from skillspector.artifacts import SecurityTextView
from skillspector.graph import graph
from skillspector.inspection_ledger import LedgerOutcome, LedgerReason
from skillspector.nodes.analyzers import static_patterns_anti_refusal as ar_module
from skillspector.nodes.analyzers import static_patterns_prompt_injection as pi_module
from skillspector.nodes.analyzers import static_patterns_tool_misuse as tm_module
from skillspector.nodes.analyzers import static_runner
Expand Down Expand Up @@ -211,9 +212,15 @@ def counted(content: str, check_runtime):

# --- One ownership parse per module ----------------------------------------

# The marker pass proves the closing quote after "omit", and the tool-misuse
# shell parser proves the fence backticks literal: both consumers ask.
_BOTH_CONSUMERS = _ASSERT_FLAG + 'FENCE = "\\n```\\n"\n' + _UNQUOTED_TAIL
# The marker pass proves the closing quote after "omit", the tool-misuse shell
# parser proves the fence backticks literal, and AR2 proves that "no warning"
# lies in a comment that reports program output: all three consumers ask.
_ALL_CONSUMERS = (
_ASSERT_FLAG
+ 'FENCE = "\\n```\\n"\n'
+ "# The server emits no warning when the cache is cold.\n"
+ _UNQUOTED_TAIL
)


def _count_parses(monkeypatch: pytest.MonkeyPatch) -> list[str]:
Expand All @@ -234,25 +241,34 @@ def _literals(content: str, spans: python_tokens.PythonLiteralSpans | None) -> l
return [content[start:end] for start, end in zip(*spans, strict=True)]


def test_marker_pass_and_shell_parser_share_one_parse_per_scan(
def test_marker_pass_shell_parser_and_ar2_share_one_parse_per_scan(
monkeypatch: pytest.MonkeyPatch,
) -> None:
parses = _count_parses(monkeypatch)
requests: list[int] = []
requests: list[tuple[str, int]] = []
real_spans = python_tokens.python_literal_spans

def requested(content: str, check_runtime: Callable[[], None]):
requests.append(len(content))
return real_spans(content, check_runtime)
def requested_by(consumer: str):
def requested(content: str, check_runtime: Callable[[], None]):
requests.append((consumer, len(content)))
return real_spans(content, check_runtime)

return requested

monkeypatch.setattr(python_tokens, "python_literal_spans", requested)
monkeypatch.setattr(tm_module, "_python_literal_spans", requested)
monkeypatch.setattr(python_tokens, "python_literal_spans", requested_by("marker"))
monkeypatch.setattr(tm_module, "_python_literal_spans", requested_by("tool-misuse"))
monkeypatch.setattr(ar_module, "python_literal_spans", requested_by("anti-refusal"))

result = _ledger("scripts/check.py", _BOTH_CONSUMERS, tm_module, pi_module)
result = _ledger("scripts/check.py", _ALL_CONSUMERS, tm_module, pi_module, ar_module)

assert result["findings"] == []
assert result["inspection_ledger"][0]["outcome"] is LedgerOutcome.COMPLETED
assert requests == [len(_BOTH_CONSUMERS)] * 2
assert [len(content) for content in parses] == [len(_BOTH_CONSUMERS)]
assert sorted(requests) == [
("anti-refusal", len(_ALL_CONSUMERS)),
("marker", len(_ALL_CONSUMERS)),
("tool-misuse", len(_ALL_CONSUMERS)),
]
assert [len(content) for content in parses] == [len(_ALL_CONSUMERS)]


def test_memo_reuses_proven_and_unproven_results(monkeypatch: pytest.MonkeyPatch) -> None:
Expand Down Expand Up @@ -324,15 +340,17 @@ def expires_after_the_module_parse() -> None:
def test_concurrent_sources_each_get_their_own_spans() -> None:
# Analyzer nodes run on worker threads. More sources than memo entries
# keep entries being replaced while other threads look them up.
cache_size = python_tokens._PYTHON_LITERAL_SPANS_CACHE_SIZE
proven_count = cache_size + 2
sources = [
"".join(f'value_{index} = "{worker}-{index}" # w{worker}\n' for index in range(worker + 1))
for worker in range(6)
for worker in range(proven_count)
] + ['broken = "x" + $(\n']
expected = {
source: python_tokens._python_literal_spans_uncached(source, lambda: None)
for source in sources
}
assert [spans is None for spans in expected.values()] == [False] * 6 + [True]
assert [spans is None for spans in expected.values()] == [False] * proven_count + [True]
barrier = threading.Barrier(8)
correct: list[bool] = []

Expand All @@ -351,8 +369,8 @@ def scan(worker: int) -> None:
thread.join()

assert correct == [True] * 480
cache_size = python_tokens._PYTHON_LITERAL_SPANS_CACHE_SIZE
assert len(python_tokens._PYTHON_LITERAL_SPANS_CACHE) <= cache_size
# More distinct sources than entries were stored, so eviction left it full.
assert len(python_tokens._PYTHON_LITERAL_SPANS_CACHE) == cache_size


# --- Readings that stay fail-closed ----------------------------------------
Expand Down
Loading
Loading