Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
25 commits
Select commit Hold shift + click to select a range
71d497a
fix(batch): preserve malformed response failures in compatibility mode
yashrajp22 Oct 7, 2026
c7d664c
fix(batch): preserve analyzer deadlines through compatibility initial…
yashrajp22 Oct 7, 2026
0604e45
fix(meta): reject invalid stringified response fields
yashrajp22 Oct 7, 2026
d149e59
test(meta): import result schema for validation regressions
yashrajp22 Oct 7, 2026
9c7bdfc
test(meta): enable parametrized invalid response coverage
yashrajp22 Oct 7, 2026
8a4288d
Update meta response validation expectations
yashrajp22 Oct 7, 2026
d6d36f2
Format malformed response regressions
yashrajp22 Oct 7, 2026
7ed4487
Merge branch 'main' into yashraj/fix-batch-compat-parse-failures
github-actions[bot] Oct 7, 2026
dc45f5c
Merge branch 'main' into yashraj/fix-batch-compat-parse-failures
github-actions[bot] Oct 8, 2026
107b072
Merge branch 'main' into yashraj/fix-batch-compat-parse-failures
github-actions[bot] Oct 8, 2026
d1bcf2c
Merge branch 'main' into yashraj/fix-batch-compat-parse-failures
github-actions[bot] Oct 8, 2026
6c047b0
Merge branch 'main' into yashraj/fix-batch-compat-parse-failures
github-actions[bot] Oct 8, 2026
12fde02
fix: preserve valid compat verdicts and guard timeout forwarding
yashrajp22 Oct 9, 2026
88fd991
fix: respect shorter compatibility request deadlines
yashrajp22 Oct 9, 2026
54cc278
fix: bound pooled request waiting and retries by one deadline
yashrajp22 Oct 9, 2026
5c309f9
style: format compatibility regression checks
yashrajp22 Oct 9, 2026
dffb219
Merge branch 'main' into yashraj/fix-batch-compat-parse-failures
github-actions[bot] Oct 9, 2026
2c2afa3
Merge branch 'main' into yashraj/fix-batch-compat-parse-failures
github-actions[bot] Oct 9, 2026
73c0d23
Merge branch 'main' into yashraj/fix-batch-compat-parse-failures
github-actions[bot] Oct 9, 2026
2be7b1f
Merge branch 'main' into yashraj/fix-batch-compat-parse-failures
github-actions[bot] Oct 9, 2026
169dff8
Merge remote-tracking branch 'origin/yashraj/fix-batch-compat-parse-f…
yashrajp22 Oct 9, 2026
df8ff55
Merge branch 'main' into yashraj/fix-batch-compat-parse-failures
github-actions[bot] Oct 9, 2026
8219589
Merge remote-tracking branch 'origin/yashraj/fix-batch-compat-parse-f…
yashrajp22 Oct 9, 2026
7c11466
Merge branch 'main' into yashraj/fix-batch-compat-parse-failures
github-actions[bot] Oct 9, 2026
581ded0
Merge branch 'main' into yashraj/fix-batch-compat-parse-failures
github-actions[bot] Oct 9, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
125 changes: 57 additions & 68 deletions contrib/batch_scan/api_pool.py
Original file line number Diff line number Diff line change
Expand Up @@ -475,106 +475,95 @@ async def ainvoke_with_usage(self, prompt: str, collector: object) -> object:

# -- Internal -------------------------------------------------------------

@staticmethod
def _remaining(deadline: float) -> float:
from skillspector.llm_analyzer_base import LLMRuntimeLimitError

remaining = deadline - time.monotonic()
if remaining <= 0:
raise LLMRuntimeLimitError("pooled request runtime limit reached")
return remaining

def _invoke_with_retry(
self,
prompt: str,
*,
callbacks: list[object] | None = None,
self, prompt: str, *, callbacks: list[object] | None = None,
) -> object:
"""Sync retry loop — acquire slot, call LLM, release, retry on 429."""
last_exception: Exception | None = None

"""Share one deadline across slot waiting, requests, and key retries."""
deadline = time.monotonic() + self._timeout
for attempt in range(self._max_retries + 1):
key = self._pool.acquire()
llm = self._build_llm(key)
try:
if callbacks is None:
result = llm.invoke(prompt)
else:
result = llm.invoke(prompt, config={"callbacks": callbacks})
self._pool.release(key, success=True)
if attempt > 0:
self._pool.record_retry_success()
return result
key = self._pool.acquire(timeout=self._remaining(deadline))
except Exception:
self._remaining(deadline)
raise
rate_limited = False
try:
llm = self._build_llm(key, timeout=self._remaining(deadline))
result = llm.invoke(prompt) if callbacks is None else llm.invoke(
prompt, config={"callbacks": callbacks}
)
except Exception as exc:
if self._is_rate_limit(exc) and attempt < self._max_retries:
self._pool.release(key, success=False)
logger.debug(
"PooledChatModel: rate-limited, retrying "
"(attempt %d/%d)",
attempt + 1,
self._max_retries,
)
rate_limited = self._is_rate_limit(exc)
if rate_limited and attempt < self._max_retries:
continue
self._pool.release(key, success=True)
last_exception = exc
raise

raise RuntimeError(
f"PooledChatModel: exhausted {self._max_retries} retries "
"due to rate-limit errors"
) from last_exception
finally:
self._pool.release(key, success=not rate_limited)
if attempt > 0:
self._pool.record_retry_success()
return result
raise RuntimeError("PooledChatModel exhausted key retries")

async def _ainvoke_with_retry(
self,
prompt: str,
*,
callbacks: list[object] | None = None,
self, prompt: str, *, callbacks: list[object] | None = None,
) -> object:
"""Async retry loop — non-blocking acquire first, block only if full."""
"""Bound asynchronous slot waiting without leaving an acquiring thread."""
import asyncio
last_exception: Exception | None = None

deadline = time.monotonic() + self._timeout
for attempt in range(self._max_retries + 1):
self._remaining(deadline)
key = self._pool.try_acquire()
if key is None:
key = await asyncio.to_thread(self._pool.acquire)
llm = self._build_llm(key)
while key is None:
await asyncio.sleep(min(0.05, self._remaining(deadline)))
self._remaining(deadline)
key = self._pool.try_acquire()
rate_limited = False
try:
if callbacks is None:
result = await llm.ainvoke(prompt)
else:
result = await llm.ainvoke(prompt, config={"callbacks": callbacks})
self._pool.release(key, success=True)
if attempt > 0:
self._pool.record_retry_success()
return result
llm = self._build_llm(key, timeout=self._remaining(deadline))
result = await llm.ainvoke(prompt) if callbacks is None else await llm.ainvoke(
prompt, config={"callbacks": callbacks}
)
except Exception as exc:
if self._is_rate_limit(exc) and attempt < self._max_retries:
self._pool.release(key, success=False)
logger.debug(
"PooledChatModel: rate-limited, retrying "
"(attempt %d/%d)",
attempt + 1,
self._max_retries,
)
rate_limited = self._is_rate_limit(exc)
if rate_limited and attempt < self._max_retries:
continue
self._pool.release(key, success=True)
last_exception = exc
raise

raise RuntimeError(
f"PooledChatModel: exhausted {self._max_retries} retries "
"due to rate-limit errors"
) from last_exception

def _build_llm(self, key: ApiKey):
finally:
self._pool.release(key, success=not rate_limited)
if attempt > 0:
self._pool.record_retry_success()
return result
raise RuntimeError("PooledChatModel exhausted key retries")

def _build_llm(self, key: ApiKey, *, timeout: float | None = None):
"""Build a fresh :class:`~langchain_openai.ChatOpenAI` for *key*."""
from langchain_openai import ChatOpenAI
from pydantic import SecretStr

timeout = self._timeout if timeout is None else timeout
try:
import httpx
_timeout = httpx.Timeout(self._timeout, connect=8.0)
_timeout = httpx.Timeout(timeout, connect=min(8.0, timeout))
except ImportError:
_timeout = self._timeout
_timeout = timeout

return ChatOpenAI(
model=key.model,
base_url=key.base_url,
api_key=SecretStr(key.key),
max_completion_tokens=self._max_tokens,
timeout=_timeout,
max_retries=0,
)

@staticmethod
Expand Down
32 changes: 16 additions & 16 deletions contrib/batch_scan/docs/DESIGN.md
Original file line number Diff line number Diff line change
Expand Up @@ -284,22 +284,22 @@ raw LLM string → _strip_markdown_fences() → json.loads() → model_validate(
The two-step parse (stdlib `json.loads` then Pydantic `model_validate`) exists
because:

1. `json.loads` is fast, deterministic, and raises clear `JSONDecodeError` on
malformed output — we catch this and return `[]` (empty findings).
2. `model_validate` enforces the schema: required fields, literal enums,
confidence range, string length. Schema violations are caught and returned
as `[]` with a warning log.

**Error propagation:** If the LLM returns invalid JSON or schema-mismatched
output, the analyzer returns `[]` (no findings for that file). The scan
continues — a single malformed LLM response never blocks the pipeline.
The warning is logged at `WARNING` level so operators can monitor parse-failure
rates without sifting through debug logs.

Patch 3 adds a `_sanitize_meta_finding()` pass after validation to handle
known LLM quirks: `null` string fields → `""`, unrecognized enum values
(e.g., `"none"`) → `"low"`. These are applied post-validation because they
represent recoverable soft errors, not hard schema violations.
1. `json.loads` parses the response. Invalid JSON raises the core structured-response
validation error rather than producing an empty findings list.
2. `model_validate` checks the required fields and value types. Schema failures
use the same retry and reporting path.

**Error propagation:** Invalid JSON and schema failures are retried up to four
attempts. If they persist, the ledger records `skipped` with reason
`llm_structured_response_invalid`; the analyzer is degraded and the skill counts
as incomplete. The scan continues and keeps findings from completed work.
Core logs `LLM structured response validation failed for ... retrying` and,
on exhaustion, `... after 4 attempts`, without logging the raw response.

Patch 3 sanitizes known soft quirks before validation: null explanation and
remediation become empty strings, impact labels are case-folded, and unknown
impact labels use `low`. Invalid findings still fail validation. Optional prose
in `overall_assessment` is ignored so valid finding verdicts are retained.

## Gap-Fill Rule Selection Criteria

Expand Down
9 changes: 5 additions & 4 deletions contrib/batch_scan/docs/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -341,10 +341,11 @@ fi
classified as English and lose gap-fill coverage.
3. **No SARIF output.** Upstream supports it; this contrib adds terminal/JSON/Markdown.
4. **Gap-fill quality not benchmarked for non-English.** No ground-truth comparison exists.
5. **`parse_response` JSON recovery is best-effort.** When the LLM returns
malformed JSON, the analyzer returns empty findings (no crash). This is a
graceful-degradation choice: a single malformed response won't block the
pipeline, but the user won't know which findings were lost.
5. **Malformed provider responses reduce coverage.** Compatibility-mode discovery
and meta responses are retried up to four attempts, then recorded as
`llm_structured_response_invalid` and counted as incomplete. Valid findings
from completed batches are kept. Gap-fill parsing still drops malformed
responses until the separate gap-fill failure fix is included.

See `DESIGN.md` for architecture details and `docs/archive/FUTURE_WORK.md` for suggested directions.

Expand Down
Loading