Skip to content

Scanners

Vulnerability scanners that query OSV, NVD, and other databases.

agent_bom.scanners

Vulnerability scanning — re-exports for agent_bom.scanners public surface.

IncompleteScanError

Bases: RuntimeError

Raised when a scan cannot produce a trustworthy verdict.

Source code in src/agent_bom/scanners/package_scan.py
class IncompleteScanError(RuntimeError):
    """Raised when a scan cannot produce a trustworthy verdict."""

ScanOptions dataclass

Immutable per-scan scanner controls.

The legacy module-level setters remain for CLI compatibility, but request handlers and concurrent scan callers should pass explicit options so tenant policy cannot bleed through shared module state.

Source code in src/agent_bom/scanners/package_scan.py
@dataclass(frozen=True)
class ScanOptions:
    """Immutable per-scan scanner controls.

    The legacy module-level setters remain for CLI compatibility, but request
    handlers and concurrent scan callers should pass explicit options so tenant
    policy cannot bleed through shared module state.
    """

    offline: bool = False
    compliance_enabled: bool = False
    resolve_transitive: bool = False
    prefer_local_db: bool = False
    demo_advisories: bool = False
    # Target project directory (the ``-p`` path) for local install resolution.
    # When set, bare/floating versions are resolved against the target — its
    # virtualenv for pip, the dir itself for npm/go — never the scanner host.
    project_dir: Optional[str] = None

ScannerExecutionState

Bases: str, Enum

Whether a driver is executable today or declared as a roadmap slot.

Source code in src/agent_bom/scanners/base.py
class ScannerExecutionState(str, Enum):
    """Whether a driver is executable today or declared as a roadmap slot."""

    ACTIVE = "active"
    PASSIVE = "passive"
    PLANNED = "planned"

ScannerFailureMode

Bases: str, Enum

How the orchestrator should treat driver failures.

Source code in src/agent_bom/scanners/base.py
class ScannerFailureMode(str, Enum):
    """How the orchestrator should treat driver failures."""

    FAIL_CLOSED = "fail_closed"
    WARN_AND_CONTINUE = "warn_and_continue"
    SKIP_WHEN_UNAVAILABLE = "skip_when_unavailable"

ScannerPhase

Bases: str, Enum

Pipeline phase where a scanner driver contributes evidence.

Source code in src/agent_bom/scanners/base.py
class ScannerPhase(str, Enum):
    """Pipeline phase where a scanner driver contributes evidence."""

    DISCOVERY = "discovery"
    EXTRACTION = "extraction"
    SCANNING = "scanning"
    ENRICHMENT = "enrichment"
    ANALYSIS = "analysis"
    OUTPUT = "output"

ScannerRegistration dataclass

Bases: RegistryEntry

Registry metadata for a scanner driver implementation.

Source code in src/agent_bom/scanners/base.py
@dataclass(frozen=True)
class ScannerRegistration(RegistryEntry):
    """Registry metadata for a scanner driver implementation."""

    phase: ScannerPhase = ScannerPhase.SCANNING
    execution_state: ScannerExecutionState = ScannerExecutionState.ACTIVE
    failure_mode: ScannerFailureMode = ScannerFailureMode.WARN_AND_CONTINUE
    enabled_by_default: bool = True
    run_attr: str = ""
    input_types: tuple[str, ...] = ()
    output_types: tuple[str, ...] = ()
    finding_types: tuple[str, ...] = ()
    skip_when: tuple[str, ...] = ()
    telemetry_keys: tuple[str, ...] = ()
    standards: tuple[str, ...] = ()
    summary: str = ""

    def to_dict(self) -> dict[str, Any]:
        """Return a stable API/CLI-safe representation."""

        return {
            "name": self.name,
            "module": self.module,
            "source": self.source,
            "phase": self.phase.value,
            "execution_state": self.execution_state.value,
            "failure_mode": self.failure_mode.value,
            "enabled_by_default": self.enabled_by_default,
            "run_attr": self.run_attr,
            "input_types": list(self.input_types),
            "output_types": list(self.output_types),
            "finding_types": list(self.finding_types),
            "skip_when": list(self.skip_when),
            "telemetry_keys": list(self.telemetry_keys),
            "standards": list(self.standards),
            "summary": self.summary,
            "capabilities": {
                "scan_modes": list(self.capabilities.scan_modes),
                "required_scopes": list(self.capabilities.required_scopes),
                "permissions_used": list(self.capabilities.permissions_used),
                "outbound_destinations": list(self.capabilities.outbound_destinations),
                "data_boundary": self.capabilities.data_boundary,
                "writes": self.capabilities.writes,
                "network_access": self.capabilities.network_access,
                "guarantees": list(self.capabilities.guarantees),
            },
        }

to_dict

to_dict() -> dict[str, Any]

Return a stable API/CLI-safe representation.

Source code in src/agent_bom/scanners/base.py
def to_dict(self) -> dict[str, Any]:
    """Return a stable API/CLI-safe representation."""

    return {
        "name": self.name,
        "module": self.module,
        "source": self.source,
        "phase": self.phase.value,
        "execution_state": self.execution_state.value,
        "failure_mode": self.failure_mode.value,
        "enabled_by_default": self.enabled_by_default,
        "run_attr": self.run_attr,
        "input_types": list(self.input_types),
        "output_types": list(self.output_types),
        "finding_types": list(self.finding_types),
        "skip_when": list(self.skip_when),
        "telemetry_keys": list(self.telemetry_keys),
        "standards": list(self.standards),
        "summary": self.summary,
        "capabilities": {
            "scan_modes": list(self.capabilities.scan_modes),
            "required_scopes": list(self.capabilities.required_scopes),
            "permissions_used": list(self.capabilities.permissions_used),
            "outbound_destinations": list(self.capabilities.outbound_destinations),
            "data_boundary": self.capabilities.data_boundary,
            "writes": self.capabilities.writes,
            "network_access": self.capabilities.network_access,
            "guarantees": list(self.capabilities.guarantees),
        },
    }

expand_blast_radius_hops

expand_blast_radius_hops(blast_radii: list[BlastRadius], agents: list[Agent], max_depth: int = 1) -> None

Expand blast radii with multi-hop delegation chain analysis.

Source code in src/agent_bom/scanners/blast_radius.py
def expand_blast_radius_hops(
    blast_radii: list[BlastRadius],
    agents: list[Agent],
    max_depth: int = 1,
) -> None:
    """Expand blast radii with multi-hop delegation chain analysis."""
    max_depth = max(1, min(max_depth, 5))
    if max_depth <= 1:
        return

    server_to_agents: dict[str, list[Agent]] = {}
    for agent in agents:
        for server in agent.mcp_servers:
            server_to_agents.setdefault(server.name, []).append(agent)

    agent_to_servers: dict[str, list[str]] = {}
    for agent in agents:
        agent_to_servers[agent.name] = [server.name for server in agent.mcp_servers]

    for blast_radius in blast_radii:
        direct_agent_names = {agent.name for agent in blast_radius.affected_agents}
        direct_server_names = {server.name for server in blast_radius.affected_servers}

        visited_agents: set[str] = set(direct_agent_names)
        visited_servers: set[str] = set(direct_server_names)
        transitive_agents: list[dict] = []
        transitive_credentials: list[str] = []
        chains: list[str] = []

        queue: list[tuple[str, int, list[str]]] = []
        for agent in blast_radius.affected_agents:
            for server_name in agent_to_servers.get(agent.name, []):
                if server_name not in direct_server_names:
                    queue.append((agent.name, 1, [agent.name, server_name]))
                    visited_servers.add(server_name)

        max_hop_reached = 1
        while queue:
            _agent_name, hop, chain = queue.pop(0)
            if hop >= max_depth:
                continue

            current_server = chain[-1]
            for next_agent in server_to_agents.get(current_server, []):
                if next_agent.name in visited_agents:
                    continue
                visited_agents.add(next_agent.name)
                next_hop = hop + 1
                max_hop_reached = max(max_hop_reached, next_hop)

                new_chain = chain + [next_agent.name]
                chain_str = "\u2192".join(new_chain)
                chains.append(chain_str)

                agent_creds: list[str] = []
                for server in next_agent.mcp_servers:
                    agent_creds.extend(server.credential_names)
                agent_creds = list(set(agent_creds))

                transitive_agents.append(
                    {
                        "name": next_agent.name,
                        "type": next_agent.agent_type.value,
                        "hop": next_hop,
                        "chain": chain_str,
                    }
                )
                transitive_credentials.extend(agent_creds)

                if next_hop < max_depth:
                    for server_name in agent_to_servers.get(next_agent.name, []):
                        if server_name not in visited_servers:
                            visited_servers.add(server_name)
                            queue.append((next_agent.name, next_hop, new_chain + [server_name]))

        if transitive_agents:
            blast_radius.hop_depth = max_hop_reached
            blast_radius.delegation_chain = chains
            blast_radius.transitive_agents = transitive_agents
            blast_radius.transitive_credentials = list(set(transitive_credentials))
            factor = _HOP_RISK_FACTORS.get(max_hop_reached, 0.25)
            blast_radius.transitive_risk_score = round(blast_radius.risk_score * factor, 2)

build_vulnerabilities

build_vulnerabilities(vuln_data_list: list[dict], package: Package) -> list[Vulnerability]

Convert OSV response data to Vulnerability objects.

Filters out false positives by verifying the package version falls within OSV affected ranges. Deduplicates by canonical CVE ID.

Source code in src/agent_bom/scanners/package_scan.py
def build_vulnerabilities(vuln_data_list: list[dict], package: Package) -> list[Vulnerability]:
    """Convert OSV response data to Vulnerability objects.

    Filters out false positives by verifying the package version falls
    within OSV affected ranges.  Deduplicates by canonical CVE ID.
    """
    vulns = []

    for raw_vuln_data in vuln_data_list:
        vuln_data = _scope_advisory_to_package_release(raw_vuln_data, package)
        if vuln_data is None:
            continue
        vuln_id = vuln_data.get("id", "unknown")

        # Version-range filter: skip vulns that don't affect our version
        if package.version and package.version not in ("unknown", "latest"):
            if not _is_version_affected(
                vuln_data,
                package.name,
                package.version,
                package.ecosystem,
                source_package=package.source_package,
                aliases=tuple(package.lookup_names),
            ):
                _logger.debug(
                    "Filtered %s: version %s not in affected range for %s",
                    vuln_id,
                    package.version,
                    package.name,
                )
                continue

        from agent_bom.advisory_ids import canonical_vulnerability_id, match_confidence_tier

        aliases = vuln_data.get("aliases", [])
        canonical_id, all_aliases = canonical_vulnerability_id(vuln_id, aliases)

        severity, cvss_score, sev_source = parse_osv_severity(vuln_data)
        cvss_vector = osv_cvss_vector(vuln_data) if cvss_score is not None else None
        fixed = parse_fixed_version(
            vuln_data,
            package.name,
            package.ecosystem,
            current_version=package.version or "",
            source_package=package.source_package,
            aliases=tuple(package.lookup_names),
        )

        references = [ref.get("url", "") for ref in vuln_data.get("references", []) if ref.get("url")]

        summary = vuln_data.get("summary", vuln_data.get("details", "No description available"))[:200]

        # Extract CWE IDs from database_specific (GHSA entries store them here)
        cwe_ids: list[str] = []
        db_specific = vuln_data.get("database_specific", {})
        if isinstance(db_specific, dict):
            raw_cwes = db_specific.get("cwe_ids", [])
            if isinstance(raw_cwes, list):
                cwe_ids = [c for c in raw_cwes if isinstance(c, str) and c.startswith("CWE-")]

        affected_symbols = advisory_affected_symbols_list(vuln_data)
        affected_symbols_by_path = advisory_affected_symbols_by_path(vuln_data)

        vulns.append(
            Vulnerability(
                id=canonical_id,
                summary=summary,
                severity=severity,
                severity_source=sev_source,
                cvss_score=cvss_score,
                cvss_vector=cvss_vector,
                fixed_version=fixed,
                references=references,
                published_at=vuln_data.get("published"),
                modified_at=vuln_data.get("modified"),
                upstream_ids=upstream_advisory_ids(vuln_data.get("upstream")),
                aliases=all_aliases,
                cwe_ids=cwe_ids,
                affected_symbols=affected_symbols,
                affected_symbols_by_path=affected_symbols_by_path,
                advisory_sources=["osv"],
                match_confidence_tier=match_confidence_tier(
                    advisory_source="osv",
                    db_ecosystem=None,
                    package_ecosystem=package.ecosystem,
                    fixed_version=fixed,
                ),
            )
        )

    from agent_bom.scanners.advisory_merge import merge_advisory_clusters

    vulns = merge_advisory_clusters(vulns)
    _apply_distro_release_ambiguity(package, vulns)
    return vulns

create_client

create_client(timeout: float | None = None, max_redirects: int = 0, *, cert: str | tuple[str, str] | None = None, verify: bool | str = True) -> httpx.AsyncClient

Create an httpx.AsyncClient with connection-level retries.

Uses httpx's built-in transport retry for connection failures (DNS, TCP reset). Application-level retries (429, 5xx) are handled by request_with_retry.

Parameters:

Name Type Description Default
timeout float | None

Per-request timeout in seconds.

None
max_redirects int

Maximum redirects available if a caller explicitly enables redirects on a request. Redirect following is disabled by default so SSRF validation cannot be bypassed by a Location header.

0
Source code in src/agent_bom/http_client.py
def create_client(
    timeout: float | None = None,
    max_redirects: int = 0,
    *,
    cert: str | tuple[str, str] | None = None,
    verify: bool | str = True,
) -> httpx.AsyncClient:
    """Create an httpx.AsyncClient with connection-level retries.

    Uses httpx's built-in transport retry for connection failures (DNS, TCP reset).
    Application-level retries (429, 5xx) are handled by ``request_with_retry``.

    Args:
        timeout: Per-request timeout in seconds.
        max_redirects: Maximum redirects available if a caller explicitly
            enables redirects on a request. Redirect following is disabled by
            default so SSRF validation cannot be bypassed by a Location header.
    """
    check_offline()
    from agent_bom.config import HTTP_DEFAULT_TIMEOUT

    if timeout is None:
        timeout = HTTP_DEFAULT_TIMEOUT
    tls_context = _verified_tls_context(verify, cert)
    transport = None if _env_proxy_configured() else httpx.AsyncHTTPTransport(retries=2, verify=tls_context)
    return httpx.AsyncClient(
        timeout=timeout,
        transport=transport,
        follow_redirects=False,
        max_redirects=max_redirects,
        verify=tls_context,
    )

deduplicate_packages

deduplicate_packages(packages: list) -> list

Remove duplicate packages across discovery sources.

Deduplicates by (ecosystem, normalized_name, version) fingerprint. When duplicates exist, the first occurrence is kept (preserves source ordering).

This prevents redundant OSV API calls and duplicate vulnerability findings when the same package is discovered from multiple sources (local, K8s, cloud).

Parameters:

Name Type Description Default
packages list

List of Package objects from one or more discovery sources.

required

Returns:

Type Description
list

Deduplicated list, preserving first-seen order.

Source code in src/agent_bom/scanners/package_scan.py
def deduplicate_packages(packages: list) -> list:
    """Remove duplicate packages across discovery sources.

    Deduplicates by (ecosystem, normalized_name, version) fingerprint.
    When duplicates exist, the first occurrence is kept (preserves source ordering).

    This prevents redundant OSV API calls and duplicate vulnerability findings
    when the same package is discovered from multiple sources (local, K8s, cloud).

    Args:
        packages: List of Package objects from one or more discovery sources.

    Returns:
        Deduplicated list, preserving first-seen order.
    """
    seen: set[tuple[str, ...]] = set()
    result = []
    for pkg in packages:
        # Use normalized name for dedup (PEP 503: torch == Torch == pytorch)
        name = getattr(pkg, "name", "") or ""
        ecosystem = getattr(pkg, "ecosystem", "") or ""
        version = getattr(pkg, "version", "") or ""
        key = (*canonical_package_identity(name, version, ecosystem, getattr(pkg, "purl", None)), runtime_cve.runtime_assessment_key(pkg))
        if key not in seen:
            seen.add(key)
            result.append(pkg)
    return result

default_scan_options

default_scan_options(*, compliance_enabled: bool = False, resolve_transitive: bool = False, prefer_local_db: bool | None = None, offline: bool | None = None, demo_advisories: bool = False, project_dir: str | None = None) -> ScanOptions

Build per-scan options while preserving legacy offline defaults.

Source code in src/agent_bom/scanners/package_scan.py
def default_scan_options(
    *,
    compliance_enabled: bool = False,
    resolve_transitive: bool = False,
    prefer_local_db: bool | None = None,
    offline: bool | None = None,
    demo_advisories: bool = False,
    project_dir: str | None = None,
) -> ScanOptions:
    """Build per-scan options while preserving legacy offline defaults."""

    return ScanOptions(
        offline=_scanners_patchable("offline_mode") if offline is None else offline,
        compliance_enabled=compliance_enabled,
        resolve_transitive=resolve_transitive,
        prefer_local_db=(prefer_local_db if prefer_local_db is not None else _scanners_patchable("prefer_local_db")),
        demo_advisories=demo_advisories,
        project_dir=project_dir,
    )

query_osv_batch async

query_osv_batch(packages: list[Package]) -> dict[str, list[dict]]

Query OSV API for vulnerabilities in batch.

Source code in src/agent_bom/scanners/package_scan.py
async def query_osv_batch(packages: list[Package]) -> dict[str, list[dict]]:
    """Query OSV API for vulnerabilities in batch."""
    return await query_osv_batch_impl(
        packages,
        console=console,
        get_scan_cache=_scanners_patchable("_get_scan_cache"),
        get_api_semaphore=_get_api_semaphore,
        bump_scan_perf=_bump_scan_perf,
        enrich_results_if_needed_fn=_scanners_patchable("_enrich_results_if_needed"),
        record_scan_warning=_scanners_patchable("record_scan_warning"),
        osv_ecosystems_for_package=_osv_ecosystems_for_package,
        non_osv_ecosystems=_NON_OSV_ECOSYSTEMS,
        create_client_fn=_scanners_patchable("create_client"),
        request_with_retry_fn=_scanners_patchable("request_with_retry"),
    )

request_with_retry async

request_with_retry(client: AsyncClient, method: str, url: str, max_retries: int = MAX_RETRIES, **kwargs: Any) -> Optional[httpx.Response]

Make an HTTP request with exponential backoff on retryable errors.

Handles: - 429 Too Many Requests (respects Retry-After header) - 5xx server errors - Connection timeouts and network errors

Returns:

Type Description
Optional[Response]

httpx.Response on success, None on exhausted retries.

Source code in src/agent_bom/http_client.py
async def request_with_retry(
    client: httpx.AsyncClient,
    method: str,
    url: str,
    max_retries: int = MAX_RETRIES,
    **kwargs: Any,
) -> Optional[httpx.Response]:
    """Make an HTTP request with exponential backoff on retryable errors.

    Handles:
    - 429 Too Many Requests (respects Retry-After header)
    - 5xx server errors
    - Connection timeouts and network errors

    Returns:
        httpx.Response on success, None on exhausted retries.
    """
    check_offline()
    # Defense-in-depth: validate and re-derive the URL at the transport layer.
    # validate_url() raises SecurityError on SSRF attempts (private IPs,
    # localhost, metadata endpoints, non-HTTPS, DNS rebinding).
    # Re-constructing the URL from parsed components ensures CodeQL sees
    # the taint is broken.
    from urllib.parse import urlparse, urlunparse

    from agent_bom.security import validate_url as _validate_url  # noqa: E402

    _validate_url(url)  # raises SecurityError on SSRF attempts
    # Re-derive URL from parsed components to break CodeQL taint chain
    _parsed = urlparse(url)
    safe_url = urlunparse(_parsed)

    log_url = _safe_url(safe_url)
    host = _host_of(safe_url)
    backoff = INITIAL_BACKOFF

    # Breaker already open for this host: skip the network entirely so the
    # caller falls through to cached/bundled data without backoff or warnings.
    if host and registry_breaker_tripped(host):
        logger.debug("Rate-limit breaker open for %s — skipping live request to %s", host, log_url)
        return None

    for attempt in range(max_retries + 1):
        try:
            response = await client.request(method, safe_url, **kwargs)

            if not _should_retry_status(response.status_code, safe_url):
                _record_non_rate_limited(host)
                return response

            # Sustained 429s trip the per-host breaker: stop retrying this host
            # immediately and return the 429 so the caller can fall back fast.
            if response.status_code == 429 and _record_rate_limit(host):
                logger.debug("Rate-limit breaker tripped for %s on HTTP 429 — short-circuiting %s", host, log_url)
                return response

            # Retryable status — check Retry-After header
            retry_after = response.headers.get("Retry-After")
            if retry_after:
                try:
                    wait = min(float(retry_after), MAX_BACKOFF)
                except ValueError:
                    wait = backoff
            else:
                wait = backoff
            wait = _jittered_wait(wait)

            if attempt < max_retries:
                logger.info(
                    "HTTP %d from %s — retry %d/%d in %.1fs",
                    response.status_code,
                    log_url,
                    attempt + 1,
                    max_retries,
                    wait,
                )
                await asyncio.sleep(wait)
                backoff = min(backoff * 2, MAX_BACKOFF)
            else:
                logger.warning(
                    "HTTP %d from %s — exhausted %d retries",
                    response.status_code,
                    log_url,
                    max_retries,
                )
                return response

        except httpx.TimeoutException:
            if attempt < max_retries:
                wait = _jittered_wait(backoff)
                logger.info(
                    "Timeout on %s — retry %d/%d in %.1fs",
                    log_url,
                    attempt + 1,
                    max_retries,
                    wait,
                )
                await asyncio.sleep(wait)
                backoff = min(backoff * 2, MAX_BACKOFF)
            else:
                logger.warning("Timeout on %s — exhausted %d retries", log_url, max_retries)
                return None

        except httpx.HTTPError as e:
            safe_err = _sanitize_for_log(e)
            if attempt < max_retries:
                wait = _jittered_wait(backoff)
                logger.info(
                    "HTTP error on %s: %s — retry %d/%d in %.1fs",
                    log_url,
                    safe_err,
                    attempt + 1,
                    max_retries,
                    wait,
                )
                await asyncio.sleep(wait)
                backoff = min(backoff * 2, MAX_BACKOFF)
            else:
                logger.warning("HTTP error on %s: %s — exhausted %d retries", log_url, safe_err, max_retries)
                return None

    return None

scan_agents async

scan_agents(agents: list[Agent], *, compliance_enabled: bool = False, resolve_transitive: bool = False, show_scan_banner: bool = True, options: ScanOptions | None = None) -> list[BlastRadius]

Scan all agents' MCP server packages for vulnerabilities.

Source code in src/agent_bom/scanners/package_scan.py
async def scan_agents(
    agents: list[Agent],
    *,
    compliance_enabled: bool = False,
    resolve_transitive: bool = False,
    show_scan_banner: bool = True,
    options: ScanOptions | None = None,
) -> list[BlastRadius]:
    """Scan all agents' MCP server packages for vulnerabilities."""
    scan_options = options or default_scan_options(
        compliance_enabled=compliance_enabled,
        resolve_transitive=resolve_transitive,
    )
    if show_scan_banner:
        from agent_bom.output.brand_tokens import PRODUCT_NAME

        console.print(f"\n[bold cyan]{PRODUCT_NAME}[/bold cyan]  [bold]Scanning for vulnerabilities…[/bold]\n")

    _pkg_key = runtime_cve.advisory_instance_key

    # Collect all unique packages
    all_packages = []
    pkg_to_servers: dict[str, list[MCPServer]] = {}
    pkg_to_agents: dict[str, list[Agent]] = {}

    for agent in agents:
        for server in agent.mcp_servers:
            for pkg in server.packages:
                key = _pkg_key(pkg)
                all_packages.append(pkg)

                if key not in pkg_to_servers:
                    pkg_to_servers[key] = []
                pkg_to_servers[key].append(server)

                if key not in pkg_to_agents:
                    pkg_to_agents[key] = []
                if agent not in pkg_to_agents[key]:
                    pkg_to_agents[key].append(agent)

    # Deduplicate packages for scanning — uses canonical deduplicate_packages()
    # which normalizes by (ecosystem, normalized_name, version) fingerprint.
    unique_packages = deduplicate_packages(all_packages)

    if show_scan_banner:
        console.print(f"  Scanning {len(unique_packages)} unique packages across {len(agents)} agent(s)...")

    total_vulns = await _scanners_patchable("scan_packages")(unique_packages, options=scan_options)
    runtime_cve.propagate_runtime_assessments(all_packages, unique_packages)

    # Propagate vulnerabilities back to all instances
    vuln_map = {}
    for pkg in unique_packages:
        if pkg.vulnerabilities:
            vuln_map[_pkg_key(pkg)] = pkg.vulnerabilities

    for agent in agents:
        for server in agent.mcp_servers:
            for pkg in server.packages:
                if _pkg_key(pkg) in vuln_map:
                    pkg.vulnerabilities = vuln_map[_pkg_key(pkg)]

    # Build blast radius analysis. Registry enrichment is keyed by MCP server,
    # not by vulnerable package. Keep one cache for the whole scan: a server can
    # expose many packages, and matching the same registry catalog once per
    # package turns a linear build into an avoidable package×catalog walk.
    from agent_bom.parsers import get_registry_entry

    _registry_cache: dict[tuple[str, str, tuple[str, ...], str], dict | None] = {}

    def _registry_key(server: MCPServer) -> tuple[str, str, tuple[str, ...], str]:
        return (
            server.name,
            server.command,
            tuple(server.args),
            server.url or "",
        )

    blast_radii = []
    for pkg in unique_packages:
        if not pkg.vulnerabilities:
            continue

        key = _pkg_key(pkg)
        affected_servers = pkg_to_servers.get(key, [])
        affected_agents = pkg_to_agents.get(key, [])

        # Collect exposed credentials and tools — enrich from registry when server
        # config doesn't have explicit tool/credential data.
        # Cache registry lookups per server to avoid duplicate tool creation.
        #
        # IMPORTANT: Registry-sourced tools are "phantom" — they reflect what
        # the registry CLAIMS the server has, not what was introspected.
        # We include them for visibility but mark them so blast radius
        # consumers can distinguish confirmed vs phantom tools.
        exposed_creds: list[str] = []
        exposed_tools: list = []
        phantom_tools: list = []
        for server in affected_servers:
            server_creds = server.credential_names
            server_tools = list(server.tools)  # copy — don't mutate server

            # Registry enrichment: if no tools/creds known from config, use registry
            if not server_tools or not server_creds:
                registry_key = _registry_key(server)
                if registry_key not in _registry_cache:
                    _registry_cache[registry_key] = get_registry_entry(server)
                reg = _registry_cache[registry_key]
                if reg:
                    if not server_tools and reg.get("tools"):
                        from agent_bom.models import MCPTool

                        server_tools = [
                            MCPTool(
                                name=t,
                                description="(registry — unverified)",
                                discovery_source="registry",
                                discovery_confidence="unverified",
                            )
                            for t in reg["tools"]
                        ]
                    if not server_creds and reg.get("credential_env_vars"):
                        server_creds = reg["credential_env_vars"]

            exposed_creds.extend(server_creds)
            for tool in server_tools:
                if getattr(tool, "discovery_source", None) == "registry" and getattr(tool, "discovery_confidence", None) == "unverified":
                    phantom_tools.append(tool)
                else:
                    exposed_tools.append(tool)

        # Deduplicate credentials and tools to prevent inflation
        exposed_creds_deduped = sorted(set(exposed_creds))
        seen_tool_names: set[str] = set()
        deduped_tools = []
        for t in exposed_tools:
            if t.name not in seen_tool_names:
                seen_tool_names.add(t.name)
                deduped_tools.append(t)
        exposed_tools = deduped_tools
        seen_phantom: set[str] = set()
        deduped_phantom = []
        for t in phantom_tools:
            if t.name not in seen_phantom:
                seen_phantom.add(t.name)
                deduped_phantom.append(t)
        phantom_tools = deduped_phantom

        # AI-native risk context: elevated when an AI framework has creds + tools
        is_ai_framework = (
            pkg.name.lower().replace("-", "_") in {n.replace("-", "_") for n in _AI_FRAMEWORK_PACKAGES}
            or pkg.name.lower() in _AI_FRAMEWORK_PACKAGES
        )
        has_creds = bool(exposed_creds_deduped)
        has_tools = bool(exposed_tools)
        has_phantom_tools = bool(phantom_tools)
        if is_ai_framework and has_creds and has_tools:
            phantom_note = f" (+{len(phantom_tools)} registry-only tool(s) excluded from score)" if has_phantom_tools else ""
            ai_risk_context = (
                f"AI framework '{pkg.name}' runs inside an agent with {len(exposed_creds_deduped)} "
                f"exposed credential(s) and {len(exposed_tools)} confirmed reachable tool(s){phantom_note}. "
                f"A compromise here gives an attacker both identity and capability."
            )
        elif is_ai_framework and has_creds:
            ai_risk_context = (
                f"AI framework '{pkg.name}' has access to {len(exposed_creds_deduped)} "
                f"credential(s). Exploitation could exfiltrate secrets via LLM output."
            )
        elif is_ai_framework:
            ai_risk_context = "AI framework package — vulnerability affects LLM inference/orchestration pipeline."
        else:
            ai_risk_context = None

        for vuln in pkg.vulnerabilities:
            if not vuln.compliance_tags:
                vuln.compliance_tags = _tag_vuln(vuln, pkg)

            # CWE-aware filtering: only expose credentials/tools the vuln
            # type can realistically reach. A DoS (CWE-400) doesn't steal
            # DATABASE_URL. An RCE (CWE-94) does.
            from agent_bom.cwe_impact import (
                build_attack_vector_summary,
                classify_cwe_impact,
                filter_credentials_by_impact,
                filter_tools_by_impact,
            )

            impact_cat = classify_cwe_impact(vuln.cwe_ids)
            filtered_creds = filter_credentials_by_impact(
                impact_cat,
                exposed_creds_deduped,
            )
            filtered_tools = filter_tools_by_impact(
                impact_cat,
                exposed_tools,
            )
            filtered_phantom = filter_tools_by_impact(
                impact_cat,
                phantom_tools,
            )
            attack_summary = build_attack_vector_summary(
                cwe_ids=vuln.cwe_ids,
                category=impact_cat,
                filtered_creds=filtered_creds,
                filtered_tools=filtered_tools,
                severity=vuln.severity.value if vuln.severity else None,
                is_kev=vuln.is_kev,
            )
            if attack_summary:
                if ai_risk_context:
                    ai_risk_context = f"{ai_risk_context} {attack_summary}"
                else:
                    ai_risk_context = attack_summary

            br = BlastRadius(
                vulnerability=vuln,
                package=pkg,
                affected_servers=affected_servers,
                affected_agents=affected_agents,
                exposed_credentials=filtered_creds,
                exposed_tools=filtered_tools,
                phantom_tools=filtered_phantom,
                ai_risk_context=ai_risk_context,
                impact_category=impact_cat,
                all_server_credentials=list(exposed_creds_deduped),
                all_server_tools=list(exposed_tools) + list(phantom_tools),
                attack_vector_summary=attack_summary,
            )
            br.calculate_risk_score()
            # Context-aware tagging remains opt-in for explicit compliance
            # views, but effective framework tags are always materialized.
            if scan_options.compliance_enabled:
                apply_framework_tags(br)
            apply_effective_blast_radius_tags(br)
            blast_radii.append(br)

    # Sort by risk score descending
    blast_radii.sort(key=lambda br: br.risk_score, reverse=True)

    _print_vulnerability_summary(total_vulns, len(blast_radii), agents=agents)

    _logger.info(
        "Scan summary: %d packages scanned, %d vulnerabilities, %d blast radius findings across %d agent(s)",
        len(unique_packages),
        total_vulns,
        len(blast_radii),
        len(agents),
    )

    return blast_radii

scan_agents_sync

scan_agents_sync(agents: list[Agent], enable_enrichment: bool = False, nvd_api_key: Optional[str] = None, blast_radius_depth: int = 1, compliance_enabled: bool = False, resolve_transitive: bool = False, show_scan_banner: bool = True, offline: bool | None = None, prefer_local_db: bool | None = None, demo_advisories: bool = False, project_dir: str | None = None, options: ScanOptions | None = None) -> list[BlastRadius]

Synchronous wrapper for scan_agents.

Source code in src/agent_bom/scanners/package_scan.py
def scan_agents_sync(
    agents: list[Agent],
    enable_enrichment: bool = False,
    nvd_api_key: Optional[str] = None,
    blast_radius_depth: int = 1,
    compliance_enabled: bool = False,
    resolve_transitive: bool = False,
    show_scan_banner: bool = True,
    offline: bool | None = None,
    prefer_local_db: bool | None = None,
    demo_advisories: bool = False,
    project_dir: str | None = None,
    options: ScanOptions | None = None,
) -> list[BlastRadius]:
    """Synchronous wrapper for scan_agents."""
    scan_options = options or default_scan_options(
        compliance_enabled=compliance_enabled,
        resolve_transitive=resolve_transitive,
        prefer_local_db=prefer_local_db,
        offline=offline,
        demo_advisories=demo_advisories,
        project_dir=project_dir,
    )
    if enable_enrichment:
        blast_radii = asyncio.run(
            scan_agents_with_enrichment(
                agents,
                nvd_api_key,
                enable_enrichment,
                compliance_enabled=scan_options.compliance_enabled,
                show_scan_banner=show_scan_banner,
                options=scan_options,
            )
        )
    else:
        blast_radii = asyncio.run(
            scan_agents(
                agents,
                compliance_enabled=scan_options.compliance_enabled,
                resolve_transitive=scan_options.resolve_transitive,
                show_scan_banner=show_scan_banner,
                options=scan_options,
            )
        )
    if blast_radius_depth > 1:
        expand_blast_radius_hops(blast_radii, agents, max_depth=blast_radius_depth)
    return blast_radii

scan_agents_with_enrichment async

scan_agents_with_enrichment(agents: list[Agent], nvd_api_key: Optional[str] = None, enable_enrichment: bool = True, compliance_enabled: bool = False, show_scan_banner: bool = True, options: ScanOptions | None = None) -> list[BlastRadius]

Scan agents and enrich vulnerabilities with NVD/EPSS/KEV data.

Source code in src/agent_bom/scanners/package_scan.py
async def scan_agents_with_enrichment(
    agents: list[Agent],
    nvd_api_key: Optional[str] = None,
    enable_enrichment: bool = True,
    compliance_enabled: bool = False,
    show_scan_banner: bool = True,
    options: ScanOptions | None = None,
) -> list[BlastRadius]:
    """Scan agents and enrich vulnerabilities with NVD/EPSS/KEV data."""
    scan_options = options or default_scan_options(compliance_enabled=compliance_enabled)
    # First, do normal OSV scan
    blast_radii = await scan_agents(
        agents,
        compliance_enabled=scan_options.compliance_enabled,
        show_scan_banner=show_scan_banner,
        options=scan_options,
    )

    # Then enrich with external data
    if enable_enrichment and blast_radii:
        from agent_bom.enrichment import enrich_vulnerabilities
        from agent_bom.resolver import enrich_supply_chain_metadata

        # Collect all vulnerabilities
        all_vulns = []
        all_pkgs: list[Package] = []
        for agent in agents:
            for server in agent.mcp_servers:
                for pkg in server.packages:
                    all_pkgs.append(pkg)
                    all_vulns.extend(pkg.vulnerabilities)

        if all_vulns:
            await enrich_vulnerabilities(
                all_vulns,
                nvd_api_key=nvd_api_key,
                enable_nvd=True,
                enable_epss=True,
                enable_kev=True,
                offline=offline_mode,
            )

            # Refresh CVE-level compliance tags now that CWE/KEV/EPSS data is populated
            for agent in agents:
                for server in agent.mcp_servers:
                    for pkg in server.packages:
                        for v in pkg.vulnerabilities:
                            v.compliance_tags = _tag_vuln(v, pkg)

        # Supply-chain metadata enrichment — feeds Scorecard repo resolution
        try:
            async with create_client(timeout=10.0) as client:
                await enrich_supply_chain_metadata(all_pkgs, client)
        except Exception as exc:  # noqa: BLE001
            _logger.warning("Supply chain metadata enrichment failed (scorecard coverage may be incomplete): %s", exc)
            _emit_scan_warning("supply-chain metadata enrichment failed")

        # Scorecard enrichment — adds supply-chain quality signal
        try:
            from agent_bom.scorecard import enrich_packages_with_scorecard

            if all_pkgs:
                await enrich_packages_with_scorecard(all_pkgs)
        except Exception as exc:  # noqa: BLE001
            _logger.warning("Scorecard auto-enrichment failed (risk scores may be understated): %s", exc)
            _emit_scan_warning("OpenSSF Scorecard enrichment failed")

        # Recalculate blast radius with all enriched data
        for br in blast_radii:
            br.calculate_risk_score()
            apply_effective_blast_radius_tags(br)

        # Re-sort by updated risk scores
        blast_radii.sort(key=lambda br: br.risk_score, reverse=True)

    return blast_radii

scan_packages async

scan_packages(packages: list[Package], *, resolve_transitive: bool = False, options: ScanOptions | None = None) -> int

Scan a list of packages for vulnerabilities. Returns count of vulns found.

Source code in src/agent_bom/scanners/package_scan.py
async def scan_packages(
    packages: list[Package],
    *,
    resolve_transitive: bool = False,
    options: ScanOptions | None = None,
) -> int:
    """Scan a list of packages for vulnerabilities. Returns count of vulns found."""
    scan_options = options or default_scan_options(resolve_transitive=resolve_transitive)
    scan_offline = scan_options.offline
    packages = _prepare_packages(packages)
    packages, runtime_count = await _scan_runtime_packages(packages, scan_offline)

    # Installed versions reflect what's actually on disk, so they are tried before the registry fallback.
    unresolved = [p for p in packages if _is_unresolved_package(p) and p.ecosystem.lower() in ("npm", "pypi", "go")]
    if unresolved:
        _resolve_local_versions(unresolved, scan_options.project_dir)
    await _resolve_registry_versions(packages, scan_offline)

    # Capture the version-resolved direct demo inventory before optional online
    # transitive expansion. Only this curated set is closed evidence; packages
    # discovered outside it still require a real advisory lookup.
    demo_inventory_keys = {_package_db_key(package) for package in packages} if scan_options.demo_advisories else set()
    if scan_options.resolve_transitive and not scan_offline:
        packages = await _expand_transitive_packages(packages)

    scannable = _select_scannable_packages(packages)
    if not scannable:
        return runtime_count

    local_count, db_covered, local_db_targets = _scan_local_advisories(scannable, scan_options, demo_inventory_keys)
    coverage_gaps = _detect_coverage_gaps(scannable)
    osv_targets, covered_ecos = _select_osv_targets(scannable, scan_offline, local_db_targets, db_covered, coverage_gaps)
    results = await _query_osv_results(scannable, osv_targets, covered_ecos, scan_options)

    total_vulns = local_count + runtime_count
    total_vulns += _apply_osv_results(scannable, osv_targets, results)
    total_vulns += await _scan_supplemental_advisories(scannable, scan_offline)
    resolve_upstream_advisory_severity(scannable)
    _flag_suspicious_packages(scannable)
    total_vulns -= _suppress_unfixed_findings(scannable)

    # Ignore-rule suppression is applied once after BlastRadius construction by ``agent_bom.ignores``
    # so JSON/SARIF/VEX/MCP can record why a finding was accepted.
    return total_vulns

set_include_unfixed

set_include_unfixed(value: bool) -> None

Toggle surfacing of unfixed (no-dsa/won't-fix) OS-package advisories.

Source code in src/agent_bom/scanners/package_scan.py
def set_include_unfixed(value: bool) -> None:
    """Toggle surfacing of unfixed (no-dsa/won't-fix) OS-package advisories."""
    global include_unfixed  # noqa: PLW0603
    include_unfixed = value

set_offline_mode

set_offline_mode(value: bool) -> None

Set offline mode in both scanner and http_client transport layer.

Source code in src/agent_bom/scanners/package_scan.py
def set_offline_mode(value: bool) -> None:
    """Set offline mode in both scanner and http_client transport layer."""
    global offline_mode  # noqa: PLW0603
    offline_mode = value
    from agent_bom.http_client import set_offline

    set_offline(value)

builtin_scanner_registrations

builtin_scanner_registrations() -> list[ScannerRegistration]

Return built-in scanner registrations with capability metadata.

Source code in src/agent_bom/scanners/registry.py
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
def builtin_scanner_registrations() -> list[ScannerRegistration]:
    """Return built-in scanner registrations with capability metadata."""

    local_read = _local_read_capabilities()
    return [
        _registration(
            "sca-vulnerability",
            "agent_bom.scanners",
            phase=ScannerPhase.SCANNING,
            run_attr="scan_agents_sync",
            input_types=("agents", "packages"),
            output_types=("blast_radii", "vulnerabilities", "findings"),
            finding_types=("cve", "ghsa", "osv", "malicious-package", "dependency-confusion", "typosquat"),
            summary="SCA vulnerability matching with OSV/GHSA/local DB, enrichment, KEV/EPSS, compliance, and blast radius.",
            capabilities=_network_read_capabilities(destinations=("osv.dev", "github_advisory_database", "local_vulnerability_db")),
            failure_mode=ScannerFailureMode.FAIL_CLOSED,
            skip_when=("no_scan_requested", "no_packages_found"),
            telemetry_keys=("packages_scanned", "api_batches", "cache_hits", "warnings", "duration_ms"),
            standards=("OWASP", "NIST", "CIS", "SOC2", "EU_AI_ACT"),
        ),
        _registration(
            "secret-patterns",
            "agent_bom.secret_scanner",
            phase=ScannerPhase.SCANNING,
            run_attr="scan_secrets",
            input_types=("filesystem_path", "source_file", "config_file"),
            output_types=("secret_findings",),
            finding_types=("credential", "pii", "hardcoded-secret"),
            summary="Local file secret and PII pattern scanning using the shared runtime detector pattern library.",
            capabilities=local_read,
            failure_mode=ScannerFailureMode.WARN_AND_CONTINUE,
            skip_when=("no_filesystem_scope", "file_too_large", "excluded_directory"),
            telemetry_keys=("files_scanned", "findings_emitted", "warnings", "duration_ms"),
            standards=("OWASP_LLM02", "CIS_16_4", "SOC2_CC6_1"),
        ),
        _registration(
            "code-native",
            "agent_bom.ast_analyzer",
            phase=ScannerPhase.ANALYSIS,
            run_attr="analyze_project",
            input_types=("code_path",),
            output_types=("ast_analysis",),
            finding_types=("prompt-risk", "guardrail", "tool-signature"),
            summary="Native Python AST analysis composed into the agent-bom code command; no Semgrep execution.",
            capabilities=local_read,
            failure_mode=ScannerFailureMode.FAIL_CLOSED,
            skip_when=("no_code_scope",),
            telemetry_keys=("files_analyzed", "components_found", "duration_ms"),
            standards=("OWASP_LLM", "MITRE_ATLAS"),
        ),
        _registration(
            "ai-component-source",
            "agent_bom.ai_components",
            phase=ScannerPhase.ANALYSIS,
            run_attr="scan_source",
            input_types=("code_path",),
            output_types=("ai_components",),
            finding_types=("ai-component", "shadow-ai", "deprecated-model", "credential-reference"),
            summary="Native multi-language AI SDK, model, and component analysis composed into the agent-bom code command.",
            capabilities=local_read,
            failure_mode=ScannerFailureMode.FAIL_CLOSED,
            skip_when=("no_code_scope",),
            telemetry_keys=("files_scanned", "components_found", "duration_ms"),
            standards=("OWASP_LLM", "MITRE_ATLAS"),
        ),
        _registration(
            "sast-semgrep",
            "agent_bom.sast",
            phase=ScannerPhase.SCANNING,
            run_attr="scan_code",
            input_types=("code_path", "sarif"),
            output_types=("sast_data", "packages", "vulnerabilities"),
            finding_types=("sast", "cwe", "owasp"),
            summary=(
                "Semgrep SAST execution and normalization into the package/vulnerability model; "
                "the legacy .sarif path input imports through the canonical parser without executing Semgrep."
            ),
            capabilities=ExtensionCapabilities(
                scan_modes=("local", "online", "offline_local_rules"),
                required_scopes=("local_project_read", "conditional_network_egress"),
                outbound_destinations=("semgrep_registry",),
                data_boundary="local_source_read_with_operator_selected_rule_source",
                network_access=True,
                guarantees=("read_only", "offline_rejects_remote_rules"),
            ),
            failure_mode=ScannerFailureMode.SKIP_WHEN_UNAVAILABLE,
            skip_when=("semgrep_missing", "no_code_scope", "offline_without_local_rules"),
            telemetry_keys=("files_scanned", "rules_loaded", "findings_emitted", "duration_ms"),
            standards=("CWE", "OWASP", "SARIF"),
        ),
        _registration(
            "external-scan-ingest",
            "agent_bom.parsers.external_scanners",
            phase=ScannerPhase.DISCOVERY,
            run_attr="detect_and_parse",
            input_types=("sarif", "cyclonedx", "spdx", "trivy_json", "grype_json", "syft_json", "prowler_ocsf", "asff"),
            output_types=("packages", "vulnerabilities", "findings"),
            finding_types=("external-finding", "sast", "sca", "sbom-package"),
            summary="Tool-agnostic evidence import through canonical SARIF, SBOM, and scanner-native parsers.",
            capabilities=local_read,
            failure_mode=ScannerFailureMode.FAIL_CLOSED,
            skip_when=("no_external_scan_input",),
            telemetry_keys=("format", "packages_emitted", "findings_emitted", "duration_ms"),
            standards=("SARIF", "CycloneDX", "SPDX"),
        ),
        _registration(
            "container-image",
            "agent_bom.image",
            phase=ScannerPhase.DISCOVERY,
            run_attr="scan_image",
            input_types=("image_ref", "image_tar"),
            output_types=("packages", "image_scan_strategy"),
            finding_types=("container-package", "image-vulnerability-input"),
            summary="Container image package extraction via local daemon, registry, or OCI tarball paths.",
            capabilities=ExtensionCapabilities(
                scan_modes=("image", "airgap"),
                required_scopes=("image_read",),
                permissions_used=("docker_socket_read", "registry_pull", "local_tar_read"),
                outbound_destinations=("container_registry",),
                data_boundary="image_metadata_and_layers_read_only",
                network_access=True,
                guarantees=("read_only",),
            ),
            failure_mode=ScannerFailureMode.WARN_AND_CONTINUE,
            skip_when=("no_image_scope", "image_runtime_unavailable"),
            telemetry_keys=("images_scanned", "packages_emitted", "strategy", "warnings", "duration_ms"),
        ),
        _registration(
            "container-sbom-posture",
            "agent_bom.cloud.container_sbom",
            phase=ScannerPhase.ANALYSIS,
            run_attr="scan_container_image",
            input_types=("image_ref",),
            output_types=("container_sbom_posture",),
            finding_types=("unpinned-tag", "missing-sbom", "missing-provenance", "stale-image"),
            summary="Read-only OCI metadata posture check for SBOM/provenance/staleness signals.",
            capabilities=_network_read_capabilities(destinations=("docker_hub_registry_api",), scan_modes=("cloud", "container")),
            failure_mode=ScannerFailureMode.WARN_AND_CONTINUE,
            skip_when=("non_docker_hub_registry_metadata_only", "registry_unavailable"),
            telemetry_keys=("images_checked", "registry_requests", "findings_emitted", "duration_ms"),
            standards=("SLSA", "CycloneDX", "SPDX"),
        ),
        _registration(
            "sbom-ingest",
            "agent_bom.sbom",
            phase=ScannerPhase.DISCOVERY,
            run_attr="load_sbom",
            input_types=("cyclonedx", "spdx", "sbom_json"),
            output_types=("packages", "sbom_metadata"),
            finding_types=("sbom-package",),
            summary="SBOM ingestion for CycloneDX/SPDX documents into the shared package model.",
            capabilities=local_read,
            failure_mode=ScannerFailureMode.FAIL_CLOSED,
            skip_when=("no_sbom_input",),
            telemetry_keys=("packages_emitted", "format", "warnings", "duration_ms"),
            standards=("CycloneDX", "SPDX"),
        ),
        _registration(
            "iac-terraform",
            "agent_bom.terraform",
            phase=ScannerPhase.DISCOVERY,
            run_attr="scan_terraform_dir",
            input_types=("terraform_dir",),
            output_types=("agents", "iac_findings"),
            finding_types=("iac", "misconfiguration", "ai-infra"),
            summary="Terraform AI infrastructure discovery and misconfiguration evidence.",
            capabilities=local_read,
            failure_mode=ScannerFailureMode.WARN_AND_CONTINUE,
            skip_when=("no_terraform_scope",),
            telemetry_keys=("files_scanned", "findings_emitted", "warnings", "duration_ms"),
            standards=("MITRE_ATLAS", "CIS", "NIST"),
        ),
        _registration(
            "iac-dbt",
            "agent_bom.iac.dbt_security",
            phase=ScannerPhase.DISCOVERY,
            run_attr="scan_dbt_file",
            input_types=("dbt_project", "dbt_profiles", "dbt_packages", "dbt_models", "dbt_macros", "dbt_ci"),
            output_types=("iac_findings",),
            finding_types=("dbt-security", "credential-exposure", "supply-chain", "sql-injection", "ci-hygiene"),
            summary="dbt project, profile, package, model, macro, seed, and CI/CD security checks.",
            capabilities=local_read,
            failure_mode=ScannerFailureMode.WARN_AND_CONTINUE,
            skip_when=("no_dbt_project_scope",),
            telemetry_keys=("files_scanned", "findings_emitted", "warnings", "duration_ms"),
            standards=("SLSA", "NIST", "SOC2"),
        ),
        _registration(
            "k8s-live-cluster",
            "agent_bom.k8s",
            phase=ScannerPhase.DISCOVERY,
            run_attr="scan_live_cluster_posture",
            input_types=("kubeconfig_context", "kubectl"),
            output_types=("iac_findings",),
            finding_types=(
                "kubernetes-live",
                "pod-security",
                "rbac-wildcard",
                "cluster-admin-binding",
                "kubelet-cis",
                "network-policy-gap",
            ),
            summary=(
                "Read-only live Kubernetes cluster audit via kubectl: PodSecurity on running "
                "workloads, RBAC/namespace over-broad grants, and node/kubelet CIS configuration."
            ),
            capabilities=ExtensionCapabilities(
                scan_modes=("cluster", "online"),
                required_scopes=("kubernetes_cluster_read",),
                permissions_used=("kubectl_get", "kubelet_configz_read"),
                outbound_destinations=("kubernetes_api_server",),
                data_boundary="cluster_state_read_only",
                network_access=True,
                guarantees=("read_only", "no_cluster_mutation"),
            ),
            failure_mode=ScannerFailureMode.SKIP_WHEN_UNAVAILABLE,
            skip_when=("kubectl_missing", "cluster_unreachable", "no_cluster_scope"),
            telemetry_keys=("pods_scanned", "nodes_scanned", "findings_emitted", "duration_ms"),
            standards=("CIS", "NIST"),
        ),
        _registration(
            "cicd-github-actions",
            "agent_bom.github_actions",
            phase=ScannerPhase.DISCOVERY,
            run_attr="scan_github_actions",
            input_types=("github_actions_path",),
            output_types=("agents", "workflow_findings"),
            finding_types=("workflow-permissions", "unpinned-action", "secret-exposure", "fork-pr-risk"),
            summary="GitHub Actions workflow inventory and CI/CD posture checks.",
            capabilities=local_read,
            failure_mode=ScannerFailureMode.WARN_AND_CONTINUE,
            skip_when=("no_github_actions_scope",),
            telemetry_keys=("workflows_scanned", "findings_emitted", "warnings", "duration_ms"),
            standards=("SLSA", "CIS", "NIST"),
        ),
        _registration(
            "dataset-card",
            "agent_bom.parsers.dataset_cards",
            phase=ScannerPhase.DISCOVERY,
            run_attr="scan_dataset_directory",
            input_types=("dataset_directory", "dataset_card"),
            output_types=("dataset_cards",),
            finding_types=("unlicensed-dataset", "missing-card", "unversioned-data", "remote-source"),
            summary="Dataset-card provenance, license, and lineage scanner.",
            capabilities=local_read,
            failure_mode=ScannerFailureMode.WARN_AND_CONTINUE,
            skip_when=("no_dataset_scope",),
            telemetry_keys=("datasets_scanned", "flagged_count", "warnings", "duration_ms"),
            standards=("EU_AI_ACT", "NIST_AI_RMF"),
        ),
        _registration(
            "dataset-pii",
            "agent_bom.parsers.dataset_pii_scanner",
            phase=ScannerPhase.SCANNING,
            run_attr="scan_directory_for_pii",
            input_types=("csv", "json", "jsonl", "dataset_directory"),
            output_types=("dataset_pii_findings",),
            finding_types=("pii", "phi", "drivers_license", "email", "ssn"),
            summary="Dataset row sampling scanner for PII/PHI exposure.",
            capabilities=local_read,
            failure_mode=ScannerFailureMode.WARN_AND_CONTINUE,
            skip_when=("no_dataset_scope", "unsupported_file_type", "row_limit_reached"),
            telemetry_keys=("files_scanned", "rows_scanned", "findings_emitted", "duration_ms"),
            standards=("HIPAA", "EU_AI_ACT", "NIST_PRIVACY"),
        ),
        _registration(
            "prompt-injection",
            "agent_bom.parsers.prompt_scanner",
            phase=ScannerPhase.SCANNING,
            run_attr="scan_prompt_file",
            input_types=("prompt_file", "instruction_file"),
            output_types=("prompt_findings",),
            finding_types=("prompt-injection", "hidden-instruction", "tool-exfiltration"),
            summary="Prompt/instruction file scanning for injection and agentic abuse patterns.",
            capabilities=local_read,
            failure_mode=ScannerFailureMode.WARN_AND_CONTINUE,
            skip_when=("no_prompt_scope", "unsupported_file_type"),
            telemetry_keys=("files_scanned", "findings_emitted", "duration_ms"),
            standards=("OWASP_LLM", "OWASP_AGENTIC"),
        ),
        _registration(
            "skill-audit",
            "agent_bom.parsers.skill_audit",
            phase=ScannerPhase.SCANNING,
            run_attr="audit_skill_result",
            input_types=("skill_file", "instruction_file"),
            output_types=("skill_findings", "skill_audit"),
            finding_types=("skill-risk", "mcp-blocklist", "typosquat", "shell-access", "unverified-server"),
            summary=(
                "Skill/instruction file audit: registry typosquat, MCP blocklist on extracted "
                "servers, shell access, and behavioral regex/AST risks."
            ),
            capabilities=local_read,
            failure_mode=ScannerFailureMode.WARN_AND_CONTINUE,
            skip_when=("no_skill_scope", "unsupported_file_type"),
            telemetry_keys=("files_scanned", "packages_checked", "servers_checked", "findings_emitted", "duration_ms"),
            standards=("OWASP_LLM", "OWASP_AGENTIC", "OWASP_MCP"),
        ),
        _registration(
            "license-policy",
            "agent_bom.license_policy",
            phase=ScannerPhase.ANALYSIS,
            run_attr="evaluate_license_policy",
            input_types=("agents", "packages", "license_policy"),
            output_types=("license_report",),
            finding_types=("license-block", "license-warning", "unknown-license"),
            summary="SPDX license policy evaluation across discovered packages.",
            capabilities=_local_read_capabilities(scan_modes=("analysis",)),
            failure_mode=ScannerFailureMode.WARN_AND_CONTINUE,
            skip_when=("no_packages_found", "license_policy_disabled"),
            telemetry_keys=("packages_evaluated", "findings_emitted", "duration_ms"),
            standards=("SPDX", "OpenChain"),
        ),
        _registration(
            "firmware-advisory",
            "agent_bom.scanners.firmware_advisory",
            phase=ScannerPhase.SCANNING,
            run_attr="scan_firmware_advisories",
            input_types=("gpu_inventory", "firmware_inventory"),
            output_types=("firmware_findings",),
            finding_types=("firmware-cve", "bmc-cve", "driver-cve"),
            summary="GPU firmware/BMC/driver advisory scanner for AI infrastructure.",
            capabilities=_network_read_capabilities(destinations=("bundled_firmware_advisory_feed",), scan_modes=("infra", "gpu")),
            failure_mode=ScannerFailureMode.WARN_AND_CONTINUE,
            skip_when=("no_gpu_inventory", "advisory_feed_unavailable"),
            telemetry_keys=("devices_scanned", "advisories_matched", "duration_ms"),
            standards=("NVIDIA_CSAF", "CVE"),
        ),
        _registration(
            "model-advisory",
            "agent_bom.model_advisories",
            phase=ScannerPhase.SCANNING,
            run_attr="match_model_advisories",
            input_types=("model_card", "model_inventory"),
            output_types=("model_advisories",),
            finding_types=("unsafe-model-format", "model-card-risk", "model-advisory"),
            summary="Model-specific advisory and model-card risk matching.",
            capabilities=_local_read_capabilities(scan_modes=("model",)),
            failure_mode=ScannerFailureMode.WARN_AND_CONTINUE,
            skip_when=("no_model_inventory", "feed_missing"),
            telemetry_keys=("models_scanned", "advisories_matched", "duration_ms"),
            standards=("NIST_AI_RMF", "EU_AI_ACT"),
        ),
        _registration(
            "runtime-detectors",
            "agent_bom.runtime.detectors",
            phase=ScannerPhase.SCANNING,
            run_attr="detector_pipeline",
            input_types=("tool_call", "tool_response", "runtime_event"),
            output_types=("runtime_findings", "policy_decisions"),
            finding_types=("prompt-injection", "credential-leak", "rate-limit", "tool-drift", "vector-db-injection"),
            summary="Runtime proxy detector family for agent/tool call enforcement.",
            capabilities=ExtensionCapabilities(
                scan_modes=("runtime", "proxy"),
                required_scopes=("runtime_event_read",),
                permissions_used=("tool_call_observe",),
                outbound_destinations=(),
                data_boundary="runtime_event_inline_redacted",
                network_access=False,
                guarantees=("redacted_evidence", "policy_enforced"),
            ),
            failure_mode=ScannerFailureMode.FAIL_CLOSED,
            skip_when=("runtime_proxy_disabled",),
            telemetry_keys=("events_scanned", "findings_emitted", "policy_blocks", "duration_ms"),
            standards=("OWASP_LLM", "OWASP_MCP", "SOC2"),
        ),
        _registration(
            "yara-signature",
            "agent_bom.scanners.yara",
            phase=ScannerPhase.SCANNING,
            run_attr="scan_with_yara",
            input_types=("file", "directory", "model_file", "artifact"),
            output_types=("signature_findings",),
            finding_types=("malware-signature", "unsafe-artifact", "model-file-indicator"),
            summary="Reserved scanner-driver slot for local YARA-style artifact signatures; not executable in this release.",
            capabilities=_local_read_capabilities(scan_modes=("signature",)),
            failure_mode=ScannerFailureMode.SKIP_WHEN_UNAVAILABLE,
            execution_state=ScannerExecutionState.PLANNED,
            enabled_by_default=False,
            skip_when=("driver_not_implemented", "ruleset_missing"),
            telemetry_keys=("files_scanned", "rules_loaded", "matches", "duration_ms"),
            standards=("YARA",),
        ),
        _registration(
            "zero-day-heuristics",
            "agent_bom.scanners.zero_day",
            phase=ScannerPhase.ANALYSIS,
            run_attr="score_zero_day_residual_risk",
            input_types=("packages", "runtime_context", "exploit_signals", "graph_context"),
            output_types=("residual_risk_scores", "prioritized_findings"),
            finding_types=("residual-risk", "zero-day-susceptibility", "high-blast-radius-unfixed"),
            summary="Reserved scanner-driver slot for residual/zero-day risk scoring across SCA, runtime, and graph context.",
            capabilities=_local_read_capabilities(scan_modes=("analysis",)),
            failure_mode=ScannerFailureMode.SKIP_WHEN_UNAVAILABLE,
            execution_state=ScannerExecutionState.PLANNED,
            enabled_by_default=False,
            skip_when=("driver_not_implemented", "insufficient_context"),
            telemetry_keys=("packages_scored", "contexts_used", "scores_emitted", "duration_ms"),
            standards=("EPSS", "KEV", "NIST_AI_RMF"),
        ),
    ]

get_scanner_registration

get_scanner_registration(name: str) -> ScannerRegistration

Return one scanner registration or raise KeyError.

Source code in src/agent_bom/scanners/registry.py
def get_scanner_registration(name: str) -> ScannerRegistration:
    """Return one scanner registration or raise ``KeyError``."""

    _ensure_scanner_registry_loaded()
    return _SCANNER_REGISTRY[name]

list_registered_scanners

list_registered_scanners(*, include_planned: bool = True) -> list[ScannerRegistration]

Return registered scanner drivers sorted by name.

Source code in src/agent_bom/scanners/registry.py
def list_registered_scanners(*, include_planned: bool = True) -> list[ScannerRegistration]:
    """Return registered scanner drivers sorted by name."""

    _ensure_scanner_registry_loaded()
    registrations = [_SCANNER_REGISTRY[name] for name in sorted(_SCANNER_REGISTRY)]
    if include_planned:
        return registrations
    return [registration for registration in registrations if registration.execution_state != ScannerExecutionState.PLANNED]

register_scanner

register_scanner(registration: ScannerRegistration) -> None

Register a scanner driver with capability and execution metadata.

Source code in src/agent_bom/scanners/registry.py
def register_scanner(registration: ScannerRegistration) -> None:
    """Register a scanner driver with capability and execution metadata."""

    if not registration.name:
        raise ValueError("scanner registration must declare a name")
    if registration.name in _SCANNER_REGISTRY:
        raise ValueError(f"duplicate scanner registration: {registration.name}")
    _SCANNER_REGISTRY[registration.name] = registration

scanner_registry_summary

scanner_registry_summary() -> dict[str, object]

Return a compact scanner registry summary for API/UI surfaces.

Source code in src/agent_bom/scanners/registry.py
def scanner_registry_summary() -> dict[str, object]:
    """Return a compact scanner registry summary for API/UI surfaces."""

    registrations = list_registered_scanners(include_planned=True)
    by_phase: dict[str, int] = {}
    by_state: dict[str, int] = {}
    for registration in registrations:
        by_phase[registration.phase.value] = by_phase.get(registration.phase.value, 0) + 1
        by_state[registration.execution_state.value] = by_state.get(registration.execution_state.value, 0) + 1
    return {
        "total": len(registrations),
        "active": by_state.get(ScannerExecutionState.ACTIVE.value, 0),
        "passive": by_state.get(ScannerExecutionState.PASSIVE.value, 0),
        "planned": by_state.get(ScannerExecutionState.PLANNED.value, 0),
        "by_phase": by_phase,
        "by_state": by_state,
    }

scanner_registry_warnings

scanner_registry_warnings() -> list[str]

Return sanitized non-fatal registry loading warnings.

Source code in src/agent_bom/scanners/registry.py
def scanner_registry_warnings() -> list[str]:
    """Return sanitized non-fatal registry loading warnings."""

    _ensure_scanner_registry_loaded()
    return list(_SCANNER_REGISTRY_WARNINGS)

advisory_id_severity_fallback

advisory_id_severity_fallback(advisory_id: str) -> tuple[Severity, Optional[str]]

Return conservative triage severity for advisory-only IDs.

Some advisory ecosystems publish IDs before CVSS/vendor severity arrives. These should not stay invisible as unknown findings in operator views, but only known advisory namespaces get this fallback. Arbitrary missing severity still remains UNKNOWN.

Source code in src/agent_bom/scanners/risk.py
def advisory_id_severity_fallback(advisory_id: str) -> tuple[Severity, Optional[str]]:
    """Return conservative triage severity for advisory-only IDs.

    Some advisory ecosystems publish IDs before CVSS/vendor severity arrives.
    These should not stay invisible as ``unknown`` findings in operator views,
    but only known advisory namespaces get this fallback. Arbitrary missing
    severity still remains ``UNKNOWN``.
    """
    normalized = advisory_id.upper()
    if normalized.startswith("GHSA-"):
        return Severity.MEDIUM, "ghsa_heuristic"
    if normalized.startswith(_OSV_MEDIUM_FALLBACK_PREFIXES):
        return Severity.MEDIUM, "osv_heuristic"
    if normalized.startswith(_DISTRO_MEDIUM_FALLBACK_PREFIXES):
        return Severity.MEDIUM, "distro_advisory_heuristic"
    return Severity.UNKNOWN, None

parse_cvss_vector

parse_cvss_vector(vector: str) -> Optional[float]

Compute CVSS base score from a vector string (v3.x and v4.0).

Source code in src/agent_bom/core/cvss.py
def parse_cvss_vector(vector: str) -> Optional[float]:
    """Compute CVSS base score from a vector string (v3.x and v4.0)."""
    try:
        vector = vector.strip()
        if vector.startswith("CVSS:4"):
            return parse_cvss4_vector(vector)
        if not vector.startswith("CVSS:3"):
            return None

        parts = vector.split("/")[1:]
        metrics = dict(p.split(":") for p in parts)

        av = _CVSS3_AV.get(metrics.get("AV", ""), None)
        ac = _CVSS3_AC.get(metrics.get("AC", ""), None)
        scope = metrics.get("S", "U")
        pr_map = _CVSS3_PR_C if scope == "C" else _CVSS3_PR_U
        pr = pr_map.get(metrics.get("PR", ""), None)
        ui = _CVSS3_UI.get(metrics.get("UI", ""), None)
        c = _CVSS3_CIA.get(metrics.get("C", ""), None)
        i = _CVSS3_CIA.get(metrics.get("I", ""), None)
        a = _CVSS3_CIA.get(metrics.get("A", ""), None)

        if any(value is None for value in (av, ac, pr, ui, c, i, a)):
            return None

        av, ac, pr, ui = float(av), float(ac), float(pr), float(ui)  # type: ignore[arg-type]
        c, i, a = float(c), float(i), float(a)  # type: ignore[arg-type]

        isc_base = 1.0 - (1.0 - c) * (1.0 - i) * (1.0 - a)
        if scope == "C":
            isc = 7.52 * (isc_base - 0.029) - 3.25 * ((isc_base - 0.02) ** 15)
        else:
            isc = 6.42 * isc_base

        if isc <= 0:
            return 0.0

        exploitability = 8.22 * av * ac * pr * ui
        raw = min(1.08 * (isc + exploitability), 10.0) if scope == "C" else min(isc + exploitability, 10.0)
        return math.ceil(raw * 10) / 10.0
    except Exception:  # noqa: BLE001
        _logger.debug("CVSS vector parse failed")
        return None

parse_osv_severity

parse_osv_severity(vuln_data: dict) -> tuple[Severity, Optional[float], Optional[str]]

Extract severity, CVSS score, and severity source from OSV data.

Source code in src/agent_bom/scanners/risk.py
def parse_osv_severity(vuln_data: dict) -> tuple[Severity, Optional[float], Optional[str]]:
    """Extract severity, CVSS score, and severity source from OSV data."""
    basis = osv_severity_basis(vuln_data)
    return basis.severity, basis.cvss_score, basis.severity_source

severity_from_label

severity_from_label(raw: Any) -> Severity

Normalize vendor labels into the scanner's supported severity enum.

Source code in src/agent_bom/core/severity.py
def severity_from_label(raw: Any) -> Severity:
    """Normalize vendor labels into the scanner's supported severity enum."""
    label = normalize_severity(str(raw) if raw is not None else None)
    return Severity(label) if label != "info" else Severity.UNKNOWN

peek_coverage_warnings

peek_coverage_warnings() -> list[dict]

Read the structured coverage warnings without draining them.

The scanner needs to see what a sub-step recorded (e.g. an OSV lookup failure) while leaving the channel intact for the command boundary that actually consumes it.

Source code in src/agent_bom/scanners/state.py
def peek_coverage_warnings() -> list[dict]:
    """Read the structured coverage warnings without draining them.

    The scanner needs to see what a sub-step recorded (e.g. an OSV lookup
    failure) while leaving the channel intact for the command boundary that
    actually consumes it.
    """
    return list(_coverage_warnings_state())

record_coverage_warning

record_coverage_warning(warning: dict) -> None

Record a structured per-release coverage-gap warning (deduped by release).

Source code in src/agent_bom/scanners/state.py
def record_coverage_warning(warning: dict) -> None:
    """Record a structured per-release coverage-gap warning (deduped by release)."""
    captured = _coverage_capture.get()
    if captured is not None:
        captured.append(dict(warning))
    warnings = _coverage_warnings_state()
    release = warning.get("release")
    if any(existing.get("release") == release for existing in warnings):
        return
    warnings.append(warning)

reset_scan_warnings

reset_scan_warnings() -> None

Reset all scan warning channels at a command/request boundary.

Source code in src/agent_bom/scanners/state.py
def reset_scan_warnings() -> None:
    """Reset all scan warning channels at a command/request boundary."""
    _scan_state_local.warnings = []
    _scan_state_local.coverage_warnings = []

reset_scan_warnings_only

reset_scan_warnings_only() -> None

Reset transient scanner warnings without discarding parser coverage gaps.

Project manifest parsers can run before vulnerability scanning and record a structured coverage warning. The package scanner must clear warnings from a prior scan, but must preserve those parser warnings for the final report.

Source code in src/agent_bom/scanners/state.py
def reset_scan_warnings_only() -> None:
    """Reset transient scanner warnings without discarding parser coverage gaps.

    Project manifest parsers can run before vulnerability scanning and record a
    structured coverage warning. The package scanner must clear warnings from a
    prior scan, but must preserve those parser warnings for the final report.
    """
    _scan_state_local.warnings = []