#!/usr/bin/env python3
"""vsscli — Vendor Security Scanner CLI (Snyk-style, runs in YOUR CI).

Three scans, all executed on this machine, results uploaded to the VSS server:

  vsscli image|scan|container <ref>   container image  (Trivy: vuln + secret + misconfig)
  vsscli oss [PATH]                   open-source deps (trivy fs lockfiles ∪ trivy rootfs
                                      installed artefacts, run it AFTER your build)
  vsscli code [PATH]                  source code      (Semgrep SAST + quality, Gitleaks or
                                      Trivy secrets, Trivy IaC misconfig)

Exit codes (identical for every scan command, mirrors `snyk test`):
  0  scan completed, nothing at/above the gate
  1  scan completed, gating issues found
  2  error (scanner failure, timeout, network, 4xx/5xx, bad flag, --require-sast
     without semgrep, denylisted scanner, ...)
  3  nothing scannable (oss: no package resolved from any pass; code: no file)

Output: progress and warnings go to stderr. `--json` writes exactly one JSON
document to stdout. `--json-file-output`, `--sarif-file-output` (SARIF 2.1.0,
GitHub code scanning) and `--sbom-file-output` (CycloneDX) write files.

Authentication: an API key with the matching ingest scope (`ingest:oss`,
`ingest:code`, `ingest:image`). NEVER pass the key on argv: let `login` prompt
for it or export VSS_API_KEY. The server defaults to
https://threatlens.internal.genorim.xyz, so CI needs only VSS_API_KEY.
VSS_SERVER / VSS_API_KEY always win over ~/.vsscli/config.json. Plain http:// is refused for non-loopback servers (and
always refused in CI). TLS certificates are always verified. Redirects to
another origin or from https to http are refused, so the key is never
forwarded.

Requirements: Python 3.8+, Trivy >= 0.53 on PATH (never 0.69.4-0.69.6);
optional semgrep and gitleaks on PATH for `code`. Zero pip dependencies:
this is a single file you can drop on PATH.
"""

from __future__ import annotations

import argparse
import datetime as _dt
import email.utils
import getpass
import gzip
import hashlib
import json
import os
import platform
import random
import re
import shutil
import signal
import socket
import ssl
import stat
import subprocess
import sys
import tempfile
import time
import urllib.error
import urllib.parse
import urllib.request
from pathlib import Path
from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple

# ── Version, exit codes, constants ────────────────────────────────────────────

VSSCLI_VERSION = "0.3.2"
# The ThreatLens server; VSS_SERVER or `vsscli login --server` point elsewhere.
DEFAULT_SERVER = "https://threatlens.internal.genorim.xyz"

EXIT_OK = 0          # scan completed, no gating finding
EXIT_ISSUES = 1      # scan completed, gating findings present
EXIT_ERROR = 2       # any error
EXIT_NOTHING = 3     # nothing scannable

CONFIG_DIR = Path.home() / ".vsscli"
CONFIG_FILE = CONFIG_DIR / "config.json"

USER_AGENT = "vsscli/{} ({} {})".format(VSSCLI_VERSION, platform.system(), platform.machine())

#: `--require-fresh-db`: a DB built longer ago than this is refused (Trivy publishes every 6 h).
FRESH_DB_MAX_AGE_HOURS = 48
#: Trivy >= 0.53 is required for --detection-priority / --image-config-scanners.
MIN_TRIVY_VERSION = (0, 53, 0)
#: Credential-stealing Trivy releases (GHSA-69fq-xp46-6x23, CVE-2026-33634).
DEFAULT_SCANNER_DENYLIST = "trivy:0.69.4,trivy:0.69.5,trivy:0.69.6"

SEVERITIES = ("critical", "high", "medium", "low", "unknown")
#: Gate rank; `unknown` gates as medium (documented in --help).
SEVERITY_RANK = {"low": 1, "medium": 2, "high": 3, "critical": 4, "unknown": 2}
THRESHOLDS = ("low", "medium", "high", "critical")
FAIL_ON_CHOICES = ("all", "upgradable", "none")
TRIVY_SEVERITY_ARG = "UNKNOWN,LOW,MEDIUM,HIGH,CRITICAL"
#: Misconfig scanners for image FILESYSTEM scans — the server's list (trivy_scanner
#: IMAGE_MISCONFIG_SCANNERS, WP6-33). Trivy resolves Terraform `module { source }`
#: blocks (registry lookups, archive GETs, `git clone`, cloud-metadata probes) even
#: with --offline-scan, so a `.tf` baked into an untrusted image would choose the
#: hosts the CI runner connects to. Terraform is never enabled for image scans.
IMAGE_MISCONFIG_SCANNERS = "dockerfile,kubernetes,helm,cloudformation,azure-arm,ansible"
# `vsscli code` over the checkout in place: NEVER terraform / terraformplan-snapshot.
# Trivy resolves remote Terraform module sources even with --offline-scan and Go never
# proxies loopback, so a fork PR's `.tf` could make the CI runner connect to
# localhost:PORT (REQUESTS/WP8-to-WP7.md N5). terraformplan-json was verified not to connect.
CODE_MISCONFIG_SCANNERS = "azure-arm,cloudformation,dockerfile,helm,kubernetes,terraformplan-json,ansible"
#: Every file Trivy's Terraform scanner reads (WP8-25): when any is present, `vsscli code`
#: says loudly that Terraform IaC was NOT scanned by Trivy (INT-12) — never a silent gap.
TERRAFORM_SUFFIXES = (".tf", ".tf.json", ".tofu", ".tofu.json")
#: The server's --max-image-size (IMAGE_MAX_BYTES default 16 GiB, in MiB).
IMAGE_MAX_SIZE = "16384MiB"
#: A proxy on a closed loopback port: every HTTP(S) request Trivy makes during a
#: code scan (Terraform registry, module archives, s3::/gcs:: + IMDS probes) fails
#: without connecting to the repo-chosen host (server code_scanner.NO_EGRESS_PROXY).
NO_EGRESS_PROXY = "http://127.0.0.1:1"
#: SUP-7: pinned Trivy artefact sources (the server's defaults). Used only when the
#: runner does not set its own mirror via TRIVY_DB_REPOSITORY / ... (allow-listed).
TRIVY_DB_REPOSITORIES = "mirror.gcr.io/aquasec/trivy-db:2,ghcr.io/aquasecurity/trivy-db:2"
TRIVY_JAVA_DB_REPOSITORIES = "mirror.gcr.io/aquasec/trivy-java-db:1,ghcr.io/aquasecurity/trivy-java-db:1"
TRIVY_CHECKS_BUNDLE_REPOSITORY = "mirror.gcr.io/aquasec/trivy-checks:2"

#: Every finding the gate sees was reported by THIS scan, so it is present. A
#: pipeline close (resolved, fixed, stale, not_applicable) copied from another
#: branch's row must not stop it gating (CLI-3: fail closed). Only an advisory
#: withdrawn upstream, or an in-force human verdict, is non-gating.
_NON_GATING_STATUSES = frozenset({"rejected"})
#: Human verdicts that hold until `suppress_until` (server STICKY_STATUSES).
_STICKY_STATUSES = frozenset({"false_positive", "accepted_risk", "not_affected"})

#: Code-scan build output / third-party dirs, as doublestar globs so a nested
#: `apps/web/.next` is excluded too (a bare name only matches at the scan root).
CODE_SKIP_DIRS = (
    ".git", "__pycache__", ".mypy_cache", ".pytest_cache", ".ruff_cache", ".turbo",
    "node_modules", ".venv", "venv", "site-packages", "vendor",
    ".next", "dist", "out", "coverage", ".gradle", ".terraform",
)
#: OSS pass 1 (lockfiles) skips installed trees — pass 2 (rootfs) reads them.
OSS_FS_SKIP_DIRS = ("node_modules", ".git", ".venv", "venv", "site-packages")

#: Environment passed to scanner subprocesses (allow-list). Everything else —
#: VSS_API_KEY, cloud credentials, CI tokens — never reaches Trivy/Semgrep/Gitleaks.
_ENV_ALLOW = (
    "PATH", "HOME", "USER", "LOGNAME", "TMPDIR", "TEMP", "TMP", "LANG", "LC_ALL", "LC_CTYPE",
    "SSL_CERT_FILE", "SSL_CERT_DIR", "HTTP_PROXY", "HTTPS_PROXY", "NO_PROXY",
    "http_proxy", "https_proxy", "no_proxy", "XDG_CACHE_HOME",
    # actions/setup-python's Python finds libpython only through it, so a pip-installed
    # semgrep cannot start without it. Same trust as PATH (the runner's own env);
    # LD_PRELOAD / DYLD_* stay stripped.
    "LD_LIBRARY_PATH",
    # Windows
    "SYSTEMROOT", "USERPROFILE", "APPDATA", "LOCALAPPDATA", "COMSPEC", "PATHEXT", "WINDIR",
)
#: Trivy variables that are credentials or locations, not result filters.
_TRIVY_ENV_ALLOW = (
    "TRIVY_USERNAME", "TRIVY_PASSWORD", "TRIVY_REGISTRY_TOKEN", "TRIVY_CACHE_DIR",
    "TRIVY_DB_REPOSITORY", "TRIVY_JAVA_DB_REPOSITORY", "TRIVY_CHECKS_BUNDLE_REPOSITORY",
)
_DOCKER_ENV_ALLOW = ("DOCKER_HOST", "DOCKER_CONFIG", "DOCKER_CERT_PATH", "DOCKER_TLS_VERIFY", "DOCKER_CONTEXT")

#: Semgrep packs verified to exist and be non-empty in CE (DESIGN C18). These tables are
#: a verbatim mirror of the server's semgrep_scanner (DEFAULT_CONFIGS, LANGUAGE_PACKS,
#: FRAMEWORK_PACKS, WEB_LANGUAGES/WEB_CLASS_PACKS, IAC_PACKS, QUALITY_PACKS,
#: QUALITY_LANG_CONFIGS): the local fallback (--no-upload, or /api/code/config
#: unreachable) selects exactly what select_configs() would. A parity test compares them.
SEMGREP_BASE_PACKS = ("p/default", "p/owasp-top-ten", "p/cwe-top-25", "p/secrets", "p/security-audit")
SEMGREP_WEB_PACKS = ("p/jwt", "p/sql-injection", "p/xss", "p/command-injection", "p/insecure-transport")
SEMGREP_QUALITY_PACKS = ("p/r2c-bug-scan", "p/r2c-best-practices")
SEMGREP_LANG_PACKS = {
    "python": ("p/python", "p/bandit"),
    "javascript": ("p/javascript", "p/nodejs", "p/nodejsscan", "p/eslint-plugin-security"),
    "typescript": ("p/typescript", "p/nodejs", "p/nodejsscan", "p/eslint-plugin-security"),
    "java": ("p/java", "p/findsecbugs"),
    "kotlin": ("p/kotlin", "p/mobsfscan"),
    "scala": ("p/scala",),
    "go": ("p/golang", "p/gosec"),
    "csharp": ("p/csharp",),
    "php": ("p/php", "p/phpcs-security-audit"),
    "ruby": ("p/ruby",),
    "rust": ("p/rust",),
    "c": ("p/c", "p/flawfinder"),
    "swift": ("p/swift", "p/mobsfscan"),
}
#: Quality-only configs per language (added only with quality, like the server).
SEMGREP_QUALITY_LANG_CONFIGS = {
    "python": ("r/python.lang.correctness", "r/python.lang.best-practice", "r/python.lang.maintainability"),
    "javascript": ("r/javascript.lang.correctness", "r/javascript.lang.best-practice"),
    "typescript": ("r/typescript.react.best-practice",),
    "go": ("r/go.lang.correctness",),
    "java": ("r/java.lang.correctness",),
}
SEMGREP_FRAMEWORK_PACKS = {
    "django": ("p/django",), "flask": ("p/flask",), "fastapi": ("p/fastapi",),
    "react": ("p/react",), "express": ("p/expressjs",), "rails": ("p/brakeman",),
}
#: The server's IaC words (+ `kubernetes`, the CLI's sniffed-manifest word, an alias of
#: the server's `yaml`).
SEMGREP_IAC_PACKS = {
    "dockerfile": ("p/dockerfile",), "terraform": ("p/terraform",),
    "github-actions": ("p/github-actions",), "ci": ("p/ci",),
    "docker-compose": ("p/docker-compose",), "yaml": ("p/kubernetes",), "kubernetes": ("p/kubernetes",),
}
#: Never used: 404 or empty in CE (DESIGN C18).
SEMGREP_FORBIDDEN_PACKS = frozenset({"p/spring", "p/java-spring", "p/bash", "p/best-practices", "p/nextjs",
                                     "p/express", "p/rails", "p/cloudformation", "auto", "p/auto"})
_SEMGREP_CONFIG_RE = re.compile(r"^(?:p|r)/[A-Za-z0-9][A-Za-z0-9._-]{0,127}$")

_EXT_LANG = {
    ".py": "python", ".js": "javascript", ".jsx": "javascript", ".mjs": "javascript", ".cjs": "javascript",
    ".ts": "typescript", ".tsx": "typescript", ".java": "java", ".kt": "kotlin", ".kts": "kotlin",
    ".scala": "scala", ".go": "go", ".cs": "csharp", ".php": "php", ".rb": "ruby", ".rs": "rust",
    ".c": "c", ".h": "c", ".cc": "c", ".cpp": "c", ".hpp": "c", ".swift": "swift",
}
_WEB_LANGS = frozenset({"python", "javascript", "typescript", "java", "kotlin", "go", "php", "ruby", "csharp",
                        "scala"})

#: filename -> (ecosystem, kind). kind: lockfile | manifest.
_MANIFEST_NAMES = {
    "package-lock.json": ("npm", "lockfile"), "npm-shrinkwrap.json": ("npm", "lockfile"),
    "yarn.lock": ("npm", "lockfile"), "pnpm-lock.yaml": ("npm", "lockfile"), "bun.lock": ("npm", "lockfile"),
    "package.json": ("npm", "manifest"),
    "poetry.lock": ("PyPI", "lockfile"), "uv.lock": ("PyPI", "lockfile"), "Pipfile.lock": ("PyPI", "lockfile"),
    "pdm.lock": ("PyPI", "lockfile"), "pyproject.toml": ("PyPI", "manifest"), "Pipfile": ("PyPI", "manifest"),
    "setup.py": ("PyPI", "manifest"), "setup.cfg": ("PyPI", "manifest"),
    "pom.xml": ("Maven", "manifest"), "build.gradle": ("Maven", "manifest"),
    "build.gradle.kts": ("Maven", "manifest"), "gradle.lockfile": ("Maven", "lockfile"),
    "go.mod": ("Go", "lockfile"), "Cargo.lock": ("crates.io", "lockfile"), "Cargo.toml": ("crates.io", "manifest"),
    "packages.lock.json": ("NuGet", "lockfile"), "packages.config": ("NuGet", "lockfile"),
    "Directory.Packages.props": ("NuGet", "lockfile"),
    "Gemfile.lock": ("RubyGems", "lockfile"), "Gemfile": ("RubyGems", "manifest"),
    "composer.lock": ("Packagist", "lockfile"), "composer.json": ("Packagist", "manifest"),
    "pubspec.lock": ("Pub", "lockfile"), "pubspec.yaml": ("Pub", "manifest"),
    "Package.resolved": ("SwiftURL", "lockfile"), "Package.swift": ("SwiftURL", "manifest"),
    "Podfile.lock": ("CocoaPods", "lockfile"), "mix.lock": ("Hex", "lockfile"), "mix.exs": ("Hex", "manifest"),
    "conan.lock": ("ConanCenter", "lockfile"),
}
#: A manifest counts as resolved when one of these lockfiles sits next to it.
_LOCK_FOR_MANIFEST = {
    "package.json": ("package-lock.json", "npm-shrinkwrap.json", "yarn.lock", "pnpm-lock.yaml", "bun.lock"),
    "pyproject.toml": ("poetry.lock", "uv.lock", "pdm.lock"), "Pipfile": ("Pipfile.lock",),
    "setup.py": (), "setup.cfg": (),
    "build.gradle": ("gradle.lockfile",), "build.gradle.kts": ("gradle.lockfile",),
    "Cargo.toml": ("Cargo.lock",), "Gemfile": ("Gemfile.lock",), "composer.json": ("composer.lock",),
    "pubspec.yaml": ("pubspec.lock",), "Package.swift": ("Package.resolved",), "mix.exs": ("mix.lock",),
}
_DETECT_SKIP_DIRS = frozenset({".git", "node_modules", ".venv", "venv", "site-packages", "__pycache__",
                               ".tox", ".gradle", ".terraform", ".next", ".mypy_cache", ".pytest_cache"})
#: Walk bound for detect_projects — reported (never silent) when hit.
DETECT_MAX_FILES = 200000
#: Recorded bounds (DESIGN §0.7): defaults are the server's CODE_SEMGREP_MAX_TARGET_BYTES
#: and GITLEAKS_MAX_TARGET_MB, overridable with --semgrep-max-target-bytes /
#: --gitleaks-max-target-mb. Files over them are skipped by the engine; the CLI records
#: them in coverage so the server never resolves a finding in a file it did not read.
SEMGREP_MAX_TARGET_BYTES = 1000000
GITLEAKS_MAX_TARGET_MB = 50
#: Oversize paths listed per bound (the rest is counted, flagged partial).
SKIPPED_PATHS_MAX = 5000

_HEX_RE = re.compile(r"^[0-9a-fA-F]{7,64}$")


# ── Terminal helpers — progress ALWAYS on stderr ──────────────────────────────


def _tty(stream: Any) -> bool:
    try:
        return bool(stream.isatty()) and os.environ.get("NO_COLOR", "") == ""
    except (AttributeError, ValueError):
        return False


def c(code: str, s: str, stream: Any = None) -> str:
    if not _tty(stream if stream is not None else sys.stdout):
        return s
    return "\033[{}m{}\033[0m".format(code, s)


def red(s: str) -> str: return c("31", s)
def green(s: str) -> str: return c("32", s)
def yellow(s: str) -> str: return c("33", s)
def blue(s: str) -> str: return c("34", s)
def grey(s: str) -> str: return c("90", s)
def bold(s: str) -> str: return c("1", s)


def _err(line: str, code: str) -> None:
    print(c(code, line, sys.stderr), file=sys.stderr)


def info(msg: str) -> None:
    _err("• " + msg, "90")


def ok(msg: str) -> None:
    _err("✓ " + msg, "32")


def warn(msg: str) -> None:
    _err("⚠ " + msg, "33")


def die(msg: str, code: int = EXIT_ERROR) -> None:
    """Print to stderr and exit. Default exit is 2 (error) — never 1, which means 'issues found'."""
    print(c("31", "✗ " + redact(msg), sys.stderr), file=sys.stderr)
    sys.exit(code)


def emit_stdout(text: str) -> None:
    sys.stdout.write(text)
    if not text.endswith("\n"):
        sys.stdout.write("\n")
    sys.stdout.flush()


# ── Redaction (JSON / SARIF / meta / error text) ──────────────────────────────
# Linear patterns only (no nested quantifiers): these run over scanner output.

_REDACTIONS = (
    (re.compile(r"(?i)\b([a-z][a-z0-9+.-]{0,31}://)[^/\s:@]{1,256}:[^/\s@]{1,256}@"), r"\1[REDACTED]@"),
    (re.compile(r"\b(?:ghp|gho|ghs|ghu|ghr)_[A-Za-z0-9]{20,255}\b"), "[REDACTED]"),
    (re.compile(r"\bgithub_pat_[A-Za-z0-9_]{20,255}\b"), "[REDACTED]"),
    (re.compile(r"\bglpat-[A-Za-z0-9_-]{20,255}\b"), "[REDACTED]"),
    (re.compile(r"\bxox[abposr]-[A-Za-z0-9-]{10,255}\b"), "[REDACTED]"),
    (re.compile(r"https://hooks\.slack\.com/services/[A-Za-z0-9/_-]{10,255}"), "https://hooks.slack.com/services/[REDACTED]"),
    (re.compile(r"\b(?:AKIA|ASIA)[A-Z0-9]{16}\b"), "[REDACTED]"),
    (re.compile(r"\bsk-(?:proj-|ant-)?[A-Za-z0-9_-]{20,255}\b"), "[REDACTED]"),
    (re.compile(r"\bvss_[A-Za-z0-9]{6}_[A-Za-z0-9]{20,64}\b"), "[REDACTED]"),
    (re.compile(r"\bnpm_[A-Za-z0-9]{36}\b"), "[REDACTED]"),
    (re.compile(r"\bpypi-[A-Za-z0-9_-]{50,255}\b"), "[REDACTED]"),
    (re.compile(r"\bAIza[0-9A-Za-z_-]{35}\b"), "[REDACTED]"),
    (re.compile(r"\beyJ[A-Za-z0-9_-]{8,4096}\.eyJ[A-Za-z0-9_-]{8,8192}\.[A-Za-z0-9_-]{8,4096}\b"), "[REDACTED JWT]"),
    (re.compile(r"(?i)\b(bearer|basic|token)\s+[A-Za-z0-9._~+/=-]{16,4096}"), r"\1 [REDACTED]"),
    (re.compile(r"(?i)\b((?:api[_-]?key|apikey|secret|password|passwd|pwd|token|access[_-]?key|private[_-]?key|client[_-]?secret)"
                r"[\"']?\s{0,3}[:=]\s{0,3}[\"']?)[^\s\"',;]{4,512}"), r"\1[REDACTED]"),
)


_PEM_BEGIN = re.compile(r"-----BEGIN [A-Z0-9 ]{0,40}PRIVATE KEY-----")
_PEM_END = re.compile(r"-----END [A-Z0-9 ]{0,40}PRIVATE KEY-----")


def _redact_private_keys(text: str) -> str:
    """Replace every private-key block (BEGIN..END, or BEGIN..end of text when the END
    marker is missing) in ONE forward pass: the END search only ever moves forward, so
    a text full of unterminated BEGIN markers stays linear."""
    if "PRIVATE KEY-----" not in text:
        return text
    out: List[str] = []
    pos = 0
    end_m = None
    for m in _PEM_BEGIN.finditer(text):
        if m.start() < pos:
            continue
        if end_m is None or end_m.start() < m.end():
            end_m = _PEM_END.search(text, m.end())
        out.append(text[pos:m.start()])
        out.append("[REDACTED PRIVATE KEY]")
        if end_m is None:
            pos = len(text)
            break
        pos = end_m.end()
    out.append(text[pos:])
    return "".join(out)


def redact(value: Any) -> Any:
    """Mask credentials in a string (other types pass through unchanged)."""
    if not isinstance(value, str) or not value:
        return value
    out = _redact_private_keys(value)
    for pattern, repl in _REDACTIONS:
        out = pattern.sub(repl, out)
    return out


#: Marker for a subtree deeper than REDACT_MAX_DEPTH (the server's redact_obj does the
#: same): an unwalked subtree is never passed through unredacted.
REDACT_TOO_DEEP = "[REDACTED:too_deep]"
REDACT_MAX_DEPTH = 64


def redact_obj(obj: Any, _depth: int = 0) -> Any:
    """Walk a JSON-like object and redact every string (bounded depth, fail closed)."""
    if _depth > REDACT_MAX_DEPTH:
        return REDACT_TOO_DEEP
    if isinstance(obj, str):
        return redact(obj)
    if isinstance(obj, list):
        return [redact_obj(v, _depth + 1) for v in obj]
    if isinstance(obj, dict):
        return {k: redact_obj(v, _depth + 1) for k, v in obj.items()}
    return obj


#: Well-known, NON-secret token prefixes worth keeping for triage (mirror of the
#: server's code_scanner._KNOWN_TOKEN_PREFIX).
_KNOWN_TOKEN_PREFIX = re.compile(
    r"^(ghp_|gho_|ghu_|ghs_|ghr_|github_pat_|glpat-|xox[abpr]-|AKIA|ASIA|AIza|sk-proj-|sk-ant-|sk-|"
    r"sk_live_|sk_test_|rk_live_|rk_test_|npm_|pypi-|hf_|SG\.|dp\.pt\.)"
)


def mask_secret(raw: Optional[str]) -> str:
    """Port of the server's code_scanner.mask_secret_hint: never more than a known
    token prefix + the last 2 characters, and nothing at all for short secrets.

    ``ghp_Z9y8...J3i2`` -> ``ghp_****…i2``; ``Tr0ub4dor&3x`` -> ``********``.
    (The old 4+4 hint revealed 8 of the 12 characters of a typical password in
    CLI JSON/SARIF/stdout.)
    """
    if not raw:
        return "********"
    s = str(raw)
    m = _KNOWN_TOKEN_PREFIX.match(s)
    if m and len(s) >= 16:
        return "{}****…{}".format(m.group(1), s[-2:])
    if len(s) >= 24:
        return "****…{}".format(s[-2:])
    return "********"


def sha256_hex(text: str) -> str:
    return hashlib.sha256(text.encode("utf-8")).hexdigest()


def strip_dot_slash(path: Optional[str]) -> str:
    """Remove ONE literal leading './' (py3.8-safe removeprefix). Never character-set
    stripping of '.' and '/', which turns `.env` into `env` and `.github/...` into `github/...`."""
    p = path or ""
    return p[2:] if p.startswith("./") else p


# ── Config + authentication ───────────────────────────────────────────────────


def load_config() -> dict:
    if not CONFIG_FILE.exists():
        return {}
    try:
        data = json.loads(CONFIG_FILE.read_text())
        return data if isinstance(data, dict) else {}
    except (json.JSONDecodeError, OSError, ValueError):
        return {}


def save_config(cfg: dict) -> None:
    """Write ~/.vsscli/config.json containing the API key, 0600 from the start (SC-07)."""
    CONFIG_DIR.mkdir(mode=0o700, parents=True, exist_ok=True)
    try:
        CONFIG_DIR.chmod(0o700)
    except OSError:
        pass
    payload = json.dumps(cfg, indent=2)
    fd = os.open(str(CONFIG_FILE), os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
    with os.fdopen(fd, "w") as fh:
        fh.write(payload)
    CONFIG_FILE.chmod(0o600)


def is_ci() -> bool:
    """True on a CI runner (GitHub Actions, GitLab, Jenkins, Buildkite, Azure, Circle, Bitbucket, ...)."""
    if os.environ.get("CI", "").strip().lower() in ("1", "true", "yes"):
        return True
    return any(os.environ.get(k) for k in (
        "GITHUB_ACTIONS", "GITLAB_CI", "BUILDKITE", "JENKINS_URL", "TF_BUILD", "CIRCLECI",
        "BITBUCKET_BUILD_NUMBER", "TEAMCITY_VERSION", "CODEBUILD_BUILD_ID",
    ))


#: Hosts where plain http:// is not a transport risk (SC-02).
_LOOPBACK_HOSTS = {"localhost", "127.0.0.1", "::1", "[::1]"}


def _check_server_scheme(server: str, insecure: bool) -> None:
    """Refuse to send an API key over plain http:// to a remote host (SC-02, B18).

    https:// always passes; http:// to loopback passes; http:// to anything else
    is refused in CI unconditionally, and outside CI only with an explicit opt-in
    (`--insecure` at login, or VSS_ALLOW_INSECURE=1). TLS verification itself is
    never disabled.
    """
    parsed = urllib.parse.urlparse(server)
    if parsed.scheme == "https" and parsed.hostname:
        return
    host = (parsed.hostname or "").lower()
    if parsed.scheme not in ("http", "https") or not host:
        die("The server must be an http:// or https:// URL (got {!r}).".format(server))
    if parsed.scheme == "http" and host in _LOOPBACK_HOSTS:
        return
    if is_ci():
        die("Refusing to send an API key to {} over plain HTTP on a CI runner. "
            "Use https:// (plain HTTP is never allowed in CI, even with VSS_ALLOW_INSECURE).".format(server))
    if insecure or os.environ.get("VSS_ALLOW_INSECURE") == "1":
        warn("Sending your API key to {} over plain HTTP. Anyone on the network path can read it "
             "and use it until you revoke the key.".format(server))
        return
    die("Refusing to send an API key to {} over plain HTTP — it would cross the network in "
        "cleartext, along with every scan result you upload.\n"
        "  Use https://…, or pass --insecure at login (VSS_ALLOW_INSECURE=1) on a trusted "
        "private network outside CI.".format(server))


def _server_base(url: str) -> str:
    """The server's base URL: every request path already starts with /api, so a
    pasted ".../api" (or ".../api/") is trimmed instead of producing /api/api/..."""
    base = (url or "").strip().rstrip("/")
    return base[:-4] if base.lower().endswith("/api") else base


def resolve_auth(required: bool = True) -> Optional[dict]:
    """Server + key, taken as a PAIR from one source (B18).

    Precedence: VSS_SERVER/VSS_API_KEY > ~/.vsscli/config.json > DEFAULT_SERVER.
    A key from the config file is only ever sent to the server it was saved for:
    if VSS_SERVER points elsewhere, VSS_API_KEY must be set too (otherwise a
    stale config would leak its key to the new server). On a CI runner an env
    key without VSS_SERVER goes to DEFAULT_SERVER, never to the config's server.
    """
    cfg = load_config()
    env_server = _server_base(os.environ.get("VSS_SERVER") or "")
    env_key = (os.environ.get("VSS_API_KEY") or "").strip()
    cfg_server = _server_base(str(cfg.get("server") or ""))
    cfg_key = str(cfg.get("api_key") or "").strip()
    server, key, source = "", "", ""
    if env_server:
        server, source = env_server, "env"
        if env_key:
            key = env_key
        elif cfg_key and cfg_server.rstrip("/") == env_server.rstrip("/"):
            key, source = cfg_key, "env+config"
    elif env_key:
        # The pipeline's env is authoritative on a runner: never pair its key with
        # whatever server a stale ~/.vsscli/config.json on a shared runner names.
        if is_ci() or not cfg_server:
            server, key, source = DEFAULT_SERVER, env_key, "default+env"
        else:
            server, key, source = cfg_server, env_key, "config+env"
    else:
        server, key, source = cfg_server or DEFAULT_SERVER, cfg_key, "config" if cfg_server else "default"
    if not server or not key:
        if not required:
            return None
        die("Not logged in to {}. Export VSS_API_KEY, or run `vsscli login` (you will be prompted "
            "for the key).".format(server))
    insecure = bool(cfg.get("allow_insecure_http")) and cfg_server.rstrip("/") == server.rstrip("/")
    _check_server_scheme(server, insecure)
    return {"server": server.rstrip("/"), "api_key": key, "source": source,
            "ui_url": str(cfg.get("ui_url") or "").rstrip("/") or None}


# ── HTTP transport (stdlib only) ──────────────────────────────────────────────


class RedirectRefused(urllib.error.HTTPError):
    pass


class _SameOriginRedirectHandler(urllib.request.HTTPRedirectHandler):
    """urllib's default handler copies X-API-Key to ANY redirect target (CLI-1).

    Only same-origin (scheme, host, port) GET redirects are followed, at most 3;
    a cross-host hop or an https->http downgrade raises instead.
    """

    max_redirections = 3

    def redirect_request(self, req, fp, code, msg, headers, newurl):  # noqa: D401
        new_full = urllib.parse.urljoin(req.full_url, newurl)
        old, new = urllib.parse.urlsplit(req.full_url), urllib.parse.urlsplit(new_full)
        if (new.scheme, (new.hostname or "").lower(), new.port) != (old.scheme, (old.hostname or "").lower(), old.port):
            raise RedirectRefused(new_full, code,
                                  "redirect to another origin refused ({} -> {}://{})".format(
                                      old.netloc, new.scheme, new.netloc), headers, fp)
        if req.get_method() not in ("GET", "HEAD"):
            raise RedirectRefused(new_full, code, "redirect of a {} request refused".format(req.get_method()),
                                  headers, fp)
        return super().redirect_request(req, fp, code, msg, headers, newurl)


def build_opener() -> urllib.request.OpenerDirector:
    """Opener with default TLS verification (never disabled) and the safe redirect handler."""
    ctx = ssl.create_default_context()
    return urllib.request.build_opener(_SameOriginRedirectHandler(), urllib.request.HTTPSHandler(context=ctx))


_RETRY_STATUSES = frozenset({429, 502, 503, 504})
#: A non-idempotent request (the upload POST) is retried only when the server said
#: explicitly that it did NOT process it: 429 / 503 (ingest_busy, scan_in_progress —
#: both carry Retry-After). A 502/504 or a lost response may mean it was stored.
_POST_RETRY_STATUSES = frozenset({429, 503})
_IDEMPOTENT_METHODS = frozenset({"GET", "HEAD", "OPTIONS", "PUT", "DELETE"})
#: Payloads that compress better than this are sent uncompressed: the server
#: refuses bodies above its gzip ratio ceiling (bomb guard, default 20:1).
GZIP_MAX_RATIO = 15.0
GZIP_MIN_BYTES = 64 * 1024


def _retry_after_seconds(headers: Any) -> Optional[float]:
    raw = headers.get("Retry-After") if headers is not None else None
    if not raw:
        return None
    raw = str(raw).strip()
    if raw.isdigit():
        return float(raw)
    try:
        when = email.utils.parsedate_to_datetime(raw)
        if when is None:
            return None
        if when.tzinfo is None:
            when = when.replace(tzinfo=_dt.timezone.utc)
        return max(0.0, (when - _dt.datetime.now(_dt.timezone.utc)).total_seconds())
    except (TypeError, ValueError, IndexError):
        return None


def encode_body(body: dict, allow_gzip: bool = True) -> Tuple[bytes, Dict[str, str]]:
    """JSON body, gzip-compressed when it is large and not suspiciously compressible."""
    raw = json.dumps(body, separators=(",", ":")).encode("utf-8")
    headers = {"Content-Type": "application/json"}
    if allow_gzip and len(raw) >= GZIP_MIN_BYTES:
        packed = gzip.compress(raw, compresslevel=6)
        if packed and len(raw) / float(len(packed)) <= GZIP_MAX_RATIO:
            headers["Content-Encoding"] = "gzip"
            return packed, headers
    return raw, headers


def http(
    method: str,
    url: str,
    api_key: Optional[str],
    body: Optional[dict] = None,
    timeout: int = 120,
    retries: int = 4,
    allow_gzip: bool = True,
    opener: Optional[urllib.request.OpenerDirector] = None,
    sleep=time.sleep,
) -> Tuple[int, Optional[Any], str]:
    """Returns (status, json_or_None, raw_text). Dies (exit 2) when the server stays
    unreachable or a redirect is refused.

    Retries with full-jitter backoff honouring Retry-After:
    * idempotent methods: 429/502/503/504 and any transient network error;
    * POST (the upload): only 429/503, and network errors raised while the request
      was being SENT (urllib wraps those in URLError: refused, DNS, connect/send
      timeout). A timeout or reset while waiting for or reading the RESPONSE is not
      retried — the server may still be ingesting, and a retry would ingest twice.
    """
    data: Optional[bytes] = None
    headers = {"User-Agent": USER_AGENT, "Accept": "application/json", "X-VSSCLI-Version": VSSCLI_VERSION}
    if api_key:
        headers["X-API-Key"] = api_key
    if body is not None:
        data, extra = encode_body(body, allow_gzip=allow_gzip)
        headers.update(extra)
    op = opener or build_opener()
    idempotent = method.upper() in _IDEMPOTENT_METHODS
    retry_statuses = _RETRY_STATUSES if idempotent else _POST_RETRY_STATUSES
    attempt = 0
    while True:
        req = urllib.request.Request(url, data=data, method=method, headers=headers)
        status, parsed, raw, retry_hint = 0, None, "", None
        try:
            with op.open(req, timeout=timeout) as r:
                raw = r.read().decode("utf-8", errors="replace")
                status = r.status
        except RedirectRefused as e:
            die("Server redirect refused — {}. The API key was NOT forwarded. Configure the final "
                "https:// URL directly.".format(e.msg))
        except urllib.error.HTTPError as e:
            status = e.code
            try:
                raw = e.read().decode("utf-8", errors="replace")
            except Exception:  # noqa: BLE001
                raw = ""
            retry_hint = _retry_after_seconds(e.headers)
        except ssl.SSLError as e:
            die("TLS verification failed for {}: {}. Fix the certificate or point SSL_CERT_FILE at "
                "your CA bundle — verification is never disabled.".format(url, e))
        except (urllib.error.URLError, socket.timeout, ConnectionError, TimeoutError) as e:
            reason = getattr(e, "reason", e)
            if isinstance(reason, ssl.SSLError):
                die("TLS verification failed for {}: {}. Point SSL_CERT_FILE at your CA bundle.".format(url, reason))
            request_sent = not isinstance(e, urllib.error.URLError)
            if request_sent and not idempotent:
                die("No response from {} after the upload was sent ({}). The server may still be processing "
                    "it or may have stored it — check the web UI before re-running. Not retried, to avoid a "
                    "duplicate ingest (raise --upload-timeout for very large uploads).".format(url, reason))
            if attempt < retries:
                delay = min(30.0, random.uniform(0, 2 ** attempt))
                warn("Cannot reach {} ({}); retrying in {:.1f}s …".format(url, reason, delay))
                sleep(delay)
                attempt += 1
                continue
            die("Cannot reach {}: {}".format(url, reason))
        if status in retry_statuses and attempt < retries:
            delay = random.uniform(0, min(30.0, 2 ** attempt))
            if retry_hint is not None:
                delay = max(delay, min(retry_hint, 120.0))
            warn("Server returned HTTP {}; retrying in {:.1f}s …".format(status, delay))
            sleep(delay)
            attempt += 1
            continue
        if raw:
            try:
                parsed = json.loads(raw)
            except (json.JSONDecodeError, ValueError):
                parsed = None
        return status, parsed, raw


def _server_error_text(code: int, body: Any, raw: str) -> str:
    detail = body.get("detail") if isinstance(body, dict) else None
    if isinstance(detail, dict):
        detail = detail.get("message") or json.dumps(detail)[:300]
    text = detail if isinstance(detail, str) else (raw or "")[:300]
    return redact("HTTP {}: {}".format(code, text))


# ── Subprocess helpers ────────────────────────────────────────────────────────


def which(name: str) -> Optional[str]:
    return shutil.which(name)


def scanner_env(extra: Optional[Dict[str, str]] = None, docker: bool = False) -> Tuple[Dict[str, str], List[str]]:
    """Allow-listed environment for scanner children + the NAMES of stripped scanner vars.

    Repo- or CI-level `TRIVY_SEVERITY`, `TRIVY_IGNORE_UNFIXED`, `TRIVY_SKIP_DB_UPDATE`,
    `TRIVY_OFFLINE_SCAN`, `SEMGREP_APP_TOKEN`, `GITLEAKS_CONFIG`, ... must not change
    results, and VSS_API_KEY / cloud credentials never reach a scanner.
    """
    allow = set(_ENV_ALLOW) | set(_TRIVY_ENV_ALLOW)
    if docker:
        allow |= set(_DOCKER_ENV_ALLOW)
    env = {k: v for k, v in os.environ.items() if k in allow}
    stripped = sorted(k for k in os.environ
                      if k not in allow and k.upper().startswith(("TRIVY_", "SEMGREP_", "GITLEAKS_")))
    env["SEMGREP_SEND_METRICS"] = "off"
    if extra:
        env.update(extra)
    return env, stripped


def _kill_group(proc: subprocess.Popen) -> None:
    try:
        if hasattr(os, "killpg"):
            os.killpg(proc.pid, signal.SIGKILL)
        else:  # Windows
            proc.kill()
    except (ProcessLookupError, PermissionError, OSError):
        pass


#: Bytes of a file-backed stderr read back for error messages (the tail).
STDERR_TAIL_BYTES = 64 * 1024


def _read_tail(path: str, limit: int = STDERR_TAIL_BYTES) -> str:
    try:
        with open(path, "rb") as fh:
            fh.seek(0, os.SEEK_END)
            size = fh.tell()
            fh.seek(max(0, size - limit))
            return fh.read().decode("utf-8", "replace")
    except OSError:
        return ""


def run_capture(cmd: Sequence[str], timeout: int = 600, cwd: Optional[str] = None,
                env: Optional[Dict[str, str]] = None, stderr_path: Optional[str] = None,
                fatal: bool = True) -> Tuple[int, str, str]:
    """Run argv (never a shell), capture output. Timeout / missing binary -> exit 2, or with
    fatal=False -> (127 | 126 | 124, "", reason) so a dependency-resolution run only warns.

    The child gets its own session so a timeout kills its whole process group
    (subprocess.run would orphan grandchildren). `stderr_path` (a 0600 file in the
    private dir) takes a chatty stderr (Semgrep --verbose logs every file) out of
    memory: only its last STDERR_TAIL_BYTES are returned.
    """
    if env is None:
        env = scanner_env()[0]   # never inherit VSS_API_KEY / cloud credentials by accident
    err_fh = open(stderr_path, "wb") if stderr_path else None   # noqa: SIM115 — closed below
    kwargs: Dict[str, Any] = {"stdout": subprocess.PIPE, "stderr": err_fh if err_fh else subprocess.PIPE,
                              "cwd": cwd, "env": env}
    if os.name == "posix":
        kwargs["start_new_session"] = True
    try:
        try:
            proc = subprocess.Popen(list(cmd), **kwargs)  # noqa: S603 — argv list, no shell
        except FileNotFoundError:
            if not fatal:
                return 127, "", "{} is not installed or not on PATH".format(cmd[0])
            die("{} is not installed or not on PATH.".format(cmd[0]))
        except PermissionError as e:
            if not fatal:
                return 126, "", "cannot execute {}: {}".format(cmd[0], e)
            die("Cannot execute {}: {}".format(cmd[0], e))
        try:
            out, err = proc.communicate(timeout=timeout)
        except subprocess.TimeoutExpired:
            _kill_group(proc)
            try:
                proc.communicate(timeout=5)
            except Exception:  # noqa: BLE001
                pass
            if not fatal:
                return 124, "", "timed out after {}s".format(timeout)
            die("{} timed out after {}s (raise --timeout).".format(os.path.basename(str(cmd[0])), timeout))
    finally:
        if err_fh is not None:
            err_fh.close()
    err_text = _read_tail(stderr_path) if stderr_path else (err or b"").decode("utf-8", "replace")
    return proc.returncode, (out or b"").decode("utf-8", "replace"), err_text


class PrivateDir:
    """A 0700 temp dir used as the scanners' CWD (so repo trivy.yaml/.trivyignore/
    trivy-secret.yaml are never auto-loaded) and for 0600 report files."""

    def __enter__(self) -> "PrivateDir":
        self.path = tempfile.mkdtemp(prefix="vsscli-")
        os.chmod(self.path, 0o700)
        return self

    def file(self, name: str) -> str:
        p = os.path.join(self.path, name)
        fd = os.open(p, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600)
        os.close(fd)
        return p

    def __exit__(self, *exc: Any) -> None:
        shutil.rmtree(self.path, ignore_errors=True)


def _last_line(text: str) -> str:
    lines = [ln for ln in (text or "").strip().splitlines() if ln.strip()]
    return redact(lines[-1][:300]) if lines else ""


def _basename_any(path: str) -> str:
    """Basename for POSIX and Windows spellings alike (argv from any runner OS)."""
    parts = [p for p in re.split(r"[\\/]", str(path).rstrip("/\\")) if p]
    return parts[-1] if parts else ""


def sanitize_argv(argv: Sequence[str], replacements: Dict[str, str]) -> List[str]:
    """argv for meta: local paths replaced by placeholders, credential values masked."""
    out: List[str] = []
    mask_next = False
    for a in argv:
        s = str(a)
        if mask_next:
            out.append("[REDACTED]")
            mask_next = False
            continue
        if s in ("--password", "--username", "--registry-token", "--token", "--api-key"):
            out.append(s)
            mask_next = True
            continue
        for real, placeholder in sorted(replacements.items(), key=lambda kv: -len(kv[0])):
            if real and s.startswith(real):
                s = placeholder + s[len(real):]
                break
        else:
            # Any other absolute local path (a --semgrep-config rules file, an --artifact, a
            # home directory) keeps only its basename: runner paths never reach the server.
            if s.startswith("/") or re.match(r"^[A-Za-z]:[\\/]", s) or s.startswith("~"):
                s = "<path>/" + _basename_any(s)
        out.append(redact(s))
    if out:
        out[0] = _basename_any(str(argv[0])) or out[0]
    return out


def _safe_read(root: Path, rel: str, max_bytes: int = 5 * 1024 * 1024) -> Optional[bytes]:
    """Read root/rel only if it is a regular, non-symlinked file inside root."""
    try:
        base = os.path.realpath(str(root))
        full = os.path.realpath(os.path.join(base, rel))
        if not (full == base or full.startswith(base + os.sep)):
            return None
        st = os.lstat(os.path.join(base, rel))
        if not stat.S_ISREG(st.st_mode) or st.st_size > max_bytes:
            return None
        flags = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0)
        fd = os.open(os.path.join(base, rel), flags)
        with os.fdopen(fd, "rb") as fh:
            return fh.read(max_bytes + 1)[:max_bytes]
    except (OSError, ValueError):
        return None


# ── Git context + identity ────────────────────────────────────────────────────

_SCP_REMOTE = re.compile(r"^(?:(?P<user>[^@/\s]+)@)?(?P<host>[A-Za-z0-9.-]+):(?P<path>[^\s]+)$")
_SEGMENT = re.compile(r"^[A-Za-z0-9._-]+$")
#: The canonical `host/owner/repo` form (what this CLI uploads): a dotted host first.
_BARE_REMOTE = re.compile(r"^(?P<host>[A-Za-z0-9](?:[A-Za-z0-9-]*[A-Za-z0-9])?(?:\.[A-Za-z0-9-]+)+)/(?P<path>[^\s]+)$")


def normalize_git_remote(remote: Optional[str]) -> Optional[str]:
    """`host/owner/repo` (lower-case, credentials/port/query/fragment/.git stripped) or None.

    Accepts https/http/ssh/git URLs and scp-style `git@host:owner/repo.git`. Local
    paths and file:// remotes return None. The server re-normalises every upload
    with the same rules — this is convenience, not a control.
    """
    if not remote or not isinstance(remote, str):
        return None
    r = remote.strip()
    if not r or len(r) > 512 or any(ord(ch) < 32 for ch in r):
        return None
    host, path = "", ""
    if "://" in r:
        parts = urllib.parse.urlsplit(r)
        if parts.scheme.lower() not in ("https", "http", "ssh", "git", "git+ssh", "ssh+git"):
            return None
        host = (parts.hostname or "").lower()
        path = parts.path
    else:
        m = _SCP_REMOTE.match(r) or _BARE_REMOTE.match(r)
        if not m or r.startswith(("/", ".", "~")):
            return None
        host, path = m.group("host").lower(), m.group("path")
    if not host or host in ("localhost",) or not re.match(r"^[a-z0-9.-]+$", host):
        return None
    path = path.split("?", 1)[0].split("#", 1)[0].strip("/")
    if path.lower().endswith(".git"):
        path = path[:-4]
    segments = [s for s in path.split("/") if s]
    if len(segments) < 2 or any(s in (".", "..") or not _SEGMENT.match(s) for s in segments):
        return None
    return "{}/{}".format(host, "/".join(s.lower() for s in segments))


def _git(path: Path, *args: str, timeout: int = 15) -> Optional[str]:
    if not which("git"):
        return None
    env, _ = scanner_env({"GIT_TERMINAL_PROMPT": "0"})
    try:
        proc = subprocess.run(["git", "-C", str(path)] + list(args), stdout=subprocess.PIPE,  # noqa: S603
                              stderr=subprocess.PIPE, timeout=timeout, check=False, env=env)
    except (OSError, subprocess.TimeoutExpired):
        return None
    if proc.returncode != 0:
        return None
    out = proc.stdout.decode("utf-8", "replace").strip()
    return out or None


def _ci_branch() -> Optional[str]:
    for name in ("GITHUB_HEAD_REF", "CI_MERGE_REQUEST_SOURCE_BRANCH_NAME", "GITHUB_REF_NAME",
                 "CI_COMMIT_REF_NAME", "BITBUCKET_BRANCH", "BUILDKITE_BRANCH", "CIRCLE_BRANCH",
                 "BUILD_SOURCEBRANCHNAME", "BRANCH_NAME", "GIT_BRANCH"):
        v = (os.environ.get(name) or "").strip()
        if v:
            if name == "GITHUB_REF_NAME" and v.endswith("/merge"):
                continue
            return v[len("origin/"):] if v.startswith("origin/") else v
    return None


#: GitHub event payloads are small JSON files; anything bigger is not read.
_GITHUB_EVENT_MAX_BYTES = 16 * 1024 * 1024
_BRANCH_NAME_RE = re.compile(r"^[^\x00-\x1f\x7f]{1,255}$")


def _valid_branch(v: Any) -> Optional[str]:
    if not isinstance(v, str):
        return None
    v = v.strip()
    if not v or v == "HEAD" or ".." in v or v.startswith("-") or not _BRANCH_NAME_RE.match(v):
        return None
    return v


def _github_event() -> dict:
    """The GitHub Actions event payload ($GITHUB_EVENT_PATH), or {} (bounded, never raises)."""
    path = (os.environ.get("GITHUB_EVENT_PATH") or "").strip()
    if not path or not os.environ.get("GITHUB_ACTIONS"):
        return {}
    try:
        if os.path.getsize(path) > _GITHUB_EVENT_MAX_BYTES:
            return {}
        with open(path, "r", encoding="utf-8") as fh:
            doc = json.load(fh)
    except (OSError, ValueError):
        return {}
    return doc if isinstance(doc, dict) else {}


def _ci_default_branch() -> Optional[str]:
    """The repository's default branch as the CI system states it.

    actions/checkout makes a shallow clone with no refs/remotes/origin/HEAD, so on
    GitHub Actions it is read from the event payload (repository.default_branch).
    """
    for name in ("CI_DEFAULT_BRANCH", "BUILDKITE_PIPELINE_DEFAULT_BRANCH"):
        v = _valid_branch(os.environ.get(name))
        if v:
            return v
    repo = _github_event().get("repository")
    if isinstance(repo, dict):
        return _valid_branch(repo.get("default_branch"))
    return None


_PR_ENV = ("GITHUB_HEAD_REF", "CI_MERGE_REQUEST_IID", "BITBUCKET_PR_ID", "CIRCLE_PULL_REQUEST",
           "SYSTEM_PULLREQUEST_PULLREQUESTID", "CHANGE_ID")


def _ci_is_pull_request() -> bool:
    """A pull/merge-request build: never the monitored default branch, whatever the
    head branch is called (a fork PR from its own `main`)."""
    if any((os.environ.get(n) or "").strip() for n in _PR_ENV):
        return True
    bk = (os.environ.get("BUILDKITE_PULL_REQUEST") or "").strip().lower()
    if bk and bk != "false":
        return True
    return os.environ.get("GITHUB_EVENT_NAME", "").strip() in ("pull_request", "pull_request_target")


def git_context(path: Path, branch_override: Optional[str] = None,
                commit_override: Optional[str] = None) -> dict:
    """Normalised remote, branch, default branch, commit and the project sub-path.

    A folder does not need to be a checkout to be scanned. Credentials in a
    remote URL never leave this function (normalize_git_remote drops them).
    """
    ctx: Dict[str, Any] = {"git_remote": None, "branch": None, "default_branch": None,
                           "git_commit": None, "project_path": "", "is_pull_request": _ci_is_pull_request()}
    top = _git(path, "rev-parse", "--show-toplevel")
    if top:
        ctx["git_remote"] = normalize_git_remote(_git(path, "config", "--get", "remote.origin.url"))
        try:
            rel = os.path.relpath(os.path.realpath(str(path)), os.path.realpath(top))
            ctx["project_path"] = "" if rel in (".", "") else rel.replace(os.sep, "/")
        except ValueError:
            ctx["project_path"] = ""
        head = _git(path, "rev-parse", "--abbrev-ref", "HEAD")
        ctx["branch"] = head if head and head != "HEAD" else None
        ctx["git_commit"] = _git(path, "rev-parse", "HEAD")
        origin_head = _git(path, "symbolic-ref", "--short", "refs/remotes/origin/HEAD")
        if origin_head and origin_head.startswith("origin/"):
            ctx["default_branch"] = origin_head[len("origin/"):]
    ci_branch = _ci_branch()
    if ci_branch:
        ctx["branch"] = ci_branch
    ctx["default_branch"] = _ci_default_branch() or ctx["default_branch"]
    if branch_override:
        ctx["branch"] = branch_override.strip()
    if commit_override:
        if not _HEX_RE.match(commit_override.strip()):
            die("--commit must be a hex git object id (7-64 characters).")
        ctx["git_commit"] = commit_override.strip().lower()
    if ctx["git_commit"] and not _HEX_RE.match(ctx["git_commit"] or ""):
        ctx["git_commit"] = None
    return ctx


class TrackedFilesError(RuntimeError):
    """`git ls-files` failed or timed out inside a checkout: the tracked set is UNKNOWN
    (never treat it as empty — that silently filtered out every finding)."""


#: `git ls-files` on a very large monorepo can take minutes.
GIT_LS_FILES_TIMEOUT = 600


def git_tracked_files(path: Path, timeout: int = GIT_LS_FILES_TIMEOUT) -> Optional[set]:
    """Paths relative to `path` that git tracks; None when `path` is not a checkout.

    `-z` + core.quotePath=false: git otherwise C-quotes non-ASCII and special paths
    (`"caf\\303\\251.env"`), which never equals Trivy's Target, so a committed secret in
    such a file was dropped by --tracked-only. Raises TrackedFilesError when ls-files
    fails or times out; an empty set means the checkout really tracks nothing here.
    """
    if _git(path, "rev-parse", "--is-inside-work-tree") != "true":
        return None
    env, _ = scanner_env({"GIT_TERMINAL_PROMPT": "0"})
    try:
        proc = subprocess.run(["git", "-C", str(path), "-c", "core.quotePath=false", "ls-files", "-z"],  # noqa: S603
                              stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=timeout, check=False, env=env)
    except subprocess.TimeoutExpired as e:
        raise TrackedFilesError("git ls-files timed out after {}s".format(timeout)) from e
    except OSError as e:
        raise TrackedFilesError("git ls-files could not run: {}".format(e)) from e
    if proc.returncode != 0:
        raise TrackedFilesError("git ls-files failed (exit {}): {}".format(
            proc.returncode, _last_line(proc.stderr.decode("utf-8", "replace"))))
    names = proc.stdout.decode("utf-8", "surrogateescape").split("\0")
    return {n for n in names if n}


def annotate_tracking(report: dict, tracked: Optional[set]) -> Tuple[dict, int]:
    """Count findings in files git is not tracking (report unchanged)."""
    if tracked is None:
        return report, 0
    untracked = 0
    for r in report.get("Results") or []:
        target = strip_dot_slash(r.get("Target"))
        if target and target not in tracked:
            untracked += len(r.get("Secrets") or []) + len(r.get("Misconfigurations") or [])
    return report, untracked


def filter_to_tracked(report: dict, tracked: set) -> dict:
    """Keep only results for files git is tracking (`.env`, `.github/**` stay tracked)."""
    kept = [r for r in (report.get("Results") or []) if strip_dot_slash(r.get("Target")) in tracked]
    return dict(report, Results=kept)


# ── Advisory identity (mirror of the server's services/vuln_identity.py) ─────

_CVE_RE = re.compile(r"^CVE-\d{4}-\d{4,}$")
_GHSA_RE = re.compile(r"^GHSA(-[23456789cfghjmpqrvwx]{4}){3}$")


def _norm_id(value: Any) -> Optional[str]:
    if not isinstance(value, str):
        return None
    t = value.strip()
    if not t or len(t) > 128 or re.search(r"[\s\x00-\x1f\x7f]", t):
        return None
    up = t.upper()
    if up.startswith("CVE-"):
        return up if _CVE_RE.match(up) else None
    if up.startswith("GHSA-"):
        cand = "GHSA-" + t[5:].lower()
        return cand if _GHSA_RE.match(cand) else None
    return t


def canonical_id(ids: Iterable[Any]) -> Optional[str]:
    """CVE > GHSA > other; lexicographic within a class (DESIGN C6)."""
    cands = {n for n in (_norm_id(i) for i in ids) if n}
    if not cands:
        return None

    def rank(v: str) -> Tuple[int, str]:
        return (0 if _CVE_RE.match(v) else 1 if _GHSA_RE.match(v) else 2, v)

    return min(cands, key=rank)


# ── Version helpers ───────────────────────────────────────────────────────────


def parse_version_tuple(v: Optional[str]) -> Optional[Tuple[int, int, int]]:
    m = re.match(r"^v?(\d+)\.(\d+)(?:\.(\d+))?", (v or "").strip())
    if not m:
        return None
    return (int(m.group(1)), int(m.group(2)), int(m.group(3) or 0))


def _vtokens(v: str) -> List[Tuple[int, Any]]:
    out: List[Tuple[int, Any]] = []
    for tok in re.findall(r"\d+|[A-Za-z]+", v):
        out.append((1, int(tok)) if tok.isdigit() else (0, tok.lower()))
    return out


def version_cmp(a: Optional[str], b: Optional[str]) -> Optional[int]:
    """Rough generic comparison for local upgrade hints (the server decides facts)."""
    if not a or not b:
        return None
    ta, tb = _vtokens(a), _vtokens(b)
    if not ta or not tb:
        return None
    n = max(len(ta), len(tb))
    ta += [(1, 0)] * (n - len(ta))
    tb += [(1, 0)] * (n - len(tb))
    for x, y in zip(ta, tb):
        if x != y:
            if x[0] != y[0]:
                return -1 if x[0] < y[0] else 1
            return -1 if x[1] < y[1] else 1
    return 0


def fixed_version_for_installed(installed: Optional[str], fixed: Optional[str]) -> Optional[str]:
    """Smallest listed fix strictly above the installed version (Trivy comma lists)."""
    best: Optional[str] = None
    for cand in [p.strip() for p in (fixed or "").split(",") if p.strip()]:
        cmp = version_cmp(cand, installed)
        if cmp is None or cmp <= 0:
            continue
        if best is None or (version_cmp(cand, best) or 0) < 0:
            best = cand
    return best


# ── Trivy ─────────────────────────────────────────────────────────────────────


def ensure_trivy() -> str:
    p = which("trivy")
    if p:
        return p
    hint = {
        "Darwin": "brew install trivy",
        "Linux": "download a pinned release from https://github.com/aquasecurity/trivy/releases "
                 "and verify its checksum (never 0.69.4-0.69.6)",
        "Windows": "scoop install trivy",
    }.get(platform.system(), "https://trivy.dev")
    die("Trivy is not installed on this machine.\n  Install with: {}".format(hint))
    return ""  # unreachable


def scanner_denylist() -> set:
    raw = DEFAULT_SCANNER_DENYLIST + "," + os.environ.get("VSS_SCANNER_VERSION_DENYLIST", "")
    out = set()
    for item in raw.split(","):
        tool, _, ver = item.strip().partition(":")
        if tool and ver:
            out.add((tool.strip().lower(), ver.strip().lstrip("vV")))
    return out


def parse_trivy_version_json(text: str) -> dict:
    """`trivy --version --format json` -> flat provenance dict."""
    info_: Dict[str, Any] = {}
    try:
        data = json.loads(text)
    except (json.JSONDecodeError, ValueError):
        data = None
    if isinstance(data, dict):
        info_["version"] = str(data.get("Version") or "") or None
        vdb = data.get("VulnerabilityDB") or {}
        jdb = data.get("JavaDB") or {}
        info_["db_updated_at"] = vdb.get("UpdatedAt")
        info_["db_next_update"] = vdb.get("NextUpdate")
        info_["db_downloaded_at"] = vdb.get("DownloadedAt")
        info_["db_schema"] = vdb.get("Version")
        info_["java_db_updated_at"] = jdb.get("UpdatedAt")
        info_["java_db_next_update"] = jdb.get("NextUpdate")
        info_["java_db_downloaded_at"] = jdb.get("DownloadedAt")
        info_["check_bundle_digest"] = (data.get("CheckBundle") or {}).get("Digest")
        return info_
    m = re.search(r"Version:\s*v?(\S+)", text or "")
    info_["version"] = m.group(1) if m else None
    return info_


def trivy_info(trivy: str, env: Dict[str, str], cwd: str) -> dict:
    rc, out, err = run_capture([trivy, "--version", "--format", "json"], timeout=60, cwd=cwd, env=env)
    if rc != 0:
        die("`trivy --version` failed: {}".format(_last_line(err) or "exit {}".format(rc)))
    return parse_trivy_version_json(out)


def check_tool_version(tool: str, raw: Optional[str]) -> str:
    """B5 / SUP-3 (phase E): semgrep and gitleaks against SCANNER_VERSION_DENYLIST,
    failing closed on an unparseable version (the Trivy rule)."""
    m = re.search(r"\d+\.\d+\.\d+", raw or "")
    if not m:
        die("Could not determine the {} version (got {!r}); refusing to run an unidentified binary "
            "(the scanner denylist fails closed).".format(tool, (raw or "")[:80]))
    v = m.group(0)
    if (tool, v) in scanner_denylist():
        die("{} {} is on the scanner denylist. Do NOT run it: remove the binary, install a verified "
            "release, and ROTATE every secret this runner can reach.".format(tool, v))
    return v


def check_trivy_version(version: Optional[str]) -> None:
    """Refuse denylisted and too-old Trivy releases (B5, C20)."""
    v = (version or "").strip().lstrip("vV")
    if ("trivy", v) in scanner_denylist():
        die("Trivy {} is on the scanner denylist (credential-stealing release, GHSA-69fq-xp46-6x23). "
            "Do NOT run it: remove the binary, install a verified release, and ROTATE every secret "
            "this runner can reach.".format(v))
    tup = parse_version_tuple(v)
    if tup is None:
        die("Could not determine the Trivy version (got {!r}).".format(version))
    if tup < MIN_TRIVY_VERSION:
        die("Trivy {} is too old: vsscli needs >= {}.{}.{} (--detection-priority, "
            "--image-config-scanners).".format(v, *MIN_TRIVY_VERSION))


def file_sha256(path: str) -> Optional[str]:
    try:
        h = hashlib.sha256()
        with open(os.path.realpath(path), "rb") as fh:
            for chunk in iter(lambda: fh.read(1024 * 1024), b""):
                h.update(chunk)
        return h.hexdigest()
    except OSError:
        return None


def _parse_ts(value: Any) -> Optional[_dt.datetime]:
    if not value or not isinstance(value, str):
        return None
    s = value.strip().replace("Z", "+00:00")
    m = re.match(r"^(\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2})(\.\d+)?([+-]\d{2}:\d{2})?$", s)
    if not m:
        return None
    frac = (m.group(2) or "")[:7]
    try:
        dt = _dt.datetime.fromisoformat(m.group(1) + frac + (m.group(3) or "+00:00"))
    except ValueError:
        return None
    return dt.astimezone(_dt.timezone.utc)


def refresh_trivy_db(trivy: str, env: Dict[str, str], cwd: str, *, need_java: bool,
                     need_checks: bool = False) -> Dict[str, Any]:
    """C14 refresh: `--download-db-only`, then `--download-java-db-only`, then (for
    misconfig scans) a checks-bundle prefetch — so every scan can run with
    --skip-check-update and never fetches anything while reading untrusted content.

    Trivy only downloads when its local copy is older than the latest published
    build, so this yields "the newest DB available at run start"; its UpdatedAt
    is recorded in meta. Artefact sources are pinned (SUP-7) unless the runner
    configured its own mirror through the allow-listed TRIVY_*_REPOSITORY vars.
    """
    db_repo = [] if env.get("TRIVY_DB_REPOSITORY") else ["--db-repository", TRIVY_DB_REPOSITORIES]
    java_repo = [] if env.get("TRIVY_JAVA_DB_REPOSITORY") else ["--java-db-repository", TRIVY_JAVA_DB_REPOSITORIES]
    steps = [("vulnerability DB", [trivy, "image", "--download-db-only", "--no-progress", "--quiet",
                                   "--timeout", "15m"] + db_repo, 960)]
    if need_java:
        steps.append(("Java DB", [trivy, "image", "--download-java-db-only", "--no-progress", "--quiet",
                                  "--timeout", "30m"] + java_repo, 1860))
    done = []
    for label, argv, timeout in steps:
        info("Refreshing the Trivy {} …".format(label))
        rc, _, err = run_capture(argv, timeout=timeout, cwd=cwd, env=env)
        if rc != 0:
            die("Trivy {} download failed: {} (use --offline only with a DB already present).".format(
                label, _last_line(err) or "exit {}".format(rc)))
        done.append(label)
    result: Dict[str, Any] = {"refreshed": done, "check_bundle_prefetch": "not_needed"}
    if need_checks:
        result["check_bundle_prefetch"] = prefetch_checks_bundle(trivy, env, cwd)
    return result


def prefetch_checks_bundle(trivy: str, env: Dict[str, str], cwd: str) -> str:
    """Fetch the misconfig checks bundle into Trivy's cache (no dedicated command
    exists: a `trivy config` of an EMPTY private directory makes Trivy download it;
    nothing else is read). Not fatal — scans then use the checks embedded in the
    binary — but recorded so the upload says which checks ran. Returns 'ok'/'failed'."""
    empty = os.path.join(cwd, "vss-empty-checks")
    try:
        os.mkdir(empty, 0o700)
    except FileExistsError:
        pass
    except OSError as e:
        warn("Trivy checks bundle prefetch skipped ({}); misconfig uses the embedded checks.".format(e))
        return "failed"
    repo = [] if env.get("TRIVY_CHECKS_BUNDLE_REPOSITORY") else [
        "--checks-bundle-repository", TRIVY_CHECKS_BUNDLE_REPOSITORY]
    argv = [trivy, "config", "--quiet", "--timeout", "600s"] + repo + [
        "--skip-version-check", "--disable-telemetry", "--format", "json", "--", empty]
    info("Refreshing the Trivy misconfig checks bundle …")
    rc, _, err = run_capture(argv, timeout=660, cwd=cwd, env=env)
    if rc != 0:
        warn("Trivy checks bundle prefetch failed ({}); misconfig uses the checks embedded in the "
             "Trivy binary.".format(_last_line(err) or "exit {}".format(rc)))
        return "failed"
    return "ok"


def db_state(tinfo: dict, *, offline: bool, need_java: bool, now: Optional[_dt.datetime] = None) -> dict:
    now = now or _dt.datetime.now(_dt.timezone.utc)
    updated = _parse_ts(tinfo.get("db_updated_at"))
    nxt = _parse_ts(tinfo.get("db_next_update"))
    state = {
        "updated_at": tinfo.get("db_updated_at"), "next_update": tinfo.get("db_next_update"),
        "downloaded_at": tinfo.get("db_downloaded_at"),
        "age_hours": round((now - updated).total_seconds() / 3600.0, 1) if updated else None,
        "offline": bool(offline),
        "stale": bool(offline) or updated is None or (nxt is not None and nxt < now),
        "java_db_updated_at": tinfo.get("java_db_updated_at"),
        "java_db_next_update": tinfo.get("java_db_next_update"),
    }
    state["fresh"] = bool(updated is not None and (now - updated).total_seconds() <= FRESH_DB_MAX_AGE_HOURS * 3600
                          and (nxt is None or nxt >= now))
    if updated is None:
        die("Trivy has no vulnerability DB. Run once without --offline to download it.")
    if need_java and not tinfo.get("java_db_updated_at"):
        die("Trivy has no Java DB (needed to identify JARs). Run once without --offline.")
    return state


def _common_scan_flags(out: str, timeout: int, *, offline: bool, java: bool = True,
                       checks: bool = True) -> List[str]:
    flags = ["--format", "json", "--output", out, "--timeout", "{}s".format(int(timeout)),
             "--cache-backend", "memory", "--skip-db-update"]
    if java:
        flags.append("--skip-java-db-update")
    # C14: never fetch the checks bundle while scanning (prefetched by refresh_trivy_db).
    # `trivy sbom` has no misconfig checks and REJECTS the flag ("unknown flag:
    # --skip-check-update", Trivy 0.74.0), exactly as the server's build_sbom_rescan_cmd
    # (WP6-20): its builder passes checks=False.
    if checks:
        flags.append("--skip-check-update")
    flags += ["--disable-telemetry", "--skip-version-check", "--quiet"]
    if offline:
        flags.append("--offline-scan")
    return flags


def _check_operand(value: str, what: str) -> str:
    if not value or value.startswith("-"):
        die("Invalid {} {!r}: it may not be empty or start with '-'.".format(what, value))
    return value


def build_image_argv(trivy: str, image_ref: str, out: str, *, timeout: int, platform_: Optional[str] = None,
                     offline: bool = False, image_src: str = "docker,remote") -> List[str]:
    """The canonical image argv — flag-for-flag the server's
    trivy_scanner.build_image_scan_cmd (DESIGN D7/D8, I1), same order.

    Only differences: no --cache-dir (the runner's own cache) and --offline-scan
    when --offline is given (the server never scans offline).
    """
    argv = [trivy, "image",
            "--format", "json",
            "--output", out,
            "--scanners", "vuln,secret,misconfig",
            "--image-config-scanners", "misconfig,secret",
            "--pkg-types", "os,library",
            "--list-all-pkgs",
            "--severity", TRIVY_SEVERITY_ARG,
            "--timeout", "{}s".format(int(timeout)),
            "--cache-backend", "memory",
            "--skip-db-update", "--skip-java-db-update", "--skip-check-update",
            "--skip-version-check", "--disable-telemetry",
            "--no-progress", "--quiet",
            "--misconfig-scanners", IMAGE_MISCONFIG_SCANNERS,
            "--detection-priority", "comprehensive"]
    if offline:
        argv.append("--offline-scan")
    if platform_:
        argv += ["--platform", platform_]
    argv += ["--max-image-size", IMAGE_MAX_SIZE]
    argv += ["--image-src", image_src, "--", _check_operand(image_ref, "image reference")]
    return argv


def image_scan_env(env: Dict[str, str], workdir: str) -> Dict[str, str]:
    """The server's image_scan_env: no git transport (a `git::` module source can
    never clone) and Trivy's temp/module cache inside the private dir. Registry
    egress must stay (the scan pulls the image), so no dead proxy here."""
    return dict(env, GIT_ALLOW_PROTOCOL="vss-none", GIT_TERMINAL_PROMPT="0", TMPDIR=workdir)


def code_scan_env(env: Dict[str, str], workdir: str) -> Dict[str, str]:
    """No network at all for `trivy fs` over the checkout (WP8 N4 / CISO EXEC-11).

    Trivy 0.74 resolves remote Terraform modules from the scanned tree even with
    --offline-scan; on a CI runner a fork PR's `.tf` would choose the hosts the
    runner connects to (registry, `git clone`, IMDS). secret/misconfig/licence need
    no network: a dead proxy in BOTH cases (Go reads HTTPS_PROXY then https_proxy,
    and an empty NO_PROXY falls through to no_proxy), git with every transport
    disallowed, and TMPDIR in the private dir."""
    out = dict(env)
    for name in ("HTTPS_PROXY", "HTTP_PROXY", "ALL_PROXY", "https_proxy", "http_proxy", "all_proxy"):
        out[name] = NO_EGRESS_PROXY
    out["NO_PROXY"] = ""
    out["no_proxy"] = ""
    out.update({"GIT_ALLOW_PROTOCOL": "vss-none", "GIT_TERMINAL_PROMPT": "0", "TMPDIR": workdir})
    return out


def build_fs_argv(trivy: str, target: str, out: str, *, timeout: int, include_dev: bool = False,
                  offline: bool = False) -> List[str]:
    """OSS pass 1 (lockfiles). `precise`: an unpinned manifest yields NO package rather
    than a fabricated range floor (trivy.md §4)."""
    # --include-dev-deps ALWAYS: Trivy then marks `Dev` (npm/yarn/gradle), so the same
    # packages found again by the rootfs pass under node_modules/ can be tagged dev and
    # excluded when --include-dev is off. `include_dev` is kept for API compatibility;
    # it decides gating/upload in cmd_oss, not the Trivy argv.
    argv = [trivy, "fs", "--scanners", "vuln", "--pkg-types", "library", "--include-dev-deps"]
    argv += ["--list-all-pkgs", "--detection-priority", "precise", "--severity", TRIVY_SEVERITY_ARG]
    for d in OSS_FS_SKIP_DIRS:
        argv += ["--skip-dirs", "**/" + d]
    argv += _common_scan_flags(out, timeout, offline=offline)
    argv.append(_check_operand(target, "path"))
    return argv


def build_rootfs_argv(trivy: str, target: str, out: str, *, timeout: int, offline: bool = False) -> List[str]:
    """OSS pass 2 (installed artefacts): NO dependency-dir excludes — node_modules,
    site-packages, target/*.jar and build/libs are exactly what it must read."""
    argv = [trivy, "rootfs", "--scanners", "vuln", "--pkg-types", "library", "--list-all-pkgs",
            "--detection-priority", "comprehensive", "--severity", TRIVY_SEVERITY_ARG,
            "--skip-dirs", "**/.git"]
    argv += _common_scan_flags(out, timeout, offline=offline)
    argv.append(_check_operand(target, "path"))
    return argv


def build_sbom_argv(trivy: str, sbom_path: str, out: str, *, timeout: int, offline: bool = False) -> List[str]:
    """Build-tool SBOM (cyclonedx-maven/gradle plugin) re-matched by Trivy (C20)."""
    argv = [trivy, "sbom", "--scanners", "vuln", "--list-all-pkgs", "--detection-priority", "comprehensive",
            "--severity", TRIVY_SEVERITY_ARG]
    # checks=False: `trivy sbom` has no --skip-check-update (it exits FATAL on it).
    argv += _common_scan_flags(out, timeout, offline=offline, checks=False)
    argv.append(_check_operand(sbom_path, "SBOM path"))
    return argv


def build_code_trivy_argv(trivy: str, target: str, out: str, *, timeout: int, scanners: str,
                          offline: bool = False) -> List[str]:
    """Code scan: Trivy misconfig (+ secret when gitleaks is absent) with doublestar skip globs."""
    argv = [trivy, "fs", "--scanners", scanners, "--severity", TRIVY_SEVERITY_ARG]
    if "misconfig" in [x.strip() for x in scanners.split(",")]:
        argv += ["--misconfig-scanners", CODE_MISCONFIG_SCANNERS]
    for d in CODE_SKIP_DIRS:
        argv += ["--skip-dirs", "**/" + d]
    argv += _common_scan_flags(out, timeout, offline=offline, java=False)
    argv.append(_check_operand(target, "path"))
    return argv


def run_trivy_json(argv: List[str], out: str, *, timeout: int, cwd: str, env: Dict[str, str],
                   label: str) -> dict:
    """Run a Trivy scan argv; any non-zero exit is an engine error (we never pass --exit-code)."""
    info("Running Trivy {} …".format(label))
    rc, _, err = run_capture(argv, timeout=int(timeout) + 60, cwd=cwd, env=env)
    if rc != 0:
        die("Trivy {} failed: {}".format(label, _last_line(err) or "exit {}".format(rc)))
    try:
        with open(out, "r", encoding="utf-8") as fh:
            text = fh.read()
    except OSError as e:
        die("Trivy {} produced no report: {}".format(label, e))
    try:
        report = json.loads(text or "{}")
    except (json.JSONDecodeError, ValueError) as e:
        die("Trivy {} emitted invalid JSON: {}".format(label, e))
    if not isinstance(report, dict):
        die("Trivy {} emitted an unexpected document.".format(label))
    if report.get("SchemaVersion") not in (None, 2):
        die("Trivy {} report has unsupported SchemaVersion {!r}.".format(label, report.get("SchemaVersion")))
    results = report.get("Results")
    if results is None:
        report["Results"] = []
    elif not isinstance(results, list):
        die("Trivy {} report has a non-list Results.".format(label))
    return report


# ── Project detection (languages, manifests, IaC) ─────────────────────────────


def detect_projects(target: Path, max_files: int = DETECT_MAX_FILES) -> dict:
    """Walk the tree (installed/vendored dirs pruned) for manifests, languages and IaC."""
    manifests: List[dict] = []
    langs: Dict[str, int] = {}
    iac: set = set()
    frameworks: set = set()
    tf_count = 0
    tf_sample: List[str] = []
    files_seen = 0
    truncated = False
    root = str(target)
    for dirpath, dirnames, filenames in os.walk(root):
        dirnames[:] = sorted(d for d in dirnames if d not in _DETECT_SKIP_DIRS and not os.path.islink(os.path.join(dirpath, d)))
        rel_dir = os.path.relpath(dirpath, root)
        rel_dir = "" if rel_dir == "." else rel_dir.replace(os.sep, "/")
        present = set(filenames)
        for name in sorted(filenames):
            files_seen += 1
            if files_seen > max_files:
                truncated = True
                break
            rel = (rel_dir + "/" + name) if rel_dir else name
            ext = os.path.splitext(name)[1].lower()
            if ext in _EXT_LANG:
                langs[_EXT_LANG[ext]] = langs.get(_EXT_LANG[ext], 0) + 1
            if name in _MANIFEST_NAMES:
                eco, kind = _MANIFEST_NAMES[name]
                has_lock = any(lk in present for lk in _LOCK_FOR_MANIFEST.get(name, ()))
                manifests.append({"path": rel, "ecosystem": eco, "kind": kind, "has_lockfile": has_lock})
            elif name.endswith(".csproj") or name.endswith(".fsproj") or name.endswith(".vbproj"):
                manifests.append({"path": rel, "ecosystem": "NuGet", "kind": "manifest",
                                  "has_lockfile": "packages.lock.json" in present})
            elif name.endswith(".deps.json"):
                manifests.append({"path": rel, "ecosystem": "NuGet", "kind": "lockfile", "has_lockfile": True})
            elif re.match(r"^requirements[\w.-]*\.(?:txt|lock)$", name) or (rel_dir.endswith("requirements") and ext == ".txt"):
                manifests.append({"path": rel, "ecosystem": "PyPI", "kind": "manifest", "has_lockfile": False})
            # IaC words follow the server's inventory_checkout (the server's select_configs
            # reads them): dockerfile, terraform, github-actions, ci, docker-compose, yaml.
            lower = name.lower()
            if lower.startswith("dockerfile") or lower.endswith(".dockerfile") or lower == "containerfile":
                iac.add("dockerfile")
            elif lower.endswith(TERRAFORM_SUFFIXES):
                iac.add("terraform")
                tf_count += 1
                if len(tf_sample) < 20:
                    tf_sample.append(rel)
            elif lower == "jenkinsfile" or lower.startswith("jenkinsfile."):
                iac.add("ci")
            elif ext in (".yml", ".yaml"):
                if rel_dir == ".github/workflows" or rel_dir.startswith(".github/workflows/"):
                    iac.add("github-actions")
                    iac.add("ci")
                elif lower.startswith(("docker-compose", "compose")):
                    iac.add("docker-compose")
                elif lower in (".gitlab-ci.yml", ".gitlab-ci.yaml", "bitbucket-pipelines.yml",
                               "azure-pipelines.yml", "azure-pipelines.yaml") or \
                        (rel_dir == ".circleci" and lower in ("config.yml", "config.yaml")):
                    iac.add("ci")
                else:
                    iac.add("yaml")
                    if "kubernetes" not in iac:
                        head = _safe_read(target, rel, max_bytes=4096) or b""
                        if b"apiVersion:" in head and b"kind:" in head:
                            iac.add("kubernetes")
            if name == "manage.py":
                frameworks.add("django")
        if truncated:
            break
    for m in manifests:
        base = os.path.basename(m["path"])
        if base in ("package.json", "requirements.txt", "pyproject.toml", "Gemfile") or base.startswith("requirements"):
            text = (_safe_read(target, m["path"], max_bytes=512 * 1024) or b"").decode("utf-8", "replace").lower()
            for fw, marker in (("react", '"react"'), ("express", '"express"'), ("django", "django"),
                               ("flask", "flask"), ("fastapi", "fastapi"), ("rails", "rails")):
                if marker in text:
                    frameworks.add(fw)
    return {"manifests": manifests, "languages": langs, "iac": sorted(iac), "frameworks": sorted(frameworks),
            "files_seen": min(files_seen, max_files), "truncated": truncated,
            "terraform_files": {"count": tf_count, "sample": tf_sample}}


def classify_manifests(detected: dict, fs_report: Optional[dict]) -> List[dict]:
    """Attach a status to each detected manifest: resolved (Trivy read it or its lockfile) or unresolved."""
    targets = {strip_dot_slash(r.get("Target")) for r in ((fs_report or {}).get("Results") or [])
               if r.get("Class") == "lang-pkgs" and (r.get("Packages") or r.get("Vulnerabilities"))}
    target_dirs = {os.path.dirname(t) for t in targets}
    out = []
    for m in detected.get("manifests") or []:
        path = m["path"]
        if path in targets:
            status, hint = "resolved", None
        elif m["kind"] == "manifest" and m.get("has_lockfile") and os.path.dirname(path) in target_dirs:
            status, hint = "resolved", None
        else:
            status = "unresolved"
            base = os.path.basename(path)
            if base.startswith("build.gradle"):
                hint = ("no gradle.lockfile: vsscli runs gradlew/gradle on a CI runner (elsewhere pass --resolve); "
                        "or enable dependency locking, or --sbom <cyclonedx.json>")
            elif base == "pom.xml":
                hint = "pom.xml not resolved: vsscli runs mvnw/mvn on a CI runner (elsewhere pass --resolve), or --sbom <cyclonedx.json>"
            elif base.startswith("requirements") and base.endswith(".lock"):
                hint = "Trivy only reads requirements*.txt: name the pinned file requirements*.txt, or run after `pip install` into a venv"
            elif base.startswith("requirements"):
                hint = ("unpinned requirements: vsscli resolves them with pip on a CI runner (elsewhere pass "
                        "--resolve); or pin versions, or run after `pip install` into a venv")
            elif base.endswith("proj"):
                hint = "no packages.lock.json: vsscli runs `dotnet restore` on a CI runner (elsewhere pass --resolve)"
            elif m["kind"] == "manifest":
                hint = ("no lockfile next to this manifest: commit one, or let vsscli resolve it on a CI runner "
                        "(elsewhere pass --resolve), or run vsscli oss after the build")
            else:
                hint = "Trivy produced no packages for this file"
        out.append(dict(m, status=status, hint=hint))
    return out


# ── OSS: merge fs ∪ rootfs ∪ sbom by purl ─────────────────────────────────────

_TYPE_ECOSYSTEM = {
    "npm": "npm", "yarn": "npm", "pnpm": "npm", "bun": "npm", "node-pkg": "npm",
    "pip": "PyPI", "pipenv": "PyPI", "poetry": "PyPI", "uv": "PyPI", "python-pkg": "PyPI", "pdm": "PyPI",
    "pom": "Maven", "gradle": "Maven", "jar": "Maven", "sbt": "Maven",
    "gomod": "Go", "gobinary": "Go", "cargo": "crates.io", "rustbinary": "crates.io", "rust-binary": "crates.io",
    "nuget": "NuGet", "dotnet-core": "NuGet", "packages-props": "NuGet",
    "bundler": "RubyGems", "gemspec": "RubyGems", "composer": "Packagist", "composer-vendor": "Packagist",
    "pub": "Pub", "swift": "SwiftURL", "cocoapods": "CocoaPods", "hex": "Hex", "conan": "ConanCenter",
}
_PURL_TYPE_ECOSYSTEM = {
    "npm": "npm", "pypi": "PyPI", "maven": "Maven", "golang": "Go", "cargo": "crates.io", "nuget": "NuGet",
    "gem": "RubyGems", "composer": "Packagist", "pub": "Pub", "swift": "SwiftURL", "cocoapods": "CocoaPods",
    "hex": "Hex", "conan": "ConanCenter",
}
_DEV_AWARE_TYPES = frozenset({"npm", "yarn", "gradle"})


def purl_key(purl: Optional[str]) -> Optional[str]:
    """purl without qualifiers/subpath (the merge key, trivy.md §6)."""
    if not purl or not isinstance(purl, str) or not purl.startswith("pkg:"):
        return None
    return purl.split("?", 1)[0].split("#", 1)[0]


def _ecosystem_for(result_type: Optional[str], purl: Optional[str]) -> str:
    if purl and purl.startswith("pkg:"):
        ptype = purl[4:].split("/", 1)[0].lower()
        if ptype in _PURL_TYPE_ECOSYSTEM:
            return _PURL_TYPE_ECOSYSTEM[ptype]
    return _TYPE_ECOSYSTEM.get((result_type or "").lower(), result_type or "unknown")


def _rootfs_keep(result: dict, pkg: dict) -> bool:
    """Post-filter rootfs: a node-pkg without a node_modules/ segment is the project's own manifest."""
    if (result.get("Type") or "") == "node-pkg":
        return "node_modules/" in (pkg.get("FilePath") or "")
    return True


def _shortest_paths(pkgs: Dict[str, dict]) -> Dict[str, List[str]]:
    """Multi-source BFS over DependsOn from the direct deps (sorted) -> path per node id."""
    direct = sorted(pid for pid, p in pkgs.items() if p.get("Relationship") == "direct")
    if not direct:
        roots = [pid for pid, p in pkgs.items() if p.get("Relationship") in ("root", "workspace")]
        kids = set()
        for r in roots:
            kids.update(pkgs[r].get("DependsOn") or [])
        direct = sorted(k for k in kids if k in pkgs)
    paths: Dict[str, List[str]] = {}
    queue: List[str] = []
    for d in direct:
        if d not in paths:
            paths[d] = [d]
            queue.append(d)
    i = 0
    while i < len(queue):
        cur = queue[i]
        i += 1
        for nxt in sorted(set(pkgs.get(cur, {}).get("DependsOn") or [])):
            if nxt in pkgs and nxt not in paths:
                paths[nxt] = paths[cur] + [nxt]
                queue.append(nxt)
    return paths


def _label(pkg: dict) -> str:
    return "{}@{}".format(pkg.get("Name"), pkg.get("Version") or "")


def _relationship(pkg: dict) -> str:
    rel = pkg.get("Relationship")
    if rel == "direct":
        return "direct"
    if rel == "indirect":
        return "transitive"
    if rel is None and pkg.get("Indirect") is True:
        return "transitive"
    return "unknown"


def _severity(raw: Any) -> str:
    s = str(raw or "").strip().lower()
    if s == "moderate":
        return "medium"
    if s in ("negligible", "info", "informational"):
        return "low"
    return s if s in SEVERITIES else "unknown"


def _cvss_score(v: dict) -> Optional[float]:
    cvss = v.get("CVSS") or {}
    if not isinstance(cvss, dict):
        return None
    order = [v.get("SeveritySource"), "nvd", "ghsa"] + sorted(k for k in cvss if isinstance(k, str))
    for src in order:
        entry = cvss.get(src) if src else None
        if isinstance(entry, dict):
            for key in ("V40Score", "V3Score", "V2Score"):
                if isinstance(entry.get(key), (int, float)):
                    return float(entry[key])
    return None


def _pkg_coord(rtype: Optional[str], purl: Optional[str], name: Any, version: Any) -> Tuple[str, str, str]:
    return (_ecosystem_for(rtype, purl), str(name or "").lower(), str(version or ""))


def dev_package_sets(fs_report: Optional[dict]) -> Tuple[set, set]:
    """(dev package keys, dev coordinates) from the lockfile pass, which always runs
    with --include-dev-deps so Trivy can mark `Dev`. A package that is a prod
    dependency of ANY lockfile in the tree is prod (monorepo: app A ships lodash,
    app B only tests with it), so it is never tagged dev."""
    dev_keys: set = set()
    dev_coords: set = set()
    prod_keys: set = set()
    prod_coords: set = set()
    for result in (fs_report or {}).get("Results") or []:
        if not isinstance(result, dict) or (result.get("Type") or "").lower() not in _DEV_AWARE_TYPES:
            continue
        rtype = result.get("Type")
        for pkg in result.get("Packages") or []:
            if not isinstance(pkg, dict) or not pkg.get("Name") or pkg.get("Relationship") in ("root", "workspace"):
                continue
            purl = (pkg.get("Identifier") or {}).get("PURL")
            coord = _pkg_coord(rtype, purl, pkg.get("Name"), pkg.get("Version"))
            key = purl_key(purl) or "{}:{}@{}".format(coord[0], pkg.get("Name"), pkg.get("Version") or "")
            if pkg.get("Dev") is True:
                dev_keys.add(key)
                dev_coords.add(coord)
            else:
                prod_keys.add(key)
                prod_coords.add(coord)
    return dev_keys - prod_keys, dev_coords - prod_coords


def dev_npm_names(fs_report: Optional[dict]) -> set:
    """npm package NAMES that the lockfile pass lists only as dev (never prod at any
    version). Used to attribute binaries found INSIDE ``node_modules/<pkg>/``."""
    dev: set = set()
    prod: set = set()
    for result in (fs_report or {}).get("Results") or []:
        if not isinstance(result, dict) or (result.get("Type") or "").lower() not in ("npm", "yarn", "pnpm"):
            continue
        for pkg in result.get("Packages") or []:
            if not isinstance(pkg, dict) or not pkg.get("Name") or pkg.get("Relationship") in ("root", "workspace"):
                continue
            (dev if pkg.get("Dev") is True else prod).add(str(pkg["Name"]).lower())
    return dev - prod


def node_modules_owner(path: Optional[str]) -> Optional[str]:
    """``frontend/node_modules/@esbuild/darwin-arm64/bin/esbuild`` -> ``@esbuild/darwin-arm64``:
    the innermost node_modules package that holds a file (None outside node_modules)."""
    parts = [p for p in str(path or "").replace("\\", "/").split("/") if p]
    idx = [i for i, p in enumerate(parts) if p == "node_modules"]
    if not idx:
        return None
    i = idx[-1]
    if i + 1 >= len(parts):
        return None
    name = parts[i + 1]
    if name.startswith("@"):
        if i + 2 >= len(parts):
            return None
        name = name + "/" + parts[i + 2]
    return name.lower()


#: Result types that are packages OF node_modules itself (matched by coordinate);
#: everything else found under node_modules/<pkg>/ is an artefact that package ships.
_NODE_PKG_TYPES = frozenset({"node-pkg", "npm", "yarn", "pnpm"})


def exclude_dev_packages(reports: Sequence[Optional[dict]], dev_keys: set,
                         dev_coords: set, dev_names: Optional[set] = None) -> Tuple[List[Optional[dict]], Dict[str, int]]:
    """--include-dev off (the default, like `snyk test` without --dev): drop dev
    dependencies from EVERY pass, including rootfs sightings of the same package
    under node_modules/ (found by purl, else by ecosystem+name+version), so they
    neither gate, nor upload, nor resolve anything. Returns copies and counts
    (recorded in meta/coverage: never a silent drop)."""
    stats = {"packages": 0, "findings": 0, "dev_package_keys": len(dev_keys), "binaries": 0}
    out: List[Optional[dict]] = []
    dev_names = set(dev_names or ())
    if not dev_keys and not dev_coords and not dev_names:
        return list(reports), stats
    for report in reports:
        if report is None:
            out.append(None)
            continue
        rep = json.loads(json.dumps(report))
        for result in rep.get("Results") or []:
            if not isinstance(result, dict):
                continue
            rtype = result.get("Type")
            # A binary shipped INSIDE a dev-only npm package (esbuild's Go binary under
            # node_modules/@esbuild/darwin-arm64, a build tool that never ships) is dev
            # too: its `pkg:golang/stdlib` can never match a lockfile coordinate, so its
            # findings gated the build without --include-dev (Phase F: 22 of 36).
            owner = node_modules_owner(result.get("Target"))
            if dev_names and owner and owner in dev_names and (rtype or "").lower() not in _NODE_PKG_TYPES:
                n_pkgs = len([p for p in result.get("Packages") or [] if isinstance(p, dict)])
                n_vulns = len([v for v in result.get("Vulnerabilities") or [] if isinstance(v, dict)])
                stats["packages"] += n_pkgs
                stats["findings"] += n_vulns
                stats["binaries"] += 1
                if "Packages" in result:
                    result["Packages"] = []
                if "Vulnerabilities" in result:
                    result["Vulnerabilities"] = []
                continue
            dropped_ids: set = set()
            keep_pkgs = []
            for pkg in result.get("Packages") or []:
                if isinstance(pkg, dict) and pkg.get("Name") and pkg.get("Relationship") not in ("root", "workspace"):
                    purl = (pkg.get("Identifier") or {}).get("PURL")
                    if purl_key(purl) in dev_keys or _pkg_coord(rtype, purl, pkg.get("Name"), pkg.get("Version")) in dev_coords:
                        stats["packages"] += 1
                        if pkg.get("ID"):
                            dropped_ids.add(pkg["ID"])
                        continue
                keep_pkgs.append(pkg)
            if "Packages" in result:
                result["Packages"] = keep_pkgs
            keep_vulns = []
            for v in result.get("Vulnerabilities") or []:
                if isinstance(v, dict):
                    purl = (v.get("PkgIdentifier") or {}).get("PURL")
                    if ((v.get("PkgID") and v.get("PkgID") in dropped_ids) or purl_key(purl) in dev_keys
                            or _pkg_coord(rtype, purl, v.get("PkgName"), v.get("InstalledVersion")) in dev_coords):
                        stats["findings"] += 1
                        continue
                keep_vulns.append(v)
            if "Vulnerabilities" in result:
                result["Vulnerabilities"] = keep_vulns
        out.append(rep)
    return out, stats


def merge_oss(fs_report: Optional[dict], rootfs_report: Optional[dict] = None,
              sbom_report: Optional[dict] = None) -> dict:
    """Union of the three passes, packages keyed by purl, findings keyed by
    (canonical vuln id, package key, pkg_path). Keeps Relationship, DependsOn
    (introduced_through), PURL, Licenses and Dev from the lockfile pass."""
    packages: Dict[str, dict] = {}
    findings: Dict[Tuple[str, str, str], dict] = {}
    #: (vuln id, package key) -> finding: the same advisory on the same package found by
    #: another pass (or at another path) joins the existing finding in O(1).
    by_vuln_pkg: Dict[Tuple[str, str], dict] = {}
    stats = {"fs_packages": 0, "rootfs_packages": 0, "sbom_packages": 0, "rootfs_dropped_project_manifests": 0}

    def pkg_entry(key: str, eco: str, pkg: dict) -> dict:
        if key not in packages:
            packages[key] = {"ecosystem": eco, "name": pkg.get("Name"), "version": pkg.get("Version"),
                             "purl": (pkg.get("Identifier") or {}).get("PURL"), "relationship": "unknown",
                             "scope": "unknown", "licenses": [], "pkg_paths": [], "source_files": [],
                             "introduced_through": None, "passes": [], "start_line": None}
        return packages[key]

    for pass_name, report in (("fs", fs_report), ("rootfs", rootfs_report), ("sbom", sbom_report)):
        for result in (report or {}).get("Results") or []:
            if result.get("Class") not in (None, "lang-pkgs", "os-pkgs"):
                continue
            rtype = result.get("Type")
            target = strip_dot_slash(result.get("Target"))
            pkgs_by_id: Dict[str, dict] = {}
            key_by_id: Dict[str, str] = {}
            for pkg in result.get("Packages") or []:
                if not isinstance(pkg, dict) or not pkg.get("Name"):
                    continue
                if pass_name == "rootfs" and not _rootfs_keep(result, pkg):
                    stats["rootfs_dropped_project_manifests"] += 1
                    continue
                if pkg.get("Relationship") in ("root", "workspace"):
                    if pkg.get("ID"):
                        pkgs_by_id[pkg["ID"]] = pkg
                    continue
                purl = (pkg.get("Identifier") or {}).get("PURL")
                eco = _ecosystem_for(rtype, purl)
                key = purl_key(purl) or "{}:{}@{}".format(eco, pkg.get("Name"), pkg.get("Version") or "")
                entry = pkg_entry(key, eco, pkg)
                stats[pass_name + "_packages"] += 1
                if pass_name not in entry["passes"]:
                    entry["passes"].append(pass_name)
                if pass_name == "fs":
                    entry["relationship"] = _relationship(pkg) if entry["relationship"] == "unknown" else entry["relationship"]
                    if pkg.get("Dev") is True:
                        if entry["scope"] != "prod":   # prod in ANY lockfile wins (monorepo)
                            entry["scope"] = "dev"
                    elif (rtype or "").lower() in _DEV_AWARE_TYPES:
                        entry["scope"] = "prod"
                    if target and target not in entry["source_files"]:
                        entry["source_files"].append(target)
                    locs = pkg.get("Locations") or []
                    if locs and isinstance(locs[0], dict) and entry["start_line"] is None:
                        entry["start_line"] = locs[0].get("StartLine")
                else:
                    fp = strip_dot_slash(pkg.get("FilePath")) or ""
                    if fp and fp not in entry["pkg_paths"]:
                        entry["pkg_paths"].append(fp)
                    if pass_name == "sbom" and target and target not in entry["source_files"]:
                        entry["source_files"].append(target)
                for lic in pkg.get("Licenses") or []:
                    if isinstance(lic, str) and lic not in entry["licenses"]:
                        entry["licenses"].append(lic)
                if pkg.get("ID"):
                    pkgs_by_id[pkg["ID"]] = pkg
                    key_by_id[pkg["ID"]] = key
            if pass_name == "fs" and pkgs_by_id:
                paths = _shortest_paths(pkgs_by_id)
                for pid, path in paths.items():
                    k = key_by_id.get(pid)
                    if k and packages[k]["introduced_through"] is None:
                        packages[k]["introduced_through"] = [_label(pkgs_by_id[p]) for p in path]
            for v in result.get("Vulnerabilities") or []:
                if not isinstance(v, dict) or not v.get("VulnerabilityID") or not v.get("PkgName"):
                    continue
                purl = (v.get("PkgIdentifier") or {}).get("PURL")
                eco = _ecosystem_for(rtype, purl)
                pkey = key_by_id.get(v.get("PkgID") or "") or purl_key(purl) or "{}:{}@{}".format(
                    eco, v.get("PkgName"), v.get("InstalledVersion") or "")
                vid = canonical_id([v.get("VulnerabilityID")] + list(v.get("VendorIDs") or [])) or v["VulnerabilityID"]
                pkg_path = strip_dot_slash(v.get("PkgPath") or (pkgs_by_id.get(v.get("PkgID") or "") or {}).get("FilePath")) or ""
                fkey = (vid, pkey, pkg_path if pass_name != "fs" else "")
                existing = findings.get(fkey) or by_vuln_pkg.get((vid, pkey))
                if existing is None:
                    existing = findings[fkey] = by_vuln_pkg[(vid, pkey)] = {
                        "vuln_id": vid,
                        "aliases": sorted({a for a in [v.get("VulnerabilityID")] + list(v.get("VendorIDs") or [])
                                           if isinstance(a, str) and a != vid}),
                        "severity": _severity(v.get("Severity")),
                        "title": v.get("Title") or vid,
                        "package": v.get("PkgName"), "version": v.get("InstalledVersion"),
                        "purl": purl, "package_key": pkey, "ecosystem": eco,
                        "fixed_version": v.get("FixedVersion") or None,
                        "vendor_status": v.get("Status"),
                        "cvss_score": _cvss_score(v),
                        "cwe": [c_ for c_ in (v.get("CweIDs") or []) if isinstance(c_, str)],
                        "primary_url": v.get("PrimaryURL"),
                        "file": (target if pass_name == "fs" else (pkg_path or target)) or None,
                        "pkg_paths": [], "passes": [], "engines": [],
                        "status": "open",
                    }
                if pass_name not in existing["passes"]:
                    existing["passes"].append(pass_name)
                eng = "trivy-" + pass_name
                if eng not in existing["engines"]:
                    existing["engines"].append(eng)
                if pkg_path and pkg_path not in existing["pkg_paths"]:
                    existing["pkg_paths"].append(pkg_path)
                if pass_name == "fs" and target:
                    existing["file"] = target
    # A rootfs/sbom sighting whose purl differs from the lockfile's (or that has none)
    # still inherits the lockfile's Dev tag by ecosystem+name+version.
    _, dev_coords = dev_package_sets(fs_report)
    if dev_coords:
        for entry in packages.values():
            if entry["scope"] == "unknown" and (entry["ecosystem"], str(entry["name"] or "").lower(),
                                                str(entry["version"] or "")) in dev_coords:
                entry["scope"] = "dev"
    out_findings = []
    for f in findings.values():
        pkg = packages.get(f["package_key"]) or {}
        f["relationship"] = pkg.get("relationship", "unknown")
        f["scope"] = pkg.get("scope", "unknown")
        f["introduced_through"] = pkg.get("introduced_through")
        f["start_line"] = pkg.get("start_line")
        f["fix_available"] = bool(f.get("fixed_version"))
        f["fixed_version_for_installed"] = fixed_version_for_installed(f.get("version"), f.get("fixed_version"))
        if f["relationship"] == "direct":
            f["upgradable"] = f["fixed_version_for_installed"] is not None
        else:
            f["upgradable"] = None   # unknown without a resolver — the gate treats unknown+fix as upgradable
        out_findings.append(f)
    out_findings.sort(key=lambda x: (-SEVERITY_RANK.get(x["severity"], 2), x["vuln_id"], x["package"] or ""))
    return {"packages": sorted(packages.values(), key=lambda p: (p["ecosystem"], p["name"] or "", p["version"] or "")),
            "findings": out_findings, "stats": stats}


# ── Code: Semgrep ─────────────────────────────────────────────────────────────


def select_semgrep_packs(detected: dict, quality: bool = True) -> List[str]:
    """Deterministic pack list, the port of the server's semgrep_scanner.select_configs
    (same tables, same order): security base, languages, frameworks, web classes, IaC,
    then quality packs + per-language quality configs (C18)."""
    langs = set((detected.get("languages") or {}).keys())
    iac = set(detected.get("iac") or [])
    if "terraform" in langs:
        iac.add("terraform")
    if "dockerfile" in langs:
        iac.add("dockerfile")
    out: List[str] = list(SEMGREP_BASE_PACKS)

    def _add(items: Iterable[str]) -> None:
        for p in items:
            if p not in out and p not in SEMGREP_FORBIDDEN_PACKS:
                out.append(p)

    for lang in sorted(langs):
        _add(SEMGREP_LANG_PACKS.get(lang, ()))
    for fw in sorted(set(detected.get("frameworks") or [])):
        _add(SEMGREP_FRAMEWORK_PACKS.get(fw, ()))
    if langs & _WEB_LANGS:
        _add(SEMGREP_WEB_PACKS)
    for kind in sorted(iac):
        _add(SEMGREP_IAC_PACKS.get(kind, ()))
    if quality:
        _add(SEMGREP_QUALITY_PACKS)
        for lang in sorted(langs):
            _add(SEMGREP_QUALITY_LANG_CONFIGS.get(lang, ()))
    return out


def _is_registry_pack(value: str) -> bool:
    """p/<pack> and r/<rule> are fetched from the Semgrep registry (network)."""
    v = (value or "").strip()
    return (v.startswith("p/") or v.startswith("r/")) and not os.path.exists(v)


def validate_semgrep_config(value: str) -> str:
    """Registry pack/rule (`p/...`, `r/...`) or a local rules file you own."""
    v = (value or "").strip()
    if v in SEMGREP_FORBIDDEN_PACKS:
        die("Semgrep config {} is empty or missing in the Community Edition registry.".format(v))
    if _SEMGREP_CONFIG_RE.match(v):
        return v
    if v in ("auto",) or v.startswith(("http://", "https://")):
        die("--semgrep-config {!r} is not allowed (never --config auto; no remote URLs).".format(v))
    if v.startswith("-") or not os.path.isfile(v):
        die("--semgrep-config {!r} is neither a registry pack (p/..., r/...) nor an existing file.".format(v))
    return os.path.abspath(v)


def semgrep_help_flags(semgrep: str, env: Dict[str, str], cwd: str) -> str:
    rc, out, _ = run_capture([semgrep, "scan", "--help"], timeout=120, cwd=cwd, env=env)
    return out if rc == 0 else ""


def build_semgrep_argv(semgrep: str, target: str, out: str, configs: Sequence[str], *, timeout: int = 30,
                       jobs: int = 2, max_memory_mb: int = 4096, ignore_semgrepignore: bool = False,
                       max_target_bytes: int = SEMGREP_MAX_TARGET_BYTES) -> List[str]:
    argv = [semgrep, "scan"]
    for cfg in configs:
        argv += ["--config", cfg]
    argv += ["--json", "--output", out, "--metrics=off", "--disable-version-check",
             "--timeout", str(int(timeout)), "--timeout-threshold", "3", "--max-memory", str(int(max_memory_mb)),
             "--max-target-bytes", str(int(max_target_bytes)), "--jobs", str(int(jobs)), "--dataflow-traces",
             "--disable-nosem"]
    if ignore_semgrepignore:
        argv.append("--x-ignore-semgrepignore-files")
    for d in CODE_SKIP_DIRS:
        argv += ["--exclude", d]
    # --verbose (NOT --quiet), as the server's build_semgrep_cmd: only then does Semgrep
    # write paths.skipped (oversize / timeout / parser skips) into the JSON. The log goes
    # to stderr, which cmd_code sends to a file in the private dir (bounded tail read).
    argv += ["--verbose", _check_operand(target, "path")]
    return argv


# ONE secret classifier, ported byte for byte from the server's
# semgrep_scanner._is_secret_rule (scheme wp8-2026.09.3, phase E): CWE-522 alone
# (Express cookie flags, JWT exposure) is SAST, not a hard-coded credential.
_SECRET_CWES = frozenset({"798", "259", "321"})
_SECRET_ID = re.compile(r"(^|\.)secrets(\.|$)|hardcoded[-_](secret|password|credential|api[-_]?key)", re.IGNORECASE)
_HARDCODED_522 = re.compile(r"hard[-_]?cod|in[-_]source[-_]code", re.IGNORECASE)


def _as_list(v):
    return v if isinstance(v, list) else ([v] if v else [])


def _is_secret_rule(check_id: str, metadata: dict) -> bool:
    cwes = set()
    for c_ in _as_list(metadata.get("cwe")):
        m = re.search(r"(?i)CWE-(\d+)", str(c_))
        if m:
            cwes.add(m.group(1))
    if cwes & _SECRET_CWES or _SECRET_ID.search(check_id or ""):
        return True
    says_secrets = any(str(v).strip().lower() in ("secrets", "secret")
                       for k in ("technology", "subcategory") for v in _as_list(metadata.get(k)))
    return "522" in cwes and (bool(_HARDCODED_522.search(check_id or "")) or says_secrets)


def _rel_to_target(path: str, target: Path) -> str:
    base = str(target).rstrip(os.sep) + os.sep
    p = path or ""
    if p.startswith(base):
        p = p[len(base):]
    elif os.path.isabs(p):
        p = os.path.basename(p)
    return strip_dot_slash(p.replace(os.sep, "/"))


def compute_fingerprint(engine: str, rule_id: str, file_path: str, identity: str) -> str:
    """Same formula as the server's code_scanner.compute_fingerprint."""
    return sha256_hex("{}|{}|{}|{}".format(engine, rule_id, file_path, identity))


#: Metavariable values shorter than this are not treated as secret literals
#: (`=`, `"`, a variable name like `pw`): masking them would shred the snippet.
_SECRET_MIN_LEN = 6
#: Bound on the values masked per result (a pathological rule with thousands of metavars).
_SECRET_MAX_VALUES = 64


_QUOTED_LITERAL_RE = re.compile(r"[\"'`]([^\"'`\s]{%d,})[\"'`]" % _SECRET_MIN_LEN)


#: Mirrors of the server's code_scanner.SECRET_QUOTED / SECRET_ASSIGNMENT: the Semgrep
#: secret digest must be sha256 of the SAME literal the server keys on (and Gitleaks'
#: Secret / Trivy's recovered literal), or one secret becomes one row per engine.
_SECRET_QUOTED = re.compile(r"\"([^\"\\\r\n]{4,})\"|'([^'\\\r\n]{4,})'|`([^`\\\r\n]{4,})`")
_SECRET_ASSIGNMENT = re.compile(
    r"^\s*(?:export\s+)?[A-Za-z_][\w.\-\[\]\"']*\s*(?::=|=>|[:=])\s*(?P<value>[^\s;,]+)\s*[;,]?\s*$")


def primary_secret_literal(region: str) -> str:
    """Port of the server's semgrep_scanner._primary_literal (WP8 #2 / N3).

    * exactly one quoted literal (>= 4 chars, no escapes/newlines) -> its content;
    * no quotes and a single `[export] NAME (=|:|:=|=>) VALUE [;|,]` assignment with a
      one-token value of >= 4 chars -> VALUE;
    * anything else -> the whole whitespace-normalised region (never a guess).
    """
    quoted = [next(g for g in m.groups() if g) for m in _SECRET_QUOTED.finditer(region)]
    if len(quoted) == 1:
        return quoted[0]
    norm = re.sub(r"\s+", " ", region).strip()
    if not quoted:
        m = _SECRET_ASSIGNMENT.match(norm)
        if m and len(m.group("value")) >= 4:
            return m.group("value")
    return norm


def _metavar_values(metavars: Any) -> List[str]:
    out: List[str] = []
    if isinstance(metavars, dict):
        for mv in metavars.values():
            if isinstance(mv, dict):
                for key in ("abstract_content", "propagated_value"):
                    val = mv.get(key)
                    if isinstance(val, dict):
                        val = val.get("svalue_abstract_content")
                    if isinstance(val, str) and val.strip():
                        out.append(val.strip())
    return out[:_SECRET_MAX_VALUES]


def _secret_values(raw: Optional[str], metavar_values: Sequence[str]) -> List[str]:
    """Every literal a secret result could reveal: the whole match, each of its lines,
    and each metavariable value (quotes stripped too), longest first."""
    vals = set()
    raw = (raw or "").strip()
    if raw:
        vals.add(raw)
        vals.update(ln.strip() for ln in raw.splitlines())
        # The quoted literal inside `key = "..."` is what a rule message usually echoes.
        vals.update(m.group(1) for m in _QUOTED_LITERAL_RE.finditer(raw))
    for v in metavar_values:
        vals.add(v)
        vals.add(v.strip("\"'`"))
    return sorted((v for v in vals if len(v) >= _SECRET_MIN_LEN), key=len, reverse=True)[:_SECRET_MAX_VALUES]


def _mask_all(text: str, secrets: Sequence[str]) -> str:
    for v in secrets:
        if v in text:
            text = text.replace(v, mask_secret(v))
    return text


def _leaks(text: Any, secrets: Sequence[str]) -> bool:
    return isinstance(text, str) and any(v in text for v in secrets)


def _rule_title(check_id: str, metadata: Any) -> str:
    md = metadata if isinstance(metadata, dict) else {}
    for key in ("shortDescription", "short_description", "title", "message"):
        if isinstance(md.get(key), str) and md[key].strip():
            return redact(md[key].strip())[:300]
    return "Hard-coded secret ({})".format(check_id.rsplit(".", 1)[-1] or "semgrep")


def postprocess_semgrep(report: dict, target: Path) -> Tuple[dict, dict]:
    """Repo-relative paths, own identity (matched text + occurrence index), redacted snippet
    and secret digest per result under `extra.vss` (the server has no source for CLI uploads)."""
    results = [r for r in (report.get("results") or []) if isinstance(r, dict)]
    stats = {"results": len(results), "degraded_identity": 0, "secret_results": 0}
    file_cache: Dict[str, Optional[bytes]] = {}
    groups: Dict[Tuple[str, str, str], List[dict]] = {}
    for r in results:
        rel = _rel_to_target(str(r.get("path") or ""), target)
        r["path"] = rel
        extra = r.get("extra") if isinstance(r.get("extra"), dict) else {}
        r["extra"] = extra
        # Metavariable values ($SECRET) are what Semgrep interpolates into messages:
        # captured here, before metavars is dropped, so a secret rule can mask them.
        metavar_values = _metavar_values(extra.get("metavars"))
        for k in ("fingerprint", "lines", "metavars", "is_ignored"):
            if extra.get(k) == "requires login" or k == "metavars":
                extra.pop(k, None)
        if rel not in file_cache:
            file_cache[rel] = _safe_read(target, rel)
        data = file_cache[rel]
        start = r.get("start") or {}
        end = r.get("end") or {}
        raw_text = None
        if data is not None and isinstance(start.get("offset"), int) and isinstance(end.get("offset"), int):
            raw_text = data[start["offset"]:end["offset"]].decode("utf-8", "replace")
        norm = re.sub(r"\s+", " ", raw_text).strip() if raw_text is not None else None
        metadata = extra.get("metadata") if isinstance(extra.get("metadata"), dict) else {}
        check_id = str(r.get("check_id") or "")
        is_secret = _is_secret_rule(check_id, metadata)
        digest = sha256_hex(primary_secret_literal(raw_text)) if (is_secret and raw_text and raw_text.strip()) else None
        r["_vss_tmp"] = {"norm": norm, "digest": digest, "is_secret": is_secret, "raw": raw_text,
                         "metavar_values": metavar_values}
        ident_base = digest if is_secret else norm
        groups.setdefault((check_id, rel, ident_base or ""), []).append(r)
    for (check_id, rel, ident_base), members in groups.items():
        members.sort(key=lambda x: ((x.get("start") or {}).get("offset") or 0))
        for k, r in enumerate(members):
            tmp = r.pop("_vss_tmp")
            extra = r["extra"]
            if ident_base:
                identity = "{}|{}".format(ident_base, k)
                quality = "ok"
            else:
                start = r.get("start") or {}
                identity = "L{}:C{}".format(start.get("line"), start.get("col"))
                quality = "degraded"
                stats["degraded_identity"] += 1
            secrets = _secret_values(tmp["raw"], tmp["metavar_values"]) if tmp["is_secret"] else []
            snippet = None
            data = file_cache.get(rel)
            if data is not None:
                lines = data.decode("utf-8", "replace").splitlines()
                s_line = int((r.get("start") or {}).get("line") or 1)
                e_line = int((r.get("end") or {}).get("line") or s_line)
                chosen = lines[max(0, s_line - 1):min(len(lines), e_line, s_line - 1 + 6)]
                # Mask the FULL line first, then clip: a secret past column 200 (minified
                # JS) or on a later line of the match must not survive the clip.
                snippet = "\n".join(_mask_all(ln, secrets)[:200] for ln in chosen)
            if tmp["is_secret"]:
                stats["secret_results"] += 1
                for k in ("dataflow_trace", "fix", "fix_regex", "fixed_lines", "lines"):
                    extra.pop(k, None)
                msg = extra.get("message")
                if isinstance(msg, str):
                    extra["message"] = _mask_all(msg, secrets)
                # Fail closed: if any secret literal is still visible anywhere we would
                # upload, drop the snippet and replace the message with the rule title.
                if _leaks(snippet, secrets) or _leaks(extra.get("message"), secrets):
                    stats["secret_snippets_dropped"] = stats.get("secret_snippets_dropped", 0) + 1
                    snippet = None
                    extra["message"] = _rule_title(check_id, extra.get("metadata"))
            elif snippet is None and isinstance(extra.get("lines"), str):
                extra["lines"] = redact(extra["lines"])
            if snippet is not None:
                snippet = redact(snippet)
                extra["lines"] = snippet
            extra["message"] = redact(extra.get("message")) if isinstance(extra.get("message"), str) else extra.get("message")
            extra["vss"] = {"identity": identity, "snippet": snippet, "secret_digest": tmp["digest"],
                            "fp_quality": quality,
                            "fingerprint": compute_fingerprint("semgrep", check_id, rel, identity)}
    stats["skipped"] = normalise_semgrep_paths(report, target)
    return report, stats


#: Semgrep skip reasons after which the file was NOT (fully) analysed — the server
#: must never resolve a finding in such a file (semgrep_scanner.run_semgrep's list).
SEMGREP_UNANALYSED_REASONS = frozenset({"exceeded_size_limit", "analysis_failed_parser_or_internal_error",
                                        "timeout", "too_many_matches"})


def normalise_semgrep_paths(report: dict, target: Path) -> dict:
    """Repo-relative `paths.skipped[].path` and `errors[].path` / spans (no runner paths
    in the upload, and the same relative form the findings use, so the server can keep
    these files out of resolution). `paths.scanned` (every file, absolute) is dropped.
    Returns {total, by_reason, unanalysed, sample, sample_is_partial} for coverage."""
    paths = report.get("paths") if isinstance(report.get("paths"), dict) else {}
    skipped_in = paths.get("skipped") if isinstance(paths.get("skipped"), list) else []
    skipped: List[dict] = []
    by_reason: Dict[str, int] = {}
    unanalysed = 0
    for item in skipped_in:
        if not isinstance(item, dict):
            continue
        reason = str(item.get("reason") or "unknown")
        rel = _rel_to_target(str(item.get("path") or ""), target)
        by_reason[reason] = by_reason.get(reason, 0) + 1
        if reason in SEMGREP_UNANALYSED_REASONS:
            unanalysed += 1
        skipped.append({"path": rel, "reason": reason})
    report["paths"] = {"skipped": skipped} if ("skipped" in paths) else {}
    for e in report.get("errors") or []:
        if not isinstance(e, dict):
            continue
        if isinstance(e.get("path"), str):
            e["path"] = _rel_to_target(e["path"], target)
        for span in e.get("spans") or []:
            if isinstance(span, dict) and isinstance(span.get("file"), str):
                span["file"] = _rel_to_target(span["file"], target)
    return {"reported": "skipped" in paths, "total": len(skipped), "by_reason": by_reason,
            "unanalysed": unanalysed, "sample": skipped[:200], "sample_is_partial": len(skipped) > 200}


def oversize_files(target: Path, limit_bytes: int, *, skip_dirs: Iterable[str] = (".git",),
                   max_listed: int = SKIPPED_PATHS_MAX) -> Tuple[List[str], int]:
    """Regular files larger than `limit_bytes` anywhere under target (symlinks never
    followed; gitleaks does not use our excludes) -> (sorted repo-relative paths, listed
    at most `max_listed`, total count). The server's inventory walk for
    GITLEAKS_MAX_TARGET_MB (WP8-20), so a CLI upload records the same bound."""
    root = str(target)
    found: List[str] = []
    total = 0
    skip = set(skip_dirs)
    for dirpath, dirnames, filenames in os.walk(root, followlinks=False):
        dirnames[:] = sorted(d for d in dirnames if d not in skip)
        rel_dir = os.path.relpath(dirpath, root)
        for name in sorted(filenames):
            full = os.path.join(dirpath, name)
            try:
                st = os.lstat(full)
            except OSError:
                continue
            if not stat.S_ISREG(st.st_mode) or st.st_size <= limit_bytes:
                continue
            total += 1
            if len(found) < max_listed:
                rel = name if rel_dir == "." else rel_dir.replace(os.sep, "/") + "/" + name
                found.append(rel)
    return sorted(found), total


# ── Code: Gitleaks (C17) ──────────────────────────────────────────────────────

GITLEAKS_CONFIG_TOML = 'title = "vsscli default (gitleaks built-in rules)"\n\n[extend]\nuseDefault = true\n'


def build_gitleaks_argv(gitleaks: str, target: str, report_path: str, config_path: str, ignore_dir: str,
                        help_text: str = "", max_target_mb: int = GITLEAKS_MAX_TARGET_MB) -> Tuple[List[str], List[str]]:
    """`gitleaks dir` WITHOUT --redact (the raw Secret is needed for the digest), report to a
    0600 file, `--exit-code 0` so any non-zero exit is an engine error. Flags not listed
    by the installed version's help are dropped and reported."""
    wanted = [("--config", config_path), ("--report-format", "json"), ("--report-path", report_path),
              ("--exit-code", "0"), ("--no-banner", None), ("--log-level", "error"),
              ("--max-target-megabytes", str(int(max_target_mb))), ("--ignore-gitleaks-allow", None),
              ("--gitleaks-ignore-path", ignore_dir)]
    required = {"--config", "--report-format", "--report-path", "--exit-code"}
    argv = [gitleaks, "dir", _check_operand(target, "path")]
    dropped: List[str] = []
    for flag, value in wanted:
        if help_text and flag not in help_text and flag not in required:
            dropped.append(flag)
            continue
        argv.append(flag)
        if value is not None:
            argv.append(value)
    return argv, dropped


def process_gitleaks_findings(raw: Any, target: Path) -> List[dict]:
    """Digest the raw Secret, then DROP it; mask Match/Line; drop author/email/commit metadata."""
    if not isinstance(raw, list):
        die("gitleaks report is not a JSON list.")
    out: List[dict] = []
    for f in raw:
        if not isinstance(f, dict):
            continue
        secret = f.get("Secret") if isinstance(f.get("Secret"), str) else ""
        digest = sha256_hex(secret) if secret else None
        masked = mask_secret(secret) if secret else "[REDACTED]"

        def _mask(text: Any, secret: str = secret, masked: str = masked) -> Any:
            if not isinstance(text, str):
                return text
            return redact(text.replace(secret, masked) if secret else text)

        item = {
            "RuleID": f.get("RuleID"), "Description": f.get("Description"),
            "File": _rel_to_target(str(f.get("File") or ""), target),
            "StartLine": f.get("StartLine"), "EndLine": f.get("EndLine"),
            "StartColumn": f.get("StartColumn"), "EndColumn": f.get("EndColumn"),
            "Match": _mask(f.get("Match")), "Line": _mask(f.get("Line")),
            "Entropy": f.get("Entropy"), "Tags": f.get("Tags") if isinstance(f.get("Tags"), list) else [],
            "Fingerprint": redact(str(f.get("Fingerprint") or "")) or None,
            "VssSecretDigest": digest,
        }
        out.append(item)
    return out


def digest_trivy_secrets(report: dict, target: Path) -> int:
    """Recover each Trivy secret from the file span under the '*' run of Match, store only
    `_vss_secret_digest`, and re-redact Match/Code lines. Returns the count digested."""
    count = 0
    for r in report.get("Results") or []:
        rel = strip_dot_slash(r.get("Target"))
        data = None
        for s in r.get("Secrets") or []:
            if not isinstance(s, dict):
                continue
            if data is None:
                data = _safe_read(target, rel) or b""
            lines = data.decode("utf-8", "replace").splitlines()
            ln = s.get("StartLine")
            match = s.get("Match") if isinstance(s.get("Match"), str) else ""
            literal = None
            if isinstance(ln, int) and 1 <= ln <= len(lines) and "*" in match:
                m = re.search(r"\*+", match)
                line = lines[ln - 1]
                prefix = match[:m.start()]
                idx = line.find(prefix) if prefix else 0
                if idx >= 0:
                    start = idx + len(prefix)
                    literal = line[start:start + (m.end() - m.start())] or None
            if literal:
                s["_vss_secret_digest"] = sha256_hex(literal)
                count += 1
            s["Match"] = redact(match)
            code = s.get("Code") or {}
            for cl in code.get("Lines") or [] if isinstance(code, dict) else []:
                if isinstance(cl, dict):
                    for k in ("Content", "Highlighted"):
                        if isinstance(cl.get(k), str):
                            cl[k] = redact(cl[k].replace(literal, mask_secret(literal)) if literal else cl[k])
    return count


def redact_misconfig_code(report: dict) -> None:
    for r in report.get("Results") or []:
        for m in r.get("Misconfigurations") or []:
            if not isinstance(m, dict):
                continue
            m["Message"] = redact(m.get("Message"))
            code = ((m.get("CauseMetadata") or {}).get("Code") or {}) if isinstance(m.get("CauseMetadata"), dict) else {}
            for cl in code.get("Lines") or [] if isinstance(code, dict) else []:
                if isinstance(cl, dict):
                    for k in ("Content", "Highlighted"):
                        cl[k] = redact(cl.get(k))


def normalise_misconfig_id(raw: Optional[str]) -> str:
    s = (raw or "").upper()
    if s.startswith("AVD-"):
        s = s[4:]
    m = re.match(r"^([A-Z]+(?:-[A-Z]+)*)-?0*(\d+)$", s)
    return "{}-{}".format(m.group(1), int(m.group(2))) if m else s


# ── Dockerfile / base image ───────────────────────────────────────────────────


def parse_dockerfile(path: str) -> dict:
    """FROM lines of the Dockerfile and the base image of the FINAL stage (stage aliases resolved)."""
    try:
        with open(path, "r", encoding="utf-8", errors="replace") as fh:
            text = fh.read(2 * 1024 * 1024)
    except OSError as e:
        die("Cannot read --file {}: {}".format(path, e))
    joined = re.sub(r"\\\r?\n", " ", text)
    froms: List[str] = []
    stages: Dict[str, str] = {}
    last_base: Optional[str] = None
    args: Dict[str, str] = {}
    for line in joined.splitlines():
        s = line.strip()
        m_arg = re.match(r"(?i)^ARG\s+([A-Za-z_][A-Za-z0-9_]*)=(\S+)", s)
        if m_arg and not froms:
            args[m_arg.group(1)] = m_arg.group(2).strip("\"'")
        m = re.match(r"(?i)^FROM\s+(?:--platform=\S+\s+)?(\S+)(?:\s+AS\s+(\S+))?", s)
        if not m:
            continue
        ref = m.group(1)
        ref = re.sub(r"\$\{?([A-Za-z_][A-Za-z0-9_]*)\}?", lambda mm: args.get(mm.group(1), mm.group(0)), ref)
        froms.append(ref)
        resolved = stages.get(ref.lower(), ref)
        if m.group(2):
            stages[m.group(2).lower()] = resolved
        last_base = resolved
    base = last_base if last_base and "$" not in last_base and last_base.lower() != "scratch" else None
    return {"path": os.path.basename(path), "from": froms, "base_image": base}


def base_layer_count(image_diff_ids: List[str], base_diff_ids: List[str]) -> int:
    n = 0
    for a, b in zip(image_diff_ids, base_diff_ids):
        if a != b:
            break
        n += 1
    return n


def image_findings(report: dict, base_diff_ids: Optional[List[str]] = None) -> List[dict]:
    """Normalised image findings (vulns + secrets + misconfig) for gating/JSON/SARIF."""
    diff_ids = list((report.get("Metadata") or {}).get("DiffIDs") or [])
    n_base = base_layer_count(diff_ids, base_diff_ids) if base_diff_ids else 0
    base_set = set(diff_ids[:n_base])
    out: List[dict] = []
    seen = set()
    for r in report.get("Results") or []:
        cls = r.get("Class")
        for v in r.get("Vulnerabilities") or []:
            if not isinstance(v, dict) or not v.get("VulnerabilityID"):
                continue
            vid = canonical_id([v.get("VulnerabilityID")] + list(v.get("VendorIDs") or [])) or v["VulnerabilityID"]
            key = (vid, v.get("PkgName"), strip_dot_slash(v.get("PkgPath")) or "")
            if key in seen:
                continue
            seen.add(key)
            layer = ((v.get("Layer") or {}).get("DiffID")) if isinstance(v.get("Layer"), dict) else None
            in_base = None
            if base_diff_ids:
                in_base = bool(layer and layer in base_set) if layer else None
            fixed = v.get("FixedVersion") or None
            out.append({"kind": "vuln", "vuln_id": vid, "severity": _severity(v.get("Severity")),
                        "title": v.get("Title") or vid, "package": v.get("PkgName"),
                        "version": v.get("InstalledVersion"), "fixed_version": fixed,
                        "fix_available": bool(fixed),
                        "upgradable": bool(fixed) if cls == "os-pkgs" else None,
                        "pkg_class": cls, "pkg_path": key[2], "in_base_image": in_base,
                        "cvss_score": _cvss_score(v), "cwe": list(v.get("CweIDs") or []),
                        "vendor_status": v.get("Status"), "status": "open", "scope": "unknown",
                        "file": None, "engines": ["trivy-image"]})
        for s in r.get("Secrets") or []:
            if isinstance(s, dict):
                out.append({"kind": "secret", "vuln_id": s.get("RuleID"), "severity": _severity(s.get("Severity")),
                            "title": redact(s.get("Title") or s.get("RuleID")), "package": None,
                            "file": strip_dot_slash(r.get("Target")), "start_line": s.get("StartLine"),
                            "fix_available": False, "upgradable": False, "status": "open", "scope": "unknown",
                            "engines": ["trivy-image"]})
        for m in r.get("Misconfigurations") or []:
            if isinstance(m, dict) and (m.get("Status") or "FAIL") == "FAIL":
                out.append({"kind": "misconfig", "vuln_id": m.get("ID"), "severity": _severity(m.get("Severity")),
                            "title": m.get("Title") or m.get("ID"), "package": None,
                            "file": strip_dot_slash(r.get("Target")),
                            "start_line": (m.get("CauseMetadata") or {}).get("StartLine") if isinstance(m.get("CauseMetadata"), dict) else None,
                            "fix_available": bool(m.get("Resolution")), "upgradable": False, "status": "open",
                            "scope": "unknown", "engines": ["trivy-image"]})
    return out


# ── Gate (single implementation; outputs-and-gating.md §2) ────────────────────


class GateConfigError(ValueError):
    pass


def parse_gate_args(severity_threshold: Optional[str], fail_on: Optional[str]) -> Tuple[str, Optional[str]]:
    """Validate the gate flags. Unknown tokens are an ERROR (exit 2), never ignored.

    `--fail-on` takes Snyk's all|upgradable (+ none). For backward compatibility a
    severity list (`critical,high`) is accepted and treated as `--severity-threshold`
    of its LOWEST severity — `>=` semantics, so `--fail-on high` now fails on critical too.
    """
    threshold = (severity_threshold or "").strip().lower() or None
    if threshold is not None and threshold not in THRESHOLDS:
        raise GateConfigError("--severity-threshold must be one of {}".format(", ".join(THRESHOLDS)))
    mode: Optional[str] = None
    raw = (fail_on or "").strip().lower()
    if raw:
        tokens = [t.strip() for t in raw.split(",") if t.strip()]
        if tokens == ["patchable"]:
            raise GateConfigError("--fail-on patchable is not supported: VSS has no patch feed. "
                                  "Use --fail-on all or --fail-on upgradable.")
        if len(tokens) == 1 and tokens[0] in FAIL_ON_CHOICES:
            mode = tokens[0]
        elif tokens and all(t in SEVERITIES for t in tokens):
            legacy = min((t if t != "unknown" else "medium" for t in tokens), key=lambda t: SEVERITY_RANK[t])
            if threshold is not None and threshold != legacy:
                raise GateConfigError("--fail-on {} (a legacy severity list) conflicts with --severity-threshold {}"
                                      .format(fail_on, threshold))
            warn("--fail-on {} is deprecated: treated as --severity-threshold {} (>= semantics; "
                 "criticals now fail a 'high' gate).".format(fail_on, legacy))
            threshold = legacy
        else:
            raise GateConfigError("--fail-on {!r} is not valid: use all, upgradable or none "
                                  "(or --severity-threshold low|medium|high|critical)".format(fail_on))
    return threshold or "low", mode


def _is_open(f: dict, now: _dt.datetime) -> bool:
    st = str(f.get("status") or "open").strip().lower()
    if st in _NON_GATING_STATUSES:
        return False
    if f.get("suppressed") is not None and "suppressed" in f:
        return not bool(f.get("suppressed"))
    if st in _STICKY_STATUSES:
        until = _parse_ts(f.get("suppress_until"))
        return until is not None and until <= now
    return True


def evaluate_gate(findings: Sequence[dict], threshold: str = "low", fail_on: Optional[str] = None,
                  include_dev: bool = False, exclude_base_image_vulns: bool = False,
                  now: Optional[_dt.datetime] = None) -> Tuple[bool, List[dict]]:
    """Returns (fail, gating findings). Errors never reach here (exit 2 earlier)."""
    now = now or _dt.datetime.now(_dt.timezone.utc)
    rank = SEVERITY_RANK[threshold]
    cands = [f for f in findings if _is_open(f, now)]
    if not include_dev:
        cands = [f for f in cands if str(f.get("scope") or "unknown") != "dev"]
    if exclude_base_image_vulns:
        cands = [f for f in cands if not (f.get("in_base_image") is True and f.get("pkg_class") == "os-pkgs")]
    gating = [f for f in cands if SEVERITY_RANK.get(_severity(f.get("severity")), 2) >= rank]
    # `gating` is exactly the set that makes the build fail, so a finding marked
    # "gating": true can never sit next to exit 0 (Phase F: --fail-on upgradable
    # reported 218 gating code findings and exited 0).
    if fail_on == "all":
        gating = [f for f in gating if bool(f.get("fix_available"))]
    elif fail_on == "upgradable":
        gating = [f for f in gating
                  if f.get("upgradable") is True or (f.get("upgradable") is None and bool(f.get("fix_available")))]
    elif fail_on == "none":
        gating = []
    return bool(gating), gating


def severity_counts(findings: Iterable[dict]) -> Dict[str, int]:
    counts = {s: 0 for s in SEVERITIES}
    for f in findings:
        counts[_severity(f.get("severity"))] += 1
    return counts


# ── SARIF 2.1.0 (GitHub code scanning; text only, redacted) ───────────────────

_LEVEL = {"critical": "error", "high": "error", "medium": "warning", "low": "note", "unknown": "note"}
_DEFAULT_SEC_SEV = {"critical": "9.5", "high": "8.0", "medium": "5.5", "low": "2.0", "unknown": "5.0"}
SARIF_SCHEMA = "https://json.schemastore.org/sarif-2.1.0.json"


def _clip(text: Any, n: int) -> str:
    s = redact(str(text or "")).replace("\r", " ")
    return s if len(s) <= n else s[: n - 1] + "…"


def _sarif_uri(path: Optional[str]) -> str:
    p = strip_dot_slash((path or "").replace("\\", "/"))
    if not p or os.path.isabs(p) or p.startswith("../"):
        p = os.path.basename(p) or "unknown"
    return urllib.parse.quote(p, safe="/._-~$@+,:;=")


#: SARIF rule tags per category. GitHub treats `security` (with security-severity) as a
#: security alert: quality and licence results are not security findings.
_SECURITY_CATEGORIES = frozenset({"sast", "secret", "misconfig", "oss", "vuln", "image"})
_QUALITY_TAGS = {"correctness": "correctness", "best-practice": "maintainability", "maintainability": "maintainability",
                 "performance": "performance", "compatibility": "maintainability", "portability": "maintainability"}


def sarif_tags(f: dict) -> List[str]:
    cat = str(f.get("category") or "")
    if cat == "quality":
        tags = [_QUALITY_TAGS.get(str(f.get("quality_kind") or ""), "maintainability"), "quality"]
    elif cat == "license":
        tags = ["license", "compliance"]
    else:
        tags = ["security"] + ([cat] if cat else [])
    for cwe in f.get("cwe") or []:
        m = re.match(r"(?i)^\s*CWE-(\d+)", str(cwe))
        if m:
            tags.append("external/cwe/cwe-{}".format(int(m.group(1))))
    return list(dict.fromkeys(tags))


def build_sarif(runs: Sequence[dict]) -> dict:
    """runs: [{"category": "vss/oss/", "tool": "vsscli-oss", "results": [normalised finding, ...]}]."""
    doc_runs = []
    for spec in runs:
        rules: List[dict] = []
        index: Dict[str, int] = {}
        results = []
        for f in spec.get("results") or []:
            rule_id = str(f.get("rule_id") or "unknown")
            sev = _severity(f.get("severity"))
            level = _LEVEL[sev]
            if rule_id not in index:
                index[rule_id] = len(rules)
                score = f.get("cvss_score")
                sec_sev = "{:.1f}".format(float(score)) if isinstance(score, (int, float)) and score > 0 else _DEFAULT_SEC_SEV[sev]
                tags = sarif_tags(f)
                rules.append({
                    "id": rule_id,
                    "name": re.sub(r"[^A-Za-z0-9]", "", rule_id.title())[:100] or "Rule",
                    "shortDescription": {"text": _clip(f.get("short") or f.get("title") or rule_id, 200)},
                    "fullDescription": {"text": _clip(f.get("full") or f.get("title") or rule_id, 1000)},
                    "help": {"text": _clip(f.get("help") or f.get("title") or rule_id, 2000)},
                    "defaultConfiguration": {"level": level},
                    "properties": dict({"tags": tags[:20], "precision": f.get("precision") or "high"},
                                       **({"security-severity": sec_sev} if "security" in tags else {})),
                })
            region: Dict[str, int] = {"startLine": max(1, int(f.get("start_line") or 1))}
            if isinstance(f.get("end_line"), int) and f["end_line"] >= region["startLine"]:
                region["endLine"] = f["end_line"]
            if isinstance(f.get("start_col"), int) and f["start_col"] >= 1:
                region["startColumn"] = f["start_col"]
            result = {
                "ruleId": rule_id, "ruleIndex": index[rule_id], "level": level,
                "message": {"text": _clip(f.get("message") or f.get("title") or rule_id, 4000)},
                "locations": [{"physicalLocation": {
                    "artifactLocation": {"uri": _sarif_uri(f.get("uri")), "uriBaseId": "%SRCROOT%"},
                    "region": region}}],
            }
            if f.get("fingerprint"):
                result["partialFingerprints"] = {"vssFingerprint/v1": str(f["fingerprint"])}
            results.append(result)
        doc_runs.append({
            "tool": {"driver": {"name": spec.get("tool") or "vsscli", "semanticVersion": VSSCLI_VERSION,
                                "informationUri": "https://github.com/aquasecurity/trivy", "rules": rules}},
            "automationDetails": {"id": spec["category"]},
            "results": results,
        })
    return {"$schema": SARIF_SCHEMA, "version": "2.1.0", "runs": doc_runs}


def validate_sarif(doc: dict) -> List[str]:
    """Structural checks GitHub code scanning enforces. Returns a list of problems."""
    problems: List[str] = []
    if doc.get("version") != "2.1.0":
        problems.append("version must be 2.1.0")
    runs = doc.get("runs")
    if not isinstance(runs, list) or not runs:
        return problems + ["runs must be a non-empty list"]
    if len(runs) > 20:
        problems.append("more than 20 runs")
    cats = [((r.get("automationDetails") or {}).get("id")) for r in runs]
    if len(set(cats)) != len(cats) or None in cats:
        problems.append("automationDetails.id must be present and distinct per run")
    for r in runs:
        rules = ((r.get("tool") or {}).get("driver") or {}).get("rules") or []
        ids = [x.get("id") for x in rules]
        if len(set(ids)) != len(ids):
            problems.append("duplicate rule ids")
        for rule in rules:
            for key in ("shortDescription", "fullDescription", "help"):
                if not ((rule.get(key) or {}).get("text")):
                    problems.append("rule {} lacks {}.text".format(rule.get("id"), key))
            if len(((rule.get("fullDescription") or {}).get("text") or "")) > 1024:
                problems.append("rule {} fullDescription > 1024 chars".format(rule.get("id")))
            if len(((rule.get("properties") or {}).get("tags") or [])) > 20:
                problems.append("rule {} has more than 20 tags".format(rule.get("id")))
        results = r.get("results") or []
        if len(results) > 25000:
            problems.append("run {} has more than 25000 results".format(r.get("automationDetails")))
        for res in results:
            idx = res.get("ruleIndex")
            if not isinstance(idx, int) or idx >= len(rules) or rules[idx].get("id") != res.get("ruleId"):
                problems.append("result ruleIndex does not point at its ruleId")
            if res.get("level") not in ("error", "warning", "note", "none"):
                problems.append("bad level")
            if "markdown" in json.dumps(res.get("message") or {}):
                problems.append("message must use text only")
            for loc in res.get("locations") or []:
                uri = (((loc.get("physicalLocation") or {}).get("artifactLocation")) or {}).get("uri") or ""
                if uri.startswith("/") or "://" in uri:
                    problems.append("location uri must be repo-relative")
                reg = ((loc.get("physicalLocation") or {}).get("region")) or {}
                if not isinstance(reg.get("startLine"), int) or reg["startLine"] < 1:
                    problems.append("region.startLine must be >= 1")
    return problems


# ── CycloneDX ─────────────────────────────────────────────────────────────────


def trivy_convert_cyclonedx(trivy: str, report_path: str, out_path: str, *, env: Dict[str, str], cwd: str) -> Optional[dict]:
    rc, _, err = run_capture([trivy, "convert", "--format", "cyclonedx", "--output", out_path, report_path],
                             timeout=600, cwd=cwd, env=env)
    if rc != 0:
        warn("trivy convert to CycloneDX failed: {}".format(_last_line(err) or "exit {}".format(rc)))
        return None
    try:
        with open(out_path, "r", encoding="utf-8") as fh:
            doc = json.load(fh)
    except (OSError, ValueError, json.JSONDecodeError) as e:
        warn("CycloneDX output unreadable: {}".format(e))
        return None
    return doc if isinstance(doc, dict) and doc.get("bomFormat") == "CycloneDX" else None


def scrub_local_paths(obj: Any, replacements: Dict[str, str], _depth: int = 0) -> Any:
    """Replace runner-local absolute paths (the target, $VIRTUAL_ENV, the private temp
    dir, --artifact/--sbom locations) inside every string of a report or SBOM, in
    place, longest path first. Nothing about the runner's filesystem leaves the machine."""
    pairs = sorted(((k, v) for k, v in replacements.items() if k and len(k) > 1), key=lambda kv: -len(kv[0]))

    def fix(text: str) -> str:
        for old, new in pairs:
            if old in text:
                text = text.replace(old, new)
        return text

    def walk(node: Any, depth: int) -> Any:
        if depth > 200:
            return node
        if isinstance(node, str):
            return fix(node)
        if isinstance(node, list):
            for i, v in enumerate(node):
                node[i] = walk(v, depth + 1)
            return node
        if isinstance(node, dict):
            for k in list(node.keys()):
                node[k] = walk(node[k], depth + 1)
            return node
        return node

    return walk(obj, _depth) if pairs else obj


def label_report(report: Optional[dict], label: str) -> Optional[dict]:
    """ArtifactName is the scanned path (absolute on the runner): replace it with the project label."""
    if isinstance(report, dict) and "ArtifactName" in report:
        report["ArtifactName"] = label
    return report


def label_cyclonedx(doc: Optional[dict], label: str) -> Optional[dict]:
    """`trivy convert` names metadata.component after the absolute scan path."""
    if isinstance(doc, dict):
        comp = (doc.get("metadata") or {}).get("component")
        if isinstance(comp, dict):
            comp["name"] = label
    return doc


def merge_cyclonedx(primary: Optional[dict], *others: Optional[dict]) -> Optional[dict]:
    """Union by purl; the SBOM is an inventory, so `vulnerabilities` is stripped."""
    docs = [d for d in (primary,) + others if isinstance(d, dict)]
    if not docs:
        return None
    base = json.loads(json.dumps(docs[0]))
    base.pop("vulnerabilities", None)
    comps = base.setdefault("components", [])
    seen = {purl_key(c_.get("purl")) or c_.get("bom-ref") for c_ in comps if isinstance(c_, dict)}
    deps = base.setdefault("dependencies", [])
    for other in docs[1:]:
        for comp in other.get("components") or []:
            k = purl_key(comp.get("purl")) or comp.get("bom-ref")
            if k and k not in seen:
                comps.append(comp)
                seen.add(k)
        refs = {d.get("ref") for d in deps if isinstance(d, dict)}
        for dep in other.get("dependencies") or []:
            if isinstance(dep, dict) and dep.get("ref") not in refs:
                deps.append(dep)
    return base


# ── Output documents ──────────────────────────────────────────────────────────


def result_document(*, product: str, exit_code: int, target: dict, gate: dict, findings: List[dict],
                    meta: dict, coverage: dict, ui_url: Optional[str], server: Optional[dict]) -> dict:
    return redact_obj({
        "schema": "vss-cli-result/1", "product": product, "exit_code": exit_code,
        "target": target, "gate": gate, "summary": severity_counts(findings),
        "findings": findings, "meta": meta, "coverage": coverage, "ui_url": ui_url,
        "server": server,
    })


def write_file(path: str, doc: dict) -> None:
    try:
        tmp = path + ".vsscli-tmp"
        with open(tmp, "w", encoding="utf-8") as fh:
            json.dump(doc, fh, indent=2)
        os.replace(tmp, path)
    except OSError as e:
        die("Cannot write {}: {}".format(path, e))


def _ui(auth: Optional[dict], body: dict, default_path: str) -> Optional[str]:
    if body.get("ui_url"):
        return str(body["ui_url"])
    link = ((body.get("links") or {}).get("ui")) or default_path
    if not auth or not link:
        return None
    return (auth.get("ui_url") or auth["server"]).rstrip("/") + link


def print_counts(title: str, counts: Dict[str, int]) -> None:
    print(bold(title))
    print("  {:<20} {}".format(red(bold("Critical")), counts.get("critical", 0)))
    print("  {:<20} {}".format(c("38;5;208", bold("High")), counts.get("high", 0)))
    print("  {:<20} {}".format(yellow(bold("Medium")), counts.get("medium", 0)))
    print("  {:<20} {}".format(green(bold("Low")), counts.get("low", 0)))
    print("  {:<20} {}".format(grey(bold("Unknown")), counts.get("unknown", 0)))


def print_delta(server: Optional[dict]) -> None:
    delta = (server or {}).get("delta") if isinstance((server or {}).get("delta"), dict) else None
    if not delta:
        return
    monitored = (server or {}).get("monitored")
    print(bold("Since last scan: ") + "{} new, {} reopened, {} resolved, {} suppressed by triage{}".format(
        delta.get("new", 0), delta.get("reopened", 0), delta.get("resolved", 0), delta.get("suppressed", 0),
        "" if monitored is not False else grey("  (branch upload: not monitored, default branch untouched)")))


def print_gate_line(fail: bool, gating: List[dict], threshold: str, fail_on: Optional[str]) -> None:
    desc = "severity >= {}".format(threshold) + (", fail-on {}".format(fail_on) if fail_on else "")
    if fail:
        print(red(bold("✗ {} issue(s) at the gate ({}).".format(len(gating), desc))))
    else:
        print(green("✓ Gate passed ({}; {} open finding(s) at/above threshold, none failing).".format(desc, len(gating))))


# ── Common command plumbing ───────────────────────────────────────────────────


class ScanContext:
    """Shared state of one scan command: auth, private dir, Trivy provenance, meta."""

    def __init__(self, args: argparse.Namespace, product: str) -> None:
        self.args = args
        self.product = product
        self.started = _dt.datetime.now(_dt.timezone.utc)
        self.no_upload = bool(getattr(args, "no_upload", False))
        try:
            self.threshold, self.fail_on = parse_gate_args(getattr(args, "severity_threshold", None),
                                                           getattr(args, "fail_on", None))
        except GateConfigError as e:
            die(str(e))
        self.fail_on_ignored: Optional[str] = None
        if product == "code" and self.fail_on in ("all", "upgradable"):
            # Code findings (secrets, SAST, misconfig) are never "fixable by upgrade", so
            # --fail-on all|upgradable would silently disable the gate — leaked private
            # keys and tokens would never fail the build. Teams reuse one flag set across
            # `oss` and `code`, so the flag is ignored here, loudly (Phase F).
            warn("--fail-on {} does not apply to `vsscli code` (code findings have no upgrade fix); "
                 "ignored — the gate uses --severity-threshold {} only. Use --fail-on none for a "
                 "report-only run.".format(self.fail_on, self.threshold))
            self.fail_on_ignored = self.fail_on
            self.fail_on = None
        self.auth = None if self.no_upload else resolve_auth(required=True)
        self.env_stripped: List[str] = []
        self.flags_dropped: List[str] = []
        self.engines: Dict[str, Optional[str]] = {"trivy": None, "semgrep": None, "gitleaks": None}
        self.argv: Dict[str, List[str]] = {}
        self.engines_ran: List[List[str]] = []
        self.coverage: Dict[str, Any] = {"warnings": []}
        if self.fail_on_ignored:
            self.coverage["warnings"].append("--fail-on {} ignored for code findings".format(self.fail_on_ignored))
        self.tinfo: Dict[str, Any] = {}
        self.db: Dict[str, Any] = {}
        self.trivy_sha256: Optional[str] = None
        #: C14 checks-bundle prefetch outcome: ok | failed | offline | not_needed.
        self.check_bundle = "not_needed"

    def warn(self, msg: str) -> None:
        warn(msg)
        self.coverage["warnings"].append(redact(msg))

    def prepare_trivy(self, pdir: PrivateDir, *, need_java: bool, docker: bool = False,
                      need_checks: bool = False) -> Tuple[str, Dict[str, str]]:
        trivy = ensure_trivy()
        env, stripped = scanner_env(docker=docker)
        self.env_stripped = sorted(set(self.env_stripped) | set(stripped))
        if stripped:
            self.warn("Ignoring scanner variables that could change results: {}".format(", ".join(stripped)))
        tinfo = trivy_info(trivy, env, pdir.path)
        check_trivy_version(tinfo.get("version"))
        offline = bool(getattr(self.args, "offline", False))
        if offline:
            self.check_bundle = "offline" if need_checks else "not_needed"
        else:
            refreshed = refresh_trivy_db(trivy, env, pdir.path, need_java=need_java, need_checks=need_checks)
            self.check_bundle = refreshed.get("check_bundle_prefetch") or "not_needed"
            tinfo = trivy_info(trivy, env, pdir.path)
        self.tinfo = tinfo
        self.db = db_state(tinfo, offline=offline, need_java=need_java)
        if getattr(self.args, "require_fresh_db", False) and not self.db["fresh"]:
            die("--require-fresh-db: the Trivy DB (UpdatedAt {}, NextUpdate {}) is older than {} h or past its "
                "NextUpdate. Run without --offline, or refresh the runner's DB cache.".format(
                    self.db.get("updated_at"), self.db.get("next_update"), FRESH_DB_MAX_AGE_HOURS))
        if self.db["stale"]:
            self.warn("Trivy DB is stale or offline (UpdatedAt {}); the server will not auto-resolve "
                      "findings from this upload.".format(self.db.get("updated_at")))
        self.engines["trivy"] = tinfo.get("version")
        self.trivy_sha256 = file_sha256(trivy)
        return trivy, env

    def meta(self, extra: Optional[dict] = None, git: Optional[dict] = None) -> dict:
        m = {
            "cli_version": VSSCLI_VERSION, "product": self.product, "scan_mode": "full",
            "engines": dict(self.engines), "trivy_binary_sha256": self.trivy_sha256,
            "trivy_db": {k: self.db.get(k) for k in ("updated_at", "next_update", "downloaded_at", "age_hours")},
            "java_db": {"updated_at": self.db.get("java_db_updated_at"), "next_update": self.db.get("java_db_next_update")},
            "check_bundle_digest": self.tinfo.get("check_bundle_digest"),
            # Every scan passes --skip-check-update (C14); the bundle is prefetched first.
            # Offline or a failed prefetch = misconfig ran on the cached/embedded checks.
            "check_bundle_prefetch": self.check_bundle,
            "check_bundle_stale": self.check_bundle in ("failed", "offline"),
            "offline": bool(getattr(self.args, "offline", False)), "db_stale": bool(self.db.get("stale")),
            "argv": self.argv, "engines_ran": self.engines_ran, "env_stripped": self.env_stripped,
            "flags_dropped": self.flags_dropped, "started_at": self.started.isoformat(),
            "finished_at": _dt.datetime.now(_dt.timezone.utc).isoformat(),
            "python": platform.python_version(), "os": "{} {}".format(platform.system(), platform.machine()),
            "ci": is_ci(),
        }
        if git is not None:
            m["git"] = {k: git.get(k) for k in ("git_remote", "branch", "default_branch", "git_commit", "project_path",
                                               "is_pull_request")}
        if extra:
            m.update(extra)
        return redact_obj(m)

    def upload(self, path: str, payload: dict, scope_hint: str) -> dict:
        assert self.auth is not None
        info("Uploading results to {}{} …".format(self.auth["server"], path))
        code, body, raw = http("POST", self.auth["server"] + path, api_key=self.auth["api_key"], body=payload,
                               timeout=int(getattr(self.args, "upload_timeout", 300) or 300))
        if code == 401:
            die("The server did not recognise the API key (HTTP 401). Run `vsscli config` to see which "
                "server this key belongs to, or issue a new key.")
        if code == 403:
            die("API key authenticated but is not authorised (HTTP 403): {}. It needs the '{}' scope, and the "
                "project identity must be in the key's allowed identities.".format(
                    _server_error_text(code, body, raw), scope_hint))
        if code in (502, 504):
            die("Upload failed — {}. A gateway gave up waiting; the server may still have stored the results "
                "(check the web UI before re-running). Not retried, to avoid a duplicate ingest.".format(
                    _server_error_text(code, body, raw)))
        if code >= 400:
            die("Upload failed — {}".format(_server_error_text(code, body, raw)))
        if not isinstance(body, dict):
            die("Unexpected server response: {}".format(redact(raw[:200])))
        ok("Results stored (scan run {}).".format(body.get("scan_run_id", "?")))
        return body

    def finish(self, *, findings_for_gate: List[dict], local_findings: List[dict], target: dict,
               meta: dict, body: Optional[dict], ui_path: str, human, sarif_runs: Optional[List[dict]] = None,
               sbom: Optional[dict] = None) -> int:
        args = self.args
        fail, gating = evaluate_gate(findings_for_gate, self.threshold, self.fail_on,
                                     include_dev=bool(getattr(args, "include_dev", False)),
                                     exclude_base_image_vulns=bool(getattr(args, "exclude_base_image_vulns", False)))
        exit_code = EXIT_ISSUES if fail else EXIT_OK
        gating_ids = {id(f) for f in gating}
        for f in findings_for_gate:
            f["gating"] = id(f) in gating_ids
        ui_url = _ui(self.auth, body or {}, ui_path) if body is not None else None
        gate = {"severity_threshold": self.threshold, "fail_on": self.fail_on,
                **({"fail_on_ignored": self.fail_on_ignored} if self.fail_on_ignored else {}),
                "include_dev": bool(getattr(args, "include_dev", False)),
                "exclude_base_image_vulns": bool(getattr(args, "exclude_base_image_vulns", False)),
                "gating_count": len(gating), "triage_aware": body is not None and bool((body or {}).get("findings") is not None)}
        server = None
        if body is not None:
            server = {k: body.get(k) for k in ("scan_run_id", "repo_id", "image_id", "asset_id", "delta",
                                               "summary_open", "monitored", "identity", "coverage") if k in body}
        doc = result_document(product=self.product, exit_code=exit_code, target=target, gate=gate,
                              findings=findings_for_gate, meta=meta, coverage=self.coverage, ui_url=ui_url,
                              server=server)
        if getattr(args, "json_file_output", None):
            write_file(args.json_file_output, doc)
            info("JSON written to {}".format(args.json_file_output))
        if getattr(args, "sarif_file_output", None):
            sarif = build_sarif(sarif_runs or [])
            problems = validate_sarif(sarif)
            if problems:
                die("Refusing to write invalid SARIF: {}".format("; ".join(sorted(set(problems))[:5])))
            write_file(args.sarif_file_output, sarif)
            info("SARIF written to {}".format(args.sarif_file_output))
        if getattr(args, "sbom_file_output", None):
            if sbom is None:
                die("--sbom-file-output requested but no CycloneDX SBOM could be produced.")
            write_file(args.sbom_file_output, redact_obj(sbom))
            info("CycloneDX SBOM written to {}".format(args.sbom_file_output))
        if getattr(args, "json", False):
            emit_stdout(json.dumps(doc, indent=2))
        else:
            human(doc)
            print_delta(server)
            print_gate_line(fail, gating, self.threshold, self.fail_on)
            if ui_url:
                print(grey("View: {}".format(ui_url)))
        if body is None:
            info("--no-upload: gated on local results; server triage (accepted risk, false positives) not applied.")
        return exit_code


def _add_common_scan_flags(sp: argparse.ArgumentParser, *, timeout: int) -> None:
    sp.add_argument("--json", action="store_true", help="Print exactly one JSON document to stdout (progress goes to stderr)")
    sp.add_argument("--json-file-output", metavar="PATH", help="Also write the JSON result document to PATH")
    sp.add_argument("--sarif-file-output", metavar="PATH", help="Write SARIF 2.1.0 for GitHub code scanning")
    sp.add_argument("--severity-threshold", choices=THRESHOLDS, default=None,
                    help="Fail only on issues at or above this severity (>=; 'unknown' gates as medium). Default low")
    sp.add_argument("--fail-on", default=None, metavar="{all,upgradable,none}",
                    help="all = fail only if a gating issue has a fix; upgradable = only if upgrading fixes it; "
                         "none = never fail. Default: fail on any gating issue. Unknown values exit 2")
    sp.add_argument("--no-upload", action="store_true", help="Scan and gate locally only (test mode; no triage)")
    sp.add_argument("--project-name", help="Project name when the folder has no git remote")
    sp.add_argument("--branch", help="Branch to record (default: CI env / git)")
    sp.add_argument("--commit", help="Commit SHA to record (default: git HEAD)")
    sp.add_argument("--offline", action="store_true",
                    help="No network: no Trivy DB refresh, no misconfig checks-bundle update (cached/embedded "
                         "checks), no Semgrep registry packs (code: only --semgrep-config local rules run). "
                         "Results are marked stale and never auto-resolve")
    sp.add_argument("--require-fresh-db", action="store_true",
                    help="Exit 2 unless the Trivy DB was built within {} h and is not past its NextUpdate".format(
                        FRESH_DB_MAX_AGE_HOURS))
    sp.add_argument("--timeout", type=int, default=timeout, help="Per-scanner timeout in seconds (default {})".format(timeout))
    sp.add_argument("--upload-timeout", type=int, default=300,
                    help="Seconds to wait for the server's response to the upload (default 300). A lost "
                         "response is never retried (the server may have stored it)")


# ── Commands: oss ─────────────────────────────────────────────────────────────


# ── Dependency resolution for manifests without a lockfile ───────────────────
#
# After the lockfile pass, each manifest Trivy could not read gets, in order:
#   1. its workspace lockfile from a parent folder (npm/pnpm/yarn, uv/poetry/pdm, Cargo): read only;
#   2. on a CI runner (or with --resolve): the ecosystem's own tool resolves it into a private dir:
#      npm --ignore-scripts, pip --dry-run, composer --no-scripts, bundle lock, cargo, dotnet,
#      gradle/maven (copies the resolved JARs) - the tools the pipeline's build runs anyway;
#   3. JVM only: the JARs built in a parent build root (Dockerfile in services/api, build at the root).
# Nothing here is fatal: a missing tool or a failed run is a warning, and a scan that resolves
# nothing still exits 3 naming the manifests. VSS_* never reaches these tools.

#: Workspace lockfiles looked up in parent folders, nearest folder first.
_WORKSPACE_LOCKFILES = {
    "package.json": ("package-lock.json", "npm-shrinkwrap.json", "pnpm-lock.yaml", "yarn.lock"),
    "pyproject.toml": ("uv.lock", "poetry.lock", "pdm.lock"),
    "Cargo.toml": ("Cargo.lock",),
}
#: Build files whose dependencies gradle/maven resolve (subprojects included).
JVM_BUILD_FILES = ("pom.xml", "build.gradle", "build.gradle.kts")
#: JVM build outputs; a Spring Boot / shaded JAR or a WAR carries every dependency inside it.
JVM_OUTPUT_GLOBS = ("build/libs/*.jar", "build/libs/*.war", "target/*.jar", "target/*.war")
#: Package-manager runs per scan (each resolves over the network); reported when hit.
RESOLVE_MAX_PROJECTS = 25


def _resolver_for(name: str) -> Optional[str]:
    if name == "package.json":
        return "npm"
    if name == "pyproject.toml" or (name.startswith("requirements") and name.endswith((".txt", ".in"))):
        return "pip"
    if name == "composer.json":
        return "composer"
    if name == "Gemfile":
        return "bundler"
    if name == "Cargo.toml":
        return "cargo"
    if name.endswith((".csproj", ".fsproj", ".vbproj")):
        return "dotnet"
    if name == "pom.xml":
        return "maven"
    if name in ("build.gradle", "build.gradle.kts"):
        return "gradle"
    return None


def _search_dirs(start: Path) -> List[Path]:
    """`start`, then each parent up to the git checkout root (never above it)."""
    cur = Path(os.path.realpath(str(start)))
    top = _git(cur, "rev-parse", "--show-toplevel")
    stop = Path(os.path.realpath(top)) if top else cur
    if stop != cur and stop not in cur.parents:
        stop = cur
    out = [cur]
    while cur != stop and cur.parent != cur:
        cur = cur.parent
        out.append(cur)
    return out


def _find_build_tool(dirs: List[Path], wrapper: str, tool: str) -> Optional[List[str]]:
    """The nearest wrapper (gradlew / mvnw: multi-module builds keep it at the build root),
    else the tool on PATH. A wrapper committed without the exec bit runs through sh."""
    for d in dirs:
        w = d / wrapper
        if w.is_file():
            return [str(w)] if os.access(str(w), os.X_OK) else ["sh", str(w)]
    found = which(tool)
    return [found] if found else None


def _jvm_outputs(d: Path) -> List[str]:
    return [str(f) for g in JVM_OUTPUT_GLOBS for f in sorted(d.glob(g)) if f.is_file()]


def build_root_outputs(start: Path) -> Tuple[Optional[Path], List[str]]:
    """JVM outputs for a folder whose build runs in a parent folder: the nearest parent, up to
    the git root, with build outputs next to a build file. None when `start` has its own
    outputs (the rootfs pass reads those already)."""
    dirs = _search_dirs(start)
    if _jvm_outputs(dirs[0]):
        return None, []
    marks = JVM_BUILD_FILES + ("gradlew", "mvnw", "settings.gradle", "settings.gradle.kts")
    for d in dirs[1:]:
        files = _jvm_outputs(d)
        if files and any((d / n).is_file() for n in marks):
            return d, files
    return None, []


def _workspace_lockfile(start: Path, name: str) -> Optional[Path]:
    for d in _search_dirs(start)[1:]:
        for lock in _WORKSPACE_LOCKFILES.get(name, ()):
            if (d / lock).is_file():
                return d / lock
    return None


def _pyproject_requirements(path: Path) -> Optional[List[str]]:
    """PEP 621 [project].dependencies; None when unreadable (Python < 3.11 has no tomllib)."""
    try:
        import tomllib  # type: ignore[import-not-found]  # Python 3.11+
    except ImportError:
        return None
    try:
        with open(str(path), "rb") as fh:
            data = tomllib.load(fh)
    except (OSError, ValueError):
        return None
    deps = (data.get("project") or {}).get("dependencies")
    return [str(d) for d in deps] if isinstance(deps, list) else None


def _pins_from_pip_report(report_path: str) -> List[str]:
    with open(report_path, encoding="utf-8") as fh:
        report = json.load(fh)
    pins = set()
    for item in report.get("install") or []:
        md = item.get("metadata") or {}
        if md.get("name") and md.get("version"):
            pins.add("{}=={}".format(md["name"], md["version"]))
    return sorted(pins, key=str.lower)


def _copy_if_present(src_dir: Path, dest: str, names: Sequence[str]) -> None:
    for n in names:
        if (src_dir / n).is_file():
            shutil.copy2(str(src_dir / n), os.path.join(dest, n))


_GRADLE_INIT = (
    "allprojects {\n"
    "  tasks.register('vssCopyDependencies') {\n"
    "    doLast {\n"
    "      def names = System.getenv('VSS_INCLUDE_DEV') == '1' ? ['runtimeClasspath', 'testRuntimeClasspath'] : ['runtimeClasspath']\n"
    "      def out = new File(System.getenv('VSS_DEPS_DIR'), project.path == ':' ? 'root' : project.path.replace(':', '_'))\n"
    "      names.each { n ->\n"
    "        def cfg = project.configurations.findByName(n)\n"
    "        if (cfg != null && cfg.canBeResolved) { project.copy { from cfg; into out } }\n"
    "      }\n"
    "    }\n"
    "  }\n"
    "}\n")


def resolve_dependencies(target: Path, unresolved: List[dict], pdir: PrivateDir, *, run_tools: bool,
                         include_dev: bool, timeout: int, note=warn) -> dict:
    """Resolve the manifests the lockfile pass could not read (see the section comment).

    Returns {"lock_root": dir of generated/copied lockfiles mirrored at their manifest's
    path (for one more `trivy fs`) or None, "rootfs_roots": [(label, path)] (resolved and
    built JARs), "records": [...] (coverage), "resolved_by": {manifest path: how}}."""
    lock_root = os.path.join(pdir.path, "resolved")
    res: Dict[str, Any] = {"lock_root": None, "rootfs_roots": [], "records": [], "resolved_by": {}}
    records: List[dict] = res["records"]
    env = {k: v for k, v in os.environ.items() if not k.startswith("VSS_")}
    covered: Dict[Path, str] = {}       # dir -> "gradle"/"maven": its run covered every subproject
    seen_outputs: set = set()
    runs = capped = 0

    def _dest(rel_dir: str) -> str:
        d = os.path.join(lock_root, rel_dir)
        os.makedirs(d, exist_ok=True)
        res["lock_root"] = lock_root
        return d

    def _run(manifest: str, tool: str, argv: List[str], cwd: str, run_env: Dict[str, str]) -> bool:
        info("Resolving {} with {} (no lockfile) …".format(manifest, tool))
        rc, _, err = run_capture(argv, timeout=timeout, cwd=cwd, env=run_env, fatal=False)
        records.append({"manifest": manifest, "tool": tool, "status": "ok" if rc == 0 else "failed",
                        "argv": sanitize_argv(argv, {str(target): "<target>", pdir.path: "<tmp>"})})
        if rc != 0:
            note("{}: {} could not resolve its dependencies: {}".format(
                manifest, tool, _last_line(err) or "exit {}".format(rc)))
        return rc == 0

    def _skip(manifest: str, tool: str, why: str) -> None:
        records.append({"manifest": manifest, "tool": tool, "status": "skipped", "reason": why})
        note("{}: dependencies not resolved — {}".format(manifest, why))

    def _built_jars(manifest: str, d: Path) -> None:
        root, files = build_root_outputs(d)
        if root is None:
            return
        res["resolved_by"][manifest] = "JARs built in {}".format(os.path.relpath(str(root), str(target)))
        new = [f for f in files if f not in seen_outputs]
        if new:
            seen_outputs.update(new)
            info("Scanning the JAR(s) built in {}: {}".format(
                os.path.relpath(str(root), str(target)), ", ".join(os.path.basename(f) for f in new)))
            res["rootfs_roots"] += [("build-root:" + os.path.relpath(f, str(root)).replace(os.sep, "/"), f)
                                    for f in new]
            records.append({"manifest": manifest, "tool": "build-outputs", "status": "ok",
                            "paths": [os.path.relpath(f, str(root)).replace(os.sep, "/") for f in new]})

    for m in sorted(unresolved, key=lambda x: (x["path"].count("/"), x["path"])):
        rel = m["path"]
        name = os.path.basename(rel)
        d = (target / rel).parent
        rel_dir = os.path.dirname(rel)
        tool = _resolver_for(name)
        # 1) a workspace lockfile in a parent folder (read only)
        lock = _workspace_lockfile(d, name)
        if lock is not None:
            dest = _dest(rel_dir)
            shutil.copy2(str(lock), os.path.join(dest, lock.name))
            where = os.path.relpath(str(lock), str(target)).replace(os.sep, "/")
            info("{}: using the workspace lockfile {}".format(rel, where))
            res["resolved_by"][rel] = "workspace lockfile " + where
            records.append({"manifest": rel, "tool": "workspace-lockfile", "status": "ok", "path": where})
            continue
        if tool in ("gradle", "maven"):
            owner = next((c for c in [d] + list(d.parents) if covered.get(c) == tool), None)
            if owner is not None:
                res["resolved_by"][rel] = "{} run in {}".format(tool, os.path.relpath(str(owner), str(target)))
                continue
        if tool is None or not run_tools:
            if tool in ("gradle", "maven"):
                _built_jars(rel, d)
            continue
        if runs >= RESOLVE_MAX_PROJECTS:
            capped += 1
            if tool in ("gradle", "maven"):
                _built_jars(rel, d)
            continue
        runs += 1
        ok_ = False
        dirs = _search_dirs(d)
        if tool == "npm":
            npm = which("npm")
            if not npm:
                _skip(rel, "npm", "npm is not on PATH")
            else:
                dest = _dest(rel_dir)
                shutil.copy2(str(target / rel), os.path.join(dest, "package.json"))
                _copy_if_present(d, dest, (".npmrc",))
                ok_ = _run(rel, "npm", [npm, "install", "--package-lock-only", "--ignore-scripts", "--no-audit",
                                        "--no-fund", "--loglevel=error"], dest, env) \
                    and os.path.isfile(os.path.join(dest, "package-lock.json"))
        elif tool == "pip":
            py = which("python3") or which("python")
            reqs: Optional[str] = str(target / rel)
            if name == "pyproject.toml":
                deps = _pyproject_requirements(target / rel)
                if deps is None:
                    reqs = None
                    _skip(rel, "pip", "no [project].dependencies readable (Poetry/PDM projects: commit the "
                                      "lockfile; reading pyproject.toml needs Python 3.11+)")
                else:
                    reqs = pdir.file("pyproject-{}.in".format(runs))
                    with open(reqs, "w", encoding="utf-8") as fh:
                        fh.write("\n".join(deps) + "\n")
            if reqs and not py:
                _skip(rel, "pip", "python3 is not on PATH")
            elif reqs:
                report = pdir.file("pip-report-{}.json".format(runs))
                if _run(rel, "pip", [py, "-m", "pip", "install", "--dry-run", "--ignore-installed", "--quiet",
                                     "--no-input", "--disable-pip-version-check", "--report", report, "-r", reqs],
                        str(d), env):
                    pins = _pins_from_pip_report(report)
                    if pins:
                        with open(os.path.join(_dest(rel_dir), "requirements.txt"), "w", encoding="utf-8") as fh:
                            fh.write("\n".join(pins) + "\n")
                        ok_ = True
        elif tool == "composer":
            composer = which("composer")
            if not composer:
                _skip(rel, "composer", "composer is not on PATH")
            else:
                dest = _dest(rel_dir)
                shutil.copy2(str(target / rel), os.path.join(dest, "composer.json"))
                ok_ = _run(rel, "composer", [composer, "update", "--no-install", "--no-scripts", "--no-plugins",
                                             "--no-interaction", "--no-progress", "--ignore-platform-reqs",
                                             "--working-dir=" + dest], dest, env) \
                    and os.path.isfile(os.path.join(dest, "composer.lock"))
        elif tool == "bundler":
            bundle = which("bundle")
            if not bundle:
                _skip(rel, "bundler", "bundle is not on PATH")
            else:
                dest = _dest(rel_dir)
                shutil.copy2(str(target / rel), os.path.join(dest, "Gemfile"))
                _copy_if_present(d, dest, [p.name for p in d.glob("*.gemspec")] + [".ruby-version"])
                ok_ = _run(rel, "bundler", [bundle, "lock"], dest,
                           dict(env, BUNDLE_GEMFILE=os.path.join(dest, "Gemfile"),
                                BUNDLE_APP_CONFIG=os.path.join(dest, ".bundle"))) \
                    and os.path.isfile(os.path.join(dest, "Gemfile.lock"))
        elif tool == "cargo":
            cargo = which("cargo")
            if not cargo:
                _skip(rel, "cargo", "cargo is not on PATH")
            else:
                # cargo writes Cargo.lock beside the (workspace) manifest: move it out, never leave it behind.
                before = {p for p in (x / "Cargo.lock" for x in dirs) if p.is_file()}
                try:
                    if _run(rel, "cargo", [cargo, "generate-lockfile", "--manifest-path", str(target / rel)],
                            str(d), env):
                        made = next((p for p in (x / "Cargo.lock" for x in dirs) if p.is_file() and p not in before),
                                    None)
                        if made is not None:
                            shutil.copy2(str(made), os.path.join(_dest(rel_dir), "Cargo.lock"))
                            ok_ = True
                finally:
                    for p in (x / "Cargo.lock" for x in dirs):
                        if p.is_file() and p not in before:
                            p.unlink()
        elif tool == "dotnet":
            dotnet = which("dotnet")
            if not dotnet:
                _skip(rel, "dotnet", "dotnet is not on PATH")
            else:
                lockfile = os.path.join(_dest(rel_dir), "packages.lock.json")
                ok_ = _run(rel, "dotnet", [dotnet, "restore", str(target / rel), "--use-lock-file",
                                           "--lock-file-path", lockfile, "--nologo", "--verbosity", "quiet"],
                           str(d), env) and os.path.isfile(lockfile)
        elif tool == "maven":
            mvn = _find_build_tool(dirs, "mvnw", "mvn")
            if not mvn:
                _skip(rel, "maven", "no mvnw (here or in a parent folder) and no mvn on PATH")
            else:
                out_dir = os.path.join(pdir.path, "resolved-maven-{}".format(runs))
                ok_ = _run(rel, "maven", mvn + ["-B", "-q", "-f", str(target / rel), "dependency:copy-dependencies",
                                                "-DoutputDirectory=" + out_dir,
                                                "-DincludeScope=" + ("test" if include_dev else "runtime")],
                           str(d), env)
                if ok_:
                    res["rootfs_roots"].append(("resolved:maven:" + (rel_dir or "."), out_dir))
        elif tool == "gradle":
            gw = _find_build_tool(dirs, "gradlew", "gradle")
            if not gw:
                _skip(rel, "gradle", "no gradlew (here or in a parent folder) and no gradle on PATH")
            else:
                out_dir = os.path.join(pdir.path, "resolved-gradle-{}".format(runs))
                init = pdir.file("vss-init.gradle")
                with open(init, "w", encoding="utf-8") as fh:
                    fh.write(_GRADLE_INIT)
                ok_ = _run(rel, "gradle", gw + ["-q", "--no-daemon", "--no-configuration-cache", "--init-script",
                                                init, "vssCopyDependencies"], str(d),
                           dict(env, VSS_DEPS_DIR=out_dir, VSS_INCLUDE_DEV="1" if include_dev else "0"))
                if ok_:
                    res["rootfs_roots"].append(("resolved:gradle:" + (rel_dir or "."), out_dir))
        if ok_:
            res["resolved_by"][rel] = tool
            if tool in ("gradle", "maven"):
                covered[d] = tool
        elif tool in ("gradle", "maven"):
            _built_jars(rel, d)
    if capped:
        note("Dependency resolution stopped after {} project(s) (RESOLVE_MAX_PROJECTS); {} more manifest(s) "
             "left unresolved.".format(RESOLVE_MAX_PROJECTS, capped))
        records.append({"tool": "cap", "status": "capped", "limit": RESOLVE_MAX_PROJECTS, "skipped": capped})
    return res


def _load_user_sbom(path: str) -> str:
    p = os.path.abspath(path)
    if not os.path.isfile(p):
        die("--sbom {} does not exist.".format(path))
    if os.path.getsize(p) > 128 * 1024 * 1024:
        die("--sbom {} is larger than 128 MiB.".format(path))
    try:
        with open(p, "r", encoding="utf-8") as fh:
            doc = json.load(fh)
    except (OSError, ValueError, json.JSONDecodeError) as e:
        die("--sbom {} is not valid JSON: {}".format(path, e))
    if not isinstance(doc, dict) or not (doc.get("bomFormat") == "CycloneDX" or doc.get("spdxVersion")):
        die("--sbom {} is neither CycloneDX nor SPDX JSON.".format(path))
    return p


def _prefix_rootfs_paths(report: dict, label: str) -> dict:
    for r in report.get("Results") or []:
        r["Target"] = "{}/{}".format(label, strip_dot_slash(r.get("Target")))
        for p in r.get("Packages") or []:
            if isinstance(p, dict) and p.get("FilePath"):
                p["FilePath"] = "{}/{}".format(label, strip_dot_slash(p["FilePath"]))
        for v in r.get("Vulnerabilities") or []:
            if isinstance(v, dict) and v.get("PkgPath"):
                v["PkgPath"] = "{}/{}".format(label, strip_dot_slash(v["PkgPath"]))
    return report


def cmd_oss(args: argparse.Namespace) -> int:
    """Open-source dependencies — lockfiles (trivy fs) ∪ installed artefacts (trivy rootfs)."""
    ctx = ScanContext(args, "oss")
    if getattr(args, "prod_only", False):
        warn("--prod-only is the default now (dev dependencies are excluded unless --include-dev).")
    target = Path(args.path).expanduser().resolve()
    if not target.is_dir():
        die("Not a directory: {}".format(target))
    # Validate --sbom BEFORE any Trivy work: a bad path must not surface only after
    # the DB refresh, the fs pass and every rootfs pass have run.
    user_sbom = _load_user_sbom(args.sbom) if args.sbom else None
    git = git_context(target, args.branch, args.commit)
    project_name = (args.project_name or getattr(args, "name", None) or "").strip() or None
    with PrivateDir() as pdir:
        trivy, env = ctx.prepare_trivy(pdir, need_java=True)
        offline = bool(args.offline)
        detected = detect_projects(target)
        if detected["truncated"]:
            ctx.warn("Project detection stopped after {} files (DETECT_MAX_FILES); manifest list may be "
                     "incomplete.".format(DETECT_MAX_FILES))
        repl = {str(target): "<target>", pdir.path: "<tmp>"}
        t0 = time.time()
        fs_out = pdir.file("fs.json")
        argv = build_fs_argv(trivy, str(target), fs_out, timeout=args.timeout, include_dev=bool(args.include_dev),
                             offline=offline)
        ctx.argv["trivy_fs"] = sanitize_argv(argv, repl)
        fs_report = run_trivy_json(argv, fs_out, timeout=args.timeout, cwd=pdir.path, env=env, label="fs (lockfiles)")
        ctx.engines_ran.append(["trivy", "vuln:fs"])
        manifests = classify_manifests(detected, fs_report)
        # Like `snyk test`: on a CI runner the package managers run automatically (the pipeline
        # builds this code anyway); on a workstation only with --resolve (they run build scripts).
        run_tools = not args.no_resolve and not offline and (bool(args.resolve) or is_ci())
        res = resolve_dependencies(target, [m for m in manifests if m["status"] == "unresolved"], pdir,
                                   run_tools=run_tools, include_dev=bool(args.include_dev), timeout=int(args.timeout),
                                   note=ctx.warn)
        resolve_records = res["records"]
        extra_roots = res["rootfs_roots"]
        if res["lock_root"]:
            out = pdir.file("fs-resolved.json")
            argv = build_fs_argv(trivy, res["lock_root"], out, timeout=args.timeout,
                                 include_dev=bool(args.include_dev), offline=offline)
            ctx.argv["trivy_fs_resolved"] = sanitize_argv(argv, repl)
            rep = _prefix_rootfs_paths(run_trivy_json(argv, out, timeout=args.timeout, cwd=pdir.path, env=env,
                                                      label="fs (resolved lockfiles)"), "resolved")
            fs_report.setdefault("Results", []).extend(rep.get("Results") or [])
        for m in manifests:
            if m["status"] == "unresolved" and m["path"] in res["resolved_by"]:
                m.update(status="resolved", hint=None, resolved_by=res["resolved_by"][m["path"]])
        rootfs_report: Optional[dict] = None
        rootfs_paths: List[str] = []
        if not args.lockfile_only:
            roots: List[Tuple[str, str]] = [("", str(target))]
            venv = os.environ.get("VIRTUAL_ENV")
            if venv and os.path.isdir(venv) and not os.path.realpath(venv).startswith(str(target) + os.sep):
                roots.append(("$VIRTUAL_ENV", venv))
            for i, art in enumerate(args.artifact or []):
                ap = os.path.abspath(art)
                if not os.path.exists(ap):
                    die("--artifact {} does not exist.".format(art))
                roots.append(("artifact{}:{}".format(i, os.path.basename(ap.rstrip(os.sep))), ap))
            roots.extend(extra_roots)
            merged_results: List[dict] = []
            for n, (label, root) in enumerate(roots):
                out = pdir.file("rootfs{}.json".format(n))
                argv = build_rootfs_argv(trivy, root, out, timeout=args.timeout, offline=offline)
                ctx.argv["trivy_rootfs" + ("" if n == 0 else str(n))] = sanitize_argv(argv, dict(repl, **{root: "<{}>".format(label or "target")}))
                rep = run_trivy_json(argv, out, timeout=args.timeout, cwd=pdir.path, env=env,
                                     label="rootfs (installed{})".format(": " + label if label else ""))
                if label:
                    rep = _prefix_rootfs_paths(rep, label)
                merged_results.extend(rep.get("Results") or [])
                rootfs_paths.append(label or "<target>")
            rootfs_report = {"SchemaVersion": 2, "ArtifactType": "filesystem", "ArtifactName": target.name,
                             "Results": merged_results}
            ctx.engines_ran.append(["trivy", "vuln:rootfs"])
        sbom_report: Optional[dict] = None
        if user_sbom:
            sbom_path = user_sbom
            out = pdir.file("sbom.json")
            argv = build_sbom_argv(trivy, sbom_path, out, timeout=args.timeout, offline=offline)
            ctx.argv["trivy_sbom"] = sanitize_argv(argv, dict(repl, **{sbom_path: "<sbom>"}))
            sbom_report = run_trivy_json(argv, out, timeout=args.timeout, cwd=pdir.path, env=env,
                                         label="sbom (build-tool SBOM)")
            ctx.engines_ran.append(["trivy", "vuln:sbom"])
        ok("Dependency passes finished in {:.1f}s.".format(time.time() - t0))

        resolved_before_dev = merge_oss(fs_report, rootfs_report, sbom_report)["packages"]
        dev_stats = {"packages": 0, "findings": 0, "dev_package_keys": 0, "binaries": 0}
        if not args.include_dev:
            dev_keys, dev_coords = dev_package_sets(fs_report)
            (fs_report, rootfs_report, sbom_report), dev_stats = exclude_dev_packages(
                [fs_report, rootfs_report, sbom_report], dev_keys, dev_coords, dev_npm_names(fs_report))
            if dev_stats["packages"]:
                info("{} dev dependenc{} excluded ({} advisor{}); pass --include-dev to scan and gate on them.".format(
                    dev_stats["packages"], "y" if dev_stats["packages"] == 1 else "ies",
                    dev_stats["findings"], "y" if dev_stats["findings"] == 1 else "ies"))
        merged = merge_oss(fs_report, rootfs_report, sbom_report)
        unresolved = [m for m in manifests if m["status"] == "unresolved"]
        ctx.coverage.update({
            "unresolved_manifests": [{"path": m["path"], "ecosystem": m["ecosystem"], "hint": m["hint"]} for m in unresolved],
            "rootfs_dropped_project_manifests": merged["stats"]["rootfs_dropped_project_manifests"],
            "rootfs_roots": rootfs_paths, "resolve": resolve_records,
            "include_dev": bool(args.include_dev), "lockfile_only": bool(args.lockfile_only),
            "dev_excluded": dev_stats,
        })
        for m in unresolved:
            ctx.warn("Unresolved manifest {} ({}): {}".format(m["path"], m["ecosystem"], m["hint"]))
        packages = merged["packages"]
        target_info = {"project": git.get("git_remote") or project_name or target.name,
                       "project_path": git.get("project_path"), "branch": git.get("branch"),
                       "commit": git.get("git_commit"), "image_ref": None, "platform": None}
        if not packages and resolved_before_dev:
            info("All {} resolved packages are dev dependencies; nothing to gate without --include-dev.".format(
                len(resolved_before_dev)))
        if not resolved_before_dev:
            msg = "Nothing scannable: no package was resolved by any pass."
            if unresolved:
                msg += " Detected but unresolved: " + ", ".join(m["path"] for m in unresolved[:20])
            else:
                msg += " No supported manifest or lockfile was found."
            if args.json:
                emit_stdout(json.dumps(result_document(
                    product="oss", exit_code=EXIT_NOTHING, target=target_info,
                    gate={"severity_threshold": ctx.threshold, "fail_on": ctx.fail_on}, findings=[],
                    meta=ctx.meta(git=git), coverage=ctx.coverage, ui_url=None, server=None), indent=2))
            die(msg, EXIT_NOTHING)

        # Nothing about the runner's filesystem leaves the machine: ArtifactName and
        # CycloneDX metadata.component.name carry the project label, and any absolute
        # runner path left in a report or the SBOM is replaced by a placeholder.
        label = str(target_info["project"]) + ("/" + target_info["project_path"] if target_info.get("project_path") else "")
        path_repl = {str(target): "<target>", pdir.path: "<tmp>", os.path.realpath(pdir.path): "<tmp>"}
        venv_env = os.environ.get("VIRTUAL_ENV")
        if venv_env:
            path_repl[os.path.abspath(venv_env)] = "$VIRTUAL_ENV"
        for i, art in enumerate(args.artifact or []):
            path_repl[os.path.abspath(art)] = "<artifact{}>".format(i)
        if args.sbom:
            path_repl[os.path.abspath(args.sbom)] = "<sbom>/" + os.path.basename(args.sbom)
        for rep_ in (fs_report, rootfs_report, sbom_report):
            label_report(rep_, label)
            scrub_local_paths(rep_, path_repl)
        cdx = None
        if not args.no_sbom:
            # Converted from the IN-MEMORY reports (dev dependencies already excluded
            # unless --include-dev), and the build-tool SBOM pass is part of the inventory.
            cdx_parts = []
            for name, rep_ in (("fs", fs_report), ("rootfs", rootfs_report), ("sbom", sbom_report)):
                if rep_ is None:
                    continue
                rp = pdir.file("{}-filtered.json".format(name))
                with open(rp, "w", encoding="utf-8") as fh:
                    json.dump(rep_, fh)
                cdx_parts.append(trivy_convert_cyclonedx(trivy, rp, pdir.file("{}.cdx.json".format(name)),
                                                         env=env, cwd=pdir.path))
            cdx = merge_cyclonedx(*cdx_parts) if cdx_parts else None
            label_cyclonedx(cdx, label)
            scrub_local_paths(cdx, path_repl)

        meta = ctx.meta(extra={
            "include_dev": bool(args.include_dev),
            "dev_dependencies": {"listed_by_fs": True, "excluded": not args.include_dev,
                                 "packages_excluded": dev_stats["packages"],
                                 "findings_excluded": dev_stats["findings"]},
            "passes": {"fs": {"packages": merged["stats"]["fs_packages"], "include_dev_deps": True},
                       "rootfs": ({"packages": merged["stats"]["rootfs_packages"], "roots": rootfs_paths}
                                  if rootfs_report is not None else {"status": "not_run"}),
                       "sbom": ({"packages": merged["stats"]["sbom_packages"]} if sbom_report is not None
                                else {"status": "not_run"})},
            "languages": detected["languages"], "coverage": ctx.coverage,
        }, git=git)
        body: Optional[dict] = None
        findings_for_gate: List[dict] = merged["findings"]
        if not ctx.no_upload:
            payload = {
                "folder_name": target.name[:200], "project_name": project_name,
                "git_remote": git.get("git_remote"), "project_path": git.get("project_path") or "",
                "branch": git.get("branch"), "default_branch": git.get("default_branch"),
                "is_pull_request": bool(git.get("is_pull_request")),
                "git_commit": git.get("git_commit"), "git_branch": git.get("branch"),
                "include_dev": bool(args.include_dev),
                "trivy_output": fs_report, "trivy_rootfs": rootfs_report, "trivy_sbom": sbom_report,
                "cyclonedx": cdx, "manifests": [{k: m[k] for k in ("path", "ecosystem", "kind", "status")} for m in manifests],
                "languages": detected["languages"], "meta": meta,
                "client_hostname": socket.gethostname()[:255], "cli_version": VSSCLI_VERSION,
            }
            body = ctx.upload("/api/oss/scan-result", payload, "ingest:oss")
            if isinstance(body.get("findings"), list):
                findings_for_gate = _join_server_findings(body["findings"], merged["findings"])
            else:
                ctx.warn("The server returned no triage-aware findings (older server); gating on local results.")

    def human(doc: dict) -> None:
        print()
        print(bold("Project: ") + str(target_info["project"]) + ("/" + target_info["project_path"] if target_info.get("project_path") else ""))
        if target_info.get("branch"):
            print(bold("Branch:  ") + str(target_info["branch"]) + (" @ " + str(target_info["commit"])[:12] if target_info.get("commit") else ""))
        direct = sum(1 for p in packages if p["relationship"] == "direct")
        print(bold("Deps:    ") + "{} resolved ({} direct, {} transitive/unknown)".format(len(packages), direct, len(packages) - direct))
        print(bold("Vulns:   ") + "{} ({} with a fix available)".format(len(doc["findings"]), sum(1 for f in doc["findings"] if f.get("fix_available"))))
        print()
        print_counts("By severity", doc["summary"])
        shown = [f for f in doc["findings"] if f.get("gating")][:15]
        if shown:
            print()
            print(bold("Gating issues"))
            for f in shown:
                path = " > ".join((f.get("introduced_through") or [])[-4:])
                print("  {:<9} {:<22} {}@{}  fix: {}{}".format(
                    f.get("severity"), f.get("vuln_id"), f.get("package"), f.get("version"),
                    f.get("fixed_version_for_installed") or f.get("fixed_version") or "none",
                    ("  via " + path) if path and f.get("relationship") != "direct" else ""))
        if unresolved:
            print()
            print(yellow("{} manifest(s) could not be resolved — see warnings above.".format(len(unresolved))))
        print()

    sarif_runs = [{"category": "vss/oss/", "tool": "vsscli-oss", "results": [
        {"rule_id": f["vuln_id"], "severity": f["severity"], "title": f.get("title"),
         "short": "{} in {}".format(f["vuln_id"], f.get("package")),
         "full": f.get("title"), "help": "Upgrade {} to {}".format(
             f.get("package"), f.get("fixed_version_for_installed") or f.get("fixed_version") or "a fixed version"),
         "message": "{} {}@{}: {}. Fixed in: {}".format(f["vuln_id"], f.get("package"), f.get("version"),
                                                        f.get("title"), f.get("fixed_version") or "no fix yet"),
         "uri": f.get("file") or (f.get("pkg_paths") or ["dependencies"])[0], "start_line": f.get("start_line"),
         "cvss_score": f.get("cvss_score"), "cwe": f.get("cwe"), "category": "oss",
         "fingerprint": sha256_hex("oss|{}|{}|{}".format(f["vuln_id"], f.get("package_key"), f.get("version")))}
        for f in merged["findings"]]}]
    return ctx.finish(findings_for_gate=findings_for_gate, local_findings=merged["findings"], target=target_info,
                      meta=meta, body=body, ui_path="/projects", human=human, sarif_runs=sarif_runs, sbom=cdx)


def _join_server_findings(server: List[Any], local: List[dict]) -> List[dict]:
    """Server findings carry triage status; enrich them with the local package facts."""
    by_key: Dict[Tuple[str, str, str], dict] = {}
    for f in local:
        by_key[(str(f.get("vuln_id")), str(f.get("package")), str(f.get("version")))] = f
    out = []
    for s in server:
        if not isinstance(s, dict):
            continue
        lf = by_key.get((str(s.get("vuln_id")), str(s.get("package")), str(s.get("version"))), {})
        merged = dict(lf)
        merged.update({k: v for k, v in s.items() if v is not None or k in ("suppressed",)})
        if merged.get("scope") in (None, "") and lf.get("scope"):
            merged["scope"] = lf["scope"]
        out.append(merged)
    return out


# ── Commands: code ────────────────────────────────────────────────────────────


def _has_any_file(target: Path) -> bool:
    for _dirpath, dirnames, filenames in os.walk(str(target)):
        dirnames[:] = [d for d in dirnames if d not in CODE_SKIP_DIRS]
        if filenames:
            return True
    return False


_CONFIG_WORD_RE = re.compile(r"^[a-z0-9+#._-]{1,40}$")


def code_config_query(detected: dict, quality: bool = True) -> str:
    """`GET /api/code/config` query (WP8 #1): the server returns its own semgrep_configs
    (select_configs) only when told what the checkout contains. Comma lists, words
    validated like the server's (^[a-z0-9+#._-]{1,40}$, <= 64 each)."""
    def words(values: Iterable[Any]) -> str:
        out = sorted({str(v).lower() for v in values if _CONFIG_WORD_RE.match(str(v).lower())})
        return ",".join(out[:64])

    params = [("languages", words((detected.get("languages") or {}).keys())),
              ("frameworks", words(detected.get("frameworks") or [])),
              ("iac", words(detected.get("iac") or [])),
              ("quality", "true" if quality else "false")]
    return urllib.parse.urlencode(params)


def _server_semgrep_configs(auth: Optional[dict], detected: Optional[dict] = None,
                            quality: bool = True) -> Optional[List[str]]:
    """The server's pack list for this checkout (CLI/server parity); None -> local selection."""
    if not auth or detected is None:
        return None
    url = auth["server"] + "/api/code/config?" + code_config_query(detected, quality)
    code, body, _ = http("GET", url, api_key=auth["api_key"], timeout=30, retries=1)
    if code != 200 or not isinstance(body, dict):
        return None
    raw = body.get("semgrep_configs")
    if raw is None and isinstance(body.get("semgrep"), dict):
        raw = body["semgrep"].get("configs")
    if not isinstance(raw, list):
        return None
    out = [str(x) for x in raw if isinstance(x, str) and _SEMGREP_CONFIG_RE.match(str(x)) and str(x) not in SEMGREP_FORBIDDEN_PACKS]
    return out or None


def cmd_code(args: argparse.Namespace) -> int:
    """Source code: Semgrep (SAST + quality) + Gitleaks (else Trivy secret) + Trivy misconfig."""
    ctx = ScanContext(args, "code")
    target = Path(args.path).expanduser().resolve()
    if not target.is_dir():
        die("Not a directory: {}".format(target))
    git = git_context(target, args.branch, args.commit)
    project_name = (args.project_name or getattr(args, "name", None) or "").strip() or None
    semgrep = which("semgrep") if args.sast else None
    if args.require_sast and not semgrep:
        die("--require-sast: semgrep is not installed (pipx install semgrep==<server pin>). SAST cannot run.")
    local_rules = [c_ for c_ in (args.semgrep_config or []) if not _is_registry_pack(c_)]
    if semgrep and args.offline and not local_rules:
        # Registry packs (p/..., r/...) are fetched from semgrep.dev at scan time: --offline
        # means no network, so SAST cannot run without local rule files.
        if args.require_sast:
            die("--require-sast with --offline: Semgrep registry packs need semgrep.dev. Pass "
                "--semgrep-config <local rules file or dir> to run SAST offline.")
        ctx.warn("--offline: Semgrep is SKIPPED (its registry packs are downloaded from semgrep.dev). "
                 "Pass --semgrep-config <local rules> to run SAST offline. This run has NO SAST.")
        ctx.coverage["semgrep_skipped"] = "offline: registry packs need semgrep.dev and no local rules were given"
        semgrep = None
    if args.sast and not semgrep:
        ctx.warn("semgrep is NOT installed: this run has NO SAST and NO quality rules (secrets and IaC only). "
                 "Install semgrep, or pass --require-sast to make this an error.")
    gitleaks = which("gitleaks") if args.secrets_engine in ("auto", "gitleaks") else None
    if args.secrets_engine == "gitleaks" and not gitleaks:
        die("--secrets-engine gitleaks: gitleaks is not installed.")
    if not _has_any_file(target):
        msg = "Nothing scannable: {} contains no files outside excluded directories.".format(target)
        if args.json:
            emit_stdout(json.dumps(result_document(
                product="code", exit_code=EXIT_NOTHING, target={"project": git.get("git_remote") or project_name or target.name},
                gate={"severity_threshold": ctx.threshold, "fail_on": ctx.fail_on}, findings=[],
                meta=ctx.meta(git=git), coverage=dict(ctx.coverage, nothing_scannable=msg), ui_url=None,
                server=None), indent=2))
        die(msg, EXIT_NOTHING)
    detected = detect_projects(target)
    misconfig_requested = "misconfig" in [x.strip() for x in (args.scanners or "misconfig").split(",")]
    tf = detected.get("terraform_files") or {}
    if tf.get("count") and misconfig_requested:
        # INT-12 / WP8-25: Trivy's Terraform scanner resolves module sources from the
        # tree (loopback SSRF on the runner), so the CLI never runs it over the checkout.
        # The gap is recorded and shown, never silent (DESIGN §0.7).
        ctx.coverage["terraform_not_scanned"] = {
            "files": int(tf["count"]), "sample": list(tf.get("sample") or []),
            "reason": "vsscli never runs Trivy's Terraform misconfig scanner over a checkout (module-source "
                      "SSRF, INT-12); only Semgrep p/terraform rules apply. Full Terraform IaC coverage "
                      "needs the server code scan of this repository.",
        }
        ctx.warn("{} Terraform file(s) found (e.g. {}): Trivy's Terraform misconfig checks do NOT run in vsscli "
                 "(security: remote module sources). Only Semgrep p/terraform rules apply{}. Use the server "
                 "code scan for full Terraform IaC coverage.".format(
                     tf["count"], (tf.get("sample") or ["?"])[0],
                     "" if (semgrep and not args.offline) else " (and Semgrep is not running)"))
    semgrep_report: Optional[dict] = None
    gitleaks_findings: Optional[List[dict]] = None
    sast_ran = False
    with PrivateDir() as pdir:
        trivy, env = ctx.prepare_trivy(pdir, need_java=False, need_checks=True)
        repl = {str(target): "<target>", pdir.path: "<tmp>"}
        t0 = time.time()
        # 1) Semgrep
        if semgrep:
            sg_env, stripped = scanner_env()
            ctx.env_stripped = sorted(set(ctx.env_stripped) | set(stripped))
            rc, out, err = run_capture([semgrep, "--version"], timeout=60, cwd=pdir.path, env=sg_env)
            if rc != 0:
                die("`{} --version` failed (exit {}): {}".format(semgrep, rc, _last_line(err) or "no output"))
            ctx.engines["semgrep"] = (out.strip().splitlines() or [None])[0]
            check_tool_version("semgrep", ctx.engines["semgrep"])
            configs = [validate_semgrep_config(c_) for c_ in (args.semgrep_config or [])]
            if args.offline:
                registry = [c_ for c_ in configs if _is_registry_pack(c_)]
                if registry:
                    ctx.warn("--offline: skipping registry pack(s) {} (they need semgrep.dev); running local "
                             "rules only.".format(", ".join(registry)))
                configs = [c_ for c_ in configs if not _is_registry_pack(c_)]
                ctx.coverage["semgrep_offline_local_rules_only"] = True
            else:
                server_packs = None if ctx.no_upload else _server_semgrep_configs(
                    ctx.auth, detected, quality=not args.no_quality)
                ctx.coverage["semgrep_pack_source"] = "server" if server_packs else "local"
                if not server_packs:
                    ctx.coverage["semgrep_pack_note"] = (
                        "local tables mirror the server's default select_configs; a server-side "
                        "CODE_SEMGREP_CONFIGS or CODE_QUALITY_ENABLED override is not applied")
                packs = server_packs or select_semgrep_packs(detected, quality=not args.no_quality)
                configs = packs + [c_ for c_ in configs if c_ not in packs]
            help_text = semgrep_help_flags(semgrep, sg_env, pdir.path)
            ignore_flag = "--x-ignore-semgrepignore-files" in help_text
            if not ignore_flag:
                ctx.coverage["semgrepignore_honored"] = (target / ".semgrepignore").exists()
            sg_out = pdir.file("semgrep.json")
            sg_max = int(args.semgrep_max_target_bytes)
            argv = build_semgrep_argv(semgrep, str(target), sg_out, configs, jobs=max(1, min(4, os.cpu_count() or 2)),
                                      ignore_semgrepignore=ignore_flag, max_target_bytes=sg_max)
            ctx.argv["semgrep"] = sanitize_argv(argv, repl)
            info("Running Semgrep with {} rule pack(s) …".format(len(configs)))
            # --verbose logs every file to stderr: a file in the private dir, never memory.
            rc, _, err = run_capture(argv, timeout=int(args.timeout) + 60, cwd=pdir.path, env=sg_env,
                                     stderr_path=pdir.file("semgrep.stderr.log"))
            if rc >= 2:
                die("Semgrep failed (exit {}): {}".format(rc, _last_line(err)))
            try:
                with open(sg_out, "r", encoding="utf-8") as fh:
                    semgrep_report = json.load(fh)
            except (OSError, ValueError, json.JSONDecodeError) as e:
                die("Semgrep produced no readable report: {}".format(e))
            if not isinstance(semgrep_report, dict):
                die("Semgrep report is not a JSON object.")
            semgrep_report, sg_stats = postprocess_semgrep(semgrep_report, target)
            errs = semgrep_report.get("errors") or []
            sk = sg_stats["skipped"]
            ctx.coverage["semgrep"] = {"packs": configs, "results": sg_stats["results"],
                                       "errors": len(errs) if isinstance(errs, list) else 0,
                                       "skipped_paths": sk["total"], "skipped_by_reason": sk["by_reason"],
                                       "skipped_sample": sk["sample"], "skipped_sample_is_partial": sk["sample_is_partial"],
                                       "skipped_reported": sk["reported"],
                                       "degraded_identity": sg_stats["degraded_identity"]}
            big = sk["by_reason"].get("exceeded_size_limit", 0)
            ctx.coverage.setdefault("bounds", {})["SEMGREP_MAX_TARGET_BYTES"] = {
                "value": sg_max, "hit": big if sk["reported"] else None}
            if not sk["reported"]:
                ctx.warn("This Semgrep did not report paths.skipped: files it skipped (over {} bytes, timeouts, "
                         "parse failures) are unknown for this run.".format(sg_max))
            if big:
                ctx.warn("Semgrep skipped {} file(s) larger than --semgrep-max-target-bytes={}: they were NOT "
                         "analysed, and the server will not resolve findings in them.".format(big, sg_max))
            other = sk["unanalysed"] - big
            if other > 0:
                ctx.warn("Semgrep could not fully analyse {} file(s) (timeout / parser error / too many matches); "
                         "see coverage.semgrep.skipped_by_reason.".format(other))
            ctx.engines_ran += [["semgrep", "sast"], ["semgrep", "secret"], ["semgrep", "quality"]]
            sast_ran = True
        # 2) Gitleaks (raw report stays inside the 0700 dir; digest, then drop the secret)
        if gitleaks:
            gl_env, stripped = scanner_env()
            ctx.env_stripped = sorted(set(ctx.env_stripped) | set(stripped))
            rc, out, err = run_capture([gitleaks, "version"], timeout=60, cwd=pdir.path, env=gl_env)
            if rc != 0:
                die("`{} version` failed (exit {}): {}".format(gitleaks, rc, _last_line(err) or "no output"))
            ctx.engines["gitleaks"] = out.strip()
            check_tool_version("gitleaks", ctx.engines["gitleaks"])
            cfg = pdir.file("vss-gitleaks.toml")
            with open(cfg, "w", encoding="utf-8") as fh:
                fh.write(GITLEAKS_CONFIG_TOML)
            ignore_dir = os.path.join(pdir.path, "gl-ignore")
            os.mkdir(ignore_dir, 0o700)
            report_path = pdir.file("gitleaks.raw.json")
            _, help_out, _ = run_capture([gitleaks, "dir", "--help"], timeout=60, cwd=pdir.path, env=gl_env)
            gl_mb = int(args.gitleaks_max_target_mb)
            argv, dropped = build_gitleaks_argv(gitleaks, str(target), report_path, cfg, ignore_dir, help_out,
                                                max_target_mb=gl_mb)
            ctx.flags_dropped += ["gitleaks " + d for d in dropped]
            ctx.argv["gitleaks"] = sanitize_argv(argv, repl)
            # GITLEAKS_MAX_TARGET_MB is a recorded bound (DESIGN §0.7, WP8-20): gitleaks
            # skips bigger files without a word. List them (the server's inventory walk)
            # so the upload tells the server which files gitleaks never read.
            if "--max-target-megabytes" in dropped:
                ctx.coverage.setdefault("bounds", {})["GITLEAKS_MAX_TARGET_MB"] = {"value": None, "hit": 0,
                                                                                   "flag_unsupported": True}
            else:
                big_files, big_total = oversize_files(target, gl_mb * 1024 * 1024)
                entry: Dict[str, Any] = {"value": gl_mb, "hit": big_total}
                if big_total:
                    entry.update({"skipped_sample": big_files[:200], "sample_is_partial": big_total > 200})
                    ctx.coverage["gitleaks_skipped"] = {"paths": big_files, "total": big_total,
                                                        "is_partial": big_total > len(big_files)}
                    ctx.warn("Gitleaks skips {} file(s) larger than --gitleaks-max-target-mb={}: they are NOT "
                             "scanned for secrets, and the server will not resolve findings in them.".format(
                                 big_total, gl_mb))
                ctx.coverage.setdefault("bounds", {})["GITLEAKS_MAX_TARGET_MB"] = entry
            if (target / ".gitleaksignore").exists():
                ctx.coverage["gitleaksignore_present"] = True
                ctx.warn("The repo has a .gitleaksignore; gitleaks always honours it — suppressed secrets are not reported.")
            info("Running Gitleaks …")
            rc, _, err = run_capture(argv, timeout=int(args.timeout) + 60, cwd=pdir.path, env=gl_env)
            try:
                if rc != 0:
                    die("Gitleaks failed (exit {}): {}".format(rc, _last_line(err)))
                try:
                    with open(report_path, "r", encoding="utf-8") as fh:
                        raw_text = fh.read()
                    raw = json.loads(raw_text) if raw_text.strip() else []
                except (OSError, ValueError, json.JSONDecodeError) as e:
                    die("Gitleaks produced no readable report (partial scan?): {}".format(e))
                gitleaks_findings = process_gitleaks_findings(raw, target)
            finally:
                try:
                    os.unlink(report_path)
                except OSError:
                    pass
            ctx.engines_ran.append(["gitleaks", "secret"])
        # 3) Trivy misconfig (+ secret without gitleaks) (+ license if asked)
        scanners = args.scanners or ("misconfig" if gitleaks else "misconfig,secret")
        bad = [s for s in scanners.split(",") if s.strip() not in ("misconfig", "secret", "license")]
        if bad:
            die("--scanners accepts misconfig, secret, license (got {}).".format(",".join(bad)))
        trivy_out = pdir.file("code.json")
        argv = build_code_trivy_argv(trivy, str(target), trivy_out, timeout=args.timeout, scanners=scanners,
                                     offline=bool(args.offline))
        ctx.argv["trivy_code"] = sanitize_argv(argv, repl)
        # No egress for the checkout scan (WP8 N4): a fork PR's `.tf` must not choose
        # the hosts this runner connects to.
        report = run_trivy_json(argv, trivy_out, timeout=args.timeout, cwd=pdir.path,
                                env=code_scan_env(env, pdir.path), label="fs (code)")
        for s in scanners.split(","):
            ctx.engines_ran.append(["trivy", s.strip()])
        digested = digest_trivy_secrets(report, target) if "secret" in scanners else 0
        redact_misconfig_code(report)
        ok("Code engines finished in {:.1f}s.".format(time.time() - t0))

    try:
        tracked = git_tracked_files(target)
    except TrackedFilesError as e:
        if args.tracked_only:
            # Fail closed: filtering against an unknown set would drop every finding
            # and pass the gate.
            die("--tracked-only: cannot list the files git tracks ({}). Re-run without "
                "--tracked-only, or fix the checkout.".format(redact(str(e))))
        ctx.warn("Could not list git-tracked files ({}); untracked-file findings are not counted.".format(e))
        tracked = None
    _, untracked_n = annotate_tracking(report, tracked)
    if args.tracked_only:
        if tracked is None:
            ctx.warn("--tracked-only ignored: not a git checkout (every finding is reported).")
        else:
            report = filter_to_tracked(report, tracked)
            if semgrep_report is not None:
                semgrep_report["results"] = [r for r in semgrep_report.get("results") or [] if r.get("path") in tracked]
            if gitleaks_findings is not None:
                gitleaks_findings = [g for g in gitleaks_findings if g.get("File") in tracked]
            info("--tracked-only: excluded findings in files git does not track.")
    elif untracked_n:
        ctx.warn("{} Trivy finding(s) are in files git does NOT track (.env, local secrets): real, but not "
                 "committed. Use --tracked-only to report only what is in the repo.".format(untracked_n))

    label_report(report, str(git.get("git_remote") or project_name or target.name))
    scrub_local_paths(report, {str(target): "<target>"})
    local = code_findings(report, semgrep_report, gitleaks_findings)
    meta = ctx.meta(extra={"sast_ran": sast_ran, "secrets_engine": "gitleaks" if gitleaks else "trivy",
                           "trivy_secrets_digested": digested, "languages": detected["languages"],
                           # The misconfig scanners that actually ran over the checkout: without
                           # `terraform` the server never auto-resolves a CLI Terraform misconfig row.
                           "trivy_misconfig_scanners": CODE_MISCONFIG_SCANNERS.split(",")
                           if "misconfig" in (args.scanners or "misconfig") else [],
                           "coverage": ctx.coverage}, git=git)
    report["_vss_scanner_meta"] = {"engine": "trivy", "trivy_version": ctx.engines["trivy"],
                                   "db_updated_at": ctx.db.get("updated_at"), "semgrep_version": ctx.engines["semgrep"],
                                   "sast_ran": sast_ran, "cli_version": VSSCLI_VERSION}
    body: Optional[dict] = None
    findings_for_gate = local
    target_info = {"project": git.get("git_remote") or project_name or target.name,
                   "project_path": git.get("project_path"), "branch": git.get("branch"),
                   "commit": git.get("git_commit"), "image_ref": None, "platform": None}
    if not ctx.no_upload:
        payload = {"folder_name": target.name[:200], "project_name": project_name,
                   "git_remote": git.get("git_remote"), "project_path": git.get("project_path") or "",
                   "branch": git.get("branch"), "default_branch": git.get("default_branch"),
                   "is_pull_request": bool(git.get("is_pull_request")),
                   "git_branch": git.get("branch"), "git_commit": git.get("git_commit"),
                   "trivy_output": report, "semgrep": semgrep_report, "gitleaks": gitleaks_findings,
                   "meta": meta, "client_hostname": socket.gethostname()[:255], "cli_version": VSSCLI_VERSION}
        body = ctx.upload("/api/code/scan-result", payload, "ingest:code")
        if isinstance(body.get("findings"), list):
            findings_for_gate = [dict(f) for f in body["findings"] if isinstance(f, dict)]
        else:
            ctx.warn("The server returned no triage-aware findings (older server); gating on local results.")

    def human(doc: dict) -> None:
        print()
        print(bold("Project: ") + str(target_info["project"]))
        cats: Dict[str, int] = {}
        for f in doc["findings"]:
            cats[str(f.get("category") or "?")] = cats.get(str(f.get("category") or "?"), 0) + 1
        print(bold("Engines: ") + ", ".join("{} {}".format(k, v or "?") for k, v in ctx.engines.items() if v) +
              ("" if sast_ran else yellow("  (no SAST: semgrep missing)")))
        print()
        print_counts("By severity", doc["summary"])
        if cats:
            print()
            print(bold("By category"))
            for k in sorted(cats):
                print("  {:<20} {}".format(k, cats[k]))
        if cats.get("secret"):
            print()
            print(red(bold("! {} credential(s) found in source. Rotate them first.".format(cats["secret"]))))
        print()

    sarif_runs = []
    for engine, category in (("semgrep", "vss/code/semgrep/"), ("gitleaks", "vss/code/gitleaks/"), ("trivy", "vss/code/trivy/")):
        res = [f for f in local if f.get("engine") == engine]
        if res or engine == "trivy":
            sarif_runs.append({"category": category, "tool": "vsscli-code-" + engine, "results": [
                {"rule_id": "{}:{}".format(engine, f.get("rule_id")), "severity": f.get("severity"),
                 "title": f.get("title"), "full": f.get("title"), "help": f.get("help") or f.get("title"),
                 "message": f.get("message") or f.get("title"), "uri": f.get("file"),
                 "start_line": f.get("start_line"), "end_line": f.get("end_line"), "start_col": f.get("start_col"),
                 "cwe": f.get("cwe"), "category": f.get("category"), "precision": f.get("precision"),
                 # Secrets: no partialFingerprints — a published SARIF carries no value derived
                 # from the credential (the local one is run-keyed anyway); GitHub computes its own.
                 "fingerprint": None if f.get("category") == "secret" else f.get("fingerprint")} for f in res]})
    return ctx.finish(findings_for_gate=findings_for_gate, local_findings=local, target=target_info, meta=meta,
                      body=body, ui_path="/code", human=human, sarif_runs=sarif_runs)


_QUALITY_CATEGORIES = frozenset({"correctness", "best-practice", "maintainability", "performance",
                                 "compatibility", "portability"})
_SEMGREP_SEV = {"CRITICAL": "critical", "HIGH": "high", "ERROR": "high", "MEDIUM": "medium", "WARNING": "medium",
                "LOW": "low", "INFO": "low"}


#: Per-process random key for CLI-LOCAL secret fingerprints (JSON output only). The
#: server keys secrets with HMAC(K_fp, sha256(secret)); a published CI artifact carrying
#: an unkeyed sha256-derived value would allow an offline dictionary check of weak
#: secrets (incl. untracked .env files). Unstable across runs by design; SARIF omits
#: partialFingerprints for secrets altogether (GitHub then computes its own).
_LOCAL_SECRET_FP_KEY = os.urandom(32)
_SHA256_HEX_RE = re.compile(r"^[0-9a-fA-F]{64}$")
#: Display-engine preference when one secret is found by several engines (the server's
#: code_scanner._SECRET_ENGINE_PREFERENCE).
SECRET_ENGINE_PREFERENCE = ("gitleaks", "trivy", "semgrep")


def local_secret_fingerprint(file_path: str, digest: str) -> str:
    """Engine-independent identity of one secret in one file (the server's
    secret_fingerprint_for('secret', 'v2', path, keyed)), keyed with a run-local
    random key instead of the server's K_fp."""
    keyed = hashlib.blake2b(bytes.fromhex(digest), key=_LOCAL_SECRET_FP_KEY, digest_size=32).hexdigest()
    return compute_fingerprint("secret", "v2", file_path, keyed)


_WS_RE = re.compile(r"\s+")


def _trivy_code_lines(entry: Any, cause_first: bool = False) -> List[Tuple[Any, str]]:
    """Port of the server's code_scanner._code_lines (Trivy Code.Lines as (number, content))."""
    lines = entry.get("Lines") if isinstance(entry, dict) else None
    out: List[Tuple[Any, str]] = []
    causes: List[Tuple[Any, str]] = []
    for item in lines or []:
        if isinstance(item, dict) and isinstance(item.get("Content"), str):
            out.append((item.get("Number", "?"), item["Content"]))
            if item.get("IsCause") is True:
                causes.append(out[-1])
    return causes if (cause_first and causes) else out


def misconfig_group_key(m: dict) -> Tuple[str, str]:
    """(normalised rule id, line-free group key) — port of the server's
    code_scanner._misconfig_parts: rule|provider|service|resource|cause:<sha256 of the
    whitespace-normalised IsCause lines>. No StartLine: a block moving in the file keeps
    its identity (the old `|L<StartLine>` made SARIF fingerprints churn on every edit)."""
    rid = normalise_misconfig_id(str(m.get("ID") or m.get("AVDID") or "unknown-misconfig"))
    cause = m.get("CauseMetadata") if isinstance(m.get("CauseMetadata"), dict) else {}
    body = "\n".join(_WS_RE.sub(" ", c).strip() for _, c in _trivy_code_lines(cause.get("Code"), cause_first=True))
    cause_hash = sha256_hex(body) if body.strip() else ""
    return rid, "{}|{}|{}|{}|cause:{}".format(rid, cause.get("Provider") or "", cause.get("Service") or "",
                                              cause.get("Resource") or "", cause_hash)


def misconfig_occurrences(misconfigs: Sequence[dict]) -> List[int]:
    """Per-file occurrence index k within each group key, ordered by (StartLine, EndLine,
    original order) — port of the server's code_scanner._misconfig_occurrences."""
    order: Dict[str, List[Tuple[int, int, int]]] = {}
    for idx, m in enumerate(misconfigs):
        cause = m.get("CauseMetadata") if isinstance(m.get("CauseMetadata"), dict) else {}
        start = cause.get("StartLine") if isinstance(cause.get("StartLine"), int) else 0
        end = cause.get("EndLine") if isinstance(cause.get("EndLine"), int) else start
        order.setdefault(misconfig_group_key(m)[1], []).append((start, end, idx))
    rank: Dict[int, int] = {}
    for items in order.values():
        for k, (_s, _e, idx) in enumerate(sorted(items)):
            rank[idx] = k
    return [rank[i] for i in range(len(misconfigs))]


def merge_local_secrets(findings: List[dict]) -> List[dict]:
    """One local finding per (file, secret digest) across engines (the server merges on
    the same identity): a credential found by Semgrep AND Gitleaks is ONE finding for the
    gate, the JSON and the SARIF. Engines are unioned, severity maxed, the display row
    follows SECRET_ENGINE_PREFERENCE. Secrets without a digest stay engine-scoped."""
    out: List[dict] = []
    by_key: Dict[Tuple[str, str], dict] = {}
    for f in findings:
        digest = f.pop("_secret_digest", None)
        if f.get("category") != "secret" or not digest:
            out.append(f)
            continue
        key = (str(f.get("file") or ""), str(digest))
        cur = by_key.get(key)
        if cur is None:
            by_key[key] = f
            out.append(f)
            continue
        engines = sorted(set(cur.get("engines") or [cur["engine"]]) | set(f.get("engines") or [f["engine"]]))
        sev = cur["severity"] if SEVERITY_RANK.get(cur["severity"], 0) >= SEVERITY_RANK.get(f["severity"], 0) \
            else f["severity"]

        def _rank(x: dict) -> int:
            e = x.get("engine")
            return SECRET_ENGINE_PREFERENCE.index(e) if e in SECRET_ENGINE_PREFERENCE else 99

        if _rank(f) < _rank(cur):
            f.update({"engines": engines, "severity": sev})
            out[out.index(cur)] = f
            by_key[key] = f
        else:
            cur.update({"engines": engines, "severity": sev})
    return out


def code_findings(trivy_report: dict, semgrep_report: Optional[dict], gitleaks: Optional[List[dict]]) -> List[dict]:
    """Normalised, redacted code findings for gating / JSON / SARIF.

    Identities follow the server (code_scanner): misconfigs by rule + resource + cause
    hash + occurrence k (no line numbers); secrets merged on (file, digest) across
    engines, with a run-locally keyed fingerprint (never an unkeyed sha256 of the secret)."""
    out: List[dict] = []
    for r in (semgrep_report or {}).get("results") or []:
        extra = r.get("extra") or {}
        md = extra.get("metadata") if isinstance(extra.get("metadata"), dict) else {}
        vss = extra.get("vss") or {}
        cat = str(md.get("category") or "").lower()
        digest = vss.get("secret_digest") if _SHA256_HEX_RE.match(str(vss.get("secret_digest") or "")) else None
        is_secret = bool(digest) or _is_secret_rule(str(r.get("check_id") or ""), md)
        category = "secret" if is_secret else ("quality" if cat in _QUALITY_CATEGORIES else "sast")
        sev = _SEMGREP_SEV.get(str(extra.get("severity") or md.get("impact") or "").upper(), "unknown")
        if sev != "low" and (str(md.get("confidence") or "").upper() == "LOW" or "audit" in [str(s).lower() for s in md.get("subcategory") or []]):
            sev = {"critical": "high", "high": "medium", "medium": "low"}.get(sev, sev)
        conf = str(md.get("confidence") or "").upper()
        path = r.get("path")
        # Without a digest the Semgrep identity is line/column-based (no literal in it).
        fp = local_secret_fingerprint(str(path or ""), digest) if (is_secret and digest) else vss.get("fingerprint")
        out.append({"engine": "semgrep", "engines": ["semgrep"], "rule_id": r.get("check_id"), "category": category,
                    "quality_kind": cat if category == "quality" else None, "severity": sev,
                    "title": redact(extra.get("message") or r.get("check_id")), "message": redact(extra.get("message")),
                    "file": path, "start_line": (r.get("start") or {}).get("line"),
                    "end_line": (r.get("end") or {}).get("line"), "start_col": (r.get("start") or {}).get("col"),
                    "cwe": [str(x) for x in md.get("cwe") or []], "fingerprint": fp,
                    "precision": {"HIGH": "high", "MEDIUM": "medium", "LOW": "low"}.get(conf, "medium"),
                    "fix_available": bool(extra.get("fix")), "upgradable": False, "status": "open", "scope": "unknown",
                    "_secret_digest": digest if is_secret else None})
    for g in gitleaks or []:
        digest = g.get("VssSecretDigest") if _SHA256_HEX_RE.match(str(g.get("VssSecretDigest") or "")) else None
        path = g.get("File")
        out.append({"engine": "gitleaks", "engines": ["gitleaks"], "rule_id": g.get("RuleID"), "category": "secret",
                    "severity": "high",
                    "title": redact(g.get("Description") or g.get("RuleID")), "message": redact(g.get("Description")),
                    "file": path, "start_line": g.get("StartLine"), "end_line": g.get("EndLine"),
                    "start_col": g.get("StartColumn"), "cwe": ["CWE-798"],
                    "fingerprint": local_secret_fingerprint(str(path or ""), digest) if digest else
                    compute_fingerprint("gitleaks", str(g.get("RuleID")), str(path), "L{}".format(g.get("StartLine"))),
                    "fix_available": False, "upgradable": False, "status": "open", "scope": "unknown",
                    "_secret_digest": digest})
    for res in trivy_report.get("Results") or []:
        path = strip_dot_slash(res.get("Target"))
        for s in res.get("Secrets") or []:
            if not isinstance(s, dict):
                continue
            digest = s.get("_vss_secret_digest") if _SHA256_HEX_RE.match(str(s.get("_vss_secret_digest") or "")) else None
            out.append({"engine": "trivy", "engines": ["trivy"], "rule_id": s.get("RuleID"), "category": "secret",
                        "severity": _severity(s.get("Severity")), "title": redact(s.get("Title") or s.get("RuleID")),
                        "message": redact(s.get("Title")), "file": path, "start_line": s.get("StartLine"),
                        "end_line": s.get("EndLine"), "cwe": ["CWE-798"],
                        "fingerprint": local_secret_fingerprint(path, digest) if digest else
                        compute_fingerprint("trivy", str(s.get("RuleID")), path, "L{}".format(s.get("StartLine"))),
                        "fix_available": False, "upgradable": False, "status": "open", "scope": "unknown",
                        "_secret_digest": digest})
        misconfigs = [m for m in res.get("Misconfigurations") or [] if isinstance(m, dict)]
        for m, k in zip(misconfigs, misconfig_occurrences(misconfigs)):
            if (m.get("Status") or "FAIL") != "FAIL":
                continue
            cm = m.get("CauseMetadata") if isinstance(m.get("CauseMetadata"), dict) else {}
            rid, group = misconfig_group_key(m)
            out.append({"engine": "trivy", "engines": ["trivy"], "rule_id": m.get("ID"), "category": "misconfig",
                        "severity": _severity(m.get("Severity")), "title": m.get("Title") or m.get("ID"),
                        "message": redact(m.get("Message")), "help": m.get("Resolution"), "file": path,
                        "start_line": cm.get("StartLine"), "end_line": cm.get("EndLine"), "cwe": [],
                        "fingerprint": compute_fingerprint("trivy", rid, path, "{}|k{}".format(group, k)),
                        "fix_available": bool(m.get("Resolution")), "upgradable": False, "status": "open",
                        "scope": "unknown"})
        for lic in res.get("Licenses") or []:
            if isinstance(lic, dict):
                out.append({"engine": "trivy", "engines": ["trivy"], "rule_id": "license:" + str(lic.get("Name")),
                            "category": "license",
                            "severity": _severity(lic.get("Severity")), "title": "License {} ({})".format(
                                lic.get("Name"), lic.get("Category")), "file": path or lic.get("FilePath"),
                            "start_line": 1, "cwe": [], "fix_available": False, "upgradable": False,
                            "status": "open", "scope": "unknown",
                            "fingerprint": compute_fingerprint("trivy", "license:" + str(lic.get("Name")),
                                                               path, str(lic.get("PkgName") or ""))})
    return merge_local_secrets(out)


# ── Commands: image ───────────────────────────────────────────────────────────


#: Canonical image flags newer than MIN_TRIVY_VERSION: dropped (and recorded in
#: meta.flags_dropped) only when this Trivy's `image --help` does not list them.
_OPTIONAL_IMAGE_FLAGS = ("--max-image-size",)


def _drop_unsupported_image_flags(trivy: str, argv: List[str], env: Dict[str, str], cwd: str,
                                  ctx: "ScanContext") -> List[str]:
    rc, help_out, _ = run_capture([trivy, "image", "--help"], timeout=60, cwd=cwd, env=env)
    if rc != 0:
        return argv
    out = list(argv)
    for flag in _OPTIONAL_IMAGE_FLAGS:
        if flag in out and flag not in help_out:
            i = out.index(flag)
            del out[i:i + 2]
            ctx.flags_dropped.append("trivy image " + flag)
            ctx.warn("This Trivy does not support {} (the server passes it); upgrade Trivy for argv "
                     "parity.".format(flag))
    return out


def cmd_image(args: argparse.Namespace) -> int:
    """Container image — the same Trivy argv as the server."""
    ctx = ScanContext(args, "image")
    image_ref = (args.image_ref or "").strip()
    if not image_ref:
        die("Usage: vsscli image <image_ref>")
    dockerfile = parse_dockerfile(args.file) if args.file else None
    body: Optional[dict] = None
    base_diff_ids: Optional[List[str]] = None
    with PrivateDir() as pdir:
        trivy, env = ctx.prepare_trivy(pdir, need_java=True, docker=True, need_checks=True)
        img_env = image_scan_env(env, pdir.path)
        out = pdir.file("image.json")
        argv = build_image_argv(trivy, image_ref, out, timeout=args.timeout, platform_=args.platform,
                                offline=bool(args.offline))
        argv = _drop_unsupported_image_flags(trivy, argv, env, pdir.path, ctx)
        ctx.argv["trivy_image"] = sanitize_argv(argv, {pdir.path: "<tmp>"})
        t0 = time.time()
        report = run_trivy_json(argv, out, timeout=args.timeout, cwd=pdir.path, env=img_env, label="image")
        ctx.engines_ran += [["trivy", "vuln"], ["trivy", "secret"], ["trivy", "misconfig"]]
        ok("Trivy finished in {:.1f}s.".format(time.time() - t0))
        if report.get("ArtifactType") not in (None, "container_image"):
            die("Trivy reported ArtifactType {!r}, expected container_image.".format(report.get("ArtifactType")))
        redact_misconfig_code(report)
        for r in report.get("Results") or []:
            for s in r.get("Secrets") or []:
                if isinstance(s, dict):
                    s["Match"] = redact(s.get("Match"))
                    for cl in ((s.get("Code") or {}).get("Lines") or []):
                        if isinstance(cl, dict):
                            cl["Content"] = redact(cl.get("Content"))
        if args.exclude_base_image_vulns:
            if dockerfile and dockerfile.get("base_image"):
                base_out = pdir.file("base.json")
                bargv = [trivy, "image", "--scanners", "vuln", "--pkg-types", "os"] + _common_scan_flags(
                    base_out, args.timeout, offline=bool(args.offline)) + (
                    ["--platform", args.platform] if args.platform else []) + ["--image-src", "docker,remote",
                                                                                 _check_operand(dockerfile["base_image"], "base image")]
                ctx.argv["trivy_base_image"] = sanitize_argv(bargv, {pdir.path: "<tmp>"})
                base = run_trivy_json(bargv, base_out, timeout=args.timeout, cwd=pdir.path, env=img_env,
                                      label="base image {} (layer attribution)".format(dockerfile["base_image"]))
                base_diff_ids = list((base.get("Metadata") or {}).get("DiffIDs") or [])
            else:
                ctx.warn("--exclude-base-image-vulns needs --file <Dockerfile> with a resolvable final FROM; "
                         "nothing was excluded.")
        cdx = trivy_convert_cyclonedx(trivy, out, pdir.file("image.cdx.json"), env=env, cwd=pdir.path) \
            if (args.sbom_file_output or not ctx.no_upload) else None
        if cdx is not None:
            cdx.pop("vulnerabilities", None)
    local = image_findings(report, base_diff_ids)
    meta = ctx.meta(extra={"platform": args.platform, "image_src": "docker,remote",
                           "detection_priority": "comprehensive", "scanners": "vuln,secret,misconfig",
                           "image_config_scanners": "misconfig,secret", "coverage": ctx.coverage,
                           "base_image_layers": base_layer_count(list((report.get("Metadata") or {}).get("DiffIDs") or []),
                                                                 base_diff_ids) if base_diff_ids else None})
    report["_vss_scanner_meta"] = {"engine": "trivy", "trivy_version": ctx.engines["trivy"],
                                   "db_updated_at": ctx.db.get("updated_at"), "scanners": "vuln,secret,misconfig",
                                   "pkg_types": "os,library", "detection_priority": "comprehensive",
                                   "cli_version": VSSCLI_VERSION}
    findings_for_gate = local
    if not ctx.no_upload:
        payload = {"image_ref": image_ref, "trivy_output": report, "cyclonedx": cdx, "meta": meta,
                   "dockerfile": dockerfile, "platform": args.platform,
                   "client_hostname": socket.gethostname()[:255], "cli_version": VSSCLI_VERSION}
        body = ctx.upload("/api/images/scan-result", payload, "ingest:image")
        if isinstance(body.get("findings"), list):
            by_key = {(f["vuln_id"], f.get("package"), f.get("version")): f for f in local if f.get("kind") == "vuln"}
            findings_for_gate = []
            for s in body["findings"]:
                if isinstance(s, dict):
                    lf = by_key.get((s.get("vuln_id"), s.get("package"), s.get("version")), {})
                    merged = dict(lf)
                    merged.update({k: v for k, v in s.items() if v is not None or k == "suppressed"})
                    findings_for_gate.append(merged)
        else:
            ctx.warn("The server returned no triage-aware findings (older server); gating on local results.")
    md = report.get("Metadata") or {}
    os_md = md.get("OS") or {}
    target_info = {"project": None, "branch": None, "commit": None, "image_ref": image_ref, "platform": args.platform}

    def human(doc: dict) -> None:
        print()
        print(bold("Image: ") + image_ref)
        print(bold("OS:    ") + "{} {}".format(os_md.get("Family") or "?", os_md.get("Name") or "").strip() +
              (red("  (end of life)") if os_md.get("EOSL") else ""))
        if dockerfile:
            print(bold("Base:  ") + str(dockerfile.get("base_image") or "unknown"))
        print()
        print_counts("Issues", doc["summary"])
        print()

    uri = dockerfile["path"] if dockerfile else "container-image/" + re.sub(r"[^A-Za-z0-9._/-]", "_", image_ref)
    sarif_runs = [{"category": "vss/image/", "tool": "vsscli-image", "results": [
        {"rule_id": "trivy:{}".format(f.get("vuln_id")), "severity": f.get("severity"), "title": f.get("title"),
         "full": f.get("title"), "help": "Fixed in: {}".format(f.get("fixed_version") or "no fix yet"),
         "message": "{} {} {}".format(f.get("vuln_id"), f.get("package") or f.get("file") or "", f.get("title") or ""),
         "uri": uri, "start_line": 1, "cvss_score": f.get("cvss_score"), "cwe": f.get("cwe"), "category": f.get("kind"),
         "fingerprint": sha256_hex("image|{}|{}|{}".format(f.get("vuln_id"), f.get("package"), f.get("pkg_path") or f.get("file")))}
        for f in local]}]
    return ctx.finish(findings_for_gate=findings_for_gate, local_findings=local, target=target_info, meta=meta,
                      body=body, ui_path="/images", human=human, sarif_runs=sarif_runs, sbom=cdx)


# ── Commands: login / logout / config / doctor / list / install ──────────────


def cmd_login(args: argparse.Namespace) -> int:
    server = _server_base(args.server)
    insecure = bool(getattr(args, "insecure", False))
    _check_server_scheme(server, insecure)
    api_key = args.api_key
    if api_key:
        warn("--api-key was passed on the command line, so it is now in your shell history and was visible "
             "in `ps`. Prefer omitting it (you will be prompted) or exporting VSS_API_KEY.")
    else:
        api_key = getpass.getpass("API key (vss_…): ").strip()
    if not api_key.startswith("vss_"):
        die("API key must start with 'vss_'. Create one in Settings → API Keys (admin-only).")
    info("Pinging {}/api/health …".format(server))
    code, _, raw = http("GET", server + "/api/health", api_key=None, timeout=15)
    if code != 200:
        die("Server returned {}".format(_server_error_text(code, None, raw)))
    info("Verifying API key …")
    code, body, raw = http("GET", server + "/api/cli/whoami", api_key=api_key, timeout=15)
    if code == 401:
        die("API key rejected (401). Double-check the key.")
    if code == 200 and isinstance(body, dict):
        ok("Key '{}' scopes: {}".format(body.get("name"), ", ".join(body.get("scopes") or []) or "none"))
    elif code == 404:
        warn("This server has no /api/cli/whoami (older version); the key's scopes were not checked.")
    else:
        die("Unexpected response {}".format(_server_error_text(code, body, raw)))
    cfg = {"server": server, "api_key": api_key}
    if insecure and urllib.parse.urlparse(server).scheme == "http":
        cfg["allow_insecure_http"] = True
    if getattr(args, "ui_url", None):
        cfg["ui_url"] = args.ui_url.rstrip("/")
    save_config(cfg)
    ok("Logged in. Config saved to {} (mode 0600).".format(CONFIG_FILE))
    return EXIT_OK


def cmd_logout(_args: argparse.Namespace) -> int:
    if CONFIG_FILE.exists():
        CONFIG_FILE.unlink()
        ok("Logged out.")
    else:
        info("Already logged out.")
    return EXIT_OK


def cmd_config(_args: argparse.Namespace) -> int:
    cfg = load_config()
    if not cfg:
        info("No config found. Run `vsscli login` first.")
        return EXIT_OK
    masked = dict(cfg)
    if masked.get("api_key"):
        masked["api_key"] = masked["api_key"][:12] + "…" + masked["api_key"][-4:]
    for name in ("VSS_SERVER", "VSS_API_KEY"):
        if os.environ.get(name):
            masked["env_override_" + name.lower()] = True
    emit_stdout(json.dumps(masked, indent=2))
    return EXIT_OK


def cmd_doctor(_args: argparse.Namespace) -> int:
    """Diagnose Trivy/Semgrep/Gitleaks, the denylist, DB age, server reachability and key scopes.
    Exit 2 when something would make scans fail."""
    problems = 0
    print(bold("vsscli doctor"))
    print("  vsscli version : {}".format(VSSCLI_VERSION))
    print("  python         : {} ({})".format(platform.python_version(), sys.executable))
    print("  os / arch      : {} {}".format(platform.system(), platform.machine()))
    print("  ci runner      : {}".format("yes" if is_ci() else "no"))
    trivy = which("trivy")
    with PrivateDir() as pdir:
        env, stripped = scanner_env()
        if trivy:
            rc, out, _ = run_capture([trivy, "--version", "--format", "json"], timeout=30, cwd=pdir.path, env=env)
            ti = parse_trivy_version_json(out) if rc == 0 else {}
            v = (ti.get("version") or "").lstrip("vV")
            status = green("ok")
            if ("trivy", v) in scanner_denylist():
                status, problems = red("DENYLISTED — do not run; rotate CI secrets"), problems + 1
            elif (parse_version_tuple(v) or (0, 0, 0)) < MIN_TRIVY_VERSION:
                status, problems = red("too old (need >= 0.53)"), problems + 1
            print("  trivy          : {} {} ({})".format(trivy, v or "?", status))
            upd = _parse_ts(ti.get("db_updated_at"))
            if upd:
                age = (_dt.datetime.now(_dt.timezone.utc) - upd).total_seconds() / 3600.0
                print("  trivy DB       : UpdatedAt {} ({:.0f} h old){}".format(ti.get("db_updated_at"), age,
                      yellow("  — will refresh before scanning") if age > 24 else ""))
            else:
                print("  trivy DB       : {}".format(yellow("not downloaded yet (first scan downloads it)")))
            print("  java DB        : {}".format(ti.get("java_db_updated_at") or yellow("not downloaded yet")))
        else:
            problems += 1
            print("  trivy          : {}".format(red("not installed")))
        for tool, args_ in (("semgrep", ["--version"]), ("gitleaks", ["version"])):
            p = which(tool)
            if p:
                rc, out, _ = run_capture([p] + args_, timeout=60, cwd=pdir.path, env=env)
                print("  {:<15}: {} {}".format(tool, p, (out.strip().splitlines() or ["?"])[0] if rc == 0 else red("error")))
            else:
                print("  {:<15}: {}".format(tool, yellow("not installed (code scans degrade: no SAST)" if tool == "semgrep"
                                                          else "not installed (Trivy secret scanning is used)")))
        if stripped:
            print("  env ignored    : {}".format(", ".join(stripped)))
    print("  denylist       : {}".format(", ".join(sorted("{}:{}".format(t, v) for t, v in scanner_denylist()))))
    auth = resolve_auth(required=False)
    if not auth:
        print("  server         : {}".format(yellow("no API key (export VSS_API_KEY, or vsscli login)")))
        return EXIT_ERROR if problems else EXIT_OK
    print("  server         : {} (from {})".format(auth["server"], auth["source"]))
    code, _, raw = http("GET", auth["server"] + "/api/health", api_key=None, timeout=15, retries=1)
    print("  health         : {}".format(green("reachable") if code == 200 else red("HTTP {}".format(code))))
    if code != 200:
        problems += 1
    code, body, raw = http("GET", auth["server"] + "/api/cli/whoami", api_key=auth["api_key"], timeout=15, retries=1)
    if code == 200 and isinstance(body, dict):
        scopes = body.get("scopes") or []
        print("  api key        : {} (expires {})".format(body.get("name"), body.get("expires_at") or "never"))
        print("  scopes         : {}".format(", ".join(scopes) or "none"))
        for kind in ("oss", "code", "image"):
            has = "*" in scopes or "ingest:" + kind in scopes or \
                (kind in ("oss", "code") and "code:write" in scopes) or (kind == "image" and "images:write" in scopes)
            print("  can upload {:<5}: {}".format(kind, green("yes") if has else yellow("no")))
        ids = body.get("allowed_identities")
        print("  identities     : {}".format(", ".join(ids) if ids else "only projects this key created" if ids == [] else "legacy (unbound)"))
    elif code == 401:
        problems += 1
        print("  api key        : {}".format(red("rejected (401)")))
    elif code == 404:
        print("  api key        : {}".format(yellow("server has no /api/cli/whoami (older version)")))
    else:
        print("  api key        : {}".format(red("HTTP {}".format(code))))
    return EXIT_ERROR if problems else EXIT_OK


def cmd_list(args: argparse.Namespace) -> int:
    auth = resolve_auth(required=True)
    assert auth is not None
    if args.type == "repos":
        code, body, raw = http("GET", auth["server"] + "/api/projects/repos/all?limit=500", api_key=auth["api_key"])
        if code != 200 or not isinstance(body, list):
            die("List failed: {}".format(_server_error_text(code, body, raw)))
        if not body:
            info("No repositories tracked yet.")
            return EXIT_OK
        for r in body:
            print("  {:>5}  {}/{}@{}  {}".format(r.get("id"), r.get("github_owner"), r.get("github_repo"),
                                               r.get("branch"), grey(str(r.get("status")))))
        return EXIT_OK
    code, body, raw = http("GET", auth["server"] + "/api/images", api_key=auth["api_key"])
    if code != 200 or not isinstance(body, list):
        die("List failed: {}".format(_server_error_text(code, body, raw)))
    if not body:
        info("No images tracked yet.")
        return EXIT_OK
    name_w = max(len(str(i.get("image_ref"))) for i in body)
    for i in body:
        chml = "{}/{}/{}/{}".format(i.get("vuln_critical", 0), i.get("vuln_high", 0), i.get("vuln_medium", 0), i.get("vuln_low", 0))
        last = (i.get("last_scanned_at") or "never").split("T")[0]
        print("  {:>4}  {:<{w}}  {:<11}  {:<14}  {}".format(i.get("id"), str(i.get("image_ref")), str(i.get("status")),
                                                          chml, grey(last), w=name_w))
    return EXIT_OK


def cmd_install(args: argparse.Namespace) -> int:
    """Symlink this script onto PATH (a symlink, so upgrades are one file replace)."""
    src = Path(__file__).resolve()
    dest = Path(args.dest).expanduser() / "vsscli"
    existing = which("vsscli")
    if existing and Path(existing).resolve() != src:
        warn("Another vsscli is already on PATH: {}".format(existing))
    try:
        dest.parent.mkdir(parents=True, exist_ok=True)
        if dest.exists() or dest.is_symlink():
            dest.unlink()
        dest.symlink_to(src)
        ok("Linked {} → {}".format(dest, src))
    except PermissionError:
        die("No permission to write {}.\n  Run:  sudo ln -sf {} {}".format(dest, src, dest))
    except OSError as e:
        die("Could not link {}: {}".format(dest, e))
    resolved = which("vsscli")
    if resolved and Path(resolved).resolve() == src:
        ok("`vsscli` now resolves to this script ({}).".format(resolved))
    else:
        warn("`vsscli` still resolves to {} — an earlier PATH entry shadows {}.".format(resolved or "nothing", dest))
    return EXIT_OK


# ── argparse plumbing ─────────────────────────────────────────────────────────


def _bounded_int(lo: int, hi: int):
    """argparse type: an int in [lo, hi] (a bad bound is a usage error, exit 2)."""
    def _parse(text: str) -> int:
        try:
            v = int(text)
        except (TypeError, ValueError):
            raise argparse.ArgumentTypeError("expected an integer, got {!r}".format(text))
        if not lo <= v <= hi:
            raise argparse.ArgumentTypeError("must be between {} and {}".format(lo, hi))
        return v
    return _parse


class _Parser(argparse.ArgumentParser):
    def error(self, message: str) -> None:  # argparse errors are exit 2 (EXIT_ERROR) — keep it explicit
        self.print_usage(sys.stderr)
        die("{}: error: {}".format(self.prog, message))


def build_parser() -> argparse.ArgumentParser:
    p = _Parser(prog="vsscli", description=(__doc__ or "vsscli").split("\n\n")[0], formatter_class=argparse.RawDescriptionHelpFormatter,
                epilog="Exit codes: 0 pass, 1 issues at/above the gate, 2 error, 3 nothing scannable.")
    p.add_argument("--version", action="version", version="vsscli {}".format(VSSCLI_VERSION))
    sub = p.add_subparsers(dest="cmd", metavar="<command>", parser_class=_Parser)
    sub.required = True

    sp = sub.add_parser("login", help="Save server URL + API key (prompts for the key)")
    sp.add_argument("--server", default=DEFAULT_SERVER, help="Server URL (default: {})".format(DEFAULT_SERVER))
    sp.add_argument("--api-key", help="PREFER OMITTING THIS (shell history, `ps`); you will be prompted")
    sp.add_argument("--insecure", action="store_true",
                    help="Allow http:// to a non-loopback host OUTSIDE CI (never in CI; TLS is never disabled)")
    sp.add_argument("--ui-url", help="Browser URL for the dashboard when it differs from --server")
    sp.set_defaults(func=cmd_login)

    sp = sub.add_parser("logout", help="Remove saved config")
    sp.set_defaults(func=cmd_logout)
    sp = sub.add_parser("config", help="Show current config (key partially masked)")
    sp.set_defaults(func=cmd_config)
    sp = sub.add_parser("doctor", help="Check Trivy/Semgrep/Gitleaks, DB age, server and key scopes")
    sp.set_defaults(func=cmd_doctor)

    for name in ("image", "scan", "container"):
        sp = sub.add_parser(name, help="Scan a container image (vuln + secret + misconfig)" if name == "image"
                            else "Alias of `image`")
        sp.add_argument("image_ref", help="Image to scan, e.g. nginx:1.27-alpine")
        _add_common_scan_flags(sp, timeout=900)
        sp.add_argument("--sbom-file-output", metavar="PATH", help="Write the image's CycloneDX SBOM to PATH")
        sp.add_argument("--platform", help="os/arch[/variant] of a multi-arch image")
        sp.add_argument("--file", metavar="DOCKERFILE", help="Dockerfile: sends FROM lines for base-image attribution")
        sp.add_argument("--exclude-base-image-vulns", action="store_true",
                        help="Do not gate on OS vulnerabilities that come from the base image (needs --file)")
        sp.add_argument("--skip-trivy-check", action="store_true", help=argparse.SUPPRESS)
        sp.set_defaults(func=cmd_image)

    sp = sub.add_parser("oss", help="Open-source dependencies (run AFTER your build: lockfiles ∪ installed artefacts)")
    sp.add_argument("path", nargs="?", default=".", help="Project directory (default: current)")
    _add_common_scan_flags(sp, timeout=900)
    sp.add_argument("--sbom-file-output", metavar="PATH", help="Write the merged CycloneDX SBOM (inventory only)")
    sp.add_argument("--include-dev", action="store_true",
                    help="Include dev dependencies (Trivy --include-dev-deps: npm, yarn, gradle) and gate on them")
    sp.add_argument("--lockfile-only", action="store_true", help="Skip the installed-artefact (rootfs) pass")
    sp.add_argument("--artifact", action="append", metavar="PATH",
                    help="Extra built artefact to scan (JAR/WAR dir, Go/Rust binary, wheel dir). Repeatable")
    sp.add_argument("--sbom", metavar="CYCLONEDX_JSON",
                    help="Build-tool SBOM (cyclonedx-maven-plugin / cyclonedx-gradle-plugin) to include")
    sp.add_argument("--resolve", action="store_true",
                    help="Resolve manifests without a lockfile with their own tool here (npm, pip, composer, "
                         "bundler, cargo, dotnet, gradle, maven; runs build configuration, never on the server). "
                         "Automatic on CI runners")
    sp.add_argument("--no-resolve", action="store_true",
                    help="Never run package managers or build tools, not even on a CI runner")
    sp.add_argument("--no-sbom", action="store_true", help=argparse.SUPPRESS)
    sp.add_argument("--name", help=argparse.SUPPRESS)
    sp.add_argument("--prod-only", action="store_true", help=argparse.SUPPRESS)
    sp.set_defaults(func=cmd_oss)

    sp = sub.add_parser("code", help="Source code: Semgrep SAST + quality, secrets, IaC misconfig")
    sp.add_argument("path", nargs="?", default=".", help="Directory to scan (default: current)")
    _add_common_scan_flags(sp, timeout=900)
    sp.add_argument("--require-sast", action="store_true", help="Exit 2 when semgrep is not installed")
    sp.add_argument("--no-sast", dest="sast", action="store_false", help="Do not run Semgrep")
    sp.add_argument("--no-quality", action="store_true", help="Skip Semgrep quality packs")
    sp.add_argument("--semgrep-config", action="append", metavar="CONFIG",
                    help="Extra Semgrep config: registry pack (p/..., r/...) or a local rules file. Repeatable")
    sp.add_argument("--secrets-engine", choices=("auto", "gitleaks", "trivy"), default="auto",
                    help="auto = gitleaks when installed, else Trivy")
    sp.add_argument("--scanners", default=None, help="Trivy scanners (misconfig,secret,license); default by engine")
    sp.add_argument("--tracked-only", action="store_true", help="Only report files git tracks")
    sp.add_argument("--semgrep-max-target-bytes", type=_bounded_int(1024, 1024 * 1024 * 1024),
                    default=SEMGREP_MAX_TARGET_BYTES,
                    help="Semgrep skips bigger files (recorded in coverage; default {}, the server's)".format(
                        SEMGREP_MAX_TARGET_BYTES))
    sp.add_argument("--gitleaks-max-target-mb", type=_bounded_int(1, 4096), default=GITLEAKS_MAX_TARGET_MB,
                    help="Gitleaks skips bigger files (recorded in coverage; default {}, the server's)".format(
                        GITLEAKS_MAX_TARGET_MB))
    sp.add_argument("--name", help=argparse.SUPPRESS)
    sp.set_defaults(func=cmd_code, sast=True)

    sp = sub.add_parser("install", help="Symlink this script onto your PATH")
    sp.add_argument("--dest", default="/usr/local/bin", help="Directory to link into (default: /usr/local/bin)")
    sp.set_defaults(func=cmd_install)

    sp = sub.add_parser("list", help="List images or repositories on the server")
    sp.add_argument("--type", choices=("images", "repos"), default="images")
    sp.set_defaults(func=cmd_list)
    return p


def main(argv: Optional[List[str]] = None) -> int:
    parser = build_parser()
    args = parser.parse_args(argv)
    try:
        return int(args.func(args) or 0)
    except KeyboardInterrupt:
        print(file=sys.stderr)
        return 130
    except SystemExit:
        raise
    except Exception as exc:  # noqa: BLE001 — any unexpected failure is an ERROR (2), never "issues" (1)
        print(c("31", "✗ internal error: {}: {}".format(type(exc).__name__, redact(str(exc))[:500]), sys.stderr),
              file=sys.stderr)
        if os.environ.get("VSSCLI_DEBUG") == "1":
            import traceback
            traceback.print_exc()
        return EXIT_ERROR


if __name__ == "__main__":
    sys.exit(main())
